summaryrefslogtreecommitdiff
path: root/expt_0516.cu
diff options
context:
space:
mode:
authorKunoiSayami <[email protected]>2023-05-17 01:56:15 +0800
committerKunoiSayami <[email protected]>2023-05-17 01:56:15 +0800
commitff0640807f810b5ac3d9f7a831b60e1266608a08 (patch)
treec02f21fd433bf54549a2997778b0793c5a59e1ef /expt_0516.cu
parentb6bcd9a2eb08796e44f317bfcda2a14b6bb311db (diff)
feat(exp): Add expt_0516
Signed-off-by: KunoiSayami <[email protected]>
Diffstat (limited to 'expt_0516.cu')
-rw-r--r--expt_0516.cu304
1 files changed, 304 insertions, 0 deletions
diff --git a/expt_0516.cu b/expt_0516.cu
new file mode 100644
index 0000000..4d9be45
--- /dev/null
+++ b/expt_0516.cu
@@ -0,0 +1,304 @@
+// Experimental content: Test branch performance (optimized memory and output)
+#include "sortlib.cuh"
+
+#include <algorithm>
+#include <cassert>
+#include <cstdio>
+#include <random>
+#include <vector>
+
+#ifndef NDEBUG
+#define P_ERR(...) fprintf(stderr, __VA_ARGS__)
+#else
+#define P_ERR(...)
+#endif
+
+std::vector<unsigned long long> population_vector, sample_vector,
+ le_sample_vector;
+
+constexpr long DEFAULT_SAMPLE_LENGTH = 65535;
+constexpr size_t TEST_LENGTH = 32'768;
+// constexpr size_t TEST_LENGTH = 4096;
+// constexpr size_t RESERVED_BLOCK = 100;
+
+unsigned long long max_value = 0, min_value = 0xfffffffff;
+typedef std::pair<int, int> dim_pair_type;
+
+inline void store_into_vector(unsigned long long value) {
+ if (max_value < value) {
+ max_value = value;
+ }
+ if (min_value > value) {
+ min_value = value;
+ }
+ population_vector.push_back(value);
+}
+// #define TEST_BOUNDS
+
+__device__ long cuda_sample_length;
+__device__ key_type *cudaSampleItem, *cudaNormalSampleItem, *cudaPopulationItem;
+//__device__ bool *cdf_result;
+#ifdef TEST_BOUNDS
+__device__ unsigned insert_value;
+__device__ unsigned int index_max, index_min;
+#endif
+
+__global__ void initSample(key_type *sample, key_type *population) {
+ cudaSampleItem = sample;
+ cudaPopulationItem = population;
+ // cudaMalloc(&cdf_result, sizeof(bool) * TEST_LENGTH);
+ // cdf_result = nullptr;
+ // cdf_normal_result = nullptr;
+}
+
+__global__ void initNormalSample(key_type *normal_sample) {
+ cudaNormalSampleItem = normal_sample;
+}
+
+__global__ void kernel(unsigned long step, const double slice_size) {
+ auto custom_sort = CustomSort(cuda_sample_length, sizeof(key_type) * 8);
+ // printf("kernel1 step: %ld\n", step);
+ for (int i = 0; i < step; i++) {
+ auto tid = step * (blockIdx.x * blockDim.x + threadIdx.x) + i;
+
+ // printf("%d\n", tid);
+ auto _index = (int)(custom_sort.sample_cdf_custom_version(
+ cudaSampleItem, cudaPopulationItem[tid]) /
+ slice_size) -
+ 1;
+ // cdf_result[index] = true;
+ }
+}
+
+__global__ void kernel2(unsigned long step, const double slice_size) {
+
+ for (int i = 0; i < step; i++) {
+ auto tid = step * (blockIdx.x * blockDim.x + threadIdx.x) + i;
+ auto _index =
+ (int)(FactorySort::sample_cdf(cudaNormalSampleItem, cuda_sample_length,
+ cudaPopulationItem[tid]) /
+ slice_size) -
+ 1;
+ // cdf_result[index] = true;
+ }
+}
+
+__global__ void print2() {
+ /*for (int i = 0; i< SAMPLE_LENGTH; i++) {
+ printf("%llu ", cudaSampleItem[i]);
+ }
+ printf("\n");*/
+ printf("%lu\n", CustomSort::fast_log(cuda_sample_length));
+}
+
+__global__ void printFunction() {}
+
+__global__ void checkItems() {
+ for (int i = 0; i < cuda_sample_length; i++) {
+ printf("%lld ", cudaSampleItem[i]);
+ }
+ printf("\n");
+}
+
+void initCustomSampleCPU() {
+ auto sample_length = sample_vector.size();
+ auto tmp = new key_type[sample_length];
+
+ auto custom_sort = CustomSort(sample_length, 0);
+
+ for (size_t i = 0; i < sample_length; i++) {
+ tmp[i] = sample_vector[custom_sort.calculate_index(i) - 1];
+ }
+
+ memcpy(sample_vector.data(), tmp, sizeof(key_type) * sample_length);
+
+ delete[] tmp;
+}
+
+constexpr dim3 grid_dim = 32, block_dim = 32;
+
+bool checkTestSize() {}
+
+void run_kernel(size_t test_size, bool custom = true) {
+ // need description
+ unsigned long step = test_size / (grid_dim.x * block_dim.x);
+
+ // Only scale mode will use
+ auto scale = 2;
+ const unsigned long split_size = test_size * scale;
+ const auto slice_size = 1.0 / (double)split_size;
+
+ cudaEvent_t start, stop;
+ cudaEventCreate(&start);
+ cudaEventCreate(&stop);
+ cudaEventRecord(start, nullptr);
+ if (custom) {
+ kernel<<<grid_dim, block_dim>>>(step, slice_size);
+ } else {
+ kernel2<<<grid_dim, block_dim>>>(step, slice_size);
+ }
+
+ cudaDeviceSynchronize();
+ cudaEventRecord(stop, nullptr);
+ cudaEventSynchronize(stop);
+ float time;
+ cudaEventElapsedTime(&time, start, stop);
+ cudaEventDestroy(start);
+ cudaEventDestroy(stop);
+ if (time == 0) {
+ // printf("last error: %u\n", cudaGetLastError());
+ }
+ printf("%stime: %lf ", custom ? "custom " : "", time);
+ cudaDeviceSynchronize();
+
+ printFunction<<<1, 1>>>();
+ cudaDeviceSynchronize();
+}
+
+__global__ void freeStorageStage1() { cudaFree(cudaSampleItem); }
+
+__global__ void freeStorageStage2() {
+ cudaFree(cudaNormalSampleItem);
+ cudaFree(cudaPopulationItem);
+ // cudaFree(cdf_result);
+}
+
+unsigned long randomRow(unsigned long max_value_) {
+ std::random_device randomDevice;
+ std::mt19937 mt19937(randomDevice());
+ std::uniform_int_distribution<std::mt19937::result_type> dst(0, max_value_);
+ return dst(mt19937);
+}
+
+void readFile(char const *filename, long sample_length,
+ unsigned long &total_row,
+ unsigned long max_number = TEST_LENGTH) {
+ auto read_number = 0UL;
+ FILE *file = fopen(filename, "r");
+ assert(file);
+ P_ERR("Reading sample");
+ for (long long i;
+ read_number < max_number && fscanf(file, "%lld ", &i) != EOF;
+ store_into_vector(i))
+ read_number++;
+
+ if (total_row > 0) {
+ P_ERR("\rReading skip");
+ read_number = randomRow(total_row - sample_length - max_number - 256);
+ total_row = read_number;
+ // fprintf(stderr, "Skip %lu\n", read_number);
+ for (long long i; read_number > 0 && fscanf(file, "%lld ", &i) != EOF;)
+ read_number--;
+ }
+
+ P_ERR("\rReading population");
+ read_number = 0;
+
+ auto remain = sample_length + 256;
+ for (long long i; read_number < remain && fscanf(file, "%lld ", &i) != EOF;
+ store_into_vector(i))
+ read_number++;
+ fclose(file);
+ P_ERR("\r");
+}
+
+long pow_for_sample(long n) {
+ auto x = 2;
+ for (int i = 0; i < n; i++) {
+ x *= 2;
+ }
+ return x - 1;
+}
+
+__global__ void applyCudaSampleLength(long length) {
+ cuda_sample_length = length;
+}
+
+int main(int argc, char const *argv[]) {
+
+ auto test_size = TEST_LENGTH;
+ auto sample_length = DEFAULT_SAMPLE_LENGTH;
+ auto total_row = 0UL;
+
+ if (argc >= 2) {
+ test_size = strtol(argv[1], nullptr, 10);
+ }
+ if (argc >= 3) {
+ sample_length = pow_for_sample(strtol(argv[2], nullptr, 10));
+ }
+ if (argc >= 4) {
+ total_row = strtol(argv[3], nullptr, 10);
+ }
+
+ readFile("normal_distribution.txt", sample_length, total_row, test_size);
+
+ printf("population: %zu, skip: %lu, test size: %zu, sample length: %zu\n",
+ population_vector.size(), total_row, test_size, sample_length);
+
+ sample_vector = std::vector<key_type>(
+ population_vector.begin(), population_vector.begin() + sample_length - 2);
+ sample_vector.push_back(min_value);
+ sample_vector.push_back(max_value);
+ std::sort(sample_vector.begin(), sample_vector.end());
+ le_sample_vector = sample_vector;
+ initCustomSampleCPU();
+
+ key_type *cudaSample = nullptr, *cudaPopulation;
+
+ P_ERR("Coping sample");
+ cudaMalloc(&cudaSample, sizeof(key_type) * sample_length);
+ assert(sample_length == sample_vector.size());
+ cudaMemcpy(cudaSample, sample_vector.data(), sizeof(key_type) * sample_length,
+ cudaMemcpyHostToDevice);
+ cudaMalloc(&cudaPopulation, sizeof(key_type) * test_size);
+
+ P_ERR("\rCoping population");
+ cudaMemcpy(cudaPopulation, population_vector.data() + sample_length,
+ sizeof(key_type) * test_size, cudaMemcpyHostToDevice);
+
+ // auto custom_sort = CustomSort(SAMPLE_LENGTH, sizeof(long) * 8);
+ // custom_sort.testCalculation();
+ P_ERR("\rApply custom length and test calculation");
+ applyCudaSampleLength<<<1, 1>>>(sample_length);
+ testCustomCalculation<<<1, 1>>>(sample_length, sizeof(key_type) * 8);
+ cudaDeviceSynchronize();
+ P_ERR("\rApply custom length and test calculation completed");
+
+ // Can remove this function if pass
+ CustomSort::testSelf(sample_length, sizeof(key_type) * 8);
+
+ initSample<<<1, 1>>>(cudaSample, cudaPopulation);
+ cudaDeviceSynchronize();
+
+ // check_items<<<1, 1>>>();
+ // cudaDeviceSynchronize();
+ P_ERR("\r \rRunning "
+ "kernel\n");
+ run_kernel(test_size);
+
+ freeStorageStage1<<<1, 1>>>();
+ cudaDeviceSynchronize();
+
+ key_type *cudaNormalSample = nullptr;
+
+ P_ERR("\nCoping new sample");
+ cudaMalloc(&cudaNormalSample, sizeof(key_type) * sample_length);
+ cudaMemcpy(cudaNormalSample, le_sample_vector.data(),
+ sizeof(key_type) * sample_length, cudaMemcpyHostToDevice);
+
+ /*cudaMemcpy(sample_vector.data(), cudaSample, sizeof(key_type) *
+ sample_length, cudaMemcpyDeviceToHost);*/
+
+ initNormalSample<<<1, 1>>>(cudaNormalSample);
+ cudaDeviceSynchronize();
+
+ P_ERR("\r ");
+ P_ERR("\rRunning kernel2\n");
+ run_kernel(test_size, false);
+ // check_items<<<1, 1>>>();
+ // cudaDeviceSynchronize();
+ puts("");
+
+ freeStorageStage2<<<1, 1>>>();
+ cudaDeviceSynchronize();
+}