// Experimental content: Test branch performance (optimized memory and output) #include "sortlib.cuh" #include #include #include #include #include #ifndef NDEBUG #define P_ERR(...) fprintf(stderr, __VA_ARGS__) #else #define P_ERR(...) #endif std::vector population_vector, sample_vector, le_sample_vector; constexpr long DEFAULT_SAMPLE_LENGTH = 65535; constexpr size_t TEST_LENGTH = 32'768; // constexpr size_t TEST_LENGTH = 4096; // constexpr size_t RESERVED_BLOCK = 100; unsigned long long max_value = 0, min_value = 0xfffffffff; typedef std::pair dim_pair_type; inline void store_into_vector(unsigned long long value) { if (max_value < value) { max_value = value; } if (min_value > value) { min_value = value; } population_vector.push_back(value); } // #define TEST_BOUNDS __device__ long cuda_sample_length; __device__ key_type *cudaSampleItem, *cudaNormalSampleItem, *cudaPopulationItem; //__device__ bool *cdf_result; #ifdef TEST_BOUNDS __device__ unsigned insert_value; __device__ unsigned int index_max, index_min; #endif __global__ void initSample(key_type *sample, key_type *population) { cudaSampleItem = sample; cudaPopulationItem = population; // cudaMalloc(&cdf_result, sizeof(bool) * TEST_LENGTH); // cdf_result = nullptr; // cdf_normal_result = nullptr; } __global__ void initNormalSample(key_type *normal_sample) { cudaNormalSampleItem = normal_sample; } __global__ void kernel(unsigned long step, const double slice_size) { auto custom_sort = CustomSort(cuda_sample_length, sizeof(key_type) * 8); // printf("kernel1 step: %ld\n", step); for (int i = 0; i < step; i++) { auto tid = step * (blockIdx.x * blockDim.x + threadIdx.x) + i; // printf("%d\n", tid); auto _index = (int)(custom_sort.sample_cdf_custom_version( cudaSampleItem, cudaPopulationItem[tid]) / slice_size) - 1; // cdf_result[index] = true; } } __global__ void kernel2(unsigned long step, const double slice_size) { for (int i = 0; i < step; i++) { auto tid = step * (blockIdx.x * blockDim.x + threadIdx.x) + i; auto _index = (int)(FactorySort::sample_cdf(cudaNormalSampleItem, cuda_sample_length, cudaPopulationItem[tid]) / slice_size) - 1; // cdf_result[index] = true; } } __global__ void print2() { /*for (int i = 0; i< SAMPLE_LENGTH; i++) { printf("%llu ", cudaSampleItem[i]); } printf("\n");*/ printf("%lu\n", CustomSort::fast_log(cuda_sample_length)); } __global__ void printFunction() {} __global__ void checkItems() { for (int i = 0; i < cuda_sample_length; i++) { printf("%lld ", cudaSampleItem[i]); } printf("\n"); } void initCustomSampleCPU() { auto sample_length = sample_vector.size(); auto tmp = new key_type[sample_length]; auto custom_sort = CustomSort(sample_length, 0); for (size_t i = 0; i < sample_length; i++) { tmp[i] = sample_vector[custom_sort.calculate_index(i) - 1]; } memcpy(sample_vector.data(), tmp, sizeof(key_type) * sample_length); delete[] tmp; } constexpr dim3 grid_dim = 32, block_dim = 32; bool checkTestSize() {} void run_kernel(size_t test_size, bool custom = true) { // need description unsigned long step = test_size / (grid_dim.x * block_dim.x); // Only scale mode will use auto scale = 2; const unsigned long split_size = test_size * scale; const auto slice_size = 1.0 / (double)split_size; cudaEvent_t start, stop; cudaEventCreate(&start); cudaEventCreate(&stop); cudaEventRecord(start, nullptr); if (custom) { kernel<<>>(step, slice_size); } else { kernel2<<>>(step, slice_size); } cudaDeviceSynchronize(); cudaEventRecord(stop, nullptr); cudaEventSynchronize(stop); float time; cudaEventElapsedTime(&time, start, stop); cudaEventDestroy(start); cudaEventDestroy(stop); if (time == 0) { // printf("last error: %u\n", cudaGetLastError()); } printf("%stime: %lf ", custom ? "custom " : "", time); cudaDeviceSynchronize(); printFunction<<<1, 1>>>(); cudaDeviceSynchronize(); } __global__ void freeStorageStage1() { cudaFree(cudaSampleItem); } __global__ void freeStorageStage2() { cudaFree(cudaNormalSampleItem); cudaFree(cudaPopulationItem); // cudaFree(cdf_result); } unsigned long randomRow(unsigned long max_value_) { std::random_device randomDevice; std::mt19937 mt19937(randomDevice()); std::uniform_int_distribution dst(0, max_value_); return dst(mt19937); } void readFile(char const *filename, long sample_length, unsigned long &total_row, unsigned long max_number = TEST_LENGTH) { auto read_number = 0UL; FILE *file = fopen(filename, "r"); assert(file); P_ERR("Reading sample"); for (long long i; read_number < max_number && fscanf(file, "%lld ", &i) != EOF; store_into_vector(i)) read_number++; if (total_row > 0) { P_ERR("\rReading skip"); read_number = randomRow(total_row - sample_length - max_number - 256); total_row = read_number; // fprintf(stderr, "Skip %lu\n", read_number); for (long long i; read_number > 0 && fscanf(file, "%lld ", &i) != EOF;) read_number--; } P_ERR("\rReading population"); read_number = 0; auto remain = sample_length + 256; for (long long i; read_number < remain && fscanf(file, "%lld ", &i) != EOF; store_into_vector(i)) read_number++; fclose(file); P_ERR("\r"); } long pow_for_sample(long n) { auto x = 2; for (int i = 1; i < n; i++) { x *= 2; } return x - 1; } __global__ void applyCudaSampleLength(long length) { cuda_sample_length = length; } int main(int argc, char const *argv[]) { auto test_size = TEST_LENGTH; auto sample_length = DEFAULT_SAMPLE_LENGTH; auto total_row = 0UL; if (argc >= 2) { test_size = strtol(argv[1], nullptr, 10); } if (argc >= 3) { sample_length = pow_for_sample(strtol(argv[2], nullptr, 10)); } if (argc >= 4) { total_row = strtol(argv[3], nullptr, 10); } readFile("normal_distribution.txt", sample_length, total_row, test_size); printf("population: %zu, skip: %lu, test size: %zu, sample length: %zu\n", population_vector.size(), total_row, test_size, sample_length); sample_vector = std::vector( population_vector.begin(), population_vector.begin() + sample_length - 2); sample_vector.push_back(min_value); sample_vector.push_back(max_value); std::sort(sample_vector.begin(), sample_vector.end()); le_sample_vector = sample_vector; initCustomSampleCPU(); key_type *cudaSample = nullptr, *cudaPopulation; P_ERR("Coping sample"); cudaMalloc(&cudaSample, sizeof(key_type) * sample_length); assert(sample_length == sample_vector.size()); cudaMemcpy(cudaSample, sample_vector.data(), sizeof(key_type) * sample_length, cudaMemcpyHostToDevice); cudaMalloc(&cudaPopulation, sizeof(key_type) * test_size); P_ERR("\rCoping population"); cudaMemcpy(cudaPopulation, population_vector.data() + sample_length, sizeof(key_type) * test_size, cudaMemcpyHostToDevice); // auto custom_sort = CustomSort(SAMPLE_LENGTH, sizeof(long) * 8); // custom_sort.testCalculation(); P_ERR("\rApply custom length and test calculation"); applyCudaSampleLength<<<1, 1>>>(sample_length); // testCustomCalculation<<<1, 1>>>(sample_length); cudaDeviceSynchronize(); P_ERR("\rApply custom length and test calculation completed"); // Can remove this function if pass // CustomSort::testSelf(sample_length); initSample<<<1, 1>>>(cudaSample, cudaPopulation); cudaDeviceSynchronize(); // check_items<<<1, 1>>>(); // cudaDeviceSynchronize(); P_ERR("\r \rRunning " "kernel\n"); run_kernel(test_size); freeStorageStage1<<<1, 1>>>(); cudaDeviceSynchronize(); key_type *cudaNormalSample = nullptr; P_ERR("\nCoping new sample"); cudaMalloc(&cudaNormalSample, sizeof(key_type) * sample_length); cudaMemcpy(cudaNormalSample, le_sample_vector.data(), sizeof(key_type) * sample_length, cudaMemcpyHostToDevice); /*cudaMemcpy(sample_vector.data(), cudaSample, sizeof(key_type) * sample_length, cudaMemcpyDeviceToHost);*/ initNormalSample<<<1, 1>>>(cudaNormalSample); cudaDeviceSynchronize(); P_ERR("\r "); P_ERR("\rRunning kernel2\n"); run_kernel(test_size, false); // check_items<<<1, 1>>>(); // cudaDeviceSynchronize(); puts(""); freeStorageStage2<<<1, 1>>>(); cudaDeviceSynchronize(); }