Товарищи, может кто-нибудь разбирается в технологии CUDA и может дать совет новичку вт этом деле. Пишу самую простую программу поэлементного перемножения 2-х комплексных векторов и сравниваю скорость работы кода. Программа, в которой вычисление производится CPU:
| Код | #include<iostream>
#include <helper_functions.h> // helper functions for SDK examples
#define SIGNAL_SIZE 8192
typedef double2 Complex; using std::cout; using std::endl;
int main() { StopWatchInterface *timer = 0; sdkCreateTimer(&timer); // Allocate host memory for the signal Complex *h_signal = (Complex *)malloc(sizeof(Complex) *SIGNAL_SIZE); Complex *h_pr_signal = (Complex *)malloc(sizeof(Complex) * SIGNAL_SIZE); for (unsigned int i = 0; i < SIGNAL_SIZE; ++i) { h_signal[i].x = i; h_signal[i].y = i; //cout<<"<"<< h_signal[i].x<<","<<h_signal[i].y<<">"<<"\n"; } sdkStartTimer(&timer); for (unsigned int i = 0; i < SIGNAL_SIZE; ++i) { h_pr_signal[i].x=h_signal[i].x*h_signal[i].x-h_signal[i].y*h_signal[i].y; h_pr_signal[i].y=2*h_signal[i].y*h_signal[i].x; }
sdkStopTimer(&timer);
printf("Processing time: %f (ms)\n", sdkGetTimerValue(&timer)); sdkDeleteTimer(&timer); for (unsigned int i = 0; i < SIGNAL_SIZE; ++i) { if(i==2||i==SIGNAL_SIZE/2||i==SIGNAL_SIZE) cout<<"<"<< h_pr_signal[i].x<<","<<h_pr_signal[i].y<<">"<<"\n"; } // cleanup memory free(h_signal); free(h_pr_signal); }
|
Программа, в которой вычисление производится GPU:
| Код | #include<iostream> #include <cufft.h> #include <cuda_runtime.h> #include <helper_functions.h> // helper functions for SDK examples
#define SIGNAL_SIZE 8192
typedef double2 Complex; using std::cout; using std::endl; __global__ void MultiVector(cufftDoubleComplex *d_signal, cufftDoubleComplex *d_pr_signal) { // Получаем id текущей нити. int idx = threadIdx.x+blockIdx.x*blockDim.x; // Расчитываем результат. d_pr_signal[idx].x = d_signal[idx].x*d_signal[idx].x-d_signal[idx].y*d_signal[idx].y; d_pr_signal[idx].y = 2*d_signal[idx].x*d_signal[idx].y; }
int main() { cudaSetDevice(0); StopWatchInterface *timer = 0; sdkCreateTimer(&timer); // Allocate host memory for the signal Complex *h_signal = (Complex *)malloc(sizeof(Complex) *SIGNAL_SIZE); Complex *h_pr_signal = (Complex *)malloc(sizeof(Complex) * SIGNAL_SIZE); for (unsigned int i = 0; i < SIGNAL_SIZE; ++i) { h_signal[i].x = i; h_signal[i].y =i; // cout<<"<"<< h_signal[i].x<<","<<h_signal[i].y<<">"<<"\n"; } cufftDoubleComplex *d_signal; cufftDoubleComplex *d_pr_signal; cudaError_t cuerr=cudaMalloc((void **)&d_signal,sizeof(cufftDoubleComplex)*SIGNAL_SIZE); if(cuerr!=cudaSuccess) cout<<"Cannot create GPU memory buffer for d_signal"<<endl; cudaError_t cuerr1=cudaMalloc((void **)&d_pr_signal,sizeof(cufftDoubleComplex)*SIGNAL_SIZE); if(cuerr1!=cudaSuccess) cout<<"Cannot create GPU memory buffer for d_signal"<<endl; sdkStartTimer(&timer);
cudaError_t cudaResult=cudaMemcpy(d_signal,h_signal,sizeof(cufftDoubleComplex)*SIGNAL_SIZE,cudaMemcpyHostToDevice); if(cudaResult!=cudaSuccess) cout<<"Could not copy data from device to host"<<endl; dim3 gridSize = dim3(SIGNAL_SIZE/512,1,1); dim3 blockSize = dim3(512,1,1);
MultiVector<<<gridSize,blockSize>>>(d_signal, d_pr_signal); //Хендл event’а cudaEvent_t syncEvent; cudaEventCreate(&syncEvent); //Создаем event cudaEventRecord(syncEvent, 0); //Записываем event cudaEventSynchronize(syncEvent); //Синхронизируем event cudaError_t cudaResult2=cudaMemcpy(h_pr_signal, d_pr_signal,sizeof(cufftDoubleComplex)*SIGNAL_SIZE,cudaMemcpyDeviceToHost); if(cudaResult2!=cudaSuccess) cout<<"Could not copy data from device to host"<<endl; sdkStopTimer(&timer);
printf("Processing time: %f (ms)\n", sdkGetTimerValue(&timer)); sdkDeleteTimer(&timer); for (unsigned int i = 0; i < SIGNAL_SIZE; ++i) { if(i==0||i==SIGNAL_SIZE/2||i==SIGNAL_SIZE-1) cout<<"<"<< h_pr_signal[i].x<<","<<h_pr_signal[i].y<<">"<<"\n"; }
// cleanup memory free(h_signal); free(h_pr_signal); cudaFree(d_signal); cudaFree(d_pr_signal); cudaDeviceReset(); }
|
Результаты работы этих программ во вложении) Подскажите должно ли так быть(или я что=то делаю неправильно?) и почему? Заранее спасибо за помошь!!! P.S. Видеокарта NDIVIA GeForse GTX 650 проц Intel Core 2 VS2012 CUDA Tollkit 5.5 |