【发布时间】:2018-05-06 15:15:05
【问题描述】:
我有返回 3 个指针的 CUDA 函数:csrVal、csrRowPtr、csrColInd。
void dense2Csr (int dim,
cuComplex *dnMatr,
cuComplex *csrVal,
int *csrRowPtr,
int *csrColInd)
{
cusparseHandle_t cusparseH = NULL; // residual evaluation
cudaStream_t stream = NULL;
cusparseMatDescr_t descrA = NULL; // A is a base-0 general matrix
cusparseStatus_t cudaStat1 = CUSPARSE_STATUS_SUCCESS;
int nnZ;
//Input GPU Copy
cuComplex *d_dnMatr;
int *d_nnzRow;
//Output GPU Copy
cuComplex *d_csrVal;
int *d_csrRowPtr;
int *d_csrColInd;
cusparseCreate(&cusparseH); //Create SparseStructure
cudaStreamCreate(&stream);
cusparseSetStream(cusparseH, stream);
cusparseCreateMatDescr(&descrA);
cusparseSetMatType(descrA, CUSPARSE_MATRIX_TYPE_GENERAL);
cusparseSetMatIndexBase(descrA, CUSPARSE_INDEX_BASE_ZERO); //Set First Element RowPtr eq. to zero
cudaMalloc((void **)&d_dnMatr , sizeof(cuComplex)*dim*dim);
cudaMalloc((void **)&d_nnzRow , sizeof(int)*dim);
cudaMemcpy(d_dnMatr , dnMatr , sizeof(cuComplex)*dim*dim , cudaMemcpyHostToDevice);
cusparseCnnz(cusparseH,
CUSPARSE_DIRECTION_ROW,
dim,
dim,
descrA,
d_dnMatr,
dim,
d_nnzRow,
&nnZ);
cudaMalloc((void **)&d_csrRowPtr , sizeof(int)*(dim+1));
cudaMalloc((void **)&d_csrColInd , sizeof(int)*nnZ);
cudaMalloc((void **)&d_csrVal , sizeof(cuComplex)*nnZ);
cudaStat1 = cusparseCdense2csr(cusparseH,
dim,
dim,
descrA,
d_dnMatr,
dim,
d_nnzRow,
d_csrVal,
d_csrRowPtr,
d_csrColInd);
assert(cudaStat1 == CUSPARSE_STATUS_SUCCESS);
cudaMallocHost((void **)&csrRowPtr , sizeof(int)*(dim+1));
cudaMallocHost((void **)&csrColInd , sizeof(int)*nnZ);
cudaMallocHost((void **)&csrVal , sizeof(cuComplex)*nnZ);
cudaMemcpy(csrVal, d_csrVal, sizeof(cuComplex)*nnZ, cudaMemcpyDeviceToHost);
cudaMemcpy(csrRowPtr, d_csrRowPtr, sizeof(int)*(dim+1), cudaMemcpyDeviceToHost);
cudaMemcpy(csrColInd, d_csrColInd, sizeof(int)*(nnZ), cudaMemcpyDeviceToHost);
if (d_csrVal) cudaFree(d_csrVal);
if (d_csrRowPtr) cudaFree(d_csrRowPtr);
if (d_csrColInd) cudaFree(d_csrColInd);
if (cusparseH ) cusparseDestroy(cusparseH);
if (stream ) cudaStreamDestroy(stream);
我用 C 代码调用它(100% 正确链接):
dense2Csr(dim, Sigma, csrValSigma, csrRowPtrSigma, csrColIndSigma);
或
dense2Csr(dim, Sigma, &csrValSigma[0], &csrRowPtrSigma[0], &csrColIndSigma[0]);
它以两种方式写给我
Process finished with exit code 139 (interrupted by signal 11: SIGSEGV)
所以,这是一个内存错误,我只是通过在调用dense2Csr之前在主程序中分配一个主机内存(并且函数中没有cudaMallocHost)来解决它。但现在我无法以这种方式做到这一点。那么,有没有一种方法可以让函数吃空指针,并让它在这样的设置中返回一个指向内存区域的指针?
【问题讨论】:
-
通过引用而不是值传递指针。看看 cudaMalloc 是如何工作的。
-
嗯,诀窍是我使用的是 C,而不是 C++
-
cudaMalloc 也不是。 C 有一个通过引用的习语。它没有参考。如果需要在函数中修改 dim 怎么办?指针的答案也在这里
-
好吧,我会用 &dim 代替 dim,但正如我已经写过的,它不起作用。我不明白你。我没有提到的是dense2Csr函数在一个共享库中。
-
我将再次重复我的第一条评论。查看 cudaMalloc 的原型,然后问问自己为什么会这样。这正是您在这里所需要的