我有一个 C++ 项目
是否可以为程序生命周期中的 CUDA 内容创建一个类
当然。我不确定extern "C" 的东西与 C++ 项目有什么关系。 CUDA 是一种 C++ 类型的语言定义。
这是一个例子:
$ cat t1901.cu
#include <iostream>
#include <vector>
__global__ void my_work(int N, double *X, double *Y, double *scale){
int idx = threadIdx.x+blockDim.x*blockIdx.x;
if (idx < N)
Y[idx] += X[idx] * (*scale);
}
class stuffHandler
{
double persistent;
double *dev_persistent = NULL;
public:
stuffHandler(double persistent_) : persistent(persistent_) {
cudaMalloc(&dev_persistent, sizeof(double));
cudaMemcpy(dev_persistent, &persistent, sizeof(double), cudaMemcpyHostToDevice);}
void doStuff(int N, double *arg1, double *arg2){
double *d_arg1, *d_arg2;
cudaMalloc(&d_arg1, N*sizeof(double));
cudaMalloc(&d_arg2, N*sizeof(double));
cudaMemcpy(d_arg1, arg1, N*sizeof(double), cudaMemcpyHostToDevice);
cudaMemcpy(d_arg2, arg2, N*sizeof(double), cudaMemcpyHostToDevice);
my_work<<<(N+255)/256, 256>>>(N, d_arg1, d_arg2, dev_persistent);
cudaMemcpy(arg2, d_arg2, N*sizeof(double), cudaMemcpyDeviceToHost);
cudaFree(d_arg1);
cudaFree(d_arg2);}
~stuffHandler(){if (dev_persistent) cudaFree(dev_persistent);}
};
int main(){
int my_N = 4;
double scale = 1.5;
stuffHandler stuff_handler(scale);
std::vector<double> v1(my_N, 0.1);
std::vector<double> v2(my_N, 0.2);
stuff_handler.doStuff(my_N, v1.data(), v2.data());
std::cout << v2[0] << std::endl;
}
$ nvcc -o t1901 t1901.cu
$ compute-sanitizer ./t1901
========= COMPUTE-SANITIZER
0.35
========= ERROR SUMMARY: 0 errors
$
回答cmets中的一个问题,您可以将上面的内容重新排列为一个典型的多模块项目实现:
$ cat t1901.h
class stuffHandler
{
double persistent;
double *dev_persistent = NULL;
public:
stuffHandler(double persistent_);
void doStuff(int N, double *arg1, double *arg2);
~stuffHandler();
};
$ cat t1901.cu
#include "t1901.h"
__global__ void my_work(int N, double *X, double *Y, double *scale){
int idx = threadIdx.x+blockDim.x*blockIdx.x;
if (idx < N)
Y[idx] += X[idx] * (*scale);
}
stuffHandler::stuffHandler(double persistent_) : persistent(persistent_) {
cudaMalloc(&dev_persistent, sizeof(double));
cudaMemcpy(dev_persistent, &persistent, sizeof(double), cudaMemcpyHostToDevice);}
void stuffHandler::doStuff(int N, double *arg1, double *arg2){
double *d_arg1, *d_arg2;
cudaMalloc(&d_arg1, N*sizeof(double));
cudaMalloc(&d_arg2, N*sizeof(double));
cudaMemcpy(d_arg1, arg1, N*sizeof(double), cudaMemcpyHostToDevice);
cudaMemcpy(d_arg2, arg2, N*sizeof(double), cudaMemcpyHostToDevice);
my_work<<<(N+255)/256, 256>>>(N, d_arg1, d_arg2, dev_persistent);
cudaMemcpy(arg2, d_arg2, N*sizeof(double), cudaMemcpyDeviceToHost);
cudaFree(d_arg1);
cudaFree(d_arg2);}
stuffHandler::~stuffHandler(){if (dev_persistent) cudaFree(dev_persistent);}
$ cat main.cpp
#include <iostream>
#include <vector>
#include "t1901.h"
int main(){
int my_N = 4;
double scale = 1.5;
stuffHandler stuff_handler(scale);
std::vector<double> v1(my_N, 0.1);
std::vector<double> v2(my_N, 0.2);
stuff_handler.doStuff(my_N, v1.data(), v2.data());
std::cout << v2[0] << std::endl;
}
$ nvcc -o t1901 t1901.cu main.cpp
$ compute-sanitizer ./t1901
========= COMPUTE-SANITIZER
0.35
========= ERROR SUMMARY: 0 errors
$
如果您想将编译和链接步骤分开,您可以这样做:
nvcc -c t1901.cu
g++ -c main.cpp
g++ main.o t1901.o -o test -L/usr/local/cuda/lib64 -lcudart
或者在 MPI 项目中,将上面的 g++ 替换为 mpicxx
在 cmets 中进行了一些额外的讨论后,与您的问题标题和第一句话相反,您实际上有一个 C 项目,而不是 C++(您希望使用 C 编译器 mpicc 进行最终链接)。
在这种情况下,我们可以对上面的代码进行一些不同的布局,并参考一些指令here 以按顺序获取所有 C++ 链接。这是另一个例子:
$ cat t1902.h
#ifdef __cplusplus
extern "C"
#endif
void C_init(double scale);
#ifdef __cplusplus
extern "C"
#endif
void C_doStuff(int N, double *arg1, double *arg2);
#ifdef __cplusplus
extern "C"
#endif
void C_end();
$ cat t1902.cu
#include <iostream>
#include <vector>
__global__ void my_work(int N, double *X, double *Y, double *scale){
int idx = threadIdx.x+blockDim.x*blockIdx.x;
if (idx < N)
Y[idx] += X[idx] * (*scale);
}
class stuffHandler
{
double persistent;
double *dev_persistent = NULL;
public:
stuffHandler(double persistent_) : persistent(persistent_) {
cudaMalloc(&dev_persistent, sizeof(double));
cudaMemcpy(dev_persistent, &persistent, sizeof(double), cudaMemcpyHostToDevice);}
void doStuff(int N, double *arg1, double *arg2){
double *d_arg1, *d_arg2;
cudaMalloc(&d_arg1, N*sizeof(double));
cudaMalloc(&d_arg2, N*sizeof(double));
cudaMemcpy(d_arg1, arg1, N*sizeof(double), cudaMemcpyHostToDevice);
cudaMemcpy(d_arg2, arg2, N*sizeof(double), cudaMemcpyHostToDevice);
my_work<<<(N+255)/256, 256>>>(N, d_arg1, d_arg2, dev_persistent);
cudaMemcpy(arg2, d_arg2, N*sizeof(double), cudaMemcpyDeviceToHost);
cudaFree(d_arg1);
cudaFree(d_arg2);}
void finish(){if (dev_persistent) cudaFree(dev_persistent);}
};
stuffHandler *stuff_handler = NULL;
extern "C" void C_doStuff(int N, double *arg1, double *arg2){
if (stuff_handler)
stuff_handler->doStuff(N, arg1, arg2);
}
extern "C" void C_end(){
if (stuff_handler) {
stuff_handler->finish();
delete stuff_handler;}
stuff_handler = NULL;
}
extern "C" void C_init(double scale){
if (stuff_handler) C_end();
stuff_handler = new stuffHandler(scale);
}
$ cat main.c
#include <stdio.h>
#include <stdlib.h>
#include "t1902.h"
int main(){
int i,my_N = 4;
double scale = 1.5;
C_init(scale);
double *d1 = malloc(my_N*sizeof(double));
double *d2 = malloc(my_N*sizeof(double));
for (i=0; i < my_N; i++) {
d1[i] = 0.1;
d2[i] = 0.2;}
C_doStuff(my_N, d1, d2);
printf("%f\n", d2[0]);
C_end();
}
$ nvcc -c t1902.cu
$ gcc -c main.c
$ gcc -o test main.o t1902.o -L/usr/local/cuda/lib64 -lcudart_static -lculibos -lpthread -lrt -ldl -lstdc++
$ compute-sanitizer ./test
========= COMPUTE-SANITIZER
0.350000
========= ERROR SUMMARY: 0 errors
$
在上面的编译序列中,应该可以将gcc替换为mpicc。