CUDA: Difference between revisions
Jump to navigation
Jump to search
No edit summary |
No edit summary |
||
| Line 52: | Line 52: | ||
Compile following test program by command ''nvcc cudatest.cu -o cudatest'': | |||
/* cudatest.cu */ | /* cudatest.cu */ | ||
| Line 103: | Line 103: | ||
return 0; | return 0; | ||
} | } | ||
Run it: | |||
clonfarm11:boiarino> ./cudatest | |||
[1] Program started on CPU... | |||
[2] Found 2 CUDA-capable GPU(s). | |||
[3] Launching GPU kernel... | |||
Hello World from the GPU! | |||
[4] Program finished successfully on CPU. | |||
Revision as of 14:36, 15 September 2026
Some notes for RHEL9.8 on clonfarm11:
Find installed cards:
[root@clonfarm11 ~]# lspci | grep -i nvidia 2c:00.0 3D controller: NVIDIA Corporation TU104GL [Tesla T4] (rev a1) 31:00.0 3D controller: NVIDIA Corporation TU104GL [Tesla T4] (rev a1)
Install software:
Disable jlab/rhel repo with cuda first, and get cuda repo from nvidia website ! Than, install:
[root@clonfarm11 ~]# yum repolist all ..... cuda-rhel9-x86_64 cuda-rhel9-x86_64 enabled .....
dnf install cuda dnf install nvidia-driver dnf install nvidia-driver-cuda
[root@clonfarm11 ~]# ls /usr/src | grep nvidia nvidia-615.71.09
[root@clonfarm11 ~]# dkms install -m nvidia -v 615.71.09
[root@clonfarm11 ~]# nvidia-smi
Tue Sep 15 14:26:10 2026 +-----------------------------------------------------------------------------------------+ | NVIDIA-SMI 615.71.09 KMD Version: 615.71.09 CUDA UMD Version: 13.4 | +-----------------------------------------+------------------------+----------------------+ | GPU Name Persistence-M | Bus-Id Disp.A | Volatile Uncorr. ECC | | Fan Temp Perf Pwr:Usage/Cap | Memory-Usage | GPU-Util Compute M. | | | | MIG M. | |=========================================+========================+======================| | 0 Tesla T4 Off | 00000000:2C:00.0 Off | 0 | | N/A 31C P8 9W / 70W | 0MiB / 15360MiB | 0% Default | | | | N/A | +-----------------------------------------+------------------------+----------------------+ | 1 Tesla T4 Off | 00000000:31:00.0 Off | 0 | | N/A 30C P8 9W / 70W | 0MiB / 15360MiB | 0% Default | | | | N/A | +-----------------------------------------+------------------------+----------------------+ +-----------------------------------------------------------------------------------------+ | Processes: | | GPU GI CI PID Type Process name GPU Memory | | ID ID Usage | |=========================================================================================| | No running processes found | +-----------------------------------------------------------------------------------------+
Compile following test program by command nvcc cudatest.cu -o cudatest:
/* cudatest.cu */
#include <iostream>
#include <cstdio>
// A tiny GPU kernel that actually prints from the GPU
__global__ void helloFromGPU() {
// Only let the very first thread print to avoid flooding the console
if (threadIdx.x == 0 && blockIdx.x == 0) {
printf("Hello World from the GPU!\n");
}
}
int main() {
std::cout << "[1] Program started on CPU..." << std::endl;
// Check if a CUDA-capable GPU is even visible
int deviceCount = 0;
cudaError_t err = cudaGetDeviceCount(&deviceCount);
if (err != cudaSuccess) {
std::cout << "❌ CUDA Error: " << cudaGetErrorString(err) << std::endl;
std::cout << "Reason: Your system might lack an NVIDIA GPU, or drivers are missing." << std::endl;
return 1;
}
std::cout << "[2] Found " << deviceCount << " CUDA-capable GPU(s)." << std::endl;
// Launch 1 block with 1 thread
std::cout << "[3] Launching GPU kernel..." << std::endl;
helloFromGPU<<<1, 1>>>();
// Check if the kernel launch itself failed
err = cudaGetLastError();
if (err != cudaSuccess) {
std::cout << "❌ Kernel Launch Error: " << cudaGetErrorString(err) << std::endl;
return 1;
}
// Force CPU to wait and flush the GPU's printf buffer
err = cudaDeviceSynchronize();
if (err != cudaSuccess) {
std::cout << "❌ Synchronization Error: " << cudaGetErrorString(err) << std::endl;
return 1;
}
std::cout << "[4] Program finished successfully on CPU." << std::endl;
return 0;
}
Run it:
clonfarm11:boiarino> ./cudatest [1] Program started on CPU... [2] Found 2 CUDA-capable GPU(s). [3] Launching GPU kernel... Hello World from the GPU! [4] Program finished successfully on CPU.