CUDA: Difference between revisions

From CLONWiki
Jump to navigation Jump to search
Boiarino (talk | contribs)
No edit summary
Boiarino (talk | contribs)
No edit summary
Line 50: Line 50:
  |  No running processes found                                                            |
  |  No running processes found                                                            |
  +-----------------------------------------------------------------------------------------+
  +-----------------------------------------------------------------------------------------+
'''Compile and run test program'''
/* cudatest.cu */
#include <iostream>
#include <cstdio>
// A tiny GPU kernel that actually prints from the GPU
__global__ void helloFromGPU() {
    // Only let the very first thread print to avoid flooding the console
    if (threadIdx.x == 0 && blockIdx.x == 0) {
        printf("Hello World from the GPU!\n");
    }
}
int main() {
    std::cout << "[1] Program started on CPU..." << std::endl;
    // Check if a CUDA-capable GPU is even visible
    int deviceCount = 0;
    cudaError_t err = cudaGetDeviceCount(&deviceCount);
   
    if (err != cudaSuccess) {
        std::cout << "❌ CUDA Error: " << cudaGetErrorString(err) << std::endl;
        std::cout << "Reason: Your system might lack an NVIDIA GPU, or drivers are missing." << std::endl;
        return 1;
    }
    std::cout << "[2] Found " << deviceCount << " CUDA-capable GPU(s)." << std::endl;
    // Launch 1 block with 1 thread
    std::cout << "[3] Launching GPU kernel..." << std::endl;
    helloFromGPU<<<1, 1>>>();
    // Check if the kernel launch itself failed
    err = cudaGetLastError();
    if (err != cudaSuccess) {
        std::cout << "❌ Kernel Launch Error: " << cudaGetErrorString(err) << std::endl;
        return 1;
    }
    // Force CPU to wait and flush the GPU's printf buffer
    err = cudaDeviceSynchronize();
    if (err != cudaSuccess) {
        std::cout << "❌ Synchronization Error: " << cudaGetErrorString(err) << std::endl;
        return 1;
    }
    std::cout << "[4] Program finished successfully on CPU." << std::endl;
    return 0;
}

Revision as of 14:33, 15 September 2026

Some notes for RHEL9.8 on clonfarm11:

Find installed cards:

[root@clonfarm11 ~]# lspci | grep -i nvidia
2c:00.0 3D controller: NVIDIA Corporation TU104GL [Tesla T4] (rev a1)
31:00.0 3D controller: NVIDIA Corporation TU104GL [Tesla T4] (rev a1)

Install software:

Disable jlab/rhel repo with cuda first, and get cuda repo from nvidia website ! Than, install:

[root@clonfarm11 ~]# yum repolist all
.....
cuda-rhel9-x86_64       cuda-rhel9-x86_64                   enabled
.....
dnf install cuda
dnf install nvidia-driver
dnf install nvidia-driver-cuda
[root@clonfarm11 ~]# ls /usr/src | grep nvidia
nvidia-615.71.09
[root@clonfarm11 ~]# dkms install -m nvidia -v 615.71.09

[root@clonfarm11 ~]# nvidia-smi

Tue Sep 15 14:26:10 2026       
+-----------------------------------------------------------------------------------------+
| NVIDIA-SMI 615.71.09              KMD Version: 615.71.09     CUDA UMD Version: 13.4     |
+-----------------------------------------+------------------------+----------------------+
| GPU  Name                 Persistence-M | Bus-Id          Disp.A | Volatile Uncorr. ECC |
| Fan  Temp   Perf          Pwr:Usage/Cap |           Memory-Usage | GPU-Util  Compute M. |
|                                         |                        |               MIG M. |
|=========================================+========================+======================|
|   0  Tesla T4                       Off |   00000000:2C:00.0 Off |                    0 |
| N/A   31C    P8              9W /   70W |       0MiB /  15360MiB |      0%      Default |
|                                         |                        |                  N/A |
+-----------------------------------------+------------------------+----------------------+
|   1  Tesla T4                       Off |   00000000:31:00.0 Off |                    0 |
| N/A   30C    P8              9W /   70W |       0MiB /  15360MiB |      0%      Default |
|                                         |                        |                  N/A |
+-----------------------------------------+------------------------+----------------------+

+-----------------------------------------------------------------------------------------+
| Processes:                                                                              |
|  GPU   GI   CI              PID   Type   Process name                        GPU Memory |
|        ID   ID                                                               Usage      |
|=========================================================================================|
|  No running processes found                                                             |
+-----------------------------------------------------------------------------------------+


Compile and run test program

/* cudatest.cu */

#include <iostream>
#include <cstdio>

// A tiny GPU kernel that actually prints from the GPU
__global__ void helloFromGPU() {
    // Only let the very first thread print to avoid flooding the console
    if (threadIdx.x == 0 && blockIdx.x == 0) {
        printf("Hello World from the GPU!\n");
    }
}

int main() {
    std::cout << "[1] Program started on CPU..." << std::endl;

    // Check if a CUDA-capable GPU is even visible
    int deviceCount = 0;
    cudaError_t err = cudaGetDeviceCount(&deviceCount);
    
    if (err != cudaSuccess) {
        std::cout << "❌ CUDA Error: " << cudaGetErrorString(err) << std::endl;
        std::cout << "Reason: Your system might lack an NVIDIA GPU, or drivers are missing." << std::endl;
        return 1;
    }

    std::cout << "[2] Found " << deviceCount << " CUDA-capable GPU(s)." << std::endl;

    // Launch 1 block with 1 thread
    std::cout << "[3] Launching GPU kernel..." << std::endl;
    helloFromGPU<<<1, 1>>>();

    // Check if the kernel launch itself failed
    err = cudaGetLastError();
    if (err != cudaSuccess) {
        std::cout << "❌ Kernel Launch Error: " << cudaGetErrorString(err) << std::endl;
        return 1;
    }

    // Force CPU to wait and flush the GPU's printf buffer
    err = cudaDeviceSynchronize();
    if (err != cudaSuccess) {
        std::cout << "❌ Synchronization Error: " << cudaGetErrorString(err) << std::endl;
        return 1;
    }

    std::cout << "[4] Program finished successfully on CPU." << std::endl;
    return 0;
}