diff --git a/README.md b/README.md index e2d679f..6460942 100644 --- a/README.md +++ b/README.md @@ -70,21 +70,36 @@ Or add `iommu=pt` to your kernel cmdline manually. IOMMU must also be enabled in --- -## Verify & Benchmark +## Verify -After install and cold reboot, run the built-in GPU benchmark: +After install and cold reboot: ```bash -./benchmark/nvidia_bench +# Memory and SMs — 8GB card should show 65536 MiB, 10GB card 40960 MiB +nvidia-smi --query-gpu=index,memory.total,pci.bus_id --format=csv + +# Unlock logs +sudo dmesg | grep CMPUNLOCK + +# P2P support (multi-GPU) — should show PIX/PHB/SYS/OK, not GNS +nvidia-smi topo -m +nvidia-smi topo -p2p r + +# PCIe link speed, performance, memory amount, memory speed, SM count... +./benchmark/nvidia_bench [--cuda_gpu_id] + ``` -This measures memory bandwidth, tensor core throughput, PCIe speed, SM clock, and reports hardware features. An unlocked 8GB card should show ~64 GiB total memory and 56/70 SMs; a 10GB card should show ~40 GiB and 54/70 SMs. +### Benchmark ```bash -./benchmark/nvidia_bench 0 # test GPU 0 (default) +./benchmark/nvidia_bench # test GPU 0 +./benchmark/nvidia_bench 0 # explicit GPU index ./benchmark/nvidia_bench --csv # machine-readable output ``` +Measures memory bandwidth, tensor core throughput (TF32/BF16/INT8), PCIe H2D/D2H speed, SM clock, and reports hardware features (NVENC/NVDEC). + A pre-built x86-64 binary is included. On aarch64 (or to rebuild), install the CUDA toolkit and build from source: ```bash @@ -99,6 +114,7 @@ cd benchmark && nvcc -O3 -o nvidia_bench nvidia_bench.cu -lnvidia-ml -ldl \ | Full SM compute throughput (SS0/SS1) | Working | | Memory geometry (64GB on 8GB cards, 40GB on 10GB cards) | Working | | PCIe Gen 2 speeds | Working | +| GPU-to-GPU P2P (cudaDeviceEnablePeerAccess) | Working | | HBM2 memory overclock/downclock | Working | | Persistence across reboot (patched modules) | Working | diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index f739151..23ed692 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -48,6 +48,7 @@ driver/ 0004-memmgr-quirks.patch mem_mgr*, mem_scrub 3 conditionals 0005-bar0-pramin-clamp.patch kern_bus_gm107.c 1 conditional 0006-persistent-sw-state.patch nv.c 1 conditional + 0007-p2p-caps-hook.patch gpu.c 1 hook ``` `build.sh` copies `driver/src/*` into `src/nvidia/{src,inc}/kernel/gpu/cmpunlock/`, appends one `SRCS +=` line to `src/nvidia/srcs.mk`, then applies the patches. @@ -141,7 +142,20 @@ The unlock is applied by **patched kernel modules**, not a userspace daemon: - Every time the driver initializes (on boot or after a reload), the patched sequence runs - The unlock persists indefinitely until `./remove.sh` is run -Card profile (8GB vs 10GB) is stored in `/lib/modules/$(uname -r)/updates/cmpunlocker/card_profile` at install time and used during every boot. +Card profile (8GB vs 10GB) is determined at runtime from PCI device ID. + +--- + +### GPU-to-GPU P2P + +GSP firmware reports `pcieP2PReadCaps` / `pcieP2PWriteCaps` = NOT_SUPPORTED for CMP device IDs. This blocks `cudaDeviceEnablePeerAccess()` even though the GA100 PCIe mailbox P2P hardware works. + +`cmpUnlockForceP2PCaps()` overrides the GSP response to OK after the RPC returns in `_gpuInitPcieP2PCapability()`. The driver then uses `P2P_CONNECTIVITY_PCIE_PROPRIETARY` (mailbox protocol through BAR0/PRAMIN) which does not require large BAR1. + +Expected dmesg output: +``` +CMPUNLOCK: PCIe P2P caps forced to OK +``` --- diff --git a/driver/patches/0007-p2p-caps-hook.patch b/driver/patches/0007-p2p-caps-hook.patch new file mode 100644 index 0000000..5fedacb --- /dev/null +++ b/driver/patches/0007-p2p-caps-hook.patch @@ -0,0 +1,19 @@ +diff --git a/src/nvidia/src/kernel/gpu/gpu.c b/src/nvidia/src/kernel/gpu/gpu.c +--- a/src/nvidia/src/kernel/gpu/gpu.c ++++ b/src/nvidia/src/kernel/gpu/gpu.c +@@ -77,6 +77,7 @@ + #include "platform/chipset/chipset.h" + #include "kernel/gpu/host_eng/host_eng.h" + #include "gpu/bif/kernel_bif.h" ++#include "gpu/cmpunlock/cmpunlock.h" + #include "gpu/ce/kernel_ce.h" + #include "kernel/gpu/mem_mgr/mem_mgr.h" + #include "kernel/gpu/mig_mgr/kernel_mig_manager.h" +@@ -2517,6 +2518,7 @@ + + pGpu->pcieP2PReadCaps = p2pCapsParams.p2pReadCapsStatus; + pGpu->pcieP2PWriteCaps = p2pCapsParams.p2pWriteCapsStatus; ++ cmpUnlockForceP2PCaps(pGpu); + } + + static void diff --git a/driver/src/cmpunlock.c b/driver/src/cmpunlock.c index 1e98574..00aa093 100644 --- a/driver/src/cmpunlock.c +++ b/driver/src/cmpunlock.c @@ -159,6 +159,18 @@ cmpUnlockIsTarget(OBJGPU *pGpu) return (devId == CMPUNLOCK_DEVID_8GB || devId == CMPUNLOCK_DEVID_10GB); } +void +cmpUnlockForceP2PCaps(OBJGPU *pGpu) +{ + if (!cmpUnlockIsTarget(pGpu)) + return; + + pGpu->pcieP2PReadCaps = 0; + pGpu->pcieP2PWriteCaps = 0; + + NV_PRINTF(LEVEL_ERROR, "CMPUNLOCK: PCIe P2P caps forced to OK\n"); +} + static NvU64 _cmpUnlockedFbBytes(OBJGPU *pGpu) { diff --git a/driver/src/cmpunlock.h b/driver/src/cmpunlock.h index 6cd7022..f59f773 100644 --- a/driver/src/cmpunlock.h +++ b/driver/src/cmpunlock.h @@ -93,4 +93,13 @@ void cmpUnlockMclkPostGsp(OBJGPU *pGpu, KernelGsp *pKernelGsp); */ NV_STATUS cmpUnlockLateExtendPma(OBJGPU *pGpu); +/* + * Force PCIe P2P caps to OK. GSP firmware reports NOT_SUPPORTED for CMP + * cards, but the GA100 mailbox P2P hardware works fine. This overrides the + * GSP response so cudaDeviceEnablePeerAccess() succeeds. + * + * Hook: _gpuInitPcieP2PCapability() in gpu.c, after the GSP RPC. + */ +void cmpUnlockForceP2PCaps(OBJGPU *pGpu); + #endif /* CMPUNLOCK_H */