diff --git a/driver/build.sh b/driver/build.sh index 082a333..1374174 100755 --- a/driver/build.sh +++ b/driver/build.sh @@ -270,6 +270,7 @@ done ok "Installed ${#INSTALLED[@]} modules: ${INSTALLED[*]}" depmod -a "${KVER}" +sync # flush modules.dep to disk — important if VM is hard-killed before reboot INITRAMFS_LOG="${BUILD_ROOT}/initramfs.log" @@ -315,6 +316,7 @@ if [[ -n "${resolved}" ]]; then warn "Resolved nvidia.ko is not under updates/cmpunlocker/" fi fi +loads_before="$(dmesg 2>/dev/null | grep -c 'loading NVIDIA UNIX' || true)" info "Attempting to unload NVIDIA modules..." systemctl stop nvidia-persistenced 2>/dev/null || true systemctl stop nvidia-fabricmanager 2>/dev/null || true @@ -332,10 +334,16 @@ if ! lsmod | grep -q '^nvidia '; then modprobe nvidia-drm 2>/dev/null || true reload_ok=1 ok "Patched NVIDIA modules loaded" - running_src="$(cat /sys/module/nvidia/srcversion 2>/dev/null || true)" - patched_src="$(modinfo -F srcversion "${INSTALL_MOD_DIR}/nvidia.ko" 2>/dev/null || true)" - if [[ -n "${running_src}" && -n "${patched_src}" && "${running_src}" != "${patched_src}" ]]; then - warn "Loaded nvidia srcversion (${running_src}) != patched (${patched_src})" + # + # srcversion cannot answer this: it hashes the kernel-open sources, + # and cmpunlock.c arrives as part of the prebuilt nv-kernel.o blob, so + # it stays identical no matter what the unlock does. Count the module's + # own load banner instead - if it did not go up, nothing was reloaded + # and the driver in memory is still the previous build. + # + loads_after="$(dmesg 2>/dev/null | grep -c 'loading NVIDIA UNIX' || true)" + if [[ "${loads_after}" == "${loads_before}" ]]; then + warn "nvidia did not actually reload — the running driver is still the old build" reload_ok=0 fi else diff --git a/driver/patches/0008-scrub-timeout.patch b/driver/patches/0008-scrub-timeout.patch new file mode 100644 index 0000000..fa155b3 --- /dev/null +++ b/driver/patches/0008-scrub-timeout.patch @@ -0,0 +1,29 @@ +--- a/src/nvidia/src/kernel/gpu/mem_mgr/mem_scrub.c ++++ b/src/nvidia/src/kernel/gpu/mem_mgr/mem_scrub.c +@@ -274,7 +274,7 @@ + if (!API_GPU_IN_RESET_SANITY_CHECK(pGpu)) + { + RMTIMEOUT timeout; +- gpuSetTimeout(pGpu, GPU_TIMEOUT_DEFAULT, &timeout, 0); ++ gpuSetTimeout(pGpu, 25000000, &timeout, GPU_TIMEOUT_FLAGS_OSTIMER); + + while (_isScrubWorkPending(pScrubber)) + { +@@ -871,7 +871,7 @@ + if (itemsToSave == 0) + goto done; + +- gpuSetTimeout(pGpu, GPU_TIMEOUT_DEFAULT, &timeout, 0); ++ gpuSetTimeout(pGpu, 25000000, &timeout, GPU_TIMEOUT_FLAGS_OSTIMER); + + while (_scrubCheckProgress(pScrubber) < (pScrubber->lastSeenIdByClient + itemsToSave)) + { +@@ -928,7 +928,7 @@ + return NV_OK; + } + +- gpuSetTimeout(pGpu, GPU_TIMEOUT_DEFAULT, &timeout, 0); ++ gpuSetTimeout(pGpu, 25000000, &timeout, GPU_TIMEOUT_FLAGS_OSTIMER); + + // Loop will break out, when the semaphore is equal to payload, or times out + while (_scrubCheckProgress(pScrubber) < idToWait) diff --git a/driver/src/cmpunlock.c b/driver/src/cmpunlock.c index 628da24..b2b9c2f 100644 --- a/driver/src/cmpunlock.c +++ b/driver/src/cmpunlock.c @@ -175,6 +175,33 @@ cmpUnlockIsTarget(OBJGPU *pGpu) return (devId == CMPUNLOCK_DEVID_8GB || devId == CMPUNLOCK_DEVID_10GB); } +/* + * Escape hatch. + * + * The optional features are compiled in, so a value that wedges the GPU during + * driver init leaves a machine that cannot boot far enough to reinstall - the + * only way out would be pulling the card. This gives a way to skip all of them + * from the boot loader instead: + * + * nvidia.NVreg_RegistryDwords=cmpSafe=1 + * + * The base unlock (memory geometry, SM, PCIe) still runs; only the tunables + * that can be set to a bad value are skipped. + */ +static NvBool +cmpUnlockSafeMode(OBJGPU *pGpu) +{ + NvU32 data = 0; + + if (osReadRegistryDword(pGpu, "cmpSafe", &data) == NV_OK && data != 0) + { + NV_PRINTF(LEVEL_ERROR, + "CMPUNLOCK: cmpSafe=1 - skipping MCLK overclock and timing scaling\n"); + return NV_TRUE; + } + return NV_FALSE; +} + /* * Opt-in, because forcing the caps is not the same as P2P working. * @@ -929,7 +956,8 @@ cmpUnlockPostBooterLoad(OBJGPU *pGpu, KernelGsp *pKernelGsp) /* * Booter Load is the last thing that can silently undo the unlock, so - * report what the gated registers actually ended up holding. + * report what the gated registers actually ended up holding. Worth having + * in safe mode too - that is when you are diagnosing something. */ NV_PRINTF(LEVEL_ERROR, "CMPUNLOCK: post-BooterLoad verify PLM=0x%08x SS0=0x%08x SS1=0x%08x " @@ -940,6 +968,9 @@ cmpUnlockPostBooterLoad(OBJGPU *pGpu, KernelGsp *pKernelGsp) GPU_REG_RD32(pGpu, CMP_REG_FBPA_CFG1), GPU_REG_RD32(pGpu, CMP_REG_MMU_LMR)); + if (cmpUnlockSafeMode(pGpu)) + return; + #ifdef CMPUNLOCK_MCLK_TIMINGS /* Must precede the PLL cycle below: the clock has to come up on the loosened timings, not land on the stock ones and then be widened. */ @@ -1058,6 +1089,8 @@ cmpUnlockMclkPostGsp(OBJGPU *pGpu, KernelGsp *pKernelGsp) #if defined(CMPUNLOCK_MCLK_TIMINGS) || defined(CMPUNLOCK_MCLK_NDIV) if (!cmpUnlockIsTarget(pGpu)) return; + if (cmpUnlockSafeMode(pGpu)) + return; #endif #ifdef CMPUNLOCK_MCLK_TIMINGS