diff --git a/README.md b/README.md index e7b2831..8a811ac 100644 --- a/README.md +++ b/README.md @@ -161,15 +161,16 @@ cd benchmark && nvcc -O3 -o nvidia_bench nvidia_bench.cu -lnvidia-ml -ldl \ ## What Gets Unlocked -| Feature | Status | -|------------------------------------------------------------------|-------------| -| Full SM compute throughput (SS0/SS1) | Working | -| Memory geometry (64GB on 8GB cards, 40GB on 10GB cards) | Working | -| PCIe Gen 2 speeds | Working | -| GPU-to-GPU P2P (`cudaDeviceEnablePeerAccess`) | In progress | -| HBM2e memory overclock/downclock | Working | -| Persistence across kernel updates (auto-rebuild) (anti-rollback) | Working | -| BAR1 64mb->64gb (requires Above 4G Decoding in BIOS) | Working | +| Feature | Status | +|------------------------------------------------------------------|--------------| +| Full SM compute throughput (SS0/SS1) | ✅ | +| Memory geometry (64GB on 8GB cards, 40GB on 10GB cards) | ✅ | +| PCIe Gen 2 speeds | ✅ | +| GPU-to-GPU P2P (`cudaDeviceEnablePeerAccess`) | In progress | +| HBM2e memory overclock/downclock | ✅ | +| Persistence across kernel updates (auto-rebuild) (anti-rollback) | ✅ | +| BAR1 64mb->64gb (requires Above 4G Decoding in BIOS) | ✅ | +| PMA mem region fix | ✅ | --- @@ -194,6 +195,14 @@ This removes the patched modules from disk, undoes the kernel-update hooks, rele `--reload` swaps the running driver for the stock one immediately instead of waiting for the reboot. It is off by default because loading the stock `nvidia-drm` against a CMP 170HX can wedge the machine: the card has no usable display engine, and the kernel keeps answering pings while userspace stops making progress. There is no reason to take that risk during an uninstall you are going to reboot from anyway. +## Credits + +| Who | Contribution | +|--------------------------------------|-----------------------------------------------| +| [@bayley](https://github.com/bayley) | GPU-to-GPU P2P over BAR1, PMA WPR overlap fix | +| JP | Extra special thanks | +| Humvee55 | Extra special thanks | + ## Community Join our [Discord community](https://discord.gg/CdHSakKSFv) to discuss with other users. diff --git a/driver/src/cmpunlock.c b/driver/src/cmpunlock.c index 1a70222..016a83c 100644 --- a/driver/src/cmpunlock.c +++ b/driver/src/cmpunlock.c @@ -1191,6 +1191,22 @@ cmpUnlockLateExtendPma(OBJGPU *pGpu) if (!cmpUnlockIsTarget(pGpu)) return NV_OK; + /* + * The candidate region (base=0xff7300000 limit=0xfffffffff, ~141 MB) + * overlaps the WPR (Write Protected Region) and the GSP heap. If PMA + * hands out pages from that range, Copy-Engine writes hit a hardware + * region-violation fault: + * + * Xid 31 "MMU Fault: ENGINE CE2 HUBCLIENT_HSCE2 ... + * FAULT_INFO_TYPE_REGION_VIOLATION ACCESS_TYPE_VIRT_WRITE" + * + * Cost of skipping: ~141 MB out of 63.5 GiB (0.22 %). + */ + NV_PRINTF(LEVEL_WARNING, + "CMPUNLOCK_PMA: extension SKIPPED — " + "candidate region overlaps WPR + GSP heap\n"); + return NV_OK; + pMemoryManager = GPU_GET_MEMORY_MANAGER(pGpu); if (pMemoryManager == NULL) return NV_ERR_INVALID_ARGUMENT;