From 765ffba1f5cbc7eeb598d13abc2b0b2f1f5b76b2 Mon Sep 17 00:00:00 2001 From: aboba Date: Thu, 30 Jul 2026 16:09:13 +0200 Subject: [PATCH] Allow mem oc on any VBIOS and on 10gb cards 0007/0008 were gated on PCI 0x20C2 plus an exact VBIOS string, so only 8gb cards on the 300W 92.00.6D.00.0A build could clock. Device check is now 0x20C2 or 0x2082, and the stock NDIV is read out of the PLL instead of assumed. The post-GSP COEFF write hardcoded MDIV/PDIV for that one VBIOS and would have written wrong divisors anywhere else; it is a read-modify- write now, like the pre-GSP path already was. The NDIV literal is an XX placeholder, so a tree that build.sh did not substitute fails to compile instead of quietly running 1890 MHz. Rebased onto PCIe Gen2: upstream took 0007/0008 for its own patches, so the mclk pair is 0009/0010. Both it and Gen2 insert at the same anchor in kernel_gsp_tu102.c; the mclk block is placed after the Gen2 register writes. The three FBPA_PLL entries are merged into the plmTable that Gen2 also extended, and the loop bound is NV_ARRAY_ELEMENTS so the two additions cannot desync it again. install.sh warns when the PCI ID cannot clock and when a mixed 8gb+10gb inventory would get one multiplier, build.sh validates it and records it next to card_profile, and verify.sh reports it against dmesg. --- README.md | 23 +++ common/constants.yaml | 13 ++ docs/INSTALLATION.md | 11 ++ driver/build.sh | 25 ++- .../patches/0001-sec2-postbl-plm-ss-cfg.patch | 19 ++- driver/patches/0009-mclk-overclock.patch | 147 ++++++++++++++++++ .../0010-mclk-overclock-post-gsp.patch | 59 +++++++ install.sh | 35 +++++ verify.sh | 15 +- 9 files changed, 337 insertions(+), 10 deletions(-) create mode 100644 driver/patches/0009-mclk-overclock.patch create mode 100644 driver/patches/0010-mclk-overclock-post-gsp.patch diff --git a/README.md b/README.md index 49a70b8..137de9a 100644 --- a/README.md +++ b/README.md @@ -50,6 +50,28 @@ sudo ./install.sh --profile=10gb # 10GB card → 40GB unlock Then perform a cold reboot (full power off, then boot). +### HBM Memory Clock + +`--mclk-ndiv=N` sets the FBPA PLL multiplier; the resulting clock is `N × 27` MHz. Any VBIOS works, on both `0x20C2` (8GB) and `0x2082` (10GB) — the driver reads the stock NDIV out of the PLL and rewrites only that field, leaving MDIV/PDIV as the VBIOS programmed them. + +```bash +sudo ./install.sh --mclk-ndiv=70 # → 1890 MHz +``` + +| NDIV | Frequency | Notes | +|------|-----------|---------------------------------| +| 45 | 1215 MHz | Stock 10gb | +| 54 | 1458 MHz | Stock 8gb 250w vbios | +| 64 | 1728 MHz | Stock 8gb 300w vbios | +| 70 | 1890 MHz | Works on ~60% of 8gb cards | +| 73 | 1971 MHz | Usually only on lucky 8gb cards | + +Values below stock downclock the card, which is the way to cut memory power or stabilise a card that fails at stock. + +Without the flag, patches `0009` and `0010` are not applied at all and nothing touches the PLL. The multiplier is compiled into the modules, so changing it means re-running `install.sh`. In a mixed 8GB+10GB box the same multiplier lands on every card, so check each one with `sudo ./verify.sh` before trusting it. + +If a value turns out to be unstable, boot is what breaks — reinstall without `--mclk-ndiv` (or run `./remove.sh`) from a working state. + ## What Gets Unlocked | Feature | Status | @@ -57,6 +79,7 @@ Then perform a cold reboot (full power off, then boot). | Full SM compute throughput (SS0/SS1) | Working ✓ | | Memory geometry (64GB on 8GB cards, 40GB on 10GB cards) | Working ✓ | | PCIe Gen 2 speeds | Working ✓ | +| HBM2 memory overclock/downclock | Working ✓ | | Persistence across reboot (patched modules) | Working ✓ | --- diff --git a/common/constants.yaml b/common/constants.yaml index d444458..8daa8ad 100644 --- a/common/constants.yaml +++ b/common/constants.yaml @@ -26,6 +26,19 @@ pcie: opt_gen23_addr: "0x0082057c" opt_magic_a100: "0x00200000" +mclk_overclock: + supported_device_ids: + - "20c2" + - "2082" + ndiv_min: 30 + ndiv_max: 80 + mhz_per_ndiv: 27 + stock_ndiv: + "10gb": 45 # 1215 MHz + "8gb_250w": 54 # 1458 MHz + "8gb_300w": 64 # 1728 MHz, VBIOS 92.00.6D.00.0A + comment: "HBM2 memory clock via the FBPA PLL NDIV. The patches ship an XX placeholder that build.sh substitutes from --mclk-ndiv." + profiles: "8gb": stock_mib: 8192 diff --git a/docs/INSTALLATION.md b/docs/INSTALLATION.md index 19bb734..5bf91a8 100644 --- a/docs/INSTALLATION.md +++ b/docs/INSTALLATION.md @@ -30,6 +30,17 @@ sudo ./install.sh --profile=8gb # 8GB card → 64GB unlock sudo ./install.sh --profile=10gb # 10GB card → 40GB unlock ``` +To change the HBM memory clock, use `--mclk-ndiv` (multiplier × 27 MHz, any VBIOS, +both `0x20C2` and `0x2082`): + +```bash +sudo ./install.sh --mclk-ndiv=70 # → 1890 MHz +``` + +Stock is NDIV 64 on 8GB 300W VBIOS, 54 on 8GB 250W, 45 on 10GB. Without the flag +the overclock patches are not applied at all. See the README for the full table +and the recovery path if a value turns out unstable. + Then perform a cold reboot (full power off, then boot). ## Uninstall diff --git a/driver/build.sh b/driver/build.sh index 4227878..dbcdfa6 100755 --- a/driver/build.sh +++ b/driver/build.sh @@ -69,12 +69,34 @@ cd "${SRC_DIR}" shopt -s nullglob patches=("${PATCH_DIR}"/*.patch) [[ ${#patches[@]} -gt 0 ]] || die "No patches found in ${PATCH_DIR}" +MCLK_NDIV="${CMPUNLOCKER_MCLK_NDIV:-}" +if [[ -n "${MCLK_NDIV}" ]]; then + if ! [[ "${MCLK_NDIV}" =~ ^[0-9]+$ ]] || [[ "${MCLK_NDIV}" -lt 30 || "${MCLK_NDIV}" -gt 80 ]]; then + die "CMPUNLOCKER_MCLK_NDIV must be an integer between 30 and 80 (got: '${MCLK_NDIV}')" + fi +fi for p in "${patches[@]}"; do - info " $(basename "${p}")" + base="$(basename "${p}")" + if [[ ("${base}" == "0009-mclk-overclock.patch" || "${base}" == "0010-mclk-overclock-post-gsp.patch") && -z "${MCLK_NDIV}" ]]; then + warn "Skipping ${base} (no --mclk-ndiv)" + continue + fi + info " ${base}" patch -p1 < "${p}" done ok "All patches applied" +if [[ -n "${MCLK_NDIV}" ]]; then + TU102_C="${SRC_DIR}/src/nvidia/src/kernel/gpu/gsp/arch/turing/kernel_gsp_tu102.c" + MCLK_GSP_C="${SRC_DIR}/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c" + sed -i "s/static const NvU32 newNdiv = [^;]\\+;/static const NvU32 newNdiv = ${MCLK_NDIV};/" "${TU102_C}" "${MCLK_GSP_C}" + for f in "${TU102_C}" "${MCLK_GSP_C}"; do + grep -q "static const NvU32 newNdiv = ${MCLK_NDIV};" "${f}" \ + || die "NDIV placeholder not substituted in $(basename "${f}") — patch and build.sh are out of sync" + done + ok "MCLK NDIV set to ${MCLK_NDIV} ($((MCLK_NDIV * 27)) MHz)" +fi + PROFILE="${CMPUNLOCKER_CARD_PROFILE:-8gb}" GSP_C="${SRC_DIR}/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c" [[ -f "${GSP_C}" ]] || die "Missing ${GSP_C} after patching" @@ -159,6 +181,7 @@ mkdir -p "${INSTALL_MOD_DIR}" printf '%s\n' "${VERSION}" > "${INSTALL_MOD_DIR}/driver_version" printf '%s\n' "${PROFILE}" > "${INSTALL_MOD_DIR}/card_profile" printf '%s\n' "${UNLOCK_LABEL}" > "${INSTALL_MOD_DIR}/unlock_geometry" +printf '%s\n' "${MCLK_NDIV:-none}" > "${INSTALL_MOD_DIR}/mclk_ndiv" if [[ -n "${CMPUNLOCKER_GPU_INVENTORY:-}" ]]; then printf '%s\n' "${CMPUNLOCKER_GPU_INVENTORY}" > "${INSTALL_MOD_DIR}/gpu_inventory" ok "Wrote gpu_inventory ($(echo "${CMPUNLOCKER_GPU_INVENTORY}" | grep -c . || true) GPU(s))" diff --git a/driver/patches/0001-sec2-postbl-plm-ss-cfg.patch b/driver/patches/0001-sec2-postbl-plm-ss-cfg.patch index 90fed08..bba89b7 100644 --- a/driver/patches/0001-sec2-postbl-plm-ss-cfg.patch +++ b/driver/patches/0001-sec2-postbl-plm-ss-cfg.patch @@ -64,7 +64,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ { // If the new FB layout requires a scrubber ucode to scrub additional space, prepare it now NV_CHECK_OK_OR_RETURN(LEVEL_ERROR, _kgspPrepareScrubberImageIfNeeded(pGpu, pKernelGsp)); -@@ -4821,6 +4849,122 @@ +@@ -4821,6 +4849,125 @@ // Setup arguments for bootstrapping GSP NV_CHECK_OK_OR_RETURN(LEVEL_ERROR, kgspPrepareForBootstrap_HAL(pGpu, pKernelGsp, KGSP_BOOT_MODE_NORMAL)); @@ -84,6 +84,9 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ + { 0x00088ff8U, 0xffffffffU, "XVE_C" }, + { 0x00823b00U, 0xffffffffU, "FEAT2" }, + { 0x008200fcU, 0xffffffffU, "OPT_PLM" }, ++ { 0x009a3c7cU, 0xffffffffU, "FBPA_PLL0" }, ++ { 0x009a3c80U, 0xffffffffU, "FBPA_PLL1" }, ++ { 0x009a3c84U, 0xffffffffU, "FBPA_PLL2" }, + }; + + NvU32 wpr2Lo = GPU_REG_RD32(pGpu, 0x001fa824U); @@ -92,7 +95,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ + "SEC2_DEBUG: saved WPR2 lo=0x%08x hi=0x%08x\n", + wpr2Lo, wpr2Hi); + -+ for (plmIdx = 0; plmIdx < 9; plmIdx++) ++ for (plmIdx = 0; plmIdx < NV_ARRAY_ELEMENTS(plmTable); plmIdx++) + { + NvBool opened = NV_FALSE; + for (attempt = 0; attempt < 2 && !opened; attempt++) @@ -187,7 +190,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ // Release the API lock if relaxed locking for parallel init is enabled NvBool bRelaxedLocking = _kgspShouldRelaxGspInitLocking(pGpu); if (bRelaxedLocking) -@@ -5164,6 +5289,53 @@ +@@ -5164,6 +5292,53 @@ goto done; } @@ -241,7 +244,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ NV_ASSERT_OK_OR_GOTO(status, kgspInitGspTraceCrashBuffer(pGpu, pKernelGsp), done); // Set PDB properties as per data from GSP. -@@ -5662,6 +5831,84 @@ +@@ -5662,6 +5834,84 @@ pKernelGsp->gspRmBootUcodeSize = 0; } @@ -326,7 +329,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ static NV_STATUS _kgspCreateSignatureMemdesc ( -@@ -5672,6 +5919,9 @@ +@@ -5672,6 +5922,9 @@ { NV_STATUS status = NV_OK; NvU8 *pSignatureVa = NULL; @@ -336,7 +339,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ NvU64 flags = MEMDESC_FLAGS_NONE; if (confComputeForceUnprotAlloc(pGpu)) -@@ -5682,7 +5932,7 @@ +@@ -5682,7 +5935,7 @@ // NOTE: align to 256 because that's the alignment needed for Booter DMA NV_CHECK_OK_OR_RETURN(LEVEL_ERROR, memdescCreate(&pKernelGsp->pSignatureMemdesc, pGpu, @@ -345,7 +348,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ NV_TRUE, ADDR_SYSMEM, NV_MEMORY_CACHED, flags)); memdescTagAlloc(status, -@@ -5694,8 +5944,48 @@ +@@ -5694,8 +5947,48 @@ (pSignatureVa != NULL) ? NV_OK : NV_ERR_INSUFFICIENT_RESOURCES, fail_alloc); @@ -396,7 +399,7 @@ diff -Naur a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c b/src/nvidia/src/kernel/ memdescUnmapInternal(pGpu, pKernelGsp->pSignatureMemdesc, 0); pSignatureVa = NULL; -@@ -5709,6 +5999,69 @@ +@@ -5709,6 +6002,69 @@ memdescDestroy(pKernelGsp->pSignatureMemdesc); pKernelGsp->pSignatureMemdesc = NULL; diff --git a/driver/patches/0009-mclk-overclock.patch b/driver/patches/0009-mclk-overclock.patch new file mode 100644 index 0000000..07ac0ed --- /dev/null +++ b/driver/patches/0009-mclk-overclock.patch @@ -0,0 +1,147 @@ +--- a/src/nvidia/src/kernel/gpu/gsp/arch/turing/kernel_gsp_tu102.c ++++ b/src/nvidia/src/kernel/gpu/gsp/arch/turing/kernel_gsp_tu102.c +@@ -649,6 +649,144 @@ + } + } + ++ // HBM PLL NDIV overclock (post-BooterLoad, pre-GSP window where PLMs are open) ++ if (status == NV_OK) ++ { ++ // ++ // Device IDs mirror SEC2_POSTBL_TIMING_CMP_170HX_{8,10}GB_PCI_DEVICE_ID ++ // in kernel_gsp.c. The sequence is VBIOS-agnostic: the stock NDIV is ++ // read out of the PLL rather than assumed, and every COEFF write is a ++ // read-modify-write that keeps the MDIV/PDIV the VBIOS programmed. ++ // ++ NvU32 ocDevId = pGpu->idInfo.PCIDeviceID >> 16; ++ if (ocDevId == 0x20C2 || ocDevId == 0x2082) ++ { ++ NvU32 fbpaIdx; ++ NvU32 pollIdx; ++ NvU32 fbpaCount = 0; ++ NvU32 failCount = 0; ++ NvU32 plm0; ++ NvU32 fbio0; ++ NvU32 curNdiv; ++ NvU32 base, cfgAddr, coeffAddr; ++ NvU32 origCfg, origCoeff, newCoeff, lockCfg; ++ NvU32 ddll0, calSt0, calSt1; ++ // XX is rewritten by driver/build.sh from --mclk-ndiv, and is left ++ // deliberately uncompilable so an unsubstituted tree cannot build. ++ static const NvU32 newNdiv = XX; ++ ++ plm0 = GPU_REG_RD32(pGpu, 0x00903C7CU); ++ curNdiv = (GPU_REG_RD32(pGpu, 0x00903C98U) >> 8) & 0xFFU; ++ if ((plm0 & 0x10U) == 0) ++ { ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: aborted, PLM=0x%08x (bit4 closed)\n", plm0); ++ } ++ else ++ { ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: start NDIV %u->%u (%u->%u MHz) " ++ "devid=0x%04x vbios=%s PLM=0x%08x\n", ++ curNdiv, newNdiv, 27 * curNdiv, 27 * newNdiv, ++ ocDevId, pKernelGsp->vbiosVersionStr, plm0); ++ ++ // 1. Assert MEMCLK_CHANGE_ALERT (FBIO broadcast bit 31) ++ fbio0 = GPU_REG_RD32(pGpu, 0x009A0590U); ++ GPU_REG_WR32(pGpu, 0x009A0590U, fbio0 | 0x80000000U); ++ ++ // 2. Enter HBM self-refresh ++ GPU_REG_WR32(pGpu, 0x009A031CU, 0x00000001U); ++ osDelay(5); ++ ++ // 3. PLL cycle on each active FBPA ++ for (fbpaIdx = 0; fbpaIdx < 12; fbpaIdx++) ++ { ++ base = 0x900000U + fbpaIdx * 0x4000U; ++ cfgAddr = base + 0x3C90U; ++ coeffAddr = base + 0x3C98U; ++ ++ origCfg = GPU_REG_RD32(pGpu, cfgAddr); ++ origCoeff = GPU_REG_RD32(pGpu, coeffAddr); ++ ++ if (origCfg == 0 || origCoeff == 0 || ++ (origCfg & 0xFFF00000U) == 0xBAD00000U) ++ continue; ++ ++ fbpaCount++; ++ ++ GPU_REG_WR32(pGpu, cfgAddr, origCfg & ~0x09U); ++ osDelay(2); ++ ++ newCoeff = (origCoeff & ~0xFF00U) | ((newNdiv & 0xFFU) << 8); ++ GPU_REG_WR32(pGpu, coeffAddr, newCoeff); ++ ++ GPU_REG_WR32(pGpu, cfgAddr, (origCfg | 0x09U) & ~0x20U); ++ ++ lockCfg = 0; ++ for (pollIdx = 0; pollIdx < 200; pollIdx++) ++ { ++ osDelay(1); ++ lockCfg = GPU_REG_RD32(pGpu, cfgAddr); ++ if (lockCfg & 0x20U) ++ break; ++ } ++ ++ if (!(lockCfg & 0x20U)) ++ { ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: FBPA%u PLL lock timeout\n", fbpaIdx); ++ failCount++; ++ continue; ++ } ++ ++ GPU_REG_WR32(pGpu, cfgAddr, lockCfg & ~0x1000U); ++ } ++ ++ // 4. Exit self-refresh ++ GPU_REG_WR32(pGpu, 0x009A031CU, 0x00000000U); ++ osDelay(10); ++ ++ // 5. DDLL calibration ++ ddll0 = GPU_REG_RD32(pGpu, 0x009A11DCU); ++ GPU_REG_WR32(pGpu, 0x009A11DCU, ddll0 | 0x40U); ++ osDelay(10); ++ calSt0 = 0; ++ calSt1 = 0; ++ for (pollIdx = 0; pollIdx < 100; pollIdx++) ++ { ++ osDelay(1); ++ calSt0 = GPU_REG_RD32(pGpu, 0x009A0674U); ++ calSt1 = GPU_REG_RD32(pGpu, 0x009A0678U); ++ } ++ GPU_REG_WR32(pGpu, 0x009A11DCU, ddll0 & ~0x40U); ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: DDLL cal status 0x%08x 0x%08x\n", ++ calSt0, calSt1); ++ ++ // 6. Clear MEMCLK_CHANGE_ALERT ++ GPU_REG_WR32(pGpu, 0x009A0590U, ++ GPU_REG_RD32(pGpu, 0x009A0590U) & ~0x80000000U); ++ ++ if (fbpaCount == 0) ++ NV_PRINTF(LEVEL_ERROR, "HBMPLL_OC: no active FBPAs found\n"); ++ else if (failCount > 0) ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: %u/%u FBPAs failed to lock\n", ++ failCount, fbpaCount); ++ else ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: done, %u FBPAs at NDIV=%u (%u MHz)\n", ++ fbpaCount, newNdiv, 27 * newNdiv); ++ } ++ } ++ else ++ { ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: skipped, PCI device 0x%04x is not a CMP 170HX " ++ "(need 0x20C2 or 0x2082)\n", ocDevId); ++ } ++ } ++ + if (status != NV_OK) + { + NV_PRINTF(LEVEL_ERROR, "failed to execute Booter Load (ucode for initial boot): 0x%x\n", status); diff --git a/driver/patches/0010-mclk-overclock-post-gsp.patch b/driver/patches/0010-mclk-overclock-post-gsp.patch new file mode 100644 index 0000000..94e232a --- /dev/null +++ b/driver/patches/0010-mclk-overclock-post-gsp.patch @@ -0,0 +1,59 @@ +--- a/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c ++++ b/src/nvidia/src/kernel/gpu/gsp/kernel_gsp.c +@@ -5620,6 +5620,56 @@ + + NV_CHECK_OK_OR_GOTO(status, LEVEL_ERROR, kgspStartLogPolling(pGpu, pKernelGsp), done); + ++ // Post-GSP HBMPLL OC: multicast COEFF write + PRI_FENCE (matching v3) ++ { ++ NvU32 ocDevId = pGpu->idInfo.PCIDeviceID >> 16; ++ if (_kgspSec2PostblTimingEnabled(pGpu)) ++ { ++ // XX is rewritten by driver/build.sh from --mclk-ndiv, and is left ++ // deliberately uncompilable so an unsubstituted tree cannot build. ++ static const NvU32 newNdiv = XX; ++ NvU32 coeff_pre = GPU_REG_RD32(pGpu, 0x00903C98U); ++ NvU32 plm_pre = GPU_REG_RD32(pGpu, 0x00903C7CU); ++ NvU32 newCoeff; ++ NvU32 cfg = 0; ++ NvU32 i; ++ ++ // ++ // Keep the MDIV/PDIV the VBIOS programmed and swap NDIV only, so ++ // this works on any VBIOS. Fall back to the 300W layout if the ++ // read came back as a PRI error. ++ // ++ if (coeff_pre == 0 || (coeff_pre & 0xFFF00000U) == 0xBAD00000U) ++ newCoeff = (1U << 16) | (newNdiv << 8) | 1U; ++ else ++ newCoeff = (coeff_pre & ~0xFF00U) | ((newNdiv & 0xFFU) << 8); ++ ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: post-GSP PRE: COEFF=0x%08x(NDIV=%u) PLM=0x%08x " ++ "devid=0x%04x vbios=%s newCOEFF=0x%08x\n", ++ coeff_pre, (coeff_pre >> 8) & 0xFFU, plm_pre, ++ ocDevId, pKernelGsp->vbiosVersionStr, newCoeff); ++ ++ GPU_REG_WR32(pGpu, 0x0098BC98U, newCoeff); ++ GPU_REG_WR32(pGpu, 0x001211FCU, 0x0U); ++ for (i = 0; i < 500; i++) ++ (void)GPU_REG_RD32(pGpu, 0x00903C90U); ++ ++ for (i = 0; i < 200000; i++) ++ { ++ cfg = GPU_REG_RD32(pGpu, 0x00903C90U); ++ if (cfg & 0x20U) ++ break; ++ } ++ ++ NV_PRINTF(LEVEL_ERROR, ++ "HBMPLL_OC: post-GSP POST: COEFF=0x%08x(NDIV=%u) CFG=0x%08x lock=%u iter=%u\n", ++ GPU_REG_RD32(pGpu, 0x00903C98U), ++ (GPU_REG_RD32(pGpu, 0x00903C98U) >> 8) & 0xFFU, ++ cfg, (cfg >> 5) & 1U, i); ++ } ++ } ++ + done: + pKernelGsp->bInInit = NV_FALSE; + diff --git a/install.sh b/install.sh index 6f833ce..f5656a3 100755 --- a/install.sh +++ b/install.sh @@ -11,21 +11,28 @@ LOG_FILE="${LOG_DIR}/install_$(date +%Y%m%d_%H%M%S).log" PROFILE_OVERRIDE="" CONFIGURE_IOMMU=1 CONFIGURE_GEN2_SERVICE=1 +MCLK_NDIV="" for arg in "$@"; do case "${arg}" in --profile=8gb|--profile=8GB) PROFILE_OVERRIDE="8gb" ;; --profile=10gb|--profile=10GB) PROFILE_OVERRIDE="10gb" ;; --no-iommu) CONFIGURE_IOMMU=0 ;; --no-gen2-service) CONFIGURE_GEN2_SERVICE=0 ;; + --mclk-ndiv=*) MCLK_NDIV="${arg#*=}" ;; -h|--help) cat <<'EOF' Usage: sudo ./install.sh [--profile=8gb|10gb] [--no-iommu] [--no-gen2-service] + [--mclk-ndiv=N] --profile=8gb Force 8GB metadata label (geometry is still chosen per PCI ID) --profile=10gb Force 10GB metadata label (geometry is still chosen per PCI ID) --no-iommu Do not touch the kernel command line (leave IOMMU settings alone) --no-gen2-service Do not install the early-boot PCIe Gen2 retrain service + --mclk-ndiv=N HBM memory clock: set PLL multiplier (30-80), N * 27 MHz. + Works on any VBIOS, on both 0x20C2 and 0x2082. Stock is 64 on + 8GB 300W, 54 on 8GB 250W, 45 on 10GB. Without this flag the + overclock patches are not applied at all. By default the installer appends intel_iommu=on / amd_iommu=on plus iommu=pt to the kernel command line so the IOMMU runs in passthrough mode. This takes effect @@ -227,6 +234,25 @@ done export CMPUNLOCKER_CARD_PROFILE="${CARD_PROFILE}" export CMPUNLOCKER_GPU_INVENTORY="$(printf '%s\n' "${GPU_INVENTORY_LINES[@]}")" +if [[ -n "${MCLK_NDIV}" ]]; then + if ! [[ "${MCLK_NDIV}" =~ ^[0-9]+$ ]] || [[ "${MCLK_NDIV}" -lt 30 || "${MCLK_NDIV}" -gt 80 ]]; then + die "--mclk-ndiv must be between 30 and 80 (got: ${MCLK_NDIV})" + fi + ok "MCLK set: NDIV=${MCLK_NDIV} ($((MCLK_NDIV * 27)) MHz) on every unlockable card" + if (( COUNT_8GB > 0 && COUNT_10GB > 0 )); then + warn "Mixed inventory: the multiplier is compiled in once and applies to both" + warn "variants, but stock differs (8gb 54/64 vs 10gb 45). NDIV ${MCLK_NDIV} is" + warn "$((MCLK_NDIV * 27)) MHz on all of them — verify each card in dmesg." + elif (( COUNT_10GB > 0 )); then + info "Stock for 10gb is NDIV 45 (1215 MHz)" + else + info "Stock for 8gb is NDIV 54 (1458 MHz, 250W VBIOS) or 64 (1728 MHz, 300W VBIOS)" + fi +else + info "MCLK overclock disabled (use --mclk-ndiv=N to enable)" +fi +export CMPUNLOCKER_MCLK_NDIV="${MCLK_NDIV}" + step "Step 4/6: Verifying nvidia-open (${SUPPORTED_VERSIONS_CSV})" [[ ${#SUPPORTED_VERSIONS[@]} -gt 0 ]] || die "No supported versions listed in driver/VERSION" if [[ -d /sys/firmware/efi ]] && command -v mokutil &>/dev/null; then @@ -276,6 +302,7 @@ chmod +x "${SCRIPT_DIR}/driver/build.sh" CMPUNLOCKER_DRIVER_VERSION="${detected}" \ CMPUNLOCKER_CARD_PROFILE="${CARD_PROFILE}" \ CMPUNLOCKER_GPU_INVENTORY="${CMPUNLOCKER_GPU_INVENTORY}" \ +CMPUNLOCKER_MCLK_NDIV="${MCLK_NDIV}" \ "${SCRIPT_DIR}/driver/build.sh" ok "Patched modules installed (profile ${CARD_PROFILE})" @@ -450,6 +477,11 @@ if [[ -n "${IOMMU_PARAMS}" && "${IOMMU_STATUS}" != "skipped" ]]; then else echo "IOMMU: not configured" fi +if [[ -n "${MCLK_NDIV}" ]]; then + echo "MCLK: NDIV=${MCLK_NDIV} → $((MCLK_NDIV * 27)) MHz" +else + echo "MCLK: stock (overclock patches not compiled in)" +fi echo "" echo "Per-GPU expectations after unlock:" printf " %-16s %-8s %-8s %s\n" "BDF" "PCI ID" "Variant" "Expect MiB" @@ -463,6 +495,9 @@ echo -e " 2. Verify all GPUs: ${CYAN}sudo ./verify.sh${NC}" echo -e " 3. Verify PCIe Gen2: ${CYAN}nvidia-smi --query-gpu=pcie.link.gen.current,pcie.link.gen.max --format=csv${NC} (expect 2,2)" echo -e " 4. Or check manually: ${CYAN}nvidia-smi${NC}" echo -e " 5. Unlock logs: ${CYAN}sudo dmesg | grep SEC2_DEBUG${NC}" +if [[ -n "${MCLK_NDIV}" ]]; then + echo -e " Memory clock logs: ${CYAN}sudo dmesg | grep HBMPLL_OC${NC} (expect NDIV=${MCLK_NDIV})" +fi echo -e " 6. Verify IOMMU after reboot: ${CYAN}cat /proc/cmdline${NC} and ${CYAN}ls /sys/class/iommu${NC}" if (( CONFIGURE_GEN2_SERVICE == 1 )); then echo -e " 7. Verify negotiated Gen2: ${CYAN}sudo ./tools/service.sh verify${NC}" diff --git a/verify.sh b/verify.sh index 9770dcc..bcdac67 100755 --- a/verify.sh +++ b/verify.sh @@ -176,9 +176,22 @@ else warn "No SEC2_DEBUG lines in dmesg (logs may have rotated; unlock can still be OK if memory is unlocked)" fi +installed_ndiv="$(cat "${INSTALL_MOD_DIR}/mclk_ndiv" 2>/dev/null || echo 'none')" +if [[ "${installed_ndiv}" != "none" && -n "${installed_ndiv}" ]]; then + echo "" + mclk_logs="$(dmesg 2>/dev/null | grep 'HBMPLL_OC' || true)" + if [[ -n "${mclk_logs}" ]]; then + ok "dmesg contains HBMPLL_OC logs (installed NDIV=${installed_ndiv}, $((installed_ndiv * 27)) MHz)" + printf '%s\n' "${mclk_logs}" | tail -n 6 | sed 's/^/ /' + else + warn "Installed with --mclk-ndiv=${installed_ndiv} but no HBMPLL_OC lines in dmesg" + warn "(logs may have rotated; otherwise the memory is still running at stock)" + fi +fi + echo "" if [[ -r "${INSTALL_MOD_DIR}/card_profile" ]]; then - info "Installed profile: $(cat "${INSTALL_MOD_DIR}/card_profile") / geometry: $(cat "${INSTALL_MOD_DIR}/unlock_geometry" 2>/dev/null || echo '?')" + info "Installed profile: $(cat "${INSTALL_MOD_DIR}/card_profile") / geometry: $(cat "${INSTALL_MOD_DIR}/unlock_geometry" 2>/dev/null || echo '?') / mclk_ndiv: ${installed_ndiv}" fi if (( failures > 0 )); then -- 2.51.2