diff --git a/tools/qxmx_diff.cpp b/tools/qxmx_diff.cpp index a668fe9..871b6b4 100644 --- a/tools/qxmx_diff.cpp +++ b/tools/qxmx_diff.cpp @@ -1,4 +1,4 @@ -/* qxmx_diff: per-layer diff between the B70 engine and the CPU ref oracle. +/* qxmx_diff: per-layer diff between the Arc Battlemage GPU engine and the CPU ref oracle. * * Runs both engines on the same prompt, dumps the residual stream (x[5120]) * after each layer's attention+FFN, and reports where they diverge. @@ -48,7 +48,7 @@ int main(int argc, char** argv) { qx::ref_session ref_sess; ref_sess.init(&m, std::max((int)ids.size() + 16, 1 << 14)); - /* --- B70 engine --- */ + /* --- GPU engine --- */ qx::g_model = &m; qx::engine eng; eng.k_ctk = k_ctk; @@ -84,7 +84,7 @@ int main(int argc, char** argv) { eng.forward(slot, ids[0]); int eng_best = eng.sample(slot, qx::sample_params{}); std::string eng_tok = tok.decode(&eng_best, 1, true); - std::printf("B70 eng: argmax=%d \"%s\"\n", eng_best, eng_tok.c_str()); + std::printf("GPU eng: argmax=%d \"%s\"\n", eng_best, eng_tok.c_str()); /* Also run the full 5-token prefill through both and compare the * next-token prediction (the one that should produce " Paris"). */ @@ -109,7 +109,7 @@ int main(int argc, char** argv) { eng_next = eng.sample(slot, qx::sample_params{}); } std::string eng_tok2 = tok.decode(&eng_next, 1, true); - std::printf("B70 eng after %zu tokens: argmax=%d \"%s\" (logit=%.3f)\n", + std::printf("GPU eng after %zu tokens: argmax=%d \"%s\" (logit=%.3f)\n", ids.size(), eng_next, eng_tok2.c_str(), eng.logits[eng_next]); /* logits diff after full prefill */ @@ -145,4 +145,4 @@ int main(int argc, char** argv) { qx_model_close(&m); return 0; -} \ No newline at end of file +}