From 77caf9c4d15be302937d621abca72f0b3797bcb0 Mon Sep 17 00:00:00 2001 From: Raphael Amorim Date: Wed, 20 May 2026 00:32:35 +0200 Subject: [PATCH] format code --- src/ast.h | 3 +- src/astgen.cpp | 743 ++++++++++++++------------------ src/astgen.h | 11 +- src/codegen.cpp | 4 +- src/codegen.h | 8 +- src/diagnostics.cpp | 31 +- src/diagnostics.h | 31 +- src/jam_llvm.cpp | 28 +- src/jam_llvm.h | 47 +- src/jir.h | 94 ++-- src/jir_codegen.cpp | 321 +++++++------- src/jir_verify.cpp | 321 ++++++++------ src/jir_verify.h | 9 +- src/lexer.cpp | 4 +- src/lexer.h | 4 +- src/main.cpp | 72 ++-- src/mangling.h | 7 +- src/parser.cpp | 86 ++-- src/parser.h | 5 +- src/token.h | 5 +- tests/cpp/test_diagnostics.cpp | 246 +++++------ tests/cpp/test_jir_skeleton.cpp | 3 +- 22 files changed, 974 insertions(+), 1109 deletions(-) diff --git a/src/ast.h b/src/ast.h index 4e7b3fe..a7bbfd2 100644 --- a/src/ast.h +++ b/src/ast.h @@ -164,8 +164,7 @@ class UnionDeclAST { // evaluable at codegen time). At each use site, the init expression is // codegen'd inline with `DeclaredType` as the expected type — a small // inlining pass that costs nothing at runtime and lets the optimizer -// fold across uses just like a literal would. This is the same model -// Zig uses for `pub const` of comptime-known values. +// fold across uses just like a literal would. class ConstDeclAST { public: std::string Name; diff --git a/src/astgen.cpp b/src/astgen.cpp index a34b6db..aa2b8e3 100644 --- a/src/astgen.cpp +++ b/src/astgen.cpp @@ -45,8 +45,8 @@ struct LoopFrame { // order so drop calls are emitted in REVERSE order at scope exit. struct DropTrack { std::string varName; - JirRef slot; // alloca for the variable - TypeIdx type; // source-level type + JirRef slot; // alloca for the variable + TypeIdx type; // source-level type std::string llvmFnName; // canonical drop fn (legacy mangles to `__drop_T`) }; @@ -74,9 +74,7 @@ struct AstGenCtx { std::vector> localScopes; // Most-recently-entered AST node — `astgenExpr` updates this on // every entry so error helpers without an explicit NodeIdx in - // scope (`failHere` and friends) can still emit a SrcLoc. Mirrors - // the implicit "current node" Zig keeps via the active - // `Ast.Node.Index` it threads through every helper. + // scope (`failHere` and friends) can still emit a SrcLoc. NodeIdx currentNode = 0; }; @@ -98,8 +96,8 @@ static jam::SrcLoc locOf(AstGenCtx &gctx, NodeIdx node) { // in. Every astgen error helper funnels through here so the chain of // generic instantiations (if any) is preserved. static jam::Diagnostic makeDiag(AstGenCtx &gctx, NodeIdx node, - std::string message, - std::vector notes) { + std::string message, + std::vector notes) { jam::Diagnostic d; d.loc = locOf(gctx, node); d.severity = jam::Diagnostic::Severity::Error; @@ -111,20 +109,18 @@ static jam::Diagnostic makeDiag(AstGenCtx &gctx, NodeIdx node, // Append a Diagnostic anchored at `node` and keep walking. The // caller is responsible for synthesising a poison value if the -// expression position needs one. Mirrors `AstGen.zig:10113 -// appendErrorNode`. +// expression position needs one. static void appendErrorNode(AstGenCtx &gctx, NodeIdx node, std::string message) { - gctx.ctx.diagnostics().push( - makeDiag(gctx, node, std::move(message), {})); + gctx.ctx.diagnostics().push(makeDiag(gctx, node, std::move(message), {})); } // As `appendErrorNode`, but the diagnostic carries secondary notes. // Notes are typically built by the caller from related source // positions ("X was declared here", "did you mean Y?"). static void appendErrorNodeNotes(AstGenCtx &gctx, NodeIdx node, - std::string message, - std::vector notes) { + std::string message, + std::vector notes) { gctx.ctx.diagnostics().push( makeDiag(gctx, node, std::move(message), std::move(notes))); } @@ -132,16 +128,16 @@ static void appendErrorNodeNotes(AstGenCtx &gctx, NodeIdx node, // Append + bail the current decl. The catch site (one per function, // struct method, or generic instantiation) drops back to compiling // siblings, so a single broken function doesn't suppress diagnostics -// from the rest of the file. Mirrors `AstGen.zig:10149 failNodeNotes`. +// from the rest of the file. [[noreturn]] static void failNode(AstGenCtx &gctx, NodeIdx node, - std::string message) { + std::string message) { appendErrorNode(gctx, node, std::move(message)); throw AstGenAnalysisFail{}; } [[noreturn]] static void failNodeNotes(AstGenCtx &gctx, NodeIdx node, - std::string message, - std::vector notes) { + std::string message, + std::vector notes) { appendErrorNodeNotes(gctx, node, std::move(message), std::move(notes)); throw AstGenAnalysisFail{}; } @@ -168,13 +164,6 @@ static JirRef emit(AstGenCtx &gctx, JirInst inst); // `Diagnostics::hasErrors()` is true, so the undef is unreachable // in well-formed builds and serves only as a placeholder for // downstream typecheck. -// -// This is Jam-specific, not borrowed from Zig: Zig's `ZIR.unreachable_value` -// represents the actual `unreachable` keyword and `ZIR.generic_poison` -// is for unresolved generic-type parameters during pre-instantiation; -// neither plays the "skip past an error and keep analyzing" role. -// The closest analog elsewhere is Roslyn's `BoundBadNode` / -// LLVM's `undef`-after-error recovery. static JirRef emitPoison(AstGenCtx &gctx, TypeIdx ty) { JirInst inst{}; inst.tag = JirTag::Poison; @@ -184,15 +173,14 @@ static JirRef emitPoison(AstGenCtx &gctx, TypeIdx ty) { // Combined: push a recoverable diagnostic anchored at `node` and // hand back a typed Poison so the caller can continue typechecking. -static JirRef recoverNode(AstGenCtx &gctx, NodeIdx node, - std::string message, TypeIdx ty) { +static JirRef recoverNode(AstGenCtx &gctx, NodeIdx node, std::string message, + TypeIdx ty) { appendErrorNode(gctx, node, std::move(message)); return emitPoison(gctx, ty); } // As `recoverNode` but anchored at `gctx.currentNode`. -static JirRef recoverHere(AstGenCtx &gctx, std::string message, - TypeIdx ty) { +static JirRef recoverHere(AstGenCtx &gctx, std::string message, TypeIdx ty) { appendErrorHere(gctx, std::move(message)); return emitPoison(gctx, ty); } @@ -219,12 +207,10 @@ static JirRef emit(AstGenCtx &gctx, JirInst inst) { // other ref. static JirRef emitAllocaHoisted(AstGenCtx &gctx, JirInst alloca) { JirRef r = gctx.jfn.pushInst(alloca); - auto &entryInsts = - gctx.jfn.getBlockMut(/*entry=*/1).insts; + auto &entryInsts = gctx.jfn.getBlockMut(/*entry=*/1).insts; std::size_t insertAt = entryInsts.size(); while (insertAt > 0) { - JirTag prev = - gctx.jfn.getInst(entryInsts[insertAt - 1]).tag; + JirTag prev = gctx.jfn.getInst(entryInsts[insertAt - 1]).tag; if (prev != JirTag::Br && prev != JirTag::CondBr && prev != JirTag::Switch && prev != JirTag::Ret && prev != JirTag::Unreachable) { @@ -232,9 +218,8 @@ static JirRef emitAllocaHoisted(AstGenCtx &gctx, JirInst alloca) { } insertAt--; } - entryInsts.insert(entryInsts.begin() + - static_cast(insertAt), - r); + entryInsts.insert( + entryInsts.begin() + static_cast(insertAt), r); return r; } @@ -246,9 +231,8 @@ static bool blockHasTerminator(const JirBlock &block, const JirFunction &jfn) { if (block.insts.empty()) return false; JirRef last = block.insts.back(); JirTag t = jfn.getInst(last).tag; - return t == JirTag::Ret || t == JirTag::Unreachable || - t == JirTag::Br || t == JirTag::CondBr || - t == JirTag::Switch; + return t == JirTag::Ret || t == JirTag::Unreachable || t == JirTag::Br || + t == JirTag::CondBr || t == JirTag::Switch; } // Count blocks that branch into `target` via Br / CondBr / Switch. @@ -273,8 +257,7 @@ static std::size_t predecessorCount(const JirFunction &jfn, case JirTag::CondBr: { JirExtraIdx ex = last.b; if (ex + 2 <= jfn.extra.size()) { - if (static_cast(jfn.extra[ex]) == target) - count++; + if (static_cast(jfn.extra[ex]) == target) count++; if (static_cast(jfn.extra[ex + 1]) == target) count++; } @@ -288,8 +271,7 @@ static std::size_t predecessorCount(const JirFunction &jfn, for (uint32_t i = 0; i < caseCount; i++) { JirExtraIdx caseSlot = ex + 2 + i * 4 + 3; if (caseSlot < jfn.extra.size() && - static_cast(jfn.extra[caseSlot]) == - target) { + static_cast(jfn.extra[caseSlot]) == target) { count++; } } @@ -311,8 +293,7 @@ static JirRef astgenExpr(AstGenCtx &gctx, NodeIdx node, TypeIdx expected); static void emitBr(AstGenCtx &gctx, JirBlockRef target); static void emitCondBr(AstGenCtx &gctx, JirRef cond, JirBlockRef thenB, JirBlockRef elseB); -static void emitDrops(AstGenCtx &gctx, - const std::vector &bindings); +static void emitDrops(AstGenCtx &gctx, const std::vector &bindings); static JirRef emitCall(AstGenCtx &gctx, const FunctionAST *fn, const std::vector &argRefs); @@ -331,7 +312,7 @@ static inline void pushDropScope(AstGenCtx &gctx) { // in the calling AST walker (and will pop on its own as the structured // bodies return). static inline void emitDropsThroughScope(AstGenCtx &gctx, - std::size_t targetIdx) { + std::size_t targetIdx) { if (gctx.dropScopes.size() <= targetIdx) return; for (std::size_t i = gctx.dropScopes.size(); i > targetIdx; i--) { emitDrops(gctx, gctx.dropScopes[i - 1]); @@ -345,8 +326,7 @@ static inline void emitDropsThroughScope(AstGenCtx &gctx, static inline void popDropScopeEmitting(AstGenCtx &gctx) { if (gctx.dropScopes.empty()) return; const auto &scope = gctx.dropScopes.back(); - if (!blockHasTerminator(gctx.jfn.getBlock(gctx.currentBlock), - gctx.jfn)) { + if (!blockHasTerminator(gctx.jfn.getBlock(gctx.currentBlock), gctx.jfn)) { emitDrops(gctx, scope); } gctx.dropScopes.pop_back(); @@ -365,15 +345,14 @@ static inline void popDropScope(AstGenCtx &gctx) { // JIR `Int` or `Float` constant. static JirRef astgenNumberLit(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { - uint64_t val = static_cast(n.lhs) | - (static_cast(n.rhs) << 32); + uint64_t val = + static_cast(n.lhs) | (static_cast(n.rhs) << 32); bool isNeg = (n.flags & 1) != 0; bool isFloat = (n.flags & 2) != 0; JirInst inst{}; - inst.srcLine = gctx.jfn.insts.empty() - ? 0 - : static_cast(0); // patched below + inst.srcLine = + gctx.jfn.insts.empty() ? 0 : static_cast(0); // patched below if (isFloat) { inst.tag = JirTag::Float; @@ -462,8 +441,8 @@ static JirRef astgenBoolLit(AstGenCtx &gctx, const AstNode &n) { static void astgenReturn(AstGenCtx &gctx, const AstNode &n) { JirRef valRef = kNoJirRef; if (n.lhs != 0) { - valRef = astgenExpr(gctx, static_cast(n.lhs), - gctx.jfn.returnType); + valRef = + astgenExpr(gctx, static_cast(n.lhs), gctx.jfn.returnType); } // Drop every active scope before exiting the function. emitDropsThroughScope(gctx, 0); @@ -486,8 +465,7 @@ static std::string lookupDropFnLLVMName(AstGenCtx &gctx, TypeIdx ty) { std::string typeName; if (k.kind == TypeKind::Struct || k.kind == TypeKind::Named || k.kind == TypeKind::Enum) { - typeName = gctx.ctx.getStringPool().get( - static_cast(k.a)); + typeName = gctx.ctx.getStringPool().get(static_cast(k.a)); } else { return ""; } @@ -498,8 +476,7 @@ static std::string lookupDropFnLLVMName(AstGenCtx &gctx, TypeIdx ty) { // mangling change can't silently desync. auto resolveName = [&](const std::string &name) -> std::string { const FunctionAST *fn = nullptr; - const jam::drops::DropRegistry *reg = - gctx.ctx.getDropRegistry(); + const jam::drops::DropRegistry *reg = gctx.ctx.getDropRegistry(); if (reg != nullptr) { auto it = reg->find(name); if (it != reg->end()) fn = it->second; @@ -515,15 +492,14 @@ static std::string lookupDropFnLLVMName(AstGenCtx &gctx, TypeIdx ty) { if (aliasTarget != kNoType) { const TypeKey &ak0 = gctx.ctx.getTypePool().get(aliasTarget); if (ak0.kind == TypeKind::GenericCall) { - TypeIdx resolved = - gctx.ctx.resolveGenericCall(aliasTarget); + TypeIdx resolved = gctx.ctx.resolveGenericCall(aliasTarget); if (resolved != kNoType) aliasTarget = resolved; } const TypeKey &ak = gctx.ctx.getTypePool().get(aliasTarget); if (ak.kind == TypeKind::Named || ak.kind == TypeKind::Struct || ak.kind == TypeKind::Enum) { - std::string aliasName = gctx.ctx.getStringPool().get( - static_cast(ak.a)); + std::string aliasName = + gctx.ctx.getStringPool().get(static_cast(ak.a)); r = resolveName(aliasName); if (!r.empty()) return r; } @@ -534,8 +510,7 @@ static std::string lookupDropFnLLVMName(AstGenCtx &gctx, TypeIdx ty) { // Emit drop calls for every binding in `bindings` in REVERSE order // (LIFO — last-pushed dropped first). The argument is &binding (the // alloca pointer itself, since `fn drop(self: mut T)` takes a *T). -static void emitDrops(AstGenCtx &gctx, - const std::vector &bindings) { +static void emitDrops(AstGenCtx &gctx, const std::vector &bindings) { // Emit explicit DropBinding JIR instructions in reverse declaration // order. Each carries the binding's alloca JirRef and the LLVM // symbol of the drop fn as a StringIdx. Codegen mechanically @@ -543,8 +518,7 @@ static void emitDrops(AstGenCtx &gctx, // AddrOf+pack ceremony, no metadata fallback. for (auto it = bindings.rbegin(); it != bindings.rend(); ++it) { const DropTrack &d = *it; - StringIdx symId = - gctx.ctx.getStringPool().intern(d.llvmFnName); + StringIdx symId = gctx.ctx.getStringPool().intern(d.llvmFnName); JirInst drop{}; drop.tag = JirTag::DropBinding; drop.a = d.slot; @@ -565,19 +539,13 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { NodeIdx initIdx = static_cast(ns.getExtra(extra + 2)); const std::string &name = gctx.ctx.getStringPool().get(nameId); - // Reject re-declaration within the same lexical scope only. Looser - // than Zig (which forbids ALL shadowing, including inner blocks - // shadowing outer bindings and even shadowing of primitive type - // names — see `AstGen.zig:12099 detectLocalShadowing` that walks - // the entire scope chain). Jam's `localScopes` stack only inspects - // the innermost frame, so the user's reported bug (`const a = X; - // var a = Y;` at the same level) is rejected but + // Reject re-declaration within the same lexical scope only. We + // inspect the innermost `localScopes` frame, so `const a = X; + // var a = Y;` at the same level errors, but // fn f() { var x = 1; if (c) { var x = 2; } } - // still compiles — intentional inner-block shadow. - if (!gctx.localScopes.empty() && - gctx.localScopes.back().count(name) != 0) { - failHere(gctx, "redeclaration of `" + name + - "` in the same scope"); + // still compiles — intentional inner-block shadowing is allowed. + if (!gctx.localScopes.empty() && gctx.localScopes.back().count(name) != 0) { + failHere(gctx, "redeclaration of `" + name + "` in the same scope"); } // Type resolution. @@ -590,15 +558,11 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { // type so the slot pre-exists; we reject them with a clear // error when inferred. // - // Shallower than Zig: Zig's `alloc_inferred` + `block_ptr` - // result-location (`AstGen.zig:3014`, `Sema.zig:3483`) unifies - // the types of *every* store into the inferred slot so - // `var x = if (c) @as(i32,1) else @as(i32,2);` works. Jam - // commits to whatever single type comes back from astgenExpr - // — sufficient for the v1 grammar (no `if`-expression in - // value position outside `match`), but cases that would need - // peer-typing across multiple stores must use an explicit - // `: T` annotation. + // Inference commits to whatever single type comes back from + // astgenExpr — sufficient for the v1 grammar (no `if`-expression + // in value position outside `match`). Cases that would need + // peer-typing across multiple stores into the same slot must + // use an explicit `: T` annotation. TypeIdx type; JirRef allocaRef; JirRef initRef; @@ -607,9 +571,8 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { initRef = astgenExpr(gctx, initIdx, kNoType); type = gctx.jfn.getInst(initRef).ty; if (type == kNoType) { - failHere(gctx, - "could not infer type of `" + name + - "`; add an explicit `: T` annotation"); + failHere(gctx, "could not infer type of `" + name + + "`; add an explicit `: T` annotation"); } JirInst alloca{}; alloca.tag = JirTag::Alloca; @@ -617,9 +580,7 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { allocaRef = emitAllocaHoisted(gctx, alloca); gctx.locals[name] = allocaRef; gctx.localTypes[name] = type; - if (!gctx.localScopes.empty()) { - gctx.localScopes.back().insert(name); - } + if (!gctx.localScopes.empty()) { gctx.localScopes.back().insert(name); } } else { type = declared; JirInst alloca{}; @@ -634,9 +595,7 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { // for the resulting semantics. gctx.locals[name] = allocaRef; gctx.localTypes[name] = type; - if (!gctx.localScopes.empty()) { - gctx.localScopes.back().insert(name); - } + if (!gctx.localScopes.empty()) { gctx.localScopes.back().insert(name); } initRef = astgenExpr(gctx, initIdx, type); @@ -662,8 +621,8 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { // method body, `T` resolves to whatever the // instantiation supplied). if (k.kind == TypeKind::Named) { - const std::string &name = gctx.ctx.getStringPool().get( - static_cast(k.a)); + const std::string &name = + gctx.ctx.getStringPool().get(static_cast(k.a)); TypeIdx sub = gctx.ctx.lookupCurrentSubst(name); if (sub != kNoType) return resolveForCmp(sub); } @@ -673,8 +632,7 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { } if (k.kind == TypeKind::Named) { TypeIdx a = gctx.ctx.lookupTypeAlias( - gctx.ctx.getStringPool().get( - static_cast(k.a))); + gctx.ctx.getStringPool().get(static_cast(k.a))); if (a != kNoType) return resolveForCmp(a); } return t; @@ -682,44 +640,38 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { TypeIdx initTy = gctx.jfn.getInst(initRef).ty; TypeIdx declRes = resolveForCmp(type); TypeIdx initRes = resolveForCmp(initTy); - // Pointer-shape leniency (Jam-specific, *not* Zig-equivalent): - // PtrSingle(T) and PtrMany(T) share the runtime representation - // (a plain `ptr`), so `&arr[i]` (which lowers to PtrSingle(T)) - // is accepted in a PtrMany(T) slot as a zero-cost retag. Zig's - // `Sema.coerceInMemoryAllowedPtrs` (`Sema.zig:25447`) requires - // the source pointer size to match the destination or one to be - // `.C`; `*T → [*]T` is only allowed when the source is a - // pointer-to-array. Jam's rule is broader because we don't - // currently distinguish pointer-to-array from pointer-to-element - // at the JIR level — revisit when an Array-pointer kind lands. + // Pointer-shape leniency: PtrSingle(T) and PtrMany(T) share + // the runtime representation (a plain `ptr`), so `&arr[i]` + // (which lowers to PtrSingle(T)) is accepted in a PtrMany(T) + // slot as a zero-cost retag. The rule is permissive because + // we don't currently distinguish pointer-to-array from + // pointer-to-element at the JIR level — revisit when an + // Array-pointer kind lands and the source can carry its + // length statically. auto pointerCompatible = [&](TypeIdx a, TypeIdx b) -> bool { if (a == kNoType || b == kNoType) return false; const TypeKey &ka = gctx.ctx.getTypePool().get(a); const TypeKey &kb = gctx.ctx.getTypePool().get(b); - bool aPtr = ka.kind == TypeKind::PtrSingle || - ka.kind == TypeKind::PtrMany; - bool bPtr = kb.kind == TypeKind::PtrSingle || - kb.kind == TypeKind::PtrMany; + bool aPtr = + ka.kind == TypeKind::PtrSingle || ka.kind == TypeKind::PtrMany; + bool bPtr = + kb.kind == TypeKind::PtrSingle || kb.kind == TypeKind::PtrMany; return aPtr && bPtr && ka.a == kb.a; }; - bool typesMatch = (declRes == initRes) || - pointerCompatible(declRes, initRes); + bool typesMatch = + (declRes == initRes) || pointerCompatible(declRes, initRes); if (initTy != kNoType && !typesMatch) { const TypeKey &dk = gctx.ctx.getTypePool().get(declRes); const TypeKey &ik = gctx.ctx.getTypePool().get(initRes); // Specialize the message for the two patterns users hit // most often; everything else gets the generic mismatch. - if (dk.kind == TypeKind::Float && - ik.kind == TypeKind::Int) { - failHere(gctx, - "cannot assign integer to float-typed `" + - name + - "`; use a float literal (e.g. `3.0`) " - "or an explicit `as` cast"); + if (dk.kind == TypeKind::Float && ik.kind == TypeKind::Int) { + failHere(gctx, "cannot assign integer to float-typed `" + name + + "`; use a float literal (e.g. `3.0`) " + "or an explicit `as` cast"); } - failHere(gctx, - "type mismatch in `" + name + - "`: declared and initialised values disagree"); + failHere(gctx, "type mismatch in `" + name + + "`: declared and initialised values disagree"); } } @@ -735,8 +687,7 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { std::string dropName = lookupDropFnLLVMName(gctx, type); if (!dropName.empty()) { if (gctx.dropScopes.empty()) pushDropScope(gctx); - gctx.dropScopes.back().push_back( - {name, allocaRef, type, dropName}); + gctx.dropScopes.back().push_back({name, allocaRef, type, dropName}); } } @@ -778,12 +729,11 @@ static JirRef astgenLvalue(AstGenCtx &gctx, NodeIdx node, TypeIdx &outLeafTy) { const AstNode &n = ns.get(node); switch (n.tag) { case AstTag::Variable: { - const std::string &name = gctx.ctx.getStringPool().get( - static_cast(n.lhs)); + const std::string &name = + gctx.ctx.getStringPool().get(static_cast(n.lhs)); auto it = gctx.locals.find(name); if (it == gctx.locals.end()) { - failHere(gctx, - "astgen: unknown lvalue variable `" + name + "`"); + failHere(gctx, "astgen: unknown lvalue variable `" + name + "`"); } outLeafTy = gctx.localTypes[name]; return it->second; @@ -801,20 +751,18 @@ static JirRef astgenLvalue(AstGenCtx &gctx, NodeIdx node, TypeIdx &outLeafTy) { } case AstTag::MemberAccess: { TypeIdx baseTy = kNoType; - JirRef basePtr = astgenLvalue( - gctx, static_cast(n.lhs), baseTy); + JirRef basePtr = + astgenLvalue(gctx, static_cast(n.lhs), baseTy); StringIdx memberId = static_cast(n.rhs); - const std::string &memberName = - gctx.ctx.getStringPool().get(memberId); + const std::string &memberName = gctx.ctx.getStringPool().get(memberId); // Union field lvalue: every field shares the union's address, // so the field pointer IS the union pointer — just retype it. if (const auto *uinfo = gctx.ctx.lookupUnion(baseTy)) { TypeIdx fieldTy = gctx.ctx.getUnionFieldType(uinfo->name, memberName); if (fieldTy == kNoType) { - failHere(gctx, - "astgen: union `" + uinfo->name + "` has no field `" + - memberName + "`"); + failHere(gctx, "astgen: union `" + uinfo->name + + "` has no field `" + memberName + "`"); } outLeafTy = fieldTy; TypeIdx ptrTy = gctx.ctx.getTypePool().intern( @@ -827,14 +775,12 @@ static JirRef astgenLvalue(AstGenCtx &gctx, NodeIdx node, TypeIdx &outLeafTy) { } const auto *info = gctx.ctx.lookupStruct(baseTy); if (info == nullptr) { - failHere(gctx, - "astgen: lvalue field access on non-struct"); + failHere(gctx, "astgen: lvalue field access on non-struct"); } int idx = gctx.ctx.getFieldIndex(info->name, memberName); if (idx < 0) { - failHere(gctx, - "astgen: unknown field `" + memberName + "` on `" + - info->name + "`"); + failHere(gctx, "astgen: unknown field `" + memberName + "` on `" + + info->name + "`"); } outLeafTy = info->fields[idx].second; TypeIdx ptrTy = gctx.ctx.getTypePool().intern( @@ -855,8 +801,8 @@ static JirRef astgenLvalue(AstGenCtx &gctx, NodeIdx node, TypeIdx &outLeafTy) { } case AstTag::Index: { TypeIdx baseTy = kNoType; - JirRef basePtr = astgenLvalue( - gctx, static_cast(n.lhs), baseTy); + JirRef basePtr = + astgenLvalue(gctx, static_cast(n.lhs), baseTy); NodeIdx idxIdx = static_cast(n.rhs); JirRef idxRef = astgenExpr(gctx, idxIdx, BuiltinType::U64); const TypeKey &k = gctx.ctx.getTypePool().get(baseTy); @@ -865,8 +811,7 @@ static JirRef astgenLvalue(AstGenCtx &gctx, NodeIdx node, TypeIdx &outLeafTy) { k.kind == TypeKind::PtrMany) { elemTy = static_cast(k.a); } else { - failHere(gctx, - "astgen: lvalue index on non-array/slice/ptr-many"); + failHere(gctx, "astgen: lvalue index on non-array/slice/ptr-many"); } // For pointer-typed base variables, the alloca holds the // *pointer* itself — we need to load it first so the GEP @@ -938,8 +883,7 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, TypeIdx ty = static_cast(n.lhs); if (ty == kNoType) ty = expected; if (ty == kNoType) { - failHere(gctx, - "astgen: struct literal without target type"); + failHere(gctx, "astgen: struct literal without target type"); } // Union literal: exactly one field, stored into a slot of the @@ -950,21 +894,15 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, ExtraIdx fieldsExtra = static_cast(n.rhs); uint32_t fieldCount = ns.getExtra(fieldsExtra); if (fieldCount != 1) { - failHere(gctx, - "astgen: union literal must list exactly one field"); - } - StringIdx nameId = static_cast( - ns.getExtra(fieldsExtra + 1)); - NodeIdx exprIdx = static_cast( - ns.getExtra(fieldsExtra + 2)); - const std::string &fieldName = - gctx.ctx.getStringPool().get(nameId); - TypeIdx fieldTy = - gctx.ctx.getUnionFieldType(uinfo->name, fieldName); + failHere(gctx, "astgen: union literal must list exactly one field"); + } + StringIdx nameId = static_cast(ns.getExtra(fieldsExtra + 1)); + NodeIdx exprIdx = static_cast(ns.getExtra(fieldsExtra + 2)); + const std::string &fieldName = gctx.ctx.getStringPool().get(nameId); + TypeIdx fieldTy = gctx.ctx.getUnionFieldType(uinfo->name, fieldName); if (fieldTy == kNoType) { - failHere(gctx, - "astgen: union `" + uinfo->name + "` has no field `" + - fieldName + "`"); + failHere(gctx, "astgen: union `" + uinfo->name + + "` has no field `" + fieldName + "`"); } JirRef fieldVal = astgenExpr(gctx, exprIdx, fieldTy); // Alloca the union, store the field's value into its slot, @@ -995,13 +933,12 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, // "not exported" / "does not exist" / "unknown handle" // diagnostic; bare names fall back to the generic message. if (name.find('.') != std::string::npos) { - failHere(gctx, - gctx.ctx.formatNamespaceLookupError("struct", name)); + failHere(gctx, + gctx.ctx.formatNamespaceLookupError("struct", name)); } failHere(gctx, "unknown struct `" + name + "`"); } - failHere(gctx, - "astgen: struct literal type is not a known struct"); + failHere(gctx, "astgen: struct literal type is not a known struct"); } ExtraIdx fieldsExtra = static_cast(n.rhs); @@ -1020,8 +957,7 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, // Recoverable: record the bad field name and skip it so // other malformed fields in the same literal still get // reported in this pass. - appendErrorHere(gctx, "unknown struct field `" + - fieldName + "`"); + appendErrorHere(gctx, "unknown struct field `" + fieldName + "`"); continue; } TypeIdx expectedField = info->fields[idx].second; @@ -1035,8 +971,7 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, if (vt != expectedField && vt != kNoType) { const TypeKey &fk = gctx.ctx.getTypePool().get(expectedField); const TypeKey &vk = gctx.ctx.getTypePool().get(vt); - if (fk.kind == TypeKind::Float && - vk.kind == TypeKind::Int) { + if (fk.kind == TypeKind::Float && vk.kind == TypeKind::Int) { JirInst c{}; c.tag = vk.b != 0 ? JirTag::SIToFP : JirTag::UIToFP; c.a = fieldVal; @@ -1050,9 +985,8 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, // literals must initialise every field. for (size_t i = 0; i < ordered.size(); i++) { if (ordered[i] == kNoJirRef) { - failHere(gctx, - "astgen: struct literal missing field `" + - info->fields[i].first + "`"); + failHere(gctx, "astgen: struct literal missing field `" + + info->fields[i].first + "`"); } } @@ -1060,8 +994,7 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, packed.reserve(1 + ordered.size()); packed.push_back(static_cast(ordered.size())); for (JirRef r : ordered) packed.push_back(r); - JirExtraIdx extra = - gctx.jfn.pushExtra(packed.data(), packed.size()); + JirExtraIdx extra = gctx.jfn.pushExtra(packed.data(), packed.size()); JirInst inst{}; inst.tag = JirTag::StructLit; @@ -1085,8 +1018,8 @@ static JirRef astgenMemberAccess(AstGenCtx &gctx, const AstNode &n) { const AstNode &baseNode = ns.get(baseIdx); if (baseNode.tag == AstTag::Variable) { - const std::string &baseName = gctx.ctx.getStringPool().get( - static_cast(baseNode.lhs)); + const std::string &baseName = + gctx.ctx.getStringPool().get(static_cast(baseNode.lhs)); // Enum-variant unit reference: emit JirTag::StructLit with a // single field (the tag) for payloaded enums; for unit-only // enums, the tag value IS the runtime form (i8). @@ -1094,22 +1027,20 @@ static JirRef astgenMemberAccess(AstGenCtx &gctx, const AstNode &n) { int vidx = gctx.ctx.getEnumVariantIndex(baseName, member); if (vidx < 0) { failHere(gctx, "astgen: enum `" + baseName + - "` has no variant `" + member + "`"); + "` has no variant `" + member + "`"); } uint32_t disc = einfo->variants[vidx].discriminant; TypeIdx enumTy = gctx.ctx.getTypePool().intern( TypeKey{TypeKind::Named, 0, 0, - static_cast(gctx.ctx.getStringPool().intern( - baseName)), + static_cast( + gctx.ctx.getStringPool().intern(baseName)), 0}); JirInst tag{}; tag.tag = JirTag::Int; tag.a = disc; tag.ty = BuiltinType::U8; JirRef tagRef = emit(gctx, tag); - if (!einfo->hasPayloadVariant) { - return tagRef; - } + if (!einfo->hasPayloadVariant) { return tagRef; } // Payloaded enum: build a {tag, payload-undef} struct. // We carry the enum's TypeIdx so codegen materialises the // right struct shape. @@ -1133,8 +1064,7 @@ static JirRef astgenMemberAccess(AstGenCtx &gctx, const AstNode &n) { const TypeKey &bk = gctx.ctx.getTypePool().get(baseTy); if (bk.kind == TypeKind::Slice) { if (member != "ptr" && member != "len") { - failHere(gctx, - "astgen: slice has no field `" + member + "`"); + failHere(gctx, "astgen: slice has no field `" + member + "`"); } unsigned fieldIdx = (member == "ptr") ? 0 : 1; TypeIdx fieldTy; @@ -1157,12 +1087,10 @@ static JirRef astgenMemberAccess(AstGenCtx &gctx, const AstNode &n) { // fieldTy — opaque pointers let us hand back any field-typed // value from the same storage. if (const auto *uinfo = gctx.ctx.lookupUnion(baseTy)) { - TypeIdx fieldTy = - gctx.ctx.getUnionFieldType(uinfo->name, member); + TypeIdx fieldTy = gctx.ctx.getUnionFieldType(uinfo->name, member); if (fieldTy == kNoType) { - failHere(gctx, - "astgen: union `" + uinfo->name + "` has no field `" + - member + "`"); + failHere(gctx, "astgen: union `" + uinfo->name + + "` has no field `" + member + "`"); } JirInst alloca{}; alloca.tag = JirTag::Alloca; @@ -1182,14 +1110,13 @@ static JirRef astgenMemberAccess(AstGenCtx &gctx, const AstNode &n) { const auto *info = gctx.ctx.lookupStruct(baseTy); if (info == nullptr) { - failHere(gctx, - "astgen: cannot access field of non-struct type"); + failHere(gctx, "astgen: cannot access field of non-struct type"); } int idx = gctx.ctx.getFieldIndex(info->name, member); if (idx < 0) { - return recoverHere(gctx, "unknown field `" + member + "` on `" + - info->name + "`", - kNoType); + return recoverHere( + gctx, "unknown field `" + member + "` on `" + info->name + "`", + kNoType); } JirInst inst{}; inst.tag = JirTag::FieldAccess; @@ -1223,8 +1150,8 @@ static JirRef astgenArrayLit(AstGenCtx &gctx, const AstNode &n, elemTy = gctx.jfn.getInst(elems[0]).ty; } if (elemTy == kNoType) { - failHere(gctx, - "astgen: array literal element type could not be inferred"); + failHere(gctx, + "astgen: array literal element type could not be inferred"); } TypeIdx arrTy = gctx.ctx.getTypePool().intern( TypeKey{TypeKind::Array, 0, 0, elemTy, count}); @@ -1233,8 +1160,7 @@ static JirRef astgenArrayLit(AstGenCtx &gctx, const AstNode &n, packed.reserve(1 + elems.size()); packed.push_back(count); for (JirRef r : elems) packed.push_back(r); - JirExtraIdx extra = - gctx.jfn.pushExtra(packed.data(), packed.size()); + JirExtraIdx extra = gctx.jfn.pushExtra(packed.data(), packed.size()); JirInst inst{}; inst.tag = JirTag::ArrayLit; @@ -1258,11 +1184,12 @@ static JirRef astgenArrayRepeat(AstGenCtx &gctx, const AstNode &n, const AstNode &cn = ns.get(countIdx); if (cn.tag != AstTag::NumberLit) { - failHere(gctx, + failHere( + gctx, "astgen: array-repeat count must be a constant integer literal"); } - uint64_t count = static_cast(cn.lhs) | - (static_cast(cn.rhs) << 32); + uint64_t count = + static_cast(cn.lhs) | (static_cast(cn.rhs) << 32); // Resolve element type. From the array TypeKey if we have it; else // from the first compile of the value. @@ -1306,8 +1233,7 @@ static JirRef astgenIndex(AstGenCtx &gctx, const AstNode &n) { k.kind == TypeKind::PtrMany) { elemTy = static_cast(k.a); } else { - failHere(gctx, - "astgen: cannot index value of this type"); + failHere(gctx, "astgen: cannot index value of this type"); } JirInst inst{}; inst.tag = JirTag::Index; @@ -1396,8 +1322,7 @@ static JirRef astgenAsCast(AstGenCtx &gctx, const AstNode &n) { } } const TypeKey &dst = gctx.ctx.getTypePool().get(dstTy); - TypeIdx hint = - (dst.kind == TypeKind::Int) ? dstTy : kNoType; + TypeIdx hint = (dst.kind == TypeKind::Int) ? dstTy : kNoType; JirRef val = astgenExpr(gctx, operandIdx, hint); TypeIdx srcTy = gctx.jfn.getInst(val).ty; if (srcTy == dstTy) return val; @@ -1480,8 +1405,7 @@ static JirRef astgenAsCast(AstGenCtx &gctx, const AstNode &n) { // enums the runtime form is already i8, so the cast is just a // width adjustment. For payloaded enums it's an ExtractValue(0) // followed by the same width adjustment. - if ((src.kind == TypeKind::Named || - src.kind == TypeKind::Enum) && + if ((src.kind == TypeKind::Named || src.kind == TypeKind::Enum) && dst.kind == TypeKind::Int) { if (const auto *einfo = gctx.ctx.lookupEnum(srcTy)) { JirRef tagRef = val; @@ -1511,8 +1435,7 @@ static JirRef astgenAsCast(AstGenCtx &gctx, const AstNode &n) { // Pointer ↔ pointer cast: in opaque-pointer LLVM the runtime // representation is identical, so just retag at the JIR level. auto isPtr = [](const TypeKey &k) { - return k.kind == TypeKind::PtrSingle || - k.kind == TypeKind::PtrMany; + return k.kind == TypeKind::PtrSingle || k.kind == TypeKind::PtrMany; }; if (isPtr(src) && isPtr(dst)) { JirInst inst{}; @@ -1633,9 +1556,9 @@ static JirRef astgenBinaryOp(AstGenCtx &gctx, const AstNode &n, const TypeKey &rk = gctx.ctx.getTypePool().get(rhsType); if (lk.kind == TypeKind::Float && rk.kind == TypeKind::Float && lk.a != rk.a) { - failHere(gctx, - "mismatched float widths in binary op; use an explicit " - "`as` cast to align them"); + failHere(gctx, + "mismatched float widths in binary op; use an explicit " + "`as` cast to align them"); } if (lk.kind == TypeKind::Int && rk.kind == TypeKind::Int && lk.a != rk.a) { @@ -1675,7 +1598,8 @@ static JirRef astgenBinaryOp(AstGenCtx &gctx, const AstNode &n, // LogOr: result = lhs ? true : rhs // Use an alloca for the result slot; codegen folds to phi at -O2. bool isAnd = op == BinOp::LogAnd; - // Re-emit operands as bools: lhsRef is already i1 (we cast expected to bool). + // Re-emit operands as bools: lhsRef is already i1 (we cast expected to + // bool). (void)lhsRef; (void)rhsRef; (void)resultTy; @@ -1721,20 +1645,36 @@ static JirRef astgenBinaryOp(AstGenCtx &gctx, const AstNode &n, } switch (op) { - case BinOp::Add: tag = isFloat ? JirTag::FAdd : JirTag::Add; break; - case BinOp::Sub: tag = isFloat ? JirTag::FSub : JirTag::Sub; break; - case BinOp::Mul: tag = isFloat ? JirTag::FMul : JirTag::Mul; break; + case BinOp::Add: + tag = isFloat ? JirTag::FAdd : JirTag::Add; + break; + case BinOp::Sub: + tag = isFloat ? JirTag::FSub : JirTag::Sub; + break; + case BinOp::Mul: + tag = isFloat ? JirTag::FMul : JirTag::Mul; + break; case BinOp::Div: tag = isFloat ? JirTag::FDiv : (isSigned ? JirTag::SDiv : JirTag::UDiv); break; case BinOp::Mod: tag = isFloat ? JirTag::FRem : (isSigned ? JirTag::SRem : JirTag::URem); break; - case BinOp::BitAnd: tag = JirTag::BitAnd; break; - case BinOp::BitOr: tag = JirTag::BitOr; break; - case BinOp::BitXor: tag = JirTag::BitXor; break; - case BinOp::Shl: tag = JirTag::Shl; break; - case BinOp::Shr: tag = isSigned ? JirTag::AShr : JirTag::LShr; break; + case BinOp::BitAnd: + tag = JirTag::BitAnd; + break; + case BinOp::BitOr: + tag = JirTag::BitOr; + break; + case BinOp::BitXor: + tag = JirTag::BitXor; + break; + case BinOp::Shl: + tag = JirTag::Shl; + break; + case BinOp::Shr: + tag = isSigned ? JirTag::AShr : JirTag::LShr; + break; case BinOp::Eq: tag = isFloat ? JirTag::FCmpOeq : JirTag::ICmpEq; isCmp = true; @@ -1765,8 +1705,8 @@ static JirRef astgenBinaryOp(AstGenCtx &gctx, const AstNode &n, break; default: failHere(gctx, "unsupported binary operator (internal op = " + - std::to_string(static_cast(op)) + - ") — please file a bug"); + std::to_string(static_cast(op)) + + ") — please file a bug"); } JirInst inst{}; @@ -1825,8 +1765,8 @@ static void astgenIf(AstGenCtx &gctx, const AstNode &n) { JirRef condRef = astgenExpr(gctx, condIdx, BuiltinType::Bool); JirBlockRef thenB = gctx.jfn.pushBlock("then"); - JirBlockRef elseB = (elseCount > 0) ? gctx.jfn.pushBlock("else") - : kNoJirBlock; + JirBlockRef elseB = + (elseCount > 0) ? gctx.jfn.pushBlock("else") : kNoJirBlock; JirBlockRef mergeB = gctx.jfn.pushBlock("ifend"); emitCondBr(gctx, condRef, thenB, (elseCount > 0) ? elseB : mergeB); @@ -1849,8 +1789,8 @@ static void astgenIf(AstGenCtx &gctx, const AstNode &n) { gctx.currentBlock = elseB; pushDropScope(gctx); for (uint32_t i = 0; i < elseCount; i++) { - NodeIdx s = static_cast( - ns.getExtra(extra + 2 + thenCount + i)); + NodeIdx s = + static_cast(ns.getExtra(extra + 2 + thenCount + i)); astgenExpr(gctx, s, kNoType); } popDropScopeEmitting(gctx); @@ -1923,8 +1863,7 @@ static void astgenFor(AstGenCtx &gctx, const AstNode &n) { // destroyed before each step / break. gctx.currentBlock = bodyB; pushDropScope(gctx); - gctx.loopStack.push_back( - {stepB, exitB, gctx.dropScopes.size() - 1}); + gctx.loopStack.push_back({stepB, exitB, gctx.dropScopes.size() - 1}); for (uint32_t i = 0; i < bodyCount; i++) { NodeIdx s = static_cast(ns.getExtra(extra + 4 + i)); astgenExpr(gctx, s, kNoType); @@ -1985,8 +1924,7 @@ static void astgenWhile(AstGenCtx &gctx, const AstNode &n) { gctx.currentBlock = bodyB; pushDropScope(gctx); - gctx.loopStack.push_back( - {condB, exitB, gctx.dropScopes.size() - 1}); + gctx.loopStack.push_back({condB, exitB, gctx.dropScopes.size() - 1}); for (uint32_t i = 0; i < bodyCount; i++) { NodeIdx s = static_cast(ns.getExtra(extra + 1 + i)); astgenExpr(gctx, s, kNoType); @@ -2004,8 +1942,7 @@ static void astgenBreak(AstGenCtx &gctx) { if (gctx.loopStack.empty()) { failHere(gctx, "astgen: `break` outside of loop"); } - emitDropsThroughScope(gctx, - gctx.loopStack.back().bodyScopeIdx); + emitDropsThroughScope(gctx, gctx.loopStack.back().bodyScopeIdx); emitBr(gctx, gctx.loopStack.back().breakBlock); } @@ -2013,8 +1950,7 @@ static void astgenContinue(AstGenCtx &gctx) { if (gctx.loopStack.empty()) { failHere(gctx, "astgen: `continue` outside of loop"); } - emitDropsThroughScope(gctx, - gctx.loopStack.back().bodyScopeIdx); + emitDropsThroughScope(gctx, gctx.loopStack.back().bodyScopeIdx); emitBr(gctx, gctx.loopStack.back().continueBlock); } @@ -2070,8 +2006,8 @@ static TypeIdx inferTailType(const AstGenCtx &gctx, NodeIdx idx) { const AstNode &n = gctx.ctx.getNodeStore().get(idx); switch (n.tag) { case AstTag::NumberLit: { - uint64_t val = static_cast(n.lhs) | - (static_cast(n.rhs) << 32); + uint64_t val = + static_cast(n.lhs) | (static_cast(n.rhs) << 32); bool isNeg = (n.flags & 1) != 0; bool isFloat = (n.flags & 2) != 0; if (isFloat) return BuiltinType::F64; @@ -2089,8 +2025,8 @@ static TypeIdx inferTailType(const AstGenCtx &gctx, NodeIdx idx) { case AstTag::BoolLit: return BuiltinType::Bool; case AstTag::Variable: { - const std::string &name = gctx.ctx.getStringPool().get( - static_cast(n.lhs)); + const std::string &name = + gctx.ctx.getStringPool().get(static_cast(n.lhs)); auto it = gctx.localTypes.find(name); if (it != gctx.localTypes.end()) return it->second; return kNoType; @@ -2111,10 +2047,9 @@ static bool stmtDiverges(const AstGenCtx &gctx, NodeIdx node) { return true; } if (n.tag == AstTag::Call && (n.flags & 1) == 0) { - const std::string &callee = gctx.ctx.getStringPool().get( - static_cast(n.lhs)); - if (const FunctionAST *fn = - gctx.ctx.getFunctionAST(callee)) { + const std::string &callee = + gctx.ctx.getStringPool().get(static_cast(n.lhs)); + if (const FunctionAST *fn = gctx.ctx.getFunctionAST(callee)) { return fn->ReturnType == BuiltinType::NoReturn; } } @@ -2156,20 +2091,17 @@ struct SwitchCase { // `EnumInfo` is supplied so we can resolve the // variant name without re-doing the lookup // later. -static bool collectSwitchCases(const NodeStore &ns, - JamCodegenContext &ctx, - NodeIdx patIdx, - JirBlockRef armBlock, - TypeIdx scrutTy, - bool scrutIsEnum, +static bool collectSwitchCases(const NodeStore &ns, JamCodegenContext &ctx, + NodeIdx patIdx, JirBlockRef armBlock, + TypeIdx scrutTy, bool scrutIsEnum, const JamCodegenContext::EnumInfo *einfo, std::vector &out) { const AstNode &p = ns.get(patIdx); switch (p.tag) { case AstTag::PatLit: { if (scrutIsEnum) return false; - uint64_t val = static_cast(p.lhs) | - (static_cast(p.rhs) << 32); + uint64_t val = + static_cast(p.lhs) | (static_cast(p.rhs) << 32); bool isNeg = (p.flags & 1) != 0; if (isNeg) { // Two's-complement encoding at the scrut's width; the @@ -2188,8 +2120,7 @@ static bool collectSwitchCases(const NodeStore &ns, bool hasBindings = (p.flags & 1) != 0; if (hasBindings) return false; StringIdx variantNameId = static_cast(p.rhs); - const std::string &variantName = - ctx.getStringPool().get(variantNameId); + const std::string &variantName = ctx.getStringPool().get(variantNameId); int vidx = ctx.getEnumVariantIndex(einfo->name, variantName); if (vidx < 0) return false; uint64_t disc = einfo->variants[vidx].discriminant; @@ -2202,7 +2133,7 @@ static bool collectSwitchCases(const NodeStore &ns, for (uint32_t i = 0; i < cnt; i++) { NodeIdx sub = static_cast(ns.getExtra(ex + 1 + i)); if (!collectSwitchCases(ns, ctx, sub, armBlock, scrutTy, - scrutIsEnum, einfo, out)) { + scrutIsEnum, einfo, out)) { return false; } } @@ -2224,9 +2155,8 @@ static bool collectSwitchCases(const NodeStore &ns, // payload bindings introduced by the pattern. The bindings' slot // allocas are emitted in `outBindings` order so arm-body code can // install them by Load via Variable. -static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, - JirRef scrut, TypeIdx scrutTy, - JirBlockRef armBlock, +static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, JirRef scrut, + TypeIdx scrutTy, JirBlockRef armBlock, JirBlockRef nextBlock, ArmBindings *outBindings) { const NodeStore &ns = gctx.ctx.getNodeStore(); @@ -2254,8 +2184,8 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, switch (p.tag) { case AstTag::PatLit: { - uint64_t val = static_cast(p.lhs) | - (static_cast(p.rhs) << 32); + uint64_t val = + static_cast(p.lhs) | (static_cast(p.rhs) << 32); bool isNeg = (p.flags & 1) != 0; JirRef k = emitInt(val, isNeg); JirRef cmp = emitCmp(JirTag::ICmpEq, scrut, k); @@ -2268,13 +2198,13 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, uint64_t hi = static_cast(p.rhs); JirRef loK = emitInt(lo, false); JirRef hiK = emitInt(hi, false); - JirRef geLo = emitCmp(signedCmp ? JirTag::ICmpSge : JirTag::ICmpUge, - scrut, loK); + JirRef geLo = + emitCmp(signedCmp ? JirTag::ICmpSge : JirTag::ICmpUge, scrut, loK); JirBlockRef checkHi = gctx.jfn.pushBlock("range.hi"); emitCondBr(gctx, geLo, checkHi, nextBlock); gctx.currentBlock = checkHi; - JirRef leHi = emitCmp(signedCmp ? JirTag::ICmpSle : JirTag::ICmpUle, - scrut, hiK); + JirRef leHi = + emitCmp(signedCmp ? JirTag::ICmpSle : JirTag::ICmpUle, scrut, hiK); emitCondBr(gctx, leHi, armBlock, nextBlock); return; } @@ -2283,11 +2213,10 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, uint32_t cnt = ns.getExtra(ex); for (uint32_t i = 0; i < cnt; i++) { NodeIdx sub = static_cast(ns.getExtra(ex + 1 + i)); - JirBlockRef tryNext = (i + 1 == cnt) - ? nextBlock - : gctx.jfn.pushBlock("or.next"); - astgenPatternCompare(gctx, sub, scrut, scrutTy, armBlock, - tryNext, outBindings); + JirBlockRef tryNext = + (i + 1 == cnt) ? nextBlock : gctx.jfn.pushBlock("or.next"); + astgenPatternCompare(gctx, sub, scrut, scrutTy, armBlock, tryNext, + outBindings); if (i + 1 != cnt) gctx.currentBlock = tryNext; } return; @@ -2317,8 +2246,8 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, } std::string enumName; - const auto *einfo = static_cast( - nullptr); + const auto *einfo = + static_cast(nullptr); if (typeIdxReceiver) { TypeIdx ty = static_cast(recvSlot); if (const auto *info = gctx.ctx.lookupEnum(ty)) { @@ -2326,21 +2255,21 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, einfo = info; } } else { - std::string recvName = gctx.ctx.getStringPool().get( - static_cast(recvSlot)); + std::string recvName = + gctx.ctx.getStringPool().get(static_cast(recvSlot)); enumName = recvName; einfo = gctx.ctx.getEnum(recvName); } if (einfo == nullptr) { - failHere(gctx, - "astgen: pattern receiver doesn't resolve to an enum"); + failHere(gctx, + "astgen: pattern receiver doesn't resolve to an enum"); } const std::string &variantName = gctx.ctx.getStringPool().get(variantNameId); int vidx = gctx.ctx.getEnumVariantIndex(enumName, variantName); if (vidx < 0) { - failHere(gctx, "astgen: unknown variant `" + enumName + - "." + variantName + "`"); + failHere(gctx, "astgen: unknown variant `" + enumName + "." + + variantName + "`"); } // Extract the tag as a u8. For payloaded enums the tag is @@ -2383,10 +2312,10 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, gctx.currentBlock = bindB; const auto &variant = einfo->variants[vidx]; if (bindingCount != variant.payloadTypes.size()) { - failHere(gctx, - "astgen: pattern binds " + std::to_string(bindingCount) + - " field(s), variant has " + - std::to_string(variant.payloadTypes.size())); + failHere(gctx, "astgen: pattern binds " + + std::to_string(bindingCount) + + " field(s), variant has " + + std::to_string(variant.payloadTypes.size())); } // Spill scrut to a local alloca so EnumPayload can take its @@ -2403,8 +2332,8 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, uint64_t off = 0; for (uint32_t b = 0; b < bindingCount; b++) { - StringIdx bindNameId = static_cast( - ns.getExtra(bindingsStart + b)); + StringIdx bindNameId = + static_cast(ns.getExtra(bindingsStart + b)); const std::string &bindName = gctx.ctx.getStringPool().get(bindNameId); TypeIdx fieldTy = variant.payloadTypes[b]; @@ -2429,8 +2358,7 @@ static void astgenPatternCompare(AstGenCtx &gctx, NodeIdx patIdx, bindStore.b = payloadRef; emit(gctx, bindStore); if (outBindings) { - outBindings->push_back( - {bindName, bindSlot, fieldTy}); + outBindings->push_back({bindName, bindSlot, fieldTy}); } off += s; } @@ -2470,8 +2398,8 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { uint32_t bc = ns.getExtra(armsExtra + pos + 1); a.body.reserve(bc); for (uint32_t j = 0; j < bc; j++) { - a.body.push_back(static_cast( - ns.getExtra(armsExtra + pos + 2 + j))); + a.body.push_back( + static_cast(ns.getExtra(armsExtra + pos + 2 + j))); } pos += 2 + bc; if (ns.get(a.patIdx).tag == AstTag::PatWildcard) { @@ -2522,8 +2450,8 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { std::vector armBindings(armCount); JirBlockRef defaultB = (wildcardArmIdx >= 0) - ? armBlocks[wildcardArmIdx] - : gctx.jfn.pushBlock("nomatch"); + ? armBlocks[wildcardArmIdx] + : gctx.jfn.pushBlock("nomatch"); // Try the Switch lowering first. Every non-wildcard arm has to be // a pattern that resolves to a single integer-equality test (or a @@ -2542,9 +2470,8 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { const NodeStore &ns = gctx.ctx.getNodeStore(); for (uint32_t i = 0; i < armCount; i++) { if (static_cast(i) == wildcardArmIdx) continue; - if (!collectSwitchCases(ns, gctx.ctx, arms[i].patIdx, - armBlocks[i], scrutTy, scrutIsEnum, - einfo, switchCases)) { + if (!collectSwitchCases(ns, gctx.ctx, arms[i].patIdx, armBlocks[i], + scrutTy, scrutIsEnum, einfo, switchCases)) { tryingSwitch = false; break; } @@ -2586,8 +2513,7 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { packed.push_back(sc.isSigned ? 1u : 0u); packed.push_back(static_cast(sc.target)); } - JirExtraIdx extraIdx = - gctx.jfn.pushExtra(packed.data(), packed.size()); + JirExtraIdx extraIdx = gctx.jfn.pushExtra(packed.data(), packed.size()); JirInst sw{}; sw.tag = JirTag::Switch; sw.a = caseScrut; @@ -2609,10 +2535,10 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { bool emittedAnyDispatch = false; for (uint32_t i = 0; i < armCount; i++) { if (static_cast(i) == wildcardArmIdx) continue; - JirBlockRef next = (i + 1 < armCount && - static_cast(i + 1) != wildcardArmIdx) - ? gctx.jfn.pushBlock("matchnext") - : defaultB; + JirBlockRef next = + (i + 1 < armCount && static_cast(i + 1) != wildcardArmIdx) + ? gctx.jfn.pushBlock("matchnext") + : defaultB; astgenPatternCompare(gctx, arms[i].patIdx, scrut, scrutTy, armBlocks[i], next, &armBindings[i]); if (next != defaultB) gctx.currentBlock = next; @@ -2621,9 +2547,7 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { // All arms were wildcards (only a `_` arm, or none). Emit an // unconditional Br from the entry block into the default // block. - if (!emittedAnyDispatch) { - emitBr(gctx, defaultB); - } + if (!emittedAnyDispatch) { emitBr(gctx, defaultB); } // If the last non-wildcard arm fell through, we need its // `next` to terminate. The compare loop above already wired // the final next to `defaultB`. When there's no catch-all @@ -2675,9 +2599,7 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { } else { astgenExpr(gctx, stmt, kNoType); } - if (isTail && divergent) { - armDiverged = true; - } + if (isTail && divergent) { armDiverged = true; } } // A noreturn-call tail (like `abort()`) leaves the arm with no // LLVM terminator even though semantically control can't flow @@ -2701,8 +2623,7 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { gctx.localTypes.erase(bind.name); } for (const auto &kv : saved) gctx.locals[kv.first] = kv.second; - for (const auto &kv : savedTypes) - gctx.localTypes[kv.first] = kv.second; + for (const auto &kv : savedTypes) gctx.localTypes[kv.first] = kv.second; } gctx.currentBlock = mergeB; @@ -2728,8 +2649,7 @@ static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n) { ExtraIdx extra = static_cast(n.rhs); StringIdx methodNameId = static_cast(ns.getExtra(extra)); uint32_t argCount = ns.getExtra(extra + 1); - const std::string &methodName = - gctx.ctx.getStringPool().get(methodNameId); + const std::string &methodName = gctx.ctx.getStringPool().get(methodNameId); std::string receiverName; const auto *einfoForVariant = @@ -2740,9 +2660,8 @@ static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n) { receiverName = einfo->name; einfoForVariant = einfo; } else { - failHere(gctx, - "astgen: TypeMethodCall receiver doesn't resolve to a " - "struct or enum"); + failHere(gctx, "astgen: TypeMethodCall receiver doesn't resolve to a " + "struct or enum"); } // Enum-variant constructor: `Option(i32).Some(42)` — the method @@ -2750,23 +2669,20 @@ static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n) { // {tag, payload} aggregate. v1 supports single-payload variants // (the dominant case for Option/Result/either-style enums). if (einfoForVariant != nullptr) { - int vidx = - gctx.ctx.getEnumVariantIndex(receiverName, methodName); + int vidx = gctx.ctx.getEnumVariantIndex(receiverName, methodName); if (vidx >= 0) { const auto &variant = einfoForVariant->variants[vidx]; - TypeIdx enumTy = gctx.ctx.getTypePool().intern(TypeKey{ - TypeKind::Named, 0, 0, - static_cast( - gctx.ctx.getStringPool().intern(receiverName)), - 0}); + TypeIdx enumTy = gctx.ctx.getTypePool().intern( + TypeKey{TypeKind::Named, 0, 0, + static_cast( + gctx.ctx.getStringPool().intern(receiverName)), + 0}); JirInst tag{}; tag.tag = JirTag::Int; tag.a = static_cast(variant.discriminant); tag.ty = BuiltinType::U8; JirRef tagRef = emit(gctx, tag); - if (!einfoForVariant->hasPayloadVariant) { - return tagRef; - } + if (!einfoForVariant->hasPayloadVariant) { return tagRef; } // Payloaded enum: alloca enum struct, store tag at field 0, // store each payload arg at the payload area. JirInst alloca{}; @@ -2859,8 +2775,7 @@ static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n) { // callee's FunctionAST via the same registry path. const FunctionAST *fn = gctx.ctx.getFunctionAST(qualified); if (fn == nullptr) { - return recoverHere(gctx, "unknown method `" + qualified + "`", - kNoType); + return recoverHere(gctx, "unknown method `" + qualified + "`", kNoType); } std::vector argRefs; argRefs.reserve(argCount); @@ -2915,8 +2830,7 @@ static JirRef astgenAssertCall(AstGenCtx &gctx, const AstNode &n) { ExtraIdx argsExtra = static_cast(n.rhs); uint32_t argCount = ns.getExtra(argsExtra); if (argCount != 2) { - failHere(gctx, - "astgen: assert expects exactly 2 arguments"); + failHere(gctx, "astgen: assert expects exactly 2 arguments"); } NodeIdx actualIdx = static_cast(ns.getExtra(argsExtra + 1)); NodeIdx expectedIdx = static_cast(ns.getExtra(argsExtra + 2)); @@ -2968,27 +2882,25 @@ static JirRef astgenAssertCall(AstGenCtx &gctx, const AstNode &n) { // up parameter types. We mark it varArgs so the prototype // matches the C signature. auto fakePrintf = std::make_unique( - "printf", std::vector{Param{ - "fmt", BuiltinType::U64, ParamMode::Let}}, + "printf", + std::vector{Param{"fmt", BuiltinType::U64, ParamMode::Let}}, BuiltinType::I32, std::vector{}, /*isExtern=*/true, /*isExport=*/false, /*isPub=*/false, /*isTest=*/false, /*isVarArgs=*/true); printfAST = fakePrintf.get(); gctx.ctx.registerFunctionAST("printf", fakePrintf.release()); // Also declare the LLVM prototype so the call resolves. - JamTypeRef i8PtrType = - JamLLVMPointerType(gctx.ctx.getInt8Type(), 0); + JamTypeRef i8PtrType = JamLLVMPointerType(gctx.ctx.getInt8Type(), 0); JamTypeRef paramTypes[1] = {i8PtrType}; - JamTypeRef ft = JamLLVMFunctionType( - gctx.ctx.getInt32Type(), paramTypes, 1, /*isVarArgs=*/true); - JamFunctionRef pf = JamLLVMAddFunction( - gctx.ctx.getModule(), "printf", ft); + JamTypeRef ft = JamLLVMFunctionType(gctx.ctx.getInt32Type(), paramTypes, + 1, /*isVarArgs=*/true); + JamFunctionRef pf = + JamLLVMAddFunction(gctx.ctx.getModule(), "printf", ft); JamLLVMApplyDefaultFnAttrs(pf, /*isExtern=*/true); } // Build the format-string slice "Assertion failed\n" as a JIR Str. - StringIdx msgId = - gctx.ctx.getStringPool().intern("Assertion failed\n"); + StringIdx msgId = gctx.ctx.getStringPool().intern("Assertion failed\n"); TypeIdx sliceTy = gctx.ctx.getTypePool().intern( TypeKey{TypeKind::Slice, 0, 0, BuiltinType::U8, 0}); JirInst strInst{}; @@ -3020,18 +2932,18 @@ static JirRef astgenAssertCall(AstGenCtx &gctx, const AstNode &n) { const FunctionAST *exitAST = gctx.ctx.getFunctionAST("exit"); if (exitAST == nullptr) { auto fakeExit = std::make_unique( - "exit", std::vector{Param{"code", BuiltinType::I32, - ParamMode::Let}}, + "exit", + std::vector{Param{"code", BuiltinType::I32, ParamMode::Let}}, kNoType, std::vector{}, /*isExtern=*/true, /*isExport=*/false, /*isPub=*/false, /*isTest=*/false, /*isVarArgs=*/false); exitAST = fakeExit.get(); gctx.ctx.registerFunctionAST("exit", fakeExit.release()); JamTypeRef exitParamTypes[1] = {gctx.ctx.getInt32Type()}; - JamTypeRef et = JamLLVMFunctionType( - gctx.ctx.getVoidType(), exitParamTypes, 1, false); - JamFunctionRef ef = JamLLVMAddFunction( - gctx.ctx.getModule(), "exit", et); + JamTypeRef et = JamLLVMFunctionType(gctx.ctx.getVoidType(), + exitParamTypes, 1, false); + JamFunctionRef ef = + JamLLVMAddFunction(gctx.ctx.getModule(), "exit", et); JamLLVMApplyDefaultFnAttrs(ef, /*isExtern=*/true); } JirInst exitCode{}; @@ -3074,7 +2986,7 @@ static JirRef astgenAssertCall(AstGenCtx &gctx, const AstNode &n) { // substitution happens. Every call-dispatch path goes through // here. Don't add a second one. static std::string resolvePrefix(JamCodegenContext &ctx, - const std::string &prefix) { + const std::string &prefix) { TypeIdx ty = ctx.lookupCurrentSubst(prefix); if (ty == kNoType) return prefix; if (const auto *sinfo = ctx.lookupStruct(ty)) return sinfo->name; @@ -3089,8 +3001,7 @@ static std::string resolvePrefix(JamCodegenContext &ctx, // param's TypeIdx as the expected hint so literals settle at the // right width. static JirRef lowerArg(AstGenCtx &gctx, NodeIdx argIdx, const Param &p) { - jam::abi::ParamABI pabi = - jam::abi::classifyParam(p.Mode, p.Type, gctx.ctx); + jam::abi::ParamABI pabi = jam::abi::classifyParam(p.Mode, p.Type, gctx.ctx); if (pabi.kind != jam::abi::ParamABI::Kind::ByPointer) { return astgenExpr(gctx, argIdx, p.Type); } @@ -3112,7 +3023,8 @@ static JirRef lowerArg(AstGenCtx &gctx, NodeIdx argIdx, const Param &p) { return astgenLvalue(gctx, argIdx, leafTy); case AstTag::AddressOf: return astgenExpr(gctx, argIdx, kNoType); - default: break; + default: + break; } } JirRef val = astgenExpr(gctx, argIdx, p.Type); @@ -3131,16 +3043,14 @@ static JirRef lowerArg(AstGenCtx &gctx, NodeIdx argIdx, const Param &p) { static JirRef emitCall(AstGenCtx &gctx, const FunctionAST *fn, const std::vector &argRefs) { - std::string mangled = mangledFunctionName( - *fn, gctx.ctx.getTypePool(), gctx.ctx.getStringPool()); - StringIdx calleeId = - gctx.ctx.getStringPool().intern(mangled); + std::string mangled = mangledFunctionName(*fn, gctx.ctx.getTypePool(), + gctx.ctx.getStringPool()); + StringIdx calleeId = gctx.ctx.getStringPool().intern(mangled); std::vector packed; packed.reserve(1 + argRefs.size()); packed.push_back(static_cast(argRefs.size())); for (JirRef r : argRefs) packed.push_back(r); - JirExtraIdx extra = - gctx.jfn.pushExtra(packed.data(), packed.size()); + JirExtraIdx extra = gctx.jfn.pushExtra(packed.data(), packed.size()); JirInst call{}; call.tag = JirTag::Call; call.a = calleeId; @@ -3174,8 +3084,8 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { NodeIdx calleeNodeIdx = static_cast(n.lhs); const AstNode &cn = ns.get(calleeNodeIdx); if (cn.tag != AstTag::MemberAccess) { - failHere(gctx, - "astgen: indirect call must be on a `.method` callee"); + failHere(gctx, + "astgen: indirect call must be on a `.method` callee"); } NodeIdx recvExprIdx = static_cast(cn.lhs); StringIdx methodNameId = static_cast(cn.rhs); @@ -3197,17 +3107,17 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { } else if (const auto *einfo = gctx.ctx.lookupEnum(recvTy)) { recvName = einfo->name; } else { - failHere(gctx, - "astgen: indirect-call receiver is not a struct/enum"); + failHere(gctx, + "astgen: indirect-call receiver is not a struct/enum"); } std::string qualified = recvName + "." + methodName; const FunctionAST *method = gctx.ctx.getFunctionAST(qualified); if (method == nullptr) { return recoverHere(gctx, "unknown method `" + qualified + "`", - kNoType); + kNoType); } - ParamMode mode = method->Args.empty() ? ParamMode::Let - : method->Args[0].Mode; + ParamMode mode = + method->Args.empty() ? ParamMode::Let : method->Args[0].Mode; JirRef recvArg = recvVal; if (mode == ParamMode::Mut || mode == ParamMode::Move) { // Re-lower the receiver as an lvalue so the method sees @@ -3245,8 +3155,8 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { NodeIdx argIdx = static_cast(ns.getExtra(argsExtra + 1 + i)); TypeIdx expectArg = (1 + i < method->Args.size()) - ? method->Args[1 + i].Type - : kNoType; + ? method->Args[1 + i].Type + : kNoType; argRefs.push_back(astgenExpr(gctx, argIdx, expectArg)); } return emitCall(gctx, method, argRefs); @@ -3257,9 +3167,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { // Builtin: `assert(actual, expected)` — handled via JIR inline, // not as a regular function call. - if (callee == "assert") { - return astgenAssertCall(gctx, n); - } + if (callee == "assert") { return astgenAssertCall(gctx, n); } // Single-dot qualified call: try in order // 1. `inst.method(args)` — instance dispatch on a local variable @@ -3274,8 +3182,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { // `Self.method(...)` inside an instantiated generic body // resolves through the substitution context set up by // instantiateStructExpr before lowering the body. - std::string resolvedPrefix = - resolvePrefix(gctx.ctx, prefix); + std::string resolvedPrefix = resolvePrefix(gctx.ctx, prefix); // Type-alias / type-name static dispatch: if `prefix` names // a registered struct (or aliases one), rewrite the call to @@ -3283,11 +3190,9 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { std::string canonicalType; const auto *enumForVariant = static_cast(nullptr); - TypeIdx aliasTarget = - gctx.ctx.lookupTypeAlias(resolvedPrefix); + TypeIdx aliasTarget = gctx.ctx.lookupTypeAlias(resolvedPrefix); if (aliasTarget != kNoType) { - if (const auto *sinfo = - gctx.ctx.lookupStruct(aliasTarget)) { + if (const auto *sinfo = gctx.ctx.lookupStruct(aliasTarget)) { canonicalType = sinfo->name; } else if (const auto *einfo = gctx.ctx.lookupEnum(aliasTarget)) { @@ -3296,8 +3201,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { } } if (canonicalType.empty()) { - if (const auto *sinfo = - gctx.ctx.getStruct(resolvedPrefix)) { + if (const auto *sinfo = gctx.ctx.getStruct(resolvedPrefix)) { canonicalType = sinfo->name; } else if (const auto *einfo = gctx.ctx.getEnum(resolvedPrefix)) { @@ -3308,25 +3212,21 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { // Enum-variant constructor (`Result.Ok(x)`-style Call): // the suffix is a variant name on the resolved enum. if (enumForVariant != nullptr) { - int vidx = gctx.ctx.getEnumVariantIndex(canonicalType, - methodName); + int vidx = + gctx.ctx.getEnumVariantIndex(canonicalType, methodName); if (vidx >= 0) { - const auto &variant = - enumForVariant->variants[vidx]; - TypeIdx enumTy = gctx.ctx.getTypePool().intern( - TypeKey{TypeKind::Named, 0, 0, - static_cast( - gctx.ctx.getStringPool().intern( - canonicalType)), - 0}); + const auto &variant = enumForVariant->variants[vidx]; + TypeIdx enumTy = gctx.ctx.getTypePool().intern(TypeKey{ + TypeKind::Named, 0, 0, + static_cast( + gctx.ctx.getStringPool().intern(canonicalType)), + 0}); JirInst tag{}; tag.tag = JirTag::Int; tag.a = static_cast(variant.discriminant); tag.ty = BuiltinType::U8; JirRef tagRef = emit(gctx, tag); - if (!enumForVariant->hasPayloadVariant) { - return tagRef; - } + if (!enumForVariant->hasPayloadVariant) { return tagRef; } // Build {tag, payload-undef|val} via alloca + FieldAddr // stores, then load. Mirrors the TypeMethodCall path. JirInst alloca{}; @@ -3334,8 +3234,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { alloca.ty = enumTy; JirRef slot = emitAllocaHoisted(gctx, alloca); TypeIdx u8PtrTy = gctx.ctx.getTypePool().intern( - TypeKey{TypeKind::PtrSingle, 0, 0, - BuiltinType::U8, 0}); + TypeKey{TypeKind::PtrSingle, 0, 0, BuiltinType::U8, 0}); JirInst tagFA{}; tagFA.tag = JirTag::FieldAddr; tagFA.a = slot; @@ -3355,15 +3254,14 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { // We address them all uniformly via a payload-area // pointer + byte offset GEP through i8. if (argCount > variant.payloadTypes.size()) { - failHere(gctx, - "astgen: too many args for variant `" + - canonicalType + "." + methodName + "`"); + failHere(gctx, "astgen: too many args for variant `" + + canonicalType + "." + methodName + + "`"); } if (!variant.payloadTypes.empty() && argCount >= 1) { TypeIdx payAreaPtrTy = gctx.ctx.getTypePool().intern(TypeKey{ - TypeKind::PtrSingle, 0, 0, - BuiltinType::U8, 0}); + TypeKind::PtrSingle, 0, 0, BuiltinType::U8, 0}); JirInst payAreaFA{}; payAreaFA.tag = JirTag::FieldAddr; payAreaFA.a = slot; @@ -3387,8 +3285,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { off = (off + a - 1) / a * a; NodeIdx argIdxN = static_cast( ns.getExtra(argsExtra + 1 + i)); - JirRef payVal = - astgenExpr(gctx, argIdxN, fieldTy); + JirRef payVal = astgenExpr(gctx, argIdxN, fieldTy); JirInst gepInst{}; gepInst.tag = JirTag::IndexAddr; gepInst.a = payAreaPtr; @@ -3416,10 +3313,8 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { } } if (!canonicalType.empty()) { - std::string qualified = - canonicalType + "." + methodName; - const FunctionAST *method = - gctx.ctx.getFunctionAST(qualified); + std::string qualified = canonicalType + "." + methodName; + const FunctionAST *method = gctx.ctx.getFunctionAST(qualified); if (method != nullptr) { std::vector argRefs; argRefs.reserve(argCount); @@ -3445,9 +3340,8 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { // instantiation's T (e.g. NoDefault) doesn't define // `default`. Naming both type and method gives the // user a precise pointer to the missing piece. - failHere(gctx, "type `" + canonicalType + - "` has no method `" + methodName + - "`"); + failHere(gctx, "type `" + canonicalType + "` has no method `" + + methodName + "`"); } auto it = gctx.locals.find(prefix); @@ -3466,8 +3360,8 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { JirRef recvRef; if (mode == ParamMode::Mut || mode == ParamMode::Move) { TypeIdx pointee = instTy; - TypeIdx ptrTy = gctx.ctx.getTypePool().intern(TypeKey{ - TypeKind::PtrSingle, 0, 0, pointee, 0}); + TypeIdx ptrTy = gctx.ctx.getTypePool().intern( + TypeKey{TypeKind::PtrSingle, 0, 0, pointee, 0}); JirInst ao{}; ao.tag = JirTag::AddrOf; ao.a = it->second; @@ -3486,10 +3380,9 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { for (uint32_t i = 0; i < argCount; i++) { NodeIdx argIdx = static_cast( ns.getExtra(argsExtra + 1 + i)); - TypeIdx expectArg = - (1 + i < method->Args.size()) - ? method->Args[1 + i].Type - : kNoType; + TypeIdx expectArg = (1 + i < method->Args.size()) + ? method->Args[1 + i].Type + : kNoType; argRefs.push_back( astgenExpr(gctx, argIdx, expectArg)); } @@ -3504,10 +3397,10 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { if (fn == nullptr) { // Qualified callees (`lib.priv`) get the precise pub-access // diagnostic via formatNamespaceLookupError. - std::string msg = (callee.find('.') != std::string::npos) - ? gctx.ctx.formatNamespaceLookupError( - "function", callee) - : "unknown function `" + callee + "`"; + std::string msg = + (callee.find('.') != std::string::npos) + ? gctx.ctx.formatNamespaceLookupError("function", callee) + : "unknown function `" + callee + "`"; return recoverHere(gctx, std::move(msg), kNoType); } @@ -3626,9 +3519,8 @@ static JirRef astgenExpr(AstGenCtx &gctx, NodeIdx node, TypeIdx expected) { astgenContinue(gctx); return kNoJirRef; default: - failHere(gctx, - "astgen: unsupported AST node (tag = " + - std::to_string(static_cast(n.tag)) + ")"); + failHere(gctx, "astgen: unsupported AST node (tag = " + + std::to_string(static_cast(n.tag)) + ")"); } if (result != kNoJirRef && line > 0) { gctx.jfn.getInstMut(result).srcLine = static_cast(line); @@ -3695,8 +3587,7 @@ void astgenBodyInto(JirFunction &jfn, const FunctionAST &fn, // pointer-to-pointee rather than a by-value Param. for (size_t i = 0; i < fn.Args.size(); i++) { const Param &p = fn.Args[i]; - jam::abi::ParamABI pabi = - jam::abi::classifyParam(p.Mode, p.Type, ctx); + jam::abi::ParamABI pabi = jam::abi::classifyParam(p.Mode, p.Type, ctx); bool byPtr = pabi.kind == jam::abi::ParamABI::Kind::ByPointer; JirInst paramInst{}; paramInst.tag = JirTag::Param; @@ -3723,9 +3614,7 @@ void astgenBodyInto(JirFunction &jfn, const FunctionAST &fn, } } - for (NodeIdx stmt : fn.Body) { - astgenExpr(gctx, stmt, kNoType); - } + for (NodeIdx stmt : fn.Body) { astgenExpr(gctx, stmt, kNoType); } // Implicit fall-through terminator. Three cases for a tail block // without an explicit terminator: @@ -3748,13 +3637,15 @@ void astgenBodyInto(JirFunction &jfn, const FunctionAST &fn, u.tag = JirTag::Unreachable; emit(gctx, u); } else if (fn.ReturnType == BuiltinType::NoReturn) { - failHere(gctx, - "fn `" + fn.Name + "` is declared `noreturn` but its " - "body falls through without diverging"); + failHere(gctx, "fn `" + fn.Name + + "` is declared `noreturn` but its " + "body falls through without diverging"); } else if (fn.ReturnType != kNoType) { - failHere(gctx, - "fn `" + fn.Name + "` has non-void return type but a " - "path reaches the function end without returning a value"); + failHere( + gctx, + "fn `" + fn.Name + + "` has non-void return type but a " + "path reaches the function end without returning a value"); } else { emitDropsThroughScope(gctx, 0); JirInst ret{}; diff --git a/src/astgen.h b/src/astgen.h index 493bb4b..29bc776 100644 --- a/src/astgen.h +++ b/src/astgen.h @@ -17,14 +17,13 @@ class JamCodegenContext; // diagnostic was already pushed onto `JamCodegenContext::diagnostics()` // before the throw; catch sites (per-decl in main.cpp / generic // instantiation in codegen.cpp) recover by abandoning the current -// decl so siblings still produce diagnostics. Mirrors Zig's -// `error.AnalysisFail` (`src/AstGen.zig:60`). +// decl so siblings still produce diagnostics. class AstGenAnalysisFail {}; // AstGen — recursive, eager lowering of a FunctionAST into a typed -// JirFunction. Mirrors Zig's `AstGen` (lib/std/zig/AstGen.zig) except -// the output is typed from the start (Jam doesn't have comptime-as- -// runtime-values, so the ZIR/AIR split isn't needed). +// JirFunction. The output is typed from the start (Jam has no +// comptime-as-runtime-values, so a separate untyped-then-typed IR +// split isn't needed). // // Each AST node visited: // * pushes zero or more `JirInst`s into the function's instruction @@ -36,7 +35,7 @@ class AstGenAnalysisFail {}; // // Generic instantiation, peer-type resolution, divergence analysis, // and constant folding all happen here. The downstream `jirCodegen` -// is a mechanical AIR-to-LLVM walk. +// is a mechanical JIR-to-LLVM walk. JirFunction astgenFunction(const FunctionAST &fn, JamCodegenContext &ctx); // Populate only a JirFunction's *signature* — name, return type, diff --git a/src/codegen.cpp b/src/codegen.cpp index d25b3d1..6f381e9 100644 --- a/src/codegen.cpp +++ b/src/codegen.cpp @@ -1008,8 +1008,8 @@ TypeIdx JamCodegenContext::instantiateStructExpr( // site here (instantiation is triggered lazily inside // type resolution); the file/line on the frame stays // zero and the formatter skips it. - jam::Diagnostic::Trace traceFrame{ - /*loc=*/{}, /*decl=*/im.clonePtr->Name}; + jam::Diagnostic::Trace traceFrame{/*loc=*/{}, + /*decl=*/im.clonePtr->Name}; jam::RefTraceFrame guard(refTrace_, std::move(traceFrame)); try { astgenBodyInto(im.passOneJir, *im.clonePtr, mutCtx); diff --git a/src/codegen.h b/src/codegen.h index dfc0210..f438410 100644 --- a/src/codegen.h +++ b/src/codegen.h @@ -196,11 +196,8 @@ class JamCodegenContext { // Diagnostic so the user sees a chain like // note: in instantiation of `Vec(NoDefault).default` // note: in instantiation of `Pair(Vec(NoDefault), i32)` - // when an error surfaces deep inside a monomorphisation. Mirrors - // `Module.ErrorMsg.reference_trace` (`Module.zig:2099`). - std::vector &refTrace() const { - return refTrace_; - } + // when an error surfaces deep inside a monomorphisation. + std::vector &refTrace() const { return refTrace_; } // look up a drop method for an instantiated struct // (e.g. Box__i32). Falls back to the pre-built drop registry. // Returns nullptr if the struct has no drop method. @@ -213,6 +210,7 @@ class JamCodegenContext { } return nullptr; } + private: JamContextRef ctx; JamModuleRef mod; diff --git a/src/diagnostics.cpp b/src/diagnostics.cpp index c42f9de..fdec347 100644 --- a/src/diagnostics.cpp +++ b/src/diagnostics.cpp @@ -28,7 +28,7 @@ void Diagnostics::warning(SrcLoc loc, std::string message) { } void Diagnostics::errorWithNotes(SrcLoc loc, std::string message, - std::vector notes) { + std::vector notes) { Diagnostic d; d.loc = std::move(loc); d.severity = Diagnostic::Severity::Error; @@ -38,7 +38,7 @@ void Diagnostics::errorWithNotes(SrcLoc loc, std::string message, } void Diagnostics::errorWithTrace(SrcLoc loc, std::string message, - std::vector trace) { + std::vector trace) { Diagnostic d; d.loc = std::move(loc); d.severity = Diagnostic::Severity::Error; @@ -68,21 +68,22 @@ namespace { const char *severityLabel(Diagnostic::Severity s) { switch (s) { - case Diagnostic::Severity::Error: return "error"; - case Diagnostic::Severity::Warning: return "warning"; - case Diagnostic::Severity::Note: return "note"; + case Diagnostic::Severity::Error: + return "error"; + case Diagnostic::Severity::Warning: + return "warning"; + case Diagnostic::Severity::Note: + return "note"; } return "error"; } // Order diagnostics by source position so the report is stable -// regardless of which pass produced them in which order. Zig itself -// does *not* sort — `Compilation.printAllErrorsToStderr` walks the -// failed_* hashmaps in insertion / hash order — but Jam's tests -// assert on stderr substrings and need a deterministic line ordering -// to stay green, and a user reading multi-error output benefits from -// reading them top-down. Errors come before warnings at the same -// location, both before notes. +// regardless of which pass produced them in which order. The test +// suite asserts on stderr substrings and needs a deterministic line +// ordering to stay green, and a user reading multi-error output +// benefits from top-down reads. Errors come before warnings at the +// same location, both before notes. bool less(const Diagnostic &a, const Diagnostic &b) { if (a.loc.file != b.loc.file) return a.loc.file < b.loc.file; if (a.loc.line != b.loc.line) return a.loc.line < b.loc.line; @@ -109,14 +110,12 @@ void emitOne(std::ostream &out, const Diagnostic &d, int indent = 0) { emitOne(out, nd, indent + 4); } for (const auto &frame : d.referenceTrace) { - out << pad << " note: in instantiation of `" << frame.decl - << "`"; + out << pad << " note: in instantiation of `" << frame.decl << "`"; if (!frame.loc.file.empty() && frame.loc.line > 0) { out << " at " << frame.loc.file << ":" << frame.loc.line; } if (frame.hidden > 0) { - out << " (" << frame.hidden - << " more references hidden)"; + out << " (" << frame.hidden << " more references hidden)"; } out << "\n"; } diff --git a/src/diagnostics.h b/src/diagnostics.h index ccb82e6..584d6c9 100644 --- a/src/diagnostics.h +++ b/src/diagnostics.h @@ -21,10 +21,6 @@ namespace jam { // offset will be tacked on as additional fields when the parser // records token ranges per node — none of the call sites change // when that happens because they only handle SrcLoc by value. -// -// Modeled after Zig's `Module.SrcLoc` (`src/Module.zig:2156`), -// trimmed of the LazySrcLoc indirection that Zig uses to defer -// expensive position computation until error reporting. struct SrcLoc { // Display filename — already-formatted relative path the user // recognises. We keep it as a plain std::string for now; a @@ -36,12 +32,10 @@ struct SrcLoc { // One diagnostic emitted by any phase of the compiler (parser, // astgen, init_analysis, jir_verify, …). -// -// Mirrors Zig's `Module.ErrorMsg` (`src/Module.zig:2095`): // * `notes` carry secondary messages that may live in a different // file/line — used for "X was declared here" / "did you mean Y?". // * `referenceTrace` records the generic-instantiation chain that -// led to a Sema/AstGen failure inside a monomorphisation. +// led to an astgen failure inside a monomorphisation. struct Diagnostic { enum class Severity : uint8_t { Error, Warning, Note }; @@ -50,10 +44,8 @@ struct Diagnostic { std::string decl; // e.g. "Vec(NoDefault).default" // `hidden` counts trace frames that were elided when the // chain exceeded a (future) `--reference-trace=N` limit. - // Mirrors `Module.ErrorMsg.Trace.hidden` in Zig 0.10.1 - // (`src/Module.zig:2104`); kept at 0 today because Jam - // always materialises the full live stack at error time - // — no truncation yet. + // Kept at 0 today: the full live stack is always + // materialised at error time — no truncation yet. uint32_t hidden = 0; }; @@ -72,13 +64,9 @@ struct Diagnostic { // Scope: this is a stack of *currently-active* instantiation frames. // It captures nested instantiation (A→B→C while all three are still // in flight) but does NOT capture references from already-completed -// decls (Zig's `Sema.failWithOwnedErrorMsg` walks a persistent -// `Module.reference_table: ?Decl.Index → Decl.Index` and follows -// referrer chains backwards through finished work). Jam doesn't -// build that table; the stack covers the practical case where -// instantiation errors fire mid-stack, and Zig-style historical -// reference chains are a follow-up if/when a real reference-graph -// pass exists. +// decls. The stack covers the practical case where instantiation +// errors fire mid-stack; cross-decl historical reference chains are +// a follow-up if/when a real reference-graph pass exists. class RefTraceFrame { public: RefTraceFrame(std::vector &stack, Diagnostic::Trace f) @@ -102,9 +90,9 @@ class Diagnostics { void error(SrcLoc loc, std::string message); void warning(SrcLoc loc, std::string message); void errorWithNotes(SrcLoc loc, std::string message, - std::vector notes); + std::vector notes); void errorWithTrace(SrcLoc loc, std::string message, - std::vector referenceTrace); + std::vector referenceTrace); // Push a fully-built diagnostic (used by helpers that already // have notes / trace assembled). @@ -114,8 +102,7 @@ class Diagnostics { std::size_t errorCount() const; const std::vector &all() const { return diags_; } - - // Format the report in Zig-CLI shape: + // Format the report: // file:line: error: message // note: ... // in instantiation of `decl` at file:line diff --git a/src/jam_llvm.cpp b/src/jam_llvm.cpp index cd8805f..a8070e4 100644 --- a/src/jam_llvm.cpp +++ b/src/jam_llvm.cpp @@ -14,9 +14,9 @@ #include "llvm/Bitcode/BitcodeWriter.h" #include "llvm/IR/Constants.h" #include "llvm/IR/DerivedTypes.h" -#include "llvm/IR/Instructions.h" #include "llvm/IR/GlobalVariable.h" #include "llvm/IR/IRBuilder.h" +#include "llvm/IR/Instructions.h" #include "llvm/IR/LLVMContext.h" #include "llvm/IR/LegacyPassManager.h" #include "llvm/IR/Module.h" @@ -94,10 +94,11 @@ void JamLLVMInitializeAllTargets(void) { JamContextRef JamLLVMCreateContext(void) { auto *ctx = new llvm::LLVMContext(); - // Drop SSA value names at construction time. Clang and Zig both emit - // IR with auto-numbered temporaries (`%0`, `%1`, …) instead of the - // source-named values our codegen passes through. Names cost LLVM - // memory (string storage on every Value) and don't affect codegen at + // Drop SSA value names at construction time. Most production LLVM + // frontends emit IR with auto-numbered temporaries (`%0`, `%1`, …) + // instead of the source-named values our codegen passes through. + // Names cost LLVM memory (string storage on every Value) and don't + // affect codegen at // all — they only show up in printed IR. Discarding them here makes // `--emit-ir` output match what production compilers print without // having to plumb empty strings through every CreateAlloca/CreateGEP @@ -507,7 +508,6 @@ JamValueRef JamLLVMGetBasicBlockTerminator(JamBasicBlockRef block) { return WRAP_VALUE(UNWRAP_BLOCK(block)->getTerminator()); } - JamValueRef JamLLVMBuildAlloca(JamBuilderRef builder, JamTypeRef type, uint64_t alignBytes, const char *name) { // Every alloca lives in the function's entry block, not at the @@ -994,8 +994,7 @@ JamLLVMCreateTargetMachine(const char *triple, const char *cpu, // Emit each function/global into its own section so the linker can drop // the unreferenced ones at link time (-Wl,--gc-sections on ELF, // -Wl,-dead_strip on Mach-O). Debug builds skip this — the extra section - // table entries cost compile time we don't want to pay there. Matches - // Zig's function_sections option (zig-0.10.1/src/Compilation.zig:939). + // table entries cost compile time we don't want to pay there. if (optLevel != JAM_OPT_NONE) { opt.FunctionSections = true; opt.DataSections = true; @@ -1016,8 +1015,9 @@ JamLLVMCreateTargetMachine(const char *triple, const char *cpu, case JAM_OPT_AGGRESSIVE: cgOpt = llvm::CodeGenOptLevel::Aggressive; break; - // Codegen-level has no "size" tier — Zig also uses Aggressive for - // ReleaseSmall (zig-0.10.1/src/codegen/llvm.zig opt_level branch). + // Codegen-level has no "size" tier — size optimization happens in + // the IR pipeline and via function attributes, while codegen stays + // at Aggressive for the actual machine emission. case JAM_OPT_SIZE: cgOpt = llvm::CodeGenOptLevel::Aggressive; break; @@ -1081,11 +1081,9 @@ bool JamLLVMEmitObjectFile(JamModuleRef mod, JamTargetMachineRef tm, // LLVM's codegen passes (instruction selection, register allocation), // leaving every IR-level pass — inlining, GVN, mem2reg, SROA, loop opts, // vectorization, MergeFunctions, globaldce — disabled. That made - // `--release` little better than `-O0` for real programs. - // - // Mirrors Zig's release pipeline (see - // misc/references/zig-0.10.1/src/zig_llvm.cpp - // ZigLLVMTargetMachineEmitToFile). + // `--release` little better than `-O0` for real programs. The + // configuration below builds a full new-PM optimization pipeline so + // release builds actually optimize. llvm::PipelineTuningOptions pto; pto.LoopUnrolling = !isDebug; pto.SLPVectorization = !isDebug; diff --git a/src/jam_llvm.h b/src/jam_llvm.h index 24e4f2d..eb030bb 100644 --- a/src/jam_llvm.h +++ b/src/jam_llvm.h @@ -225,17 +225,19 @@ JAM_EXTERN_C void JamLLVMPositionBuilderBeforeTerminator(JamBuilderRef builder, JamBasicBlockRef block); -JAM_EXTERN_C JamValueRef JamLLVMBuildZExt(JamBuilderRef builder, JamValueRef val, - JamTypeRef destType, const char *name); -JAM_EXTERN_C JamValueRef JamLLVMBuildSExt(JamBuilderRef builder, JamValueRef val, - JamTypeRef destType, const char *name); +JAM_EXTERN_C JamValueRef JamLLVMBuildZExt(JamBuilderRef builder, + JamValueRef val, JamTypeRef destType, + const char *name); +JAM_EXTERN_C JamValueRef JamLLVMBuildSExt(JamBuilderRef builder, + JamValueRef val, JamTypeRef destType, + const char *name); // Stack alloca with an explicit alignment in bytes. Pass 0 to fall back to // LLVM's data-layout-derived inference, but prefer passing the type's real // alignment (via JamCodegenContext::typeAlign). LLVM's getPrefTypeAlign -// over-aligns aggregates relative to what C/C++/Zig produce on the same -// target — Zig works around this the same way (see Zig codegen/llvm.zig -// `buildAllocaInner` which always calls `setAlignment` on the result). +// over-aligns aggregates relative to what other LLVM-based compilers +// produce on the same target — explicitly setting the alignment from +// our own type model keeps layout consistent with C/C++ peer code. JAM_EXTERN_C JamValueRef JamLLVMBuildAlloca(JamBuilderRef builder, JamTypeRef type, uint64_t alignBytes, @@ -393,31 +395,26 @@ JAM_EXTERN_C char *JamLLVMGetHostCPUName(void); JAM_EXTERN_C char *JamLLVMGetHostCPUFeatures(void); // Optimization levels selected at TargetMachine creation. Default for jam is -// `None` to match Zig's Debug / rustc's `opt-level=0` — Debug compiles 5-10× -// faster than -O2 because LLVM's machine codegen + the full new-PM module -// pipeline are the dominant cost. The IR-level pipeline (inlining, GVN, SROA, -// vectorization, MergeFunctions, globaldce, …) is driven by these values -// inside JamLLVMEmitObjectFile. +// `None` — debug builds compile 5-10× faster than -O2 because LLVM's machine +// codegen + the full new-PM module pipeline are the dominant cost. The IR- +// level pipeline (inlining, GVN, SROA, vectorization, MergeFunctions, +// globaldce, …) is driven by these values inside JamLLVMEmitObjectFile. // -// Maps to rustc's `-C opt-level=N` values; see compile_codegen_options in the -// rustc source for the same mapping. SIZE/SMALL still run codegen at -// Aggressive (instruction selection / regalloc have no "size" tier) — size -// optimization happens at the IR-pipeline level and via function attributes. +// SIZE/SMALL still run codegen at Aggressive (instruction selection / regalloc +// have no "size" tier) — size optimization happens at the IR-pipeline level +// and via function attributes. typedef enum { - JAM_OPT_NONE = 0, // -O0 (rustc `0`) - JAM_OPT_LESS = 1, // -O1 (rustc `1`) - JAM_OPT_DEFAULT = 2, // -O2 (rustc `2`, LLVM default) - JAM_OPT_AGGRESSIVE = 3, // -O3 (rustc `3`, Zig "ReleaseFast") - JAM_OPT_SIZE = 4, // -Os (rustc `s` — moderate size; optsize attr) - JAM_OPT_SMALL = 5, // -Oz (rustc `z`, Zig "ReleaseSmall"; - // minsize + optsize attrs) + JAM_OPT_NONE = 0, // -O0 + JAM_OPT_LESS = 1, // -O1 + JAM_OPT_DEFAULT = 2, // -O2 (LLVM default) + JAM_OPT_AGGRESSIVE = 3, // -O3 + JAM_OPT_SIZE = 4, // -Os (moderate size; optsize attr) + JAM_OPT_SMALL = 5, // -Oz (minsize + optsize attrs) } JamOptLevel; // Link-time optimization mode. When enabled, jam emits LLVM bitcode (.bc) // instead of an object file, and clang/lld re-runs the optimization pipeline // across the bitcode plus any LTO-compatible static libraries at link time. -// Matches rustc's `-C lto={off,thin,fat}` and Zig's want_lto plumbing -// (zig-0.10.1/src/Compilation.zig:1248-1273). // // Useful even though jam already compiles to a single combined module: // link-time LTO lets the optimizer see across into libc++/libc/static diff --git a/src/jir.h b/src/jir.h index 8fde1a5..b7cdc6c 100644 --- a/src/jir.h +++ b/src/jir.h @@ -19,9 +19,9 @@ class FunctionAST; // JIR (Jam IR) — a single typed flat intermediate representation // produced by AstGen from the parsed AST and consumed by codegen to -// emit LLVM IR. JIR mixes responsibilities that Zig's pipeline splits -// between ZIR (untyped) and AIR (typed); Jam doesn't yet have -// comptime-as-values, so the two-pass split isn't needed. +// emit LLVM IR. The IR is typed from the start: Jam doesn't have +// comptime-as-values, so a separate untyped lowering pass isn't +// needed. // // Pipeline (target): // Source → Tokens → AST → AstGen → JIR → Codegen → LLVM IR @@ -67,38 +67,68 @@ enum class JirTag : uint8_t { // Integer arithmetic // Binary form: `a` = lhs ref, `b` = rhs ref, `ty` = result type. - Add, Sub, Mul, - SDiv, UDiv, SRem, URem, + Add, + Sub, + Mul, + SDiv, + UDiv, + SRem, + URem, // Float arithmetic - FAdd, FSub, FMul, FDiv, FRem, - FNeg, // unary: `a` = operand + FAdd, + FSub, + FMul, + FDiv, + FRem, + FNeg, // unary: `a` = operand // Integer comparison // Binary form; `ty` is always i1. - ICmpEq, ICmpNe, - ICmpSlt, ICmpSle, ICmpSgt, ICmpSge, - ICmpUlt, ICmpUle, ICmpUgt, ICmpUge, + ICmpEq, + ICmpNe, + ICmpSlt, + ICmpSle, + ICmpSgt, + ICmpSge, + ICmpUlt, + ICmpUle, + ICmpUgt, + ICmpUge, // Float comparison // Ordered predicates (NaN inputs ⇒ false) to match Jam's strict // float typing policy. `ty` is always i1. - FCmpOeq, FCmpOne, - FCmpOlt, FCmpOle, FCmpOgt, FCmpOge, + FCmpOeq, + FCmpOne, + FCmpOlt, + FCmpOle, + FCmpOgt, + FCmpOge, // Bitwise / shift - BitAnd, BitOr, BitXor, BitNot, - Shl, AShr, LShr, + BitAnd, + BitOr, + BitXor, + BitNot, + Shl, + AShr, + LShr, // Logical (short-circuit handled by control flow) - LogNot, // unary + LogNot, // unary // Type conversions // All take `a` = operand ref, `ty` = destination type. - ZExt, SExt, Trunc, - SIToFP, UIToFP, - FPToSI, FPToUI, - FPExt, FPTrunc, + ZExt, + SExt, + Trunc, + SIToFP, + UIToFP, + FPToSI, + FPToUI, + FPExt, + FPTrunc, BitCast, // Control flow @@ -108,11 +138,11 @@ enum class JirTag : uint8_t { // [defaultBlock, caseCount, // case0_lo, case0_hi, case0_signed, case0_block, // case1_..., ...] - // Canonical multi-way branch. astgen today lowers match - // to chained CondBr — Switch is the dense-integer path - // jir_codegen is ready for (same shape Rust's - // MIR::TerminatorKind::SwitchInt and Zig's - // air.switch_br use). + // Canonical multi-way branch. astgen emits Switch + // when every arm pattern is a single integer or + // unit-enum-variant; jir_codegen specialises down + // to icmp+cond_br for trivial shapes and otherwise + // emits LLVM `switch`. // Ret: `a` = value ref (or kNoJirRef for void). // Unreachable: no operands. Br, @@ -156,13 +186,13 @@ enum class JirTag : uint8_t { IndexAddr, // Address-of / dereference - AddrOf, // `a` = lvalue ref - Deref, // `a` = ptr ref + AddrOf, // `a` = lvalue ref + Deref, // `a` = ptr ref // Pattern binding payload extraction // For enum-variant pattern bindings, the codegen needs to load // the bound payload field. Encoded explicitly in JIR. - EnumPayload, // `a` = enum value ref, `b` = field index + EnumPayload, // `a` = enum value ref, `b` = field index // Drop // Explicit destructor call for a tracked binding. Emitted by @@ -193,18 +223,18 @@ enum class JirTag : uint8_t { // copyable so dense instruction arrays are cache-friendly. struct JirInst { JirTag tag = JirTag::Invalid; - uint8_t _pad = 0; // padding to keep `flags` 2-byte aligned - uint16_t flags = 0; // tag-specific flags - uint32_t srcLine = 0; // for diagnostics: file:line: + uint8_t _pad = 0; // padding to keep `flags` 2-byte aligned + uint16_t flags = 0; // tag-specific flags + uint32_t srcLine = 0; // for diagnostics: file:line: JirRef a = kNoJirRef; JirRef b = kNoJirRef; - TypeIdx ty = kNoType; // result type (kNoType for void / control) + TypeIdx ty = kNoType; // result type (kNoType for void / control) }; // A basic block within a JirFunction. Holds a list of JirRef // instruction indices into the function's `insts` array. struct JirBlock { - std::string name; // diagnostic label + std::string name; // diagnostic label std::vector insts; }; diff --git a/src/jir_codegen.cpp b/src/jir_codegen.cpp index 3f6a161..53a0f2d 100644 --- a/src/jir_codegen.cpp +++ b/src/jir_codegen.cpp @@ -50,11 +50,10 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r); // `jirDeclarePrototype` so the prototype + caller-side ABI policy // lives in one block. jam::abi::ReturnABI jirClassifyReturn(const JirFunction &jfn, - const JamCodegenContext &ctx); -bool jirReturnIsSret(const JirFunction &jfn, - const JamCodegenContext &ctx); + const JamCodegenContext &ctx); +bool jirReturnIsSret(const JirFunction &jfn, const JamCodegenContext &ctx); jam::abi::ParamABI jirClassifyParam(const JirFunction &jfn, size_t i, - const JamCodegenContext &ctx); + const JamCodegenContext &ctx); // Public entry: look up the cached LLVM value for `r`, or emit it // (recursing for any unrelated subexpressions) and cache the result. @@ -85,9 +84,8 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // practice we should never reach here — the driver checks // `Diagnostics::hasErrors()` after astgen and bails before // running jir_codegen. - JamTypeRef ty = (inst.ty == kNoType) - ? lctx.ctx.getVoidType() - : lctx.ctx.getLLVMType(inst.ty); + JamTypeRef ty = (inst.ty == kNoType) ? lctx.ctx.getVoidType() + : lctx.ctx.getLLVMType(inst.ty); return JamLLVMGetUndef(ty); } case JirTag::Int: { @@ -109,17 +107,16 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { return JamLLVMConstReal(ty, d); } case JirTag::Bool: - return JamLLVMConstInt(lctx.ctx.getInt1Type(), - inst.a != 0 ? 1 : 0, false); + return JamLLVMConstInt(lctx.ctx.getInt1Type(), inst.a != 0 ? 1 : 0, + false); case JirTag::Str: { StringIdx sid = static_cast(inst.a); const std::string &val = lctx.ctx.getStringPool().get(sid); - JamValueRef strConst = JamLLVMConstString( - lctx.ctx.getContext(), val.c_str(), - static_cast(val.length()), true); - JamTypeRef arrTy = - JamLLVMArrayType(lctx.ctx.getInt8Type(), - static_cast(val.length() + 1)); + JamValueRef strConst = + JamLLVMConstString(lctx.ctx.getContext(), val.c_str(), + static_cast(val.length()), true); + JamTypeRef arrTy = JamLLVMArrayType( + lctx.ctx.getInt8Type(), static_cast(val.length() + 1)); JamValueRef strGlobal = JamLLVMAddGlobal(lctx.ctx.getModule(), arrTy, "str"); JamLLVMSetGlobalConstant(strGlobal, true); @@ -127,11 +124,11 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { JamTypeRef sliceTy = lctx.ctx.getLLVMType(inst.ty); JamTypeRef i8PtrTy = JamLLVMPointerType(lctx.ctx.getInt8Type(), 0); - JamValueRef strPtr = JamLLVMBuildBitCast( - lctx.ctx.getBuilder(), strGlobal, i8PtrTy, "str_ptr"); + JamValueRef strPtr = JamLLVMBuildBitCast(lctx.ctx.getBuilder(), + strGlobal, i8PtrTy, "str_ptr"); JamValueRef slice = JamLLVMGetUndef(sliceTy); - slice = JamLLVMBuildInsertValue(lctx.ctx.getBuilder(), slice, strPtr, - 0, "slice_ptr"); + slice = JamLLVMBuildInsertValue(lctx.ctx.getBuilder(), slice, strPtr, 0, + "slice_ptr"); slice = JamLLVMBuildInsertValue( lctx.ctx.getBuilder(), slice, JamLLVMConstInt(lctx.ctx.getInt64Type(), @@ -147,8 +144,8 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // When the function uses sret, the source-level param at JIR // index `i` lives at LLVM index `i + 1` (the sret slot owns // LLVM index 0). - JamFunctionRef f = JamLLVMGetFunction( - lctx.ctx.getModule(), lctx.jfn.name.c_str()); + JamFunctionRef f = + JamLLVMGetFunction(lctx.ctx.getModule(), lctx.jfn.name.c_str()); unsigned argOffset = jirReturnIsSret(lctx.jfn, lctx.ctx) ? 1u : 0u; return JamLLVMGetParam(f, inst.a + argOffset); } @@ -156,8 +153,7 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { JamTypeRef ty = lctx.ctx.getLLVMType(inst.ty); uint64_t align = lctx.ctx.typeAlign(inst.ty); std::string nm = "v" + std::to_string(r); - return JamLLVMBuildAlloca(lctx.ctx.getBuilder(), ty, align, - nm.c_str()); + return JamLLVMBuildAlloca(lctx.ctx.getBuilder(), ty, align, nm.c_str()); } case JirTag::Load: { JamValueRef ptr = emitInst(lctx, inst.a); @@ -238,106 +234,96 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { } // === Integer comparison === case JirTag::ICmpEq: - return buildICmp(lctx, JAM_ICMP_EQ, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "eq"); + return buildICmp(lctx, JAM_ICMP_EQ, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "eq"); case JirTag::ICmpNe: - return buildICmp(lctx, JAM_ICMP_NE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "ne"); + return buildICmp(lctx, JAM_ICMP_NE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "ne"); case JirTag::ICmpSlt: - return buildICmp(lctx, JAM_ICMP_SLT, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "slt"); + return buildICmp(lctx, JAM_ICMP_SLT, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "slt"); case JirTag::ICmpSle: - return buildICmp(lctx, JAM_ICMP_SLE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "sle"); + return buildICmp(lctx, JAM_ICMP_SLE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "sle"); case JirTag::ICmpSgt: - return buildICmp(lctx, JAM_ICMP_SGT, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "sgt"); + return buildICmp(lctx, JAM_ICMP_SGT, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "sgt"); case JirTag::ICmpSge: - return buildICmp(lctx, JAM_ICMP_SGE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "sge"); + return buildICmp(lctx, JAM_ICMP_SGE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "sge"); case JirTag::ICmpUlt: - return buildICmp(lctx, JAM_ICMP_ULT, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "ult"); + return buildICmp(lctx, JAM_ICMP_ULT, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "ult"); case JirTag::ICmpUle: - return buildICmp(lctx, JAM_ICMP_ULE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "ule"); + return buildICmp(lctx, JAM_ICMP_ULE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "ule"); case JirTag::ICmpUgt: - return buildICmp(lctx, JAM_ICMP_UGT, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "ugt"); + return buildICmp(lctx, JAM_ICMP_UGT, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "ugt"); case JirTag::ICmpUge: - return buildICmp(lctx, JAM_ICMP_UGE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "uge"); + return buildICmp(lctx, JAM_ICMP_UGE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "uge"); // === Float comparison === case JirTag::FCmpOeq: - return buildFCmp(lctx, JAM_FCMP_OEQ, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "oeq"); + return buildFCmp(lctx, JAM_FCMP_OEQ, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "oeq"); case JirTag::FCmpOne: - return buildFCmp(lctx, JAM_FCMP_UNE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "one"); + return buildFCmp(lctx, JAM_FCMP_UNE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "one"); case JirTag::FCmpOlt: - return buildFCmp(lctx, JAM_FCMP_OLT, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "olt"); + return buildFCmp(lctx, JAM_FCMP_OLT, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "olt"); case JirTag::FCmpOle: - return buildFCmp(lctx, JAM_FCMP_OLE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "ole"); + return buildFCmp(lctx, JAM_FCMP_OLE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "ole"); case JirTag::FCmpOgt: - return buildFCmp(lctx, JAM_FCMP_OGT, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "ogt"); + return buildFCmp(lctx, JAM_FCMP_OGT, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "ogt"); case JirTag::FCmpOge: - return buildFCmp(lctx, JAM_FCMP_OGE, - emitInst(lctx, inst.a), emitInst(lctx, inst.b), "oge"); + return buildFCmp(lctx, JAM_FCMP_OGE, emitInst(lctx, inst.a), + emitInst(lctx, inst.b), "oge"); // === Bitwise / shift === case JirTag::BitAnd: - return JamLLVMBuildAnd(lctx.ctx.getBuilder(), - emitInst(lctx, inst.a), + return JamLLVMBuildAnd(lctx.ctx.getBuilder(), emitInst(lctx, inst.a), emitInst(lctx, inst.b), "and"); case JirTag::BitOr: - return JamLLVMBuildOr(lctx.ctx.getBuilder(), - emitInst(lctx, inst.a), + return JamLLVMBuildOr(lctx.ctx.getBuilder(), emitInst(lctx, inst.a), emitInst(lctx, inst.b), "or"); case JirTag::BitXor: - return JamLLVMBuildXor(lctx.ctx.getBuilder(), - emitInst(lctx, inst.a), + return JamLLVMBuildXor(lctx.ctx.getBuilder(), emitInst(lctx, inst.a), emitInst(lctx, inst.b), "xor"); case JirTag::Shl: - return JamLLVMBuildShl(lctx.ctx.getBuilder(), - emitInst(lctx, inst.a), + return JamLLVMBuildShl(lctx.ctx.getBuilder(), emitInst(lctx, inst.a), emitInst(lctx, inst.b), "shl"); case JirTag::AShr: - return JamLLVMBuildAShr(lctx.ctx.getBuilder(), - emitInst(lctx, inst.a), + return JamLLVMBuildAShr(lctx.ctx.getBuilder(), emitInst(lctx, inst.a), emitInst(lctx, inst.b), "ashr"); case JirTag::LShr: - return JamLLVMBuildLShr(lctx.ctx.getBuilder(), - emitInst(lctx, inst.a), + return JamLLVMBuildLShr(lctx.ctx.getBuilder(), emitInst(lctx, inst.a), emitInst(lctx, inst.b), "lshr"); case JirTag::BitNot: { // LLVM has no NOT op; emit XOR with all-ones of the operand's type. JamValueRef v = emitInst(lctx, inst.a); JamTypeRef ty = lctx.ctx.getLLVMType(inst.ty); - JamValueRef ones = JamLLVMConstInt(ty, ~static_cast(0), - true); + JamValueRef ones = JamLLVMConstInt(ty, ~static_cast(0), true); return JamLLVMBuildXor(lctx.ctx.getBuilder(), v, ones, "not"); } case JirTag::LogNot: { // Boolean inversion: XOR with i1 1. Operand is already i1. JamValueRef v = emitInst(lctx, inst.a); - JamValueRef one = - JamLLVMConstInt(lctx.ctx.getInt1Type(), 1, false); + JamValueRef one = JamLLVMConstInt(lctx.ctx.getInt1Type(), 1, false); return JamLLVMBuildXor(lctx.ctx.getBuilder(), v, one, "lnot"); } // === Type conversions === case JirTag::ZExt: { JamValueRef v = emitInst(lctx, inst.a); JamTypeRef ty = lctx.ctx.getLLVMType(inst.ty); - return JamLLVMBuildIntCast(lctx.ctx.getBuilder(), v, ty, false, - "zext"); + return JamLLVMBuildIntCast(lctx.ctx.getBuilder(), v, ty, false, "zext"); } case JirTag::SExt: { JamValueRef v = emitInst(lctx, inst.a); JamTypeRef ty = lctx.ctx.getLLVMType(inst.ty); - return JamLLVMBuildIntCast(lctx.ctx.getBuilder(), v, ty, true, - "sext"); + return JamLLVMBuildIntCast(lctx.ctx.getBuilder(), v, ty, true, "sext"); } case JirTag::Trunc: { JamValueRef v = emitInst(lctx, inst.a); @@ -383,11 +369,10 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { uint32_t count = lctx.jfn.getExtra(extra); JamValueRef agg = JamLLVMGetUndef(ty); for (uint32_t i = 0; i < count; i++) { - JirRef fr = - static_cast(lctx.jfn.getExtra(extra + 1 + i)); + JirRef fr = static_cast(lctx.jfn.getExtra(extra + 1 + i)); JamValueRef fv = emitInst(lctx, fr); - agg = JamLLVMBuildInsertValue(lctx.ctx.getBuilder(), agg, fv, - i, "field"); + agg = JamLLVMBuildInsertValue(lctx.ctx.getBuilder(), agg, fv, i, + "field"); } return agg; } @@ -403,11 +388,10 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { uint32_t count = lctx.jfn.getExtra(extra); JamValueRef agg = JamLLVMGetUndef(ty); for (uint32_t i = 0; i < count; i++) { - JirRef er = - static_cast(lctx.jfn.getExtra(extra + 1 + i)); + JirRef er = static_cast(lctx.jfn.getExtra(extra + 1 + i)); JamValueRef ev = emitInst(lctx, er); - agg = JamLLVMBuildInsertValue(lctx.ctx.getBuilder(), agg, ev, - i, "elem"); + agg = JamLLVMBuildInsertValue(lctx.ctx.getBuilder(), agg, ev, i, + "elem"); } return agg; } @@ -423,8 +407,8 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // Coerce index to i64 (the GEP wrappers expect a sized integer). JamTypeRef idxLLVMTy = JamLLVMTypeOf(idxVal); if (idxLLVMTy != i64) { - idxVal = JamLLVMBuildIntCast(lctx.ctx.getBuilder(), idxVal, - i64, false, "idx.cast"); + idxVal = JamLLVMBuildIntCast(lctx.ctx.getBuilder(), idxVal, i64, + false, "idx.cast"); } JamTypeRef baseLLVMTy = lctx.ctx.getLLVMType(baseInst.ty); @@ -433,8 +417,8 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { if (basek.kind == TypeKind::Slice) { // SSA slice value: extract the pointer field (0), GEP by elem. JamValueRef base = emitInst(lctx, inst.a); - JamValueRef ptr = JamLLVMBuildExtractValue( - lctx.ctx.getBuilder(), base, 0, "slice.ptr"); + JamValueRef ptr = JamLLVMBuildExtractValue(lctx.ctx.getBuilder(), + base, 0, "slice.ptr"); JamValueRef gep = JamLLVMBuildPtrGEP( lctx.ctx.getBuilder(), elemLLVMTy, ptr, idxVal, "idx.gep"); return JamLLVMBuildLoad(lctx.ctx.getBuilder(), elemLLVMTy, gep, @@ -450,13 +434,12 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // Array case: spill the SSA value to a temp alloca and GEP. JamValueRef base = emitInst(lctx, inst.a); uint64_t align = lctx.ctx.typeAlign(baseInst.ty); - JamValueRef tmp = JamLLVMBuildAlloca(lctx.ctx.getBuilder(), - baseLLVMTy, align, "arr.tmp"); + JamValueRef tmp = JamLLVMBuildAlloca(lctx.ctx.getBuilder(), baseLLVMTy, + align, "arr.tmp"); JamLLVMBuildStore(lctx.ctx.getBuilder(), base, tmp); JamValueRef gep = JamLLVMBuildArrayGEP( lctx.ctx.getBuilder(), baseLLVMTy, tmp, idxVal, "idx.gep"); - return JamLLVMBuildLoad(lctx.ctx.getBuilder(), elemLLVMTy, gep, - "idx"); + return JamLLVMBuildLoad(lctx.ctx.getBuilder(), elemLLVMTy, gep, "idx"); } case JirTag::AddrOf: { // AstGen plants the alloca's own JirRef into `a` when taking @@ -479,18 +462,16 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { if (baseIsAllocaLike) { pointeeTy = baseInst.ty; } else { - const TypeKey &k = - lctx.ctx.getTypePool().get(baseInst.ty); - if (k.kind != TypeKind::PtrSingle && - k.kind != TypeKind::PtrMany) { + const TypeKey &k = lctx.ctx.getTypePool().get(baseInst.ty); + if (k.kind != TypeKind::PtrSingle && k.kind != TypeKind::PtrMany) { throw std::runtime_error( "jirCodegen: FieldAddr base must be a pointer"); } pointeeTy = static_cast(k.a); } JamTypeRef structTy = lctx.ctx.getLLVMType(pointeeTy); - return JamLLVMBuildStructGEP(lctx.ctx.getBuilder(), structTy, - basePtr, inst.b, "fieldp"); + return JamLLVMBuildStructGEP(lctx.ctx.getBuilder(), structTy, basePtr, + inst.b, "fieldp"); } case JirTag::IndexAddr: { JamValueRef basePtr = emitInst(lctx, inst.a); @@ -502,32 +483,31 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { if (baseIsAllocaLike) { pointeeTy = baseInst.ty; } else { - const TypeKey &k = - lctx.ctx.getTypePool().get(baseInst.ty); - pointeeTy = (k.kind == TypeKind::PtrSingle || - k.kind == TypeKind::PtrMany) - ? static_cast(k.a) - : baseInst.ty; + const TypeKey &k = lctx.ctx.getTypePool().get(baseInst.ty); + pointeeTy = + (k.kind == TypeKind::PtrSingle || k.kind == TypeKind::PtrMany) + ? static_cast(k.a) + : baseInst.ty; } const TypeKey &pk = lctx.ctx.getTypePool().get(pointeeTy); JamValueRef idxVal = emitInst(lctx, inst.b); JamTypeRef i64 = lctx.ctx.getInt64Type(); if (JamLLVMTypeOf(idxVal) != i64) { - idxVal = JamLLVMBuildIntCast(lctx.ctx.getBuilder(), idxVal, - i64, false, "idx.cast"); + idxVal = JamLLVMBuildIntCast(lctx.ctx.getBuilder(), idxVal, i64, + false, "idx.cast"); } const TypeKey &resKey = lctx.ctx.getTypePool().get(inst.ty); TypeIdx elemTy = static_cast(resKey.a); JamTypeRef elemLLVM = lctx.ctx.getLLVMType(elemTy); if (pk.kind == TypeKind::Array) { JamTypeRef arrLLVM = lctx.ctx.getLLVMType(pointeeTy); - return JamLLVMBuildArrayGEP(lctx.ctx.getBuilder(), arrLLVM, - basePtr, idxVal, "idx.addr"); + return JamLLVMBuildArrayGEP(lctx.ctx.getBuilder(), arrLLVM, basePtr, + idxVal, "idx.addr"); } // Slice / PtrMany / PtrSingle to a single elem: treat as a // many-item pointer and PtrGEP by element-sized stride. - return JamLLVMBuildPtrGEP(lctx.ctx.getBuilder(), elemLLVM, - basePtr, idxVal, "idx.addr"); + return JamLLVMBuildPtrGEP(lctx.ctx.getBuilder(), elemLLVM, basePtr, + idxVal, "idx.addr"); } case JirTag::Deref: { JamValueRef ptr = emitInst(lctx, inst.a); @@ -540,13 +520,12 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // Mechanically emit `call void (ptr )`. JamValueRef bindingPtr = emitInst(lctx, inst.a); StringIdx symId = static_cast(inst.b); - const std::string &symbol = - lctx.ctx.getStringPool().get(symId); + const std::string &symbol = lctx.ctx.getStringPool().get(symId); JamFunctionRef f = JamLLVMGetFunction(lctx.ctx.getModule(), symbol.c_str()); if (!f) { - throw std::runtime_error("jirCodegen: drop callee `" + - symbol + "` not declared"); + throw std::runtime_error("jirCodegen: drop callee `" + symbol + + "` not declared"); } JamValueRef args[1] = {bindingPtr}; JamLLVMBuildCall(lctx.ctx.getBuilder(), f, args, 1, ""); @@ -565,12 +544,10 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { JamValueRef fieldPtr = payloadAreaPtr; if (inst.b != 0) { JamValueRef off = JamLLVMConstInt( - lctx.ctx.getInt64Type(), - static_cast(inst.b), false); + lctx.ctx.getInt64Type(), static_cast(inst.b), false); fieldPtr = JamLLVMBuildPtrGEP( - lctx.ctx.getBuilder(), - lctx.ctx.getInt8Type(), payloadAreaPtr, off, - "enum.payload.off"); + lctx.ctx.getBuilder(), lctx.ctx.getInt8Type(), payloadAreaPtr, + off, "enum.payload.off"); } JamTypeRef fieldTy = lctx.ctx.getLLVMType(inst.ty); return JamLLVMBuildLoad(lctx.ctx.getBuilder(), fieldTy, fieldPtr, @@ -584,13 +561,12 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // instantiated cloned methods keep their qualified // `Vec__i32.push` form). Codegen does a single lookup. StringIdx calleeId = static_cast(inst.a); - const std::string &symbol = - lctx.ctx.getStringPool().get(calleeId); + const std::string &symbol = lctx.ctx.getStringPool().get(calleeId); JamFunctionRef f = JamLLVMGetFunction(lctx.ctx.getModule(), symbol.c_str()); if (!f) { - throw std::runtime_error("jirCodegen: unknown callee `" + - symbol + "`"); + throw std::runtime_error("jirCodegen: unknown callee `" + symbol + + "`"); } JirExtraIdx extra = static_cast(inst.b); uint32_t argCount = lctx.jfn.getExtra(extra); @@ -605,13 +581,12 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // the pointee type. Result is loaded back after the call. JamTypeRef pointee = JamLLVMFunctionSretPointeeType(f); uint64_t align = lctx.ctx.typeAlign(inst.ty); - sretSlot = JamLLVMBuildAlloca(lctx.ctx.getBuilder(), pointee, - align, "sret.slot"); + sretSlot = JamLLVMBuildAlloca(lctx.ctx.getBuilder(), pointee, align, + "sret.slot"); args.push_back(sretSlot); } for (uint32_t i = 0; i < argCount; i++) { - JirRef ar = - static_cast(lctx.jfn.getExtra(extra + 1 + i)); + JirRef ar = static_cast(lctx.jfn.getExtra(extra + 1 + i)); args.push_back(emitInst(lctx, ar)); } if (calleeUsesSret) { @@ -619,15 +594,14 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // the sret slot. Load it so the JirRef → LLVM value map // holds the materialized return value. JamLLVMBuildCall(lctx.ctx.getBuilder(), f, args.data(), - static_cast(args.size()), ""); + static_cast(args.size()), ""); JamTypeRef retLlvmTy = lctx.ctx.getLLVMType(inst.ty); - return JamLLVMBuildLoad(lctx.ctx.getBuilder(), retLlvmTy, - sretSlot, "sret.val"); + return JamLLVMBuildLoad(lctx.ctx.getBuilder(), retLlvmTy, sretSlot, + "sret.val"); } const char *resultName = (inst.ty == kNoType) ? "" : "call"; return JamLLVMBuildCall(lctx.ctx.getBuilder(), f, args.data(), - static_cast(args.size()), - resultName); + static_cast(args.size()), resultName); } // === Control === case JirTag::Br: { @@ -638,13 +612,11 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { case JirTag::CondBr: { JamValueRef cond = emitInst(lctx, inst.a); JirExtraIdx extra = static_cast(inst.b); - JirBlockRef thenB = - static_cast(lctx.jfn.getExtra(extra)); + JirBlockRef thenB = static_cast(lctx.jfn.getExtra(extra)); JirBlockRef elseB = static_cast(lctx.jfn.getExtra(extra + 1)); - JamLLVMBuildCondBr(lctx.ctx.getBuilder(), cond, - lctx.blockMap.at(thenB), - lctx.blockMap.at(elseB)); + JamLLVMBuildCondBr(lctx.ctx.getBuilder(), cond, lctx.blockMap.at(thenB), + lctx.blockMap.at(elseB)); return nullptr; } case JirTag::Switch: { @@ -681,24 +653,23 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { JirBlockRef caseB = static_cast(lctx.jfn.getExtra(base + 3)); uint64_t bits = lo | (hi << 32); - JamValueRef caseVal = - JamLLVMConstInt(scrutLlvmTy, bits, isSigned); + JamValueRef caseVal = JamLLVMConstInt(scrutLlvmTy, bits, isSigned); return std::tuple{caseVal, caseB, bits}; }; if (caseCount == 1) { auto [caseVal, caseB, _bits] = readCase(0); - JamValueRef cmp = buildICmp(lctx, JAM_ICMP_EQ, scrut, caseVal, - "match.eq"); + JamValueRef cmp = + buildICmp(lctx, JAM_ICMP_EQ, scrut, caseVal, "match.eq"); JamLLVMBuildCondBr(lctx.ctx.getBuilder(), cmp, - lctx.blockMap.at(caseB), - lctx.blockMap.at(defaultB)); + lctx.blockMap.at(caseB), + lctx.blockMap.at(defaultB)); return nullptr; } - JamValueRef sw = JamLLVMBuildSwitch( - lctx.ctx.getBuilder(), scrut, lctx.blockMap.at(defaultB), - caseCount); + JamValueRef sw = + JamLLVMBuildSwitch(lctx.ctx.getBuilder(), scrut, + lctx.blockMap.at(defaultB), caseCount); for (uint32_t i = 0; i < caseCount; i++) { auto [caseVal, caseB, _bits] = readCase(i); JamLLVMAddCase(sw, caseVal, lctx.blockMap.at(caseB)); @@ -710,8 +681,8 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // `ptr sret(%T)` arg and emit `ret void`. The caller already // owns the slot, so we don't allocate anything here. if (jirReturnIsSret(lctx.jfn, lctx.ctx)) { - JamFunctionRef f = JamLLVMGetFunction( - lctx.ctx.getModule(), lctx.jfn.name.c_str()); + JamFunctionRef f = + JamLLVMGetFunction(lctx.ctx.getModule(), lctx.jfn.name.c_str()); JamValueRef sretSlot = JamLLVMGetParam(f, 0); if (inst.a != kNoJirRef) { JamValueRef v = emitInst(lctx, inst.a); @@ -732,38 +703,35 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { JamLLVMBuildUnreachable(lctx.ctx.getBuilder()); return nullptr; default: - throw std::runtime_error( - "jirCodegen: unsupported JIR tag (tag = " + - std::to_string(static_cast(inst.tag)) + ")"); + throw std::runtime_error("jirCodegen: unsupported JIR tag (tag = " + + std::to_string(static_cast(inst.tag)) + + ")"); } } // Single ABI source-of-truth used by both prototype emission and // call/body lowering. extern and test functions skip the classifier // for returns (extern follows the C ABI literally, tests are always -// nullary-void). Mirrors Zig's `iterateParamTypes` + `firstParamSRet` -// pattern: classification is a pure function of (mode, type, fn-kind), -// so call sites and definitions never disagree. +// nullary-void). Classification is a pure function of +// (mode, type, fn-kind), so call sites and definitions never disagree. jam::abi::ReturnABI jirClassifyReturn(const JirFunction &jfn, - const JamCodegenContext &ctx) { + const JamCodegenContext &ctx) { if (jfn.isExtern || jfn.isTest) { - return jam::abi::ReturnABI{ - jam::abi::ReturnABI::Kind::Direct, - (jfn.isTest || jfn.returnType == kNoType) - ? ctx.getVoidType() - : ctx.getLLVMType(jfn.returnType), - 0}; + return jam::abi::ReturnABI{jam::abi::ReturnABI::Kind::Direct, + (jfn.isTest || jfn.returnType == kNoType) + ? ctx.getVoidType() + : ctx.getLLVMType(jfn.returnType), + 0}; } if (jfn.returnType == kNoType) { return jam::abi::ReturnABI{jam::abi::ReturnABI::Kind::Direct, - ctx.getVoidType(), 0}; + ctx.getVoidType(), 0}; } return jam::abi::classifyReturn(jfn.returnType, ctx); } // Does the LLVM signature have a leading `ptr sret(%T)` argument? -bool jirReturnIsSret(const JirFunction &jfn, - const JamCodegenContext &ctx) { +bool jirReturnIsSret(const JirFunction &jfn, const JamCodegenContext &ctx) { return jirClassifyReturn(jfn, ctx).kind == jam::abi::ReturnABI::Kind::Indirect; } @@ -772,14 +740,14 @@ bool jirReturnIsSret(const JirFunction &jfn, // preserves the user-written type verbatim (the user already wrote // what they want at the FFI boundary, e.g. `*const T` for an out-ptr). jam::abi::ParamABI jirClassifyParam(const JirFunction &jfn, size_t i, - const JamCodegenContext &ctx) { + const JamCodegenContext &ctx) { TypeIdx t = jfn.paramTypes[i]; if (jfn.isExtern) { return jam::abi::ParamABI{jam::abi::ParamABI::Kind::ByValue, - ctx.getLLVMType(t), 0}; + ctx.getLLVMType(t), 0}; } - ParamMode mode = i < jfn.paramModes.size() ? jfn.paramModes[i] - : ParamMode::Let; + ParamMode mode = + i < jfn.paramModes.size() ? jfn.paramModes[i] : ParamMode::Let; return jam::abi::classifyParam(mode, t, ctx); } @@ -800,17 +768,17 @@ void jirDeclarePrototype(const JirFunction &jfn, JamCodegenContext &ctx) { for (size_t i = 0; i < jfn.paramTypes.size(); i++) { jam::abi::ParamABI pabi = jirClassifyParam(jfn, i, ctx); if (pabi.kind == jam::abi::ParamABI::Kind::ByPointer) { - argTypes.push_back(JamLLVMPointerType( - ctx.getLLVMType(jfn.paramTypes[i]), 0)); + argTypes.push_back( + JamLLVMPointerType(ctx.getLLVMType(jfn.paramTypes[i]), 0)); } else { argTypes.push_back(pabi.llvmType); } } JamTypeRef retType = sret ? ctx.getVoidType() : rabi.directType; - JamTypeRef ft = JamLLVMFunctionType( - retType, argTypes.data(), static_cast(argTypes.size()), - jfn.isVarArgs); + JamTypeRef ft = JamLLVMFunctionType(retType, argTypes.data(), + static_cast(argTypes.size()), + jfn.isVarArgs); JamFunctionRef f = JamLLVMAddFunction(ctx.getModule(), jfn.name.c_str(), ft); JamLLVMApplyDefaultFnAttrs(f, jfn.isExtern); @@ -819,11 +787,10 @@ void jirDeclarePrototype(const JirFunction &jfn, JamCodegenContext &ctx) { } if (sret) { JamLLVMAddParamAttrSret(f, 0, ctx.getLLVMType(jfn.returnType), - rabi.sretAlign); + rabi.sretAlign); } - bool externalLinkage = - jfn.isExtern || jfn.isExport || jfn.name == "main"; + bool externalLinkage = jfn.isExtern || jfn.isExport || jfn.name == "main"; if (externalLinkage) { JamLLVMSetLinkage(reinterpret_cast(f), JAM_LINKAGE_EXTERNAL); @@ -834,8 +801,8 @@ void jirDeclarePrototype(const JirFunction &jfn, JamCodegenContext &ctx) { const unsigned argOffset = sret ? 1u : 0u; for (size_t i = 0; i < jfn.paramTypes.size(); i++) { if (jfn.paramTypes[i] == BuiltinType::Bool) { - JamLLVMAddParamAttrZeroExt( - f, static_cast(i) + argOffset); + JamLLVMAddParamAttrZeroExt(f, static_cast(i) + + argOffset); } } if (!sret && jfn.returnType == BuiltinType::Bool) { diff --git a/src/jir_verify.cpp b/src/jir_verify.cpp index 510d68f..4717a0d 100644 --- a/src/jir_verify.cpp +++ b/src/jir_verify.cpp @@ -15,87 +15,160 @@ namespace { bool isTerminator(JirTag t) { - return t == JirTag::Br || t == JirTag::CondBr || - t == JirTag::Switch || t == JirTag::Ret || - t == JirTag::Unreachable; + return t == JirTag::Br || t == JirTag::CondBr || t == JirTag::Switch || + t == JirTag::Ret || t == JirTag::Unreachable; } const char *tagName(JirTag t) { switch (t) { - case JirTag::Invalid: return "Invalid"; - case JirTag::Int: return "Int"; - case JirTag::Float: return "Float"; - case JirTag::Bool: return "Bool"; - case JirTag::Str: return "Str"; - case JirTag::Poison: return "Poison"; - case JirTag::Alloca: return "Alloca"; - case JirTag::Load: return "Load"; - case JirTag::Store: return "Store"; - case JirTag::Add: return "Add"; - case JirTag::Sub: return "Sub"; - case JirTag::Mul: return "Mul"; - case JirTag::SDiv: return "SDiv"; - case JirTag::UDiv: return "UDiv"; - case JirTag::SRem: return "SRem"; - case JirTag::URem: return "URem"; - case JirTag::FAdd: return "FAdd"; - case JirTag::FSub: return "FSub"; - case JirTag::FMul: return "FMul"; - case JirTag::FDiv: return "FDiv"; - case JirTag::FRem: return "FRem"; - case JirTag::FNeg: return "FNeg"; - case JirTag::ICmpEq: return "ICmpEq"; - case JirTag::ICmpNe: return "ICmpNe"; - case JirTag::ICmpSlt: return "ICmpSlt"; - case JirTag::ICmpSle: return "ICmpSle"; - case JirTag::ICmpSgt: return "ICmpSgt"; - case JirTag::ICmpSge: return "ICmpSge"; - case JirTag::ICmpUlt: return "ICmpUlt"; - case JirTag::ICmpUle: return "ICmpUle"; - case JirTag::ICmpUgt: return "ICmpUgt"; - case JirTag::ICmpUge: return "ICmpUge"; - case JirTag::FCmpOeq: return "FCmpOeq"; - case JirTag::FCmpOne: return "FCmpOne"; - case JirTag::FCmpOlt: return "FCmpOlt"; - case JirTag::FCmpOle: return "FCmpOle"; - case JirTag::FCmpOgt: return "FCmpOgt"; - case JirTag::FCmpOge: return "FCmpOge"; - case JirTag::BitAnd: return "BitAnd"; - case JirTag::BitOr: return "BitOr"; - case JirTag::BitXor: return "BitXor"; - case JirTag::BitNot: return "BitNot"; - case JirTag::Shl: return "Shl"; - case JirTag::AShr: return "AShr"; - case JirTag::LShr: return "LShr"; - case JirTag::LogNot: return "LogNot"; - case JirTag::ZExt: return "ZExt"; - case JirTag::SExt: return "SExt"; - case JirTag::Trunc: return "Trunc"; - case JirTag::SIToFP: return "SIToFP"; - case JirTag::UIToFP: return "UIToFP"; - case JirTag::FPToSI: return "FPToSI"; - case JirTag::FPToUI: return "FPToUI"; - case JirTag::FPExt: return "FPExt"; - case JirTag::FPTrunc: return "FPTrunc"; - case JirTag::BitCast: return "BitCast"; - case JirTag::Br: return "Br"; - case JirTag::CondBr: return "CondBr"; - case JirTag::Switch: return "Switch"; - case JirTag::Ret: return "Ret"; - case JirTag::Unreachable: return "Unreachable"; - case JirTag::Call: return "Call"; - case JirTag::Param: return "Param"; - case JirTag::StructLit: return "StructLit"; - case JirTag::FieldAccess: return "FieldAccess"; - case JirTag::ExtractValue: return "ExtractValue"; - case JirTag::ArrayLit: return "ArrayLit"; - case JirTag::Index: return "Index"; - case JirTag::FieldAddr: return "FieldAddr"; - case JirTag::IndexAddr: return "IndexAddr"; - case JirTag::AddrOf: return "AddrOf"; - case JirTag::Deref: return "Deref"; - case JirTag::EnumPayload: return "EnumPayload"; - case JirTag::DropBinding: return "DropBinding"; + case JirTag::Invalid: + return "Invalid"; + case JirTag::Int: + return "Int"; + case JirTag::Float: + return "Float"; + case JirTag::Bool: + return "Bool"; + case JirTag::Str: + return "Str"; + case JirTag::Poison: + return "Poison"; + case JirTag::Alloca: + return "Alloca"; + case JirTag::Load: + return "Load"; + case JirTag::Store: + return "Store"; + case JirTag::Add: + return "Add"; + case JirTag::Sub: + return "Sub"; + case JirTag::Mul: + return "Mul"; + case JirTag::SDiv: + return "SDiv"; + case JirTag::UDiv: + return "UDiv"; + case JirTag::SRem: + return "SRem"; + case JirTag::URem: + return "URem"; + case JirTag::FAdd: + return "FAdd"; + case JirTag::FSub: + return "FSub"; + case JirTag::FMul: + return "FMul"; + case JirTag::FDiv: + return "FDiv"; + case JirTag::FRem: + return "FRem"; + case JirTag::FNeg: + return "FNeg"; + case JirTag::ICmpEq: + return "ICmpEq"; + case JirTag::ICmpNe: + return "ICmpNe"; + case JirTag::ICmpSlt: + return "ICmpSlt"; + case JirTag::ICmpSle: + return "ICmpSle"; + case JirTag::ICmpSgt: + return "ICmpSgt"; + case JirTag::ICmpSge: + return "ICmpSge"; + case JirTag::ICmpUlt: + return "ICmpUlt"; + case JirTag::ICmpUle: + return "ICmpUle"; + case JirTag::ICmpUgt: + return "ICmpUgt"; + case JirTag::ICmpUge: + return "ICmpUge"; + case JirTag::FCmpOeq: + return "FCmpOeq"; + case JirTag::FCmpOne: + return "FCmpOne"; + case JirTag::FCmpOlt: + return "FCmpOlt"; + case JirTag::FCmpOle: + return "FCmpOle"; + case JirTag::FCmpOgt: + return "FCmpOgt"; + case JirTag::FCmpOge: + return "FCmpOge"; + case JirTag::BitAnd: + return "BitAnd"; + case JirTag::BitOr: + return "BitOr"; + case JirTag::BitXor: + return "BitXor"; + case JirTag::BitNot: + return "BitNot"; + case JirTag::Shl: + return "Shl"; + case JirTag::AShr: + return "AShr"; + case JirTag::LShr: + return "LShr"; + case JirTag::LogNot: + return "LogNot"; + case JirTag::ZExt: + return "ZExt"; + case JirTag::SExt: + return "SExt"; + case JirTag::Trunc: + return "Trunc"; + case JirTag::SIToFP: + return "SIToFP"; + case JirTag::UIToFP: + return "UIToFP"; + case JirTag::FPToSI: + return "FPToSI"; + case JirTag::FPToUI: + return "FPToUI"; + case JirTag::FPExt: + return "FPExt"; + case JirTag::FPTrunc: + return "FPTrunc"; + case JirTag::BitCast: + return "BitCast"; + case JirTag::Br: + return "Br"; + case JirTag::CondBr: + return "CondBr"; + case JirTag::Switch: + return "Switch"; + case JirTag::Ret: + return "Ret"; + case JirTag::Unreachable: + return "Unreachable"; + case JirTag::Call: + return "Call"; + case JirTag::Param: + return "Param"; + case JirTag::StructLit: + return "StructLit"; + case JirTag::FieldAccess: + return "FieldAccess"; + case JirTag::ExtractValue: + return "ExtractValue"; + case JirTag::ArrayLit: + return "ArrayLit"; + case JirTag::Index: + return "Index"; + case JirTag::FieldAddr: + return "FieldAddr"; + case JirTag::IndexAddr: + return "IndexAddr"; + case JirTag::AddrOf: + return "AddrOf"; + case JirTag::Deref: + return "Deref"; + case JirTag::EnumPayload: + return "EnumPayload"; + case JirTag::DropBinding: + return "DropBinding"; } return "?"; } @@ -111,10 +184,10 @@ struct Verifier { // instruction as "defined" the moment we step past it. std::vector defined; - Verifier(const JirFunction &f, const TypePool *tp, - const StringPool *sp, JirVerifyResolver r, void *rctx) - : jfn(f), types(tp), strings(sp), resolver(r), - resolverCtx(rctx), defined(f.insts.size(), false) { + Verifier(const JirFunction &f, const TypePool *tp, const StringPool *sp, + JirVerifyResolver r, void *rctx) + : jfn(f), types(tp), strings(sp), resolver(r), resolverCtx(rctx), + defined(f.insts.size(), false) { defined[0] = true; // sentinel always usable as kNoJirRef } @@ -131,51 +204,45 @@ struct Verifier { void err(JirRef r, const std::string &msg) { uint32_t line = (r < jfn.insts.size()) ? jfn.insts[r].srcLine : 0; std::string tag = - (r < jfn.insts.size()) - ? tagName(jfn.insts[r].tag) - : "?"; + (r < jfn.insts.size()) ? tagName(jfn.insts[r].tag) : "?"; jam::Diagnostic d; d.loc.line = static_cast(line); d.severity = jam::Diagnostic::Severity::Error; d.message = "jir-verify: fn `" + jfn.name + "` ref #" + - std::to_string(r) + " (" + tag + "): " + msg; + std::to_string(r) + " (" + tag + "): " + msg; diags.push_back(std::move(d)); } void blockErr(JirBlockRef b, const std::string &msg) { - std::string name = (b < jfn.blocks.size()) - ? jfn.blocks[b].name - : "?"; + std::string name = (b < jfn.blocks.size()) ? jfn.blocks[b].name : "?"; jam::Diagnostic d; d.severity = jam::Diagnostic::Severity::Error; d.message = "jir-verify: fn `" + jfn.name + "` block #" + - std::to_string(b) + " (" + name + "): " + msg; + std::to_string(b) + " (" + name + "): " + msg; diags.push_back(std::move(d)); } // Check that JirRef `r` (0-based interpretation: kNoJirRef OK iff // optional) points within `insts` and was defined in an earlier // block (or earlier in this block). - void checkRef(JirRef r, bool optional, JirRef siteRef, - const char *what) { + void checkRef(JirRef r, bool optional, JirRef siteRef, const char *what) { if (r == kNoJirRef) { if (!optional) { err(siteRef, std::string("required operand `") + what + - "` is kNoJirRef"); + "` is kNoJirRef"); } return; } if (r >= jfn.insts.size()) { - err(siteRef, - std::string("operand `") + what + "` ref " + - std::to_string(r) + " out of bounds (max " + - std::to_string(jfn.insts.size() - 1) + ")"); + err(siteRef, std::string("operand `") + what + "` ref " + + std::to_string(r) + " out of bounds (max " + + std::to_string(jfn.insts.size() - 1) + ")"); return; } if (!defined[r]) { err(siteRef, std::string("operand `") + what + "` ref " + - std::to_string(r) + - " used before its defining block"); + std::to_string(r) + + " used before its defining block"); } } @@ -185,21 +252,19 @@ struct Verifier { return; } if (b >= jfn.blocks.size()) { - err(siteRef, - std::string("block ref `") + what + "` " + - std::to_string(b) + " out of bounds (max " + - std::to_string(jfn.blocks.size() - 1) + ")"); + err(siteRef, std::string("block ref `") + what + "` " + + std::to_string(b) + " out of bounds (max " + + std::to_string(jfn.blocks.size() - 1) + ")"); } } - void checkExtraSlice(JirExtraIdx start, std::size_t len, - JirRef siteRef, const char *what) { + void checkExtraSlice(JirExtraIdx start, std::size_t len, JirRef siteRef, + const char *what) { if (start + len > jfn.extra.size()) { err(siteRef, - std::string("extra slice `") + what + - "` overflows: needs [" + std::to_string(start) + - ".." + std::to_string(start + len) + ") but pool size is " + - std::to_string(jfn.extra.size())); + std::string("extra slice `") + what + "` overflows: needs [" + + std::to_string(start) + ".." + std::to_string(start + len) + + ") but pool size is " + std::to_string(jfn.extra.size())); } } @@ -341,10 +406,10 @@ struct Verifier { checkRef(inst.b, false, r, "b"); if (!refTypesMatch(inst.a, inst.b)) { err(r, std::string("operand type mismatch for ") + - tagName(inst.tag) + " (a.ty=" + - std::to_string(jfn.insts[inst.a].ty) + - " b.ty=" + - std::to_string(jfn.insts[inst.b].ty) + ")"); + tagName(inst.tag) + + " (a.ty=" + std::to_string(jfn.insts[inst.a].ty) + + " b.ty=" + std::to_string(jfn.insts[inst.b].ty) + + ")"); } return; // Shifts: LHS is the value being shifted (any int width), @@ -374,8 +439,7 @@ struct Verifier { checkRef(inst.a, false, r, "cond"); checkExtraSlice(inst.b, 2, r, "branches"); if (inst.b + 2 <= jfn.extra.size()) { - JirBlockRef thenB = - static_cast(jfn.extra[inst.b]); + JirBlockRef thenB = static_cast(jfn.extra[inst.b]); JirBlockRef elseB = static_cast(jfn.extra[inst.b + 1]); checkBlockRef(thenB, r, "then"); @@ -389,14 +453,13 @@ struct Verifier { err(r, "Switch extra header out of bounds"); return; } - JirBlockRef defaultB = - static_cast(jfn.extra[inst.b]); + JirBlockRef defaultB = static_cast(jfn.extra[inst.b]); uint32_t caseCount = jfn.extra[inst.b + 1]; checkBlockRef(defaultB, r, "default"); checkExtraSlice(inst.b, 2 + caseCount * 4, r, "switch"); for (uint32_t i = 0; i < caseCount; i++) { - JirBlockRef caseB = static_cast( - jfn.extra[inst.b + 2 + i * 4 + 3]); + JirBlockRef caseB = + static_cast(jfn.extra[inst.b + 2 + i * 4 + 3]); checkBlockRef(caseB, r, "case-block"); } return; @@ -413,8 +476,7 @@ struct Verifier { uint32_t argCount = jfn.extra[inst.b]; checkExtraSlice(inst.b, 1 + argCount, r, "call-args"); for (uint32_t i = 0; i < argCount; i++) { - JirRef ar = - static_cast(jfn.extra[inst.b + 1 + i]); + JirRef ar = static_cast(jfn.extra[inst.b + 1 + i]); checkRef(ar, false, r, "call-arg"); } return; @@ -428,8 +490,7 @@ struct Verifier { uint32_t count = jfn.extra[inst.b]; checkExtraSlice(inst.b, 1 + count, r, "agg-fields"); for (uint32_t i = 0; i < count; i++) { - JirRef fr = - static_cast(jfn.extra[inst.b + 1 + i]); + JirRef fr = static_cast(jfn.extra[inst.b + 1 + i]); checkRef(fr, false, r, "agg-field"); } return; @@ -441,10 +502,10 @@ struct Verifier { } // namespace std::vector verifyJirFunction(const JirFunction &jfn, - const TypePool *types, - const StringPool *strings, - JirVerifyResolver resolver, - void *resolverCtx) { + const TypePool *types, + const StringPool *strings, + JirVerifyResolver resolver, + void *resolverCtx) { Verifier v(jfn, types, strings, resolver, resolverCtx); // Walk blocks in declaration order, mirroring jir_codegen's pass. @@ -461,11 +522,9 @@ std::vector verifyJirFunction(const JirFunction &jfn, for (JirBlockRef b = 1; b < jfn.blocks.size(); b++) { if (b == 1) continue; for (JirRef r : jfn.blocks[b].insts) { - if (r < jfn.insts.size() && - jfn.insts[r].tag == JirTag::Alloca) { - v.blockErr(b, - "Alloca outside entry block — use " - "emitAllocaHoisted"); + if (r < jfn.insts.size() && jfn.insts[r].tag == JirTag::Alloca) { + v.blockErr(b, "Alloca outside entry block — use " + "emitAllocaHoisted"); } } } diff --git a/src/jir_verify.h b/src/jir_verify.h index 3b7e7ce..7b6417b 100644 --- a/src/jir_verify.h +++ b/src/jir_verify.h @@ -60,9 +60,10 @@ using JirVerifyResolver = TypeIdx (*)(void *ctx, TypeIdx ty); // Diagnostics channel). The `message` text already contains // fn-name + ref + tag context so the user can pin which // instruction tripped the check. -std::vector verifyJirFunction( - const JirFunction &jfn, const TypePool *types = nullptr, - const StringPool *strings = nullptr, - JirVerifyResolver resolver = nullptr, void *resolverCtx = nullptr); +std::vector +verifyJirFunction(const JirFunction &jfn, const TypePool *types = nullptr, + const StringPool *strings = nullptr, + JirVerifyResolver resolver = nullptr, + void *resolverCtx = nullptr); #endif // JIR_VERIFY_H diff --git a/src/lexer.cpp b/src/lexer.cpp index 5b3f486..acaf9d0 100644 --- a/src/lexer.cpp +++ b/src/lexer.cpp @@ -431,8 +431,8 @@ std::vector Lexer::scanTokens() { // Snapshot the byte offset of the token we're about to scan. // Every `addToken` call below reads this via member state, so - // each emitted Token carries its start offset — matches Zig's - // `Ast.TokenList.start` field. + // each emitted Token carries its start offset, used downstream + // for source-range and column resolution. tokenStart = static_cast(current); char c = advance(); diff --git a/src/lexer.h b/src/lexer.h index 481ee3d..80b1f4a 100644 --- a/src/lexer.h +++ b/src/lexer.h @@ -20,8 +20,8 @@ class Lexer { int line = 1; // Byte offset where the current token began — captured at the top // of each scan-loop iteration and recorded onto every emitted - // Token. Matches Zig's `Ast.TokenList = { tag, start }` shape: - // the persistent per-token info is just the start offset. + // Token. The persistent per-token info is just the start offset; + // line and column are recomputed from this when needed. uint32_t tokenStart = 0; bool isAtEnd() const; diff --git a/src/main.cpp b/src/main.cpp index 92085e4..62dad0c 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -25,8 +25,8 @@ #include "jam_llvm.h" #include "jir_codegen.h" #include "jir_verify.h" -#include "mangling.h" #include "lexer.h" +#include "mangling.h" #include "module_resolver.h" #include "parser.h" #include "symbol_table.h" @@ -383,16 +383,14 @@ static int compileAndRun(const std::string &filename, // at registration time by walking the InitExpr AST: if the // expression is a Call whose callee is a registered generic with // return type `type`, we evaluate each arg as a type and bind - // the result via `registerTypeAlias`. Matches Zig's "types are - // values" model: the parser stays grammar-only; the - // type-vs-value decision lives at semantic time. + // the result via `registerTypeAlias`. The parser stays grammar- + // only; the type-vs-value decision lives at semantic time. const NodeStore &ns = codegenCtx.getNodeStore(); // Look up a function by source-level name across the main module // and any imported pub fns. Functions aren't yet in the codegen // context's registry at this point (that happens in pass 1d). - auto lookupGenericFn = - [&](const std::string &name) -> const FunctionAST * { + auto lookupGenericFn = [&](const std::string &name) -> const FunctionAST * { for (auto &fn : module->Functions) { if (fn->Name == name) return fn.get(); } @@ -409,8 +407,8 @@ static int compileAndRun(const std::string &filename, [&](NodeIdx exprIdx) -> TypeIdx { const AstNode &n = ns.get(exprIdx); if (n.tag == AstTag::Variable) { - const std::string &name = codegenCtx.getStringPool().get( - static_cast(n.lhs)); + const std::string &name = + codegenCtx.getStringPool().get(static_cast(n.lhs)); // Builtin scalar names. if (name == "u8") return BuiltinType::U8; if (name == "i8") return BuiltinType::I8; @@ -471,8 +469,7 @@ static int compileAndRun(const std::string &filename, if (c->InitExpr != kNoNode) { TypeIdx maybeAlias = resolveExprAsType(c->InitExpr); if (maybeAlias != kNoType) { - const TypeKey &k = - codegenCtx.getTypePool().get(maybeAlias); + const TypeKey &k = codegenCtx.getTypePool().get(maybeAlias); // Only bind as alias when the resolved type is // an actual user-visible category — generic // instantiations, struct/enum/union names. A bare @@ -480,8 +477,7 @@ static int compileAndRun(const std::string &filename, // resolve fall through to value-const behavior. if (k.kind == TypeKind::GenericCall) { c->AliasedType = maybeAlias; - codegenCtx.registerTypeAlias(c->Name, - c->AliasedType); + codegenCtx.registerTypeAlias(c->Name, c->AliasedType); continue; } } @@ -498,10 +494,9 @@ static int compileAndRun(const std::string &filename, // Two-pass codegen: declare every function's prototype first, then // emit bodies. Without this, calling a function defined later in the - // file (or another module) would fail with "Unknown function". This - // is also how Zig handles forward references — top-down reads naturally - // (main on top, helpers below) without a manual "forward declarations" - // section. + // file (or another module) would fail with "Unknown function". The + // two-pass shape lets source read naturally top-down (main on top, + // helpers below) without a manual "forward declarations" section. // Pass 1a: prototypes for pub functions in imported modules. for (const auto &[path, importedModule] : resolver.getLoadedModules()) { @@ -509,9 +504,8 @@ static int compileAndRun(const std::string &filename, for (auto &func : importedModule->Functions) { if (func->isPub && !func->isGeneric()) { JirFunction jfn = astgenMetadata(*func, codegenCtx); - jfn.name = mangledFunctionName( - *func, codegenCtx.getTypePool(), - codegenCtx.getStringPool()); + jfn.name = mangledFunctionName(*func, codegenCtx.getTypePool(), + codegenCtx.getStringPool()); jirDeclarePrototype(jfn, codegenCtx); } // Generic functions are still registered so call sites can @@ -588,9 +582,8 @@ static int compileAndRun(const std::string &filename, // name — Jam allows `fn add_u8(...)` and `tfn add_u8()` to coexist, // and the bare-name lookup in astgenCall must resolve to the // regular function, not its similarly-named test. - const std::string regName = function->isTest - ? "__test_" + function->Name - : function->Name; + const std::string regName = + function->isTest ? "__test_" + function->Name : function->Name; codegenCtx.registerFunctionAST(regName, function.get()); // Non-generic main-module functions get their LLVM prototype // emitted by `jirDeclarePrototype` in pass 1d, alongside their @@ -661,9 +654,8 @@ static int compileAndRun(const std::string &filename, } { JirFunction jfn = astgenMetadata(*m, codegenCtx); - jfn.name = mangledFunctionName( - *m, codegenCtx.getTypePool(), - codegenCtx.getStringPool()); + jfn.name = mangledFunctionName(*m, codegenCtx.getTypePool(), + codegenCtx.getStringPool()); jirDeclarePrototype(jfn, codegenCtx); } codegenCtx.registerFunctionAST(s->Name + "." + m->Name, m.get()); @@ -693,9 +685,7 @@ static int compileAndRun(const std::string &filename, // Per-decl recovery: if astgen on one function throws // AstGenAnalysisFail, the diagnostic was already pushed to // codegenCtx.diagnostics(). We keep going so the user sees every - // error in one pass instead of having to fix-rebuild-fix. Mirrors - // Zig's per-decl `error.AnalysisFail` boundary (`Sema.zig` calls - // AstGen per-decl and continues on AnalysisFail). + // error in one pass instead of having to fix-rebuild-fix. for (auto &function : module->Functions) { if (function->isTest && !testMode) continue; if (!function->isTest && testMode && function->Name == "main") { @@ -704,9 +694,8 @@ static int compileAndRun(const std::string &filename, if (function->isGeneric()) continue; try { JirFunction jfn = astgenFunction(*function, codegenCtx); - jfn.name = mangledFunctionName( - *function, codegenCtx.getTypePool(), - codegenCtx.getStringPool()); + jfn.name = mangledFunctionName(*function, codegenCtx.getTypePool(), + codegenCtx.getStringPool()); jirDeclarePrototype(jfn, codegenCtx); jirFunctions.push_back(std::move(jfn)); } catch (const AstGenAnalysisFail &) { @@ -717,9 +706,8 @@ static int compileAndRun(const std::string &filename, for (auto &m : s->Methods) { try { JirFunction jfn = astgenFunction(*m, codegenCtx); - jfn.name = mangledFunctionName( - *m, codegenCtx.getTypePool(), - codegenCtx.getStringPool()); + jfn.name = mangledFunctionName(*m, codegenCtx.getTypePool(), + codegenCtx.getStringPool()); jirFunctions.push_back(std::move(jfn)); } catch (const AstGenAnalysisFail &) { // diagnostic already pushed @@ -731,11 +719,10 @@ static int compileAndRun(const std::string &filename, for (auto &func : importedModule->Functions) { if (func->isPub && !func->isGeneric()) { try { - JirFunction jfn = - astgenFunction(*func, codegenCtx); - jfn.name = mangledFunctionName( - *func, codegenCtx.getTypePool(), - codegenCtx.getStringPool()); + JirFunction jfn = astgenFunction(*func, codegenCtx); + jfn.name = + mangledFunctionName(*func, codegenCtx.getTypePool(), + codegenCtx.getStringPool()); jirFunctions.push_back(std::move(jfn)); } catch (const AstGenAnalysisFail &) { // diagnostic already pushed @@ -764,8 +751,7 @@ static int compileAndRun(const std::string &filename, // function shouldn't reach jir_codegen if it's malformed. for (const JirFunction &jfn : jirFunctions) { auto diags = verifyJirFunction( - jfn, &codegenCtx.getTypePool(), - &codegenCtx.getStringPool(), + jfn, &codegenCtx.getTypePool(), &codegenCtx.getStringPool(), +[](void *c, TypeIdx t) -> TypeIdx { auto *cc = static_cast(c); const TypeKey &k = cc->getTypePool().get(t); @@ -828,8 +814,8 @@ static int compileAndRun(const std::string &filename, // emitted in pass 2b. for (auto &function : module->Functions) { if (function->isTest && !testMode) continue; - if (!function->isTest && testMode && - function->Name == "main") continue; + if (!function->isTest && testMode && function->Name == "main") + continue; if (function->isGeneric()) continue; runAnalysis(function.get()); } diff --git a/src/mangling.h b/src/mangling.h index 80ddd41..2a000a5 100644 --- a/src/mangling.h +++ b/src/mangling.h @@ -37,12 +37,9 @@ inline std::string mangledFunctionName(const FunctionAST &fn, const Param &p = fn.Args[0]; if (p.Name == "self" && p.Mode == ParamMode::Mut) { const TypeKey &k = types.get(p.Type); - if (k.kind == TypeKind::Struct || - k.kind == TypeKind::Named) { + if (k.kind == TypeKind::Struct || k.kind == TypeKind::Named) { StringIdx ni = static_cast(k.a); - if (ni != kNoString) { - return "__drop_" + strings.get(ni); - } + if (ni != kNoString) { return "__drop_" + strings.get(ni); } } } } diff --git a/src/parser.cpp b/src/parser.cpp index aac8b40..ca2266e 100644 --- a/src/parser.cpp +++ b/src/parser.cpp @@ -18,7 +18,7 @@ // the parsed `double` is returned via `bit_cast`, and `isFloatOut` is // set so the caller can mark the AST node with the float flag. uint64_t Parser::parseNumLexeme(const std::string &s, bool &isNegOut, - bool &isFloatOut) const { + bool &isFloatOut) const { bool neg = !s.empty() && s[0] == '-'; const std::string &abs = neg ? s.substr(1) : s; isNegOut = neg; @@ -29,8 +29,7 @@ uint64_t Parser::parseNumLexeme(const std::string &s, bool &isNegOut, case NumberResultKind::Int: return r.intValue; case NumberResultKind::BigInt: - parseError("integer literal `" + abs + - "` exceeds u64 range"); + parseError("integer literal `" + abs + "` exceeds u64 range"); case NumberResultKind::Float: { isFloatOut = true; // pack the double's bit pattern into u64. memcpy keeps it @@ -44,7 +43,7 @@ uint64_t Parser::parseNumLexeme(const std::string &s, bool &isNegOut, } case NumberResultKind::Failure: parseError(std::string("invalid numeric literal `") + abs + - "`: " + numberErrorMessage(r.failure.kind)); + "`: " + numberErrorMessage(r.failure.kind)); } parseError("unreachable number-literal classification"); } @@ -59,8 +58,8 @@ Parser::Parser(std::vector tokens, TypePool &typePool_, jam::SrcLoc Parser::currentLoc() const { int line = 0; if (!tokens.empty()) { - int idx = current < static_cast(tokens.size()) ? current - : current - 1; + int idx = + current < static_cast(tokens.size()) ? current : current - 1; if (idx < 0) idx = 0; line = tokens[idx].line; } @@ -68,9 +67,7 @@ jam::SrcLoc Parser::currentLoc() const { } void Parser::parseError(std::string message) const { - if (diagnostics_) { - diagnostics_->error(currentLoc(), std::move(message)); - } + if (diagnostics_) { diagnostics_->error(currentLoc(), std::move(message)); } throw ParserAbort{}; } @@ -119,9 +116,7 @@ NodeIdx Parser::emit(AstNode n) { // line lets `file:line:` prefixes appear on errors thrown deep in // codegen even though the AST itself is positionless. int line = 0; - if (n.mainToken < tokens.size()) { - line = tokens[n.mainToken].line; - } + if (n.mainToken < tokens.size()) { line = tokens[n.mainToken].line; } return nodes->addNodeAt(n, line); } @@ -419,9 +414,8 @@ TypeIdx Parser::parseType() { if (match(TOK_CONST)) { ptrConst = true; } else if (!match(TOK_MUT)) { - parseError( - "Expected `const` or `mut` after `*` (e.g. `*const T`, " - "`*mut T`)"); + parseError("Expected `const` or `mut` after `*` (e.g. `*const T`, " + "`*mut T`)"); } // Optional `[]` promotes to many-item form. We need to commit to // "many" only when the next two tokens are exactly `[ ]`; a `[` @@ -475,8 +469,7 @@ TypeIdx Parser::parseType() { const std::string &firstIdent = previous().lexeme; if (firstIdent == "Self") { if (structContextStack.empty()) { - parseError( - "`Self` is only valid inside a struct body"); + parseError("`Self` is only valid inside a struct body"); } return typePool->internNamed( stringPool->intern(structContextStack.back())); @@ -637,18 +630,16 @@ NodeIdx Parser::parsePatternAtom() { variantNameId}); } current = saved; - parseError( - "Bare identifier patterns are not yet supported " - "(use `EnumName.Variant`, an integer literal, or `_`)"); + parseError("Bare identifier patterns are not yet supported " + "(use `EnumName.Variant`, an integer literal, or `_`)"); } if (match(TOK_NUMBER)) { bool isNegative = false; bool isFloat = false; uint64_t lo = parseNumLexeme(previous().lexeme, isNegative, isFloat); if (isFloat) { - parseError( - "Float literals are not allowed in `match` patterns " - "(use an integer literal or a `..=` range)"); + parseError("Float literals are not allowed in `match` patterns " + "(use an integer literal or a `..=` range)"); } // Inclusive range `lo..=hi`? if (match(TOK_DOTDOT_EQ)) { @@ -674,16 +665,14 @@ NodeIdx Parser::parsePatternAtom() { // token; we treat single-quote chars as TOK_NUMBER via the // lexer in a future patch. For now, only TOK_NUMBER is accepted. if (match(TOK_STRING_LITERAL)) { - parseError( - "Char literals in patterns are not yet supported"); + parseError("Char literals in patterns are not yet supported"); } // Wildcard `_` is lexed as TOK_IDENTIFIER; recognize it here. if (check(TOK_IDENTIFIER) && peek().lexeme == "_") { advance(); return emit(AstNode{AstTag::PatWildcard, 0, 0, 0, 0, 0}); } - parseError( - "Expected pattern (integer literal, range, or `_`)"); + parseError("Expected pattern (integer literal, range, or `_`)"); } NodeIdx Parser::parsePattern() { @@ -1039,9 +1028,8 @@ NodeIdx Parser::parseMultiplication() { } // Parse a postfix `as Type` chain on top of any unary expression. The -// cast binds tightly — `5 + (x as u32)` not `(5 + x) as u32` — matching -// the convention from C/Rust/Zig where `as` sits just above primary -// expressions. +// cast binds tightly — `5 + (x as u32)` not `(5 + x) as u32` — placing +// the cast just above primary expressions in the precedence table. static NodeIdx parseAsChain(Parser *self, NodeIdx expr, NodeStore &nodes, bool (Parser::*matchTok)(TokenType), TypeIdx (Parser::*parseType)()); @@ -1351,14 +1339,13 @@ std::unique_ptr Parser::parseEnumDecl() { } mag = r.intValue; } catch (const std::exception &e) { - parseError( - std::string("Invalid enum discriminant: ") + e.what()); + parseError(std::string("Invalid enum discriminant: ") + + e.what()); } (void)neg; if (mag > 255) { - parseError( - "Enum discriminant " + std::to_string(mag) + - " is out of range; M2 enums are u8-tagged"); + parseError("Enum discriminant " + std::to_string(mag) + + " is out of range; M2 enums are u8-tagged"); } v.Discriminant = static_cast(mag); nextDiscrim = v.Discriminant + 1; @@ -1372,13 +1359,11 @@ std::unique_ptr Parser::parseEnumDecl() { consume(TOK_SEMI, "Expected ';' after enum declaration"); if (variants.empty()) { - parseError("Enum `" + name + - "` must declare at least one variant"); + parseError("Enum `" + name + "` must declare at least one variant"); } if (variants.size() > 256) { - parseError( - "Enum `" + name + - "` has more than 256 variants; M2 enums are u8-tagged"); + parseError("Enum `" + name + + "` has more than 256 variants; M2 enums are u8-tagged"); } return std::make_unique(name, std::move(variants)); } @@ -1456,17 +1441,12 @@ std::unique_ptr Parser::parseConstDecl() { consume(TOK_EQUAL, "Expected '=' in module-scope const declaration"); // Parse RHS as an expression — no speculation. Type-alias intent // (`const ListI32 = Vec(i32);`) is recognised at registration - // time in main.cpp::resolveExprAsType by *pattern-matching* the + // time in main.cpp::resolveExprAsType by pattern-matching the // resulting AST (`Variable` for type-named bindings, direct - // `Call` of a generic-fn-returning-`type`). This is the same - // grammar shape Zig uses (`const Foo = Bar(i32);` parses as an - // expression in `lib/std/zig/parse.zig:820 parseVarDecl`), but - // the *semantic* dispatch is different: Zig comptime-evaluates - // the RHS in Sema and inspects the resulting value's type, - // which handles arbitrary expressions (if/else, member access, - // chained calls, 0-arg type-returning fns). Jam's pattern match - // covers only the common cases; richer forms are deferred until - // (and if) we add comptime evaluation. + // `Call` of a generic-fn-returning-`type`). The pattern match + // covers the common cases; richer forms (if/else types, member + // access on types, chained calls, 0-arg type-returning fns) + // are deferred until comptime evaluation lands. NodeIdx init = parseLogicalOr(); consume(TOK_SEMI, "Expected ';' after module-scope const declaration"); @@ -1496,8 +1476,7 @@ std::unique_ptr Parser::parse() { int peek = current + 1; if (peek < static_cast(tokens.size()) && tokens[peek].type == TOK_OPEN_BRACE) { - parseError( - "`pub` is not allowed on destructuring imports"); + parseError("`pub` is not allowed on destructuring imports"); } } } @@ -1527,8 +1506,7 @@ std::unique_ptr Parser::parse() { advance(); if (check(TOK_IMPORT)) { if (isPub) { - parseError( - "`pub` is not allowed on imports"); + parseError("`pub` is not allowed on imports"); } current = saved; module->Imports.push_back(parseImportDecl()); diff --git a/src/parser.h b/src/parser.h index 5abe0b9..50f393c 100644 --- a/src/parser.h +++ b/src/parser.h @@ -45,7 +45,7 @@ class Parser { // Member rather than free function so it can route over-large / // malformed literals through `parseError` with the token's line. uint64_t parseNumLexeme(const std::string &s, bool &isNegOut, - bool &isFloatOut) const; + bool &isFloatOut) const; NodeIdx parsePrimary(); NodeIdx parseUnary(); @@ -83,8 +83,7 @@ class Parser { public: Parser(std::vector tokens, TypePool &typePool, StringPool &stringPool, NodeStore &nodes, - jam::Diagnostics *diagnostics = nullptr, - std::string filename = ""); + jam::Diagnostics *diagnostics = nullptr, std::string filename = ""); std::unique_ptr parse(); std::vector> *sharedAnonStructs = nullptr; std::vector> *sharedAnonEnums = nullptr; diff --git a/src/token.h b/src/token.h index 222366f..12de700 100644 --- a/src/token.h +++ b/src/token.h @@ -46,14 +46,15 @@ enum TokenType { TOK_CLOSE_BRACKET, TOK_STRING_LITERAL, TOK_WHILE, - TOK_LOOP, // `loop { ... }` — infinite loop, syntactically noreturn unless body breaks + TOK_LOOP, // `loop { ... }` — infinite loop, syntactically noreturn unless + // body breaks TOK_FOR, TOK_BREAK, TOK_CONTINUE, TOK_IN, TOK_EXTERN, // extern keyword (import C function) TOK_EXPORT, // export keyword (C ABI export) - TOK_PUB, // pub keyword (visible to Jam modules, like Zig) + TOK_PUB, // pub keyword (visible across Jam modules) TOK_IMPORT, // import keyword TOK_DOT, // . for member access TOK_AND, // && (logical AND, short-circuit) diff --git a/tests/cpp/test_diagnostics.cpp b/tests/cpp/test_diagnostics.cpp index 2168bce..b856585 100644 --- a/tests/cpp/test_diagnostics.cpp +++ b/tests/cpp/test_diagnostics.cpp @@ -49,11 +49,10 @@ bool stderrContains(const CompileResult &r, const std::string &substr) { // to assert that EVERY error in a batch is reported (not just the // first one). std::size_t countOccurrences(const std::string &hay, - const std::string &needle) { + const std::string &needle) { std::size_t n = 0; for (std::size_t pos = 0; - (pos = hay.find(needle, pos)) != std::string::npos; - ++pos) { + (pos = hay.find(needle, pos)) != std::string::npos; ++pos) { ++n; } return n; @@ -62,43 +61,39 @@ std::size_t countOccurrences(const std::string &hay, // ── Line-number tests ────────────────────────────────────────── void testUnknownFunctionHasLine() { - auto r = compileSource("diag_unknown_fn", - "fn main() i32 {\n" - " foobar();\n" - " return 0;\n" - "}\n"); + auto r = compileSource("diag_unknown_fn", "fn main() i32 {\n" + " foobar();\n" + " return 0;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2: error:")); ASSERT_TRUE(stderrContains(r, "unknown function `foobar`")); } void testUnknownVariableHasLine() { - auto r = compileSource("diag_unknown_var", - "fn main() i32 {\n" - " return badName;\n" - "}\n"); + auto r = compileSource("diag_unknown_var", "fn main() i32 {\n" + " return badName;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2: error:")); ASSERT_TRUE(stderrContains(r, "unknown variable `badName`")); } void testBreakOutsideLoopHasLine() { - auto r = compileSource("diag_break_loop", - "fn main() i32 {\n" - " break;\n" - " return 0;\n" - "}\n"); + auto r = compileSource("diag_break_loop", "fn main() i32 {\n" + " break;\n" + " return 0;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2: error:")); ASSERT_TRUE(stderrContains(r, "`break` outside of loop")); } void testContinueOutsideLoopHasLine() { - auto r = compileSource("diag_continue_loop", - "fn main() i32 {\n" - " continue;\n" - " return 0;\n" - "}\n"); + auto r = compileSource("diag_continue_loop", "fn main() i32 {\n" + " continue;\n" + " return 0;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2: error:")); ASSERT_TRUE(stderrContains(r, "`continue` outside of loop")); @@ -134,11 +129,10 @@ void testUseAfterMoveHasLine() { // ── Multi-error reporting ─────────────────────────────────────── void testMultipleErrorsAllReported() { - auto r = compileSource("diag_multi", - "fn first() i32 { return foo(); }\n" - "fn second() i32 { return bar(); }\n" - "fn third() i32 { break; return 0; }\n" - "fn main() i32 { return 0; }\n"); + auto r = compileSource("diag_multi", "fn first() i32 { return foo(); }\n" + "fn second() i32 { return bar(); }\n" + "fn third() i32 { break; return 0; }\n" + "fn main() i32 { return 0; }\n"); ASSERT_TRUE(r.exitCode != 0); // Each of the three error lines should appear once. ASSERT_TRUE(stderrContains(r, ":1: error:")); @@ -150,10 +144,10 @@ void testMultipleErrorsAllReported() { } void testMultiErrorsSortedByLine() { - auto r = compileSource("diag_multi_sorted", - "fn a() i32 { return zzz(); }\n" - "fn b() i32 { return aaa(); }\n" - "fn main() i32 { return 0; }\n"); + auto r = + compileSource("diag_multi_sorted", "fn a() i32 { return zzz(); }\n" + "fn b() i32 { return aaa(); }\n" + "fn main() i32 { return 0; }\n"); ASSERT_TRUE(r.exitCode != 0); // Line 1's error must appear before line 2's, regardless of // function name alphabet — Diagnostics::emit sorts by location. @@ -166,20 +160,19 @@ void testMultiErrorsSortedByLine() { // ── Reference trace for generic instantiation ────────────────── void testGenericInstantiationCarriesRefTrace() { - auto r = compileSource( - "diag_ref_trace", - "fn Box(T: type) type {\n" - " return struct {\n" - " val: T,\n" - " fn pickBad(self: Self) i32 {\n" - " return self.notAField;\n" - " }\n" - " };\n" - "}\n" - "fn main() i32 {\n" - " var b: Box(i32) = { val: 7 };\n" - " return b.pickBad();\n" - "}\n"); + auto r = + compileSource("diag_ref_trace", "fn Box(T: type) type {\n" + " return struct {\n" + " val: T,\n" + " fn pickBad(self: Self) i32 {\n" + " return self.notAField;\n" + " }\n" + " };\n" + "}\n" + "fn main() i32 {\n" + " var b: Box(i32) = { val: 7 };\n" + " return b.pickBad();\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); // Underlying error inside the instantiated body — line 5 in // source. Reference trace adds "in instantiation of @@ -192,10 +185,9 @@ void testGenericInstantiationCarriesRefTrace() { void testBrokenDeclDoesNotMaskNextDecl() { auto r = compileSource( - "diag_recovery", - "fn broken() i32 { return missingFn(); }\n" - "fn alsoBroken() i32 { return anotherMissingFn(); }\n" - "fn main() i32 { return 0; }\n"); + "diag_recovery", "fn broken() i32 { return missingFn(); }\n" + "fn alsoBroken() i32 { return anotherMissingFn(); }\n" + "fn main() i32 { return 0; }\n"); ASSERT_TRUE(r.exitCode != 0); // Both decls must produce an error (a single throw would have // reported only the first). @@ -210,13 +202,12 @@ void testBrokenDeclDoesNotMaskNextDecl() { // first failNode would bail the whole decl and we'd see only one // error — Zig's recoverable-error model lets us report them all. void testRecoveryWithinSingleDecl() { - auto r = compileSource("diag_recover_within", - "fn main() i32 {\n" - " var a: i32 = foo();\n" - " var b: i32 = bar();\n" - " var c: i32 = baz();\n" - " return a + b + c;\n" - "}\n"); + auto r = compileSource("diag_recover_within", "fn main() i32 {\n" + " var a: i32 = foo();\n" + " var b: i32 = bar();\n" + " var c: i32 = baz();\n" + " return a + b + c;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2: error:")); ASSERT_TRUE(stderrContains(r, ":3: error:")); @@ -227,12 +218,12 @@ void testRecoveryWithinSingleDecl() { } void testRecoveryAcrossDifferentErrorClasses() { - auto r = compileSource("diag_recover_mixed", - "fn main() i32 {\n" - " var p: i32 = unknownThing;\n" - " var q: i32 = alsoMissing();\n" - " return p + q;\n" - "}\n"); + auto r = + compileSource("diag_recover_mixed", "fn main() i32 {\n" + " var p: i32 = unknownThing;\n" + " var q: i32 = alsoMissing();\n" + " return p + q;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2: error:")); ASSERT_TRUE(stderrContains(r, ":3: error:")); @@ -241,16 +232,15 @@ void testRecoveryAcrossDifferentErrorClasses() { } void testUnknownMethodIsRecoverable() { - auto r = compileSource( - "diag_recover_method", - "const Point = struct { x: i32 };\n" - "fn other() i32 { return 99; }\n" - "fn main() i32 {\n" - " var p: Point = { x: 1 };\n" - " var a: i32 = p.bogusMethod();\n" - " var b: i32 = p.alsoBogus();\n" - " return a + b;\n" - "}\n"); + auto r = compileSource("diag_recover_method", + "const Point = struct { x: i32 };\n" + "fn other() i32 { return 99; }\n" + "fn main() i32 {\n" + " var p: Point = { x: 1 };\n" + " var a: i32 = p.bogusMethod();\n" + " var b: i32 = p.alsoBogus();\n" + " return a + b;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); // Both unknown-method calls in the same function should report. ASSERT_TRUE(stderrContains(r, "bogusMethod")); @@ -260,59 +250,54 @@ void testUnknownMethodIsRecoverable() { // ── Parser errors now carry :line: too ───────────────────────── void testParserErrorHasLine() { - auto r = compileSource("diag_parser_line", - "fn main() i32 {\n" - " return *;\n" - "}\n"); + auto r = compileSource("diag_parser_line", "fn main() i32 {\n" + " return *;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2:")); } void testRedeclarationInSameScopeRejected() { - auto r = compileSource("diag_redecl_same_scope", - "fn main() {\n" - " const a = true;\n" - " var a = true;\n" - "}\n"); + auto r = compileSource("diag_redecl_same_scope", "fn main() {\n" + " const a = true;\n" + " var a = true;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":3:")); ASSERT_TRUE(stderrContains(r, "redeclaration of `a`")); } void testRedeclarationAcrossSiblingScopesAllowed() { - auto r = compileSource( - "diag_redecl_siblings", - "fn main(x: i32) i32 {\n" - " if (x == 0) {\n" - " const op: u32 = 1;\n" - " return op as i32;\n" - " }\n" - " if (x == 1) {\n" - " const op: u32 = 2;\n" - " return op as i32;\n" - " }\n" - " return 99;\n" - "}\n"); + auto r = + compileSource("diag_redecl_siblings", "fn main(x: i32) i32 {\n" + " if (x == 0) {\n" + " const op: u32 = 1;\n" + " return op as i32;\n" + " }\n" + " if (x == 1) {\n" + " const op: u32 = 2;\n" + " return op as i32;\n" + " }\n" + " return 99;\n" + "}\n"); ASSERT_TRUE(r.exitCode == 0); } void testTypeMismatchBoolEqualsFloatRejected() { - auto r = compileSource("diag_bool_eq_float", - "fn main() {\n" - " const a: bool = 1.0;\n" - "}\n"); + auto r = compileSource("diag_bool_eq_float", "fn main() {\n" + " const a: bool = 1.0;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2:")); ASSERT_TRUE(stderrContains(r, "type mismatch in `a`")); } void testTypeInferenceAllocatesCorrectWidth() { - auto r = compileSource("diag_var_infer", - "fn main() i32 {\n" - " var b = true;\n" - " var f = 3.14;\n" - " return 0;\n" - "}\n"); + auto r = compileSource("diag_var_infer", "fn main() i32 {\n" + " var b = true;\n" + " var f = 3.14;\n" + " return 0;\n" + "}\n"); // Both inferred-type bindings should compile cleanly; before // the fix they silently allocated a 1-byte slot regardless of // init type. We can't introspect alloca widths from stderr, so @@ -322,11 +307,11 @@ void testTypeInferenceAllocatesCorrectWidth() { } void testIntegerOverflowLiteralHasLine() { - auto r = compileSource("diag_intover", - "fn main() i32 {\n" - " var v: u32 = 99999999999999999999;\n" - " return 0;\n" - "}\n"); + auto r = + compileSource("diag_intover", "fn main() i32 {\n" + " var v: u32 = 99999999999999999999;\n" + " return 0;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); ASSERT_TRUE(stderrContains(r, ":2:")); ASSERT_TRUE(stderrContains(r, "exceeds u64 range")); @@ -335,10 +320,9 @@ void testIntegerOverflowLiteralHasLine() { // ── Output format is stable ──────────────────────────────────── void testDiagnosticFormatIsFileLineError() { - auto r = compileSource("diag_format", - "fn main() i32 {\n" - " return undefined_thing;\n" - "}\n"); + auto r = compileSource("diag_format", "fn main() i32 {\n" + " return undefined_thing;\n" + "}\n"); ASSERT_TRUE(r.exitCode != 0); // Format: ":: error: " // — file colon line colon error colon message. @@ -350,8 +334,7 @@ void testDiagnosticFormatIsFileLineError() { // ":: error:" somewhere after it. auto tmp = line.find("/tmp/"); auto err = line.find(": error:"); - if (tmp != std::string::npos && err != std::string::npos && - err > tmp) { + if (tmp != std::string::npos && err != std::string::npos && err > tmp) { foundShape = true; break; } @@ -384,31 +367,26 @@ class DiagnosticTests { framework.addTest( "Diagnostics - generic instantiation carries ref trace", testGenericInstantiationCarriesRefTrace); - framework.addTest( - "Diagnostics - broken decl does not mask next decl", - testBrokenDeclDoesNotMaskNextDecl); - framework.addTest( - "Diagnostics - within-decl multi-error (Poison)", - testRecoveryWithinSingleDecl); - framework.addTest( - "Diagnostics - recovery across error classes", - testRecoveryAcrossDifferentErrorClasses); - framework.addTest( - "Diagnostics - unknown method is recoverable", - testUnknownMethodIsRecoverable); + framework.addTest("Diagnostics - broken decl does not mask next decl", + testBrokenDeclDoesNotMaskNextDecl); + framework.addTest("Diagnostics - within-decl multi-error (Poison)", + testRecoveryWithinSingleDecl); + framework.addTest("Diagnostics - recovery across error classes", + testRecoveryAcrossDifferentErrorClasses); + framework.addTest("Diagnostics - unknown method is recoverable", + testUnknownMethodIsRecoverable); framework.addTest("Diagnostics - parser error carries :line:", testParserErrorHasLine); - framework.addTest("Diagnostics - integer-overflow literal carries :line:", - testIntegerOverflowLiteralHasLine); framework.addTest( - "Diagnostics - redeclaration in same scope rejected", - testRedeclarationInSameScopeRejected); + "Diagnostics - integer-overflow literal carries :line:", + testIntegerOverflowLiteralHasLine); + framework.addTest("Diagnostics - redeclaration in same scope rejected", + testRedeclarationInSameScopeRejected); framework.addTest( "Diagnostics - redeclaration across sibling scopes OK", testRedeclarationAcrossSiblingScopesAllowed); - framework.addTest( - "Diagnostics - bool var = float literal rejected", - testTypeMismatchBoolEqualsFloatRejected); + framework.addTest("Diagnostics - bool var = float literal rejected", + testTypeMismatchBoolEqualsFloatRejected); framework.addTest( "Diagnostics - inferred var width correct for bool / float", testTypeInferenceAllocatesCorrectWidth); diff --git a/tests/cpp/test_jir_skeleton.cpp b/tests/cpp/test_jir_skeleton.cpp index 8ea5c8d..c8e5ce5 100644 --- a/tests/cpp/test_jir_skeleton.cpp +++ b/tests/cpp/test_jir_skeleton.cpp @@ -81,7 +81,8 @@ class JirSkeletonTests { testPushBlockReturnsMonotonicRefs); framework.addTest("JIR - extra pool appends and reads", testExtraPoolAppendsAndReads); - framework.addTest("JIR - instruction size is small", testInstSizeIsSmall); + framework.addTest("JIR - instruction size is small", + testInstSizeIsSmall); } }; -- 2.51.2