From 6c9508f03c25f33ce32f92a1e8d58aa86c5b9263 Mon Sep 17 00:00:00 2001 From: Raphael Amorim Date: Sat, 23 May 2026 10:55:49 +0200 Subject: [PATCH] add jam version and fix some pathological issues --- Makefile | 19 +- src/abi.cpp | 123 ++++++---- src/abi.h | 41 ++-- src/astgen.cpp | 580 +++++++++++++++++++++++++++++++++++--------- src/jam_llvm.cpp | 15 ++ src/jam_llvm.h | 7 + src/jir.h | 14 ++ src/jir_codegen.cpp | 165 ++++++++++--- src/jir_verify.cpp | 3 + src/main.cpp | 20 ++ 10 files changed, 782 insertions(+), 205 deletions(-) diff --git a/Makefile b/Makefile index 1371c72..faa8c5f 100644 --- a/Makefile +++ b/Makefile @@ -4,6 +4,21 @@ LLVM_CONFIG=$(shell which llvm-config 2>/dev/null || echo "llvm-config") OPTFLAGS ?= -O2 -DNDEBUG +# Short SHA of the build commit, baked into the binary so `jam --version` +# can report exactly which tree it was built from. Falls back to +# `unknown` outside a git checkout. Appends `-dirty` when the worktree +# has uncommitted changes — keeps "this isn't quite the tagged build" +# obvious in bug reports. +JAM_VERSION_SHA := $(shell \ + if git rev-parse --short HEAD >/dev/null 2>&1; then \ + sha=$$(git rev-parse --short HEAD); \ + if ! git diff --quiet HEAD 2>/dev/null; then sha="$$sha-dirty"; fi; \ + echo "$$sha"; \ + else \ + echo "unknown"; \ + fi) +VERSION_FLAGS := -DJAM_VERSION_SHA=\"$(JAM_VERSION_SHA)\" + CLANG_FORMAT ?= clang-format CLANG_FORMAT_STYLE := file:clang-format FORMAT_SOURCES := $(wildcard src/*.cpp src/*.h) $(wildcard tests/cpp/*.cpp tests/cpp/*.h) @@ -41,7 +56,9 @@ build: @mkdir -p $(OUT) @for name in $(SRC_NAMES); do \ echo " CC: src/$$name.cpp -> $(OUT)/$$name.o"; \ - clang++ -c ./src/$$name.cpp -o $(OUT)/$$name.o `$(LLVM_CONFIG) --cxxflags` -fexceptions $(OPTFLAGS) || exit $$?; \ + extra=""; \ + if [ "$$name" = "main" ]; then extra="$(VERSION_FLAGS)"; fi; \ + clang++ -c ./src/$$name.cpp -o $(OUT)/$$name.o `$(LLVM_CONFIG) --cxxflags` -fexceptions $(OPTFLAGS) $$extra || exit $$?; \ done @echo " LD: $(OUT)/jam.out" @clang++ -o $(OUT)/jam.out $(OBJS) `$(LLVM_CONFIG) --ldflags --libs --libfiles --system-libs` diff --git a/src/abi.cpp b/src/abi.cpp index 6e50210..295f358 100644 --- a/src/abi.cpp +++ b/src/abi.cpp @@ -12,48 +12,98 @@ namespace jam { namespace abi { -namespace { - -// A type is "scalar" for ABI purposes when it lowers to a single -// LLVM register-shaped value: integers, floats, booleans, pointers. -// Slices are 16-byte (ptr, len) aggregates and don't qualify here; -// they are aggregates that fit under the by-value size threshold. -bool isScalar(TypeKind k) { - switch (k) { +bool isByRef(TypeIdx ty, const JamCodegenContext &ctx) { + if (ty == kNoType) return false; + const TypeKey &k = ctx.getTypePool().get(ty); + switch (k.kind) { + // Source-only / comptime-only — never lowered to runtime memory, + // so no byref classification makes sense. Caller has bigger + // problems if they reach here at codegen time. + case TypeKind::Invalid: + case TypeKind::Void: + case TypeKind::NoReturn: + case TypeKind::Type: + case TypeKind::Module: + return false; + // `c.Vec(i32)` and similar generic-call type expressions live in + // the TypePool as GenericCall keys; their *instantiated* struct + // TypeIdx is what carries runtime shape. Resolve and recurse so + // `var v: c.Vec(i32)` classifies as byref like the underlying + // `Vec__i32` does. + case TypeKind::GenericCall: { + TypeIdx resolved = ctx.resolveGenericCall(ty); + return resolved != kNoType && resolved != ty && isByRef(resolved, ctx); + } + // Scalars and pointer-shaped values: byval. The whole value + // rides in registers; SSA representation is the value itself. + case TypeKind::Bool: case TypeKind::Int: case TypeKind::Float: - case TypeKind::Bool: case TypeKind::PtrSingle: case TypeKind::PtrMany: + case TypeKind::Fn: + return false; + // Slice is a `{ ptr, len }` aggregate but we treat it as a + // packed scalar pair LLVM tolerates in registers — same as the + // current ABI path. Flipping this to byref would force a memory + // detour on every slice-typed local, which is a much larger + // blast radius than the codegen perf bug this work is fixing. + // Revisit if slice ergonomics warrant it later. + case TypeKind::Slice: + return false; + // Aggregates: byref. Every user-defined aggregate lives in memory + // at the JIR level and is referred to by pointer in SSA. Small + // structs still promote to registers via LLVM's mem2reg + SROA at + // -O1+; we don't need to second-guess the optimizer. + case TypeKind::Array: + case TypeKind::Struct: + case TypeKind::Union: return true; - default: + // Unit-only enums (no payload variants) lower to a bare i8 tag + // and ride through SSA like any other scalar — keep them byval. + // Payloaded enums are {tag, payload} aggregates and follow the + // strict-byref rule. + case TypeKind::Enum: { + const auto *einfo = ctx.lookupEnum(ty); + return einfo != nullptr && einfo->hasPayloadVariant; + } + // `Named` is a deferred reference into the struct / enum / union + // registries. Resolve once and recurse on the canonical TypeIdx + // so type aliases and `Self` substitutions classify the same as + // what they point at. + case TypeKind::Named: { + const std::string &name = + ctx.getStringPool().get(static_cast(k.a)); + if (ctx.lookupStruct(ty) || ctx.lookupUnion(ty)) { return true; } + // Enum is byref only when it carries a payload — unit-only + // enums lower to a bare i8 tag and stay byval like scalars. + if (const auto *einfo = ctx.lookupEnum(ty)) { + return einfo->hasPayloadVariant; + } + TypeIdx alias = ctx.lookupTypeAlias(name); + if (alias != kNoType && alias != ty) return isByRef(alias, ctx); + // Unresolved Named with no struct/enum/union registry hit — + // shouldn't reach codegen, but treat as byval to keep the + // existing failure surface unchanged. return false; } + } + return false; } -} // namespace - +// Any byref aggregate (Struct / Array / Union / payloaded Enum) +// crosses the call boundary by pointer; everything else rides in +// registers as its LLVM-natural value. `mut` is the one exception — +// it always carries a pointer to caller-owned storage regardless of +// the underlying type's classification. +// +// Slice ({ptr, len}) and Fn (a single ptr) are explicit byval cases +// today: they're already register-shaped at LLVM-IR level. Adding a +// new byval aggregate that doesn't fit naturally in registers should +// flip its type-kind to byref in `isByRef`, not patch this code. ParamABI classifyParam(ParamMode mode, TypeIdx ty, const JamCodegenContext &ctx) { - // `mut` always carries a pointer to caller-owned storage; size - // irrelevant. - if (mode == ParamMode::Mut) { - return ParamABI{ParamABI::Kind::ByPointer, nullptr, - static_cast(ctx.typeAlign(ty))}; - } - - // `let` and `move` are by-value at the call boundary. Whether the - // value travels in a register / scalar pair (LLVM-level "ByValue") - // or via a pointer to caller-owned storage depends on the type's - // size. Move differs from let only on the caller side (the source - // binding's tracked init state); the callee sees the same bytes. - const TypeKey &k = ctx.getTypePool().get(ty); - if (isScalar(k.kind)) { - return ParamABI{ParamABI::Kind::ByValue, ctx.getLLVMType(ty), 0}; - } - - uint64_t size = ctx.typeSize(ty); - if (size > kByValueMaxBytes) { + if (mode == ParamMode::Mut || isByRef(ty, ctx)) { return ParamABI{ParamABI::Kind::ByPointer, nullptr, static_cast(ctx.typeAlign(ty))}; } @@ -61,19 +111,10 @@ ParamABI classifyParam(ParamMode mode, TypeIdx ty, } ReturnABI classifyReturn(TypeIdx ty, const JamCodegenContext &ctx) { - // Void / unspecified return — represented as kNoType and lowered to - // LLVM `void`. Treat as Direct so codegen uses BuildRetVoid. if (ty == kNoType) { return ReturnABI{ReturnABI::Kind::Direct, ctx.getVoidType(), 0}; } - - const TypeKey &k = ctx.getTypePool().get(ty); - if (isScalar(k.kind)) { - return ReturnABI{ReturnABI::Kind::Direct, ctx.getLLVMType(ty), 0}; - } - - uint64_t size = ctx.typeSize(ty); - if (size > kByValueMaxBytes) { + if (isByRef(ty, ctx)) { return ReturnABI{ReturnABI::Kind::Indirect, nullptr, static_cast(ctx.typeAlign(ty))}; } diff --git a/src/abi.h b/src/abi.h index 3e5100c..3c62659 100644 --- a/src/abi.h +++ b/src/abi.h @@ -56,29 +56,40 @@ struct ReturnABI { uint32_t sretAlign; // Indirect: pointee alignment in bytes. }; +// Is this type carried by-reference at the codegen level? +// +// Byref: arrays / structs / non-packed unions / payloaded enums — +// anything whose natural runtime form lives in memory and whose +// JIR-level value is a pointer to that storage. This is *codegen* +// shape, distinct from the source-level ABI / param-mode +// classification — even a small 2-field struct returns true here. +// +// Not byref: scalars (Int / Float / Bool), Pointer types (PtrSingle +// / PtrMany / Slice — Slice is a 2-field aggregate but is treated +// as a packed { ptr, len } pair that LLVM passes in registers), Fn +// pointers, unit-only enums (lower to a bare i8 tag). +// +// Single source of truth: codegen reroutings (StructLit / ArrayLit / +// Store / Load / FieldAccess) and `classifyParam` / `classifyReturn` +// all dispatch on this. +bool isByRef(TypeIdx ty, const JamCodegenContext &ctx); + // Classify a parameter (mode, type) pair. Pure function of its inputs; -// safe to call any number of times. See docs/ABI.md §4. +// safe to call any number of times. // -// mut -> always ByPointer -// let / move, scalar T -> ByValue -// let / move, aggregate -// size <= kByValueMaxBytes -> ByValue (LLVM handles register packing) -// size > kByValueMaxBytes -> ByPointer +// mut -> ByPointer (always — caller storage) +// isByRef(T) -> ByPointer (byref aggregate) +// anything else -> ByValue (rides in registers) ParamABI classifyParam(ParamMode mode, TypeIdx ty, const JamCodegenContext &ctx); -// Classify a return type. See docs/ABI.md §4. +// Classify a return type. // -// scalar T -> Direct -// aggregate with size <= 16 B -> Direct (LLVM packs into return regs) -// aggregate with size > 16 B -> Indirect (sret) +// void / kNoType -> Direct void +// isByRef(T) -> Indirect (sret) +// anything else -> Direct (rides in return regs) ReturnABI classifyReturn(TypeIdx ty, const JamCodegenContext &ctx); -// Threshold above which an owned aggregate is passed/returned by pointer -// rather than by value. Matches the System V AMD64 ABI's two-eightbyte -// MEMORY classification and Rust's observed behavior at -C opt-level=0. -constexpr uint64_t kByValueMaxBytes = 16; - } // namespace abi } // namespace jam diff --git a/src/astgen.cpp b/src/astgen.cpp index aaeadb2..b089a18 100644 --- a/src/astgen.cpp +++ b/src/astgen.cpp @@ -320,7 +320,33 @@ static void emitCondBr(AstGenCtx &gctx, JirRef cond, JirBlockRef thenB, JirBlockRef elseB); static void emitDrops(AstGenCtx &gctx, const std::vector &bindings); static JirRef emitCall(AstGenCtx &gctx, const FunctionAST *fn, - const std::vector &argRefs); + const std::vector &argRefs, + JirRef destPtr = kNoJirRef); +// Forward decls for the Call / TypeMethodCall paths so the +// astgenExprIntoPtr place-into-destination helper can drive them +// before their definitions appear later in this file. +static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n, + JirRef destPtr = kNoJirRef); +static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n, + JirRef destPtr = kNoJirRef); +// Forward decl: place-into-destination dispatcher used by VarDecl, +// Return, and StructLitInto. See its definition for which expression +// shapes it recognises (StructLit, sret Calls). +static bool astgenExprIntoPtr(AstGenCtx &gctx, NodeIdx exprIdx, + TypeIdx expectedTy, JirRef destPtr); + +// Emit a `JirTag::SretArg` referencing this function's hidden sret +// pointer. The pointer's pointee type is the function's return type. +// Used as the result_ptr for byref returns so a `return X` writes X +// straight into the caller-owned slot. +static JirRef emitSretArg(AstGenCtx &gctx, TypeIdx retTy) { + TypeIdx ptrTy = gctx.ctx.getTypePool().intern( + TypeKey{TypeKind::PtrSingle, 0, 0, retTy, 0}); + JirInst s{}; + s.tag = JirTag::SretArg; + s.ty = ptrTy; + return emit(gctx, s); +} // `v[i]` desugar dispatch — see `emitStructCfnDispatch` for the body. // Forward-declared so astgenAssign (in this file, above the // definition) can call it for `v[i] = x` -> setAt routing. @@ -474,8 +500,27 @@ static JirRef astgenBoolLit(AstGenCtx &gctx, const AstNode &n) { static void astgenReturn(AstGenCtx &gctx, const AstNode &n) { JirRef valRef = kNoJirRef; if (n.lhs != 0) { - valRef = - astgenExpr(gctx, static_cast(n.lhs), gctx.jfn.returnType); + NodeIdx valIdx = static_cast(n.lhs); + // When the function returns via sret, hand the sret slot + // pointer to the value expression as a result_ptr. Byref- + // producing exprs (StructLit / ArrayLit / sret Call) write + // fields directly into the slot and return kNoJirRef — no + // temp alloca, no trailing memcpy. Byval returns still flow + // the value through and let Ret's standard store handle it. + TypeIdx retTy = gctx.jfn.returnType; + bool sretFn = retTy != kNoType && + jam::abi::classifyReturn(retTy, gctx.ctx).kind == + jam::abi::ReturnABI::Kind::Indirect; + if (sretFn && + astgenExprIntoPtr(gctx, valIdx, retTy, emitSretArg(gctx, retTy))) { + emitDropsThroughScope(gctx, 0); + JirInst ret{}; + ret.tag = JirTag::Ret; + ret.a = kNoJirRef; + emit(gctx, ret); + return; + } + valRef = astgenExpr(gctx, valIdx, retTy); } // Drop every active scope before exiting the function. emitDropsThroughScope(gctx, 0); @@ -629,98 +674,111 @@ static void astgenVarDecl(AstGenCtx &gctx, const AstNode &n) { gctx.localTypes[name] = type; if (!gctx.localScopes.empty()) { gctx.localScopes.back().insert(name); } - initRef = astgenExpr(gctx, initIdx, type); - - // Type-check the init against the declared type. The - // astgenNumberLit path already narrows integer literals to - // the declared int width when `expected` is an Int, so - // `var x: i32 = 5;` lands as ik=I32. Anything else that - // doesn't match is a real mismatch — including - // `var x: f32 = 3;` (int into float), `var x: bool = 1.0;` - // (float into bool), `var x: u8 = "s";` (slice into u8), - // etc. - // - // Both sides are resolved through generic-call / type-alias - // chains first so `var a: Identity(i32) = 42;` compares - // `i32 == i32` instead of the unresolved `GenericCall ≠ Int`. - // Pointer-target types are also resolved per-side so - // `var p: *const T = &x` (where `T` is an alias) matches. - std::function resolveForCmp = - [&](TypeIdx t) -> TypeIdx { - if (t == kNoType) return t; - const TypeKey &k = gctx.ctx.getTypePool().get(t); - // Generic substitution wins (inside an instantiated - // method body, `T` resolves to whatever the - // instantiation supplied). - if (k.kind == TypeKind::Named) { - const std::string &name = - gctx.ctx.getStringPool().get(static_cast(k.a)); - TypeIdx sub = gctx.ctx.lookupCurrentSubst(name); - if (sub != kNoType) return resolveForCmp(sub); - } - if (k.kind == TypeKind::GenericCall) { - TypeIdx r = gctx.ctx.resolveGenericCall(t); - if (r != kNoType) return resolveForCmp(r); - } - if (k.kind == TypeKind::Named) { - const std::string &nm = - gctx.ctx.getStringPool().get(static_cast(k.a)); - TypeIdx a = gctx.ctx.lookupTypeAlias(nm); - if (a != kNoType) return resolveForCmp(a); - // 3+ segment chain through module re-exports — collapse - // `w.leaf.Point` to the canonical `Point` so it matches - // values produced by `w.leaf.makePoint(...)` whose return - // type was registered with the single-segment name. - if (nm.find('.') != std::string::npos) { - TypeIdx c = gctx.ctx.resolveChainedType(nm); - if (c != kNoType) return resolveForCmp(c); + // Try the place-into-destination path first: StructLit and + // sret Calls can write directly into `allocaRef`, skipping + // the SSA aggregate (and the insertvalue chain) the value- + // form would build. When placed, `initRef` stays kNoJirRef + // so the trailing Store below short-circuits. + if (astgenExprIntoPtr(gctx, initIdx, type, allocaRef)) { + initRef = kNoJirRef; + } else { + initRef = astgenExpr(gctx, initIdx, type); + + // Type-check the init against the declared type. The + // astgenNumberLit path already narrows integer literals to + // the declared int width when `expected` is an Int, so + // `var x: i32 = 5;` lands as ik=I32. Anything else that + // doesn't match is a real mismatch — including + // `var x: f32 = 3;` (int into float), `var x: bool = 1.0;` + // (float into bool), `var x: u8 = "s";` (slice into u8), + // etc. + // + // Both sides are resolved through generic-call / type-alias + // chains first so `var a: Identity(i32) = 42;` compares + // `i32 == i32` instead of the unresolved `GenericCall ≠ Int`. + // Pointer-target types are also resolved per-side so + // `var p: *const T = &x` (where `T` is an alias) matches. + std::function resolveForCmp = + [&](TypeIdx t) -> TypeIdx { + if (t == kNoType) return t; + const TypeKey &k = gctx.ctx.getTypePool().get(t); + // Generic substitution wins (inside an instantiated + // method body, `T` resolves to whatever the + // instantiation supplied). + if (k.kind == TypeKind::Named) { + const std::string &name = gctx.ctx.getStringPool().get( + static_cast(k.a)); + TypeIdx sub = gctx.ctx.lookupCurrentSubst(name); + if (sub != kNoType) return resolveForCmp(sub); } + if (k.kind == TypeKind::GenericCall) { + TypeIdx r = gctx.ctx.resolveGenericCall(t); + if (r != kNoType) return resolveForCmp(r); + } + if (k.kind == TypeKind::Named) { + const std::string &nm = gctx.ctx.getStringPool().get( + static_cast(k.a)); + TypeIdx a = gctx.ctx.lookupTypeAlias(nm); + if (a != kNoType) return resolveForCmp(a); + // 3+ segment chain through module re-exports — collapse + // `w.leaf.Point` to the canonical `Point` so it matches + // values produced by `w.leaf.makePoint(...)` whose return + // type was registered with the single-segment name. + if (nm.find('.') != std::string::npos) { + TypeIdx c = gctx.ctx.resolveChainedType(nm); + if (c != kNoType) return resolveForCmp(c); + } + } + return t; + }; + TypeIdx initTy = gctx.jfn.getInst(initRef).ty; + TypeIdx declRes = resolveForCmp(type); + TypeIdx initRes = resolveForCmp(initTy); + // Pointer-shape leniency: PtrSingle(T) and PtrMany(T) share + // the runtime representation (a plain `ptr`), so `&arr[i]` + // (which lowers to PtrSingle(T)) is accepted in a PtrMany(T) + // slot as a zero-cost retag. The rule is permissive because + // we don't currently distinguish pointer-to-array from + // pointer-to-element at the JIR level — revisit when an + // Array-pointer kind lands and the source can carry its + // length statically. + auto pointerCompatible = [&](TypeIdx a, TypeIdx b) -> bool { + if (a == kNoType || b == kNoType) return false; + const TypeKey &ka = gctx.ctx.getTypePool().get(a); + const TypeKey &kb = gctx.ctx.getTypePool().get(b); + bool aPtr = ka.kind == TypeKind::PtrSingle || + ka.kind == TypeKind::PtrMany; + bool bPtr = kb.kind == TypeKind::PtrSingle || + kb.kind == TypeKind::PtrMany; + return aPtr && bPtr && ka.a == kb.a; + }; + bool typesMatch = + (declRes == initRes) || pointerCompatible(declRes, initRes); + if (initTy != kNoType && !typesMatch) { + const TypeKey &dk = gctx.ctx.getTypePool().get(declRes); + const TypeKey &ik = gctx.ctx.getTypePool().get(initRes); + // Specialize the message for the two patterns users hit + // most often; everything else gets the generic mismatch. + if (dk.kind == TypeKind::Float && ik.kind == TypeKind::Int) { + failHere(gctx, "cannot assign integer to float-typed `" + + name + + "`; use a float literal (e.g. `3.0`) " + "or an explicit `as` cast"); + } + failHere(gctx, + "type mismatch in `" + name + + "`: declared and initialised values disagree"); } - return t; - }; - TypeIdx initTy = gctx.jfn.getInst(initRef).ty; - TypeIdx declRes = resolveForCmp(type); - TypeIdx initRes = resolveForCmp(initTy); - // Pointer-shape leniency: PtrSingle(T) and PtrMany(T) share - // the runtime representation (a plain `ptr`), so `&arr[i]` - // (which lowers to PtrSingle(T)) is accepted in a PtrMany(T) - // slot as a zero-cost retag. The rule is permissive because - // we don't currently distinguish pointer-to-array from - // pointer-to-element at the JIR level — revisit when an - // Array-pointer kind lands and the source can carry its - // length statically. - auto pointerCompatible = [&](TypeIdx a, TypeIdx b) -> bool { - if (a == kNoType || b == kNoType) return false; - const TypeKey &ka = gctx.ctx.getTypePool().get(a); - const TypeKey &kb = gctx.ctx.getTypePool().get(b); - bool aPtr = - ka.kind == TypeKind::PtrSingle || ka.kind == TypeKind::PtrMany; - bool bPtr = - kb.kind == TypeKind::PtrSingle || kb.kind == TypeKind::PtrMany; - return aPtr && bPtr && ka.a == kb.a; - }; - bool typesMatch = - (declRes == initRes) || pointerCompatible(declRes, initRes); - if (initTy != kNoType && !typesMatch) { - const TypeKey &dk = gctx.ctx.getTypePool().get(declRes); - const TypeKey &ik = gctx.ctx.getTypePool().get(initRes); - // Specialize the message for the two patterns users hit - // most often; everything else gets the generic mismatch. - if (dk.kind == TypeKind::Float && ik.kind == TypeKind::Int) { - failHere(gctx, "cannot assign integer to float-typed `" + name + - "`; use a float literal (e.g. `3.0`) " - "or an explicit `as` cast"); - } - failHere(gctx, "type mismatch in `" + name + - "`: declared and initialised values disagree"); - } + } // close: place-into-destination else } - JirInst store{}; - store.tag = JirTag::Store; - store.a = allocaRef; - store.b = initRef; - emit(gctx, store); + if (initRef != kNoJirRef) { + JirInst store{}; + store.tag = JirTag::Store; + store.a = allocaRef; + store.b = initRef; + emit(gctx, store); + } // If this binding's type has a registered drop fn, track it on // the top of the drop-scope stack so the next scope-exit (or @@ -1081,6 +1139,211 @@ static JirRef astgenStructLit(AstGenCtx &gctx, const AstNode &n, return emit(gctx, inst); } +// Write a struct literal directly into `destPtr` (a pointer to the +// struct slot) instead of building an SSA aggregate via an +// InsertValue chain. For each field we emit FieldAddr → recursive +// place into the field's slot (StructLit / sret Call) or a value +// compile + Store fallback for everything else. +// +// Why this matters for build perf: when an outer `var b: Bus = Bus +// { ..., cdrom: Cdrom.init() }` would otherwise build %Bus via 22 +// InsertValue, after O3 inlining + SROA every multi-byte field inside +// the struct gets decomposed into per-byte InsertValue (a [2352]u8 +// field expands into 2352 of them). AArch64 ISel's DAGCombine then +// goes quadratic on the resulting store chain. Routing through +// per-field stores into the destination keeps those large aggregates +// in memory rather than in SSA registers. +static void astgenStructLitInto(AstGenCtx &gctx, const AstNode &n, + TypeIdx expectedTy, JirRef destPtr) { + const NodeStore &ns = gctx.ctx.getNodeStore(); + TypeIdx ty = static_cast(n.lhs); + if (ty == kNoType) ty = expectedTy; + if (ty == kNoType) { + failHere(gctx, "astgen: struct literal without target type"); + } + + // Unions and non-struct types: fall back to value-form StructLit + // + Store. Unions are small enough that the perf pathology doesn't + // trigger, and they have specialized lowering already. + const auto *info = gctx.ctx.lookupStruct(ty); + if (info == nullptr || gctx.ctx.lookupUnion(ty) != nullptr) { + JirRef val = astgenStructLit(gctx, n, expectedTy); + JirInst store{}; + store.tag = JirTag::Store; + store.a = destPtr; + store.b = val; + emit(gctx, store); + return; + } + + ExtraIdx fieldsExtra = static_cast(n.rhs); + uint32_t fieldCount = ns.getExtra(fieldsExtra); + + // Map declared-field index -> source NodeIdx of the initializer. + std::vector exprByIdx(info->fields.size(), 0); + std::vector hasField(info->fields.size(), 0); + for (uint32_t i = 0; i < fieldCount; i++) { + StringIdx nameId = + static_cast(ns.getExtra(fieldsExtra + 1 + i * 2)); + NodeIdx exprIdx = + static_cast(ns.getExtra(fieldsExtra + 2 + i * 2)); + const std::string &fieldName = gctx.ctx.getStringPool().get(nameId); + int idx = gctx.ctx.getFieldIndex(info->name, fieldName); + if (idx < 0) { + appendErrorHere(gctx, "unknown struct field `" + fieldName + "`"); + continue; + } + exprByIdx[idx] = exprIdx; + hasField[idx] = 1; + } + + for (size_t i = 0; i < info->fields.size(); i++) { + if (!hasField[i]) { + failHere(gctx, "astgen: struct literal missing field `" + + info->fields[i].first + "`"); + } + TypeIdx expectedField = info->fields[i].second; + TypeIdx fieldPtrTy = gctx.ctx.getTypePool().intern( + TypeKey{TypeKind::PtrSingle, 0, 0, expectedField, 0}); + JirInst fieldAddr{}; + fieldAddr.tag = JirTag::FieldAddr; + fieldAddr.a = destPtr; + fieldAddr.b = static_cast(i); + fieldAddr.ty = fieldPtrTy; + JirRef fieldPtr = emit(gctx, fieldAddr); + + // Try the place path (recursive StructLit or sret Call with + // destPtr forwarded). Fall back to value-compile + Store for + // anything else. + if (!astgenExprIntoPtr(gctx, exprByIdx[i], expectedField, fieldPtr)) { + JirRef val = astgenExpr(gctx, exprByIdx[i], expectedField); + // Silent int->float widening — mirrors astgenStructLit. + TypeIdx vt = gctx.jfn.getInst(val).ty; + if (vt != expectedField && vt != kNoType) { + const TypeKey &fk = gctx.ctx.getTypePool().get(expectedField); + const TypeKey &vk = gctx.ctx.getTypePool().get(vt); + if (fk.kind == TypeKind::Float && vk.kind == TypeKind::Int) { + JirInst c{}; + c.tag = vk.b != 0 ? JirTag::SIToFP : JirTag::UIToFP; + c.a = val; + c.ty = expectedField; + val = emit(gctx, c); + } + } + JirInst store{}; + store.tag = JirTag::Store; + store.a = fieldPtr; + store.b = val; + emit(gctx, store); + } + } +} + +// Try to lower an expression directly into `destPtr`. Returns true on +// success — when the destination has been written and no SSA value +// remains for the caller to bind. Returns false to let the caller +// take the standard astgenExpr + Store path. +// +// Recognises: +// - StructLit -> per-field FieldAddr + Store into destPtr +// - ArrayLit / ArrayRepeat -> per-element IndexAddr + Store into destPtr +// - Call / TypeMethodCall -> forward destPtr as the sret slot so +// the callee writes its result through +// the caller's storage (no temp alloca) +// +// Everything else (scalars, ptr-typed values, small aggregates that +// the ABI passes by value) returns false. The pathology this helper +// dodges only kicks in for large-aggregate stores anyway. +static bool astgenExprIntoPtr(AstGenCtx &gctx, NodeIdx exprIdx, + TypeIdx expectedTy, JirRef destPtr) { + const NodeStore &ns = gctx.ctx.getNodeStore(); + const AstNode &n = ns.get(exprIdx); + switch (n.tag) { + case AstTag::StructLit: { + astgenStructLitInto(gctx, n, expectedTy, destPtr); + return true; + } + case AstTag::ArrayLit: { + // Per-element write into destPtr. Element type comes from + // the surrounding context or from compiling the first elem. + TypeIdx elemTy = static_cast(n.lhs); + if (elemTy == kNoType && expectedTy != kNoType) { + const TypeKey &ek = gctx.ctx.getTypePool().get(expectedTy); + if (ek.kind == TypeKind::Array) { + elemTy = static_cast(ek.a); + } + } + if (elemTy == kNoType) return false; + ExtraIdx elemsExtra = static_cast(n.rhs); + uint32_t count = ns.getExtra(elemsExtra); + TypeIdx u64Ty = BuiltinType::U64; + TypeIdx elemPtrTy = gctx.ctx.getTypePool().intern( + TypeKey{TypeKind::PtrSingle, 0, 0, elemTy, 0}); + for (uint32_t i = 0; i < count; i++) { + NodeIdx eIdx = + static_cast(ns.getExtra(elemsExtra + 1 + i)); + JirInst idxInst{}; + idxInst.tag = JirTag::Int; + idxInst.a = i; + idxInst.ty = u64Ty; + JirRef idxRef = emit(gctx, idxInst); + JirInst ia{}; + ia.tag = JirTag::IndexAddr; + ia.a = destPtr; + ia.b = idxRef; + ia.ty = elemPtrTy; + JirRef ep = emit(gctx, ia); + if (!astgenExprIntoPtr(gctx, eIdx, elemTy, ep)) { + JirRef ev = astgenExpr(gctx, eIdx, elemTy); + JirInst st{}; + st.tag = JirTag::Store; + st.a = ep; + st.b = ev; + emit(gctx, st); + } + } + return true; + } + // ArrayRepeat (`[expr; N]`) deliberately falls through to the + // astgenExpr + Store path. Unrolling N stores at astgen time + // bloats JIR (`[0; 2352]` becomes 2352 IndexAddr+Store JIR + // instructions); JirTag::ArrayLit's byref codegen already emits + // the same per-element stores at LLVM-IR time but keeps the JIR + // itself compact. The trailing memcpy that the value-path leaves + // behind is one extra instruction LLVM cleans up — much cheaper + // than inflating astgen's per-fn IR-build cost. + case AstTag::Call: { + JirRef r = astgenCall(gctx, n, destPtr); + // kNoJirRef -> emitCall used destPtr as the sret slot and the + // result has been written in-place. A real ref -> the call + // had a Direct (ByValue) return, so we finish the place by + // storing the produced SSA value. Either way the field has + // been written; signal handled. + if (r != kNoJirRef) { + JirInst store{}; + store.tag = JirTag::Store; + store.a = destPtr; + store.b = r; + emit(gctx, store); + } + return true; + } + case AstTag::TypeMethodCall: { + JirRef r = astgenTypeMethodCall(gctx, n, destPtr); + if (r != kNoJirRef) { + JirInst store{}; + store.tag = JirTag::Store; + store.a = destPtr; + store.b = r; + emit(gctx, store); + } + return true; + } + default: + return false; + } +} + // AstGen for `MemberAccess`. Two cases: // 1. `EnumName.Variant` — when `base` is a Variable whose source-level // name happens to be a registered enum, lower to a unit-variant @@ -3031,7 +3294,8 @@ static JirRef astgenMatch(AstGenCtx &gctx, const AstNode &n, TypeIdx expected) { // into astgenCall. Instantiation runs the full astgen -> jirDefineBody // pipeline in `JamCodegenContext::instantiateStructExpr`, so the JIR // Call here resolves cleanly by LLVM name at codegen time. -static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n) { +static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n, + JirRef destPtr) { const NodeStore &nsConst = gctx.ctx.getNodeStore(); NodeStore &ns = const_cast(nsConst); TypeIdx recvTy = static_cast(n.lhs); @@ -3173,7 +3437,7 @@ static JirRef astgenTypeMethodCall(AstGenCtx &gctx, const AstNode &n) { TypeIdx expectArg = (i < fn->Args.size()) ? fn->Args[i].Type : kNoType; argRefs.push_back(astgenExpr(gctx, argIdx, expectArg)); } - return emitCall(gctx, fn, argRefs); + return emitCall(gctx, fn, argRefs, destPtr); } // AstGen for `AtCall` — comptime intrinsic (`@sizeOf(T)` / `@alignOf(T)`). @@ -3402,24 +3666,54 @@ static JirRef lowerArg(AstGenCtx &gctx, NodeIdx argIdx, const Param &p) { // let/move-by-ptr through the lvalue path saves the otherwise- // dead value-load + spill at every call site. const AstNode &argNode = gctx.ctx.getNodeStore().get(argIdx); + const NodeStore &ns = gctx.ctx.getNodeStore(); TypeIdx leafTy = kNoType; + // `Color.Red` and friends look syntactically like a MemberAccess + // but are enum-variant constructors that astgenLvalue can't + // resolve — its `Color` lookup falls through every lvalue path. + // Detect the shape upfront and route through the value+spill + // fallback below. Same for any MemberAccess on a base Variable + // whose name resolves to an enum/struct/module, not a local. + auto memberAccessOnNonLvalue = [&](const AstNode &n) -> bool { + if (n.tag != AstTag::MemberAccess) return false; + NodeIdx baseIdx = static_cast(n.lhs); + const AstNode &baseNode = ns.get(baseIdx); + if (baseNode.tag != AstTag::Variable) return false; + const std::string &name = + gctx.ctx.getStringPool().get(static_cast(baseNode.lhs)); + if (gctx.locals.find(name) != gctx.locals.end()) return false; + // Not a local — likely an enum / module / type name. + // Treat as non-lvalueable so the spill path takes over. + return gctx.ctx.getEnum(name) != nullptr || + gctx.ctx.getStruct(name) != nullptr || + gctx.ctx.getImportHandle(name) != nullptr; + }; switch (argNode.tag) { case AstTag::Variable: - case AstTag::MemberAccess: case AstTag::Index: case AstTag::Deref: return astgenLvalue(gctx, argIdx, leafTy); + case AstTag::MemberAccess: + if (!memberAccessOnNonLvalue(argNode)) { + return astgenLvalue(gctx, argIdx, leafTy); + } + break; // fall through to spill path case AstTag::AddressOf: return astgenExpr(gctx, argIdx, kNoType); default: break; } - JirRef val = astgenExpr(gctx, argIdx, p.Type); - leafTy = gctx.jfn.getInst(val).ty; + // Spill: hold an rvalue in a fresh stack slot so the callee can + // receive a pointer. Try the place path first — byref producers + // (StructLit / ArrayLit / sret Call) write directly into the + // alloca, skipping the temp + memcpy two-hop. Fall back to value + // compile + Store for byval rvalues. JirInst alloca{}; alloca.tag = JirTag::Alloca; - alloca.ty = leafTy; + alloca.ty = p.Type; JirRef ptr = emitAllocaHoisted(gctx, alloca); + if (astgenExprIntoPtr(gctx, argIdx, p.Type, ptr)) { return ptr; } + JirRef val = astgenExpr(gctx, argIdx, p.Type); JirInst store{}; store.tag = JirTag::Store; store.a = ptr; @@ -3429,14 +3723,40 @@ static JirRef lowerArg(AstGenCtx &gctx, NodeIdx argIdx, const Param &p) { } static JirRef emitCall(AstGenCtx &gctx, const FunctionAST *fn, - const std::vector &argRefs) { + const std::vector &argRefs, JirRef destPtr) { std::string mangled = mangledFunctionName(*fn, gctx.ctx.getTypePool(), gctx.ctx.getStringPool()); StringIdx calleeId = gctx.ctx.getStringPool().intern(mangled); + // Unified Call shape: sret-returning callees get their result slot + // as a leading arg, the same way LLVM expects it. astgen owns the + // slot — either the caller's destPtr (when known) or a fresh + // alloca — so jir_codegen no longer allocates anything at call + // time. The Call's JirRef value is the slot pointer when sret; + // downstream `isByRef`-aware code paths treat it as the byref + // result. When destPtr was supplied, the JirRef is kNoJirRef so + // the caller can detect that the result was placed in-line. + bool sretCallee = fn->ReturnType != kNoType && + jam::abi::classifyReturn(fn->ReturnType, gctx.ctx).kind == + jam::abi::ReturnABI::Kind::Indirect; + JirRef sretSlot = kNoJirRef; + std::vector allArgs; + allArgs.reserve(argRefs.size() + (sretCallee ? 1u : 0u)); + if (sretCallee) { + if (destPtr != kNoJirRef) { + sretSlot = destPtr; + } else { + JirInst a{}; + a.tag = JirTag::Alloca; + a.ty = fn->ReturnType; + sretSlot = emitAllocaHoisted(gctx, a); + } + allArgs.push_back(sretSlot); + } + for (JirRef r : argRefs) allArgs.push_back(r); std::vector packed; - packed.reserve(1 + argRefs.size()); - packed.push_back(static_cast(argRefs.size())); - for (JirRef r : argRefs) packed.push_back(r); + packed.reserve(1 + allArgs.size()); + packed.push_back(static_cast(allArgs.size())); + for (JirRef r : allArgs) packed.push_back(r); JirExtraIdx extra = gctx.jfn.pushExtra(packed.data(), packed.size()); JirInst call{}; call.tag = JirTag::Call; @@ -3447,12 +3767,20 @@ static JirRef emitCall(AstGenCtx &gctx, const FunctionAST *fn, // Calling a `noreturn` function diverges. Terminate the current // block with Unreachable so downstream code is dead and the JIR // is well-formed (every reachable block ends in a terminator). - // Mirrors what the legacy codegen did via LLVM's `unreachable`. if (fn->ReturnType == BuiltinType::NoReturn) { JirInst u{}; u.tag = JirTag::Unreachable; emit(gctx, u); } + if (sretCallee) { + // Place-call: the value was written through destPtr, so the + // caller shouldn't bind any JirRef. + if (destPtr != kNoJirRef) { return kNoJirRef; } + // Otherwise the slot ptr we just allocated *is* the call + // result — codegen returns it from the Call JirRef so byref + // consumers see a pointer to the freshly-written aggregate. + return callRef; + } return callRef; } @@ -4047,7 +4375,7 @@ static JirRef astgenCompInstantiatedCall(AstGenCtx &gctx, const AstNode &n, return emitCall(gctx, clone, argRefs); } -static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { +static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n, JirRef destPtr) { const NodeStore &ns = gctx.ctx.getNodeStore(); ExtraIdx argsExtra = static_cast(n.rhs); uint32_t argCount = ns.getExtra(argsExtra); @@ -4185,14 +4513,23 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { } ParamMode mode = method->Args.empty() ? ParamMode::Let : method->Args[0].Mode; + // Dispatch on the full ABI, not just ParamMode: a `let + // self: Self` where Self is byref now arrives ByPointer too, + // so the receiver needs to land as a pointer rather than the + // loaded aggregate value. + jam::abi::ParamABI recvAbi = + jam::abi::classifyParam(mode, recvTy, gctx.ctx); JirRef recvArg; - if (mode == ParamMode::Mut || mode == ParamMode::Move) { + if (recvAbi.kind == jam::abi::ParamABI::Kind::ByPointer) { if (recvLvaluePtr != kNoJirRef) { recvArg = recvLvaluePtr; } else { - // Non-lvalue receiver with a mut/move method: spill the - // already-computed value to a fresh alloca so the - // callee gets a stable storage address. + // Non-lvalue receiver with a by-pointer method: spill + // the already-computed value to a fresh alloca so the + // callee gets a stable storage address. For byref- + // typed rvalues `recvVal` is itself a pointer (from + // the new StructLit / Call codegen), so the spill is + // a memcpy via the Store JIR tag. JirInst alloca{}; alloca.tag = JirTag::Alloca; alloca.ty = recvTy; @@ -4208,10 +4545,6 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { if (recvVal != kNoJirRef) { recvArg = recvVal; } else { - // Lvalue receiver + by-value method: Load through the - // ptr now. This is the only spot where a full-aggregate - // load is unavoidable; the previous design emitted it - // unconditionally even when the method took a pointer. JirInst load{}; load.tag = JirTag::Load; load.a = recvLvaluePtr; @@ -4230,7 +4563,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { : kNoType; argRefs.push_back(astgenExpr(gctx, argIdx, expectArg)); } - return emitCall(gctx, method, argRefs); + return emitCall(gctx, method, argRefs, destPtr); } StringIdx calleeId = static_cast(n.lhs); @@ -4259,7 +4592,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { argRefs.push_back(astgenExpr(gctx, argIdx, kNoType)); } } - return emitCall(gctx, method, argRefs); + return emitCall(gctx, method, argRefs, destPtr); } } @@ -4426,7 +4759,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { } // `emitCall` mangles drop and test names — no need // to special-case here. - return emitCall(gctx, method, argRefs); + return emitCall(gctx, method, argRefs, destPtr); } // Prefix resolved to a real struct / enum but the // method doesn't exist. The common trigger is a @@ -4482,11 +4815,16 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { const FunctionAST *method = gctx.ctx.getFunctionAST(qualified); if (method != nullptr && !method->Args.empty()) { - // Build the receiver as either &inst (mut/move) - // or a Load of inst (let/const). + // Dispatch on the full ABI — a `let self: + // Self` where Self is byref arrives ByPointer, + // so the receiver must land as a pointer to + // the local's storage, not a loaded value. ParamMode mode = method->Args[0].Mode; + jam::abi::ParamABI recvAbi = + jam::abi::classifyParam(mode, instTy, gctx.ctx); JirRef recvRef; - if (mode == ParamMode::Mut || mode == ParamMode::Move) { + if (recvAbi.kind == + jam::abi::ParamABI::Kind::ByPointer) { TypeIdx pointee = instTy; TypeIdx ptrTy = gctx.ctx.getTypePool().intern( TypeKey{TypeKind::PtrSingle, 0, 0, pointee, 0}); @@ -4514,7 +4852,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { argRefs.push_back( astgenExpr(gctx, argIdx, expectArg)); } - return emitCall(gctx, method, argRefs); + return emitCall(gctx, method, argRefs, destPtr); } } } @@ -4661,7 +4999,7 @@ static JirRef astgenCall(AstGenCtx &gctx, const AstNode &n) { } } (void)calleeId; - return emitCall(gctx, fn, argRefs); + return emitCall(gctx, fn, argRefs, destPtr); } static JirRef astgenExpr(AstGenCtx &gctx, NodeIdx node, TypeIdx expected, @@ -4996,8 +5334,8 @@ void astgenBodyInto(JirFunction &jfn, const FunctionAST &fn, // the alloca as the local so reads emit Load(alloca) like any // ordinary local. // - ByPointer: param arrives as a pointer to caller-owned - // storage (mut / move always; let / const for aggregates > - // kByValueMaxBytes). Register the param JirRef directly as + // storage (mut / move always; let / const for any byref + // aggregate). Register the param JirRef directly as // the local's "alloca" so reads, writes, and field access all // operate on that pointer. JIR flag bit 0 on Param tells // jir_codegen + FieldAddr/IndexAddr to treat the value as a diff --git a/src/jam_llvm.cpp b/src/jam_llvm.cpp index bca3927..01f392b 100644 --- a/src/jam_llvm.cpp +++ b/src/jam_llvm.cpp @@ -775,6 +775,21 @@ JamValueRef JamLLVMBuildUnreachable(JamBuilderRef builder) { return WRAP_VALUE(UNWRAP_BUILDER(builder)->CreateUnreachable()); } +// Emit `@llvm.memcpy.p0.p0.i64(dst, src, size, isVolatile=false)`. +// Used by jir_codegen's Ret handler to convert load-aggregate + +// store-aggregate into a single memcpy when the return value comes +// from a known stack slot. The load+store form would force SROA to +// decompose every multi-byte field into per-byte insertvalues into +// an SSA aggregate — the exact pattern that makes AArch64 DAGCombine +// go quadratic on large sret returns. +JAM_EXTERN_C void JamLLVMBuildMemCpy(JamBuilderRef builder, JamValueRef dst, + uint64_t dstAlign, JamValueRef src, + uint64_t srcAlign, uint64_t size) { + UNWRAP_BUILDER(builder)->CreateMemCpy( + UNWRAP_VALUE(dst), llvm::MaybeAlign(dstAlign), UNWRAP_VALUE(src), + llvm::MaybeAlign(srcAlign), size); +} + JamValueRef JamLLVMBuildCall(JamBuilderRef builder, JamFunctionRef func, JamValueRef *args, unsigned numArgs, const char *name) { diff --git a/src/jam_llvm.h b/src/jam_llvm.h index 4c9a551..00d4cd2 100644 --- a/src/jam_llvm.h +++ b/src/jam_llvm.h @@ -247,6 +247,13 @@ JAM_EXTERN_C JamValueRef JamLLVMBuildLoad(JamBuilderRef builder, const char *name); JAM_EXTERN_C JamValueRef JamLLVMBuildStore(JamBuilderRef builder, JamValueRef val, JamValueRef ptr); +// `@llvm.memcpy.p0.p0.i64(dst, src, size, false)` — bulk byte copy +// between two memory regions. Used to replace the load+store of large +// aggregates that SROA would otherwise decompose into giant +// insertvalue chains in the sret return path. +JAM_EXTERN_C void JamLLVMBuildMemCpy(JamBuilderRef builder, JamValueRef dst, + uint64_t dstAlign, JamValueRef src, + uint64_t srcAlign, uint64_t size); // In-bounds GEP for indexing into a fixed-size array: gep [N x T], ptr, 0, idx. // `arrayType` must be the array aggregate type that `ptr` points to. JAM_EXTERN_C JamValueRef JamLLVMBuildArrayGEP(JamBuilderRef builder, diff --git a/src/jir.h b/src/jir.h index 3518779..8c615a8 100644 --- a/src/jir.h +++ b/src/jir.h @@ -162,6 +162,13 @@ enum class JirTag : uint8_t { // Call: `a` = StringIdx (callee qualified name); // `b` = ExtraIdx -> [argCount, arg0, arg1, ...] // `ty` = return type (kNoType for void). + // + // sret-returning callees: astgen prepends the result slot as the + // leading arg (arg0). The slot may be a caller-owned destination + // from the result_ptr plumbing (zero-copy place-into-destination) + // or a fresh Alloca emitted by `emitCall`. jir_codegen passes the + // args through unchanged and surfaces `arg0` as the Call's value + // for byref consumers. Call, // CallIndirect: `a` = JirRef of a fn-typed value (the function // pointer); `b` = ExtraIdx -> [argCount, arg0, ...]; @@ -184,6 +191,13 @@ enum class JirTag : uint8_t { // Function parameter access // Param: `a` = parameter index; `ty` = param type. Param, + // SretArg: evaluates to the function's hidden sret pointer (param + // 0 when `jirReturnIsSret` holds). Used by `astgenReturn` to + // thread the return slot through as a result_ptr so a `return + // StructLit {...}` writes its fields directly into the caller- + // owned slot — no temp alloca + memcpy hop. `ty` is the sret + // pointee (the function's return type). + SretArg, // Aggregates // StructLit: `b` = ExtraIdx -> [fieldCount, field0_val, field1_val, diff --git a/src/jir_codegen.cpp b/src/jir_codegen.cpp index 8987a0e..7853920 100644 --- a/src/jir_codegen.cpp +++ b/src/jir_codegen.cpp @@ -138,9 +138,9 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { } case JirTag::Param: { // flags & 1 means the param is ByPointer (mut / move always; - // let / const for aggregates > kByValueMaxBytes). We hand back - // the LLVM pointer directly so the local map can use it as - // the alloca-equivalent for field access / self-style reads. + // let / const for any byref aggregate). We hand back the LLVM + // pointer directly so the local map can use it as the alloca- + // equivalent for field access / self-style reads. // When the function uses sret, the source-level param at JIR // index `i` lives at LLVM index `i + 1` (the sret slot owns // LLVM index 0). @@ -157,12 +157,33 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { } case JirTag::Load: { JamValueRef ptr = emitInst(lctx, inst.a); + // Byref aggregates live in memory at the JIR layer — the + // "value" of a byref JirRef is the storage pointer itself. + // Skip the `load %T, ptr` here so we don't materialize the + // aggregate in SSA (the path that drove DAGCombine quadratic + // on large structs). Downstream Store / Ret / arg-passing + // recognises a byref-typed JirRef as a pointer and emits + // memcpy / pointer-forward. + if (jam::abi::isByRef(inst.ty, lctx.ctx)) { return ptr; } JamTypeRef ty = lctx.ctx.getLLVMType(inst.ty); return JamLLVMBuildLoad(lctx.ctx.getBuilder(), ty, ptr, "load"); } case JirTag::Store: { JamValueRef ptr = emitInst(lctx, inst.a); JamValueRef val = emitInst(lctx, inst.b); + // Byref store: `val` is a pointer to a byref aggregate in + // memory. Copy bytes into `ptr` instead of `store %T %v` — + // the latter forces SROA to decompose the aggregate value + // later and recreates the insertvalue chain we're trying to + // avoid. + TypeIdx valTy = lctx.jfn.getInst(inst.b).ty; + if (jam::abi::isByRef(valTy, lctx.ctx)) { + uint64_t size = lctx.ctx.typeSize(valTy); + uint64_t align = lctx.ctx.typeAlign(valTy); + JamLLVMBuildMemCpy(lctx.ctx.getBuilder(), ptr, align, val, align, + size); + return nullptr; + } JamLLVMBuildStore(lctx.ctx.getBuilder(), val, ptr); return nullptr; } @@ -407,6 +428,39 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { JamTypeRef ty = lctx.ctx.getLLVMType(inst.ty); JirExtraIdx extra = static_cast(inst.b); uint32_t count = lctx.jfn.getExtra(extra); + // Byref aggregate (struct / union): + // materialize in memory via an alloca and per-field stores. + // The JirRef *value* of the literal is the alloca pointer, + // not an SSA aggregate — downstream Load/Store/FieldAccess + // dispatch on `isByRef` and handle the pointer form. + // + // Allocas are hoisted to the entry block so mem2reg/SROA can + // promote them back to scalars when the bytes don't escape. + if (jam::abi::isByRef(inst.ty, lctx.ctx)) { + // `JamLLVMBuildAlloca` already hoists to the entry + // block (positioned at `entry.begin()`). No save/ + // restore dance needed here. + uint64_t align = lctx.ctx.typeAlign(inst.ty); + JamValueRef slot = + JamLLVMBuildAlloca(lctx.ctx.getBuilder(), ty, align, "lit"); + for (uint32_t i = 0; i < count; i++) { + JirRef fr = + static_cast(lctx.jfn.getExtra(extra + 1 + i)); + JamValueRef fv = emitInst(lctx, fr); + JamValueRef fp = JamLLVMBuildStructGEP(lctx.ctx.getBuilder(), + ty, slot, i, "lit.f"); + TypeIdx ft = lctx.jfn.getInst(fr).ty; + if (jam::abi::isByRef(ft, lctx.ctx)) { + uint64_t fsz = lctx.ctx.typeSize(ft); + uint64_t fal = lctx.ctx.typeAlign(ft); + JamLLVMBuildMemCpy(lctx.ctx.getBuilder(), fp, fal, fv, fal, + fsz); + } else { + JamLLVMBuildStore(lctx.ctx.getBuilder(), fv, fp); + } + } + return slot; + } JamValueRef agg = JamLLVMGetUndef(ty); for (uint32_t i = 0; i < count; i++) { JirRef fr = static_cast(lctx.jfn.getExtra(extra + 1 + i)); @@ -419,6 +473,21 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { case JirTag::FieldAccess: case JirTag::ExtractValue: { JamValueRef base = emitInst(lctx, inst.a); + // Byref base: `base` is a pointer to the aggregate in memory. + // GEP to the field; then return the field's pointer (when + // the field type is itself byref) or load the field value + // (when the field is byval / scalar). Slices etc. that aren't + // byref still go through the old extractvalue path. + TypeIdx baseTy = lctx.jfn.getInst(inst.a).ty; + if (jam::abi::isByRef(baseTy, lctx.ctx)) { + JamTypeRef structTy = lctx.ctx.getLLVMType(baseTy); + JamValueRef fp = JamLLVMBuildStructGEP( + lctx.ctx.getBuilder(), structTy, base, inst.b, "fa"); + if (jam::abi::isByRef(inst.ty, lctx.ctx)) { return fp; } + JamTypeRef fieldTy = lctx.ctx.getLLVMType(inst.ty); + return JamLLVMBuildLoad(lctx.ctx.getBuilder(), fieldTy, fp, + "field"); + } return JamLLVMBuildExtractValue(lctx.ctx.getBuilder(), base, inst.b, "field"); } @@ -426,6 +495,34 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { JamTypeRef ty = lctx.ctx.getLLVMType(inst.ty); JirExtraIdx extra = static_cast(inst.b); uint32_t count = lctx.jfn.getExtra(extra); + // Byref array: build in an entry-block alloca, store each + // element through an indexed GEP, return the storage pointer. + // Same shape as StructLit's byref path. + if (jam::abi::isByRef(inst.ty, lctx.ctx)) { + // JamLLVMBuildAlloca auto-hoists to entry.begin(). + uint64_t align = lctx.ctx.typeAlign(inst.ty); + JamValueRef slot = + JamLLVMBuildAlloca(lctx.ctx.getBuilder(), ty, align, "alit"); + JamTypeRef i64Ty = lctx.ctx.getInt64Type(); + for (uint32_t i = 0; i < count; i++) { + JirRef er = + static_cast(lctx.jfn.getExtra(extra + 1 + i)); + JamValueRef ev = emitInst(lctx, er); + JamValueRef idxVal = JamLLVMConstInt(i64Ty, i, false); + JamValueRef ep = JamLLVMBuildArrayGEP(lctx.ctx.getBuilder(), ty, + slot, idxVal, "alit.e"); + TypeIdx et = lctx.jfn.getInst(er).ty; + if (jam::abi::isByRef(et, lctx.ctx)) { + uint64_t esz = lctx.ctx.typeSize(et); + uint64_t eal = lctx.ctx.typeAlign(et); + JamLLVMBuildMemCpy(lctx.ctx.getBuilder(), ep, eal, ev, eal, + esz); + } else { + JamLLVMBuildStore(lctx.ctx.getBuilder(), ev, ep); + } + } + return slot; + } JamValueRef agg = JamLLVMGetUndef(ty); for (uint32_t i = 0; i < count; i++) { JirRef er = static_cast(lctx.jfn.getExtra(extra + 1 + i)); @@ -600,6 +697,12 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { // `__test_`; methods get dotted FQNs like `T.drop` or // `m.T.drop`; instantiated cloned methods keep their // `Vec__i32.push` form). Codegen does a single lookup. + // + // Extra layout `[argCount, arg0, arg1, ...]` is uniform: for + // sret-returning callees, astgen prepends the result slot as + // `arg0`. No alloca emission lives in jir_codegen anymore — + // the slot may be a caller-owned destination (zero-copy place + // path) or a fresh alloca emitted at astgen time. StringIdx calleeId = static_cast(inst.a); const std::string &symbol = lctx.ctx.getStringPool().get(calleeId); JamFunctionRef f = @@ -612,32 +715,19 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { uint32_t argCount = lctx.jfn.getExtra(extra); bool calleeUsesSret = JamLLVMFunctionUsesSret(f); std::vector args; - args.reserve(argCount + (calleeUsesSret ? 1u : 0u)); - JamValueRef sretSlot = nullptr; - if (calleeUsesSret) { - // Allocate caller-owned storage for the result, prepend - // its pointer as the hidden first argument. The slot type - // comes from the callee's sret attribute, which carries - // the pointee type. Result is loaded back after the call. - JamTypeRef pointee = JamLLVMFunctionSretPointeeType(f); - uint64_t align = lctx.ctx.typeAlign(inst.ty); - sretSlot = JamLLVMBuildAlloca(lctx.ctx.getBuilder(), pointee, align, - "sret.slot"); - args.push_back(sretSlot); - } + args.reserve(argCount); for (uint32_t i = 0; i < argCount; i++) { JirRef ar = static_cast(lctx.jfn.getExtra(extra + 1 + i)); args.push_back(emitInst(lctx, ar)); } if (calleeUsesSret) { - // The LLVM call itself returns void; the value lives in - // the sret slot. Load it so the JirRef -> LLVM value map - // holds the materialized return value. + // The LLVM call returns void; the result already lives in + // args[0] (the sret slot). The Call's JirRef value is + // that slot pointer so byref consumers see a pointer to + // the freshly-written aggregate. JamLLVMBuildCall(lctx.ctx.getBuilder(), f, args.data(), static_cast(args.size()), ""); - JamTypeRef retLlvmTy = lctx.ctx.getLLVMType(inst.ty); - return JamLLVMBuildLoad(lctx.ctx.getBuilder(), retLlvmTy, sretSlot, - "sret.val"); + return args[0]; } const char *resultName = (inst.ty == kNoType) ? "" : "call"; return JamLLVMBuildCall(lctx.ctx.getBuilder(), f, args.data(), @@ -761,17 +851,38 @@ static JamValueRef emitInstImpl(JirCodegenCtx &lctx, JirRef r) { } return nullptr; } + case JirTag::SretArg: { + // Resolve to the function's hidden sret pointer (param 0 + // when the return is Indirect). astgen emits this when it + // needs to plumb the return slot into a result_ptr. + JamFunctionRef f = + JamLLVMGetFunction(lctx.ctx.getModule(), lctx.jfn.name.c_str()); + return JamLLVMGetParam(f, 0); + } case JirTag::Ret: { - // sret return: store the value through the leading hidden - // `ptr sret(%T)` arg and emit `ret void`. The caller already - // owns the slot, so we don't allocate anything here. + // sret return: copy the result into the leading hidden + // `ptr sret(%T)` slot and emit `ret void`. Under the + // byref model, the return value's JirRef is a + // pointer to the aggregate's storage — emit a single memcpy + // rather than a `store %T %v` which SROA would later + // decompose into per-field insertvalue work (the AArch64 + // DAGCombine quadratic-blowup path). if (jirReturnIsSret(lctx.jfn, lctx.ctx)) { JamFunctionRef f = JamLLVMGetFunction(lctx.ctx.getModule(), lctx.jfn.name.c_str()); JamValueRef sretSlot = JamLLVMGetParam(f, 0); if (inst.a != kNoJirRef) { - JamValueRef v = emitInst(lctx, inst.a); - JamLLVMBuildStore(lctx.ctx.getBuilder(), v, sretSlot); + TypeIdx vty = lctx.jfn.getInst(inst.a).ty; + if (jam::abi::isByRef(vty, lctx.ctx)) { + JamValueRef src = emitInst(lctx, inst.a); + uint64_t size = lctx.ctx.typeSize(vty); + uint64_t align = lctx.ctx.typeAlign(vty); + JamLLVMBuildMemCpy(lctx.ctx.getBuilder(), sretSlot, align, + src, align, size); + } else { + JamValueRef v = emitInst(lctx, inst.a); + JamLLVMBuildStore(lctx.ctx.getBuilder(), v, sretSlot); + } } JamLLVMBuildRetVoid(lctx.ctx.getBuilder()); return nullptr; diff --git a/src/jir_verify.cpp b/src/jir_verify.cpp index 4d8a1f2..34497d7 100644 --- a/src/jir_verify.cpp +++ b/src/jir_verify.cpp @@ -155,6 +155,8 @@ const char *tagName(JirTag t) { return "CallIndirect"; case JirTag::Param: return "Param"; + case JirTag::SretArg: + return "SretArg"; case JirTag::StructLit: return "StructLit"; case JirTag::FieldAccess: @@ -330,6 +332,7 @@ struct Verifier { case JirTag::Bool: case JirTag::Str: case JirTag::Param: + case JirTag::SretArg: case JirTag::Unreachable: return; case JirTag::Alloca: diff --git a/src/main.cpp b/src/main.cpp index 124bfc7..21a2146 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -1138,6 +1138,21 @@ static std::vector collectJamFiles(const std::string &dir) { return files; } +// JAM_VERSION_BASE pins the semver-ish prefix; the suffix is the git +// short SHA the binary was built from, baked in by the Makefile via +// `-D JAM_VERSION_SHA="..."`. Falls back to "unknown" when the build +// system doesn't supply one (out-of-tree, no .git, etc.). +#ifndef JAM_VERSION_BASE +#define JAM_VERSION_BASE "0.0.1" +#endif +#ifndef JAM_VERSION_SHA +#define JAM_VERSION_SHA "unknown" +#endif + +static void printVersion() { + std::cout << JAM_VERSION_BASE "-" JAM_VERSION_SHA "\n"; +} + static void printHelp(const char *prog) { std::cout << "Usage: " << prog @@ -1190,6 +1205,7 @@ static void printHelp(const char *prog) { " -l, --library \n" " Link against system library \n" " -h, --help Show this help and exit\n" + " -V, --version Print version and exit\n" "\n" "Examples:\n" " " @@ -1384,6 +1400,10 @@ int main(int argc, char *argv[]) { printHelp(argv[0]); return 0; } + if (arg == "--version" || arg == "-V") { + printVersion(); + return 0; + } if (arg == "--target-info") { showTarget = true; continue; -- 2.51.2