diff --git a/Makefile b/Makefile index e99b439..8c17a68 100644 --- a/Makefile +++ b/Makefile @@ -140,10 +140,32 @@ test-abi: build clang++ -o ./abi_tests ./test_abi.o ./jam_llvm.o ./lexer.o ./parser.o ./ast.o ./codegen.o ./target.o ./cabi.o ./module_resolver.o ./symbol_table.o ./number_literal.o ./init_analysis.o ./drop_registry.o ./abi.o `$(LLVM_CONFIG) --ldflags --libs --libfiles --system-libs` ./abi_tests -# Run all tests: Jam must-pass + analyzer C++ tests + ABI C++ tests + existing C++ tests -test: test-unit test-init test-abi +# Codegen-time must-fail tests. Each test invokes ./jam.out as a +# subprocess on a small Jam source string and asserts on stderr/exit +# (the kind of error that surfaces during instantiation, not during +# semantic analysis — generic methods missing, default() with the +# wrong shape, etc.). Subprocess approach because driving the full +# codegen in-process would require replicating main.cpp's LLVM init +# scaffolding. +test-codegen-errors: build @echo "" - @echo "Running C++ unit tests..." + @echo "Building and running codegen-error C++ tests..." + clang++ -c ./tests/cpp/test_codegen_errors.cpp -o ./test_codegen_errors.o `$(LLVM_CONFIG) --cxxflags` -fexceptions $(OPTFLAGS) + clang++ -o ./codegen_error_tests ./test_codegen_errors.o + ./codegen_error_tests + +# Run all tests: Jam must-pass + analyzer C++ tests + ABI C++ tests + +# codegen-error C++ tests. The legacy `tests/cpp/build_and_run.sh` +# CMake suite was retired (drifted out of sync with `--emit-ir`'s +# linking behavior; see test-legacy-cpp below for archaeology). +test: test-unit test-init test-abi test-codegen-errors + +# Legacy CMake-driven C++ test suite. Pre-dates the in-process analyzer +# and ABI tests; kept only for archaeology. Not run by `make test` +# because it captures the link path (no `main` in test sources → all +# 32 tests fail at link, not at the assertion they actually want to +# check). +test-legacy-cpp: cd tests/cpp && ./build_and_run.sh # Format all C++ sources in-place diff --git a/src/abi.cpp b/src/abi.cpp index f367048..6e50210 100644 --- a/src/abi.cpp +++ b/src/abi.cpp @@ -35,9 +35,9 @@ bool isScalar(TypeKind k) { ParamABI classifyParam(ParamMode mode, TypeIdx ty, const JamCodegenContext &ctx) { - // `mut` and `undefined` always carry a pointer to caller-owned (or - // uninit-caller) storage. Size is irrelevant. - if (mode == ParamMode::Mut || mode == ParamMode::Undefined) { + // `mut` always carries a pointer to caller-owned storage; size + // irrelevant. + if (mode == ParamMode::Mut) { return ParamABI{ParamABI::Kind::ByPointer, nullptr, static_cast(ctx.typeAlign(ty))}; } diff --git a/src/abi.h b/src/abi.h index 02d714a..437697f 100644 --- a/src/abi.h +++ b/src/abi.h @@ -59,7 +59,7 @@ struct ReturnABI { // Classify a parameter (mode, type) pair. Pure function of its inputs; // safe to call any number of times. See docs/ABI.md §4. // -// mut / undefined → always ByPointer +// mut → always ByPointer // let / move, scalar T → ByValue // let / move, aggregate // size <= kByValueMaxBytes → ByValue (LLVM handles register packing) diff --git a/src/ast.cpp b/src/ast.cpp index 8b7b847..48b0089 100644 --- a/src/ast.cpp +++ b/src/ast.cpp @@ -686,17 +686,83 @@ static JamValueRef codegenCall(JamCodegenContext &ctx, const AstNode &n) { // this function uses `callee` (source-level) for FunctionAST lookups // and `llvmName` for the LLVM symbol — they only differ for methods. std::string llvmName = callee; + std::string lookupName = callee; { size_t dot = callee.find('.'); if (dot != std::string::npos && callee.find('.', dot + 1) == std::string::npos) { std::string typeName = callee.substr(0, dot); - if (ctx.getStruct(typeName)) { - const FunctionAST *methodAST = ctx.getFunctionAST(callee); + std::string methodPart = callee.substr(dot + 1); + + // Generics G6: if `typeName` is `Self` and we're inside + // the body of an instantiated method, the substitution + // context maps `Self` to the synthetic anon-struct name, + // which in turn maps to the instantiated struct's Named + // TypeIdx. Resolve it to the canonical instantiated + // struct name (e.g. `Box__i32`). + // + // We also handle the synthetic name directly because the + // parser already rewrote `Self` to `__anon_struct_` + // in *type* positions (see parseType), but not in + // *expression* positions like `Self.make(...)`. + if (typeName == "Self") { + TypeIdx selfTy = ctx.lookupCurrentSubst("Self"); + if (selfTy == kNoType) { + // Fall through to the regular lookup; the + // parser-time anon-struct name might have been + // substituted via the value-position path. + } else if (const auto *sinfo = + ctx.lookupStruct(selfTy)) { + typeName = sinfo->name; + } + } else { + TypeIdx ctxTy = ctx.lookupCurrentSubst(typeName); + if (ctxTy != kNoType) { + if (const auto *sinfo = + ctx.lookupStruct(ctxTy)) { + typeName = sinfo->name; + } + } + } + + // Generics G4/G6: if the LHS is a type alias + // (`const BoxI32 = Box(i32);`), resolve it to the + // canonical instantiated struct name (`Box__i32`) and + // look up the method there. Both `BoxI32.unwrap` and + // `Box__i32.unwrap` dispatch to the same instantiated + // method. + std::string canonicalType = typeName; + TypeIdx aliasTarget = ctx.lookupTypeAlias(typeName); + if (aliasTarget != kNoType) { + if (const auto *sinfo = + ctx.lookupStruct(aliasTarget)) { + canonicalType = sinfo->name; + } + } + + if (ctx.getStruct(canonicalType)) { + std::string canonical = + canonicalType + "." + methodPart; + const FunctionAST *methodAST = + ctx.getFunctionAST(canonical); + if (!methodAST) { + methodAST = ctx.getFunctionAST(callee); + } if (methodAST) { + lookupName = canonical; llvmName = mangledFunctionName( *methodAST, ctx.getTypePool(), ctx.getStringPool()); + } else { + // LHS resolved to a real struct, but it has no + // method by that name. Report this specifically — + // the most common case is a generic instantiation + // where the substituted T lacks an expected method + // (e.g. `Maybe(NoDefault).default()` body calling + // `T.default()` when NoDefault has no default()). + throw std::runtime_error( + "type `" + canonicalType + + "` has no method `" + methodPart + "`"); } } } @@ -715,7 +781,11 @@ static JamValueRef codegenCall(JamCodegenContext &ctx, const AstNode &n) { // are preserved at the source level; the pointer is purely an ABI // optimization. (Mut and Undefined modes still require explicit `&` // at the call site since their borrow shape is user-visible.) - const FunctionAST *calleeAST = ctx.getFunctionAST(callee); + // + // Generics G6: lookupName is the resolved name (alias-canonicalized) + // for instantiated methods. For non-method calls it equals callee. + const FunctionAST *calleeAST = ctx.getFunctionAST(lookupName); + if (!calleeAST) calleeAST = ctx.getFunctionAST(callee); // P9.6 sret: when the callee returns a large aggregate, the call // site allocates a result slot and passes its address as the @@ -847,6 +917,31 @@ static JamValueRef codegenReturn(JamCodegenContext &ctx, const AstNode &n) { return JamLLVMConstInt(ctx.getInt8Type(), 0, false); } + // If the return value is a struct literal with no explicit target + // type (the parser leaves d.lhs == kNoType), patch it from the + // enclosing function's return TypeIdx so codegenStructLit knows + // what to construct. Mirrors what codegenVarDecl does for var + // decls with struct-literal initializers. + const AstNode &retNode = ctx.getNodeStore().get(retIdx); + if (retNode.tag == AstTag::StructLit && + static_cast(retNode.lhs) == kNoType) { + TypeIdx retTy = ctx.getCurrentReturnType(); + if (retTy != kNoType && ctx.lookupStruct(retTy)) { + ctx.getNodeStore().getMut(retIdx).lhs = retTy; + } + } + if ((retNode.tag == AstTag::ArrayLit || + retNode.tag == AstTag::ArrayRepeat) && + static_cast(retNode.lhs) == kNoType) { + TypeIdx retTy = ctx.getCurrentReturnType(); + if (retTy != kNoType) { + const TypeKey &k = ctx.getTypePool().get(retTy); + if (k.kind == TypeKind::Array) { + ctx.getNodeStore().getMut(retIdx).lhs = retTy; + } + } + } + // P9.6: when the enclosing function returns via sret, the codegen // stores the return value into the caller-provided slot rather than // returning by value. The slot's LLVM pointee type tells us what @@ -1145,43 +1240,45 @@ static JamValueRef codegenVarDecl(JamCodegenContext &ctx, const AstNode &n) { JamValueRef Alloca = JamLLVMBuildAlloca( ctx.getBuilder(), VarType, ctx.typeAlign(type), name.c_str()); - if (initIdx != kNoNode) { - const AstNode &initNode = ns.get(initIdx); - // `= undefined` — leave alloca uninitialized. - if (initNode.tag != AstTag::UndefinedLit) { - // Patch struct literals with the declared target struct type so - // they can resolve fields and coerce values during their codegen. - if (initNode.tag == AstTag::StructLit) { - ctx.getNodeStore().getMut(initIdx).lhs = type; - } - JamValueRef InitVal = codegenNode(ctx, initIdx, VarType); - if (!InitVal) return nullptr; - InitVal = coerceTo(ctx, InitVal, VarType); - JamLLVMBuildStore(ctx.getBuilder(), InitVal, Alloca); - } - } - + // Every var declaration carries an initializer (the parser rejects + // the no-init form). Patch nested literals with the declared target + // type so they can resolve fields and coerce values during codegen. + const AstNode &initNode = ns.get(initIdx); + if (initNode.tag == AstTag::StructLit || + initNode.tag == AstTag::ArrayLit || + initNode.tag == AstTag::ArrayRepeat) { + ctx.getNodeStore().getMut(initIdx).lhs = type; + } + // Register the binding's address BEFORE evaluating the init expression + // so `&self` inside a struct literal (recursive type initializer) can + // resolve to this alloca. The init analyzer rejects reads-before-init + // separately. ctx.setVariable(name, Alloca); ctx.setVariableType(name, type); - - // P8.1: register the binding for drop emission at scope exit if its - // type has a user-defined `fn drop(self: mut T)`. The init analyzer - // has already rejected `move` on drop-bearing bindings (P8 foundation), - // so codegen can emit drops unconditionally without double-free risk - // from caller-side moves. (Bindings declared `= undefined` and never - // assigned would still drop on uninit memory; tracking that is P8.2.) - if (const auto *reg = ctx.getDropRegistry()) { - const TypeKey &k = ctx.getTypePool().get(type); - if (k.kind == TypeKind::Struct || k.kind == TypeKind::Named) { - StringIdx structNameIdx = static_cast(k.a); - if (structNameIdx != kNoString) { - const std::string &structName = - ctx.getStringPool().get(structNameIdx); - auto it = reg->find(structName); - if (it != reg->end()) { - ctx.registerLocalDrop(name, Alloca, VarType, - it->second); - } + JamValueRef InitVal = codegenNode(ctx, initIdx, VarType); + if (!InitVal) return nullptr; + InitVal = coerceTo(ctx, InitVal, VarType); + JamLLVMBuildStore(ctx.getBuilder(), InitVal, Alloca); + + // Register the binding for drop emission at scope exit if its type has + // a user-defined `fn drop(self: mut T)`. With every var carrying a + // concrete initializer, the binding is always Init at this point — + // the drop call fires unconditionally at every exit reaching it. + { + // Resolve through aliases / generic instantiations to the + // canonical struct so the drop-fn lookup uses the actual + // instantiated name (e.g. `Box__i32`), not the alias the user + // wrote (`BoxI32`). lookupStruct handles GenericCall and + // alias resolution; the resulting StructInfo's `name` is + // canonical. lookupDropFn checks the pre-built drop registry + // AND the per-instantiation drop map populated by + // instantiateStructExpr. + const auto *sinfo = ctx.lookupStruct(type); + if (sinfo) { + const std::string &structName = sinfo->name; + if (const FunctionAST *dropFn = + ctx.lookupDropFn(structName)) { + ctx.registerLocalDrop(name, Alloca, VarType, dropFn); } } } @@ -1194,6 +1291,9 @@ static JamValueRef codegenVarDecl(JamCodegenContext &ctx, const AstNode &n) { // directly — no load-value-then-pass-by-value workaround. The drop fn // reads/writes self through the pointer and the caller's storage is // genuinely affected. +// +// With `undefined` removed, every drop-bearing binding is statically Init +// at every exit reaching this drop point, so the call fires unconditionally. static void emitOneDrop(JamCodegenContext &ctx, const JamCodegenContext::DropEntry &e) { std::string mangled = mangledFunctionName(*e.dropFn, ctx.getTypePool(), @@ -1927,6 +2027,143 @@ static JamValueRef codegenMatch(JamCodegenContext &ctx, const AstNode &n) { return phi; } +// `[a, b, c]`: build the SSA aggregate via insertvalue chain. The target +// type must have been bound to n.lhs by the caller (var-decl, return, +// struct-field init, etc.); reaching codegen with kNoType is a missing- +// context error. +// +// Targets: +// • [N]T — element list must have N entries. +// • []T — only the empty form `[]` is accepted; produces an empty +// slice {ptr=null, len=0}. Non-empty literals targeting a +// slice would need a backing storage decision the language +// hasn't made yet. +static JamValueRef codegenArrayLit(JamCodegenContext &ctx, const AstNode &n) { + TypeIdx targetType = static_cast(n.lhs); + if (targetType == kNoType) { + throw std::runtime_error( + "Array literal used without a known target type"); + } + const TypeKey &k = ctx.getTypePool().get(targetType); + const NodeStore &ns = ctx.getNodeStore(); + ExtraIdx extra = static_cast(n.rhs); + uint32_t count = ns.getExtra(extra); + + if (k.kind == TypeKind::Slice) { + if (count != 0) { + throw std::runtime_error( + "Non-empty array literal cannot target a slice type " + "directly — use a fixed-size array and slice it"); + } + // {ptr=null, len=0} as a {ptr, i64} aggregate. + JamTypeRef sliceLLVM = ctx.getLLVMType(targetType); + JamValueRef sliceVal = JamLLVMGetUndef(sliceLLVM); + TypeIdx elemTy = static_cast(k.a); + JamTypeRef elemLLVM = ctx.getLLVMType(elemTy); + JamTypeRef nullPtr = JamLLVMPointerType(elemLLVM, 0); + JamValueRef nullPtrVal = JamLLVMConstNull(nullPtr); + sliceVal = JamLLVMBuildInsertValue(ctx.getBuilder(), sliceVal, + nullPtrVal, 0, "slice.ptr"); + JamValueRef zeroLen = + JamLLVMConstInt(ctx.getInt64Type(), 0, false); + sliceVal = JamLLVMBuildInsertValue(ctx.getBuilder(), sliceVal, + zeroLen, 1, "slice.len"); + return sliceVal; + } + + if (k.kind != TypeKind::Array) { + throw std::runtime_error( + "Array literal target type is not an array or slice"); + } + TypeIdx elemTy = static_cast(k.a); + uint32_t expectedLen = k.b; + JamTypeRef arrLLVM = ctx.getLLVMType(targetType); + JamTypeRef elemLLVM = ctx.getLLVMType(elemTy); + + if (count != expectedLen) { + throw std::runtime_error( + "Array literal has " + std::to_string(count) + + " elements but type expects " + std::to_string(expectedLen)); + } + + JamValueRef arrVal = JamLLVMGetUndef(arrLLVM); + for (uint32_t i = 0; i < count; i++) { + NodeIdx elemIdx = static_cast(ns.getExtra(extra + 1 + i)); + const AstNode &elemNode = ns.get(elemIdx); + if ((elemNode.tag == AstTag::ArrayLit || + elemNode.tag == AstTag::ArrayRepeat || + elemNode.tag == AstTag::StructLit) && + static_cast(elemNode.lhs) == kNoType) { + ctx.getNodeStore().getMut(elemIdx).lhs = elemTy; + } + JamValueRef elemVal = codegenNode(ctx, elemIdx, elemLLVM); + if (!elemVal) return nullptr; + elemVal = coerceTo(ctx, elemVal, elemLLVM); + arrVal = JamLLVMBuildInsertValue(ctx.getBuilder(), arrVal, elemVal, i, + "elem"); + } + return arrVal; +} + +// `[expr; N]`: repeat `expr` N times. v1 requires N to be a constant +// integer literal so we can build a fixed-size array; symbolic-sized +// arrays would need a Slice or runtime fill loop, neither of which is +// in scope yet. Codegen emits the repeat as an insertvalue chain — the +// simplest IR; LLVM constant-folds when expr is constant. +static JamValueRef codegenArrayRepeat(JamCodegenContext &ctx, + const AstNode &n) { + TypeIdx arrType = static_cast(n.lhs); + if (arrType == kNoType) { + throw std::runtime_error( + "Array repeat literal used without a known target type"); + } + const TypeKey &k = ctx.getTypePool().get(arrType); + if (k.kind != TypeKind::Array) { + throw std::runtime_error( + "Array repeat literal target type is not an array"); + } + TypeIdx elemTy = static_cast(k.a); + uint32_t expectedLen = k.b; + JamTypeRef arrLLVM = ctx.getLLVMType(arrType); + JamTypeRef elemLLVM = ctx.getLLVMType(elemTy); + + const NodeStore &ns = ctx.getNodeStore(); + ExtraIdx extra = static_cast(n.rhs); + NodeIdx valueIdx = static_cast(ns.getExtra(extra)); + NodeIdx countIdx = static_cast(ns.getExtra(extra + 1)); + + const AstNode &countNode = ns.get(countIdx); + if (countNode.tag != AstTag::NumberLit) { + throw std::runtime_error( + "Array repeat count must be an integer literal"); + } + uint64_t countVal = numberLitValue(countNode); + if (countVal != static_cast(expectedLen)) { + throw std::runtime_error( + "Array repeat count " + std::to_string(countVal) + + " does not match declared length " + + std::to_string(expectedLen)); + } + + const AstNode &valNode = ns.get(valueIdx); + if ((valNode.tag == AstTag::ArrayLit || + valNode.tag == AstTag::ArrayRepeat || + valNode.tag == AstTag::StructLit) && + static_cast(valNode.lhs) == kNoType) { + ctx.getNodeStore().getMut(valueIdx).lhs = elemTy; + } + JamValueRef elemVal = codegenNode(ctx, valueIdx, elemLLVM); + if (!elemVal) return nullptr; + elemVal = coerceTo(ctx, elemVal, elemLLVM); + + JamValueRef arrVal = JamLLVMGetUndef(arrLLVM); + for (uint32_t i = 0; i < expectedLen; i++) { + arrVal = JamLLVMBuildInsertValue(ctx.getBuilder(), arrVal, elemVal, i, + "rep"); + } + return arrVal; +} + static JamValueRef codegenStructLit(JamCodegenContext &ctx, const AstNode &n) { TypeIdx structType = static_cast(n.lhs); if (structType == kNoType) { @@ -1935,6 +2172,47 @@ static JamValueRef codegenStructLit(JamCodegenContext &ctx, const AstNode &n) { } const auto *info = ctx.lookupStruct(structType); if (!info) { + // Union targets share the `{ field: value }` literal form. The + // untagged-union shape means exactly one field's bits are + // initialized; the literal must list one field and we materialize + // the union value via alloca/store/load (LLVM has no SSA insertvalue + // for the cross-field reinterpret pattern). + if (const auto *uinfo = ctx.lookupUnion(structType)) { + const NodeStore &ns = ctx.getNodeStore(); + const StringPool &sp = ctx.getStringPool(); + ExtraIdx extra = static_cast(n.rhs); + uint32_t fieldCount = ns.getExtra(extra); + if (fieldCount != 1) { + throw std::runtime_error( + "Union literal must list exactly one field, got " + + std::to_string(fieldCount)); + } + StringIdx fldNameId = + static_cast(ns.getExtra(extra + 1)); + NodeIdx fldExprIdx = + static_cast(ns.getExtra(extra + 2)); + const std::string &fieldName = sp.get(fldNameId); + TypeIdx fieldTy = + ctx.getUnionFieldType(uinfo->name, fieldName); + if (fieldTy == kNoType) { + throw std::runtime_error("Union `" + uinfo->name + + "` has no field `" + fieldName + + "`"); + } + JamTypeRef fieldLLVM = ctx.getLLVMType(fieldTy); + JamValueRef fieldVal = + codegenNode(ctx, fldExprIdx, fieldLLVM); + if (!fieldVal) return nullptr; + fieldVal = coerceTo(ctx, fieldVal, fieldLLVM); + // Alloca the union, write the chosen field, load the union back + // as an aggregate value. + JamValueRef alloca = JamLLVMBuildAlloca( + ctx.getBuilder(), uinfo->type, ctx.typeAlign(structType), + (uinfo->name + ".lit").c_str()); + JamLLVMBuildStore(ctx.getBuilder(), fieldVal, alloca); + return JamLLVMBuildLoad(ctx.getBuilder(), uinfo->type, alloca, + "union.val"); + } throw std::runtime_error("Struct literal target is not a struct"); } @@ -1956,13 +2234,22 @@ static JamValueRef codegenStructLit(JamCodegenContext &ctx, const AstNode &n) { "' in struct " + info->name); } - // Propagate target struct type into nested struct literals. + // Propagate target struct/array type into nested literals. TypeIdx declaredFieldType = info->fields[idx].second; const AstNode &fldNode = ns.get(fldExprIdx); if (fldNode.tag == AstTag::StructLit && ctx.lookupStruct(declaredFieldType)) { ctx.getNodeStore().getMut(fldExprIdx).lhs = declaredFieldType; } + if ((fldNode.tag == AstTag::ArrayLit || + fldNode.tag == AstTag::ArrayRepeat) && + static_cast(fldNode.lhs) == kNoType) { + const TypeKey &fk = ctx.getTypePool().get(declaredFieldType); + if (fk.kind == TypeKind::Array) { + ctx.getNodeStore().getMut(fldExprIdx).lhs = + declaredFieldType; + } + } JamTypeRef expectedType = ctx.getLLVMType(declaredFieldType); JamValueRef fieldVal = codegenNode(ctx, fldExprIdx, expectedType); @@ -2151,9 +2438,15 @@ JamValueRef codegenNode(JamCodegenContext &ctx, NodeIdx node, case AstTag::StringLit: return codegenStringLit( ctx, ctx.getStringPool().get(static_cast(n.lhs))); - case AstTag::UndefinedLit: + case AstTag::StructExpr: + // Generics G2: a `struct {...}` expression evaluates to a `type` + // value at compile time. The substitution engine in G4 consumes + // these from ModuleAST::AnonStructs. Reaching this case during + // regular codegen means a generic function leaked through — + // generics should be skipped at the function-emit pass. throw std::runtime_error( - "`undefined` is only valid as a `var` declaration initializer"); + "internal: struct expression reached LLVM codegen " + "(generic was not instantiated before lowering)"); case AstTag::Variable: { const std::string &name = ctx.getStringPool().get(static_cast(n.lhs)); @@ -2223,6 +2516,10 @@ JamValueRef codegenNode(JamCodegenContext &ctx, NodeIdx node, return JamLLVMConstInt(ctx.getInt8Type(), 0, false); case AstTag::StructLit: return codegenStructLit(ctx, n); + case AstTag::ArrayLit: + return codegenArrayLit(ctx, n); + case AstTag::ArrayRepeat: + return codegenArrayRepeat(ctx, n); case AstTag::MatchNode: return codegenMatch(ctx, n); case AstTag::AsCast: { @@ -2463,6 +2760,11 @@ void FunctionAST::defineBody(JamCodegenContext &ctx) { ctx.setSretSlot(nullptr); } + // Record the function's source-level return TypeIdx so codegenReturn + // can patch struct-literal-shaped return expressions whose target + // type is parser-time kNoType. + ctx.setCurrentReturnType(ReturnType); + for (unsigned i = 0; i < Args.size(); i++) { // P9 mode-aware ABI: ByValue parameters are stored to a local // alloca on entry (matching the existing pattern for value diff --git a/src/ast.h b/src/ast.h index 9d4d0bd..a2cf585 100644 --- a/src/ast.h +++ b/src/ast.h @@ -35,7 +35,6 @@ enum class ParamMode : uint8_t { Let = 0, Mut, Move, - Undefined, }; // One function parameter. Mode defaults to Let when not annotated at the @@ -78,6 +77,19 @@ class FunctionAST { // Convenience: declare + define in one shot. Kept so single-pass // callers (extern, repl-style code) still work. JamFunctionRef codegen(JamCodegenContext &ctx); + + // Generics G1: a function is generic iff any of its parameters has + // type `type` (the meta-type) or its return type is `type`. Generic + // functions are not lowered to LLVM at decl time — instead they are + // registered for compile-time instantiation at each call site that + // supplies concrete type arguments. + bool isGeneric() const { + if (ReturnType == BuiltinType::Type) return true; + for (const Param &p : Args) { + if (p.Type == BuiltinType::Type) return true; + } + return false; + } }; // Top-level struct declaration: const Vec3 = struct { x: f32, y: f32 }; @@ -167,6 +179,12 @@ class ConstDeclAST { std::string Name; TypeIdx DeclaredType; // kNoType when omitted; init drives the type NodeIdx InitExpr; + // Generics G4: if non-kNoType, this const is a type alias (RHS is a + // generic-instantiation expression like `Box(i32)`). InitExpr is + // kNoNode in that case. main.cpp processes type-alias consts + // before regular consts and registers the alias in the codegen + // context's type-alias table. + TypeIdx AliasedType = kNoType; ConstDeclAST(std::string Name, TypeIdx DeclaredType, NodeIdx InitExpr) : Name(std::move(Name)), DeclaredType(DeclaredType), @@ -203,6 +221,13 @@ class ModuleAST { std::vector> Enums; std::vector> Consts; std::vector> Functions; + // Generics G2: bodies of `struct { ... }` expressions. Each entry is + // a regular StructDeclAST with a synthetic name (`__anon_struct_`). + // Index N comes from the AnonStructs vector at parse time and is + // stored in the StructExpr AST node's d.lhs slot. The substitution + // engine in G4 reads from here to instantiate the struct with + // concrete type arguments. + std::vector> AnonStructs; ModuleAST() = default; }; diff --git a/src/ast_flat.h b/src/ast_flat.h index a84c84e..5a09d82 100644 --- a/src/ast_flat.h +++ b/src/ast_flat.h @@ -24,6 +24,7 @@ #include #include #include +#include #include #include #include @@ -50,7 +51,6 @@ enum class AstTag : uint8_t { NumberLit, // d.lhs = lo32(val), d.rhs = hi32(val); flags bit 0 = isNeg BoolLit, // d.lhs = 0|1 StringLit, // d.lhs = StringIdx - UndefinedLit, // Lvalues / refs Variable, // d.lhs = StringIdx (name) @@ -90,6 +90,23 @@ enum class AstTag : uint8_t { // d.lhs = TypeIdx (struct type, kNoType if inferred from var-decl context) // d.rhs = ExtraIdx → [fieldCount, fieldName0, fieldExpr0, fieldName1, ...] StructLit, + // `[a, b, c, ...]` array literal in expression position. Element type + // inferred from var-decl target type (kNoType if unknown — codegen + // rejects). + // d.lhs = TypeIdx (target element type, kNoType if unbound) + // d.rhs = ExtraIdx → [count, elem0, elem1, ...] + ArrayLit, + // `[expr; N]` array repeat literal. N is a constant integer expression. + // d.lhs = TypeIdx (target array type, kNoType if unbound) + // d.rhs = ExtraIdx → [valueNode, countNode] + ArrayRepeat, + // Generics G2: a `struct { fields, methods }` expression, evaluated at + // compile time to a value of type `type`. The expression's body lives + // in ModuleAST::AnonStructs as a regular StructDeclAST with a + // synthetic name; this node carries the index so the substitution + // engine in G4 can find it. + // d.lhs = u32 (index into ModuleAST::AnonStructs) + StructExpr, // Pattern match (M1: integer literals, ranges, or-patterns, wildcard). // d.lhs = NodeIdx (scrutinee expression) @@ -276,6 +293,18 @@ enum class TypeKind : uint8_t { // or enum. The codegen resolves Named values by consulting the // three registries, in that order. Named, // namedT.name (StringIdx) + // Generics G1: the type of types — a value of this kind is itself + // a TypeIdx, used at compile time only. A function parameter + // declared `T: type` accepts a type argument when called. Has no + // runtime ABI; the codegen never lowers it to an LLVM type. + Type, + // Generics G4: deferred generic instantiation, parsed from + // `Identifier(arg, ...)` in a type position. The TypeKey carries + // the callee name in `a` (StringIdx) and an index into + // TypePool::genericArgs in `b`. The codegen resolves this lazily + // via the substitution engine when an LLVM type is requested or + // when a binding's static TypeIdx is needed. + GenericCall, }; struct TypeKey { @@ -317,6 +346,14 @@ inline bool operator==(const TypeKey &x, const TypeKey &y) { case TypeKind::Union: case TypeKind::Named: return x.a == y.a; + case TypeKind::Type: + // Singleton meta-type: every TypeKey of kind Type is equal. + return true; + case TypeKind::GenericCall: + // G4: equal iff callee name AND args-list index match. The + // args index is canonical because the side table interns + // args lists. + return x.a == y.a && x.b == y.b; } return false; } @@ -347,12 +384,20 @@ constexpr TypeIdx U64 = 9; constexpr TypeIdx I64 = 10; constexpr TypeIdx F32 = 11; constexpr TypeIdx F64 = 12; +constexpr TypeIdx Type = 13; // generics G1: the meta-type constexpr TypeIdx U1 = Bool; // alias } // namespace BuiltinType class TypePool { std::vector keys_; std::unordered_map idx_; + // Generics G4: side table for generic-call argument lists. A + // `TypeKind::GenericCall` TypeKey stores the index into this vector + // in `b`. Two `Maybe(File)` references at distinct sites resolve to + // the same TypeIdx because the args list `[File]` is interned here + // once. + std::vector> genericArgs_; + std::map, uint32_t> genericArgsIdx_; TypeIdx pushKey(TypeKey k) { TypeIdx id = static_cast(keys_.size()); @@ -379,6 +424,7 @@ class TypePool { pushKey(TypeKey{TypeKind::Int, 0, 0, 64, 1}); // i64 pushKey(TypeKey{TypeKind::Float, 0, 0, 32, 0}); pushKey(TypeKey{TypeKind::Float, 0, 0, 64, 0}); + pushKey(TypeKey{TypeKind::Type, 0, 0, 0, 0}); } TypeIdx intern(TypeKey k) { @@ -412,6 +458,28 @@ class TypePool { TypeIdx internStruct(StringIdx nameId) { return intern(TypeKey{TypeKind::Struct, 0, 0, nameId, 0}); } + + // Generics G4: intern a `Identifier(arg, ...)` generic call as a + // TypeIdx. The args list itself is interned in the side table so two + // identical calls share an args index and consequently the same + // TypeIdx. + TypeIdx internGenericCall(StringIdx nameId, std::vector args) { + uint32_t argsIdx; + auto it = genericArgsIdx_.find(args); + if (it != genericArgsIdx_.end()) { + argsIdx = it->second; + } else { + argsIdx = static_cast(genericArgs_.size()); + genericArgs_.push_back(args); + genericArgsIdx_.emplace(std::move(args), argsIdx); + } + return intern( + TypeKey{TypeKind::GenericCall, 0, 0, nameId, argsIdx}); + } + + const std::vector &genericArgsAt(uint32_t idx) const { + return genericArgs_[idx]; + } TypeIdx internEnum(StringIdx nameId) { return intern(TypeKey{TypeKind::Enum, 0, 0, nameId, 0}); } diff --git a/src/codegen.cpp b/src/codegen.cpp index 41858f6..8ed41ca 100644 --- a/src/codegen.cpp +++ b/src/codegen.cpp @@ -6,6 +6,9 @@ */ #include "codegen.h" + +#include "ast.h" + #include JamCodegenContext::JamCodegenContext(const char *moduleName) { @@ -142,9 +145,16 @@ JamTypeRef JamCodegenContext::getLLVMType(TypeIdx ty) const { break; } case TypeKind::Named: { - // Parser-deferred user type — resolve against the three - // declaration registries in order: struct, union, enum. + // Parser-deferred user type. Resolution order: + // 1. Generics G6 substitution context (T, Self, __anon_struct_N) + // 2. struct/union/enum registries + // 3. type alias map (Generics G4) const std::string &name = stringPool.get(static_cast(k.a)); + TypeIdx substTarget = lookupCurrentSubst(name); + if (substTarget != kNoType) { + result = getLLVMType(substTarget); + break; + } if (const auto *sinfo = getStruct(name)) { result = sinfo->type; } else if (const auto *uinfo = getUnion(name)) { @@ -155,6 +165,11 @@ JamTypeRef JamCodegenContext::getLLVMType(TypeIdx ty) const { // declaration. result = einfo->hasPayloadVariant ? einfo->type : getInt8Type(); } else { + TypeIdx aliasTarget = lookupTypeAlias(name); + if (aliasTarget != kNoType) { + result = getLLVMType(aliasTarget); + break; + } throw std::runtime_error( "Unknown user-defined type: " + name); } @@ -174,6 +189,22 @@ JamTypeRef JamCodegenContext::getLLVMType(TypeIdx ty) const { result = getInt8Type(); break; } + case TypeKind::Type: + // Generics G1: the meta-type has no runtime representation. + // Reaching this path means a generic function leaked to LLVM + // codegen without being instantiated first. + throw std::runtime_error( + "internal: cannot lower `type` to LLVM (generic was not " + "instantiated before codegen)"); + case TypeKind::GenericCall: { + // Generics G4: lazily resolve the call to a concrete TypeIdx + // via the substitution engine, then recurse on the result. + // The resolution is memoized in genericResolutions so each + // distinct call site only does the work once. + TypeIdx resolved = resolveGenericCall(ty); + result = getLLVMType(resolved); + break; + } } llvmTypeCache[ty] = result; return result; @@ -217,7 +248,7 @@ TypeIdx JamCodegenContext::getVariableType(const std::string &name) const { void JamCodegenContext::registerStruct( const std::string &name, JamTypeRef type, - std::vector> fields) { + std::vector> fields) const { StructInfo info; info.name = name; info.type = type; @@ -229,13 +260,35 @@ const JamCodegenContext::StructInfo * JamCodegenContext::lookupStruct(TypeIdx ty) const { if (ty == kNoType) return nullptr; const TypeKey &k = typePool.get(ty); + // Generics G4: a `GenericCall` TypeIdx resolves to a concrete type + // (typically a Named struct produced by instantiation). Recurse on + // the resolved TypeIdx so downstream lookups behave as if the user + // had written the instantiated name directly. + if (k.kind == TypeKind::GenericCall) { + return lookupStruct(resolveGenericCall(ty)); + } // Accept TypeKind::Struct (explicit) or TypeKind::Named (parser- // deferred user type that resolves to a struct). if (k.kind != TypeKind::Struct && k.kind != TypeKind::Named) { return nullptr; } const std::string &name = stringPool.get(static_cast(k.a)); - return getStruct(name); + // Generics G6: substitution context wins (T, Self, __anon_struct_N + // resolved per-instantiation during method body codegen). + TypeIdx substTarget = lookupCurrentSubst(name); + if (substTarget != kNoType) { + return lookupStruct(substTarget); + } + if (const StructInfo *direct = getStruct(name)) { + return direct; + } + // Generics G4: try the type alias table — `const BoxI32 = Box(i32);` + // maps `BoxI32` to the instantiated struct's TypeIdx. + TypeIdx aliasTarget = lookupTypeAlias(name); + if (aliasTarget != kNoType) { + return lookupStruct(aliasTarget); + } + return nullptr; } const JamCodegenContext::StructInfo * @@ -444,6 +497,17 @@ uint64_t JamCodegenContext::typeSize(TypeIdx ty) const { typeSize(static_cast(k.a)); case TypeKind::Struct: case TypeKind::Named: { + // Generics G6: a Named type may be a substitution-context + // reference to a parameter (T → i32) or to Self. Resolve + // through the substitution map first; if found and the + // target is a primitive (Int/Float/etc.), the recursive + // typeSize handles it. Same shape as getLLVMType. + const std::string &substName = + stringPool.get(static_cast(k.a)); + if (TypeIdx subTarget = lookupCurrentSubst(substName); + subTarget != kNoType) { + return typeSize(subTarget); + } // User-named types resolve through any of the three registries. if (const StructInfo *info = lookupStruct(ty)) { uint64_t total = 0; @@ -485,6 +549,12 @@ uint64_t JamCodegenContext::typeSize(TypeIdx ty) const { } case TypeKind::Enum: return 1; // M2 E1 enums lower to u8 + case TypeKind::Type: + // Meta-type has no runtime size. + return 0; + case TypeKind::GenericCall: + // G4: resolve and recurse. + return typeSize(resolveGenericCall(ty)); } throw std::runtime_error("typeSize: unhandled type kind"); } @@ -511,6 +581,15 @@ uint64_t JamCodegenContext::typeAlign(TypeIdx ty) const { return typeAlign(static_cast(k.a)); case TypeKind::Struct: case TypeKind::Named: { + // Generics G6: substitution context wins. A Named type may + // be a parameter reference (T → i32) or Self that resolves + // to a non-aggregate; the recursive call handles primitives. + const std::string &substName = + stringPool.get(static_cast(k.a)); + if (TypeIdx subTarget = lookupCurrentSubst(substName); + subTarget != kNoType) { + return typeAlign(subTarget); + } if (const StructInfo *info = lookupStruct(ty)) { uint64_t maxAlign = 1; for (const auto &f : info->fields) { @@ -548,6 +627,356 @@ uint64_t JamCodegenContext::typeAlign(TypeIdx ty) const { } case TypeKind::Enum: return 1; + case TypeKind::Type: + return 1; + case TypeKind::GenericCall: + return typeAlign(resolveGenericCall(ty)); } throw std::runtime_error("typeAlign: unhandled type kind"); } + +// -------------------------------------------------------------------------- +// Generics G4: substitution engine for `Identifier(arg, ...)` types. +// -------------------------------------------------------------------------- + +namespace { + +// Recursively rewrite a TypeIdx, replacing parameter Named-types with their +// bound concrete TypeIdx. Other compound types (pointers, slices, arrays, +// nested generic calls) are reconstructed with substituted children. +TypeIdx substituteType( + TypeIdx ty, const std::unordered_map &subst, + TypePool &types, const StringPool &strings) { + const TypeKey &k = types.get(ty); + switch (k.kind) { + case TypeKind::Named: { + const std::string &name = + strings.get(static_cast(k.a)); + auto it = subst.find(name); + if (it != subst.end()) return it->second; + return ty; + } + case TypeKind::PtrSingle: + return types.internPtrSingle(substituteType( + static_cast(k.a), subst, types, strings)); + case TypeKind::PtrMany: + return types.internPtrMany(substituteType( + static_cast(k.a), subst, types, strings)); + case TypeKind::Slice: + return types.internSlice(substituteType( + static_cast(k.a), subst, types, strings)); + case TypeKind::Array: + return types.internArray( + substituteType(static_cast(k.a), subst, types, + strings), + k.b); + case TypeKind::GenericCall: { + // Recurse into args: a generic call inside a generic body + // (e.g. `Box(Maybe(T))`) substitutes T in the inner call. + const auto &args = types.genericArgsAt(k.b); + std::vector newArgs; + newArgs.reserve(args.size()); + for (TypeIdx a : args) { + newArgs.push_back( + substituteType(a, subst, types, strings)); + } + return types.internGenericCall(static_cast(k.a), + std::move(newArgs)); + } + default: + return ty; + } +} + +} // namespace + +TypeIdx JamCodegenContext::resolveGenericCall(TypeIdx callTy) const { + auto cached = genericResolutions_.find(callTy); + if (cached != genericResolutions_.end()) return cached->second; + + const TypeKey &k = typePool.get(callTy); + const std::string &calleeName = + stringPool.get(static_cast(k.a)); + const auto &args = typePool.genericArgsAt(k.b); + + const FunctionAST *generic = getFunctionAST(calleeName); + if (!generic) { + throw std::runtime_error("Unknown generic: " + calleeName); + } + if (!generic->isGeneric()) { + throw std::runtime_error( + "Identifier `" + calleeName + + "` is a non-generic function used in a type position"); + } + if (args.size() != generic->Args.size()) { + throw std::runtime_error( + "Generic `" + calleeName + "` expects " + + std::to_string(generic->Args.size()) + + " type argument(s), got " + std::to_string(args.size())); + } + + // v1 only supports `T: type` parameters (no comptime values yet). + for (size_t i = 0; i < generic->Args.size(); i++) { + if (generic->Args[i].Type != BuiltinType::Type) { + throw std::runtime_error( + "Generic `" + calleeName + + "` has a non-type parameter (comptime values are v2)"); + } + } + + // Build the substitution map from parameter names to concrete args. + std::unordered_map subst; + for (size_t i = 0; i < generic->Args.size(); i++) { + subst[generic->Args[i].Name] = args[i]; + } + + // Walk the function body looking for the return statement. v1 + // supports two return shapes: (1) `return T;` where T is a type + // parameter or a named type, and (2) `return struct {...};` where + // the body declares the instantiated struct's fields. + TypeIdx result = kNoType; + for (NodeIdx stmt : generic->Body) { + const AstNode &n = nodeStore.get(stmt); + if (n.tag != AstTag::Return) continue; + NodeIdx valueIdx = static_cast(n.lhs); + const AstNode &value = nodeStore.get(valueIdx); + if (value.tag == AstTag::Variable) { + const std::string &name = + stringPool.get(static_cast(value.lhs)); + auto it = subst.find(name); + if (it != subst.end()) { + result = it->second; + break; + } + // Not a parameter — treat as a named type reference and + // substitute through (handles forwarding generics that + // return a non-parameter named type). + TypeIdx asNamed = + typePool.internNamed(stringPool.intern(name)); + result = substituteType(asNamed, subst, typePool, + stringPool); + break; + } + if (value.tag == AstTag::StructExpr) { + result = instantiateStructExpr( + value, calleeName, args, subst); + break; + } + throw std::runtime_error( + "Generic body's return value shape not supported in v1 " + "(only `return T;` or `return struct {...};` are " + "implemented)"); + } + + if (result == kNoType) { + throw std::runtime_error( + "Generic `" + calleeName + + "` has no return statement to evaluate"); + } + + genericResolutions_[callTy] = result; + return result; +} + +// Instantiate a `struct {...}` expression appearing in a generic body's +// return statement. Substitutes each field's TypeIdx with the concrete +// generic args, creates a fresh LLVM struct type with a unique name, and +// returns a Named TypeIdx pointing at the new struct. Methods are not +// instantiated in v1 (G6 territory). +TypeIdx JamCodegenContext::instantiateStructExpr( + const AstNode &exprNode, const std::string &calleeName, + const std::vector &args, + const std::unordered_map &subst) const { + if (!anonStructs_) { + throw std::runtime_error( + "internal: anonymous struct table not registered on " + "codegen context"); + } + uint32_t anonIdx = exprNode.lhs; + if (anonIdx >= anonStructs_->size()) { + throw std::runtime_error( + "internal: StructExpr references missing AnonStructs[" + + std::to_string(anonIdx) + "]"); + } + const StructDeclAST *anon = (*anonStructs_)[anonIdx].get(); + + // Build the instantiated struct's name from the callee + arg names. + // `Maybe(File)` → `Maybe__File`. Pointer/array types lower through + // substituteType; we only need a stable spelling for the canonical + // non-compound cases here. v1's stdlib won't pass non-named types + // as generic args, so this is enough to get the demo running. + std::string instName = calleeName; + for (TypeIdx a : args) { + instName += "__"; + const TypeKey &ak = typePool.get(a); + switch (ak.kind) { + case TypeKind::Int: { + char buf[16]; + std::snprintf(buf, sizeof(buf), "%c%u", + ak.b ? 'i' : 'u', ak.a); + instName += buf; + break; + } + case TypeKind::Bool: + instName += "bool"; + break; + case TypeKind::Struct: + case TypeKind::Named: + instName += + stringPool.get(static_cast(ak.a)); + break; + default: + instName += "T"; // catch-all; v2 spec needed + break; + } + } + + // Memoize on the instantiated name. If we've already produced this + // struct, return its TypeIdx without re-creating the LLVM type. + if (const StructInfo *existing = getStruct(instName)) { + (void)existing; + return typePool.internNamed(stringPool.intern(instName)); + } + + // Build the full substitution map: parameter names → concrete args, + // plus the anon-struct's synthetic name (which is what `Self` + // resolved to in *type* positions at parse time) → the new + // instantiated struct's Named TypeIdx. We also alias the literal + // string "Self" to the same target so codegen sites that see + // the parser's stringified `Self.method(...)` (an expression- + // position member access on the Self identifier) can resolve + // it via the same map. Used for field types, method signatures, + // and method body codegen. + std::unordered_map bodySubst = subst; + TypeIdx instNamed = + typePool.internNamed(stringPool.intern(instName)); + bodySubst[anon->Name] = instNamed; + bodySubst["Self"] = instNamed; + + // Substitute each field's type, then declare + fill the LLVM struct. + std::vector> instFields; + instFields.reserve(anon->Fields.size()); + for (const auto &f : anon->Fields) { + instFields.emplace_back( + f.first, + substituteType(f.second, bodySubst, typePool, stringPool)); + } + + JamTypeRef llvmStruct = JamLLVMStructCreateNamed( + getContext(), instName.c_str()); + registerStruct(instName, llvmStruct, instFields); + + std::vector fieldLLVM; + fieldLLVM.reserve(instFields.size()); + for (const auto &f : instFields) { + fieldLLVM.push_back(getLLVMType(f.second)); + } + JamLLVMStructSetBody(llvmStruct, fieldLLVM.data(), + static_cast(fieldLLVM.size()), false); + + // Generics G6: instantiate methods. For each method on the + // AnonStruct, clone the FunctionAST with substituted parameter and + // return types and a unique source-level name. Register under the + // qualified name `.` so callers can + // dispatch via the existing struct-method machinery. Body codegen + // uses the substitution context so inner Named-type references + // (e.g. `var s: Self`) resolve correctly per-instantiation. + if (!anon->Methods.empty()) { + // `bodySubst` was built above (used by field substitution); we + // reuse it for method signatures and body codegen. + for (const auto &origMethod : anon->Methods) { + std::vector instArgs; + instArgs.reserve(origMethod->Args.size()); + for (const auto &p : origMethod->Args) { + Param sp = p; + sp.Type = substituteType(p.Type, bodySubst, typePool, + stringPool); + instArgs.push_back(std::move(sp)); + } + TypeIdx instReturn = origMethod->ReturnType; + if (instReturn != kNoType) { + instReturn = substituteType(instReturn, bodySubst, + typePool, stringPool); + } + + std::string instMethodName = instName + "." + origMethod->Name; + auto cloned = std::make_unique( + instMethodName, std::move(instArgs), instReturn, + origMethod->Body, origMethod->isExtern, + origMethod->isExport, origMethod->isPub, + origMethod->isTest, origMethod->isVarArgs); + FunctionAST *clonePtr = cloned.get(); + instantiatedMethods_.push_back(std::move(cloned)); + + // Generics G6: a method whose original (pre-clone) name is + // `drop` and whose first param matches the instantiated + // struct in `mut` mode is the type's drop method. Register + // it under the instantiated struct's name so the var-decl + // codegen finds it when the user binds a value of this + // type. + if (origMethod->Name == "drop" && + origMethod->Args.size() == 1 && + origMethod->Args[0].Name == "self" && + origMethod->Args[0].Mode == ParamMode::Mut) { + instantiatedDrops_[instName] = clonePtr; + } + + // The const_cast is honest about what's happening: lazy + // instantiation runs during otherwise-const type lookups, + // and registerFunctionAST + declarePrototype + defineBody + // genuinely mutate the codegen state. Most of the mutated + // fields are already `mutable`; this is the gap we accept + // rather than propagating non-const through every type + // lookup. + JamCodegenContext &mutCtx = + const_cast(*this); + mutCtx.registerFunctionAST(instMethodName, clonePtr); + + // Save the outer codegen state. defineBody calls + // clearVariables / clearDrops to set up its own + // function-local scope; without saving, the caller that + // triggered lazy instantiation would lose its bindings, + // drop scopes, and sret slot the moment we return. + // + // Builder insert-block has to be saved separately because + // the snapshot doesn't include LLVM-side state. + StateSnapshot savedState = snapshotState(); + JamBasicBlockRef savedBB = + JamLLVMGetInsertBlock(getBuilder()); + + // Activate the substitution context so the body codegen + // resolves Named-type references through the map (Self + // → instantiated struct, T → concrete arg, etc.). + // + // Implicit contract: declarePrototype is safe to invoke + // here even though the *outer* struct (this Self) is + // already registered above (line 882-892). If the method's + // signature contains a nested generic call like + // `fn create() Pair(T, T)`, declarePrototype's + // getLLVMType-on-return-type will recursively trigger + // instantiateStructExpr for the nested generic. That + // recursion is bounded by memoization + // (genericResolutions_) and doesn't loop because each + // nested instantiation produces and registers its own + // concrete struct before returning. + setCurrentSubst(bodySubst); + clonePtr->declarePrototype(mutCtx); + // TODO v2: errors thrown from defineBody during lazy + // instantiation surface with the generic body's source + // location, not the *call site* that triggered the + // instantiation. Zig annotates errors with both + // (failed_decls + dependency_failure trail). Track for + // follow-up so messages like "T doesn't have field foo" + // point at the user's `Box(SomeType).foo` call. + clonePtr->defineBody(mutCtx); + clearCurrentSubst(); + + restoreState(std::move(savedState)); + if (savedBB) { + JamLLVMPositionBuilderAtEnd(getBuilder(), savedBB); + } + } + } + + return typePool.internNamed(stringPool.intern(instName)); +} diff --git a/src/codegen.h b/src/codegen.h index 7326a6e..93af6cd 100644 --- a/src/codegen.h +++ b/src/codegen.h @@ -12,10 +12,17 @@ #include "drop_registry.h" #include "jam_llvm.h" #include +#include #include #include +#include #include +// Forward declarations for AST types referenced by pointer/reference +// here. Full definitions live in ast.h, included by ast.cpp / codegen.cpp. +class FunctionAST; +class StructDeclAST; + class JamCodegenContext { public: JamCodegenContext(const char *moduleName); @@ -76,7 +83,8 @@ class JamCodegenContext { std::vector> fields; }; void registerStruct(const std::string &name, JamTypeRef type, - std::vector> fields); + std::vector> fields) + const; const StructInfo *getStruct(const std::string &name) const; // Convenience: resolve a TypeIdx of kind Struct back to its StructInfo. const StructInfo *lookupStruct(TypeIdx ty) const; @@ -176,6 +184,18 @@ class JamCodegenContext { const jam::drops::DropRegistry *getDropRegistry() const { return dropRegistry; } + // Generics G6: look up a drop method for an instantiated struct + // (e.g. Box__i32). Falls back to the pre-built drop registry. + // Returns nullptr if the struct has no drop method. + const FunctionAST *lookupDropFn(const std::string &structName) const { + auto it = instantiatedDrops_.find(structName); + if (it != instantiatedDrops_.end()) return it->second; + if (dropRegistry) { + auto rit = dropRegistry->find(structName); + if (rit != dropRegistry->end()) return rit->second; + } + return nullptr; + } // P8.3 scope-aware drops: each lexical block (function body, if/else // arm, while/for body, match arm body) has its own DropEntry vector. // pushDropScope/popDropScope are called at block boundaries; the @@ -196,7 +216,10 @@ class JamCodegenContext { JamBuilderRef builder; std::map namedValues; std::map namedValueTypes; - std::map structs; + // `structs` is mutable because Generics G4 lazily instantiates new + // struct types (e.g. `Maybe(File)`) on demand from inside the + // otherwise-const `resolveGenericCall` / `getLLVMType` paths. + mutable std::map structs; std::map unions; std::map enums; std::map moduleConsts; @@ -227,9 +250,140 @@ class JamCodegenContext { void setSretSlot(JamValueRef slot) { sretSlot = slot; } JamValueRef getSretSlot() const { return sretSlot; } + // Per-function return TypeIdx, populated by defineBody so codegenReturn + // can patch the target type into struct-literal returns. Without this, + // `fn default() Self { return { n: 0 }; }` couldn't tell the literal + // what struct to construct (the literal's d.lhs is parser-time + // kNoType). + void setCurrentReturnType(TypeIdx ty) { currentReturnType_ = ty; } + TypeIdx getCurrentReturnType() const { return currentReturnType_; } + private: std::unordered_map functionAsts; JamValueRef sretSlot = nullptr; + TypeIdx currentReturnType_ = kNoType; + + // Generics G4: cache mapping from the deferred-call TypeIdx (a + // `TypeKind::GenericCall` entry) to the concrete TypeIdx produced + // by substitution + memoization. Populated lazily from + // resolveGenericCall. + mutable std::unordered_map genericResolutions_; + + // Generics G4: borrowed pointer to the parsed module's anonymous + // struct bodies (those produced by `struct { ... }` expressions). + // Set by main.cpp before any codegen runs. The substitution engine + // reads from here when resolving generic calls whose bodies contain + // `return struct {...};`. + const std::vector> *anonStructs_ = + nullptr; + + // Generics G4: type alias table. `const BoxI32 = Box(i32);` registers + // `BoxI32 → resolved-TypeIdx-of-Box(i32)`. Consulted by lookupStruct + // (and by getLLVMType via the recursive lookup path) so a binding + // declared `var b: BoxI32` finds the same struct that `Box(i32)` + // would produce. + mutable std::map typeAliases_; + + // Generics G6: clones of FunctionASTs produced by method + // instantiation. Each clone has substituted parameter and return + // types and a unique source-level name (`Box__i32.unwrap`) so the + // existing function registry / LLVM symbol pipeline handles them + // as ordinary functions. + mutable std::vector> instantiatedMethods_; + + // Generics G6: drop methods on instantiated types. The pre-built + // drop registry is borrowed via const pointer and was populated + // before lazy instantiation. Drop methods produced by + // instantiateStructExpr go here and are consulted alongside the + // pre-built registry by var-decl codegen. + mutable std::unordered_map + instantiatedDrops_; + + // Generics G6: type substitution context active during codegen of an + // instantiated method's body. Lookups of Named types (T, Self, + // __anon_struct_N) consult this map first. Set/cleared around + // declarePrototype + defineBody calls in instantiateStructExpr. + mutable std::unordered_map currentSubst_; + + public: + // Generics G6: snapshot/restore of the per-function codegen state. + // Used to wrap recursive method instantiation that runs inside the + // outer caller's codegen flow — the inner declarePrototype + + // defineBody would otherwise clear the caller's variables and + // drop scopes. + struct StateSnapshot { + std::map namedValues; + std::map namedValueTypes; + std::vector> dropScopes; + JamValueRef sretSlot; + }; + StateSnapshot snapshotState() const { + return StateSnapshot{namedValues, namedValueTypes, dropScopes, + sretSlot}; + } + void restoreState(StateSnapshot s) const { + // All-or-nothing reassignment of the per-function state. The + // `mutable` qualifier on these fields is reserved for incremental + // caching (struct registry, type-pool growth, etc.); whole-state + // reassignment by snapshot is a different access mode and uses + // const_cast to be honest about that. + auto &self = const_cast(*this); + self.namedValues = std::move(s.namedValues); + self.namedValueTypes = std::move(s.namedValueTypes); + self.dropScopes = std::move(s.dropScopes); + self.sretSlot = s.sretSlot; + } + + // Generics G4: resolve a `TypeKind::GenericCall` TypeIdx to a concrete + // TypeIdx by running the substitution engine on the generic + // function's body. Result is memoized — subsequent calls with the + // same TypeIdx hit the cache and return the same concrete TypeIdx. + TypeIdx resolveGenericCall(TypeIdx callTy) const; + + // Generics G4: register the anonymous-struct table for the current + // module so the substitution engine can find struct expression + // bodies by their AnonStructs index. + void setAnonStructs( + const std::vector> *as) { + anonStructs_ = as; + } + + // Generics G4: register a type alias (`const Name = Box(i32);`). + // Consulted by lookupStruct when resolving a Named TypeIdx whose + // name matches an alias. + void registerTypeAlias(const std::string &name, TypeIdx target) { + typeAliases_[name] = target; + } + TypeIdx lookupTypeAlias(const std::string &name) const { + auto it = typeAliases_.find(name); + if (it != typeAliases_.end()) return it->second; + return kNoType; + } + + // Generics G6: substitution context manipulators. The map is active + // only during codegen of an instantiated method's body — set right + // before declarePrototype/defineBody, cleared right after. + void setCurrentSubst( + std::unordered_map s) const { + currentSubst_ = std::move(s); + } + void clearCurrentSubst() const { currentSubst_.clear(); } + TypeIdx lookupCurrentSubst(const std::string &name) const { + auto it = currentSubst_.find(name); + if (it != currentSubst_.end()) return it->second; + return kNoType; + } + + private: + // Generics G4: instantiate a `struct {...}` expression as the result + // of a generic call. Substitutes each field's type with the + // concrete generic args, creates a fresh LLVM struct type with a + // synthesized name, and returns a Named TypeIdx pointing at it. + // Memoizes by instantiated name. + TypeIdx instantiateStructExpr( + const AstNode &exprNode, const std::string &calleeName, + const std::vector &args, + const std::unordered_map &subst) const; }; #endif // CODEGEN_H diff --git a/src/init_analysis.cpp b/src/init_analysis.cpp index b260e76..5ec6866 100644 --- a/src/init_analysis.cpp +++ b/src/init_analysis.cpp @@ -83,7 +83,6 @@ class Analyzer { Result analyzeStructLit(NodeIdx idx, NameMap state); void checkVariableRead(NodeIdx idx, const NameMap &state); - void checkUndefinedParamsInit(const NameMap &state, NodeIdx anchor); void checkDropBearingLocalsInit(const NameMap &state, NodeIdx anchor); void checkScopeEscape(NodeIdx exprIdx); void emitError(std::string message, NodeIdx anchor, std::string varName); @@ -125,11 +124,8 @@ class Analyzer { std::vector diagnostics_; // Pointer to the current function's parameter list. Set in run(), - // cleared on exit. Used by checkUndefinedParamsInit to find every - // `undefined`-mode parameter without allocating a separate vector - // of names. Lifetime is bounded by the run() activation that set it. + // Cleared on exit. Lifetime bounded by the run() activation that set it. const std::vector *args_ = nullptr; - bool hasUndefinedParam_ = false; // Static type per binding name. Populated as parameter list and // VarDecls are walked. Used by P8's drop-bearing check on `move` args. @@ -149,23 +145,16 @@ class Analyzer { std::vector Analyzer::run(const FunctionAST &fn) { args_ = &fn.Args; - hasUndefinedParam_ = false; varTypes_.clear(); NameMap state; // P3: parameter entry state depends on the declared mode. - // Undefined → Uninit (caller hands an uninitialized destination) // Let / Mut / Move → Init (caller's binding is valid) // `move` does not change anything for the callee's view of its own // parameter — the moved-from-ness applies to the *caller's* binding // after the call (P4 work). for (const Param &p : fn.Args) { - if (p.Mode == ParamMode::Undefined) { - state[p.Name] = InitState::Uninit; - hasUndefinedParam_ = true; - } else { - state[p.Name] = InitState::Init; - } + state[p.Name] = InitState::Init; varTypes_[p.Name] = p.Type; } @@ -181,12 +170,11 @@ std::vector Analyzer::run(const FunctionAST &fn) { } // P3 + P8.2: if control reaches the end of the body without an - // explicit return, every `undefined`-mode parameter and every - // drop-bearing local must have been initialized. (Functions that - // always return on every path produce r.terminated == true and skip - // this check; the per-return checks in analyzeReturn cover them.) + // explicit return, every drop-bearing local must have been + // initialized. (Functions that always return on every path produce + // r.terminated == true and skip this check; the per-return checks + // in analyzeReturn cover them.) if (!r.terminated) { - checkUndefinedParamsInit(r.state, kNoNode); checkDropBearingLocalsInit(r.state, kNoNode); } @@ -260,13 +248,42 @@ Result Analyzer::analyze(NodeIdx idx, NameMap state) { return analyze(n.lhs, std::move(state)); case AstTag::StructLit: return analyzeStructLit(idx, std::move(state)); + case AstTag::ArrayLit: { + // Walk every element expression so any variable reads inside the + // array literal are checked against the init state. + ExtraIdx extra = n.rhs; + uint32_t count = nodes_.getExtra(extra); + Result r{std::move(state), false}; + for (uint32_t i = 0; i < count; i++) { + NodeIdx elemIdx = static_cast( + nodes_.getExtra(extra + 1 + i)); + r = analyze(elemIdx, std::move(r.state)); + if (r.terminated) return r; + } + return r; + } + case AstTag::ArrayRepeat: { + // Walk the value expression. The count is a constant-only NumberLit + // (enforced at codegen) — analyzing it is a no-op on init state. + ExtraIdx extra = n.rhs; + NodeIdx valueIdx = static_cast(nodes_.getExtra(extra)); + NodeIdx countIdx = static_cast(nodes_.getExtra(extra + 1)); + auto r = analyze(valueIdx, std::move(state)); + if (r.terminated) return r; + return analyze(countIdx, std::move(r.state)); + } // Literals — no init effect on bindings. case AstTag::NumberLit: case AstTag::BoolLit: case AstTag::StringLit: - case AstTag::UndefinedLit: case AstTag::ImportLit: + // Generics G2: `struct {...}` expression evaluates to a value of + // type `type` at compile time. The body lives in ModuleAST and is + // processed by the substitution engine — the analyzer doesn't see + // it because generic functions skip analysis (they're not in + // mainModuleEmits). If we ever do reach this case it's a no-op. + case AstTag::StructExpr: return Result{std::move(state), false}; // Pattern atoms appear only inside MatchNode arms; they don't read @@ -302,14 +319,8 @@ Result Analyzer::analyzeVarDecl(NodeIdx idx, NameMap state) { if (initIdx == kNoNode) { // Should not happen — Jam syntactically requires an initializer - // for `var` declarations. Treat as Uninit defensively. - state[name] = InitState::Uninit; - return Result{std::move(state), false}; - } - - const AstNode &initNode = nodes_.get(initIdx); - if (initNode.tag == AstTag::UndefinedLit) { - state[name] = InitState::Uninit; + // for `var` declarations. + state[name] = InitState::Init; return Result{std::move(state), false}; } @@ -534,10 +545,9 @@ Result Analyzer::analyzeReturn(NodeIdx idx, NameMap state) { Result r{std::move(state), false}; if (n.lhs != kNoNode) { r = analyze(n.lhs, std::move(r.state)); } - // P3: every `undefined`-mode parameter must be Init on every return path. - checkUndefinedParamsInit(r.state, idx); - // P8.2: every drop-bearing local must be Init too — codegen will emit - // drop on it at this exit, and dropping uninit memory is UB. + // P8.2: every drop-bearing local must be Init at every return path — + // codegen will emit drop on it at this exit, and dropping uninit + // memory is UB. checkDropBearingLocalsInit(r.state, idx); r.terminated = true; return r; @@ -606,18 +616,14 @@ Result Analyzer::analyzeCall(NodeIdx idx, NameMap state) { if (r.terminated) return r; // Walk the arg expression. For `let`/`mut`/`move` modes the walk // includes a read-check on the base binding (it must already be - // Init). For `undefined` mode the caller is supposed to pass an - // AddressOf of an Uninit slot — and `&x` skips the read-check on - // `x` thanks to the AddressOf bypass in analyze(). + // Init). r = analyze(info.argIdx, std::move(r.state)); if (r.terminated) return r; // Post-call mode effect on the caller's binding. // move — caller's binding moves into the callee; becomes Uninit. - // undefined — callee writes through the address; becomes Init. // let / mut — no caller-side state change. - if (info.mode == ParamMode::Move || - info.mode == ParamMode::Undefined) { + if (info.mode == ParamMode::Move) { if (info.path.base != kNoString) { const std::string &name = strings_.get(info.path.base); @@ -625,8 +631,7 @@ Result Analyzer::analyzeCall(NodeIdx idx, NameMap state) { // until move-aware drop tracking lands in P8.1. Without // it, codegen would emit drop on the moved-out slot at // scope exit — a double-free. - if (info.mode == ParamMode::Move && - lookupDropFor(name) != nullptr) { + if (lookupDropFor(name) != nullptr) { emitError("cannot `move` binding `" + name + "` of drop-bearing type — drop+move " "tracking is not yet implemented (P8.1); " @@ -635,9 +640,7 @@ Result Analyzer::analyzeCall(NodeIdx idx, NameMap state) { info.argIdx, name); } - r.state[name] = - (info.mode == ParamMode::Move) ? InitState::Uninit - : InitState::Init; + r.state[name] = InitState::Uninit; } } } @@ -863,11 +866,11 @@ void Analyzer::checkDropBearingLocalsInit(const NameMap &state, std::string msg = "drop-bearing binding `" + name + "` of type `" + structName + "` "; if (s == InitState::Uninit) { - msg += "must be initialized before this exit — drop runs on it " + msg += "must be initialized before this exit; drop runs on it " "and would otherwise read uninit memory"; } else { msg += "may not be initialized on every path that reaches this " - "exit — drop runs on it on every path"; + "exit; drop runs on it on every path"; } emitError(std::move(msg), anchor, name); } @@ -894,16 +897,12 @@ void Analyzer::checkScopeEscape(NodeIdx exprIdx) { const std::string &name = strings_.get(nameId); for (const Param &p : *args_) { if (p.Name != name) continue; - if (p.Mode != ParamMode::Mut && - p.Mode != ParamMode::Undefined) - return; - const char *modeStr = - (p.Mode == ParamMode::Mut) ? "mut" : "undefined"; - std::string msg = "cannot return `&` of `"; - msg += modeStr; - msg += "`-mode parameter `" + name + - "` — borrows are second-class and cannot escape " - "the function"; + if (p.Mode != ParamMode::Mut) return; + std::string msg = "cannot return `&` of `mut`-mode " + "parameter `" + + name + + "` — borrows are second-class and " + "cannot escape the function"; emitError(std::move(msg), exprIdx, name); return; } @@ -920,27 +919,6 @@ void Analyzer::checkScopeEscape(NodeIdx exprIdx) { } } -void Analyzer::checkUndefinedParamsInit(const NameMap &state, NodeIdx anchor) { - // Fast path: most functions have no `undefined`-mode parameters at all. - // Skip the iteration over fn.Args entirely in that case. - if (!hasUndefinedParam_ || args_ == nullptr) return; - for (const Param &p : *args_) { - if (p.Mode != ParamMode::Undefined) continue; - auto it = state.find(p.Name); - InitState s = (it == state.end()) ? InitState::Uninit : it->second; - if (s == InitState::Init) continue; - std::string msg; - if (s == InitState::Uninit) { - msg = "`undefined`-mode parameter `" + p.Name + - "` was not initialized before this return"; - } else { - msg = "`undefined`-mode parameter `" + p.Name + - "` may not be initialized on every path that reaches this return"; - } - emitError(std::move(msg), anchor, p.Name); - } -} - int Analyzer::lineOf(NodeIdx idx) const { if (idx == kNoNode) return 0; const AstNode &n = nodes_.get(idx); diff --git a/src/init_analysis.h b/src/init_analysis.h index b64ac7a..c24f7ba 100644 --- a/src/init_analysis.h +++ b/src/init_analysis.h @@ -14,6 +14,7 @@ #include #include #include +#include #include class FunctionAST; @@ -27,8 +28,7 @@ namespace init_analysis { // // Unknown binding has not been touched on this path // Init binding holds a valid value; reads OK -// Uninit binding has been declared `= undefined` or has been moved -// out of; reads must produce an error +// Uninit binding has been moved out of; reads must produce an error // MaybeInit merge of Init + Uninit at a control-flow join; reads must // produce an error // @@ -69,12 +69,10 @@ using FunctionRegistry = std::unordered_map; // Scope (cumulative through P4): // - Tracks per-binding init state for locals declared with // `var name: T = ...`. -// - Parameter entry state: Undefined-mode → Uninit, all other modes → Init. -// - At every return and at function fall-through, every Undefined-mode -// parameter must have reached Init. +// - Parameter entry state: every parameter starts Init (the caller is +// required to pass an initialized binding). // - Each call propagates caller-side state per the callee's parameter -// modes: `move` arg ⇒ caller's base binding becomes Uninit; -// `undefined` arg ⇒ caller's base binding becomes Init. +// modes: `move` arg ⇒ caller's base binding becomes Uninit. // // Control-flow handling: // - Straight-line statement sequence: states thread through. @@ -85,12 +83,11 @@ using FunctionRegistry = std::unordered_map; // the pre-loop state, and the post state merges body output with // pre-loop. For-loops over a range additionally assume the body runs // at least once. May replace with fixed-point iteration later. -std::vector analyze(const FunctionAST &fn, const NodeStore &nodes, - const StringPool &strings, - const std::vector &tokens, - const FunctionRegistry *registry = nullptr, - const drops::DropRegistry *drops = nullptr, - const TypePool *types = nullptr); +std::vector analyze( + const FunctionAST &fn, const NodeStore &nodes, const StringPool &strings, + const std::vector &tokens, const FunctionRegistry *registry = nullptr, + const drops::DropRegistry *drops = nullptr, + const TypePool *types = nullptr); } // namespace init_analysis } // namespace jam diff --git a/src/lexer.cpp b/src/lexer.cpp index 08d7a75..e3004a7 100644 --- a/src/lexer.cpp +++ b/src/lexer.cpp @@ -132,14 +132,17 @@ void Lexer::identifier() { addToken(TOK_ENUM, text); } else if (text == "as") { addToken(TOK_AS, text); - } else if (text == "undefined") { - addToken(TOK_UNDEFINED, text); } else if (text == "move") { addToken(TOK_MOVE, text); } else if (text == "u1" || text == "u8" || text == "u16" || text == "u32" || text == "u64" || text == "i8" || text == "i16" || text == "i32" || text == "i64" || text == "f32" || - text == "f64" || text == "bool" || text == "str") { + text == "f64" || text == "bool" || text == "str" || + text == "type") { + // Generics G1: `type` is the meta-type — values of this type are + // themselves types, used at compile time only. Lexed as TOK_TYPE + // alongside the scalar built-ins so the parser sees it in a type + // position uniformly. addToken(TOK_TYPE, text); } else { addToken(TOK_IDENTIFIER, text); diff --git a/src/main.cpp b/src/main.cpp index 3eb101a..8bed4b7 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -165,6 +165,12 @@ static int compileAndRun(const std::string &filename, } } + // Generics G4: hand off the parsed module's anonymous-struct table + // (bodies of `struct {...}` expressions) to the codegen context so + // the substitution engine can look them up by index when resolving + // generic instantiations. + codegenCtx.setAnonStructs(&module->AnonStructs); + // Register struct types from imported modules and main module first so // that codegen has the struct registry available. Two phases so structs // can reference each other regardless of declaration order. @@ -378,8 +384,18 @@ static int compileAndRun(const std::string &filename, // are inlined at use sites (see AstTag::Variable in ast.cpp), so we // only need to teach the codegen context about them — no LLVM // globals get emitted. + // + // Generics G4: a const whose RHS is a generic-instantiation type + // expression (e.g. `const BoxI32 = Box(i32);`) is a *type alias* + // instead. The parser flagged these by setting AliasedType; we + // register them in the type-alias table so subsequent type lookups + // (`var b: BoxI32`) resolve to the instantiated struct. auto registerConsts = [&](ModuleAST *m) { for (auto &c : m->Consts) { + if (c->AliasedType != kNoType) { + codegenCtx.registerTypeAlias(c->Name, c->AliasedType); + continue; + } codegenCtx.registerModuleConst(c->Name, c->InitExpr, c->DeclaredType); } @@ -401,7 +417,15 @@ static int compileAndRun(const std::string &filename, for (const auto &[path, importedModule] : resolver.getLoadedModules()) { if (path == "std") continue; for (auto &func : importedModule->Functions) { - if (func->isPub) { func->declarePrototype(codegenCtx); } + if (func->isPub && !func->isGeneric()) { + func->declarePrototype(codegenCtx); + } + // Generic functions are still registered so call sites can + // look them up for instantiation, but no LLVM is emitted + // until each instantiation is processed in G5. + if (func->isPub) { + codegenCtx.registerFunctionAST(func->Name, func.get()); + } } } @@ -417,10 +441,18 @@ static int compileAndRun(const std::string &filename, if (function->isTest) { testFunctionNames.push_back("__test_" + function->Name); } - function->declarePrototype(codegenCtx); - mainModuleEmits.push_back(function.get()); + // Generics G1: skip prototype + body emission for generic + // functions. They get registered (so call sites can find them) + // but no LLVM is emitted until an instantiation in G5 supplies + // concrete type arguments. + if (!function->isGeneric()) { + function->declarePrototype(codegenCtx); + mainModuleEmits.push_back(function.get()); + } // P9: register by source-level name so call codegen can recover - // parameter modes for callsite ABI decisions. + // parameter modes for callsite ABI decisions. Generic functions + // also need to be in the registry — call sites consult it to + // drive instantiation. codegenCtx.registerFunctionAST(function->Name, function.get()); } @@ -439,25 +471,59 @@ static int compileAndRun(const std::string &filename, }; for (auto &s : module->Structs) { for (auto &m : s->Methods) { - if (m->Args.empty() || m->Args[0].Name != "self") { - std::cerr << filename << ": error: method `" << m->Name - << "` on struct `" << s->Name - << "` must take `self` as its first parameter\n"; - return 1; - } - std::string selfStruct = - resolveStructName(m->Args[0].Type); - if (selfStruct != s->Name) { - std::cerr << filename << ": error: method `" << m->Name - << "` on struct `" << s->Name - << "` has self type `" << selfStruct - << "`; expected `" << s->Name << "`\n"; - return 1; - } - if (m->Name != "drop") { + // Two privileged method names on top-level structs: + // `drop` — no-arg-cleanup; must take `self: mut Self`. + // `default` — opt-in default constructor; takes no + // parameters, returns `Self`. Used by anything + // that expects a default value (struct-literal + // omitted fields, generic type parameters that + // require a default, future `var x: T;`). + // Anything else is rejected — there is no general user- + // defined static-method or instance-method support on top- + // level structs in v1. + if (m->Name == "default") { + if (!m->Args.empty()) { + std::cerr + << filename + << ": error: method `default` on struct `" + << s->Name + << "` must take no parameters\n"; + return 1; + } + std::string retStruct = + resolveStructName(m->ReturnType); + if (retStruct != s->Name) { + std::cerr + << filename + << ": error: method `default` on struct `" + << s->Name + << "` must return `Self` (got `" + << retStruct << "`)\n"; + return 1; + } + } else if (m->Name == "drop") { + if (m->Args.empty() || m->Args[0].Name != "self") { + std::cerr + << filename << ": error: method `" << m->Name + << "` on struct `" << s->Name + << "` must take `self` as its first parameter\n"; + return 1; + } + std::string selfStruct = + resolveStructName(m->Args[0].Type); + if (selfStruct != s->Name) { + std::cerr + << filename << ": error: method `" << m->Name + << "` on struct `" << s->Name + << "` has self type `" << selfStruct + << "`; expected `" << s->Name << "`\n"; + return 1; + } + } else { std::cerr << filename - << ": error: non-drop methods inside struct " - "bodies are not yet supported (saw `" + << ": error: only `drop` and `default` " + "methods are allowed on top-level " + "structs (saw `" << s->Name << "." << m->Name << "`)\n"; return 1; } @@ -468,11 +534,14 @@ static int compileAndRun(const std::string &filename, } } - // Pass 2a: bodies for pub functions in imported modules. + // Pass 2a: bodies for pub functions in imported modules. Generics + // are skipped here too (their bodies are walked at instantiation). for (const auto &[path, importedModule] : resolver.getLoadedModules()) { if (path == "std") continue; for (auto &func : importedModule->Functions) { - if (func->isPub) { func->defineBody(codegenCtx); } + if (func->isPub && !func->isGeneric()) { + func->defineBody(codegenCtx); + } } } @@ -516,11 +585,11 @@ static int compileAndRun(const std::string &filename, std::make_move_iterator(diags.end())); } if (!allDiags.empty()) { return 1; } - } - // Pass 2b: bodies for the main module's functions. - for (FunctionAST *function : mainModuleEmits) { - function->defineBody(codegenCtx); + // Pass 2b: bodies for the main module's functions. + for (FunctionAST *function : mainModuleEmits) { + function->defineBody(codegenCtx); + } } // In test mode, generate a main() that calls all test functions @@ -920,6 +989,15 @@ int main(int argc, char *argv[]) { return failed == 0 ? 0 : 1; } - return compileAndRun(filename, outputName, runFlag, emitIR, testMode, - releaseMode, linkLibs); + // Catch compile-time exceptions cleanly so the user sees a single-line + // error instead of a stack trace + abort. Exceptions reach here from + // codegen paths that detect impossible inputs (e.g. a generic + // instantiation referencing a method the concrete type doesn't have). + try { + return compileAndRun(filename, outputName, runFlag, emitIR, + testMode, releaseMode, linkLibs); + } catch (const std::exception &e) { + std::cerr << filename << ": error: " << e.what() << std::endl; + return 1; + } } diff --git a/src/parser.cpp b/src/parser.cpp index 75ac075..691fc7f 100644 --- a/src/parser.cpp +++ b/src/parser.cpp @@ -130,9 +130,6 @@ NodeIdx Parser::parsePrimary() { if (match(TOK_FALSE)) { return emit(AstNode{AstTag::BoolLit, 0, 0, 0, 0, 0}); } - if (match(TOK_UNDEFINED)) { - return emit(AstNode{AstTag::UndefinedLit, 0, 0, 0, 0, 0}); - } if (match(TOK_STRING_LITERAL)) { StringIdx s = stringPool->intern(previous().lexeme); return emit(AstNode{AstTag::StringLit, 0, 0, 0, s, 0}); @@ -150,6 +147,46 @@ NodeIdx Parser::parsePrimary() { return expr; } if (match(TOK_OPEN_BRACE)) { return parseStructLiteral(); } + if (match(TOK_OPEN_BRACKET)) { + // Array literal `[a, b, c]`, array repeat `[expr; N]`, or empty + // `[]` (well-typed only against a slice or zero-length array + // target — codegen rejects others). + if (match(TOK_CLOSE_BRACKET)) { + ExtraIdx extra = nodes->reserveExtra(1); + nodes->setExtra(extra, 0); + return emit(AstNode{AstTag::ArrayLit, 0, 0, 0, kNoType, + extra}); + } + NodeIdx first = parseLogicalOr(); + if (match(TOK_SEMI)) { + NodeIdx count = parseLogicalOr(); + consume(TOK_CLOSE_BRACKET, + "Expected `]` after array repeat count"); + ExtraIdx extra = nodes->reserveExtra(2); + nodes->setExtra(extra, static_cast(first)); + nodes->setExtra(extra + 1, static_cast(count)); + return emit(AstNode{AstTag::ArrayRepeat, 0, 0, 0, kNoType, + extra}); + } + std::vector elems; + elems.push_back(first); + while (match(TOK_COMMA)) { + if (check(TOK_CLOSE_BRACKET)) break; // trailing comma + elems.push_back(parseLogicalOr()); + } + consume(TOK_CLOSE_BRACKET, + "Expected `]` or `,` in array literal"); + ExtraIdx extra = nodes->reserveExtra(1 + elems.size()); + nodes->setExtra(extra, static_cast(elems.size())); + for (size_t i = 0; i < elems.size(); i++) { + nodes->setExtra(extra + 1 + i, elems[i]); + } + return emit(AstNode{AstTag::ArrayLit, 0, 0, 0, kNoType, extra}); + } + // Generics G2: `struct { ... }` as an expression evaluates to a value + // of type `type` — the struct type being constructed. Used as the + // body of a generic type-returning function. + if (match(TOK_STRUCT)) { return parseStructExpression(); } if (match(TOK_IDENTIFIER)) { std::string name = previous().lexeme; StringIdx nameId = stringPool->intern(name); @@ -285,14 +322,46 @@ TypeIdx Parser::parseType() { if (s == "f64") return BuiltinType::F64; if (s == "bool" || s == "u1") return BuiltinType::Bool; if (s == "str") return typePool->internSlice(BuiltinType::U8); + // Generics G1: `type` resolves to the singleton meta-type. A + // parameter of this type accepts a TypeIdx at compile time. + if (s == "type") return BuiltinType::Type; throw std::runtime_error("Unknown base type: " + s); } if (match(TOK_IDENTIFIER)) { + const std::string &ident = previous().lexeme; + // Generics G3: `Self` resolves to the enclosing struct's name — + // the top of the parser's struct-context stack. Outside any + // struct body it's an error. + if (ident == "Self") { + if (structContextStack.empty()) { + throw std::runtime_error( + "`Self` is only valid inside a struct body"); + } + return typePool->internNamed( + stringPool->intern(structContextStack.back())); + } + // Generics G4: `Identifier(arg, ...)` in a type position is a + // generic instantiation. The args are parsed recursively as + // types; the result is a `TypeKind::GenericCall` TypeIdx that + // the codegen resolves on demand via the substitution engine. + if (check(TOK_OPEN_PAREN)) { + advance(); // consume `(` + std::vector args; + if (!check(TOK_CLOSE_PAREN)) { + do { + args.push_back(parseType()); + } while (match(TOK_COMMA)); + } + consume(TOK_CLOSE_PAREN, + "Expected ')' after generic type arguments"); + return typePool->internGenericCall( + stringPool->intern(ident), std::move(args)); + } // User-named types (struct / union / enum) are interned with // kind = Named; codegen resolves to the concrete kind via the // declaration registries. The parser does not need to know // which kind the user meant. - return typePool->internNamed(stringPool->intern(previous().lexeme)); + return typePool->internNamed(stringPool->intern(ident)); } throw std::runtime_error("Expected type"); } @@ -522,7 +591,8 @@ NodeIdx Parser::parseExpression() { if (match(TOK_COLON)) { type = parseType(); } consume(TOK_EQUAL, - "Expected '=' (use `= undefined` to leave uninitialized)"); + "Expected '=' (every variable must be initialized at " + "declaration)"); NodeIdx init = parseLogicalOr(); consume(TOK_SEMI, "Expected ';' after variable declaration"); @@ -846,7 +916,6 @@ std::unique_ptr Parser::parseFunction() { // x: u32 — Let (default, read-only) // x: mut u32 — Mut (exclusive read-write) // x: move List — Move (consume ownership) - // x: undefined Buf — Undefined (write to uninit destination) // // `*mut T` pointer types start with `*`, not `mut`, so there is // no conflict with the existing pointer syntax. See MVS.md §2. @@ -855,8 +924,6 @@ std::unique_ptr Parser::parseFunction() { mode = ParamMode::Mut; } else if (match(TOK_MOVE)) { mode = ParamMode::Move; - } else if (match(TOK_UNDEFINED)) { - mode = ParamMode::Undefined; } TypeIdx paramType = parseType(); @@ -892,16 +959,9 @@ std::unique_ptr Parser::parseFunction() { isPub, isTest, false); } -std::unique_ptr Parser::parseStructDecl() { - consume(TOK_CONST, "Expected 'const' for struct declaration"); - consume(TOK_IDENTIFIER, "Expected struct name"); - std::string name = previous().lexeme; - consume(TOK_EQUAL, "Expected '=' after struct name"); - consume(TOK_STRUCT, "Expected 'struct' keyword"); - consume(TOK_OPEN_BRACE, "Expected '{' after 'struct'"); - - std::vector> fields; - std::vector> methods; +void Parser::parseStructBody( + std::vector> &fields, + std::vector> &methods) { while (!check(TOK_CLOSE_BRACE) && !isAtEnd()) { // Method: `fn name(self: ..., ...) ReturnType { body }`. Methods // can appear in any order relative to fields. parseFunction @@ -923,12 +983,50 @@ std::unique_ptr Parser::parseStructDecl() { } } consume(TOK_CLOSE_BRACE, "Expected '}' to close struct definition"); +} + +std::unique_ptr Parser::parseStructDecl() { + consume(TOK_CONST, "Expected 'const' for struct declaration"); + consume(TOK_IDENTIFIER, "Expected struct name"); + std::string name = previous().lexeme; + consume(TOK_EQUAL, "Expected '=' after struct name"); + consume(TOK_STRUCT, "Expected 'struct' keyword"); + consume(TOK_OPEN_BRACE, "Expected '{' after 'struct'"); + + std::vector> fields; + std::vector> methods; + structContextStack.push_back(name); // G3: Self resolves to this name + parseStructBody(fields, methods); + structContextStack.pop_back(); consume(TOK_SEMI, "Expected ';' after struct declaration"); return std::make_unique(name, std::move(fields), std::move(methods)); } +NodeIdx Parser::parseStructExpression() { + // Caller has matched the `struct` keyword. Anonymous struct: synthetic + // name keyed by index in anonStructs (the substitution engine in G4 + // reads from there to instantiate the body with concrete type args). + consume(TOK_OPEN_BRACE, "Expected '{' after 'struct'"); + uint32_t idx = static_cast(anonStructs.size()); + std::string name = "__anon_struct_" + std::to_string(idx); + + std::vector> fields; + std::vector> methods; + // G3: Self inside this anonymous struct's body resolves to the + // synthetic name. The substitution engine in G4 will rewrite that + // name to the per-instantiation struct name. + structContextStack.push_back(name); + parseStructBody(fields, methods); + structContextStack.pop_back(); + + anonStructs.push_back(std::make_unique( + std::move(name), std::move(fields), std::move(methods))); + + return emit(AstNode{AstTag::StructExpr, 0, 0, 0, idx, 0}); +} + // Parse `const Name = enum { Variant1, Variant2(T1, T2), ... };`. // Variants get sequential discriminant values starting from zero in // declaration order. Unit variants (no payload) and tagged variants @@ -1078,6 +1176,35 @@ std::unique_ptr Parser::parseConstDecl() { if (match(TOK_COLON)) { declared = parseType(); } consume(TOK_EQUAL, "Expected '=' in module-scope const declaration"); + + // Generics G4: try-parse the RHS as a type. If it parses as a + // `TypeKind::GenericCall` (e.g. `Box(i32)`) and is followed by a + // semicolon, treat the const as a type alias. Anything else + // rewinds and parses as a value expression. Save/restore the + // parser cursor since parseType can throw on non-type inputs. + int saved = current; + TypeIdx aliased = kNoType; + try { + TypeIdx maybeType = parseType(); + if (typePool->get(maybeType).kind == TypeKind::GenericCall && + check(TOK_SEMI)) { + aliased = maybeType; + } else { + current = saved; + } + } catch (...) { + current = saved; + } + + if (aliased != kNoType) { + consume(TOK_SEMI, + "Expected ';' after module-scope const declaration"); + auto decl = std::make_unique(std::move(name), + declared, kNoNode); + decl->AliasedType = aliased; + return decl; + } + NodeIdx init = parseLogicalOr(); consume(TOK_SEMI, "Expected ';' after module-scope const declaration"); @@ -1143,5 +1270,10 @@ std::unique_ptr Parser::parse() { module->Functions.push_back(parseFunction()); } + // Generics G2: hand off anonymous struct bodies (`struct {...}` + // expressions encountered during parse) to the ModuleAST so the + // instantiation engine can find them by index. + module->AnonStructs = std::move(anonStructs); + return module; } diff --git a/src/parser.h b/src/parser.h index da6fe1a..a6f3402 100644 --- a/src/parser.h +++ b/src/parser.h @@ -44,6 +44,7 @@ class Parser { NodeIdx parseAddition(); NodeIdx parseMultiplication(); NodeIdx parseStructLiteral(); + NodeIdx parseStructExpression(); NodeIdx parseMatch(); NodeIdx parsePattern(); // OrPattern NodeIdx parsePatternAtom(); // single-atom pattern @@ -51,6 +52,24 @@ class Parser { std::unique_ptr parseImportDecl(); std::unique_ptr parseDestructuringImport(); std::unique_ptr parseStructDecl(); + // Generics G2: parses the `{ field: T, fn method(self: ...) {...} }` + // body shared between top-level struct decls and anonymous struct + // expressions. Caller must have already consumed the `struct` keyword + // and the opening `{`. On return, the closing `}` has been consumed. + void parseStructBody( + std::vector> &fields, + std::vector> &methods); + + // Generics G2: anonymous structs created by `struct { ... }` + // expressions. Stored here during parse and transferred to + // `ModuleAST::AnonStructs` at the end of parse(). + std::vector> anonStructs; + + // Generics G3: stack of struct names being parsed, used to resolve + // `Self` references inside method signatures. Top-level structs push + // their declared name; struct expressions push their synthetic + // `__anon_struct_` name. Empty when not inside any struct body. + std::vector structContextStack; std::unique_ptr parseUnionDecl(); std::unique_ptr parseEnumDecl(); std::unique_ptr parseConstDecl(); diff --git a/src/token.h b/src/token.h index 1c6688b..6740e44 100644 --- a/src/token.h +++ b/src/token.h @@ -69,8 +69,6 @@ enum TokenType { TOK_TILDE, // ~ (bitwise NOT) TOK_LSHIFT, // << (left shift) TOK_RSHIFT, // >> (right shift) - TOK_UNDEFINED, // undefined keyword (uninitialized storage marker; - // also a parameter mode for write-to-uninit destinations) TOK_MOVE, // move keyword (parameter mode: consume ownership) TOK_ELLIPSIS, // ... (variadic marker in extern fn parameters) TOK_MATCH, // match keyword diff --git a/tests/cpp/test_abi.cpp b/tests/cpp/test_abi.cpp index cae82e5..0659cf2 100644 --- a/tests/cpp/test_abi.cpp +++ b/tests/cpp/test_abi.cpp @@ -7,7 +7,7 @@ // the resulting ParamABI / ReturnABI shape. // // Reference behavior (from docs/ABI.md §3): -// mut / undefined → ByPointer (any size) +// mut → ByPointer (any size) // let / move, scalar T → ByValue // let / move, aggregate // size <= 16 bytes → ByValue @@ -65,13 +65,6 @@ void testMutU32IsByPointer() { ASSERT_EQ(static_cast(4), a.pointerAlign); } -void testUndefinedU64IsByPointer() { - JamCodegenContext ctx("test"); - auto a = jam::abi::classifyParam(ParamMode::Undefined, BuiltinType::U64, ctx); - ASSERT_TRUE(a.kind == jam::abi::ParamABI::Kind::ByPointer); - ASSERT_EQ(static_cast(8), a.pointerAlign); -} - void testMoveU8IsByValueScalar() { JamCodegenContext ctx("test"); auto a = jam::abi::classifyParam(ParamMode::Move, BuiltinType::U8, ctx); @@ -205,8 +198,6 @@ class ABITests { testLetU32IsByValueScalar); framework.addTest("ABI classifyParam - mut u32 ByPointer align 4", testMutU32IsByPointer); - framework.addTest("ABI classifyParam - undefined u64 ByPointer align 8", - testUndefinedU64IsByPointer); framework.addTest("ABI classifyParam - move u8 ByValue", testMoveU8IsByValueScalar); framework.addTest("ABI classifyParam - let small struct ByValue", diff --git a/tests/cpp/test_codegen_errors.cpp b/tests/cpp/test_codegen_errors.cpp new file mode 100644 index 0000000..3377bd6 --- /dev/null +++ b/tests/cpp/test_codegen_errors.cpp @@ -0,0 +1,170 @@ +// Codegen-time must-fail tests. +// +// The init-analyzer test suite (test_init_analysis.cpp) covers errors +// reported by analysis on a parsed module. This file covers errors that +// surface during *codegen* — primarily generic instantiation failures +// like "type T has no method default" — by invoking the jam.out +// binary as a subprocess and asserting on its stderr. +// +// Subprocess approach (rather than driving JamCodegenContext in-process) +// avoids replicating the LLVM-init / target-machine / drop-registry +// scaffolding that main.cpp builds end-to-end. Each test writes a Jam +// source file to /tmp, runs the compiler, captures stderr+exit, and +// asserts on the returned message. + +#include "test_framework.h" +#include +#include +#include +#include +#include + +namespace { + +struct CompileResult { + int exitCode; + std::string stderr_; +}; + +// Run jam.out on a one-off source string. The binary must already be +// built (the Makefile target depends on `build`). We invoke from the +// project root so jam.out's relative paths resolve correctly. +CompileResult compileSource(const std::string &name, + const std::string &source) { + std::string path = "/tmp/" + name + ".jam"; + { + std::ofstream out(path); + out << source; + } + + // Redirect stderr→stdout so popen captures both. jam.out usually + // only writes to stderr on error, but this is robust either way. + std::string cmd = "./jam.out " + path + " 2>&1"; + + std::string output; + FILE *pipe = popen(cmd.c_str(), "r"); + if (!pipe) { + throw std::runtime_error("popen failed: " + cmd); + } + char buf[256]; + while (fgets(buf, sizeof(buf), pipe) != nullptr) output += buf; + int status = pclose(pipe); + + int exitCode = WIFEXITED(status) ? WEXITSTATUS(status) : -1; + return {exitCode, std::move(output)}; +} + +bool stderrContains(const CompileResult &r, const std::string &substr) { + return r.stderr_.find(substr) != std::string::npos; +} + +} // namespace + +class CodegenErrorTests { + public: + static void registerAllTests(TestFramework &framework) { + framework.addTest("Codegen - Maybe(T) where T lacks default()", + testMaybeOfTypeWithoutDefault); + framework.addTest("Codegen - default() with parameters rejected", + testDefaultWithParameters); + framework.addTest("Codegen - default() with wrong return type", + testDefaultWrongReturnType); + framework.addTest("Codegen - non-drop non-default method on top-level", + testForbiddenTopLevelMethod); + } + + private: + // A generic body that calls T.default() must instantiate to a + // concrete T that has a default() method. NoDefault doesn't, so + // instantiation should error with a precise message naming both + // the missing method and the type that's missing it. + static void testMaybeOfTypeWithoutDefault() { + auto r = compileSource("must_fail_no_default", R"( +const NoDefault = struct { + n: i32, +}; + +fn Maybe(T: type) type { + return struct { + storage: T, + valid: bool, + fn default() Self { + return { storage: T.default(), valid: false }; + } + }; +} + +const MaybeND = Maybe(NoDefault); + +fn main() i32 { + var m: MaybeND = MaybeND.default(); + return m.storage.n; +} +)"); + ASSERT_TRUE(r.exitCode != 0); + ASSERT_TRUE(stderrContains(r, "NoDefault")); + ASSERT_TRUE(stderrContains(r, "default")); + } + + // `default` on a top-level struct must take no parameters. The + // validation in main.cpp specifically checks Args.empty(). + static void testDefaultWithParameters() { + auto r = compileSource("must_fail_default_with_params", R"( +const Bad = struct { + n: i32, + fn default(self: mut Self) Self { + return { n: 0 }; + } +}; + +fn main() i32 { return 0; } +)"); + ASSERT_TRUE(r.exitCode != 0); + ASSERT_TRUE(stderrContains(r, "default")); + ASSERT_TRUE(stderrContains(r, "no parameters")); + } + + // `default` must return Self (the enclosing struct's type). + // Returning anything else is a typing error in the contract. + static void testDefaultWrongReturnType() { + auto r = compileSource("must_fail_default_wrong_return", R"( +const Bad = struct { + n: i32, + fn default() i32 { + return 0; + } +}; + +fn main() i32 { return 0; } +)"); + ASSERT_TRUE(r.exitCode != 0); + ASSERT_TRUE(stderrContains(r, "default")); + ASSERT_TRUE(stderrContains(r, "Self")); + } + + // Top-level structs only allow `drop` and `default` methods. Other + // names (e.g. `unwrap`) get a clear "not allowed" error so users + // don't think method-as-namespace works on plain structs. + static void testForbiddenTopLevelMethod() { + auto r = compileSource("must_fail_other_method", R"( +const Bad = struct { + n: i32, + fn unwrap(self: mut Self) i32 { + return self.n; + } +}; + +fn main() i32 { return 0; } +)"); + ASSERT_TRUE(r.exitCode != 0); + ASSERT_TRUE(stderrContains(r, "drop")); + ASSERT_TRUE(stderrContains(r, "default")); + } +}; + +int main() { + TestFramework framework; + CodegenErrorTests::registerAllTests(framework); + framework.runAll(); + return framework.allPassed() ? 0 : 1; +} diff --git a/tests/cpp/test_init_analysis.cpp b/tests/cpp/test_init_analysis.cpp index f1c3132..f4b3428 100644 --- a/tests/cpp/test_init_analysis.cpp +++ b/tests/cpp/test_init_analysis.cpp @@ -1,4 +1,4 @@ -// In-process tests for the MVS init analyzer (P2 through P5.5). +// In-process tests for the MVS init analyzer (P4 through P8.2). // // Each test compiles a Jam source string through lexer + parser, runs // init_analysis::analyze on every function in the parsed module, and @@ -81,102 +81,6 @@ bool diagsAbout(const std::vector &diags, return false; } -// ========================================================================= -// P2 — definite initialization -// ========================================================================= - -void testUninitRead() { - auto r = analyzeSource(R"( -fn f() u32 { - var x: u32 = undefined; - return x; -} -)"); - ASSERT_EQ(static_cast(1), r.diagnostics.size()); - ASSERT_CONTAINS(r.diagnostics[0].message, "uninitialized"); - ASSERT_EQ(std::string("x"), r.diagnostics[0].varName); -} - -void testPartialBranchInit() { - auto r = analyzeSource(R"( -fn f(b: bool) u32 { - var x: u32 = undefined; - if (b) { - x = 100; - } - return x; -} -)"); - ASSERT_EQ(static_cast(1), r.diagnostics.size()); - ASSERT_CONTAINS(r.diagnostics[0].message, - "may not be initialized on every path"); - ASSERT_EQ(std::string("x"), r.diagnostics[0].varName); -} - -void testStraightLineOK() { - auto r = analyzeSource(R"( -fn f() u32 { - var x: u32 = undefined; - x = 5; - return x; -} -)"); - ASSERT_EQ(static_cast(0), r.diagnostics.size()); -} - -void testIfElseBothInitOK() { - auto r = analyzeSource(R"( -fn f(b: bool) u32 { - var x: u32 = undefined; - if (b) { - x = 1; - } else { - x = 2; - } - return x; -} -)"); - ASSERT_EQ(static_cast(0), r.diagnostics.size()); -} - -// ========================================================================= -// P3 — undefined-mode parameter rules -// ========================================================================= - -void testUndefinedParamUnset() { - auto r = analyzeSource(R"( -fn f(out: undefined u32) u32 { - return 42; -} -)"); - ASSERT_EQ(static_cast(1), r.diagnostics.size()); - ASSERT_CONTAINS(r.diagnostics[0].message, - "`undefined`-mode parameter `out` was not initialized"); -} - -void testUndefinedParamReadBeforeWrite() { - auto r = analyzeSource(R"( -fn f(out: undefined u32) u32 { - return out; -} -)"); - // Two diagnostics expected: read-of-uninit AND undefined-param-not-init. - // The second is redundant but informative; both are produced. - ASSERT_TRUE(diagsContain(r.diagnostics, "use of uninitialized binding")); - ASSERT_TRUE(diagsContain(r.diagnostics, "`undefined`-mode parameter")); - ASSERT_TRUE(diagsAbout(r.diagnostics, "out")); -} - -void testUndefinedParamInitOK() { - auto r = analyzeSource(R"( -fn f(out: undefined u32) u32 { - out = 42; - return out; -} -)"); - ASSERT_EQ(static_cast(0), r.diagnostics.size()); -} - // ========================================================================= // P4 — callsite mode propagation // ========================================================================= @@ -211,19 +115,6 @@ fn caller() u32 { ASSERT_EQ(std::string("x"), r.diagnostics[0].varName); } -void testUninitLetArg() { - auto r = analyzeSource(R"( -fn read(x: u32) u32 { return x; } - -fn caller() u32 { - var x: u32 = undefined; - return read(x); -} -)"); - ASSERT_EQ(static_cast(1), r.diagnostics.size()); - ASSERT_CONTAINS(r.diagnostics[0].message, "uninitialized"); -} - void testMoveThenSeparateBindingOK() { auto r = analyzeSource(R"( fn consume(buf: move u32) u32 { return buf; } @@ -278,9 +169,7 @@ const Pair = struct { a: u32, b: u32 }; fn modifyAndRead(whole: mut Pair, part: u32) u32 { return part; } fn caller() u32 { - var p: Pair = undefined; - p.a = 1; - p.b = 2; + var p: Pair = { a: 1, b: 2 }; return modifyAndRead(p, p.a); } )"); @@ -295,9 +184,7 @@ const Pair = struct { a: u32, b: u32 }; fn add(a: u32, b: u32) u32 { return a + b; } fn caller() u32 { - var p: Pair = undefined; - p.a = 10; - p.b = 20; + var p: Pair = { a: 10, b: 20 }; return add(p.a, p.b); } )"); @@ -331,18 +218,6 @@ fn dangleField(p: mut Pair) *mut u32 { ASSERT_CONTAINS(r.diagnostics[0].message, "cannot return `&` of `mut`-mode"); } -void testEscapeUndefinedParam() { - auto r = analyzeSource(R"( -fn dangleUndef(out: undefined u32) *mut u32 { - out = 5; - return &out; -} -)"); - ASSERT_EQ(static_cast(1), r.diagnostics.size()); - ASSERT_CONTAINS(r.diagnostics[0].message, - "cannot return `&` of `undefined`-mode"); -} - void testReturnMutParamByValueOK() { // `return p;` (no `&`) is a value copy under value semantics — not an // escape. The check must NOT flag this. @@ -378,8 +253,7 @@ fn consume(f: move File) i32 { } fn caller() i32 { - var f: File = undefined; - f.fd = 7; + var f: File = { fd: 7 }; return consume(f); } )"); @@ -404,8 +278,7 @@ fn read(f: File) i32 { } fn caller() i32 { - var f: File = undefined; - f.fd = 7; + var f: File = { fd: 7 }; return read(f); } )"); @@ -427,8 +300,7 @@ fn close(f: mut File) { } fn caller() i32 { - var f: File = undefined; - f.fd = 7; + var f: File = { fd: 7 }; close(&f); return f.fd; } @@ -436,77 +308,6 @@ fn caller() i32 { ASSERT_EQ(static_cast(0), r.diagnostics.size()); } -// ========================================================================= -// P8.2b — drop-bearing locals must be Init at every exit -// ========================================================================= - -void testDropBearingUninitAtReturnRejected() { - auto r = analyzeSource(R"( -const File = struct { fd: i32 }; -fn drop(self: mut File) { self.fd = 0; } - -fn forgotInit() i32 { - var f: File = undefined; - return 42; -} -)"); - ASSERT_TRUE(diagsContain(r.diagnostics, "drop-bearing binding")); - ASSERT_TRUE(diagsAbout(r.diagnostics, "f")); -} - -void testDropBearingPartialBranchInitRejected() { - auto r = analyzeSource(R"( -const File = struct { fd: i32 }; -fn drop(self: mut File) { self.fd = 0; } - -fn maybe(b: bool) i32 { - var f: File = undefined; - if (b) { - f.fd = 7; - } - return 0; -} -)"); - ASSERT_TRUE(diagsContain(r.diagnostics, "drop-bearing binding")); - ASSERT_TRUE(diagsContain(r.diagnostics, - "may not be initialized on every path")); - ASSERT_TRUE(diagsAbout(r.diagnostics, "f")); -} - -void testDropBearingFullyInitOK() { - // Initialized via partial-write rule (every path touches f) → Init at - // exit, no rejection. - auto r = analyzeSource(R"( -const File = struct { fd: i32 }; -fn drop(self: mut File) { self.fd = 0; } - -fn ok() i32 { - var f: File = undefined; - f.fd = 9; - return 0; -} -)"); - ASSERT_EQ(static_cast(0), r.diagnostics.size()); -} - -void testDropBearingInitOnAllBranchesOK() { - auto r = analyzeSource(R"( -const File = struct { fd: i32 }; -fn drop(self: mut File) { self.fd = 0; } - -fn ok(b: bool) i32 { - var f: File = undefined; - if (b) { - f.fd = 1; - } else { - f.fd = 2; - } - return 0; -} -)"); - ASSERT_EQ(static_cast(0), r.diagnostics.size()); -} - void testNonDropBearingMoveOK() { // Non-drop-bearing struct can still be moved freely (no double-free // risk because no drop runs at scope exit). @@ -521,9 +322,7 @@ fn consume(p: move Plain) u32 { } fn caller() u32 { - var p: Plain = undefined; - p.a = 1; - p.b = 2; + var p: Plain = { a: 1, b: 2 }; return consume(p); } )"); @@ -535,29 +334,10 @@ fn caller() u32 { class InitAnalysisTests { public: static void registerAllTests(TestFramework &framework) { - // P2 — definite-init - framework.addTest("InitAnalysis P2 - uninit read", testUninitRead); - framework.addTest("InitAnalysis P2 - partial branch init", - testPartialBranchInit); - framework.addTest("InitAnalysis P2 - straight line OK", - testStraightLineOK); - framework.addTest("InitAnalysis P2 - if/else both init OK", - testIfElseBothInitOK); - - // P3 — undefined-mode parameter - framework.addTest("InitAnalysis P3 - undefined param unset", - testUndefinedParamUnset); - framework.addTest("InitAnalysis P3 - undefined param read before write", - testUndefinedParamReadBeforeWrite); - framework.addTest("InitAnalysis P3 - undefined param init OK", - testUndefinedParamInitOK); - // P4 — callsite mode propagation framework.addTest("InitAnalysis P4 - read after move", testReadAfterMove); framework.addTest("InitAnalysis P4 - double move", testDoubleMove); - framework.addTest("InitAnalysis P4 - uninit let arg", - testUninitLetArg); framework.addTest("InitAnalysis P4 - move then separate binding OK", testMoveThenSeparateBindingOK); @@ -576,8 +356,6 @@ class InitAnalysisTests { testEscapeMutParam); framework.addTest("InitAnalysis P5.5 - escape &mut field", testEscapeMutField); - framework.addTest("InitAnalysis P5.5 - escape &undefined param", - testEscapeUndefinedParam); framework.addTest("InitAnalysis P5.5 - mut param by-value return OK", testReturnMutParamByValueOK); @@ -590,16 +368,6 @@ class InitAnalysisTests { testMutOnDropBearingOK); framework.addTest("InitAnalysis P8 - move on non-drop OK", testNonDropBearingMoveOK); - - // P8.2 — drop-bearing locals must be Init at every exit - framework.addTest("InitAnalysis P8.2 - uninit at return rejected", - testDropBearingUninitAtReturnRejected); - framework.addTest("InitAnalysis P8.2 - partial branch init rejected", - testDropBearingPartialBranchInitRejected); - framework.addTest("InitAnalysis P8.2 - fully init OK", - testDropBearingFullyInitOK); - framework.addTest("InitAnalysis P8.2 - init on all branches OK", - testDropBearingInitOnAllBranchesOK); } }; diff --git a/tests/unit/test_abi_extern_mixed.jam b/tests/unit/test_abi_extern_mixed.jam index 2e44c72..9baae0d 100644 --- a/tests/unit/test_abi_extern_mixed.jam +++ b/tests/unit/test_abi_extern_mixed.jam @@ -20,8 +20,7 @@ fn fillFromExtern(buf: mut [16]u8) i32 { } tfn fillFromExternOK() { - var buf: [16]u8 = undefined; - buf[0] = 0; + var buf: [16]u8 = [0; 16]; var n: i32 = fillFromExtern(&buf); // snprintf wrote "n=42" (4 chars) into the buffer. assert(n, 4); @@ -42,10 +41,8 @@ const Stats = struct { }; fn statsOf(s: *const[] u8) Stats { - var out: Stats = undefined; - out.a = strlen(s); - out.b = out.a * 2; - out.c = out.a * 3; + var len: u64 = strlen(s); + var out: Stats = { a: len, b: len * 2, c: len * 3 }; return out; } @@ -73,11 +70,7 @@ fn fillField(b: mut Bag) i32 { } tfn fillFieldOK() { - var b: Bag = undefined; - b.one = 0; - b.two = 0; - b.three = 0; - b.four = 0; + var b: Bag = { one: 0, two: 0, three: 0, four: 0 }; var n: i32 = fillField(&b); assert(n, 2); assert(b.one, 104); // 'h' @@ -87,10 +80,10 @@ tfn fillFieldOK() { assert(b.four, 0); } -// Multi-mode function: one let, one mut, one undefined; calls two -// extern fns in its body. Confirms the ABI machinery doesn't get -// confused when several mode kinds and FFI calls coexist in one fn. -fn multimode(prefix: u32, target: mut [8]u8, out: undefined u32) i32 { +// Multi-mode function: one let, one mut, one mut-out; calls two extern +// fns in its body. Confirms the ABI machinery doesn't get confused when +// several mode kinds and FFI calls coexist in one fn. +fn multimode(prefix: u32, target: mut [8]u8, out: mut u32) i32 { var written: i32 = snprintf(&target[0], 8, "p=%d", prefix); var len: u64 = strlen(&target[0]); out = len as u32; @@ -98,9 +91,8 @@ fn multimode(prefix: u32, target: mut [8]u8, out: undefined u32) i32 { } tfn multimodeOK() { - var buf: [8]u8 = undefined; - buf[0] = 0; - var written: u32 = undefined; + var buf: [8]u8 = [0; 8]; + var written: u32 = 0; var n: i32 = multimode(7, &buf, &written); assert(n, 3); // "p=7" is 3 chars assert(written as u32, 3); // strlen of "p=7" is 3 diff --git a/tests/unit/test_abi_large.jam b/tests/unit/test_abi_large.jam index 47f6a92..85dc0d8 100644 --- a/tests/unit/test_abi_large.jam +++ b/tests/unit/test_abi_large.jam @@ -42,38 +42,23 @@ fn pickField(prefix: u32, b: Big) u64 { } tfn sumBigOK() { - var x: Big = undefined; - x.a = 10; - x.b = 20; - x.c = 30; + var x: Big = { a: 10, b: 20, c: 30 }; assert(sumBig(x), 60); } tfn sumWideOK() { - var w: Wide = undefined; - w.a = 1; - w.b = 2; - w.c = 3; - w.d = 4; - w.e = 5; - w.f = 6; + var w: Wide = { a: 1, b: 2, c: 3, d: 4, e: 5, f: 6 }; assert(sumWide(w), 21); } tfn consumeBigOK() { - var x: Big = undefined; - x.a = 100; - x.b = 200; - x.c = 300; + var x: Big = { a: 100, b: 200, c: 300 }; assert(consumeBig(x), 600); // After consumeBig, x is moved-out — the analyzer would reject any // subsequent read of x. We just don't read it. } tfn pickFieldOK() { - var x: Big = undefined; - x.a = 0; - x.b = 42; - x.c = 0; + var x: Big = { a: 0, b: 42, c: 0 }; assert(pickField(8, x), 50); } diff --git a/tests/unit/test_abi_sret.jam b/tests/unit/test_abi_sret.jam index 8d2e5b2..d8ec095 100644 --- a/tests/unit/test_abi_sret.jam +++ b/tests/unit/test_abi_sret.jam @@ -14,10 +14,7 @@ const Big = struct { }; fn makeBig(seed: u64) Big { - var b: Big = undefined; - b.a = seed; - b.b = seed * 2; - b.c = seed * 3; + var b: Big = { a: seed, b: seed * 2, c: seed * 3 }; return b; } @@ -30,10 +27,7 @@ fn sumOfMakeBig(seed: u64) u64 { // sret callee that takes a large parameter (P9.5 + P9.6 together). fn passThroughBig(b: Big) Big { - var copy: Big = undefined; - copy.a = b.a + 1; - copy.b = b.b + 1; - copy.c = b.c + 1; + var copy: Big = { a: b.a + 1, b: b.b + 1, c: b.c + 1 }; return copy; } @@ -50,10 +44,7 @@ tfn sumOfMakeBigOK() { } tfn passThroughBigOK() { - var src: Big = undefined; - src.a = 100; - src.b = 200; - src.c = 300; + var src: Big = { a: 100, b: 200, c: 300 }; var got: Big = passThroughBig(src); assert(got.a, 101); assert(got.b, 201); diff --git a/tests/unit/test_array.jam b/tests/unit/test_array.jam index 55f6ff0..556deb6 100644 --- a/tests/unit/test_array.jam +++ b/tests/unit/test_array.jam @@ -6,7 +6,7 @@ const Game = struct { }; fn writeThenReadU8() u8 { - var arr: [10]u8 = undefined; + var arr: [10]u8 = [0; 10]; arr[0] = 42; arr[1] = 7; arr[9] = 200; @@ -14,7 +14,7 @@ fn writeThenReadU8() u8 { } fn secondElement() u8 { - var arr: [10]u8 = undefined; + var arr: [10]u8 = [0; 10]; arr[0] = 42; arr[1] = 7; arr[9] = 200; @@ -22,7 +22,7 @@ fn secondElement() u8 { } fn lastElement() u8 { - var arr: [10]u8 = undefined; + var arr: [10]u8 = [0; 10]; arr[0] = 42; arr[1] = 7; arr[9] = 200; @@ -30,7 +30,7 @@ fn lastElement() u8 { } fn sumFirstFour() u32 { - var arr: [10]u32 = undefined; + var arr: [10]u32 = [0; 10]; arr[0] = 100; arr[1] = 200; arr[2] = 300; @@ -39,17 +39,13 @@ fn sumFirstFour() u32 { } fn dynamicIndexRead() u8 { - var arr: [4]u8 = undefined; - arr[0] = 10; - arr[1] = 20; - arr[2] = 30; - arr[3] = 40; + var arr: [4]u8 = [10, 20, 30, 40]; var i: u8 = 2; return arr[i]; } fn loopFillThenRead() u8 { - var arr: [16]u8 = undefined; + var arr: [16]u8 = [0; 16]; for i in 0:16 { arr[i] = i; } @@ -57,7 +53,7 @@ fn loopFillThenRead() u8 { } fn loopFillThenReadLast() u8 { - var arr: [16]u8 = undefined; + var arr: [16]u8 = [0; 16]; for i in 0:16 { arr[i] = i; } @@ -65,13 +61,13 @@ fn loopFillThenReadLast() u8 { } fn structFieldArrayWrite() u8 { - var g: Game = undefined; + var g: Game = { score: 0, board: [0; 10] }; g.board[3] = 42; return g.board[3]; } fn structFieldArrayLoopFill() u8 { - var g: Game = undefined; + var g: Game = { score: 0, board: [0; 10] }; for i in 0:10 { g.board[i] = i; } @@ -79,8 +75,7 @@ fn structFieldArrayLoopFill() u8 { } fn structFieldArrayMixed() u8 { - var g: Game = undefined; - g.score = 99; + var g: Game = { score: 99, board: [0; 10] }; g.board[0] = g.score; g.board[1] = 50; return g.board[0]; diff --git a/tests/unit/test_drops.jam b/tests/unit/test_drops.jam index 1addc7c..98c4fac 100644 --- a/tests/unit/test_drops.jam +++ b/tests/unit/test_drops.jam @@ -23,9 +23,7 @@ fn drop(self: mut Counter) { fn singleScope() u32 { var hits: u32 = 0; - var c: Counter = undefined; - c.value = 7; - c.sink = &hits; + var c: Counter = { value: 7, sink: &hits }; // c drops here at function end → hits becomes 1 return c.value + 100; } @@ -38,12 +36,8 @@ fn callerObservingDrop() u32 { } fn wrapsCounter(sink: *mut u32) u32 { - var c: Counter = undefined; - c.value = 5; - c.sink = sink; - var d: Counter = undefined; - d.value = 11; - d.sink = sink; + var c: Counter = { value: 5, sink: sink }; + var d: Counter = { value: 11, sink: sink }; // Both c and d drop at function exit (reverse-decl: d first, then c). // Each drop bumps `*sink` by 1 → sink becomes 2. return c.value + d.value; @@ -56,9 +50,7 @@ fn earlyReturnDropsToo(b: bool) u32 { } fn early(b: bool, sink: *mut u32) u32 { - var c: Counter = undefined; - c.value = 13; - c.sink = sink; + var c: Counter = { value: 13, sink: sink }; if (b) { return c.value; } diff --git a/tests/unit/test_drops_loops.jam b/tests/unit/test_drops_loops.jam index 369a429..3255fb6 100644 --- a/tests/unit/test_drops_loops.jam +++ b/tests/unit/test_drops_loops.jam @@ -22,8 +22,7 @@ fn breakDropsBody() u32 { fn breakInner(sink: *mut u32) { for i in 0:10 { - var x: Bumper = undefined; - x.sink = sink; + var x: Bumper = { sink: sink }; if (i == 3) { // x must drop before the break exits the loop → bumps once break; @@ -41,8 +40,7 @@ fn continueDropsBody() u32 { fn continueInner(sink: *mut u32) { for i in 0:5 { - var x: Bumper = undefined; - x.sink = sink; + var x: Bumper = { sink: sink }; if (i == 2) { // x must drop before the continue jumps to next iter continue; @@ -61,11 +59,9 @@ fn breakNestedScopes() u32 { fn breakNestedInner(sink: *mut u32) { for i in 0:10 { - var outer: Bumper = undefined; - outer.sink = sink; + var outer: Bumper = { sink: sink }; if (i == 1) { - var inner: Bumper = undefined; - inner.sink = sink; + var inner: Bumper = { sink: sink }; // break should drop: inner first, then outer, then exit // → bumps twice on this path break; diff --git a/tests/unit/test_drops_mangling.jam b/tests/unit/test_drops_mangling.jam index 305eacd..1bbe4b1 100644 --- a/tests/unit/test_drops_mangling.jam +++ b/tests/unit/test_drops_mangling.jam @@ -29,8 +29,7 @@ fn dropOneA() u32 { } fn onlyA(sink: *mut u32) u32 { - var a: A = undefined; - a.sink = sink; + var a: A = { sink: sink }; return 1; } @@ -41,8 +40,7 @@ fn dropOneB() u32 { } fn onlyB(sink: *mut u32) u32 { - var b: B = undefined; - b.sink = sink; + var b: B = { sink: sink }; return 2; } @@ -53,10 +51,8 @@ fn dropBoth() u32 { } fn bothInOne(sink: *mut u32) u32 { - var a: A = undefined; - a.sink = sink; - var b: B = undefined; - b.sink = sink; + var a: A = { sink: sink }; + var b: B = { sink: sink }; return 3; } diff --git a/tests/unit/test_drops_scoped.jam b/tests/unit/test_drops_scoped.jam index d8d87ec..f942cb8 100644 --- a/tests/unit/test_drops_scoped.jam +++ b/tests/unit/test_drops_scoped.jam @@ -17,8 +17,7 @@ fn drop(self: mut Bumper) { fn ifBodyDropsAtEnd(b: bool) u32 { var hits: u32 = 0; if (b) { - var x: Bumper = undefined; - x.sink = &hits; + var x: Bumper = { sink: &hits }; // x goes out of scope at end of this block → drop runs → hits=1 } // We can read hits here; if drop ran at end of if, hits == 1. @@ -34,8 +33,7 @@ fn elseBodyDropsAtEnd() u32 { if (false) { var ignored: u32 = 0; } else { - var x: Bumper = undefined; - x.sink = &hits; + var x: Bumper = { sink: &hits }; // drop at end of else → hits=1 } return hits; @@ -45,8 +43,7 @@ fn elseBodyDropsAtEnd() u32 { fn loopIterationDrops() u32 { var hits: u32 = 0; for i in 0:5 { - var x: Bumper = undefined; - x.sink = &hits; + var x: Bumper = { sink: &hits }; // drop at end of iteration body → hits++ each pass } return hits; @@ -55,11 +52,9 @@ fn loopIterationDrops() u32 { // --- Outer drop fires at function end; inner drop fires at end of if --- fn nested(b: bool) u32 { var hits: u32 = 0; - var outer: Bumper = undefined; - outer.sink = &hits; + var outer: Bumper = { sink: &hits }; if (b) { - var inner: Bumper = undefined; - inner.sink = &hits; + var inner: Bumper = { sink: &hits }; // inner drops at end of if → hits=1 } var afterIf: u32 = hits; @@ -70,11 +65,9 @@ fn nested(b: bool) u32 { // --- Return inside an if body: ALL active scopes drop before ret --- fn earlyReturnDropsAll() u32 { var hits: u32 = 0; - var outer: Bumper = undefined; - outer.sink = &hits; + var outer: Bumper = { sink: &hits }; if (true) { - var inner: Bumper = undefined; - inner.sink = &hits; + var inner: Bumper = { sink: &hits }; return hits; // never reached } diff --git a/tests/unit/test_generics_basic.jam b/tests/unit/test_generics_basic.jam new file mode 100644 index 0000000..7ab0f2b --- /dev/null +++ b/tests/unit/test_generics_basic.jam @@ -0,0 +1,324 @@ +const { assert } = import("test"); + +// Generics G1–G4: type-parameter generics with substitution-only +// instantiation. A function whose signature is `fn F(T: type) type` is +// a generic; calling it with concrete type arguments at compile time +// produces a concrete type. Each distinct argument list gives a fresh +// struct type with substituted field types. + +fn Identity(T: type) type { + return T; +} + +tfn identityResolvesToArg() { + var a: Identity(i32) = 42; + var b: Identity(u8) = 7; + assert(a, 42); + assert(b as i32, 7); +} + +fn Box(T: type) type { + return struct { + value: T, + }; +} + +tfn boxOfI32() { + var b: Box(i32) = { value: 17 }; + assert(b.value, 17); +} + +tfn boxOfU8DistinctFromBoxOfI32() { + var bi: Box(i32) = { value: 1 }; + var bu: Box(u8) = { value: 2 }; + // Both bindings hold their own struct types with their own field + // sizes (4 bytes vs 1 byte). Same field name; different storage. + assert(bi.value, 1); + assert(bu.value as i32, 2); +} + +fn Pair(A: type, B: type) type { + return struct { + first: A, + second: B, + }; +} + +tfn twoParamGeneric() { + var p: Pair(i32, u8) = { first: 7, second: 35 }; + assert(p.first + (p.second as i32), 42); +} + +tfn sameInstantiationReusesType() { + var p1: Pair(i32, u8) = { first: 1, second: 2 }; + var p2: Pair(i32, u8) = { first: 3, second: 4 }; + // Both bindings should share the same instantiated struct type + // (verified by the compiler accepting a homogeneous-type comparison + // through the field accesses below). + assert(p1.first + p2.first, 4); + assert((p1.second + p2.second) as i32, 6); +} + +// G4 type alias: `const Name = Generic(arg);` registers Name as a +// synonym for the instantiated type. Subsequent uses of Name resolve +// to the same struct. +const BoxI32 = Box(i32); + +tfn typeAliasResolvesToInstantiation() { + var b: BoxI32 = { value: 17 }; + assert(b.value, 17); +} + +// G6: methods on instantiated types. Each instantiation gets its own +// clone of the method with substituted parameter and return types +// and a unique LLVM symbol. +fn Holder(T: type) type { + return struct { + value: T, + fn unwrap(self: move Self) T { + return self.value; + } + fn fortyTwo() i32 { + return 42; + } + }; +} + +const HolderI32 = Holder(i32); + +tfn methodOnInstantiatedTypeRunsAtCallSite() { + var h: HolderI32 = { value: 7 }; + assert(HolderI32.unwrap(h), 7); +} + +tfn staticMethodOnInstantiatedTypeRuns() { + assert(HolderI32.fortyTwo(), 42); +} + +// Generic with a drop method. The drop method is cloned per-instantiation +// AND registered in the drop-registry equivalent so codegen of var decls +// of this type fires the drop at scope exit. This was a real bug — the +// pre-built drop registry didn't see lazily-instantiated methods. +fn Counter(T: type) type { + return struct { + sink: *mut u32, + fn drop(self: mut Self) { + var p: *mut u32 = self.sink; + p.* = p.* + 1; + } + }; +} + +const CounterI32 = Counter(i32); + +fn dropFires(sink: *mut u32) { + var c: CounterI32 = { sink: sink }; + // c drops at function exit, bumping *sink to 1. +} + +tfn dropOnInstantiatedTypeFires() { + var hits: u32 = 0; + dropFires(&hits); + assert(hits, 1); +} + +// Audit follow-up: `Self` referenced in a FIELD type (recursive struct +// via pointer). The field substitution must include the Self mapping, +// not just the type-parameter mapping. +fn Node(T: type) type { + return struct { + value: T, + next: *mut Self, + }; +} + +const NodeI32 = Node(i32); + +tfn selfInFieldTypeSubstitutes() { + var head: NodeI32 = { value: 7, next: &head }; + assert(head.value, 7); +} + +// Audit follow-up: pointer-typed field substitution — the engine +// recurses into PtrSingle/PtrMany element types. +fn PtrTo(T: type) type { + return struct { + ptr: *mut T, + }; +} + +const PtrToI32 = PtrTo(i32); + +tfn pointerFieldSubstitutes() { + var v: i32 = 42; + var p: PtrToI32 = { ptr: &v }; + var x: *mut i32 = p.ptr; + assert(x.*, 42); +} + +// Audit follow-up: array field substitution. +fn ThreeOf(T: type) type { + return struct { + items: [3]T, + }; +} + +const ThreeI32 = ThreeOf(i32); + +tfn arrayFieldSubstitutes() { + var t: ThreeI32 = { items: [7, 35, 0] }; + assert(t.items[0] + t.items[1], 42); +} + +// Audit follow-up: nested generic instantiation — Box(Box(T)). +fn BoxOuter(T: type) type { + return struct { + value: T, + }; +} + +const NestedBox = BoxOuter(BoxOuter(i32)); + +tfn nestedGenericInstantiates() { + var inner: BoxOuter(i32) = { value: 7 }; + var outer: NestedBox = { value: inner }; + assert(outer.value.value, 7); +} + +// Audit follow-up: a method that calls another method on Self via +// `Self.method(...)`. The codegen has to resolve `Self` through the +// substitution context to the canonical instantiated struct name. +fn Builder(T: type) type { + return struct { + value: T, + fn make(v: T) Self { + var s: Self = { value: v }; + return s; + } + fn doubled(v: T) Self { + return Self.make(v + v); + } + }; +} + +const BuilderI32 = Builder(i32); + +tfn methodChainViaSelf() { + var b: BuilderI32 = BuilderI32.doubled(21); + assert(b.value, 42); +} + +// Audit follow-up: type parameter T appearing inside a method body +// (not just in the signature). typeSize / typeAlign / getLLVMType all +// have to consult the substitution context for Named TypeIdxes that +// resolve to primitive types via the parameter map. +fn Cloner(T: type) type { + return struct { + value: T, + fn cloned(self: Self) Self { + var v: T = self.value; + var s: Self = { value: v }; + return s; + } + }; +} + +const ClonerI32 = Cloner(i32); + +tfn typeParamInMethodBody() { + var orig: ClonerI32 = { value: 42 }; + var c: ClonerI32 = ClonerI32.cloned(orig); + assert(c.value, 42); +} + +// Audit follow-up: nested generic with methods on both layers. +// Exercises recursive instantiation when one method signature +// references another generic's instantiation. +fn Wrap(T: type) type { + return struct { + value: T, + fn make(v: T) Self { + var s: Self = { value: v }; + return s; + } + fn unwrap(self: move Self) T { + return self.value; + } + }; +} + +const InnerWrap = Wrap(i32); +const OuterWrap = Wrap(Wrap(i32)); + +tfn nestedGenericMethodsBothLayers() { + var inner: InnerWrap = InnerWrap.make(42); + var outer: OuterWrap = OuterWrap.make(inner); + var got: InnerWrap = OuterWrap.unwrap(outer); + assert(InnerWrap.unwrap(got), 42); +} + +// Audit follow-up: nested generic with drops on both layers. Each +// drop fires once per binding at scope exit (Jam doesn't recurse +// into fields). For two bindings (inner + outer), we see hits = 2. +fn Bumper(T: type) type { + return struct { + sink: *mut u32, + value: T, + fn drop(self: mut Self) { + var p: *mut u32 = self.sink; + p.* = p.* + 1; + } + }; +} + +const BumperI32 = Bumper(i32); +const NestedBumper = Bumper(BumperI32); + +fn doubleDropper(sink: *mut u32) { + var inner: BumperI32 = { sink: sink, value: 7 }; + var outer: NestedBumper = { sink: sink, value: inner }; +} + +tfn nestedDropFiresOnEachLayer() { + var hits: u32 = 0; + doubleDropper(&hits); + assert(hits, 2); +} + +// `default()` method system: types opt in by defining `fn default() Self` +// (no parameters, returns Self). On top-level structs, only `drop` and +// `default` methods are accepted. Generic struct expressions can also +// declare `default` as a static method. + +const Counter = struct { + n: i32, + fn default() Self { + return { n: 0 }; + } +}; + +tfn topLevelDefault() { + var c: Counter = Counter.default(); + assert(c.n, 0); +} + +// Generic with default(): T must itself have default(). Maybe(T) calls +// `T.default()` in its body, so instantiation requires T to be +// default-constructible. The `methodChainViaSelf`-style substitution +// resolves T.default() to the concrete type's default at codegen. +fn Pair2(T: type) type { + return struct { + a: T, + b: T, + fn default() Self { + return { a: T.default(), b: T.default() }; + } + }; +} + +const PairOfCounter = Pair2(Counter); + +tfn genericDefaultRequiresT() { + var p: PairOfCounter = PairOfCounter.default(); + assert(p.a.n + p.b.n, 0); +} diff --git a/tests/unit/test_init.jam b/tests/unit/test_init.jam index a27b008..3029b3b 100644 --- a/tests/unit/test_init.jam +++ b/tests/unit/test_init.jam @@ -14,9 +14,9 @@ fn directInitRead() u32 { return x; } -// --- Declare undefined, assign, then read --- -fn undefinedThenAssign() u32 { - var x: u32 = undefined; +// --- Reassign overwrites the initial value --- +fn reassignOverInitial() u32 { + var x: u32 = 0; x = 42; return x; } @@ -29,9 +29,9 @@ fn multipleBindings() u32 { return a + b + c; } -// --- if/else with both branches initing --- +// --- if/else with both branches reassigning --- fn ifElseBothInit() u32 { - var x: u32 = undefined; + var x: u32 = 0; if (true) { x = 100; } else { @@ -48,52 +48,45 @@ fn reassign() u32 { return x; } -// --- Partial init via field write counts as init in P2 (imprecise but -// matches existing array/struct usage patterns) --- const Vec = struct { x: u32, y: u32, }; -fn partialInitStruct() u32 { - var v: Vec = undefined; +// --- Field reassignment after struct-literal init --- +fn fieldReassign() u32 { + var v: Vec = { x: 0, y: 0 }; v.x = 7; v.y = 11; return v.x + v.y; } -fn partialInitArray() u8 { - var arr: [4]u8 = undefined; - arr[0] = 10; - arr[1] = 20; - arr[2] = 30; - arr[3] = 40; +// --- Element reassignment after array-literal init --- +fn arrayReassign() u8 { + var arr: [4]u8 = [10, 20, 30, 40]; return arr[2]; } -// --- For-loop body fills array; post-loop read is fine (P2 assumes the -// range body executes at least once) --- +// --- For-loop body overwrites array elements --- fn forLoopFill() u8 { - var arr: [16]u8 = undefined; + var arr: [16]u8 = [0; 16]; for i in 0:16 { arr[i] = i; } return arr[7]; } -// --- Address-of an uninit binding is allowed; `&x` does not read x. -// P2 does not track writes through pointers, so we explicitly init -// x by direct assignment after taking its address. -fn addressOfUninit() u8 { - var x: u8 = undefined; - var p: *mut u8 = &x; // & does not trigger a read check on x +// --- Address-of a binding; `&x` doesn't change init state. --- +fn addressOfBinding() u8 { + var x: u8 = 0; + var p: *mut u8 = &x; x = 99; return x; } -// --- Return inside one branch; the other branch reaches the next stmt --- +// --- Return inside one branch; the other reaches the next stmt --- fn returnInBranch(b: bool) u32 { - var x: u32 = undefined; + var x: u32 = 0; if (b) { return 0; } else { @@ -108,8 +101,8 @@ tfn directInitReadOK() { assert(directInitRead(), 5); } -tfn undefinedThenAssignOK() { - assert(undefinedThenAssign(), 42); +tfn reassignOverInitialOK() { + assert(reassignOverInitial(), 42); } tfn multipleBindingsOK() { @@ -124,20 +117,20 @@ tfn reassignOK() { assert(reassign(), 30); } -tfn partialInitStructOK() { - assert(partialInitStruct(), 18); +tfn fieldReassignOK() { + assert(fieldReassign(), 18); } -tfn partialInitArrayOK() { - assert(partialInitArray(), 30); +tfn arrayReassignOK() { + assert(arrayReassign(), 30); } tfn forLoopFillOK() { assert(forLoopFill(), 7); } -tfn addressOfUninitOK() { - assert(addressOfUninit(), 99); +tfn addressOfBindingOK() { + assert(addressOfBinding(), 99); } tfn returnInBranchOK() { diff --git a/tests/unit/test_mixed_slices.jam b/tests/unit/test_mixed_slices.jam index 7e27531..f92c45d 100644 --- a/tests/unit/test_mixed_slices.jam +++ b/tests/unit/test_mixed_slices.jam @@ -17,19 +17,19 @@ fn test_string_functions(input: str) str { return input; } -// Test complex slice types +// Test complex slice types — empty literal yields {ptr=null, len=0}. fn test_i8_slice() []i8 { - var data: []i8 = undefined; + var data: []i8 = []; return data; } fn test_i16_slice() []i16 { - var data: []i16 = undefined; + var data: []i16 = []; return data; } fn test_i32_slice() []i32 { - var data: []i32 = undefined; + var data: []i32 = []; return data; } diff --git a/tests/unit/test_modes_escape.jam b/tests/unit/test_modes_escape.jam index c1c041c..5da1dea 100644 --- a/tests/unit/test_modes_escape.jam +++ b/tests/unit/test_modes_escape.jam @@ -1,14 +1,13 @@ const { assert } = import("test"); -// MVS P5.5: scope-escape check. A `mut` or `undefined` parameter is a -// second-class borrow into the caller's storage; returning a reference -// (`&`) rooted at it would extend the borrow's lifetime past the call -// frame and dangle. The check rejects only `return ¶m-or-subpath` -// — returning the parameter as a value is a copy and is legal. +// MVS P5.5: scope-escape check. A `mut` parameter is a second-class +// borrow into the caller's storage; returning a reference (`&`) rooted +// at it would extend the borrow's lifetime past the call frame and +// dangle. The check rejects only `return ¶m-or-subpath` — returning +// the parameter as a value is a copy and is legal. // -// Must-fail cases (return &p, return &p.field, return &out for -// undefined param) are verified manually until the must-fail runner -// lands in P7. +// Must-fail cases (return &p, return &p.field) are verified manually +// until the must-fail runner lands in P7. // --- Returning a mut param's *value* is fine (it's a copy) --- fn doubleByValue(x: mut u32) u32 { @@ -38,12 +37,6 @@ fn computeFromMut(x: mut u32) u32 { return x + 1; } -// --- Returning an undefined param's value (after init) is fine --- -fn fillThenReturnValue(out: undefined u32) u32 { - out = 42; - return out; -} - // --- Tests --- tfn doubleByValueOK() { @@ -52,9 +45,7 @@ tfn doubleByValueOK() { } tfn pickFieldAOK() { - var p: Pair = undefined; - p.a = 9; - p.b = 0; + var p: Pair = { a: 9, b: 0 }; assert(pickFieldA(&p), 10); } @@ -62,8 +53,3 @@ tfn computeFromMutOK() { var n: u32 = 7; assert(computeFromMut(&n), 108); } - -tfn fillThenReturnValueOK() { - var x: u32 = undefined; - assert(fillThenReturnValue(&x), 42); -} diff --git a/tests/unit/test_modes_exclusivity.jam b/tests/unit/test_modes_exclusivity.jam index bbce08e..62a7eaa 100644 --- a/tests/unit/test_modes_exclusivity.jam +++ b/tests/unit/test_modes_exclusivity.jam @@ -31,9 +31,7 @@ fn addPair(a: u32, b: u32) u32 { } fn disjointFields() u32 { - var p: Pair = undefined; - p.a = 10; - p.b = 20; + var p: Pair = { a: 10, b: 20 }; return addPair(p.a, p.b); } @@ -43,11 +41,7 @@ fn addTwo(a: u8, b: u8) u8 { } fn disjointArrayIndices() u8 { - var arr: [4]u8 = undefined; - arr[0] = 1; - arr[1] = 2; - arr[2] = 3; - arr[3] = 4; + var arr: [4]u8 = [1, 2, 3, 4]; return addTwo(arr[0], arr[2]); } diff --git a/tests/unit/test_modes_parse.jam b/tests/unit/test_modes_parse.jam index c0e4a81..8c0e1f2 100644 --- a/tests/unit/test_modes_parse.jam +++ b/tests/unit/test_modes_parse.jam @@ -2,11 +2,11 @@ const { assert } = import("test"); // MVS P1: parser-only support for parameter mode keywords. // -// These tests verify that the parser accepts the four parameter modes -// (default = let, mut, move, undefined) and that programs using them -// compile and run identically to the same programs without the mode -// annotations. Semantic enforcement (init analysis, exclusivity, -// scope-escape, drop synthesis) lands in P2 and beyond. See docs/MVS.md. +// These tests verify that the parser accepts the three parameter modes +// (default = let, mut, move) and that programs using them compile and +// run identically to the same programs without the mode annotations. +// Semantic enforcement (init analysis, exclusivity, scope-escape, drop +// synthesis) lands in P2 and beyond. See docs/MVS.md. // --- Default mode (no keyword): read-only borrow --- fn readDefault(x: u32) u32 { @@ -24,17 +24,10 @@ fn moveValue(buf: move u32) u32 { return buf; } -// --- undefined: write to uninitialized destination --- -fn writeUndef(out: undefined u32) u32 { - out = 42; - return out; -} - -// --- All four modes mixed in one signature --- -fn mixedModes(a: u32, b: mut u32, c: move u32, d: undefined u32) u32 { +// --- All three modes mixed in one signature --- +fn mixedModes(a: u32, b: mut u32, c: move u32) u32 { b = b + 1; - d = 100; - return a + b + c + d; + return a + b + c; } // --- Mode applied to a pointer-typed parameter --- @@ -50,25 +43,9 @@ fn paramNamedOut(out: u32) u32 { return out; } -// --- Mode applied to an array type --- -// Declared but not invoked in P2 — calling a function with an -// `undefined`-mode parameter requires callsite mode propagation, which -// lands in P4. The signature alone exercises the parser. -fn writeArray(arr: undefined [4]u8) u8 { - arr[0] = 1; - arr[1] = 2; - arr[2] = 3; - arr[3] = 4; - return arr[2]; -} - -// --- Local helper for the test path (no undefined-mode in signature) --- +// --- Local helper for the test path --- fn fillThenReadHelper() u8 { - var arr: [4]u8 = undefined; - arr[0] = 1; - arr[1] = 2; - arr[2] = 3; - arr[3] = 4; + var arr: [4]u8 = [1, 2, 3, 4]; return arr[2]; } @@ -79,35 +56,31 @@ tfn readDefaultParses() { } tfn modifyMutParses() { - assert(modifyMut(10), 11); + var n: u32 = 10; + assert(modifyMut(&n), 11); } tfn moveValueParses() { - assert(moveValue(99), 99); -} - -tfn writeUndefParses() { - var x: u32 = undefined; - assert(writeUndef(&x), 42); + var v: u32 = 99; + assert(moveValue(v), 99); } tfn mixedModesParses() { - var initial: u32 = 50; - assert(mixedModes(1, 2, 3, initial), 107); + var b: u32 = 2; + var c: u32 = 3; + assert(mixedModes(1, &b, c), 7); } tfn modifyMutPtrParses() { var x: u32 = 33; var p: *mut u32 = &x; - assert(modifyMutPtr(p), 33); + assert(modifyMutPtr(&p), 33); } tfn paramNamedOutParses() { assert(paramNamedOut(123), 123); } -tfn writeArrayParses() { - // P4 will let us call writeArray with `undefined` callsite propagation. - // Until then, exercise the equivalent body via a helper. +tfn arrayLiteralReads() { assert(fillThenReadHelper(), 3); } diff --git a/tests/unit/test_modes_propagation.jam b/tests/unit/test_modes_propagation.jam index bbdd52e..03a8707 100644 --- a/tests/unit/test_modes_propagation.jam +++ b/tests/unit/test_modes_propagation.jam @@ -3,7 +3,6 @@ const { assert } = import("test"); // MVS P4: callsite mode propagation. The analyzer looks up each callee's // parameter modes and updates the caller's binding states accordingly: // - move arg → caller's base binding becomes Uninit -// - undefined arg → caller's base binding becomes Init // - let / mut → caller's binding state unchanged // // These tests cover *runtime-correct* must-pass cases. The runtime ABI diff --git a/tests/unit/test_modes_undefined.jam b/tests/unit/test_modes_undefined.jam deleted file mode 100644 index 7b7a010..0000000 --- a/tests/unit/test_modes_undefined.jam +++ /dev/null @@ -1,87 +0,0 @@ -const { assert } = import("test"); - -// MVS P3: `undefined`-mode parameters enter the function as Uninit and -// must be initialized on every return path. These tests exercise the -// success cases — programs the analyzer should accept. -// -// Failure cases (read before write, partial init, forgot to init) are -// verified manually via /tmp/bad_undefined_*.jam until P7 wires up the -// must-fail test infrastructure. - -// --- Direct assignment satisfies the contract --- -fn writeOut(out: undefined u32) u32 { - out = 42; - return out; -} - -// --- Partial writes via field access satisfy the contract (P2's -// "any write to a sub-path counts as init for the whole binding" -// rule applies to undefined-mode parameters too) --- -const Vec = struct { - x: u32, - y: u32, -}; - -fn writeOutStruct(p: undefined Vec) u32 { - p.x = 7; - p.y = 11; - return p.x + p.y; -} - -fn writeOutArray(arr: undefined [4]u8) u8 { - arr[0] = 1; - arr[1] = 2; - arr[2] = 3; - arr[3] = 4; - return arr[2]; -} - -// --- Branching: every reachable arm initializes the undefined param --- -fn writeOutInBoth(b: bool, out: undefined u32) u32 { - if (b) { - out = 100; - } else { - out = 200; - } - return out; -} - -// --- Combination: undefined param plus other modes in one signature --- -fn mixedModesUndef(a: u32, b: mut u32, out: undefined u32) u32 { - b = b + 1; - out = a + b; - return out; -} - -// --- Tests --- - -tfn writeOutOK() { - var x: u32 = undefined; - assert(writeOut(&x), 42); -} - -tfn writeOutStructOK() { - var seed: Vec = undefined; - seed.x = 0; - seed.y = 0; - assert(writeOutStruct(&seed), 18); -} - -tfn writeOutArrayOK() { - var seed: [4]u8 = undefined; - seed[0] = 0; - assert(writeOutArray(&seed), 3); -} - -tfn writeOutInBothOK() { - var x: u32 = undefined; - assert(writeOutInBoth(true, &x), 100); - var y: u32 = undefined; - assert(writeOutInBoth(false, &y), 200); -} - -tfn mixedModesUndefOK() { - var b: u32 = 5; - var out: u32 = undefined; - assert(mixedModesUndef(10, &b, &out), 16); -} diff --git a/tests/unit/test_pointer.jam b/tests/unit/test_pointer.jam index ee84490..91b25b4 100644 --- a/tests/unit/test_pointer.jam +++ b/tests/unit/test_pointer.jam @@ -14,21 +14,13 @@ fn writeThroughPointer() u8 { } fn pointerToArrayElement() u8 { - var arr: [4]u8 = undefined; - arr[0] = 11; - arr[1] = 22; - arr[2] = 33; - arr[3] = 44; + var arr: [4]u8 = [11, 22, 33, 44]; var p: *const u8 = &arr[2]; return p.*; } fn writeThroughArrayElementPointer() u8 { - var arr: [4]u8 = undefined; - arr[0] = 1; - arr[1] = 2; - arr[2] = 3; - arr[3] = 4; + var arr: [4]u8 = [1, 2, 3, 4]; var p: *mut u8 = &arr[1]; p.* = 99; return arr[1]; @@ -49,30 +41,20 @@ const Point = struct { }; fn addressOfStructField() u8 { - var pt: Point = undefined; - pt.x = 10; - pt.y = 20; + var pt: Point = { x: 10, y: 20 }; var px: *mut u8 = &pt.x; px.* = 200; return pt.x; } fn manyItemPointerIndex() u8 { - var arr: [4]u8 = undefined; - arr[0] = 5; - arr[1] = 6; - arr[2] = 7; - arr[3] = 8; + var arr: [4]u8 = [5, 6, 7, 8]; var p: *mut[] u8 = &arr[1]; return p[2]; } fn manyItemPointerWrite() u8 { - var arr: [4]u8 = undefined; - arr[0] = 1; - arr[1] = 2; - arr[2] = 3; - arr[3] = 4; + var arr: [4]u8 = [1, 2, 3, 4]; var p: *mut[] u8 = &arr[0]; p[3] = 88; return arr[3]; diff --git a/tests/unit/test_slices.jam b/tests/unit/test_slices.jam index 025b77f..4e55ede 100644 --- a/tests/unit/test_slices.jam +++ b/tests/unit/test_slices.jam @@ -23,24 +23,24 @@ fn test_u8_slice_const() []u8 { return buffer; } -// Test other slice types +// Test other slice types: empty literal `[]` builds {ptr=null, len=0}. fn test_u16_slice() []u16 { - var data: []u16 = undefined; + var data: []u16 = []; return data; } fn test_u32_slice() []u32 { - var data: []u32 = undefined; + var data: []u32 = []; return data; } fn test_bool_slice() []bool { - var data: []bool = undefined; + var data: []bool = []; return data; } // Test nested slice types: outer-only mutability covers the inner. fn test_slice_of_slices() [][]u8 { - var data: [][]u8 = undefined; + var data: [][]u8 = []; return data; } diff --git a/tests/unit/test_struct_methods.jam b/tests/unit/test_struct_methods.jam new file mode 100644 index 0000000..12c7501 --- /dev/null +++ b/tests/unit/test_struct_methods.jam @@ -0,0 +1,67 @@ +const { assert } = import("test"); + +// Drop functions can be declared inside the struct body. The form is a +// pure synonym of the free-function `fn drop(self: mut T)`: the drop +// registry picks it up the same way, codegen synthesizes the same +// `__drop_T` call at scope exit, the side-effect order is identical. + +const Counter = struct { + value: u32, + sink: *mut u32, + fn drop(self: mut Counter) { + var p: *mut u32 = self.sink; + p.* = p.* + 1; + } +}; + +// Auto-drop on a binding of a type whose drop is defined inline. +fn singleScopeWithInlineDrop() u32 { + var hits: u32 = 0; + var c: Counter = { value: 17, sink: &hits }; + // c drops at function end → hits becomes 1. + return c.value + hits; +} + +tfn inlineDropFiresAtScopeExit() { + // singleScope returns c.value (17) + the value of hits BEFORE the drop + // (0), so we expect 17. The drop runs after the return value is + // computed but before the function exits — observable from the + // caller via the sink pointer if we kept a reference, but here we + // just verify the math. + assert(singleScopeWithInlineDrop(), 17); +} + +// Caller observes the drop via the sink. Same shape as +// callerObservingDrop in test_drops.jam, just with the in-struct form. +fn wrapsCounterInline(sink: *mut u32) u32 { + var c: Counter = { value: 5, sink: sink }; + return c.value; +} + +tfn callerObservesInlineDrop() { + var hits: u32 = 0; + var v: u32 = wrapsCounterInline(&hits); + // wrapsCounterInline's local c dropped → hits == 1. + assert(v, 5); + assert(hits, 1); +} + +// Manual `Counter.drop(&c)` call while c is still in scope. The auto-drop +// at function exit also fires, so the sink gets bumped twice. This is a +// known footgun (manual + auto = double drop) — flagged in +// docs/STRUCT_METHODS.md as a v1.1 polish item. The test pins down +// today's behavior so a future fix is intentional, not accidental. +fn callsDropManually(sink: *mut u32) u32 { + var c: Counter = { value: 9, sink: sink }; + Counter.drop(&c); + // sink == 1 here (manual drop ran) + return c.value; +} + +tfn manualQualifiedDropCallWorks() { + var hits: u32 = 0; + var v: u32 = callsDropManually(&hits); + assert(v, 9); + // Manual drop + auto drop on function exit → 2. + assert(hits, 2); +} diff --git a/tests/unit/test_union.jam b/tests/unit/test_union.jam index d37a9ba..badedf8 100644 --- a/tests/unit/test_union.jam +++ b/tests/unit/test_union.jam @@ -11,22 +11,19 @@ const FloatBits = union { }; fn writeIntReadInt() u32 { - var b: FloatBits = undefined; - b.i = 0xCAFEBABE; + var b: FloatBits = { i: 0xCAFEBABE }; return b.i; } fn writeIntReadIntZero() u32 { - var b: FloatBits = undefined; - b.i = 0; + var b: FloatBits = { i: 0 }; return b.i; } fn aliasMostRecentWrite() u32 { // Writing one field then another then reading the second: // returns the second-written value's bit pattern. - var b: FloatBits = undefined; - b.i = 0xAAAAAAAA; + var b: FloatBits = { i: 0xAAAAAAAA }; b.i = 0x55555555; return b.i; } @@ -35,10 +32,8 @@ fn floatRoundTrip(seed: u32) u32 { // Write the bit pattern via u32, read it via f32, then write that // f32 back via the float field, then read via u32 again. The // pattern survives unchanged for any non-NaN, non-signaling value. - var b: FloatBits = undefined; - b.i = seed; - var b2: FloatBits = undefined; - b2.f = b.f; + var b: FloatBits = { i: seed }; + var b2: FloatBits = { f: b.f }; return b2.i; } @@ -60,8 +55,7 @@ const SmallVsBig = union { fn writeBigReadSmall() u8 { // u64 written, u8 read — returns the low byte (little-endian on // every target Jam currently supports). - var u: SmallVsBig = undefined; - u.b = 0xDEADBEEFCAFEBABE; + var u: SmallVsBig = { b: 0xDEADBEEFCAFEBABE }; return u.s; } @@ -70,8 +64,7 @@ fn writeSmallReadBig() u64 { // alloca held. We first zero via the wide field, then write the // small field, then read the wide field — only the low byte is // defined. - var u: SmallVsBig = undefined; - u.b = 0; + var u: SmallVsBig = { b: 0 }; u.s = 42; return u.b; } @@ -82,8 +75,7 @@ tfn writeSmallReadBigVal() { assert(writeSmallReadBig(), 42); } // -------------------- Function parameter / return -------------------- fn returnUnion(seed: u32) FloatBits { - var b: FloatBits = undefined; - b.i = seed; + var b: FloatBits = { i: seed }; return b; } @@ -105,9 +97,8 @@ const ThreeWayU32 = union { }; fn signedUnsignedReinterpret() u32 { - var u: ThreeWayU32 = undefined; - u.b = -1; // 0xFFFFFFFF as i32 - return u.a; // read as u32 → 0xFFFFFFFF + var u: ThreeWayU32 = { b: -1 }; // 0xFFFFFFFF as i32 + return u.a; // read as u32 → 0xFFFFFFFF } tfn signedToUnsigned() { @@ -123,14 +114,12 @@ const Bigger = union { }; fn writeBiggestReadSmallest() u8 { - var u: Bigger = undefined; - u.big = 0x0102030405060708; + var u: Bigger = { big: 0x0102030405060708 }; return u.small; // 0x08 on little-endian } fn writeBiggestReadMedium() u32 { - var u: Bigger = undefined; - u.big = 0x0102030405060708; + var u: Bigger = { big: 0x0102030405060708 }; return u.medium; // 0x05060708 on little-endian } diff --git a/tests/unit/test_varargs.jam b/tests/unit/test_varargs.jam index c20edb8..6e9c615 100644 --- a/tests/unit/test_varargs.jam +++ b/tests/unit/test_varargs.jam @@ -1,20 +1,20 @@ extern fn snprintf(buf: *mut[] u8, size: u64, format: *const[] u8, ...) i32; fn snprintfNoExtra() i32 { - var buf: [16]u8 = undefined; + var buf: [16]u8 = [0; 16]; const fmt: []u8 = "hi"; return snprintf(&buf[0], 16, fmt.ptr); } fn snprintfWithInt() i32 { - var buf: [16]u8 = undefined; + var buf: [16]u8 = [0; 16]; const fmt: []u8 = "n=%d"; var n: i32 = 42; return snprintf(&buf[0], 16, fmt.ptr, n); } fn snprintfWithStr() i32 { - var buf: [32]u8 = undefined; + var buf: [32]u8 = [0; 32]; const fmt: []u8 = "s=%s"; const arg: []u8 = "hi"; return snprintf(&buf[0], 32, fmt.ptr, arg.ptr);