diff --git a/src/lexer.cpp b/src/lexer.cpp index e3004a7..5995669 100644 --- a/src/lexer.cpp +++ b/src/lexer.cpp @@ -69,7 +69,10 @@ bool Lexer::isAlphaNumeric(char c) const { return isAlpha(c) || isDigit(c); } void Lexer::addToken(TokenType type) { addToken(type, ""); } void Lexer::addToken(TokenType type, const std::string &lexeme) { - tokens.emplace_back(type, lexeme, line); + // `tokenStart` is captured at the top of each scan-loop iteration + // (before the first `advance()`), so we can stamp it onto every + // emitted Token without per-call site bookkeeping. + tokens.emplace_back(type, lexeme, line, tokenStart); } void Lexer::identifier() { @@ -430,6 +433,11 @@ std::vector Lexer::scanTokens() { skipWhitespace(); if (isAtEnd()) break; + // Snapshot the byte offset of the token we're about to scan. + // Every `addToken` call below reads this via member state, so + // each emitted Token carries its start offset — matches Zig's + // `Ast.TokenList.start` field. + tokenStart = static_cast(current); char c = advance(); switch (c) { diff --git a/src/lexer.h b/src/lexer.h index 0144597..481ee3d 100644 --- a/src/lexer.h +++ b/src/lexer.h @@ -18,6 +18,11 @@ class Lexer { std::vector tokens; int current = 0; int line = 1; + // Byte offset where the current token began — captured at the top + // of each scan-loop iteration and recorded onto every emitted + // Token. Matches Zig's `Ast.TokenList = { tag, start }` shape: + // the persistent per-token info is just the start offset. + uint32_t tokenStart = 0; bool isAtEnd() const; char advance(); diff --git a/src/token.h b/src/token.h index 6740e44..e58477f 100644 --- a/src/token.h +++ b/src/token.h @@ -8,6 +8,7 @@ #ifndef TOKEN_H #define TOKEN_H +#include #include // Token types @@ -76,14 +77,26 @@ enum TokenType { TOK_AS, // as keyword (explicit type cast) }; -// Token structure +// Token structure. +// +// `byteOffset` is the start byte offset of the token in the source — +// matches Zig's `Ast.TokenList = MultiArrayList({ tag, start })`. Zig +// stores only `start` per token; the end is recomputed on demand from +// the tokenizer (`tokenSlice` re-tokenizes for variable-length tokens, +// short-circuits on a static lexeme table for keywords/operators). +// +// `lexeme` and `line` are legacy: kept for compat while the rest of +// the codebase migrates to byte-offset-only access. Under strict +// `lib/std/zig/` parity neither will exist in the long run. struct Token { TokenType type; std::string lexeme; int line; + uint32_t byteOffset = 0; - Token(TokenType type, std::string lexeme, int line) - : type(type), lexeme(std::move(lexeme)), line(line) {} + Token(TokenType type, std::string lexeme, int line, uint32_t byteOffset = 0) + : type(type), lexeme(std::move(lexeme)), line(line), + byteOffset(byteOffset) {} }; #endif // TOKEN_H