From e28ca9a2157a17e0ecb626e40ead99a8f10180a5 Mon Sep 17 00:00:00 2001 From: Raphael Amorim Date: Tue, 12 May 2026 10:12:22 +0200 Subject: [PATCH] add byteOffset to Token (Zig Ast.TokenList shape) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Match Zig's persistent per-token storage: { tag, start } only. The lexer captures tokenStart at the top of each scan-loop iteration before the first advance(), and addToken stamps it onto every emitted Token. Tokens still carry lexeme + line for compat with existing parser / init-analysis consumers; those can drop later once consumers migrate to byte-offset-only access. End offsets are deliberately not stored — Zig recomputes them via tokenSlice() re-tokenizing from start (short-circuit on tag.lexeme() for keywords/operators). --- src/lexer.cpp | 10 +++++++++- src/lexer.h | 5 +++++ src/token.h | 19 ++++++++++++++++--- 3 files changed, 30 insertions(+), 4 deletions(-) diff --git a/src/lexer.cpp b/src/lexer.cpp index e3004a7..5995669 100644 --- a/src/lexer.cpp +++ b/src/lexer.cpp @@ -69,7 +69,10 @@ bool Lexer::isAlphaNumeric(char c) const { return isAlpha(c) || isDigit(c); } void Lexer::addToken(TokenType type) { addToken(type, ""); } void Lexer::addToken(TokenType type, const std::string &lexeme) { - tokens.emplace_back(type, lexeme, line); + // `tokenStart` is captured at the top of each scan-loop iteration + // (before the first `advance()`), so we can stamp it onto every + // emitted Token without per-call site bookkeeping. + tokens.emplace_back(type, lexeme, line, tokenStart); } void Lexer::identifier() { @@ -430,6 +433,11 @@ std::vector Lexer::scanTokens() { skipWhitespace(); if (isAtEnd()) break; + // Snapshot the byte offset of the token we're about to scan. + // Every `addToken` call below reads this via member state, so + // each emitted Token carries its start offset — matches Zig's + // `Ast.TokenList.start` field. + tokenStart = static_cast(current); char c = advance(); switch (c) { diff --git a/src/lexer.h b/src/lexer.h index 0144597..481ee3d 100644 --- a/src/lexer.h +++ b/src/lexer.h @@ -18,6 +18,11 @@ class Lexer { std::vector tokens; int current = 0; int line = 1; + // Byte offset where the current token began — captured at the top + // of each scan-loop iteration and recorded onto every emitted + // Token. Matches Zig's `Ast.TokenList = { tag, start }` shape: + // the persistent per-token info is just the start offset. + uint32_t tokenStart = 0; bool isAtEnd() const; char advance(); diff --git a/src/token.h b/src/token.h index 6740e44..e58477f 100644 --- a/src/token.h +++ b/src/token.h @@ -8,6 +8,7 @@ #ifndef TOKEN_H #define TOKEN_H +#include #include // Token types @@ -76,14 +77,26 @@ enum TokenType { TOK_AS, // as keyword (explicit type cast) }; -// Token structure +// Token structure. +// +// `byteOffset` is the start byte offset of the token in the source — +// matches Zig's `Ast.TokenList = MultiArrayList({ tag, start })`. Zig +// stores only `start` per token; the end is recomputed on demand from +// the tokenizer (`tokenSlice` re-tokenizes for variable-length tokens, +// short-circuits on a static lexeme table for keywords/operators). +// +// `lexeme` and `line` are legacy: kept for compat while the rest of +// the codebase migrates to byte-offset-only access. Under strict +// `lib/std/zig/` parity neither will exist in the long run. struct Token { TokenType type; std::string lexeme; int line; + uint32_t byteOffset = 0; - Token(TokenType type, std::string lexeme, int line) - : type(type), lexeme(std::move(lexeme)), line(line) {} + Token(TokenType type, std::string lexeme, int line, uint32_t byteOffset = 0) + : type(type), lexeme(std::move(lexeme)), line(line), + byteOffset(byteOffset) {} }; #endif // TOKEN_H -- 2.51.2