From fcde356b8fb6ce9fd69a3e2f5e3507ccd9487c08 Mon Sep 17 00:00:00 2001 From: Aliou Diallo Date: Sun, 10 May 2026 22:04:42 +0000 Subject: [PATCH] feat: structured TestExpr for [[ ]] Replace \`expr: Word[]\` on TestClause with a typed \`x: TestExpr\` tree (\`BinaryTest\` / \`UnaryTest\` / \`ParenTest\` / Word leaf), parsed via precedence climbing (\`||\` < \`&&\` < other binary ops). Recognized unary file/string/option ops (\`-e\`, \`-f\`, \`-z\`, \`-v\`, etc.) and binary ops (\`==\`, \`!=\`, \`=~\`, \`-ef\`, \`-nt\`, \`-ot\`, ...) follow the Bash shape. Tightened the tokenizer rule for \`!\`: it's the negation operator only when at boundary AND followed by whitespace/EOF. Otherwise it stays in the word, so \`!=\` inside \`[[ a != b ]]\` parses as a single op. Existing TestClause tests in clauses.test.ts updated to the structured shape; \`testClause\` builder kept for simple single-word fixtures. --- src/ast.ts | 59 ++++++++++++- src/index.ts | 6 ++ src/parser/clauses.test.ts | 19 ++++- src/parser/parser.ts | 142 +++++++++++++++++++++++++++++-- src/parser/test-clause.test.ts | 122 ++++++++++++++++++++++++++ src/test-helpers/ast-builders.ts | 15 ++-- src/tokenizer/tokenize.ts | 16 +++- 7 files changed, 357 insertions(+), 22 deletions(-) create mode 100644 src/parser/test-clause.test.ts diff --git a/src/ast.ts b/src/ast.ts index 76c1dcd..301f136 100644 --- a/src/ast.ts +++ b/src/ast.ts @@ -284,7 +284,64 @@ export type CaseClause = Located & { items: CaseItem[]; }; export type TimeClause = Located & { type: "TimeClause"; command: Statement }; -export type TestClause = Located & { type: "TestClause"; expr: Word[] }; +export type BinaryTestOp = + | "==" + | "!=" + | "<" + | "<=" + | ">" + | ">=" + | "=~" + | "&&" + | "||" + | "=" + | "-ef" + | "-nt" + | "-ot"; + +export type UnaryTestOp = + | "!" + | "-e" + | "-f" + | "-d" + | "-r" + | "-w" + | "-x" + | "-z" + | "-n" + | "-s" + | "-a" + | "-o" + | "-S" + | "-c" + | "-b" + | "-p" + | "-h" + | "-L" + | "-N" + | "-O" + | "-G" + | "-u" + | "-g" + | "-k" + | "-t" + | "-v" + | "-R"; + +export type BinaryTest = Located & { + type: "BinaryTest"; + op: BinaryTestOp; + x: TestExpr; + y: TestExpr; +}; +export type UnaryTest = Located & { + type: "UnaryTest"; + op: UnaryTestOp; + x: TestExpr; +}; +export type ParenTest = Located & { type: "ParenTest"; x: TestExpr }; +export type TestExpr = BinaryTest | UnaryTest | ParenTest | Word; +export type TestClause = Located & { type: "TestClause"; x: TestExpr }; export type ArithCmd = Located & { type: "ArithCmd"; x: ArithExpr }; export type CoprocClause = Located & { type: "CoprocClause"; diff --git a/src/index.ts b/src/index.ts index 1f3f6d6..70c4a30 100644 --- a/src/index.ts +++ b/src/index.ts @@ -8,6 +8,8 @@ export { type Assignment, type BinaryArithm, type BinaryArithmOp, + type BinaryTest, + type BinaryTestOp, type Block, type BraceExp, type CaseClause, @@ -34,6 +36,7 @@ export { type ParamExpReplace, type ParamExpSlice, type ParenArithm, + type ParenTest, type ParseError, type ParseOptions, type ParseResult, @@ -50,9 +53,12 @@ export { type Statement, type Subshell, type TestClause, + type TestExpr, type TimeClause, type UnaryArithm, type UnaryArithmOp, + type UnaryTest, + type UnaryTestOp, type WhileClause, type Word, type WordPart, diff --git a/src/parser/clauses.test.ts b/src/parser/clauses.test.ts index a89d6da..be45fa0 100644 --- a/src/parser/clauses.test.ts +++ b/src/parser/clauses.test.ts @@ -167,14 +167,25 @@ describe("parse (phase 11: time)", () => { describe("parse (phase 13: extended test)", () => { it("parses [[ ]]", () => { - expect(parse("[[ -f foo ]]")).toMatchAst({ - ast: program(stmt(testClause(word("-f"), word("foo")))), + const { ast } = parse("[[ -f foo ]]"); + const cmd = ast.body[0]?.command; + if (cmd?.type !== "TestClause") throw new Error("expected TestClause"); + expect(cmd.x).toMatchAst({ + type: "UnaryTest", + op: "-f", + x: { type: "Word", parts: [{ type: "Literal", value: "foo" }] }, }); }); it("parses [[ with binary op ]]", () => { - expect(parse("[[ foo == bar ]]")).toMatchAst({ - ast: program(stmt(testClause(word("foo"), word("=="), word("bar")))), + const { ast } = parse("[[ foo == bar ]]"); + const cmd = ast.body[0]?.command; + if (cmd?.type !== "TestClause") throw new Error("expected TestClause"); + expect(cmd.x).toMatchAst({ + type: "BinaryTest", + op: "==", + x: { type: "Word", parts: [{ type: "Literal", value: "foo" }] }, + y: { type: "Word", parts: [{ type: "Literal", value: "bar" }] }, }); }); }); diff --git a/src/parser/parser.ts b/src/parser/parser.ts index eb85e6f..3368c2d 100644 --- a/src/parser/parser.ts +++ b/src/parser/parser.ts @@ -2,6 +2,7 @@ import type { ArithCmd, ArrayElem, Assignment, + BinaryTestOp, Block, CaseClause, CaseItem, @@ -25,7 +26,9 @@ import type { Statement, Subshell, TestClause, + TestExpr, TimeClause, + UnaryTestOp, WhileClause, Word, WordPart, @@ -44,6 +47,49 @@ import { DECL_KEYWORDS } from "./constants"; const ZERO_POS: Pos = { offset: 0, line: 1, col: 1 }; +const TEST_UNARY_OPS = new Set([ + "-e", + "-f", + "-d", + "-r", + "-w", + "-x", + "-z", + "-n", + "-s", + "-a", + "-o", + "-S", + "-c", + "-b", + "-p", + "-h", + "-L", + "-N", + "-O", + "-G", + "-u", + "-g", + "-k", + "-t", + "-v", + "-R", +]); + +const TEST_BINARY_OPS = new Set([ + "==", + "!=", + "<", + "<=", + ">", + ">=", + "=~", + "=", + "-ef", + "-nt", + "-ot", +]); + /** * Wrap a raw substring (e.g. a slice offset or a replacement pattern) as a * single-literal Word. The string was extracted from inside `${...}` so it @@ -613,22 +659,102 @@ export class Parser { private parseTestClause(): TestClause { const open = this.consumeKeyword("[["); checkLang(this.options.dialect, open.pos, "[[", ["bash", "mksh", "zsh"]); - const words: Word[] = []; - while (!this.matchKeyword("]]")) { - if (this.isEof()) throw new Error("Unclosed [["); - const token = this.consume(); - if (token.type !== "word") throw new Error("Expected word in [[ ]]"); - words.push(this.wordFromToken(token)); - } + const x = this.parseTestExpr(0); const close = this.consumeKeyword("]]"); return { type: "TestClause", - expr: words, + x, pos: open.pos, end: close.end, }; } + private parseTestExpr(minPrec: number): TestExpr { + let left = this.parseTestPrimary(); + while (true) { + const op = this.peekTestBinaryOp(); + if (!op) break; + const prec = op === "||" ? 1 : op === "&&" ? 2 : 3; + if (prec < minPrec) break; + this.consumeTestOp(op); + const right = this.parseTestExpr(prec + 1); + left = { + type: "BinaryTest", + op, + x: left, + y: right, + pos: left.pos ?? ZERO_POS, + end: right.end ?? ZERO_POS, + }; + } + return left; + } + + private parseTestPrimary(): TestExpr { + // Negation + if (this.matchOp("!")) { + const op = this.consume(); + const x = this.parseTestPrimary(); + return { + type: "UnaryTest", + op: "!", + x, + pos: op.pos, + end: x.end ?? op.end, + }; + } + // Parenthesized + if (this.matchSymbol("(")) { + const open = this.consumeSymbol("("); + const x = this.parseTestExpr(0); + const close = this.consumeSymbol(")"); + return { type: "ParenTest", x, pos: open.pos, end: close.end }; + } + // Unary file/string ops: a single token like `-e`/`-z`/... + const head = this.peek(); + if (head && head.type === "word") { + const text = tokenPartsText(head.parts); + if (TEST_UNARY_OPS.has(text)) { + this.consume(); + const arg = this.parseTestPrimary(); + return { + type: "UnaryTest", + op: text as UnaryTestOp, + x: arg, + pos: head.pos, + end: arg.end ?? head.end, + }; + } + } + // Otherwise a Word leaf (possibly followed by a binary op). + if (this.matchWord()) { + const tok = this.consume(); + if (tok.type !== "word") throw new Error("expected word in [[ ]]"); + return this.wordFromToken(tok); + } + throw new Error("Expected expression inside [[ ]]"); + } + + private peekTestBinaryOp(): BinaryTestOp | undefined { + const tok = this.peek(); + if (!tok) return undefined; + if (tok.type === "op" && (tok.value === "&&" || tok.value === "||")) { + return tok.value; + } + if (tok.type === "word") { + const text = tokenPartsText(tok.parts); + if (TEST_BINARY_OPS.has(text)) return text as BinaryTestOp; + } + return undefined; + } + + private consumeTestOp(op: BinaryTestOp): void { + const tok = this.consume(); + if (tok.type === "op" && tok.value === op) return; + if (tok.type === "word" && tokenPartsText(tok.parts) === op) return; + throw new Error(`expected test op ${op}`); + } + private matchArithCmd(): boolean { return this.peek()?.type === "arith-cmd"; } diff --git a/src/parser/test-clause.test.ts b/src/parser/test-clause.test.ts new file mode 100644 index 0000000..ed1cb3f --- /dev/null +++ b/src/parser/test-clause.test.ts @@ -0,0 +1,122 @@ +import { describe, expect, it } from "vitest"; +import type { TestClause } from "../ast"; +import { parse } from "../parse"; + +function testOf(src: string): TestClause { + const { ast } = parse(src); + const cmd = ast.body[0]?.command; + if (cmd?.type !== "TestClause") + throw new Error(`expected TestClause, got ${cmd?.type}`); + return cmd; +} + +const litWord = (value: string) => ({ + type: "Word", + parts: [{ type: "Literal", value }], +}); + +describe("TestClause structured AST", () => { + it("a single word leaf", () => { + expect(testOf("[[ foo ]]").x).toMatchAst(litWord("foo")); + }); + + it("binary equality", () => { + expect(testOf("[[ a == b ]]").x).toMatchAst({ + type: "BinaryTest", + op: "==", + x: litWord("a"), + y: litWord("b"), + }); + }); + + it("binary !=", () => { + expect(testOf("[[ a != b ]]").x).toMatchAst({ + type: "BinaryTest", + op: "!=", + x: litWord("a"), + y: litWord("b"), + }); + }); + + it("binary regex =~", () => { + expect(testOf("[[ foo =~ bar ]]").x).toMatchAst({ + type: "BinaryTest", + op: "=~", + x: litWord("foo"), + y: litWord("bar"), + }); + }); + + it("unary file test -e", () => { + expect(testOf("[[ -e file ]]").x).toMatchAst({ + type: "UnaryTest", + op: "-e", + x: litWord("file"), + }); + }); + + it("unary string -z", () => { + expect(testOf("[[ -z $foo ]]").x).toMatchAst({ + type: "UnaryTest", + op: "-z", + x: { + type: "Word", + parts: [ + { + type: "ParamExp", + short: true, + param: { type: "Literal", value: "foo" }, + }, + ], + }, + }); + }); + + it("negation !", () => { + expect(testOf("[[ ! a ]]").x).toMatchAst({ + type: "UnaryTest", + op: "!", + x: litWord("a"), + }); + }); + + it("&& has higher precedence than ||", () => { + expect(testOf("[[ a || b && c ]]").x).toMatchAst({ + type: "BinaryTest", + op: "||", + x: litWord("a"), + y: { + type: "BinaryTest", + op: "&&", + x: litWord("b"), + y: litWord("c"), + }, + }); + }); + + it("parens override precedence", () => { + expect(testOf("[[ (a || b) && c ]]").x).toMatchAst({ + type: "BinaryTest", + op: "&&", + x: { + type: "ParenTest", + x: { + type: "BinaryTest", + op: "||", + x: litWord("a"), + y: litWord("b"), + }, + }, + y: litWord("c"), + }); + }); + + it("file comparison -ef", () => { + expect(testOf("[[ a -ef b ]]").x).toMatchAst({ + type: "BinaryTest", + op: "-ef", + x: litWord("a"), + y: litWord("b"), + }); + }); +}); diff --git a/src/test-helpers/ast-builders.ts b/src/test-helpers/ast-builders.ts index 35cfc1f..ec49431 100644 --- a/src/test-helpers/ast-builders.ts +++ b/src/test-helpers/ast-builders.ts @@ -228,11 +228,16 @@ export const caseClause = ( items, ...P, }); -export const testClause = (...words: Word[]): TestClause => ({ - type: "TestClause", - expr: words, - ...P, -}); +/** + * Build a TestClause whose `x` is the first word (or a Word wrapping it). + * The structured form uses a TestExpr tree; this helper is intended for + * fixtures with simple single-word bodies. For complex shapes, hand-build + * the AST directly. + */ +export const testClause = (...words: Word[]): TestClause => { + const first = words[0] ?? { type: "Word", parts: [], ...P }; + return { type: "TestClause", x: first, ...P }; +}; export const arithCmd = (value: string): ArithCmd => ({ type: "ArithCmd", x: { type: "ArithLit", value, ...P }, diff --git a/src/tokenizer/tokenize.ts b/src/tokenizer/tokenize.ts index 13b2f69..3381b6d 100644 --- a/src/tokenizer/tokenize.ts +++ b/src/tokenizer/tokenize.ts @@ -121,10 +121,18 @@ export function tokenize(source: string, options: ParseOptions = {}): Token[] { } if (ch === "!" && atBoundary) { - // Prefer the extended-glob form `!(...)` over the negation operator, - // matching Bash where `!` only negates a command when it's a standalone - // word. - if (source.charAt(i + 1) !== "(") { + const next = source.charAt(i + 1); + // `!(` is extended glob — fall through to word parsing. + // `!` is the negation operator only when followed by a separator + // (space, tab, newline, EOF). Otherwise it's part of a word, e.g. + // `!=` inside `[[ ... ]]` or `!foo` as a literal. + const isNegation = + next === "" || + next === " " || + next === "\t" || + next === "\n" || + next === "\r"; + if (next !== "(" && isNegation) { tokens.push({ type: "op", value: "!", -- 2.51.2