From 84f233bdcaea463f6db656cbf5a27e39b45c42ac Mon Sep 17 00:00:00 2001 From: Aliou Diallo Date: Sun, 10 May 2026 21:46:18 +0000 Subject: [PATCH] feat: parse Bash extended glob patterns (?(, *(, +(, @(, !() Adds an `ExtGlob` word part with `op` and a raw `pattern` string (alternations and nested groups stay unparsed inside the pattern). Implementation lives in tokenizer/scan-extglob.ts and is invoked from the word loop before `(` is treated as a word terminator. The negation operator at a statement boundary now defers to the extended-glob scanner when followed by `(`, matching Bash semantics. Tests cover all five operators, alternation, nested parens, position tracking, and confirm a bare `?` stays a literal. --- src/ast.ts | 15 +++++++- src/index.ts | 2 + src/parser/extglob.test.ts | 72 +++++++++++++++++++++++++++++++++++ src/parser/parser.ts | 8 ++++ src/tokenizer/scan-extglob.ts | 48 +++++++++++++++++++++++ src/tokenizer/tokenize.ts | 44 ++++++++++++++++----- src/tokenizer/types.ts | 7 +++- 7 files changed, 185 insertions(+), 11 deletions(-) create mode 100644 src/parser/extglob.test.ts create mode 100644 src/tokenizer/scan-extglob.ts diff --git a/src/ast.ts b/src/ast.ts index 34d8bbb..7e8aaef 100644 --- a/src/ast.ts +++ b/src/ast.ts @@ -49,6 +49,18 @@ export type BraceExp = Located & { /** True for sequences like `{1..5}`; false for lists like `{a,b}`. */ sequence?: boolean; }; +/** The two-character open of an extended glob pattern. */ +export type ExtGlobOp = "?(" | "*(" | "+(" | "@(" | "!("; +/** + * A Bash extended glob like `@(foo|bar)`. The pattern is captured raw — + * alternations and nested groups are part of the string. Only emitted + * for Bash/mksh dialects. + */ +export type ExtGlob = Located & { + type: "ExtGlob"; + op: ExtGlobOp; + pattern: string; +}; export type WordPart = | Literal | SglQuoted @@ -57,7 +69,8 @@ export type WordPart = | CmdSubst | ArithExp | ProcSubst - | BraceExp; + | BraceExp + | ExtGlob; export type Word = Located & { type: "Word"; parts: WordPart[] }; export type Assignment = Located & { type: "Assignment"; diff --git a/src/index.ts b/src/index.ts index 8250c28..b7e79db 100644 --- a/src/index.ts +++ b/src/index.ts @@ -15,6 +15,8 @@ export { type CStyleLoop, type DblQuoted, type DeclClause, + type ExtGlob, + type ExtGlobOp, type ForClause, type FunctionDecl, type IfClause, diff --git a/src/parser/extglob.test.ts b/src/parser/extglob.test.ts new file mode 100644 index 0000000..9b2ab23 --- /dev/null +++ b/src/parser/extglob.test.ts @@ -0,0 +1,72 @@ +import { describe, expect, it } from "vitest"; +import type { ExtGlob, SimpleCommand } from "../ast"; +import { parse } from "../parse"; + +function firstWordParts(src: string) { + const { ast } = parse(src); + const cmd = ast.body[0]?.command as SimpleCommand; + return cmd.words?.[1]?.parts ?? []; +} + +describe("extended glob patterns", () => { + it("parses ?(pattern)", () => { + const parts = firstWordParts("ls ?(foo)"); + expect(parts).toMatchAst([{ type: "ExtGlob", op: "?(", pattern: "foo" }]); + }); + + it("parses *(pattern)", () => { + const parts = firstWordParts("ls *(foo)"); + expect(parts).toMatchAst([{ type: "ExtGlob", op: "*(", pattern: "foo" }]); + }); + + it("parses +(pattern)", () => { + const parts = firstWordParts("ls +(foo)"); + expect(parts).toMatchAst([{ type: "ExtGlob", op: "+(", pattern: "foo" }]); + }); + + it("parses @(pattern)", () => { + const parts = firstWordParts("ls @(foo)"); + expect(parts).toMatchAst([{ type: "ExtGlob", op: "@(", pattern: "foo" }]); + }); + + it("parses !(pattern)", () => { + const parts = firstWordParts("ls !(foo)"); + expect(parts).toMatchAst([{ type: "ExtGlob", op: "!(", pattern: "foo" }]); + }); + + it("parses alternations inside the pattern", () => { + const parts = firstWordParts("ls @(foo|bar|baz)"); + expect(parts).toMatchAst([ + { type: "ExtGlob", op: "@(", pattern: "foo|bar|baz" }, + ]); + }); + + it("handles nested parens inside the pattern", () => { + const parts = firstWordParts("ls @(foo|@(bar|baz))"); + expect(parts).toMatchAst([ + { type: "ExtGlob", op: "@(", pattern: "foo|@(bar|baz)" }, + ]); + }); + + it("attaches a literal prefix and suffix", () => { + const parts = firstWordParts("ls a@(b|c)d"); + expect(parts).toMatchAst([ + { type: "Literal", value: "a" }, + { type: "ExtGlob", op: "@(", pattern: "b|c" }, + { type: "Literal", value: "d" }, + ]); + }); + + it("treats a bare ? followed by space as a literal", () => { + const { ast } = parse("ls ?"); + const cmd = ast.body[0]?.command as SimpleCommand; + expect(cmd.words?.[1]?.parts).toMatchAst([{ type: "Literal", value: "?" }]); + }); + + it("attaches positions to the ExtGlob node", () => { + const parts = firstWordParts("ls @(foo)"); + const eg = parts[0] as ExtGlob; + expect(eg.pos).toMatchObject({ offset: 3 }); + expect(eg.end).toMatchObject({ offset: 9 }); + }); +}); diff --git a/src/parser/parser.ts b/src/parser/parser.ts index d56f041..6bccbf7 100644 --- a/src/parser/parser.ts +++ b/src/parser/parser.ts @@ -923,6 +923,14 @@ export class Parser { end: part.end, }; } + case "ext-glob": + return { + type: "ExtGlob", + op: part.op, + pattern: part.pattern, + pos: part.pos, + end: part.end, + }; } } diff --git a/src/tokenizer/scan-extglob.ts b/src/tokenizer/scan-extglob.ts new file mode 100644 index 0000000..e19d1a6 --- /dev/null +++ b/src/tokenizer/scan-extglob.ts @@ -0,0 +1,48 @@ +import type { SourceMap } from "./cursor"; +import type { TokenWordPart } from "./types"; + +type ExtGlobOp = "?(" | "*(" | "+(" | "@(" | "!("; + +const isExtGlobHead = (c: string): c is "?" | "*" | "+" | "@" | "!" => + c === "?" || c === "*" || c === "+" || c === "@" || c === "!"; + +/** + * If `source[pos]` is the start of an extended glob (`?(`, `*(`, `+(`, + * `@(`, or `!(`), scan to the matching `)` and return an `ext-glob` part. + * Returns null otherwise. Properly handles nested parens. + */ +export function scanExtGlob( + source: string, + pos: number, + map: SourceMap, +): { part: TokenWordPart; end: number } | null { + const head = source.charAt(pos); + if (!isExtGlobHead(head)) return null; + if (source.charAt(pos + 1) !== "(") return null; + + const op = `${head}(` as ExtGlobOp; + let j = pos + 2; + let depth = 1; + while (j < source.length) { + const ch = source.charAt(j); + if (ch === "(") depth++; + else if (ch === ")") { + depth--; + if (depth === 0) { + const pattern = source.slice(pos + 2, j); + return { + part: { + type: "ext-glob", + op, + pattern, + pos: map.posAt(pos), + end: map.posAt(j + 1), + }, + end: j + 1, + }; + } + } + j++; + } + return null; +} diff --git a/src/tokenizer/tokenize.ts b/src/tokenizer/tokenize.ts index 10f8403..47a062e 100644 --- a/src/tokenizer/tokenize.ts +++ b/src/tokenizer/tokenize.ts @@ -3,6 +3,7 @@ import { isDigit, operatorChars, redirChars, symbolChars } from "./charsets"; import { SourceMap } from "./cursor"; import { scanBacktick } from "./scan-backtick"; import { scanExpansion } from "./scan-expansion"; +import { scanExtGlob } from "./scan-extglob"; import { tryRedirOp } from "./scan-redir"; import type { SymbolTokenValue, Token, TokenWordPart } from "./types"; import { tokenPartsText } from "./utils"; @@ -119,15 +120,20 @@ export function tokenize(source: string, options: ParseOptions = {}): Token[] { } if (ch === "!" && atBoundary) { - tokens.push({ - type: "op", - value: "!", - pos: map.posAt(i), - end: map.posAt(i + 1), - }); - atBoundary = true; - i += 1; - continue; + // Prefer the extended-glob form `!(...)` over the negation operator, + // matching Bash where `!` only negates a command when it's a standalone + // word. + if (source.charAt(i + 1) !== "(") { + tokens.push({ + type: "op", + value: "!", + pos: map.posAt(i), + end: map.posAt(i + 1), + }); + atBoundary = true; + i += 1; + continue; + } } if (isDigit(ch)) { @@ -336,6 +342,26 @@ export function tokenize(source: string, options: ParseOptions = {}): Token[] { } } + // Try to recognize an extended glob (`?(`, `*(`, `+(`, `@(`, `!(`) + // before the break check, since `(` is otherwise a word terminator. + if ( + (currentChar === "?" || + currentChar === "*" || + currentChar === "+" || + currentChar === "@" || + currentChar === "!") && + source.charAt(i + 1) === "(" + ) { + const eg = scanExtGlob(source, i, map); + if (eg) { + flushLit(); + parts.push(eg.part); + i = eg.end; + litStart = i; + continue; + } + } + if ( currentChar === " " || currentChar === "\t" || diff --git a/src/tokenizer/types.ts b/src/tokenizer/types.ts index 5eb5e59..4ec64e1 100644 --- a/src/tokenizer/types.ts +++ b/src/tokenizer/types.ts @@ -26,7 +26,12 @@ export type TokenWordPart = raw: string; innerOffset: number; }) - | (WithPos & { type: "backtick"; raw: string; innerOffset: number }); + | (WithPos & { type: "backtick"; raw: string; innerOffset: number }) + | (WithPos & { + type: "ext-glob"; + op: "?(" | "*(" | "+(" | "@(" | "!("; + pattern: string; + }); export type Token = | (WithPos & { type: "word"; parts: TokenWordPart[] }) -- 2.51.2