diff --git a/src/index.ts b/src/index.ts index 70c4a30..4d05cf0 100644 --- a/src/index.ts +++ b/src/index.ts @@ -64,4 +64,5 @@ export { type WordPart, } from "./ast"; export { parse } from "./parse"; +export { parseStmtsSeq, parseWordsSeq } from "./seq"; export { splitBraces } from "./split-braces"; diff --git a/src/parser/iterators.test.ts b/src/parser/iterators.test.ts new file mode 100644 index 0000000..2f9e508 --- /dev/null +++ b/src/parser/iterators.test.ts @@ -0,0 +1,56 @@ +import { describe, expect, it } from "vitest"; +import type { SimpleCommand } from "../ast"; +import { parseStmtsSeq, parseWordsSeq } from "../seq"; + +describe("parseStmtsSeq", () => { + it("yields statements in order", () => { + const stmts = [...parseStmtsSeq("foo; bar; baz")]; + expect(stmts).toHaveLength(3); + const names = stmts.map((s) => { + const cmd = s.command as SimpleCommand; + const part = cmd.words?.[0]?.parts[0]; + return part?.type === "Literal" ? part.value : null; + }); + expect(names).toEqual(["foo", "bar", "baz"]); + }); + + it("handles a single statement", () => { + const stmts = [...parseStmtsSeq("hello world")]; + expect(stmts).toHaveLength(1); + }); + + it("handles empty input", () => { + expect([...parseStmtsSeq("")]).toHaveLength(0); + expect([...parseStmtsSeq("\n\n")]).toHaveLength(0); + }); + + it("can be consumed lazily (only the parser consumes work as the caller advances)", () => { + let count = 0; + for (const _ of parseStmtsSeq("a; b; c; d; e")) { + count++; + if (count === 2) break; + } + expect(count).toBe(2); + }); +}); + +describe("parseWordsSeq", () => { + it("yields each whitespace-separated word", () => { + const words = [...parseWordsSeq("foo bar 'baz qux' $var")]; + expect(words).toHaveLength(4); + expect(words[0]?.parts[0]).toMatchObject({ type: "Literal", value: "foo" }); + expect(words[2]?.parts[0]).toMatchObject({ + type: "SglQuoted", + value: "baz qux", + }); + expect(words[3]?.parts[0]?.type).toBe("ParamExp"); + }); + + it("handles empty input", () => { + expect([...parseWordsSeq("")]).toHaveLength(0); + }); + + it("ignores trailing newlines", () => { + expect([...parseWordsSeq("foo\n")]).toHaveLength(1); + }); +}); diff --git a/src/parser/parser.ts b/src/parser/parser.ts index 3368c2d..a1bdc93 100644 --- a/src/parser/parser.ts +++ b/src/parser/parser.ts @@ -126,6 +126,38 @@ export class Parser { private readonly options: ParseOptions = {}, ) {} + /** Yield each top-level statement as it's parsed. */ + *statementsSeq(): Generator { + this.skipSeparators(); + while (!this.isEof()) { + yield this.parseStatement(); + this.skipSeparators(); + } + } + + /** Yield each word token as a Word, ignoring statement structure. */ + *wordsSeq(): Generator { + while (!this.isEof()) { + const tok = this.peek(); + if (!tok) break; + if ( + tok.type === "op" || + tok.type === "redir" || + tok.type === "symbol" || + tok.type === "heredoc-body" || + tok.type === "comment" || + tok.type === "arith-cmd" + ) { + // Skip non-word tokens; useful for argv-style sources where we + // only care about the word stream. + this.consume(); + continue; + } + this.consume(); + yield this.wordFromToken(tok); + } + } + parseProgram(): Program { const body: Statement[] = []; this.skipSeparators(); diff --git a/src/seq.ts b/src/seq.ts new file mode 100644 index 0000000..364411d --- /dev/null +++ b/src/seq.ts @@ -0,0 +1,34 @@ +import type { ParseOptions, Statement, Word } from "./ast"; +import { Parser } from "./parser"; +import { tokenize } from "./tokenizer"; + +/** + * Lazily parse `source` and yield each top-level statement as it becomes + * available. Useful for streaming consumers (REPLs, progressive analysis + * tools) that don't need the whole `Program` up front. + * + * Mirrors mvdan/sh's `Parser.StmtsSeq`. + */ +export function* parseStmtsSeq( + source: string, + options: ParseOptions = {}, +): Generator { + const tokens = tokenize(source, options); + const parser = new Parser(tokens, options); + yield* parser.statementsSeq(); +} + +/** + * Lazily parse `source` as a sequence of words (no statement structure), + * yielding each one. Useful for argv-style inputs. + * + * Mirrors mvdan/sh's `Parser.WordsSeq`. + */ +export function* parseWordsSeq( + source: string, + options: ParseOptions = {}, +): Generator { + const tokens = tokenize(source, options); + const parser = new Parser(tokens, options); + yield* parser.wordsSeq(); +}