From b8c227c7c2d28e64b719c0c27ef0419cd0ef4f31 Mon Sep 17 00:00:00 2001 From: Okiki Ojo Date: Wed, 22 Apr 2026 02:42:26 -0400 Subject: [PATCH] feat(parser): introduce strict diagnostics and recovery options for event parsing Signed-off-by: Okiki Ojo --- parse.ts | 150 ++++++++++++++++++++---------------- parse_test.ts | 207 ++++++++++++++++++++++++++++++++++++++++++++++---- 2 files changed, 277 insertions(+), 80 deletions(-) diff --git a/parse.ts b/parse.ts index 90ffa13..b98726e 100644 --- a/parse.ts +++ b/parse.ts @@ -12,16 +12,15 @@ * tokens(source) -> raw tokenizer output * outlineEvents(source) -> block structure only * events(source) -> block + inline event stream - * parse(source) -> loose full tree - * parseWithDiagnostics(source) -> strict tree + diagnostics - * parseWithRecovery(source) -> recovered tree + recovered + diagnostics + * parse(source) -> default tree + * parseWithDiagnostics(source) -> default tree + diagnostics + * parseStrictWithDiagnostics(source) -> conservative tree + diagnostics + * parseWithRecovery(source) -> default tree + recovered + diagnostics * ``` * - * `strict` and `loose` describe recovery shape, not parser acceptance. Both - * lanes still recover and still return a valid tree. - * - * That keeps the cost model visible. Callers can stop at the cheapest layer - * that answers their question instead of always paying for a full tree. + * The key split is now diagnostic emission first, then materialization policy. + * If a caller does not want diagnostics, the block and inline stages should + * not emit diagnostic events for that lane. * * @example Walking the full event stream * ```ts @@ -44,17 +43,32 @@ import type { ParseDiagnosticsResult, ParseResult } from './tree_builder.ts'; import { tokenize } from './tokenizer.ts'; import { blockEvents } from './block_parser.ts'; import { inlineEvents } from './inline_parser.ts'; -import { buildTree, buildTreeWithDiagnostics, buildTreeWithRecovery } from './tree_builder.ts'; +import { buildTree, buildTreeStrict, buildTreeWithDiagnostics, buildTreeWithRecovery } from './tree_builder.ts'; + +/** + * Public switches for event-stream production. + * + * The main cost choice here is whether parser diagnostics should be emitted at + * all. If `diagnostics` is omitted or `false`, the block and inline + * stages stay on the cheapest event lane and do not emit `error` events. + */ +export interface EventOptions { + /** Whether block and inline stages should emit diagnostic events. */ + readonly diagnostics?: boolean; +} /** * Internal event-pipeline switches used to keep the public API cost-aware. * - * The parser has three public tree lanes: + * The parser exposes one default tree lane and two diagnostics-preserving + * variants built on the same event pipeline. * * ```text - * parse() -> no diagnostics, loose recovery shape - * parseWithDiagnostics() -> diagnostics on, strict recovery shape - * parseWithRecovery() -> diagnostics on, loose recovery shape + * parse() -> default tree, no diagnostics + * parseWithDiagnostics() -> default tree + diagnostics + * parseStrictWithDiagnostics() + * -> conservative tree + diagnostics + * parseWithRecovery() -> default tree + diagnostics + recovered summary * ``` * * These options are the plumbing that keeps those lanes honest. Without them, @@ -62,10 +76,10 @@ import { buildTree, buildTreeWithDiagnostics, buildTreeWithRecovery } from './tr * the same underlying work. */ interface EventPipelineOptions { - /** Whether block and inline stages should emit recovery diagnostics. */ - readonly include_diagnostics?: boolean; - /** Whether recoverable inline/tree structures stay loose or collapse to text. */ - readonly recovery_style?: 'loose' | 'strict'; + /** Whether block and inline stages should emit parser diagnostics. */ + readonly diagnostics?: boolean; + /** Whether malformed inline/tree regions keep the default tree overlay or collapse to text. */ + readonly recovery?: 'default' | 'conservative'; } /** @@ -84,10 +98,11 @@ export function tokens(source: TextSource): Generator { * Inline content remains plain text ranges. This is the cheap structural mode * for outlines, table-of-contents extraction, and other block-focused tools. */ -export function outlineEvents(source: TextSource): Generator { - return outlineEventsWithOptions(source, { - include_diagnostics: true, - }); +export function outlineEvents( + source: TextSource, + options: EventOptions = {}, +): Generator { + return outlineEventsWithOptions(source, options); } /** @@ -96,10 +111,13 @@ export function outlineEvents(source: TextSource): Generator { * This is the default event-level API. It runs the tokenizer, block parser, * and inline enrichment in order. */ -export function events(source: TextSource): Generator { +export function events( + source: TextSource, + options: EventOptions = {}, +): Generator { return eventsWithOptions(source, { - include_diagnostics: true, - recovery_style: 'loose', + diagnostics: options.diagnostics, + recovery: 'default', }); } @@ -108,7 +126,7 @@ function outlineEventsWithOptions( options: EventPipelineOptions, ): Generator { return blockEvents(source, tokenize(source), { - include_diagnostics: options.include_diagnostics, + diagnostics: options.diagnostics, }); } @@ -142,8 +160,8 @@ function eventsFromOutline( options: EventPipelineOptions, ): Generator { return inlineEvents(source, outline, { - include_diagnostics: options.include_diagnostics, - recovery_style: options.recovery_style, + diagnostics: options.diagnostics, + recovery: options.recovery, }); } @@ -153,42 +171,46 @@ function eventsFromOutline( * This is the convenience API for callers that want a full AST and do not need * to inspect the intermediate event stream themselves. * - * It is also the cheapest tree-building lane. It does not request recovery - * diagnostics from the block or inline stages, and it keeps the loose tree - * shape when recovery is needed internally. - * - * `loose` means the final tree keeps more recovered wrapper structure when the - * parser can still infer something usable from the source. + * It is also the cheapest tree-building lane. It does not request diagnostics + * from the block or inline stages, and it keeps the default tolerant + * HTML-like tree shape when malformed input is encountered. * * If the caller also needs diagnostics or explicit recovery metadata, use * {@linkcode parseWithDiagnostics} or {@linkcode parseWithRecovery} instead. */ export function parse(source: TextSource): WikistRoot { return buildTree(eventsWithOptions(source, { - include_diagnostics: false, - recovery_style: 'loose', + diagnostics: false, + recovery: 'default', }), { source }); } /** - * Parse source text into a strict wikist tree and keep recovery diagnostics. - * - * This is the diagnostics-focused entry point. It returns a stricter - * tree than {@linkcode parse} when recovery would otherwise synthesize wrapper - * nodes, plus the diagnostics that explain those recovery points. + * Parse source text into the default wikist tree and keep diagnostics. * - * `strict` means the final tree is stricter about preserving only structure - * that the source clearly committed to. Recovery still happens, but - * recovery-heavy wrappers are more likely to collapse back to plain text. - * - * In practical terms, this is the lane to use when a caller wants to surface - * problems to a user, lint malformed input, or inspect where recovery happened - * without fully committing to the loose recovered shape. + * This is the diagnostics-first entry point. It preserves the same default + * HTML-like tree shape as {@linkcode parse}, but also returns the diagnostics + * that describe malformed input and parser continuation points. */ export function parseWithDiagnostics(source: TextSource): ParseDiagnosticsResult { return buildTreeWithDiagnostics(eventsWithOptions(source, { - include_diagnostics: true, - recovery_style: 'strict', + diagnostics: true, + recovery: 'default', + }), { source }); +} + +/** + * Parse source text into a conservative tree and keep diagnostics. + * + * This is the source-strict materialization lane. It still keeps diagnostics + * and still follows the never-throw contract, but recovery-heavy wrappers are + * more likely to collapse back to plain text when the source never clearly + * committed to them. + */ +export function parseStrictWithDiagnostics(source: TextSource): ParseDiagnosticsResult { + return buildTreeStrict(eventsWithOptions(source, { + diagnostics: true, + recovery: 'conservative', }), { source }); } @@ -196,33 +218,31 @@ export function parseWithDiagnostics(source: TextSource): ParseDiagnosticsResult * Parse source text into a wikist tree and report whether recovery happened. * * This is the explicit recovery-aware entry point. It returns the same - * loose tree as {@linkcode parse}, plus a `recovered` flag and the + * default tree as {@linkcode parse}, plus a `recovered` flag and the * diagnostics that explain what the parser had to do on the caller's behalf. * * Read the result like two coordinated lanes: * * ```text * source - * ├─► parse() -> loose tree only - * ├─► parseWithDiagnostics() -> strict tree + diagnostics - * └─► parseWithRecovery() -> recovered tree + recovered + diagnostics + * ├─► parse() -> tree only + * ├─► parseWithDiagnostics() -> tree + diagnostics + * ├─► parseStrictWithDiagnostics() + * │ -> conservative tree + diagnostics + * └─► parseWithRecovery() -> tree + recovered + diagnostics * ``` * * The important distinction from `parseWithDiagnostics()` is not just the - * extra boolean. This lane also keeps the loose recovered tree itself. - * That makes it the right fit for tolerant rendering, content transforms, or - * downstream tools that want best-effort structure plus an explicit signal - * that recovery happened. + * extra boolean. This lane adds an explicit summary field for consumers that + * want the parser's tolerant default behavior to stay visible in control flow. * * The diagnostics include a narrow `anchor` so downstream tools can resolve * the nearest node around the recovery point. * - * Today those diagnostics mostly come from block-parser recovery events and - * tree-builder recovery steps. `parse()` intentionally drops them, - * `parseWithDiagnostics()` preserves them while stripping recovery-created - * wrapper nodes back to strict text ranges where possible, and - * `parseWithRecovery()` keeps the more aggressively recovered tree plus the - * explicit `recovered` summary. + * Today those diagnostics mostly come from block-parser findings and + * tree-builder continuation steps. `parse()` intentionally drops them, + * `parseWithDiagnostics()` preserves them with the default tree, and + * `parseWithRecovery()` adds the explicit `recovered` summary. * * That anchor is intentionally tree-only today. Edit-stable anchor semantics * belong to later session/edit tracking work and are not part of this public @@ -230,7 +250,7 @@ export function parseWithDiagnostics(source: TextSource): ParseDiagnosticsResult */ export function parseWithRecovery(source: TextSource): ParseResult { return buildTreeWithRecovery(eventsWithOptions(source, { - include_diagnostics: true, - recovery_style: 'loose', + diagnostics: true, + recovery: 'default', }), { source }); } \ No newline at end of file diff --git a/parse_test.ts b/parse_test.ts index eafa150..0ac8038 100644 --- a/parse_test.ts +++ b/parse_test.ts @@ -12,10 +12,16 @@ import { spacing_heavy_wikitext_string, wikiish_string, } from './_test_utils/arbitraries.ts'; +import { + BARE_URI_ACCEPTANCE_FIXTURES, + BARE_URI_REJECTION_FIXTURES, + EXPLICIT_URI_ACCEPTANCE_FIXTURES, +} from './_test_utils/uri_fixtures.ts'; import { UNICODE_TEXT_FIXTURES } from './_test_utils/unicode_fixtures.ts'; import { buildTree, buildTreeWithDiagnostics, + buildTreeStrict, buildTreeWithLooseDiagnostics, buildTreeWithRecovery, type ParseDiagnosticsResult, @@ -27,9 +33,10 @@ import { errorEvent, exitEvent, textEvent, + type WikitextEvent, } from './events.ts'; import { blockEvents } from './block_parser.ts'; -import { events, outlineEvents, parse, parseWithDiagnostics, parseWithRecovery, tokens } from './parse.ts'; +import { events, outlineEvents, parse, parseStrictWithDiagnostics, parseWithDiagnostics, parseWithRecovery, tokens } from './parse.ts'; import { tokenize } from './tokenizer.ts'; const COMPLEX_PARSE_FIXTURES = [ @@ -55,6 +62,90 @@ const COMPLEX_PARSE_FIXTURES = [ ].join('\n'), ] as const; +const BLOCK_STRUCTURE_FIXTURES = [ + ...COMPLEX_PARSE_FIXTURES, + 'Paragraph with note', + 'Paragraph with ): [string, string][] { + const result: [string, string][] = []; + + for (const event of stream) { + if ((event.kind === 'enter' || event.kind === 'exit') && BLOCK_NODE_TYPE_LOOKUP.has(event.node_type)) { + result.push([event.kind, event.node_type]); + } + } + + return result; +} + +function blockStructureFromTree(root: TreeLikeNode): [string, string][] { + const result: [string, string][] = []; + + function walk(node: TreeLikeNode): void { + const is_block = BLOCK_NODE_TYPE_LOOKUP.has(node.type); + + if (is_block) { + result.push(['enter', node.type]); + } + + for (const child of node.children ?? []) { + walk(child); + } + + if (is_block) { + result.push(['exit', node.type]); + } + } + + walk(root); + return result; +} + describe('orchestration', () => { it('tokens() aliases the tokenizer output', () => { const input = '== Heading =='; @@ -66,7 +157,7 @@ describe('orchestration', () => { it('outlineEvents() matches the block parser pipeline', () => { const input = '== Heading ==\n\nParagraph'; - const direct = Array.from(blockEvents(input, tokenize(input))); + const direct = Array.from(blockEvents(input, tokenize(input), { diagnostics: false })); const orchestrated = Array.from(outlineEvents(input)); expect(orchestrated).toEqual(direct); @@ -80,6 +171,27 @@ describe('orchestration', () => { expect(structure).toContain('wikilink'); }); + + it('keeps block structure consistent across outlineEvents(), events(), and parse()', () => { + for (const input of BLOCK_STRUCTURE_FIXTURES) { + const outline_structure = blockStructureFromEvents(outlineEvents(input)); + const full_structure = blockStructureFromEvents(events(input)); + const tree_structure = blockStructureFromTree(parse(input)); + + expect(full_structure).toEqual(outline_structure); + expect(tree_structure).toEqual(outline_structure); + } + }); + + it('preserves the shared URI acceptance matrix in final tree output', () => { + for (const fixture of [...BARE_URI_ACCEPTANCE_FIXTURES, ...EXPLICIT_URI_ACCEPTANCE_FIXTURES]) { + expect(externalLinkUrlsFromTree(parse(fixture.input))).toContain(fixture.url); + } + + for (const input of BARE_URI_REJECTION_FIXTURES) { + expect(externalLinkUrlsFromTree(parse(input))).toEqual([]); + } + }); }); describe('buildTree()', () => { @@ -362,7 +474,7 @@ describe('buildTree()', () => { expect(result.diagnostics[0].anchor).toEqual({ kind: 'tree-path', path: [0, 0], - node_type: 'text', + node_type: 'bold', }); expect(result.diagnostics[0].recoverable).toBe(true); expect(result.diagnostics[0].source).toBe('tree'); @@ -407,21 +519,30 @@ describe('buildTree()', () => { expect(result.diagnostics[0].anchor).toEqual({ kind: 'tree-path', path: [0], - node_type: 'text', + node_type: 'paragraph', }); }); - it('buildTreeWithLooseDiagnostics() keeps the loose tree while preserving diagnostics', () => { + it('buildTreeWithLooseDiagnostics() stays as a compatibility alias for buildTreeWithDiagnostics()', () => { const input = '{|\n| Cell'; - const loose_result = buildTreeWithLooseDiagnostics(events(input), { source: input }); - const strict_result = buildTreeWithDiagnostics(events(input), { source: input }); - const recovery_result = buildTreeWithRecovery(events(input), { source: input }); + const event_stream = events(input, { diagnostics: true }); + const loose_result = buildTreeWithLooseDiagnostics(event_stream, { source: input }); + const diagnostics_result = buildTreeWithDiagnostics(events(input, { diagnostics: true }), { source: input }); + const recovery_result = buildTreeWithRecovery(events(input, { diagnostics: true }), { source: input }); expect(Object.hasOwn(loose_result, 'recovered')).toBe(false); expect(loose_result.diagnostics).toEqual(recovery_result.diagnostics); - expect(loose_result.tree).toEqual(recovery_result.tree); + expect(loose_result.tree).toEqual(diagnostics_result.tree); expect(loose_result.tree.children[0]?.type).toBe('table'); + expect(diagnostics_result.tree.children[0]?.type).toBe('table'); + }); + + it('buildTreeStrict() keeps diagnostics while collapsing recovery-heavy wrappers', () => { + const input = '{|\n| Cell'; + const strict_result = buildTreeStrict(events(input, { diagnostics: true }), { source: input }); + expect(strict_result.tree.children[0]?.type).toBe('text'); + expect(strict_result.diagnostics[0]?.anchor.node_type).toBe('text'); }); }); @@ -685,7 +806,7 @@ describe('parseWithDiagnostics()', () => { expect(result.diagnostics[0].anchor).toEqual({ kind: 'tree-path', path: [0], - node_type: 'text', + node_type: 'table', }); expect(result.diagnostics[0].recoverable).toBe(true); expect(result.diagnostics[0].severity).toBe('warning'); @@ -715,7 +836,7 @@ describe('parseWithDiagnostics()', () => { expect(result.diagnostics[0].source).toBe('inline'); expect(result.diagnostics[0].recoverable).toBe(true); expect(result.diagnostics[0].severity).toBe('warning'); - expect(result.diagnostics[0].anchor.node_type).toBe('text'); + expect(result.diagnostics[0].anchor.node_type).toBe('reference'); }); it('keeps the same tree as parse() when no diagnostics were emitted', () => { @@ -736,10 +857,45 @@ describe('parseWithDiagnostics()', () => { ); }); - it('keeps missing-close tags as plain text instead of recovered nodes', () => { + it('keeps missing-close tags as structurally real nodes in the default diagnostics lane', () => { const input = 'Paragraph with note'; const result = parseWithDiagnostics(input); + expect(JSON.stringify(result.tree)).toContain('"type":"reference"'); + }); + + it('keeps unclosed tables as recovered table nodes in the default diagnostics lane', () => { + const input = '{|\n| Cell'; + const result = parseWithDiagnostics(input); + + expect(result.tree.children[0]?.type).toBe('table'); + }); + + it('keeps the default block structure stable when inline recovery happens', () => { + const input = 'Paragraph with note'; + const outline_structure = blockStructureFromEvents(outlineEvents(input)); + const diagnostics_result = parseWithDiagnostics(input); + const recovery_result = parseWithRecovery(input); + + expect(blockStructureFromTree(diagnostics_result.tree)).toEqual(outline_structure); + expect(blockStructureFromTree(recovery_result.tree)).toEqual(outline_structure); + }); +}); + +describe('parseStrictWithDiagnostics()', () => { + it('preserves the same diagnostics as parseWithDiagnostics()', () => { + for (const input of COMPLEX_PARSE_FIXTURES) { + const diagnostics_result = parseWithDiagnostics(input); + const strict_result = parseStrictWithDiagnostics(input); + + expect(strict_result.diagnostics).toEqual(diagnostics_result.diagnostics); + } + }); + + it('keeps missing-close tags as plain text instead of recovered nodes', () => { + const input = 'Paragraph with note'; + const result = parseStrictWithDiagnostics(input); + expect(JSON.stringify(result.tree)).not.toContain('"type":"reference"'); expect(result.tree.children[0]?.type).toBe('text'); if (result.tree.children[0]?.type !== 'text') return; @@ -748,21 +904,41 @@ describe('parseWithDiagnostics()', () => { it('keeps unclosed tables as plain text instead of recovered table nodes', () => { const input = '{|\n| Cell'; - const result = parseWithDiagnostics(input); + const result = parseStrictWithDiagnostics(input); expect(result.tree.children[0]?.type).toBe('text'); if (result.tree.children[0]?.type !== 'text') return; expect(result.tree.children[0].value).toBe(input); }); + + it('may diverge from the structural overlay when a committed malformed region collapses to text', () => { + const input = 'Paragraph with note'; + const outline_structure = blockStructureFromEvents(outlineEvents(input)); + const strict_result = parseStrictWithDiagnostics(input); + + expect(blockStructureFromTree(strict_result.tree)).not.toEqual(outline_structure); + expect(strict_result.tree.children[0]?.type).toBe('text'); + }); + + it('matches the default tree when the source never reaches the commitment point', () => { + const input = 'Paragraph with { - it('adds recovery-shaped tree changes on top of parseWithDiagnostics()', () => { + it('adds the recovery summary on top of parseWithDiagnostics()', () => { for (const input of COMPLEX_PARSE_FIXTURES) { const diagnostics_result = parseWithDiagnostics(input); const recovery_result = parseWithRecovery(input); expect(recovery_result.diagnostics).toEqual(diagnostics_result.diagnostics); + expect(recovery_result.tree).toEqual(diagnostics_result.tree); expect(recovery_result.recovered).toBe(diagnostics_result.diagnostics.length > 0); } }); @@ -781,10 +957,11 @@ describe('parseWithRecovery()', () => { expect(result.diagnostics.length).toBeGreaterThan(0); }); - it('keeps recovery-specific wrapper nodes that parseWithDiagnostics() strips', () => { + it('keeps the same default tree shape as parseWithDiagnostics()', () => { const input = 'Paragraph with note'; const result = parseWithRecovery(input); expect(JSON.stringify(result.tree)).toContain('"type":"reference"'); + expect(result.tree).toEqual(parseWithDiagnostics(input).tree); }); }); \ No newline at end of file -- 2.51.2