From a760d891a0f8bb3196a107ed9a62adfbe10b8d7d Mon Sep 17 00:00:00 2001 From: Okiki Ojo Date: Wed, 11 Mar 2026 02:29:59 -0400 Subject: [PATCH] bench(events): add micro-benchmarks for event constructors and token validation Signed-off-by: Okiki Ojo --- mod_bench.ts | 594 ++++++--------------------------------------------- 1 file changed, 60 insertions(+), 534 deletions(-) diff --git a/mod_bench.ts b/mod_bench.ts index 38713da..4238d64 100644 --- a/mod_bench.ts +++ b/mod_bench.ts @@ -1,579 +1,105 @@ /** - * Benchmarks for foundational wikitext types. + * Foundational micro-benchmarks that do not belong to a specific parser layer. * - * These benchmarks measure the raw cost of the most frequently called - * operations: the `isToken()` type guard and the five event constructors. - * They establish a baseline so future changes that regress hot-path - * performance are caught immediately. - * - * Uses [mitata](https://github.com/nicolo-ribaudo/mitata) as the - * benchmark harness. Key rule: every benchmark must wrap its result in - * `do_not_optimize()` to prevent V8's JIT from eliminating dead code - * (a common source of misleadingly fast numbers). - * - * Run: - * ```sh - * deno bench --allow-env=NODE_DISABLE_COLORS --v8-flags=--expose-gc mod_bench.ts - * ``` + * Layer-specific suites live in dedicated files for faster iteration. * * @module bench */ // deno-lint-ignore-file no-import-prefix no-unversioned-import -import { - bench, - do_not_optimize, - run, - summary, -} from 'npm:mitata'; +import { bench, do_not_optimize, run, summary } from 'npm:mitata'; import { - blockEvents, - TokenType, enterEvent, errorEvent, exitEvent, - inlineEvents, - isToken, textEvent, tokenEvent, - tokenize, -} from './mod.ts'; +} from './events.ts'; +import { isToken, TokenType } from './token.ts'; -/** A minimal Position for benchmarks. */ -const pos = { - start: { line: 1, column: 1, offset: 0 }, - end: { line: 1, column: 5, offset: 4 }, -} as const; - -/** A valid token: isToken() should return true for this. */ -const validToken = { type: TokenType.TEXT, start: 0, end: 4 }; -/** An invalid token: type is not a known TokenType value. */ -const invalidToken = { type: 'UNKNOWN', start: 0, end: 4 }; +type BenchPoint = { + line: number; + column: number; + offset: number; +}; -/** - * Rotate through a small fixed input set so the JIT cannot fully specialize on - * one exact literal across the whole benchmark run. - */ -function cycleInputs(inputs: readonly T[]): () => T { - let index = 0; +type BenchPosition = { + start: BenchPoint; + end: BenchPoint; +}; - return () => { - const input = inputs[index]; - index = (index + 1) % inputs.length; - return input; +function createBenchPosition(offset: number): BenchPosition { + return { + start: { + line: 1, + column: offset + 1, + offset, + }, + end: { + line: 1, + column: offset + 5, + offset: offset + 4, + }, }; } -/** Repeat a multi-line benchmark unit without measuring string assembly. */ -function repeatBlock(unit: string, repeat: number): string { - return unit.repeat(repeat); -} - -/** - * Repeat an ASCII unit until the resulting text reaches at least `minimumSize`. - * - * This is used for large-file benchmarks where the on-disk size matters more - * than the exact block count. The units in this file are ASCII-only, so code - * unit length is a practical stand-in for byte size. - */ -function repeatToMinimumSize(unit: string, minimumSize: number): string { - const repeat = Math.ceil(minimumSize / unit.length); - return unit.repeat(repeat); -} +const VALID_TOKEN = { + type: TokenType.TEXT, + start: 0, + end: 4, +}; -/** Minimal state shape used by mitata parameterized benchmarks in this file. */ -type RangeState = { - get(name: string): number; +const INVALID_TOKEN = { + type: 'text', + start: 0, + end: 4, }; -// --- Type guard benchmarks --- -// isToken() is called on every token in the stream by downstream -// consumers. It must be fast (sub-nanosecond on modern hardware). summary(() => { - bench('isToken(valid)', () => { - do_not_optimize(isToken(validToken)); + bench('isToken(): valid token', () => { + do_not_optimize(isToken(VALID_TOKEN)); }); - bench('isToken(invalid)', () => { - do_not_optimize(isToken(invalidToken)); + bench('isToken(): invalid token', () => { + do_not_optimize(isToken(INVALID_TOKEN)); }); }); -// --- Event constructor benchmarks --- -// Event constructors are called once per event in the stream. For a large -// article producing 100K+ events, even small per-call costs add up. summary(() => { - bench('enterEvent()', () => { - do_not_optimize(enterEvent('paragraph', {}, pos)); + bench('event constructor: enter', () => { + const position = createBenchPosition(12); + do_not_optimize(enterEvent('paragraph', {}, position)); }); - bench('exitEvent()', () => { - do_not_optimize(exitEvent('paragraph', pos)); + bench('event constructor: exit', () => { + const position = createBenchPosition(12); + do_not_optimize(exitEvent('paragraph', position)); }); - bench('textEvent()', () => { - do_not_optimize(textEvent(0, 4, pos)); + bench('event constructor: text', () => { + const position = createBenchPosition(12); + do_not_optimize(textEvent(position.start.offset, position.end.offset, position)); }); - bench('tokenEvent()', () => { - do_not_optimize(tokenEvent(TokenType.TEXT, 0, 4, pos)); + bench('event constructor: token', () => { + const position = createBenchPosition(12); + do_not_optimize( + tokenEvent(TokenType.TEXT, position.start.offset, position.end.offset, position), + ); }); - bench('errorEvent()', () => { - do_not_optimize(errorEvent('Malformed inline run', pos)); + bench('event constructor: error', () => { + const position = createBenchPosition(12); + do_not_optimize(errorEvent('invalid-benchmark', position)); }); }); -// --- Tokenizer benchmarks --- -// The tokenizer is the hottest path in the parser pipeline: -// it scans every character of the input. These benchmarks measure -// throughput on representative wikitext inputs. - -const PLAIN_TEXT_INPUTS = [ - 'The quick brown fox jumps over the lazy dog. '.repeat(200), - 'Pack my box with five dozen liquor jugs. '.repeat(190), - 'Sphinx of black quartz, judge my vow. '.repeat(205), -] as const; -const HEADING_TEXT_INPUTS = [ - '== Section ==\nParagraph text here.\n'.repeat(100), - '=== Nested ===\nAnother paragraph line.\n'.repeat(90), -] as const; -const TABLE_TEXT_INPUTS = [ - '{|\n! H1 !! H2\n|-\n| A || B\n|-\n| C || D\n|}\n'.repeat(50), - '{| class="wikitable"\n! Name !! Value\n|-\n| Alpha || 1\n|-\n| Beta || 2\n|}\n'.repeat(40), -] as const; -const LINK_TEXT_INPUTS = [ - "See [[Main Page|home]], '''bold''' and ''italic'' text.\n".repeat(100), - 'Visit [[Earth|planet]] with [https://example.com source] and & notes.\n'.repeat(80), -] as const; -const TEMPLATE_TEXT_INPUTS = [ - '{{Infobox|name={{{1}}}|value={{{2|default}}}}}\n'.repeat(100), - '{{Card|title={{PAGENAME}}|body={{{content|fallback}}}}}\n'.repeat(85), -] as const; -const MIXED_TEXT_INPUTS = [ - [ - '== Heading ==', - "'''Bold''' and ''italic'' and '''''both'''''.", - '* Bullet item', - '# Ordered item', - ': Indented', - '{|', - '! Header', - '|-', - '| [[Page|link]] || {{template|arg=val}}', - '|}', - '----', - '', - '& { 💩', - '~~~~ __TOC__', - '', - ].join('\n').repeat(50), - [ - '== Another Heading ==', - "A [[Main Page|home]] link with ''italic'' and '''bold'''.", - '; Term', - ': Definition', - '{|', - '! Name !! Count', - '|-', - '| {{Item|name=Alpha}} || 42', - '|}', - 'inline', - '<escaped> ~~ ~~ __TOC__', - '', - ].join('\n').repeat(48), -] as const; -const PATHOLOGICAL_TEXT_INPUTS = [ - [ - '[[[[{{{{