diff --git a/.claude/skills/perf/SKILL.md b/.claude/skills/perf/SKILL.md index 64a577b..8109b72 100644 --- a/.claude/skills/perf/SKILL.md +++ b/.claude/skills/perf/SKILL.md @@ -1,7 +1,5 @@ --- -description: - Add or update Blit-Tech performance tests, choose between Vitest bench and Playwright perf, and explain the CI - benchmark workflow. +description: Add or update Blit-Tech CPU benchmarks and explain the CI benchmark workflow. --- # Performance Testing @@ -10,13 +8,12 @@ Use this skill when the task involves: - adding or extending `*.bench.ts` files - benchmarking a new hot method or allocation pattern -- deciding between CPU micro-benchmarks and browser/GPU perf tests -- working on `tests/perf/` or perf fixtures in `tests/visual/fixtures/` -- explaining or debugging benchmark CI behavior +- working on benchmark CI behavior -## Benchmark Types +For visual correctness verification (not performance), use `pnpm test:visual` — see the Visual Regression Tests section +in CLAUDE.md. -### Tier 1: CPU Benchmarks +## CPU Benchmarks Use Vitest bench for isolated hot paths that can run in Node. @@ -42,64 +39,17 @@ Rules: - prefer realistic hot-path inputs - use `pnpm bench:json` when validating CI-facing output -### Tier 2: GPU Performance Tests - -Use Playwright perf when the important metric is browser frame time. - -Examples: - -- sprite throughput -- texture batch switching -- bitmap text frame cost -- mixed primitive + sprite workloads - -Commands: - -```bash -pnpm test:perf -``` - -Rules: - -- add or extend a fixture in `tests/visual/fixtures/` -- add or extend the scenario in `tests/perf/perf.spec.ts` -- think in terms of workload shape and frame-time stats (`median`, `p95`, `p99`) -- Tier 2 uses Playwright-managed pinned Chromium for stable baselines -- current CI for Tier 2 runs on standard hosted runners, so treat it as approximate browser perf coverage unless a - self-hosted GPU runner exists - -## Choosing the Right Tool - -Use Tier 1 first if the code can run without a browser. - -Use Tier 2 when: - -- the code depends on browser rendering -- the user cares about whole-frame cost -- batching or texture switching is part of the question - -Use both when a rendering change affects both an inner-loop method and actual rendered frame time. - ## CI Behavior -Tier 1 and Tier 2 performance checks are in CI now, with separate labels. +CPU benchmark regression checks run in CI with the `perf` label. - `main` pushes run `pnpm bench:json` and refresh the stored baseline artifact -- PRs labeled `perf-tier-1` run `pnpm bench:json` +- PRs labeled `perf` run `pnpm bench:json` - CI compares against the latest successful `main` baseline artifact - CI posts or updates a PR comment with the comparison table - CI fails if any benchmark regresses by more than 10% -- `main` pushes run `pnpm test:perf` and refresh the stored GPU perf baseline artifact -- PRs labeled `perf-tier-2` run `pnpm test:perf` -- CI compares GPU perf results against the latest successful `main` GPU perf baseline artifact -- CI posts or updates a PR comment with the frame-time comparison table -- CI fails if any scenario has a median frame-time regression greater than 50% -- Tier 2 results are intentionally loose because the current CI environment is a standard hosted runner, not dedicated - GPU hardware - ## References -- Read `docs/performance-testing.md` when the user wants an explanation or documentation update -- Read `.github/workflows/ci.yml`, `scripts/compare-tier-1-benchmarks.mjs`, and - `scripts/compare-tier-2-perf-results.mjs` when the task is about benchmark CI +- Read `docs/performance-testing.md` for full documentation +- Read `.github/workflows/ci.yml` and `scripts/compare-tier-1-benchmarks.mjs` for benchmark CI details diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e700d80..625aaab 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -13,7 +13,7 @@ jobs: runs-on: ubuntu-latest if: github.event_name != 'pull_request' || !contains(fromJSON('["labeled","unlabeled"]'), github.event.action) || - (github.event.action == 'labeled' && contains(fromJSON('["perf-tier-1","perf-tier-2"]'), github.event.label.name)) + (github.event.action == 'labeled' && contains(fromJSON('["perf"]'), github.event.label.name)) steps: - name: Checkout code @@ -82,10 +82,10 @@ jobs: retention-days: 7 benchmark: - name: Tier 1 Benchmarks + name: Benchmarks runs-on: ubuntu-latest needs: quality - if: github.event_name == 'push' || contains(github.event.pull_request.labels.*.name, 'perf-tier-1') + if: github.event_name == 'push' || contains(github.event.pull_request.labels.*.name, 'perf') concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-benchmark cancel-in-progress: true @@ -112,7 +112,7 @@ jobs: - name: Install dependencies run: pnpm install --frozen-lockfile - - name: Run Tier 1 benchmarks + - name: Run benchmarks run: pnpm bench:json - name: Upload current benchmark results @@ -272,211 +272,6 @@ jobs: } EOF - perf: - name: Tier 2 GPU Perf - runs-on: ubuntu-latest - needs: quality - if: github.event_name == 'push' || contains(github.event.pull_request.labels.*.name, 'perf-tier-2') - concurrency: - group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-perf - cancel-in-progress: true - permissions: - actions: read - contents: read - pull-requests: write - - steps: - - name: Checkout code - uses: actions/checkout@v6 - - - name: Setup pnpm - uses: pnpm/action-setup@v4 - with: - version: 10.26.2 - - - name: Setup Node.js - uses: actions/setup-node@v6 - with: - node-version: '22' - cache: 'pnpm' - - - name: Install dependencies - run: pnpm install --frozen-lockfile - - - name: Install pinned Chromium for Playwright perf tests - run: pnpm exec playwright install --with-deps chromium - - - name: Run Tier 2 GPU perf tests - run: pnpm test:perf - - - name: Upload current perf results - uses: actions/upload-artifact@v7 - with: - name: perf-results-current - path: test-results/perf/perf-results.json - if-no-files-found: error - retention-days: 14 - - - name: Upload main perf baseline - if: github.event_name == 'push' && github.ref == 'refs/heads/main' - uses: actions/upload-artifact@v7 - with: - name: perf-baseline - path: test-results/perf/perf-results.json - if-no-files-found: error - retention-days: 90 - - - name: Find latest main perf baseline - id: find-perf-baseline - if: github.event_name == 'pull_request' - uses: actions/github-script@v8 - with: - script: | - const workflowId = 'ci.yml'; - const runs = await github.paginate(github.rest.actions.listWorkflowRuns, { - owner: context.repo.owner, - repo: context.repo.repo, - workflow_id: workflowId, - branch: 'main', - event: 'push', - status: 'completed', - per_page: 100, - }); - - for (const run of runs) { - if (run.conclusion !== 'success') { - continue; - } - - const artifacts = await github.paginate(github.rest.actions.listWorkflowRunArtifacts, { - owner: context.repo.owner, - repo: context.repo.repo, - run_id: run.id, - per_page: 100, - }); - const baselineArtifact = artifacts.find( - (artifact) => artifact.name === 'perf-baseline' && artifact.expired === false, - ); - - if (!baselineArtifact) { - continue; - } - - core.setOutput('found', 'true'); - core.setOutput('download-url', baselineArtifact.archive_download_url); - core.setOutput('artifact-id', String(baselineArtifact.id)); - core.setOutput('run-id', String(run.id)); - return; - } - - core.setOutput('found', 'false'); - - - name: Download main perf baseline - if: github.event_name == 'pull_request' && steps.find-perf-baseline.outputs.found == 'true' - env: - GITHUB_TOKEN: ${{ github.token }} - BASELINE_URL: ${{ steps.find-perf-baseline.outputs.download-url }} - run: | - mkdir -p perf-baseline-artifact - curl --fail --location \ - --header "Authorization: Bearer ${GITHUB_TOKEN}" \ - --header "Accept: application/vnd.github+json" \ - "${BASELINE_URL}" \ - --output perf-baseline-artifact.zip - unzip -o perf-baseline-artifact.zip -d perf-baseline-artifact - - - name: Compare perf results - if: github.event_name == 'pull_request' - run: | - if [ "${{ steps.find-perf-baseline.outputs.found }}" = "true" ]; then - node scripts/compare-tier-2-perf-results.mjs \ - --current test-results/perf/perf-results.json \ - --baseline perf-baseline-artifact/perf-results.json \ - --json-out perf-comparison.json \ - --markdown-out perf-comment.md \ - --threshold 50 - else - node scripts/compare-tier-2-perf-results.mjs \ - --current test-results/perf/perf-results.json \ - --json-out perf-comparison.json \ - --markdown-out perf-comment.md \ - --threshold 50 - fi - - - name: Upload perf comparison artifacts - if: github.event_name == 'pull_request' - uses: actions/upload-artifact@v7 - with: - name: perf-comparison - path: | - perf-comparison.json - perf-comment.md - test-results/perf/perf-results.json - if-no-files-found: ignore - retention-days: 14 - - - name: Post perf PR comment - if: - github.event_name == 'pull_request' && github.event.pull_request.head.repo.fork == false && github.actor != - 'dependabot[bot]' - uses: actions/github-script@v8 - env: - PERF_COMMENT_PATH: perf-comment.md - with: - script: | - const fs = require('node:fs'); - const marker = ''; - const body = fs.readFileSync(process.env.PERF_COMMENT_PATH, 'utf8'); - const comments = await github.paginate(github.rest.issues.listComments, { - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: context.issue.number, - per_page: 100, - }); - const existingComment = comments.find((comment) => - comment.user?.type === 'Bot' && comment.body?.includes(marker), - ); - - if (existingComment) { - await github.rest.issues.updateComment({ - owner: context.repo.owner, - repo: context.repo.repo, - comment_id: existingComment.id, - body, - }); - return; - } - - await github.rest.issues.createComment({ - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: context.issue.number, - body, - }); - - - name: Fail on perf regressions - if: github.event_name == 'pull_request' - run: | - node --input-type=module <<'EOF' - import fs from 'node:fs'; - - if (!fs.existsSync('perf-comparison.json')) { - process.exit(0); - } - - const report = JSON.parse(fs.readFileSync('perf-comparison.json', 'utf8')); - - if (!report.hasBaseline) { - process.exit(0); - } - - const hasFailures = report.summary.regressions > 0 || report.summary.missingScenarios > 0; - - if (hasFailures) { - process.exit(1); - } - EOF - test: name: Run Tests runs-on: ubuntu-latest diff --git a/CLAUDE.md b/CLAUDE.md index 9de3926..5cb9d58 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -107,21 +107,38 @@ pnpm test # Run all unit tests (alias for test:unit) pnpm test:unit # Run all unit tests pnpm test:unit:watch # Watch mode for development pnpm test:unit:coverage # Coverage report (80% minimum threshold) -pnpm test:visual # Playwright visual regression (requires Chrome) +pnpm test:visual # Playwright visual regression tests (requires Chrome with WebGPU) pnpm test:visual:update # Update visual test baselines -pnpm bench # Run Tier 1 CPU benchmarks (Vitest bench) -pnpm bench:json # Run Tier 1 benchmarks and write benchmark-results.json -pnpm test:perf # Run Tier 2 browser/GPU frame-time benchmarks +pnpm bench # Run CPU benchmarks (Vitest bench) +pnpm bench:json # Run benchmarks and write benchmark-results.json ``` **Test tiers:** 1. **Unit tests** (Vitest, node) - Pure logic: Vector2i, Rect2i, Color32, Palette, PaletteEffect, Easing, GameLoop 2. **Integration tests** (Vitest, Node + GPU mocks; happy-dom for DOM tests) - DOM and GPU code -3. **Visual regression** (Playwright, Chromium) - Rendering output verification -4. **Performance tests** - - Tier 1 CPU benchmarks (Vitest bench, `*.bench.ts`) for hot methods and allocation patterns - - Tier 2 GPU perf tests (Playwright, `tests/perf/`) for browser frame-time workloads +3. **Visual regression** (Playwright, Chromium + WebGPU) - PNG snapshot verification of rendered output +4. **CPU benchmarks** (Vitest bench, `*.bench.ts`) - Hot method and allocation pattern throughput + +### Visual Regression Tests + +`pnpm test:visual` runs Playwright with Chromium + WebGPU and captures PNG snapshots of actual rendered frames. This is +the primary tool for verifying that visual output is correct — not performance, but pixel-level correctness. + +Use it when implementing or changing: + +- Post-process effects (CRT, bloom, or any new effect in the effect chain) +- Sprite rendering, tinting, or blending +- Bitmap font rendering +- Primitive drawing (pixels, lines, rects) +- Palette-indexed rendering +- Camera offsets + +Run `pnpm test:visual:update` to regenerate baselines after an intentional visual change. Snapshots live in +`tests/visual/__snapshots__/`. + +The suite covers: camera, fonts, mixed (primitives + sprites), post-process (baseline/CRT/CRT+bloom), primitives, and +sprites. **WebGPU mocks:** Use `src/__test__/webgpu-mock.ts` for tests needing GPUDevice, GPUTexture, etc. See [docs/testing.md](docs/testing.md) for full details. @@ -131,35 +148,26 @@ pnpm test:perf # Run Tier 2 browser/GPU frame-time benchmarks Use the benchmark system when the user asks about performance, throughput, regressions, hot paths, or CI benchmark coverage. -- Prefer **Tier 1 CPU benchmarks** first for isolated methods, helpers, caches, and allocation patterns -- Use **Tier 2 GPU perf tests** when the question is real browser frame time rather than raw method throughput -- If a rendering change affects both inner-loop CPU work and full-frame behavior, add both -- Tier 2 uses Playwright-managed pinned Chromium, not floating branded Chrome -- Tier 2 CI currently runs on standard hosted runners, so its results are approximate and should be treated as a coarse - regression signal unless a self-hosted GPU runner is configured +- Use **CPU benchmarks** for isolated methods, helpers, caches, and allocation patterns +- For rendering correctness, use visual regression tests (`pnpm test:visual`) — they produce PNG snapshots Recommended commands: ```bash pnpm bench pnpm bench:json -pnpm test:perf ``` CI status: -- Tier 1 CPU benchmarks run in GitHub Actions on `main` pushes and on PRs labeled `perf-tier-1` +- CPU benchmarks run in GitHub Actions on `main` pushes and on PRs labeled `perf` - Labeled PR benchmark runs compare against the latest `main` baseline artifact - The benchmark job comments on the PR and fails on regressions greater than 10% -- Tier 2 GPU perf tests run in GitHub Actions on `main` pushes and on PRs labeled `perf-tier-2` -- Tier 2 PR perf runs compare against the latest `main` GPU perf baseline artifact -- The GPU perf job comments on the PR and fails on median frame-time regressions greater than 50% -- Tier 2 CI is browser/frame-time coverage on hosted Linux runners, not hardware-accurate GPU benchmarking Claude Code reusable skill: - Use `.claude/skills/perf/SKILL.md` for benchmark-related work -- Use it when adding a new `*.bench.ts`, extending browser perf fixtures, or reasoning about benchmark CI behavior +- Use it when adding a new `*.bench.ts` or reasoning about benchmark CI behavior ## Git diff --git a/README.md b/README.md index d1dc494..94a5dd7 100644 --- a/README.md +++ b/README.md @@ -97,7 +97,6 @@ Additional documentation is available in the `docs/` directory: | `pnpm test:visual:coverage` | Run visual tests with Istanbul coverage report | | `pnpm bench` | Run Tier 1 CPU benchmarks (Vitest bench) | | `pnpm bench:json` | Run Tier 1 benchmarks and write `benchmark-results.json` | -| `pnpm test:perf` | Run Tier 2 browser/GPU frame-time benchmarks (Playwright) | | `pnpm preflight` | Run all quality checks (format, lint, typecheck, spellcheck, knip, test) | | `pnpm knip` | Find unused exports and dependencies | | `pnpm knip:fix` | Auto-fix unused exports and dependencies | diff --git a/docs/performance-testing.md b/docs/performance-testing.md index 1a6379a..4b9c3a0 100644 --- a/docs/performance-testing.md +++ b/docs/performance-testing.md @@ -1,17 +1,13 @@ # Performance Testing -Blit-Tech has two performance testing layers: fast CPU micro-benchmarks for hot methods and browser-based frame-time -benchmarks for rendered workloads. This guide explains when to use each one, how to add a new benchmark, and how CI uses -the results. +Blit-Tech has CPU micro-benchmarks for hot methods. This guide explains when to use them, how to add a new benchmark, +and how CI uses the results. ## Table of Contents - [Overview](#overview) -- [Tier 1: CPU Benchmarks](#tier-1-cpu-benchmarks) -- [Tier 2: GPU Frame-Time Benchmarks](#tier-2-gpu-frame-time-benchmarks) -- [Choosing the Right Benchmark Type](#choosing-the-right-benchmark-type) +- [CPU Benchmarks](#cpu-benchmarks) - [Adding a New CPU Benchmark](#adding-a-new-cpu-benchmark) -- [Adding a New GPU Performance Test](#adding-a-new-gpu-performance-test) - [Commands](#commands) - [CI Benchmark Workflow](#ci-benchmark-workflow) - [Recommended Workflow for New Performance Work](#recommended-workflow-for-new-performance-work) @@ -20,12 +16,11 @@ the results. ## Overview -Blit-Tech now uses two benchmarking methods: +Blit-Tech uses Vitest bench for CPU micro-benchmarks. These measure isolated methods, hot loops, cache lookups, math +helpers, and allocation patterns. -1. **Tier 1: CPU benchmarks** with Vitest bench. -2. **Tier 2: browser frame-time performance tests** with Playwright and Chromium WebGPU. - -These serve different purposes. +For visual correctness (not performance), use the visual regression tests: `pnpm test:visual`. They run Playwright with +Chromium + WebGPU and produce PNG snapshots. See `docs/testing.md` for details. ### CPU Benchmarks @@ -39,22 +34,11 @@ Examples: - `BitmapFont.measureText()` cold vs warm cache - `Rect2i.containsXY()` vs `Rect2i.contains()` -### GPU Performance Tests - -Use GPU performance tests to evaluate whole-frame rendering cost in the browser. - -Examples: - -- 500 sprites with one texture vs alternating textures -- many filled rects or diagonal lines -- bitmap text rendering throughput -- mixed scenes with primitives and sprites together - --- -## Tier 1: CPU Benchmarks +## CPU Benchmarks -Tier 1 benchmarks are implemented with Vitest bench and colocated next to the source as `*.bench.ts` files. +CPU benchmarks are implemented with Vitest bench and colocated next to the source as `*.bench.ts` files. Current benchmark files: @@ -64,7 +48,7 @@ Current benchmark files: - `src/assets/BitmapFont.bench.ts` - `src/assets/PaletteEffect.bench.ts` -### Tier 1 Metrics +### Metrics CPU benchmarks report **ops/sec**. Higher numbers are better. @@ -85,102 +69,17 @@ CPU benchmarks are: - faster to run locally - easier to write - easier to reason about -- already integrated into CI regression checks when the PR is labeled `perf-tier-1` +- already integrated into CI regression checks when the PR is labeled `perf` If you add a new hot method and want immediate automated regression protection, this is the first tool to use. --- -## Tier 2: GPU Frame-Time Benchmarks - -Tier 2 performance tests are implemented with Playwright and browser fixtures under `tests/perf/` and -`tests/visual/fixtures/`. - -Current entrypoints: - -- `tests/perf/perf.spec.ts` -- `tests/visual/fixtures/perf-primitives.html` -- `tests/visual/fixtures/perf-sprites.html` -- `tests/visual/fixtures/perf-fonts.html` -- `tests/visual/fixtures/perf-mixed.html` - -### Tier 2 Metrics - -GPU performance tests measure **frame time**, not ops/sec. - -The fixture renders a workload for a fixed number of frames and records frame durations with `performance.now()`. -Playwright collects the raw frame times and computes: - -- `median` -- `p95` -- `p99` - -Lower frame times are better. - -### Browser Runtime - -Tier 2 uses the Playwright-managed **pinned Chromium** build that matches the Playwright version in this repository. -That keeps browser revisions stable across local runs and CI, which reduces baseline drift compared with floating -branded Chrome installs. - -### Why This Exists - -Some changes look cheap in isolation but are expensive in a real frame. For example: - -- texture batch switching -- sprite submission overhead -- rendering many lines or quads in one frame -- interaction between primitives and sprites - -That is what Tier 2 is for. - -### Current CI Status - -Tier 2 can be run locally with `pnpm test:perf` and is now wired into CI as a **label-gated PR workflow**. CI uses a -much looser threshold for GPU perf than Tier 1 because browser/GPU variance is higher on shared runners. - -Today, that CI job still runs on standard GitHub-hosted Linux runners, not on dedicated GPU hardware. Treat Tier 2 CI as -an **approximate browser perf smoke signal**, not as a hardware-accurate GPU benchmark. If a self-hosted GPU runner is -added later, the same workflow can be pointed at it and re-baselined. - ---- - -## Choosing the Right Benchmark Type - -Use this rule of thumb: - -### Use Tier 1 CPU Benchmarks When - -- the code runs in Node without a browser -- you are measuring a specific method or helper -- you want to compare allocating vs in-place variants -- you want CI to catch regressions now - -### Use Tier 2 GPU Performance Tests When - -- the code depends on browser rendering -- the important question is frame time, not raw method throughput -- the code affects batching, submission, or whole-frame draw cost - -If you need hardware-true GPU numbers, a controlled local machine or a future self-hosted GPU runner is still a better -source of truth than current hosted CI. - -### Use Both When - -You changed something in the render path and want both: - -- a micro-benchmark for the hot method itself -- a browser performance test for actual frame impact - -This is often the best approach for rendering code. - ---- - ## Adding a New CPU Benchmark -If you add a new sprite-related method and it can run without a browser, start here. +If you add a new method and it can run without a browser, start here. -### CPU Benchmark File Location +### File Location Create or extend a `*.bench.ts` file near the code being measured. @@ -242,62 +141,23 @@ pnpm bench:json --- -## Adding a New GPU Performance Test - -Use this when the change is meaningful only in a browser render loop. - -### GPU Benchmark File Location - -Add or extend: - -- a fixture page in `tests/visual/fixtures/` -- a scenario in `tests/perf/perf.spec.ts` - -### Typical Pattern - -1. Create a parameterized fixture that renders the workload. -2. Run a fixed number of frames. -3. Record frame times with `performance.now()`. -4. Return the raw frame times to Playwright. -5. Let the Playwright spec compute `median`, `p95`, and `p99`. - -### Good GPU Performance Cases - -- many sprites using one texture -- many sprites alternating textures -- many bitmap text characters -- mixed primitive + sprite workloads -- stress cases that intentionally push batching or fill rate - -### Run GPU Benchmarks - -```bash -pnpm test:perf -``` - -The results are written to `test-results/perf/perf-results.json`. - ---- - ## Commands ```bash -pnpm bench # Run all Tier 1 CPU benchmarks -pnpm bench:json # Run Tier 1 benchmarks and write benchmark-results.json -pnpm test:perf # Run Tier 2 browser/GPU performance tests +pnpm bench # Run all CPU benchmarks +pnpm bench:json # Run benchmarks and write benchmark-results.json ``` ### Which Command Should I Use? - **New hot method:** `pnpm bench` - **Need machine-readable result for comparison:** `pnpm bench:json` -- **Whole browser frame cost:** `pnpm test:perf` --- ## CI Benchmark Workflow -Tier 1 CPU benchmark regression detection is now wired into GitHub Actions. +CPU benchmark regression detection is wired into GitHub Actions. ### What Happens on `main` @@ -311,7 +171,7 @@ That artifact becomes the reference point for future pull requests. ### What Happens on a Pull Request -On PRs targeting `main` with the `perf-tier-1` label, CI: +On PRs targeting `main` with the `perf` label, CI: 1. runs the normal `quality` job first 2. runs the `benchmark` job @@ -321,31 +181,7 @@ On PRs targeting `main` with the `perf-tier-1` label, CI: 6. posts or updates a PR comment with a benchmark comparison table 7. fails the job if any benchmark is more than **10% slower** -PRs without the `perf-tier-1` label skip the Tier 1 benchmark job to reduce CI cost. - -### Tier 2 GPU Perf Workflow - -Tier 2 GPU perf uses the same baseline-artifact model, but it is gated separately and uses a much looser threshold. - -On pushes to `main`, CI: - -1. runs `pnpm test:perf` -2. produces `test-results/perf/perf-results.json` -3. uploads that file as the latest GPU perf baseline artifact - -On PRs targeting `main` with the `perf-tier-2` label, CI: - -1. runs `pnpm test:perf` -2. downloads the latest successful `main` GPU perf baseline artifact -3. compares current frame-time stats against the baseline -4. posts or updates a PR comment with the GPU perf comparison table -5. fails the job if any scenario has a **median frame time** regression greater than **50%** - -PRs without the `perf-tier-2` label skip the Tier 2 GPU perf job to reduce CI cost. - -Tier 2 CI currently uses Playwright-managed pinned Chromium on standard GitHub-hosted Linux runners. That keeps browser -revisions stable, but the hardware is still shared and not GPU-dedicated. Use these CI results as approximate browser -perf coverage rather than hardware-accurate GPU benchmarking. +PRs without the `perf` label skip the benchmark job to reduce CI cost. ### What the PR Comment Contains @@ -371,20 +207,15 @@ The benchmark CI job fails if: If you add a new sprite operation, follow this order: 1. **Write the code clearly first.** -2. **Add a Tier 1 CPU benchmark** if the method can run in Node. +2. **Add a CPU benchmark** if the method can run in Node. 3. **Run `pnpm bench` locally** to compare the new method against the old behavior or an alternative implementation. 4. **Run `pnpm bench:json`** if you want to inspect the machine-readable output used by CI. -5. **Add a Tier 2 GPU test** if the change affects real render throughput or frame time. -6. **Open a PR** and add: - - `perf-tier-1` for Tier 1 CPU benchmark comparison - - `perf-tier-2` for Tier 2 GPU perf comparison +5. **Open a PR** and add the `perf` label for CPU benchmark comparison. ### Best Default -If you are unsure which path to choose: - -- start with **Tier 1 CPU benchmarking** +If you are unsure where to start: -It is simpler, faster, and already supported by the label-gated benchmark CI. +- use **CPU benchmarking** with Vitest bench -Add Tier 2 only when the real question is about frame rendering cost in the browser. +It is simple, fast, and already supported by the label-gated benchmark CI. diff --git a/eslint.config.js b/eslint.config.js index 1369466..bdffacb 100644 --- a/eslint.config.js +++ b/eslint.config.js @@ -168,7 +168,7 @@ export default [ // Test files - relaxed rules { - files: ['**/*.test.ts', 'src/__test__/**/*.ts', 'tests/perf/**/*.ts', 'tests/visual/**/*.ts'], + files: ['**/*.test.ts', 'src/__test__/**/*.ts', 'tests/visual/**/*.ts'], languageOptions: { globals: { ...globals.browser, diff --git a/package.json b/package.json index 97f51ba..99387fd 100644 --- a/package.json +++ b/package.json @@ -59,7 +59,6 @@ "bench": "vitest bench", "bench:json": "vitest bench --outputJson benchmark-results.json", "test:visual": "playwright test", - "test:perf": "playwright test -c playwright.perf.config.ts", "test:visual:update": "playwright test --update-snapshots", "test:visual:coverage": "VISUAL_COVERAGE=1 playwright test && nyc report --reporter=lcov --reporter=text-summary --temp-dir=.nyc_output --report-dir=coverage-visual", "preflight": "pnpm format:check && pnpm lint && pnpm typecheck && pnpm spellcheck && pnpm knip && pnpm test:unit", diff --git a/playwright.perf.config.ts b/playwright.perf.config.ts deleted file mode 100644 index 6921e92..0000000 --- a/playwright.perf.config.ts +++ /dev/null @@ -1,41 +0,0 @@ -import { defineConfig, devices } from '@playwright/test'; - -export default defineConfig({ - testDir: 'tests/perf', - outputDir: 'test-results/perf', - timeout: 120_000, - - fullyParallel: false, - forbidOnly: !!process.env.CI, - retries: process.env.CI ? 1 : 0, - workers: 1, - - reporter: process.env.CI ? 'github' : 'list', - - use: { - baseURL: 'http://127.0.0.1:5175', - screenshot: 'only-on-failure', - trace: 'on-first-retry', - }, - - projects: [ - { - name: 'chromium-webgpu', - use: { - ...devices['Desktop Chrome'], - browserName: 'chromium', - launchOptions: { - args: ['--enable-unsafe-webgpu', '--enable-features=Vulkan', '--disable-gpu-sandbox'], - }, - }, - }, - ], - - webServer: { - command: - 'pnpm exec vite serve tests/visual/fixtures --config tests/visual/fixtures/vite.config.ts --host 127.0.0.1 --port 5175', - port: 5175, - reuseExistingServer: !process.env.CI, - timeout: 30_000, - }, -}); diff --git a/scripts/compare-tier-2-perf-results.mjs b/scripts/compare-tier-2-perf-results.mjs deleted file mode 100644 index aed48b7..0000000 --- a/scripts/compare-tier-2-perf-results.mjs +++ /dev/null @@ -1,556 +0,0 @@ -import fs from 'node:fs'; -import path from 'node:path'; - -const COMMENT_MARKER = ''; -const DEFAULT_THRESHOLD = 50; - -/** - * Parses CLI arguments for the perf comparison command. - * - * @param {string[]} argv Raw CLI arguments after the node/script prefix. - * @returns {{ - * baseline: string | null, - * current: string | null, - * jsonOut: string, - * markdownOut: string, - * threshold: number, - * }} Parsed command options. - */ -function parseArgs(argv) { - const args = { - baseline: null, - current: null, - jsonOut: 'perf-comparison.json', - markdownOut: 'perf-comment.md', - threshold: DEFAULT_THRESHOLD, - }; - - for (let index = 0; index < argv.length; index += 1) { - // eslint-disable-next-line security/detect-object-injection -- CLI flags are parsed from a fixed argv array shape. - const value = argv[index]; - const nextValue = argv[index + 1]; - - if (!value.startsWith('--')) { - throw new Error(`Unexpected positional token: ${value}`); - } - - if (nextValue === undefined || nextValue.startsWith('--') || nextValue.startsWith('-')) { - throw new Error(`Missing or invalid value for argument: ${value}`); - } - - switch (value) { - case '--baseline': - args.baseline = nextValue; - index += 1; - break; - case '--current': - args.current = nextValue; - index += 1; - break; - case '--json-out': - args.jsonOut = nextValue; - index += 1; - break; - case '--markdown-out': - args.markdownOut = nextValue; - index += 1; - break; - case '--threshold': - args.threshold = Number(nextValue); - index += 1; - break; - default: - throw new Error(`Unknown argument: ${value}`); - } - } - - if (!args.current) { - throw new Error('The --current argument is required'); - } - - if (!Number.isFinite(args.threshold) || args.threshold < 0) { - throw new Error(`Invalid --threshold value: ${String(args.threshold)}`); - } - - return args; -} - -/** - * Ensures that the parent directory for an output file exists. - * - * @param {string} filePath Output file path. - * @returns {void} - */ -function ensureParentDirectory(filePath) { - fs.mkdirSync(path.dirname(filePath), { recursive: true }); -} - -/** - * Reads and parses a JSON file from disk. - * - * @param {string} filePath JSON file path. - * @returns {unknown} Parsed JSON value. - */ -function readJsonFile(filePath) { - return JSON.parse(fs.readFileSync(filePath, 'utf8')); -} - -/** - * Throws when a required condition is not met. - * - * @param {unknown} condition Condition to evaluate. - * @param {string} message Error message for failed assertions. - * @returns {void} - */ -function assert(condition, message) { - if (!condition) { - throw new Error(message); - } -} - -/** - * Validates the top-level shape of a perf benchmark report. - * - * @param {unknown} report Parsed perf report JSON. - * @param {string} sourceLabel Human-readable source label for error messages. - * @returns {void} - */ -function validatePerfReport(report, sourceLabel) { - assert(report !== null && typeof report === 'object', `Invalid perf report in ${sourceLabel}: expected object`); - assert(Array.isArray(report.scenarios), `Invalid perf report in ${sourceLabel}: report.scenarios must be an array`); -} - -/** - * Flattens the perf report into scenario entries suitable for comparison. - * - * @param {{ scenarios: unknown[], __sourceLabel?: string }} report Parsed perf report. - * @returns {Array<{ - * fixture: string, - * label: string, - * matchKey: string, - * name: string, - * stats: { - * frames: number, - * max: number, - * median: number, - * min: number, - * p95: number, - * p99: number, - * }, - * }>} Flattened perf scenario entries. - */ -function flattenPerfScenarios(report) { - const entries = []; - const reportLabel = report.__sourceLabel ?? 'perf report'; - const seenMatchKeys = new Set(); - - validatePerfReport(report, reportLabel); - - for (const [scenarioIndex, scenario] of report.scenarios.entries()) { - assert( - scenario !== null && typeof scenario === 'object', - `Invalid scenario entry at index ${scenarioIndex} in ${reportLabel}`, - ); - assert( - typeof scenario.fixture === 'string', - `Invalid scenario.fixture at index ${scenarioIndex} in ${reportLabel}`, - ); - assert(typeof scenario.name === 'string', `Invalid scenario.name at index ${scenarioIndex} in ${reportLabel}`); - assert( - scenario.stats !== null && typeof scenario.stats === 'object', - `Invalid scenario.stats for ${scenario.name} in ${reportLabel}`, - ); - assert( - Number.isFinite(scenario.stats.frames) && - Number.isInteger(scenario.stats.frames) && - scenario.stats.frames >= 0, - `Invalid stats.frames for ${scenario.name} in ${reportLabel}`, - ); - assert( - Number.isFinite(scenario.stats.median) && scenario.stats.median >= 0, - `Invalid stats.median for ${scenario.name} in ${reportLabel}`, - ); - assert( - Number.isFinite(scenario.stats.p95) && scenario.stats.p95 >= 0, - `Invalid stats.p95 for ${scenario.name} in ${reportLabel}`, - ); - assert( - Number.isFinite(scenario.stats.p99) && scenario.stats.p99 >= 0, - `Invalid stats.p99 for ${scenario.name} in ${reportLabel}`, - ); - assert( - Number.isFinite(scenario.stats.min) && scenario.stats.min >= 0, - `Invalid stats.min for ${scenario.name} in ${reportLabel}`, - ); - assert( - Number.isFinite(scenario.stats.max) && scenario.stats.max >= 0, - `Invalid stats.max for ${scenario.name} in ${reportLabel}`, - ); - - const matchKey = `${scenario.fixture}::${scenario.name}`; - assert( - !seenMatchKeys.has(matchKey), - `Duplicate perf scenario matchKey ${matchKey} encountered in ${reportLabel}`, - ); - seenMatchKeys.add(matchKey); - - entries.push({ - fixture: scenario.fixture, - label: scenario.name, - matchKey, - name: scenario.name, - stats: scenario.stats, - }); - } - - return entries; -} - -/** - * Calculates the percentage change for a frame-time metric. - * - * @param {number} baselineValue Baseline frame-time value. - * @param {number} currentValue Current frame-time value. - * @returns {number | string} Percentage change where positive means slower. - */ -function calculateRegressionPct(baselineValue, currentValue) { - if (baselineValue === 0) { - if (currentValue === 0) { - return 0; - } - - return currentValue > 0 ? 'zero-baseline-positive' : 'zero-baseline-negative'; - } - - return ((currentValue - baselineValue) / baselineValue) * 100; -} - -/** - * Checks whether a perf delta should count as a regression. - * - * @param {number | string | null} value Perf delta value. - * @param {number} thresholdPct Maximum allowed slowdown percentage before failure. - * @returns {boolean} True when the delta exceeds the allowed regression threshold. - */ -function isRegressionDelta(value, thresholdPct) { - if (value === 'zero-baseline-positive') { - return true; - } - - if (typeof value === 'number') { - return value > thresholdPct; - } - - return false; -} - -/** - * Checks whether a perf delta should count as an improvement. - * - * @param {number | string | null} value Perf delta value. - * @returns {boolean} True when the delta indicates an improvement versus baseline. - */ -function isImprovementDelta(value) { - if (value === 'zero-baseline-negative') { - return true; - } - - if (typeof value === 'number') { - return value < 0; - } - - return false; -} - -/** - * Compares current perf results against an optional baseline report. - * - * @param {{ scenarios: unknown[], __sourceLabel?: string }} currentReport Current perf report. - * @param {{ scenarios: unknown[], __sourceLabel?: string } | null} baselineReport Baseline perf report, if available. - * @param {number} thresholdPct Maximum allowed slowdown percentage before failure. - * @returns {{ - * generatedAt: string, - * hasBaseline: boolean, - * thresholdPct: number, - * summary: { - * total: number, - * compared: number, - * regressions: number, - * improvements: number, - * pass: number, - * newScenarios: number, - * missingScenarios: number, - * }, - * scenarios: Array<{ - * fixture: string, - * name: string, - * baselineStats: { - * frames: number, - * max: number, - * median: number, - * min: number, - * p95: number, - * p99: number, - * } | null, - * currentStats: { - * frames: number, - * max: number, - * median: number, - * min: number, - * p95: number, - * p99: number, - * } | null, - * deltas: { - * median: number | string | null, - * p95: number | string | null, - * p99: number | string | null, - * }, - * status: string, - * }>, - * }} Comparison report used by CI and PR comments. - */ -function comparePerfReports(currentReport, baselineReport, thresholdPct) { - const currentEntries = flattenPerfScenarios(currentReport); - const baselineEntries = baselineReport ? flattenPerfScenarios(baselineReport) : []; - const currentByKey = new Map(currentEntries.map((entry) => [entry.matchKey, entry])); - const baselineByKey = new Map(baselineEntries.map((entry) => [entry.matchKey, entry])); - const keys = [...new Set([...baselineByKey.keys(), ...currentByKey.keys()])].sort((left, right) => - left.localeCompare(right), - ); - - const scenarios = keys.map((key) => { - const baselineEntry = baselineByKey.get(key) ?? null; - const currentEntry = currentByKey.get(key) ?? null; - - if (!baselineEntry) { - return { - fixture: currentEntry?.fixture ?? 'unknown', - name: currentEntry?.label ?? key, - baselineStats: null, - currentStats: currentEntry?.stats ?? null, - deltas: { median: null, p95: null, p99: null }, - status: 'new', - }; - } - - if (!currentEntry) { - return { - fixture: baselineEntry.fixture, - name: baselineEntry.label, - baselineStats: baselineEntry.stats, - currentStats: null, - deltas: { median: null, p95: null, p99: null }, - status: 'missing', - }; - } - - const deltas = { - median: calculateRegressionPct(baselineEntry.stats.median, currentEntry.stats.median), - p95: calculateRegressionPct(baselineEntry.stats.p95, currentEntry.stats.p95), - p99: calculateRegressionPct(baselineEntry.stats.p99, currentEntry.stats.p99), - }; - let status = 'pass'; - - if (isRegressionDelta(deltas.median, thresholdPct)) { - status = 'fail'; - } else if (isImprovementDelta(deltas.median)) { - status = 'improved'; - } - - return { - fixture: currentEntry.fixture, - name: currentEntry.label, - baselineStats: baselineEntry.stats, - currentStats: currentEntry.stats, - deltas, - status, - }; - }); - - const summary = { - total: scenarios.length, - compared: scenarios.filter((scenario) => scenario.deltas.median !== null).length, - regressions: scenarios.filter((scenario) => scenario.status === 'fail').length, - improvements: scenarios.filter((scenario) => scenario.status === 'improved').length, - pass: scenarios.filter((scenario) => scenario.status === 'pass').length, - newScenarios: scenarios.filter((scenario) => scenario.status === 'new').length, - missingScenarios: scenarios.filter((scenario) => scenario.status === 'missing').length, - }; - - return { - generatedAt: new Date().toISOString(), - hasBaseline: baselineReport !== null, - thresholdPct, - summary, - scenarios, - }; -} - -/** - * Formats a frame-time value for markdown output. - * - * @param {number | null} value Frame-time value in milliseconds. - * @returns {string} Formatted frame-time string. - */ -function formatFrameTime(value) { - if (value === null) { - return 'n/a'; - } - - return `${value.toFixed(2)}ms`; -} - -/** - * Formats a regression delta percentage for markdown output. - * - * @param {number | string | null} value Percentage delta where positive means slower. - * @returns {string} Formatted delta string. - */ -function formatDelta(value) { - if (value === null) { - return 'n/a'; - } - - if (value === 'zero-baseline-positive') { - return 'zero-baseline+'; - } - - if (value === 'zero-baseline-negative') { - return 'zero-baseline-'; - } - - const sign = value > 0 ? '+' : ''; - - return `${sign}${value.toFixed(2)}%`; -} - -/** - * Converts an internal perf scenario status into the PR comment label. - * - * @param {string} status Internal scenario status. - * @returns {string} Human-readable status label. - */ -function formatStatus(status) { - switch (status) { - case 'fail': - return 'FAIL'; - case 'improved': - return 'IMPROVED'; - case 'new': - return 'NEW'; - case 'missing': - return 'MISSING'; - default: - return 'PASS'; - } -} - -/** - * Escapes a markdown table cell value. - * - * @param {string} value Raw markdown cell content. - * @returns {string} Escaped markdown-safe value. - */ -function escapeMarkdownCell(value) { - return value.replaceAll('|', '\\|'); -} - -/** - * Builds the markdown body for the perf PR comment. - * - * @param {{ - * hasBaseline: boolean, - * thresholdPct: number, - * summary: { - * compared: number, - * regressions: number, - * improvements: number, - * newScenarios: number, - * missingScenarios: number, - * }, - * scenarios: Array<{ - * fixture: string, - * name: string, - * baselineStats: { median: number } | null, - * currentStats: { median: number } | null, - * deltas: { median: number | string | null, p95: number | string | null, p99: number | string | null }, - * status: string, - * }>, - * }} report Comparison report. - * @returns {string} Markdown comment body. - */ -function buildMarkdown(report) { - const lines = [COMMENT_MARKER, '## Tier 2 GPU Perf Comparison', '']; - - if (!report.hasBaseline) { - lines.push( - 'No `main` branch GPU perf baseline artifact is available yet. This run produced fresh perf results and uploaded them as artifacts.', - '', - `Configured regression threshold: ${report.thresholdPct}% slower median frame time.`, - ); - - return `${lines.join('\n')}\n`; - } - - lines.push( - `Compared ${report.summary.compared} perf scenarios against the latest \`main\` baseline. Regression threshold: ${report.thresholdPct}% slower median frame time.`, - '', - `Regressions: ${report.summary.regressions} | Improvements: ${report.summary.improvements} | New: ${report.summary.newScenarios} | Missing: ${report.summary.missingScenarios}`, - '', - '
', - 'Perf table', - '', - '| Fixture | Scenario | Baseline median | Current median | Median delta | P95 delta | P99 delta | Status |', - '| --- | --- | ---: | ---: | ---: | ---: | ---: | --- |', - ); - - for (const scenario of report.scenarios) { - lines.push( - `| ${escapeMarkdownCell(scenario.fixture)} | ${escapeMarkdownCell(scenario.name)} | ${formatFrameTime(scenario.baselineStats?.median ?? null)} | ${formatFrameTime(scenario.currentStats?.median ?? null)} | ${formatDelta(scenario.deltas.median)} | ${formatDelta(scenario.deltas.p95)} | ${formatDelta(scenario.deltas.p99)} | ${formatStatus(scenario.status)} |`, - ); - } - - lines.push('', '
'); - - return `${lines.join('\n')}\n`; -} - -/** - * Writes a file to disk, creating parent directories when needed. - * - * @param {string} filePath Output file path. - * @param {string} content File contents. - * @returns {void} - */ -function writeFile(filePath, content) { - ensureParentDirectory(filePath); - fs.writeFileSync(filePath, content); -} - -/** - * Runs the perf comparison CLI. - * - * @returns {void} - */ -function main() { - const args = parseArgs(process.argv.slice(2)); - const currentReport = readJsonFile(args.current); - currentReport.__sourceLabel = args.current; - let baselineReport = null; - - if (args.baseline) { - assert(fs.existsSync(args.baseline), `Missing baseline perf report: ${args.baseline}`); - baselineReport = readJsonFile(args.baseline); - } - - if (baselineReport !== null) { - baselineReport.__sourceLabel = args.baseline; - } - - const report = comparePerfReports(currentReport, baselineReport, args.threshold); - - writeFile(args.jsonOut, JSON.stringify(report, null, 2)); - writeFile(args.markdownOut, buildMarkdown(report)); -} - -main(); diff --git a/tests/perf/perf.spec.ts b/tests/perf/perf.spec.ts deleted file mode 100644 index 2bc9f62..0000000 --- a/tests/perf/perf.spec.ts +++ /dev/null @@ -1,214 +0,0 @@ -import fs from 'node:fs'; -import path from 'node:path'; - -import { expect, test } from '@playwright/test'; - -// #region Types - -interface RawPerfResult { - frameTimes: number[]; - frames: number; - scenario: string; - warmupFrames: number; - workload: Record; -} - -interface PerfStats { - frames: number; - max: number; - median: number; - min: number; - p95: number; - p99: number; -} - -interface PerfScenarioResult { - fixture: string; - name: string; - stats: PerfStats; - workload: Record; -} - -interface PerfScenario { - fixture: string; - name: string; - query: Record; -} - -interface PerfWindow { - __INIT_FAILED__?: boolean; - __PERF_COMPLETE__?: boolean; - __PERF_RESULT__?: RawPerfResult | null; -} - -// #endregion - -// #region Constants - -const PERF_RESULTS_FILE = path.resolve(process.cwd(), 'test-results/perf/perf-results.json'); - -const perfResults: PerfScenarioResult[] = []; - -const perfScenarios: PerfScenario[] = [ - { - fixture: 'perf-primitives.html', - name: 'primitives', - query: { frames: 100, lines: 80, pixels: 400, rects: 120, warmup: 10 }, - }, - { - fixture: 'perf-sprites.html', - name: 'sprites-same-texture', - query: { frames: 100, sprites: 800, textureMode: 'same', warmup: 10 }, - }, - { - fixture: 'perf-sprites.html', - name: 'sprites-alternating-textures', - query: { frames: 100, sprites: 800, textureMode: 'alternating', warmup: 10 }, - }, - { - fixture: 'perf-fonts.html', - name: 'fonts', - query: { chars: 480, frames: 100, lineWidth: 32, warmup: 10 }, - }, - { - fixture: 'perf-mixed.html', - name: 'mixed', - query: { chars: 160, frames: 100, lines: 100, rects: 140, sprites: 220, warmup: 10 }, - }, -]; - -// #endregion - -// #region Helpers - -/** - * Builds a fixture URL with query parameters for a perf scenario. - * - * @param fixture - Fixture HTML filename. - * @param query - Query parameters passed to the page. - * @returns Relative URL for Playwright navigation. - */ -function buildFixtureUrl(fixture: string, query: Record): string { - const params = new URLSearchParams(); - - for (const [key, value] of Object.entries(query)) { - params.set(key, String(value)); - } - - return `/${fixture}?${params.toString()}`; -} - -/** - * Computes summary statistics from a list of frame times. - * - * @param frameTimes - Recorded frame durations in milliseconds. - * @returns Median, tail percentiles, and min/max values. - */ -function computeStats(frameTimes: number[]): PerfStats { - const sorted = [...frameTimes].sort((left, right) => left - right); - - return { - frames: sorted.length, - max: sorted[sorted.length - 1] ?? 0, - median: percentile(sorted, 0.5), - min: sorted[0] ?? 0, - p95: percentile(sorted, 0.95), - p99: percentile(sorted, 0.99), - }; -} - -/** - * Calculates a percentile using nearest-rank selection. - * - * @param sorted - Frame times sorted ascending. - * @param percentileValue - Percentile expressed as 0.0-1.0. - * @returns Value at the requested percentile. - */ -function percentile(sorted: number[], percentileValue: number): number { - if (sorted.length === 0) { - return 0; - } - - const index = Math.min(sorted.length - 1, Math.ceil(sorted.length * percentileValue) - 1); - - return sorted.at(index) ?? 0; -} - -/** - * Writes the aggregated perf results to a machine-readable JSON file. - */ -function writePerfResultsFile(): void { - fs.mkdirSync(path.dirname(PERF_RESULTS_FILE), { recursive: true }); - fs.writeFileSync( - PERF_RESULTS_FILE, - JSON.stringify( - { - generatedAt: new Date().toISOString(), - scenarios: perfResults, - }, - null, - 2, - ), - ); -} - -// #endregion - -// #region Tests - -test.afterAll(() => { - writePerfResultsFile(); -}); - -for (const scenario of perfScenarios) { - test(`collects frame-time stats for ${scenario.name}`, async ({ page }, testInfo) => { - await page.goto(buildFixtureUrl(scenario.fixture, scenario.query)); - - await page.waitForFunction( - () => { - const perfWindow = window as unknown as PerfWindow; - - return perfWindow.__PERF_COMPLETE__ || perfWindow.__INIT_FAILED__; - }, - { timeout: 30_000 }, - ); - - const rawResult = await page.evaluate(() => { - const perfWindow = window as unknown as PerfWindow; - - return { - initFailed: perfWindow.__INIT_FAILED__, - result: perfWindow.__PERF_RESULT__ ?? null, - }; - }); - - test.skip(rawResult.initFailed === true, 'WebGPU not available in this environment'); - - expect(rawResult.result).not.toBeNull(); - - const result = rawResult.result as RawPerfResult; - - expect(result.frameTimes.length).toBe(result.frames); - - const stats = computeStats(result.frameTimes); - const scenarioResult: PerfScenarioResult = { - fixture: scenario.fixture, - name: scenario.name, - stats, - workload: result.workload, - }; - - perfResults.push(scenarioResult); - - await testInfo.attach('perf-stats', { - body: JSON.stringify(scenarioResult, null, 2), - contentType: 'application/json', - }); - - console.log( - `[perf] ${scenario.name}: median=${stats.median.toFixed(2)}ms p95=${stats.p95.toFixed(2)}ms p99=${stats.p99.toFixed(2)}ms`, - ); - }); -} - -// #endregion diff --git a/tests/visual/fixtures/perf-common.js b/tests/visual/fixtures/perf-common.js deleted file mode 100644 index ab6f917..0000000 --- a/tests/visual/fixtures/perf-common.js +++ /dev/null @@ -1,143 +0,0 @@ -/* global Image, document, window */ - -export function readNumberParam(name, fallback) { - const params = new URLSearchParams(window.location.search); - const raw = params.get(name); - - if (raw === null) { - return fallback; - } - - const parsed = Number(raw); - - return Number.isFinite(parsed) ? parsed : fallback; -} - -export function readStringParam(name, fallback) { - const params = new URLSearchParams(window.location.search); - - return params.get(name) ?? fallback; -} - -export function initializePerfState(scenario, workload) { - window.__PERF_COMPLETE__ = false; - window.__PERF_RESULT__ = null; - window.__INIT_FAILED__ = false; - - return { - completed: false, - frameLimit: Math.max(1, readNumberParam('frames', 100) | 0), - frameTimes: [], - framesSeen: 0, - lastFrameAt: null, - scenario, - warmupFrames: Math.max(0, readNumberParam('warmup', 10) | 0), - workload, - }; -} - -export function recordFrame(state) { - if (state.completed) { - return true; - } - - const now = performance.now(); - - if (state.lastFrameAt !== null) { - const delta = now - state.lastFrameAt; - - if (state.framesSeen >= state.warmupFrames) { - state.frameTimes.push(delta); - } - } - - state.lastFrameAt = now; - state.framesSeen += 1; - - if (state.frameTimes.length >= state.frameLimit) { - window.__PERF_RESULT__ = { - frameTimes: state.frameTimes.slice(), - frames: state.frameLimit, - scenario: state.scenario, - warmupFrames: state.warmupFrames, - workload: state.workload, - }; - window.__PERF_COMPLETE__ = true; - state.completed = true; - } - - return state.completed; -} - -export function markInitFailed(error) { - window.__INIT_FAILED__ = true; - - if (error) { - console.error(error); - } -} - -export function repeatText(seed, length) { - return seed.repeat(Math.ceil(length / seed.length)).slice(0, length); -} - -/** - * Creates and activates a default 16-color test palette. - * Index 0 is transparent (never set — left as zero alpha). - * Indices 1-10 cover common test colors. - * - * @param {object} BT - The BT namespace. - * @param {object} Color32 - The Color32 class. - * @returns {object} The created palette. - */ -export function installDefaultTestPalette(BT, Color32) { - const palette = BT.paletteCreate(16); - - palette.set(1, Color32.black()); - palette.set(2, Color32.red()); - palette.set(3, Color32.green()); - palette.set(4, Color32.blue()); - palette.set(5, Color32.yellow()); - palette.set(6, Color32.cyan()); - palette.set(7, Color32.white()); - palette.set(8, new Color32(32, 32, 64, 255)); - palette.set(9, new Color32(16, 24, 40, 255)); - palette.set(10, new Color32(128, 0, 0, 255)); - - BT.paletteSet(palette); - - return palette; -} - -export function createSpriteImage(primaryColor, secondaryColor) { - const canvas = document.createElement('canvas'); - canvas.width = 8; - canvas.height = 8; - const context = canvas.getContext('2d'); - - if (!context) { - markInitFailed(new Error('Failed to create 2D canvas context for perf sprite fixture')); - - return Promise.reject(new Error('Failed to create 2D canvas context for perf sprite fixture')); - } - - for (let y = 0; y < 8; y++) { - for (let x = 0; x < 8; x++) { - context.fillStyle = (x + y) % 2 === 0 ? primaryColor : secondaryColor; - context.fillRect(x, y, 1, 1); - } - } - - const image = new Image(); - - return new Promise((resolve, reject) => { - image.onload = () => resolve(image); - image.onerror = () => { - const error = new Error('Failed to load generated perf sprite image'); - - markInitFailed(error); - reject(error); - }; - image.src = canvas.toDataURL(); - }); -} diff --git a/tests/visual/fixtures/perf-fonts.html b/tests/visual/fixtures/perf-fonts.html deleted file mode 100644 index aea1a62..0000000 --- a/tests/visual/fixtures/perf-fonts.html +++ /dev/null @@ -1,90 +0,0 @@ - - - - - - Perf Test: Fonts - - - - -
- -
- - - - - diff --git a/tests/visual/fixtures/perf-mixed.html b/tests/visual/fixtures/perf-mixed.html deleted file mode 100644 index 0c28827..0000000 --- a/tests/visual/fixtures/perf-mixed.html +++ /dev/null @@ -1,114 +0,0 @@ - - - - - - Perf Test: Mixed - - - - -
- -
- - - - - diff --git a/tests/visual/fixtures/perf-primitives.html b/tests/visual/fixtures/perf-primitives.html deleted file mode 100644 index 2e3a576..0000000 --- a/tests/visual/fixtures/perf-primitives.html +++ /dev/null @@ -1,101 +0,0 @@ - - - - - - Perf Test: Primitives - - - - -
- -
- - - - - diff --git a/tests/visual/fixtures/perf-sprites.html b/tests/visual/fixtures/perf-sprites.html deleted file mode 100644 index 4856bff..0000000 --- a/tests/visual/fixtures/perf-sprites.html +++ /dev/null @@ -1,97 +0,0 @@ - - - - - - Perf Test: Sprites - - - - -
- -
- - - - - diff --git a/tsconfig.json b/tsconfig.json index 6eaf7a8..71287fb 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -36,12 +36,10 @@ }, "include": [ "src/**/*", - "tests/perf/**/*.ts", "tests/visual/**/*.ts", "vite.config.ts", "vitest.config.ts", "playwright.config.ts", - "playwright.perf.config.ts", "*.d.ts" ], "exclude": ["node_modules", "dist"]