diff --git a/.claude/skills/perf/SKILL.md b/.claude/skills/perf/SKILL.md
index 64a577b..8109b72 100644
--- a/.claude/skills/perf/SKILL.md
+++ b/.claude/skills/perf/SKILL.md
@@ -1,7 +1,5 @@
---
-description:
- Add or update Blit-Tech performance tests, choose between Vitest bench and Playwright perf, and explain the CI
- benchmark workflow.
+description: Add or update Blit-Tech CPU benchmarks and explain the CI benchmark workflow.
---
# Performance Testing
@@ -10,13 +8,12 @@ Use this skill when the task involves:
- adding or extending `*.bench.ts` files
- benchmarking a new hot method or allocation pattern
-- deciding between CPU micro-benchmarks and browser/GPU perf tests
-- working on `tests/perf/` or perf fixtures in `tests/visual/fixtures/`
-- explaining or debugging benchmark CI behavior
+- working on benchmark CI behavior
-## Benchmark Types
+For visual correctness verification (not performance), use `pnpm test:visual` — see the Visual Regression Tests section
+in CLAUDE.md.
-### Tier 1: CPU Benchmarks
+## CPU Benchmarks
Use Vitest bench for isolated hot paths that can run in Node.
@@ -42,64 +39,17 @@ Rules:
- prefer realistic hot-path inputs
- use `pnpm bench:json` when validating CI-facing output
-### Tier 2: GPU Performance Tests
-
-Use Playwright perf when the important metric is browser frame time.
-
-Examples:
-
-- sprite throughput
-- texture batch switching
-- bitmap text frame cost
-- mixed primitive + sprite workloads
-
-Commands:
-
-```bash
-pnpm test:perf
-```
-
-Rules:
-
-- add or extend a fixture in `tests/visual/fixtures/`
-- add or extend the scenario in `tests/perf/perf.spec.ts`
-- think in terms of workload shape and frame-time stats (`median`, `p95`, `p99`)
-- Tier 2 uses Playwright-managed pinned Chromium for stable baselines
-- current CI for Tier 2 runs on standard hosted runners, so treat it as approximate browser perf coverage unless a
- self-hosted GPU runner exists
-
-## Choosing the Right Tool
-
-Use Tier 1 first if the code can run without a browser.
-
-Use Tier 2 when:
-
-- the code depends on browser rendering
-- the user cares about whole-frame cost
-- batching or texture switching is part of the question
-
-Use both when a rendering change affects both an inner-loop method and actual rendered frame time.
-
## CI Behavior
-Tier 1 and Tier 2 performance checks are in CI now, with separate labels.
+CPU benchmark regression checks run in CI with the `perf` label.
- `main` pushes run `pnpm bench:json` and refresh the stored baseline artifact
-- PRs labeled `perf-tier-1` run `pnpm bench:json`
+- PRs labeled `perf` run `pnpm bench:json`
- CI compares against the latest successful `main` baseline artifact
- CI posts or updates a PR comment with the comparison table
- CI fails if any benchmark regresses by more than 10%
-- `main` pushes run `pnpm test:perf` and refresh the stored GPU perf baseline artifact
-- PRs labeled `perf-tier-2` run `pnpm test:perf`
-- CI compares GPU perf results against the latest successful `main` GPU perf baseline artifact
-- CI posts or updates a PR comment with the frame-time comparison table
-- CI fails if any scenario has a median frame-time regression greater than 50%
-- Tier 2 results are intentionally loose because the current CI environment is a standard hosted runner, not dedicated
- GPU hardware
-
## References
-- Read `docs/performance-testing.md` when the user wants an explanation or documentation update
-- Read `.github/workflows/ci.yml`, `scripts/compare-tier-1-benchmarks.mjs`, and
- `scripts/compare-tier-2-perf-results.mjs` when the task is about benchmark CI
+- Read `docs/performance-testing.md` for full documentation
+- Read `.github/workflows/ci.yml` and `scripts/compare-tier-1-benchmarks.mjs` for benchmark CI details
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index e700d80..625aaab 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -13,7 +13,7 @@ jobs:
runs-on: ubuntu-latest
if:
github.event_name != 'pull_request' || !contains(fromJSON('["labeled","unlabeled"]'), github.event.action) ||
- (github.event.action == 'labeled' && contains(fromJSON('["perf-tier-1","perf-tier-2"]'), github.event.label.name))
+ (github.event.action == 'labeled' && contains(fromJSON('["perf"]'), github.event.label.name))
steps:
- name: Checkout code
@@ -82,10 +82,10 @@ jobs:
retention-days: 7
benchmark:
- name: Tier 1 Benchmarks
+ name: Benchmarks
runs-on: ubuntu-latest
needs: quality
- if: github.event_name == 'push' || contains(github.event.pull_request.labels.*.name, 'perf-tier-1')
+ if: github.event_name == 'push' || contains(github.event.pull_request.labels.*.name, 'perf')
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-benchmark
cancel-in-progress: true
@@ -112,7 +112,7 @@ jobs:
- name: Install dependencies
run: pnpm install --frozen-lockfile
- - name: Run Tier 1 benchmarks
+ - name: Run benchmarks
run: pnpm bench:json
- name: Upload current benchmark results
@@ -272,211 +272,6 @@ jobs:
}
EOF
- perf:
- name: Tier 2 GPU Perf
- runs-on: ubuntu-latest
- needs: quality
- if: github.event_name == 'push' || contains(github.event.pull_request.labels.*.name, 'perf-tier-2')
- concurrency:
- group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }}-perf
- cancel-in-progress: true
- permissions:
- actions: read
- contents: read
- pull-requests: write
-
- steps:
- - name: Checkout code
- uses: actions/checkout@v6
-
- - name: Setup pnpm
- uses: pnpm/action-setup@v4
- with:
- version: 10.26.2
-
- - name: Setup Node.js
- uses: actions/setup-node@v6
- with:
- node-version: '22'
- cache: 'pnpm'
-
- - name: Install dependencies
- run: pnpm install --frozen-lockfile
-
- - name: Install pinned Chromium for Playwright perf tests
- run: pnpm exec playwright install --with-deps chromium
-
- - name: Run Tier 2 GPU perf tests
- run: pnpm test:perf
-
- - name: Upload current perf results
- uses: actions/upload-artifact@v7
- with:
- name: perf-results-current
- path: test-results/perf/perf-results.json
- if-no-files-found: error
- retention-days: 14
-
- - name: Upload main perf baseline
- if: github.event_name == 'push' && github.ref == 'refs/heads/main'
- uses: actions/upload-artifact@v7
- with:
- name: perf-baseline
- path: test-results/perf/perf-results.json
- if-no-files-found: error
- retention-days: 90
-
- - name: Find latest main perf baseline
- id: find-perf-baseline
- if: github.event_name == 'pull_request'
- uses: actions/github-script@v8
- with:
- script: |
- const workflowId = 'ci.yml';
- const runs = await github.paginate(github.rest.actions.listWorkflowRuns, {
- owner: context.repo.owner,
- repo: context.repo.repo,
- workflow_id: workflowId,
- branch: 'main',
- event: 'push',
- status: 'completed',
- per_page: 100,
- });
-
- for (const run of runs) {
- if (run.conclusion !== 'success') {
- continue;
- }
-
- const artifacts = await github.paginate(github.rest.actions.listWorkflowRunArtifacts, {
- owner: context.repo.owner,
- repo: context.repo.repo,
- run_id: run.id,
- per_page: 100,
- });
- const baselineArtifact = artifacts.find(
- (artifact) => artifact.name === 'perf-baseline' && artifact.expired === false,
- );
-
- if (!baselineArtifact) {
- continue;
- }
-
- core.setOutput('found', 'true');
- core.setOutput('download-url', baselineArtifact.archive_download_url);
- core.setOutput('artifact-id', String(baselineArtifact.id));
- core.setOutput('run-id', String(run.id));
- return;
- }
-
- core.setOutput('found', 'false');
-
- - name: Download main perf baseline
- if: github.event_name == 'pull_request' && steps.find-perf-baseline.outputs.found == 'true'
- env:
- GITHUB_TOKEN: ${{ github.token }}
- BASELINE_URL: ${{ steps.find-perf-baseline.outputs.download-url }}
- run: |
- mkdir -p perf-baseline-artifact
- curl --fail --location \
- --header "Authorization: Bearer ${GITHUB_TOKEN}" \
- --header "Accept: application/vnd.github+json" \
- "${BASELINE_URL}" \
- --output perf-baseline-artifact.zip
- unzip -o perf-baseline-artifact.zip -d perf-baseline-artifact
-
- - name: Compare perf results
- if: github.event_name == 'pull_request'
- run: |
- if [ "${{ steps.find-perf-baseline.outputs.found }}" = "true" ]; then
- node scripts/compare-tier-2-perf-results.mjs \
- --current test-results/perf/perf-results.json \
- --baseline perf-baseline-artifact/perf-results.json \
- --json-out perf-comparison.json \
- --markdown-out perf-comment.md \
- --threshold 50
- else
- node scripts/compare-tier-2-perf-results.mjs \
- --current test-results/perf/perf-results.json \
- --json-out perf-comparison.json \
- --markdown-out perf-comment.md \
- --threshold 50
- fi
-
- - name: Upload perf comparison artifacts
- if: github.event_name == 'pull_request'
- uses: actions/upload-artifact@v7
- with:
- name: perf-comparison
- path: |
- perf-comparison.json
- perf-comment.md
- test-results/perf/perf-results.json
- if-no-files-found: ignore
- retention-days: 14
-
- - name: Post perf PR comment
- if:
- github.event_name == 'pull_request' && github.event.pull_request.head.repo.fork == false && github.actor !=
- 'dependabot[bot]'
- uses: actions/github-script@v8
- env:
- PERF_COMMENT_PATH: perf-comment.md
- with:
- script: |
- const fs = require('node:fs');
- const marker = '';
- const body = fs.readFileSync(process.env.PERF_COMMENT_PATH, 'utf8');
- const comments = await github.paginate(github.rest.issues.listComments, {
- owner: context.repo.owner,
- repo: context.repo.repo,
- issue_number: context.issue.number,
- per_page: 100,
- });
- const existingComment = comments.find((comment) =>
- comment.user?.type === 'Bot' && comment.body?.includes(marker),
- );
-
- if (existingComment) {
- await github.rest.issues.updateComment({
- owner: context.repo.owner,
- repo: context.repo.repo,
- comment_id: existingComment.id,
- body,
- });
- return;
- }
-
- await github.rest.issues.createComment({
- owner: context.repo.owner,
- repo: context.repo.repo,
- issue_number: context.issue.number,
- body,
- });
-
- - name: Fail on perf regressions
- if: github.event_name == 'pull_request'
- run: |
- node --input-type=module <<'EOF'
- import fs from 'node:fs';
-
- if (!fs.existsSync('perf-comparison.json')) {
- process.exit(0);
- }
-
- const report = JSON.parse(fs.readFileSync('perf-comparison.json', 'utf8'));
-
- if (!report.hasBaseline) {
- process.exit(0);
- }
-
- const hasFailures = report.summary.regressions > 0 || report.summary.missingScenarios > 0;
-
- if (hasFailures) {
- process.exit(1);
- }
- EOF
-
test:
name: Run Tests
runs-on: ubuntu-latest
diff --git a/CLAUDE.md b/CLAUDE.md
index 9de3926..5cb9d58 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -107,21 +107,38 @@ pnpm test # Run all unit tests (alias for test:unit)
pnpm test:unit # Run all unit tests
pnpm test:unit:watch # Watch mode for development
pnpm test:unit:coverage # Coverage report (80% minimum threshold)
-pnpm test:visual # Playwright visual regression (requires Chrome)
+pnpm test:visual # Playwright visual regression tests (requires Chrome with WebGPU)
pnpm test:visual:update # Update visual test baselines
-pnpm bench # Run Tier 1 CPU benchmarks (Vitest bench)
-pnpm bench:json # Run Tier 1 benchmarks and write benchmark-results.json
-pnpm test:perf # Run Tier 2 browser/GPU frame-time benchmarks
+pnpm bench # Run CPU benchmarks (Vitest bench)
+pnpm bench:json # Run benchmarks and write benchmark-results.json
```
**Test tiers:**
1. **Unit tests** (Vitest, node) - Pure logic: Vector2i, Rect2i, Color32, Palette, PaletteEffect, Easing, GameLoop
2. **Integration tests** (Vitest, Node + GPU mocks; happy-dom for DOM tests) - DOM and GPU code
-3. **Visual regression** (Playwright, Chromium) - Rendering output verification
-4. **Performance tests**
- - Tier 1 CPU benchmarks (Vitest bench, `*.bench.ts`) for hot methods and allocation patterns
- - Tier 2 GPU perf tests (Playwright, `tests/perf/`) for browser frame-time workloads
+3. **Visual regression** (Playwright, Chromium + WebGPU) - PNG snapshot verification of rendered output
+4. **CPU benchmarks** (Vitest bench, `*.bench.ts`) - Hot method and allocation pattern throughput
+
+### Visual Regression Tests
+
+`pnpm test:visual` runs Playwright with Chromium + WebGPU and captures PNG snapshots of actual rendered frames. This is
+the primary tool for verifying that visual output is correct — not performance, but pixel-level correctness.
+
+Use it when implementing or changing:
+
+- Post-process effects (CRT, bloom, or any new effect in the effect chain)
+- Sprite rendering, tinting, or blending
+- Bitmap font rendering
+- Primitive drawing (pixels, lines, rects)
+- Palette-indexed rendering
+- Camera offsets
+
+Run `pnpm test:visual:update` to regenerate baselines after an intentional visual change. Snapshots live in
+`tests/visual/__snapshots__/`.
+
+The suite covers: camera, fonts, mixed (primitives + sprites), post-process (baseline/CRT/CRT+bloom), primitives, and
+sprites.
**WebGPU mocks:** Use `src/__test__/webgpu-mock.ts` for tests needing GPUDevice, GPUTexture, etc. See
[docs/testing.md](docs/testing.md) for full details.
@@ -131,35 +148,26 @@ pnpm test:perf # Run Tier 2 browser/GPU frame-time benchmarks
Use the benchmark system when the user asks about performance, throughput, regressions, hot paths, or CI benchmark
coverage.
-- Prefer **Tier 1 CPU benchmarks** first for isolated methods, helpers, caches, and allocation patterns
-- Use **Tier 2 GPU perf tests** when the question is real browser frame time rather than raw method throughput
-- If a rendering change affects both inner-loop CPU work and full-frame behavior, add both
-- Tier 2 uses Playwright-managed pinned Chromium, not floating branded Chrome
-- Tier 2 CI currently runs on standard hosted runners, so its results are approximate and should be treated as a coarse
- regression signal unless a self-hosted GPU runner is configured
+- Use **CPU benchmarks** for isolated methods, helpers, caches, and allocation patterns
+- For rendering correctness, use visual regression tests (`pnpm test:visual`) — they produce PNG snapshots
Recommended commands:
```bash
pnpm bench
pnpm bench:json
-pnpm test:perf
```
CI status:
-- Tier 1 CPU benchmarks run in GitHub Actions on `main` pushes and on PRs labeled `perf-tier-1`
+- CPU benchmarks run in GitHub Actions on `main` pushes and on PRs labeled `perf`
- Labeled PR benchmark runs compare against the latest `main` baseline artifact
- The benchmark job comments on the PR and fails on regressions greater than 10%
-- Tier 2 GPU perf tests run in GitHub Actions on `main` pushes and on PRs labeled `perf-tier-2`
-- Tier 2 PR perf runs compare against the latest `main` GPU perf baseline artifact
-- The GPU perf job comments on the PR and fails on median frame-time regressions greater than 50%
-- Tier 2 CI is browser/frame-time coverage on hosted Linux runners, not hardware-accurate GPU benchmarking
Claude Code reusable skill:
- Use `.claude/skills/perf/SKILL.md` for benchmark-related work
-- Use it when adding a new `*.bench.ts`, extending browser perf fixtures, or reasoning about benchmark CI behavior
+- Use it when adding a new `*.bench.ts` or reasoning about benchmark CI behavior
## Git
diff --git a/README.md b/README.md
index d1dc494..94a5dd7 100644
--- a/README.md
+++ b/README.md
@@ -97,7 +97,6 @@ Additional documentation is available in the `docs/` directory:
| `pnpm test:visual:coverage` | Run visual tests with Istanbul coverage report |
| `pnpm bench` | Run Tier 1 CPU benchmarks (Vitest bench) |
| `pnpm bench:json` | Run Tier 1 benchmarks and write `benchmark-results.json` |
-| `pnpm test:perf` | Run Tier 2 browser/GPU frame-time benchmarks (Playwright) |
| `pnpm preflight` | Run all quality checks (format, lint, typecheck, spellcheck, knip, test) |
| `pnpm knip` | Find unused exports and dependencies |
| `pnpm knip:fix` | Auto-fix unused exports and dependencies |
diff --git a/docs/performance-testing.md b/docs/performance-testing.md
index 1a6379a..4b9c3a0 100644
--- a/docs/performance-testing.md
+++ b/docs/performance-testing.md
@@ -1,17 +1,13 @@
# Performance Testing
-Blit-Tech has two performance testing layers: fast CPU micro-benchmarks for hot methods and browser-based frame-time
-benchmarks for rendered workloads. This guide explains when to use each one, how to add a new benchmark, and how CI uses
-the results.
+Blit-Tech has CPU micro-benchmarks for hot methods. This guide explains when to use them, how to add a new benchmark,
+and how CI uses the results.
## Table of Contents
- [Overview](#overview)
-- [Tier 1: CPU Benchmarks](#tier-1-cpu-benchmarks)
-- [Tier 2: GPU Frame-Time Benchmarks](#tier-2-gpu-frame-time-benchmarks)
-- [Choosing the Right Benchmark Type](#choosing-the-right-benchmark-type)
+- [CPU Benchmarks](#cpu-benchmarks)
- [Adding a New CPU Benchmark](#adding-a-new-cpu-benchmark)
-- [Adding a New GPU Performance Test](#adding-a-new-gpu-performance-test)
- [Commands](#commands)
- [CI Benchmark Workflow](#ci-benchmark-workflow)
- [Recommended Workflow for New Performance Work](#recommended-workflow-for-new-performance-work)
@@ -20,12 +16,11 @@ the results.
## Overview
-Blit-Tech now uses two benchmarking methods:
+Blit-Tech uses Vitest bench for CPU micro-benchmarks. These measure isolated methods, hot loops, cache lookups, math
+helpers, and allocation patterns.
-1. **Tier 1: CPU benchmarks** with Vitest bench.
-2. **Tier 2: browser frame-time performance tests** with Playwright and Chromium WebGPU.
-
-These serve different purposes.
+For visual correctness (not performance), use the visual regression tests: `pnpm test:visual`. They run Playwright with
+Chromium + WebGPU and produce PNG snapshots. See `docs/testing.md` for details.
### CPU Benchmarks
@@ -39,22 +34,11 @@ Examples:
- `BitmapFont.measureText()` cold vs warm cache
- `Rect2i.containsXY()` vs `Rect2i.contains()`
-### GPU Performance Tests
-
-Use GPU performance tests to evaluate whole-frame rendering cost in the browser.
-
-Examples:
-
-- 500 sprites with one texture vs alternating textures
-- many filled rects or diagonal lines
-- bitmap text rendering throughput
-- mixed scenes with primitives and sprites together
-
---
-## Tier 1: CPU Benchmarks
+## CPU Benchmarks
-Tier 1 benchmarks are implemented with Vitest bench and colocated next to the source as `*.bench.ts` files.
+CPU benchmarks are implemented with Vitest bench and colocated next to the source as `*.bench.ts` files.
Current benchmark files:
@@ -64,7 +48,7 @@ Current benchmark files:
- `src/assets/BitmapFont.bench.ts`
- `src/assets/PaletteEffect.bench.ts`
-### Tier 1 Metrics
+### Metrics
CPU benchmarks report **ops/sec**. Higher numbers are better.
@@ -85,102 +69,17 @@ CPU benchmarks are:
- faster to run locally
- easier to write
- easier to reason about
-- already integrated into CI regression checks when the PR is labeled `perf-tier-1`
+- already integrated into CI regression checks when the PR is labeled `perf`
If you add a new hot method and want immediate automated regression protection, this is the first tool to use.
---
-## Tier 2: GPU Frame-Time Benchmarks
-
-Tier 2 performance tests are implemented with Playwright and browser fixtures under `tests/perf/` and
-`tests/visual/fixtures/`.
-
-Current entrypoints:
-
-- `tests/perf/perf.spec.ts`
-- `tests/visual/fixtures/perf-primitives.html`
-- `tests/visual/fixtures/perf-sprites.html`
-- `tests/visual/fixtures/perf-fonts.html`
-- `tests/visual/fixtures/perf-mixed.html`
-
-### Tier 2 Metrics
-
-GPU performance tests measure **frame time**, not ops/sec.
-
-The fixture renders a workload for a fixed number of frames and records frame durations with `performance.now()`.
-Playwright collects the raw frame times and computes:
-
-- `median`
-- `p95`
-- `p99`
-
-Lower frame times are better.
-
-### Browser Runtime
-
-Tier 2 uses the Playwright-managed **pinned Chromium** build that matches the Playwright version in this repository.
-That keeps browser revisions stable across local runs and CI, which reduces baseline drift compared with floating
-branded Chrome installs.
-
-### Why This Exists
-
-Some changes look cheap in isolation but are expensive in a real frame. For example:
-
-- texture batch switching
-- sprite submission overhead
-- rendering many lines or quads in one frame
-- interaction between primitives and sprites
-
-That is what Tier 2 is for.
-
-### Current CI Status
-
-Tier 2 can be run locally with `pnpm test:perf` and is now wired into CI as a **label-gated PR workflow**. CI uses a
-much looser threshold for GPU perf than Tier 1 because browser/GPU variance is higher on shared runners.
-
-Today, that CI job still runs on standard GitHub-hosted Linux runners, not on dedicated GPU hardware. Treat Tier 2 CI as
-an **approximate browser perf smoke signal**, not as a hardware-accurate GPU benchmark. If a self-hosted GPU runner is
-added later, the same workflow can be pointed at it and re-baselined.
-
----
-
-## Choosing the Right Benchmark Type
-
-Use this rule of thumb:
-
-### Use Tier 1 CPU Benchmarks When
-
-- the code runs in Node without a browser
-- you are measuring a specific method or helper
-- you want to compare allocating vs in-place variants
-- you want CI to catch regressions now
-
-### Use Tier 2 GPU Performance Tests When
-
-- the code depends on browser rendering
-- the important question is frame time, not raw method throughput
-- the code affects batching, submission, or whole-frame draw cost
-
-If you need hardware-true GPU numbers, a controlled local machine or a future self-hosted GPU runner is still a better
-source of truth than current hosted CI.
-
-### Use Both When
-
-You changed something in the render path and want both:
-
-- a micro-benchmark for the hot method itself
-- a browser performance test for actual frame impact
-
-This is often the best approach for rendering code.
-
----
-
## Adding a New CPU Benchmark
-If you add a new sprite-related method and it can run without a browser, start here.
+If you add a new method and it can run without a browser, start here.
-### CPU Benchmark File Location
+### File Location
Create or extend a `*.bench.ts` file near the code being measured.
@@ -242,62 +141,23 @@ pnpm bench:json
---
-## Adding a New GPU Performance Test
-
-Use this when the change is meaningful only in a browser render loop.
-
-### GPU Benchmark File Location
-
-Add or extend:
-
-- a fixture page in `tests/visual/fixtures/`
-- a scenario in `tests/perf/perf.spec.ts`
-
-### Typical Pattern
-
-1. Create a parameterized fixture that renders the workload.
-2. Run a fixed number of frames.
-3. Record frame times with `performance.now()`.
-4. Return the raw frame times to Playwright.
-5. Let the Playwright spec compute `median`, `p95`, and `p99`.
-
-### Good GPU Performance Cases
-
-- many sprites using one texture
-- many sprites alternating textures
-- many bitmap text characters
-- mixed primitive + sprite workloads
-- stress cases that intentionally push batching or fill rate
-
-### Run GPU Benchmarks
-
-```bash
-pnpm test:perf
-```
-
-The results are written to `test-results/perf/perf-results.json`.
-
----
-
## Commands
```bash
-pnpm bench # Run all Tier 1 CPU benchmarks
-pnpm bench:json # Run Tier 1 benchmarks and write benchmark-results.json
-pnpm test:perf # Run Tier 2 browser/GPU performance tests
+pnpm bench # Run all CPU benchmarks
+pnpm bench:json # Run benchmarks and write benchmark-results.json
```
### Which Command Should I Use?
- **New hot method:** `pnpm bench`
- **Need machine-readable result for comparison:** `pnpm bench:json`
-- **Whole browser frame cost:** `pnpm test:perf`
---
## CI Benchmark Workflow
-Tier 1 CPU benchmark regression detection is now wired into GitHub Actions.
+CPU benchmark regression detection is wired into GitHub Actions.
### What Happens on `main`
@@ -311,7 +171,7 @@ That artifact becomes the reference point for future pull requests.
### What Happens on a Pull Request
-On PRs targeting `main` with the `perf-tier-1` label, CI:
+On PRs targeting `main` with the `perf` label, CI:
1. runs the normal `quality` job first
2. runs the `benchmark` job
@@ -321,31 +181,7 @@ On PRs targeting `main` with the `perf-tier-1` label, CI:
6. posts or updates a PR comment with a benchmark comparison table
7. fails the job if any benchmark is more than **10% slower**
-PRs without the `perf-tier-1` label skip the Tier 1 benchmark job to reduce CI cost.
-
-### Tier 2 GPU Perf Workflow
-
-Tier 2 GPU perf uses the same baseline-artifact model, but it is gated separately and uses a much looser threshold.
-
-On pushes to `main`, CI:
-
-1. runs `pnpm test:perf`
-2. produces `test-results/perf/perf-results.json`
-3. uploads that file as the latest GPU perf baseline artifact
-
-On PRs targeting `main` with the `perf-tier-2` label, CI:
-
-1. runs `pnpm test:perf`
-2. downloads the latest successful `main` GPU perf baseline artifact
-3. compares current frame-time stats against the baseline
-4. posts or updates a PR comment with the GPU perf comparison table
-5. fails the job if any scenario has a **median frame time** regression greater than **50%**
-
-PRs without the `perf-tier-2` label skip the Tier 2 GPU perf job to reduce CI cost.
-
-Tier 2 CI currently uses Playwright-managed pinned Chromium on standard GitHub-hosted Linux runners. That keeps browser
-revisions stable, but the hardware is still shared and not GPU-dedicated. Use these CI results as approximate browser
-perf coverage rather than hardware-accurate GPU benchmarking.
+PRs without the `perf` label skip the benchmark job to reduce CI cost.
### What the PR Comment Contains
@@ -371,20 +207,15 @@ The benchmark CI job fails if:
If you add a new sprite operation, follow this order:
1. **Write the code clearly first.**
-2. **Add a Tier 1 CPU benchmark** if the method can run in Node.
+2. **Add a CPU benchmark** if the method can run in Node.
3. **Run `pnpm bench` locally** to compare the new method against the old behavior or an alternative implementation.
4. **Run `pnpm bench:json`** if you want to inspect the machine-readable output used by CI.
-5. **Add a Tier 2 GPU test** if the change affects real render throughput or frame time.
-6. **Open a PR** and add:
- - `perf-tier-1` for Tier 1 CPU benchmark comparison
- - `perf-tier-2` for Tier 2 GPU perf comparison
+5. **Open a PR** and add the `perf` label for CPU benchmark comparison.
### Best Default
-If you are unsure which path to choose:
-
-- start with **Tier 1 CPU benchmarking**
+If you are unsure where to start:
-It is simpler, faster, and already supported by the label-gated benchmark CI.
+- use **CPU benchmarking** with Vitest bench
-Add Tier 2 only when the real question is about frame rendering cost in the browser.
+It is simple, fast, and already supported by the label-gated benchmark CI.
diff --git a/eslint.config.js b/eslint.config.js
index 1369466..bdffacb 100644
--- a/eslint.config.js
+++ b/eslint.config.js
@@ -168,7 +168,7 @@ export default [
// Test files - relaxed rules
{
- files: ['**/*.test.ts', 'src/__test__/**/*.ts', 'tests/perf/**/*.ts', 'tests/visual/**/*.ts'],
+ files: ['**/*.test.ts', 'src/__test__/**/*.ts', 'tests/visual/**/*.ts'],
languageOptions: {
globals: {
...globals.browser,
diff --git a/package.json b/package.json
index 97f51ba..99387fd 100644
--- a/package.json
+++ b/package.json
@@ -59,7 +59,6 @@
"bench": "vitest bench",
"bench:json": "vitest bench --outputJson benchmark-results.json",
"test:visual": "playwright test",
- "test:perf": "playwright test -c playwright.perf.config.ts",
"test:visual:update": "playwright test --update-snapshots",
"test:visual:coverage": "VISUAL_COVERAGE=1 playwright test && nyc report --reporter=lcov --reporter=text-summary --temp-dir=.nyc_output --report-dir=coverage-visual",
"preflight": "pnpm format:check && pnpm lint && pnpm typecheck && pnpm spellcheck && pnpm knip && pnpm test:unit",
diff --git a/playwright.perf.config.ts b/playwright.perf.config.ts
deleted file mode 100644
index 6921e92..0000000
--- a/playwright.perf.config.ts
+++ /dev/null
@@ -1,41 +0,0 @@
-import { defineConfig, devices } from '@playwright/test';
-
-export default defineConfig({
- testDir: 'tests/perf',
- outputDir: 'test-results/perf',
- timeout: 120_000,
-
- fullyParallel: false,
- forbidOnly: !!process.env.CI,
- retries: process.env.CI ? 1 : 0,
- workers: 1,
-
- reporter: process.env.CI ? 'github' : 'list',
-
- use: {
- baseURL: 'http://127.0.0.1:5175',
- screenshot: 'only-on-failure',
- trace: 'on-first-retry',
- },
-
- projects: [
- {
- name: 'chromium-webgpu',
- use: {
- ...devices['Desktop Chrome'],
- browserName: 'chromium',
- launchOptions: {
- args: ['--enable-unsafe-webgpu', '--enable-features=Vulkan', '--disable-gpu-sandbox'],
- },
- },
- },
- ],
-
- webServer: {
- command:
- 'pnpm exec vite serve tests/visual/fixtures --config tests/visual/fixtures/vite.config.ts --host 127.0.0.1 --port 5175',
- port: 5175,
- reuseExistingServer: !process.env.CI,
- timeout: 30_000,
- },
-});
diff --git a/scripts/compare-tier-2-perf-results.mjs b/scripts/compare-tier-2-perf-results.mjs
deleted file mode 100644
index aed48b7..0000000
--- a/scripts/compare-tier-2-perf-results.mjs
+++ /dev/null
@@ -1,556 +0,0 @@
-import fs from 'node:fs';
-import path from 'node:path';
-
-const COMMENT_MARKER = '';
-const DEFAULT_THRESHOLD = 50;
-
-/**
- * Parses CLI arguments for the perf comparison command.
- *
- * @param {string[]} argv Raw CLI arguments after the node/script prefix.
- * @returns {{
- * baseline: string | null,
- * current: string | null,
- * jsonOut: string,
- * markdownOut: string,
- * threshold: number,
- * }} Parsed command options.
- */
-function parseArgs(argv) {
- const args = {
- baseline: null,
- current: null,
- jsonOut: 'perf-comparison.json',
- markdownOut: 'perf-comment.md',
- threshold: DEFAULT_THRESHOLD,
- };
-
- for (let index = 0; index < argv.length; index += 1) {
- // eslint-disable-next-line security/detect-object-injection -- CLI flags are parsed from a fixed argv array shape.
- const value = argv[index];
- const nextValue = argv[index + 1];
-
- if (!value.startsWith('--')) {
- throw new Error(`Unexpected positional token: ${value}`);
- }
-
- if (nextValue === undefined || nextValue.startsWith('--') || nextValue.startsWith('-')) {
- throw new Error(`Missing or invalid value for argument: ${value}`);
- }
-
- switch (value) {
- case '--baseline':
- args.baseline = nextValue;
- index += 1;
- break;
- case '--current':
- args.current = nextValue;
- index += 1;
- break;
- case '--json-out':
- args.jsonOut = nextValue;
- index += 1;
- break;
- case '--markdown-out':
- args.markdownOut = nextValue;
- index += 1;
- break;
- case '--threshold':
- args.threshold = Number(nextValue);
- index += 1;
- break;
- default:
- throw new Error(`Unknown argument: ${value}`);
- }
- }
-
- if (!args.current) {
- throw new Error('The --current argument is required');
- }
-
- if (!Number.isFinite(args.threshold) || args.threshold < 0) {
- throw new Error(`Invalid --threshold value: ${String(args.threshold)}`);
- }
-
- return args;
-}
-
-/**
- * Ensures that the parent directory for an output file exists.
- *
- * @param {string} filePath Output file path.
- * @returns {void}
- */
-function ensureParentDirectory(filePath) {
- fs.mkdirSync(path.dirname(filePath), { recursive: true });
-}
-
-/**
- * Reads and parses a JSON file from disk.
- *
- * @param {string} filePath JSON file path.
- * @returns {unknown} Parsed JSON value.
- */
-function readJsonFile(filePath) {
- return JSON.parse(fs.readFileSync(filePath, 'utf8'));
-}
-
-/**
- * Throws when a required condition is not met.
- *
- * @param {unknown} condition Condition to evaluate.
- * @param {string} message Error message for failed assertions.
- * @returns {void}
- */
-function assert(condition, message) {
- if (!condition) {
- throw new Error(message);
- }
-}
-
-/**
- * Validates the top-level shape of a perf benchmark report.
- *
- * @param {unknown} report Parsed perf report JSON.
- * @param {string} sourceLabel Human-readable source label for error messages.
- * @returns {void}
- */
-function validatePerfReport(report, sourceLabel) {
- assert(report !== null && typeof report === 'object', `Invalid perf report in ${sourceLabel}: expected object`);
- assert(Array.isArray(report.scenarios), `Invalid perf report in ${sourceLabel}: report.scenarios must be an array`);
-}
-
-/**
- * Flattens the perf report into scenario entries suitable for comparison.
- *
- * @param {{ scenarios: unknown[], __sourceLabel?: string }} report Parsed perf report.
- * @returns {Array<{
- * fixture: string,
- * label: string,
- * matchKey: string,
- * name: string,
- * stats: {
- * frames: number,
- * max: number,
- * median: number,
- * min: number,
- * p95: number,
- * p99: number,
- * },
- * }>} Flattened perf scenario entries.
- */
-function flattenPerfScenarios(report) {
- const entries = [];
- const reportLabel = report.__sourceLabel ?? 'perf report';
- const seenMatchKeys = new Set();
-
- validatePerfReport(report, reportLabel);
-
- for (const [scenarioIndex, scenario] of report.scenarios.entries()) {
- assert(
- scenario !== null && typeof scenario === 'object',
- `Invalid scenario entry at index ${scenarioIndex} in ${reportLabel}`,
- );
- assert(
- typeof scenario.fixture === 'string',
- `Invalid scenario.fixture at index ${scenarioIndex} in ${reportLabel}`,
- );
- assert(typeof scenario.name === 'string', `Invalid scenario.name at index ${scenarioIndex} in ${reportLabel}`);
- assert(
- scenario.stats !== null && typeof scenario.stats === 'object',
- `Invalid scenario.stats for ${scenario.name} in ${reportLabel}`,
- );
- assert(
- Number.isFinite(scenario.stats.frames) &&
- Number.isInteger(scenario.stats.frames) &&
- scenario.stats.frames >= 0,
- `Invalid stats.frames for ${scenario.name} in ${reportLabel}`,
- );
- assert(
- Number.isFinite(scenario.stats.median) && scenario.stats.median >= 0,
- `Invalid stats.median for ${scenario.name} in ${reportLabel}`,
- );
- assert(
- Number.isFinite(scenario.stats.p95) && scenario.stats.p95 >= 0,
- `Invalid stats.p95 for ${scenario.name} in ${reportLabel}`,
- );
- assert(
- Number.isFinite(scenario.stats.p99) && scenario.stats.p99 >= 0,
- `Invalid stats.p99 for ${scenario.name} in ${reportLabel}`,
- );
- assert(
- Number.isFinite(scenario.stats.min) && scenario.stats.min >= 0,
- `Invalid stats.min for ${scenario.name} in ${reportLabel}`,
- );
- assert(
- Number.isFinite(scenario.stats.max) && scenario.stats.max >= 0,
- `Invalid stats.max for ${scenario.name} in ${reportLabel}`,
- );
-
- const matchKey = `${scenario.fixture}::${scenario.name}`;
- assert(
- !seenMatchKeys.has(matchKey),
- `Duplicate perf scenario matchKey ${matchKey} encountered in ${reportLabel}`,
- );
- seenMatchKeys.add(matchKey);
-
- entries.push({
- fixture: scenario.fixture,
- label: scenario.name,
- matchKey,
- name: scenario.name,
- stats: scenario.stats,
- });
- }
-
- return entries;
-}
-
-/**
- * Calculates the percentage change for a frame-time metric.
- *
- * @param {number} baselineValue Baseline frame-time value.
- * @param {number} currentValue Current frame-time value.
- * @returns {number | string} Percentage change where positive means slower.
- */
-function calculateRegressionPct(baselineValue, currentValue) {
- if (baselineValue === 0) {
- if (currentValue === 0) {
- return 0;
- }
-
- return currentValue > 0 ? 'zero-baseline-positive' : 'zero-baseline-negative';
- }
-
- return ((currentValue - baselineValue) / baselineValue) * 100;
-}
-
-/**
- * Checks whether a perf delta should count as a regression.
- *
- * @param {number | string | null} value Perf delta value.
- * @param {number} thresholdPct Maximum allowed slowdown percentage before failure.
- * @returns {boolean} True when the delta exceeds the allowed regression threshold.
- */
-function isRegressionDelta(value, thresholdPct) {
- if (value === 'zero-baseline-positive') {
- return true;
- }
-
- if (typeof value === 'number') {
- return value > thresholdPct;
- }
-
- return false;
-}
-
-/**
- * Checks whether a perf delta should count as an improvement.
- *
- * @param {number | string | null} value Perf delta value.
- * @returns {boolean} True when the delta indicates an improvement versus baseline.
- */
-function isImprovementDelta(value) {
- if (value === 'zero-baseline-negative') {
- return true;
- }
-
- if (typeof value === 'number') {
- return value < 0;
- }
-
- return false;
-}
-
-/**
- * Compares current perf results against an optional baseline report.
- *
- * @param {{ scenarios: unknown[], __sourceLabel?: string }} currentReport Current perf report.
- * @param {{ scenarios: unknown[], __sourceLabel?: string } | null} baselineReport Baseline perf report, if available.
- * @param {number} thresholdPct Maximum allowed slowdown percentage before failure.
- * @returns {{
- * generatedAt: string,
- * hasBaseline: boolean,
- * thresholdPct: number,
- * summary: {
- * total: number,
- * compared: number,
- * regressions: number,
- * improvements: number,
- * pass: number,
- * newScenarios: number,
- * missingScenarios: number,
- * },
- * scenarios: Array<{
- * fixture: string,
- * name: string,
- * baselineStats: {
- * frames: number,
- * max: number,
- * median: number,
- * min: number,
- * p95: number,
- * p99: number,
- * } | null,
- * currentStats: {
- * frames: number,
- * max: number,
- * median: number,
- * min: number,
- * p95: number,
- * p99: number,
- * } | null,
- * deltas: {
- * median: number | string | null,
- * p95: number | string | null,
- * p99: number | string | null,
- * },
- * status: string,
- * }>,
- * }} Comparison report used by CI and PR comments.
- */
-function comparePerfReports(currentReport, baselineReport, thresholdPct) {
- const currentEntries = flattenPerfScenarios(currentReport);
- const baselineEntries = baselineReport ? flattenPerfScenarios(baselineReport) : [];
- const currentByKey = new Map(currentEntries.map((entry) => [entry.matchKey, entry]));
- const baselineByKey = new Map(baselineEntries.map((entry) => [entry.matchKey, entry]));
- const keys = [...new Set([...baselineByKey.keys(), ...currentByKey.keys()])].sort((left, right) =>
- left.localeCompare(right),
- );
-
- const scenarios = keys.map((key) => {
- const baselineEntry = baselineByKey.get(key) ?? null;
- const currentEntry = currentByKey.get(key) ?? null;
-
- if (!baselineEntry) {
- return {
- fixture: currentEntry?.fixture ?? 'unknown',
- name: currentEntry?.label ?? key,
- baselineStats: null,
- currentStats: currentEntry?.stats ?? null,
- deltas: { median: null, p95: null, p99: null },
- status: 'new',
- };
- }
-
- if (!currentEntry) {
- return {
- fixture: baselineEntry.fixture,
- name: baselineEntry.label,
- baselineStats: baselineEntry.stats,
- currentStats: null,
- deltas: { median: null, p95: null, p99: null },
- status: 'missing',
- };
- }
-
- const deltas = {
- median: calculateRegressionPct(baselineEntry.stats.median, currentEntry.stats.median),
- p95: calculateRegressionPct(baselineEntry.stats.p95, currentEntry.stats.p95),
- p99: calculateRegressionPct(baselineEntry.stats.p99, currentEntry.stats.p99),
- };
- let status = 'pass';
-
- if (isRegressionDelta(deltas.median, thresholdPct)) {
- status = 'fail';
- } else if (isImprovementDelta(deltas.median)) {
- status = 'improved';
- }
-
- return {
- fixture: currentEntry.fixture,
- name: currentEntry.label,
- baselineStats: baselineEntry.stats,
- currentStats: currentEntry.stats,
- deltas,
- status,
- };
- });
-
- const summary = {
- total: scenarios.length,
- compared: scenarios.filter((scenario) => scenario.deltas.median !== null).length,
- regressions: scenarios.filter((scenario) => scenario.status === 'fail').length,
- improvements: scenarios.filter((scenario) => scenario.status === 'improved').length,
- pass: scenarios.filter((scenario) => scenario.status === 'pass').length,
- newScenarios: scenarios.filter((scenario) => scenario.status === 'new').length,
- missingScenarios: scenarios.filter((scenario) => scenario.status === 'missing').length,
- };
-
- return {
- generatedAt: new Date().toISOString(),
- hasBaseline: baselineReport !== null,
- thresholdPct,
- summary,
- scenarios,
- };
-}
-
-/**
- * Formats a frame-time value for markdown output.
- *
- * @param {number | null} value Frame-time value in milliseconds.
- * @returns {string} Formatted frame-time string.
- */
-function formatFrameTime(value) {
- if (value === null) {
- return 'n/a';
- }
-
- return `${value.toFixed(2)}ms`;
-}
-
-/**
- * Formats a regression delta percentage for markdown output.
- *
- * @param {number | string | null} value Percentage delta where positive means slower.
- * @returns {string} Formatted delta string.
- */
-function formatDelta(value) {
- if (value === null) {
- return 'n/a';
- }
-
- if (value === 'zero-baseline-positive') {
- return 'zero-baseline+';
- }
-
- if (value === 'zero-baseline-negative') {
- return 'zero-baseline-';
- }
-
- const sign = value > 0 ? '+' : '';
-
- return `${sign}${value.toFixed(2)}%`;
-}
-
-/**
- * Converts an internal perf scenario status into the PR comment label.
- *
- * @param {string} status Internal scenario status.
- * @returns {string} Human-readable status label.
- */
-function formatStatus(status) {
- switch (status) {
- case 'fail':
- return 'FAIL';
- case 'improved':
- return 'IMPROVED';
- case 'new':
- return 'NEW';
- case 'missing':
- return 'MISSING';
- default:
- return 'PASS';
- }
-}
-
-/**
- * Escapes a markdown table cell value.
- *
- * @param {string} value Raw markdown cell content.
- * @returns {string} Escaped markdown-safe value.
- */
-function escapeMarkdownCell(value) {
- return value.replaceAll('|', '\\|');
-}
-
-/**
- * Builds the markdown body for the perf PR comment.
- *
- * @param {{
- * hasBaseline: boolean,
- * thresholdPct: number,
- * summary: {
- * compared: number,
- * regressions: number,
- * improvements: number,
- * newScenarios: number,
- * missingScenarios: number,
- * },
- * scenarios: Array<{
- * fixture: string,
- * name: string,
- * baselineStats: { median: number } | null,
- * currentStats: { median: number } | null,
- * deltas: { median: number | string | null, p95: number | string | null, p99: number | string | null },
- * status: string,
- * }>,
- * }} report Comparison report.
- * @returns {string} Markdown comment body.
- */
-function buildMarkdown(report) {
- const lines = [COMMENT_MARKER, '## Tier 2 GPU Perf Comparison', ''];
-
- if (!report.hasBaseline) {
- lines.push(
- 'No `main` branch GPU perf baseline artifact is available yet. This run produced fresh perf results and uploaded them as artifacts.',
- '',
- `Configured regression threshold: ${report.thresholdPct}% slower median frame time.`,
- );
-
- return `${lines.join('\n')}\n`;
- }
-
- lines.push(
- `Compared ${report.summary.compared} perf scenarios against the latest \`main\` baseline. Regression threshold: ${report.thresholdPct}% slower median frame time.`,
- '',
- `Regressions: ${report.summary.regressions} | Improvements: ${report.summary.improvements} | New: ${report.summary.newScenarios} | Missing: ${report.summary.missingScenarios}`,
- '',
- '',
- 'Perf table
',
- '',
- '| Fixture | Scenario | Baseline median | Current median | Median delta | P95 delta | P99 delta | Status |',
- '| --- | --- | ---: | ---: | ---: | ---: | ---: | --- |',
- );
-
- for (const scenario of report.scenarios) {
- lines.push(
- `| ${escapeMarkdownCell(scenario.fixture)} | ${escapeMarkdownCell(scenario.name)} | ${formatFrameTime(scenario.baselineStats?.median ?? null)} | ${formatFrameTime(scenario.currentStats?.median ?? null)} | ${formatDelta(scenario.deltas.median)} | ${formatDelta(scenario.deltas.p95)} | ${formatDelta(scenario.deltas.p99)} | ${formatStatus(scenario.status)} |`,
- );
- }
-
- lines.push('', ' ');
-
- return `${lines.join('\n')}\n`;
-}
-
-/**
- * Writes a file to disk, creating parent directories when needed.
- *
- * @param {string} filePath Output file path.
- * @param {string} content File contents.
- * @returns {void}
- */
-function writeFile(filePath, content) {
- ensureParentDirectory(filePath);
- fs.writeFileSync(filePath, content);
-}
-
-/**
- * Runs the perf comparison CLI.
- *
- * @returns {void}
- */
-function main() {
- const args = parseArgs(process.argv.slice(2));
- const currentReport = readJsonFile(args.current);
- currentReport.__sourceLabel = args.current;
- let baselineReport = null;
-
- if (args.baseline) {
- assert(fs.existsSync(args.baseline), `Missing baseline perf report: ${args.baseline}`);
- baselineReport = readJsonFile(args.baseline);
- }
-
- if (baselineReport !== null) {
- baselineReport.__sourceLabel = args.baseline;
- }
-
- const report = comparePerfReports(currentReport, baselineReport, args.threshold);
-
- writeFile(args.jsonOut, JSON.stringify(report, null, 2));
- writeFile(args.markdownOut, buildMarkdown(report));
-}
-
-main();
diff --git a/tests/perf/perf.spec.ts b/tests/perf/perf.spec.ts
deleted file mode 100644
index 2bc9f62..0000000
--- a/tests/perf/perf.spec.ts
+++ /dev/null
@@ -1,214 +0,0 @@
-import fs from 'node:fs';
-import path from 'node:path';
-
-import { expect, test } from '@playwright/test';
-
-// #region Types
-
-interface RawPerfResult {
- frameTimes: number[];
- frames: number;
- scenario: string;
- warmupFrames: number;
- workload: Record;
-}
-
-interface PerfStats {
- frames: number;
- max: number;
- median: number;
- min: number;
- p95: number;
- p99: number;
-}
-
-interface PerfScenarioResult {
- fixture: string;
- name: string;
- stats: PerfStats;
- workload: Record;
-}
-
-interface PerfScenario {
- fixture: string;
- name: string;
- query: Record;
-}
-
-interface PerfWindow {
- __INIT_FAILED__?: boolean;
- __PERF_COMPLETE__?: boolean;
- __PERF_RESULT__?: RawPerfResult | null;
-}
-
-// #endregion
-
-// #region Constants
-
-const PERF_RESULTS_FILE = path.resolve(process.cwd(), 'test-results/perf/perf-results.json');
-
-const perfResults: PerfScenarioResult[] = [];
-
-const perfScenarios: PerfScenario[] = [
- {
- fixture: 'perf-primitives.html',
- name: 'primitives',
- query: { frames: 100, lines: 80, pixels: 400, rects: 120, warmup: 10 },
- },
- {
- fixture: 'perf-sprites.html',
- name: 'sprites-same-texture',
- query: { frames: 100, sprites: 800, textureMode: 'same', warmup: 10 },
- },
- {
- fixture: 'perf-sprites.html',
- name: 'sprites-alternating-textures',
- query: { frames: 100, sprites: 800, textureMode: 'alternating', warmup: 10 },
- },
- {
- fixture: 'perf-fonts.html',
- name: 'fonts',
- query: { chars: 480, frames: 100, lineWidth: 32, warmup: 10 },
- },
- {
- fixture: 'perf-mixed.html',
- name: 'mixed',
- query: { chars: 160, frames: 100, lines: 100, rects: 140, sprites: 220, warmup: 10 },
- },
-];
-
-// #endregion
-
-// #region Helpers
-
-/**
- * Builds a fixture URL with query parameters for a perf scenario.
- *
- * @param fixture - Fixture HTML filename.
- * @param query - Query parameters passed to the page.
- * @returns Relative URL for Playwright navigation.
- */
-function buildFixtureUrl(fixture: string, query: Record): string {
- const params = new URLSearchParams();
-
- for (const [key, value] of Object.entries(query)) {
- params.set(key, String(value));
- }
-
- return `/${fixture}?${params.toString()}`;
-}
-
-/**
- * Computes summary statistics from a list of frame times.
- *
- * @param frameTimes - Recorded frame durations in milliseconds.
- * @returns Median, tail percentiles, and min/max values.
- */
-function computeStats(frameTimes: number[]): PerfStats {
- const sorted = [...frameTimes].sort((left, right) => left - right);
-
- return {
- frames: sorted.length,
- max: sorted[sorted.length - 1] ?? 0,
- median: percentile(sorted, 0.5),
- min: sorted[0] ?? 0,
- p95: percentile(sorted, 0.95),
- p99: percentile(sorted, 0.99),
- };
-}
-
-/**
- * Calculates a percentile using nearest-rank selection.
- *
- * @param sorted - Frame times sorted ascending.
- * @param percentileValue - Percentile expressed as 0.0-1.0.
- * @returns Value at the requested percentile.
- */
-function percentile(sorted: number[], percentileValue: number): number {
- if (sorted.length === 0) {
- return 0;
- }
-
- const index = Math.min(sorted.length - 1, Math.ceil(sorted.length * percentileValue) - 1);
-
- return sorted.at(index) ?? 0;
-}
-
-/**
- * Writes the aggregated perf results to a machine-readable JSON file.
- */
-function writePerfResultsFile(): void {
- fs.mkdirSync(path.dirname(PERF_RESULTS_FILE), { recursive: true });
- fs.writeFileSync(
- PERF_RESULTS_FILE,
- JSON.stringify(
- {
- generatedAt: new Date().toISOString(),
- scenarios: perfResults,
- },
- null,
- 2,
- ),
- );
-}
-
-// #endregion
-
-// #region Tests
-
-test.afterAll(() => {
- writePerfResultsFile();
-});
-
-for (const scenario of perfScenarios) {
- test(`collects frame-time stats for ${scenario.name}`, async ({ page }, testInfo) => {
- await page.goto(buildFixtureUrl(scenario.fixture, scenario.query));
-
- await page.waitForFunction(
- () => {
- const perfWindow = window as unknown as PerfWindow;
-
- return perfWindow.__PERF_COMPLETE__ || perfWindow.__INIT_FAILED__;
- },
- { timeout: 30_000 },
- );
-
- const rawResult = await page.evaluate(() => {
- const perfWindow = window as unknown as PerfWindow;
-
- return {
- initFailed: perfWindow.__INIT_FAILED__,
- result: perfWindow.__PERF_RESULT__ ?? null,
- };
- });
-
- test.skip(rawResult.initFailed === true, 'WebGPU not available in this environment');
-
- expect(rawResult.result).not.toBeNull();
-
- const result = rawResult.result as RawPerfResult;
-
- expect(result.frameTimes.length).toBe(result.frames);
-
- const stats = computeStats(result.frameTimes);
- const scenarioResult: PerfScenarioResult = {
- fixture: scenario.fixture,
- name: scenario.name,
- stats,
- workload: result.workload,
- };
-
- perfResults.push(scenarioResult);
-
- await testInfo.attach('perf-stats', {
- body: JSON.stringify(scenarioResult, null, 2),
- contentType: 'application/json',
- });
-
- console.log(
- `[perf] ${scenario.name}: median=${stats.median.toFixed(2)}ms p95=${stats.p95.toFixed(2)}ms p99=${stats.p99.toFixed(2)}ms`,
- );
- });
-}
-
-// #endregion
diff --git a/tests/visual/fixtures/perf-common.js b/tests/visual/fixtures/perf-common.js
deleted file mode 100644
index ab6f917..0000000
--- a/tests/visual/fixtures/perf-common.js
+++ /dev/null
@@ -1,143 +0,0 @@
-/* global Image, document, window */
-
-export function readNumberParam(name, fallback) {
- const params = new URLSearchParams(window.location.search);
- const raw = params.get(name);
-
- if (raw === null) {
- return fallback;
- }
-
- const parsed = Number(raw);
-
- return Number.isFinite(parsed) ? parsed : fallback;
-}
-
-export function readStringParam(name, fallback) {
- const params = new URLSearchParams(window.location.search);
-
- return params.get(name) ?? fallback;
-}
-
-export function initializePerfState(scenario, workload) {
- window.__PERF_COMPLETE__ = false;
- window.__PERF_RESULT__ = null;
- window.__INIT_FAILED__ = false;
-
- return {
- completed: false,
- frameLimit: Math.max(1, readNumberParam('frames', 100) | 0),
- frameTimes: [],
- framesSeen: 0,
- lastFrameAt: null,
- scenario,
- warmupFrames: Math.max(0, readNumberParam('warmup', 10) | 0),
- workload,
- };
-}
-
-export function recordFrame(state) {
- if (state.completed) {
- return true;
- }
-
- const now = performance.now();
-
- if (state.lastFrameAt !== null) {
- const delta = now - state.lastFrameAt;
-
- if (state.framesSeen >= state.warmupFrames) {
- state.frameTimes.push(delta);
- }
- }
-
- state.lastFrameAt = now;
- state.framesSeen += 1;
-
- if (state.frameTimes.length >= state.frameLimit) {
- window.__PERF_RESULT__ = {
- frameTimes: state.frameTimes.slice(),
- frames: state.frameLimit,
- scenario: state.scenario,
- warmupFrames: state.warmupFrames,
- workload: state.workload,
- };
- window.__PERF_COMPLETE__ = true;
- state.completed = true;
- }
-
- return state.completed;
-}
-
-export function markInitFailed(error) {
- window.__INIT_FAILED__ = true;
-
- if (error) {
- console.error(error);
- }
-}
-
-export function repeatText(seed, length) {
- return seed.repeat(Math.ceil(length / seed.length)).slice(0, length);
-}
-
-/**
- * Creates and activates a default 16-color test palette.
- * Index 0 is transparent (never set — left as zero alpha).
- * Indices 1-10 cover common test colors.
- *
- * @param {object} BT - The BT namespace.
- * @param {object} Color32 - The Color32 class.
- * @returns {object} The created palette.
- */
-export function installDefaultTestPalette(BT, Color32) {
- const palette = BT.paletteCreate(16);
-
- palette.set(1, Color32.black());
- palette.set(2, Color32.red());
- palette.set(3, Color32.green());
- palette.set(4, Color32.blue());
- palette.set(5, Color32.yellow());
- palette.set(6, Color32.cyan());
- palette.set(7, Color32.white());
- palette.set(8, new Color32(32, 32, 64, 255));
- palette.set(9, new Color32(16, 24, 40, 255));
- palette.set(10, new Color32(128, 0, 0, 255));
-
- BT.paletteSet(palette);
-
- return palette;
-}
-
-export function createSpriteImage(primaryColor, secondaryColor) {
- const canvas = document.createElement('canvas');
- canvas.width = 8;
- canvas.height = 8;
- const context = canvas.getContext('2d');
-
- if (!context) {
- markInitFailed(new Error('Failed to create 2D canvas context for perf sprite fixture'));
-
- return Promise.reject(new Error('Failed to create 2D canvas context for perf sprite fixture'));
- }
-
- for (let y = 0; y < 8; y++) {
- for (let x = 0; x < 8; x++) {
- context.fillStyle = (x + y) % 2 === 0 ? primaryColor : secondaryColor;
- context.fillRect(x, y, 1, 1);
- }
- }
-
- const image = new Image();
-
- return new Promise((resolve, reject) => {
- image.onload = () => resolve(image);
- image.onerror = () => {
- const error = new Error('Failed to load generated perf sprite image');
-
- markInitFailed(error);
- reject(error);
- };
- image.src = canvas.toDataURL();
- });
-}
diff --git a/tests/visual/fixtures/perf-fonts.html b/tests/visual/fixtures/perf-fonts.html
deleted file mode 100644
index aea1a62..0000000
--- a/tests/visual/fixtures/perf-fonts.html
+++ /dev/null
@@ -1,90 +0,0 @@
-
-
-
-
-
- Perf Test: Fonts
-
-
-
-
-
-
-
-
-
-
-
-
diff --git a/tests/visual/fixtures/perf-mixed.html b/tests/visual/fixtures/perf-mixed.html
deleted file mode 100644
index 0c28827..0000000
--- a/tests/visual/fixtures/perf-mixed.html
+++ /dev/null
@@ -1,114 +0,0 @@
-
-
-
-
-
- Perf Test: Mixed
-
-
-
-
-
-
-
-
-
-
-
-
diff --git a/tests/visual/fixtures/perf-primitives.html b/tests/visual/fixtures/perf-primitives.html
deleted file mode 100644
index 2e3a576..0000000
--- a/tests/visual/fixtures/perf-primitives.html
+++ /dev/null
@@ -1,101 +0,0 @@
-
-
-
-
-
- Perf Test: Primitives
-
-
-
-
-
-
-
-
-
-
-
-
diff --git a/tests/visual/fixtures/perf-sprites.html b/tests/visual/fixtures/perf-sprites.html
deleted file mode 100644
index 4856bff..0000000
--- a/tests/visual/fixtures/perf-sprites.html
+++ /dev/null
@@ -1,97 +0,0 @@
-
-
-
-
-
- Perf Test: Sprites
-
-
-
-
-
-
-
-
-
-
-
-
diff --git a/tsconfig.json b/tsconfig.json
index 6eaf7a8..71287fb 100644
--- a/tsconfig.json
+++ b/tsconfig.json
@@ -36,12 +36,10 @@
},
"include": [
"src/**/*",
- "tests/perf/**/*.ts",
"tests/visual/**/*.ts",
"vite.config.ts",
"vitest.config.ts",
"playwright.config.ts",
- "playwright.perf.config.ts",
"*.d.ts"
],
"exclude": ["node_modules", "dist"]