From da97434723bcd9d11b7842bc4b9182b924af5bdb Mon Sep 17 00:00:00 2001 From: isabel Date: Sun, 10 May 2026 15:54:22 +0100 Subject: [PATCH] ci: proper diffing --- .github/workflows/diff.yml | 164 ++++------- modules/flake/packages/cmp-stats/cmp-stats.py | 267 ++++++++++++++++++ modules/flake/packages/cmp-stats/package.nix | 42 +++ modules/flake/packages/default.nix | 2 + 4 files changed, 363 insertions(+), 112 deletions(-) create mode 100644 modules/flake/packages/cmp-stats/cmp-stats.py create mode 100644 modules/flake/packages/cmp-stats/package.nix diff --git a/.github/workflows/diff.yml b/.github/workflows/diff.yml index 7a6fa379..961690df 100644 --- a/.github/workflows/diff.yml +++ b/.github/workflows/diff.yml @@ -67,6 +67,10 @@ jobs: - name: Install Lix uses: samueldr/lix-gha-installer-action@7b7f14d320d6aacfb65bd1ef761566b3b69e474c # v2026-02-22 + with: + extra_nix_config: | + substituters = https://cache.nixos.org/ https://nix-community.cachix.org https://isabelroses.cachix.org https://catppuccin.cachix.org https://extersia.cachix.org + trusted-public-keys = cache.nixos.org-1:6NCHdD59X431o0gWypbMrAURkbJ16ZPMQFGspcDShjY= nix-community.cachix.org-1:mB9FSh9qf2dCimDSUo8Zy7bkq5CX+/rkCWyvRCYg3Fs= isabelroses.cachix.org-1:mXdV/CMcPDaiTmkQ7/4+MzChpOe6Cb97njKmBQQmLPM= catppuccin.cachix.org-1:noG/4HkbhJb+lUAdKrph6LaozJvAeEEZj4N732IysmU= extersia.cachix.org-1:ZHy9765xrhn4lDKGTzWWykHC+B091oTqNxClgc78MQU= - name: Checkout base uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 @@ -75,124 +79,60 @@ jobs: path: base persist-credentials: false + - name: Discover hosts + id: hosts + run: | + set -euo pipefail + # shellcheck disable=SC2016 + nix eval --impure --json --expr ' + let + flake = (builtins.getFlake (toString ./.)).outputs; + mk = attr: toplevel: + map (name: { + inherit name; + attribute = "${attr}.${name}.${toplevel}"; + }) (builtins.attrNames (flake.${attr} or {})); + in + mk "nixosConfigurations" "config.system.build.toplevel" + ' > hosts.json + jq -r '.[] | .name' hosts.json + - name: Eval stats (before) working-directory: base run: | - attr=.#nixosConfigurations.amaterasu.config.system.build.toplevel - nix eval "$attr" > /dev/null - for i in $(seq 1 5); do - NIX_SHOW_STATS=1 NIX_SHOW_STATS_PATH="../stats-before-$i.json" \ - nix eval --no-eval-cache "$attr" > /dev/null - done + set -euo pipefail + mkdir -p ../stats-before/all + while IFS=$'\t' read -r name attr; do + echo "::group::eval before: $name" + NIX_SHOW_STATS=1 NIX_SHOW_STATS_PATH="../stats-before/all/${name}.json" \ + nix eval --no-eval-cache ".#${attr}" > /dev/null || \ + echo "warning: eval failed for $name (before)" + echo "::endgroup::" + done < <(jq -r '.[] | "\(.name)\t\(.attribute)"' ../hosts.json) - name: Eval stats (after) run: | - attr=.#nixosConfigurations.amaterasu.config.system.build.toplevel - nix eval "$attr" > /dev/null - for i in $(seq 1 5); do - NIX_SHOW_STATS=1 NIX_SHOW_STATS_PATH="stats-after-$i.json" \ - nix eval --no-eval-cache "$attr" > /dev/null - done - - - name: Generate stats table + set -euo pipefail + mkdir -p stats-after/all + while IFS=$'\t' read -r name attr; do + echo "::group::eval after: $name" + NIX_SHOW_STATS=1 NIX_SHOW_STATS_PATH="stats-after/all/${name}.json" \ + nix eval --no-eval-cache ".#${attr}" > /dev/null || \ + echo "warning: eval failed for $name (after)" + echo "::endgroup::" + done < <(jq -r '.[] | "\(.name)\t\(.attribute)"' hosts.json) + + - name: Compare stats run: | - python3 << 'PYEOF' - import json - import statistics - - def flatten(d, prefix=''): - result = {} - for k, v in d.items(): - if isinstance(v, dict): - result.update(flatten(v, f'{prefix}{k}.')) - elif isinstance(v, (int, float)): - result[f'{prefix}{k}'] = v - return result - - def load_runs(pattern, count=5): - runs = [] - for i in range(1, count + 1): - with open(pattern.format(i)) as f: - runs.append(flatten(json.load(f))) - return runs - - def average_runs(runs): - all_keys = set() - for r in runs: - all_keys.update(r.keys()) - result = {} - for k in all_keys: - values = [r[k] for r in runs if k in r] - result[k] = statistics.mean(values) - return result - - def stddev_runs(runs): - all_keys = set() - for r in runs: - all_keys.update(r.keys()) - result = {} - for k in all_keys: - values = [r[k] for r in runs if k in r] - result[k] = statistics.stdev(values) if len(values) > 1 else 0 - return result - - before_runs = load_runs('stats-before-{}.json') - after_runs = load_runs('stats-after-{}.json') - - before = average_runs(before_runs) - after = average_runs(after_runs) - before_sd = stddev_runs(before_runs) - after_sd = stddev_runs(after_runs) - - all_keys = sorted(set(before) | set(after)) - - def fmt(n): - return f'{n:.3f}' if isinstance(n, float) else f'{int(n):,}' - - def is_significant(before_runs, after_runs, key, threshold=0.05): - """Welch's t-test to determine if the difference is significant.""" - import math - b_vals = [r[key] for r in before_runs if key in r] - a_vals = [r[key] for r in after_runs if key in r] - n_b, n_a = len(b_vals), len(a_vals) - if n_b < 2 or n_a < 2: - return False - mean_b = statistics.mean(b_vals) - mean_a = statistics.mean(a_vals) - var_b = statistics.variance(b_vals) - var_a = statistics.variance(a_vals) - se = math.sqrt(var_b / n_b + var_a / n_a) - if se == 0: - return mean_a != mean_b - t_stat = abs(mean_a - mean_b) / se - # Approximate p-value using degrees of freedom via Welch-Satterthwaite - num = (var_b / n_b + var_a / n_a) ** 2 - denom = (var_b / n_b) ** 2 / (n_b - 1) + (var_a / n_a) ** 2 / (n_a - 1) - df = num / denom if denom > 0 else 1 - # Conservative t-critical values for two-tailed p<0.05 - # For df>=4 (our case with 5 runs each), t_crit ~ 2.78 (df=4) to 2.31 (df=8) - t_crit = 2.78 if df <= 4 else 2.45 if df <= 6 else 2.31 - return t_stat > t_crit - - lines = [ - '| Metric | Before (mean +/- σ) | After (mean +/- σ) | Δ | % | Sig? |', - '|--------|---------------------|--------------------|----|---|------|', - ] - for key in all_keys: - b = before.get(key, 0) - a = after.get(key, 0) - bs = before_sd.get(key, 0) - as_ = after_sd.get(key, 0) - diff = a - b - pct = f'{diff / b * 100:+.1f}%' if b != 0 else 'N/A' - sign = '+' if diff > 0 else '' - sig = 'Yes' if is_significant(before_runs, after_runs, key) else '' - lines.append(f'| `{key}` | {fmt(b)} ± {fmt(bs)} | {fmt(a)} ± {fmt(as_)} | {sign}{fmt(diff)} | {pct} | {sig} |') - - table = '\n'.join(lines) - with open('stats-table.md', 'w') as f: - f.write(f'## Nix Eval Stats: `amaterasu`\n\n{table}\n') - PYEOF + set -euo pipefail + { + echo "" + echo "## Nix Eval Stats" + echo + echo "Paired comparison across $(jq 'length' hosts.json) host(s). Metrics with identical values across all hosts are listed under _Unchanged_; the rest get a paired t-test (p-value, t-stat)." + echo + nix run .#cmp-stats -- --explain stats-before stats-after + } > stats-table.md - name: Post comment uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 @@ -200,7 +140,7 @@ jobs: script: | const fs = require('fs'); const body = fs.readFileSync('stats-table.md', 'utf8'); - const marker = '## Nix Eval Stats: `amaterasu`'; + const marker = ''; const { data: comments } = await github.rest.issues.listComments({ owner: context.repo.owner, repo: context.repo.repo, diff --git a/modules/flake/packages/cmp-stats/cmp-stats.py b/modules/flake/packages/cmp-stats/cmp-stats.py new file mode 100644 index 00000000..91d5a7c2 --- /dev/null +++ b/modules/flake/packages/cmp-stats/cmp-stats.py @@ -0,0 +1,267 @@ +# Vendored from nixpkgs ci/eval/compare/cmp-stats.py (MIT licensed). +# Source: https://github.com/NixOS/nixpkgs/blob/master/ci/eval/compare/cmp-stats.py +import argparse +import json +import numpy as np +import pandas as pd + +from dataclasses import asdict, dataclass +from pathlib import Path +from scipy.stats import ttest_rel +from tabulate import tabulate +from typing import Final + + +def flatten_data(json_data: dict) -> dict: + flat_metrics = {} + for key, value in json_data.items(): + if key == "cpuTime": + continue + + if isinstance(value, (int, float)): + flat_metrics[key] = value + elif isinstance(value, dict): + for subkey, subvalue in value.items(): + assert isinstance(subvalue, (int, float)), subvalue + flat_metrics[f"{key}.{subkey}"] = subvalue + else: + assert isinstance(value, (float, int, dict)), ( + f"Value `{value}` has unexpected type" + ) + + return flat_metrics + + +def load_all_metrics(path: Path) -> dict: + metrics = {} + if path.is_dir(): + for system_dir in path.iterdir(): + assert system_dir.is_dir() + + for chunk_output in system_dir.iterdir(): + with chunk_output.open() as f: + data = json.load(f) + + metrics[f"{system_dir.name}/{chunk_output.name}"] = flatten_data(data) + else: + with path.open() as f: + metrics[path.name] = flatten_data(json.load(f)) + + return metrics + + +def metric_table_name(name: str, explain: bool) -> str: + return f"{name}[^{name}]" if explain else name + + +METRIC_EXPLANATION_FOOTNOTE: Final[str] = """ + +[^time.cpu]: Number of seconds of CPU time accounted by the OS to the Nix evaluator process. On UNIX systems, this comes from [`getrusage(RUSAGE_SELF)`](https://man7.org/linux/man-pages/man2/getrusage.2.html). +[^time.gc]: Number of seconds of CPU time accounted by the Boehm garbage collector to performing GC. +[^time.gcFraction]: What fraction of the total CPU time is accounted towards performing GC. +[^gc.cycles]: Number of times garbage collection has been performed. +[^gc.heapSize]: Size in bytes of the garbage collector heap. +[^gc.totalBytes]: Size in bytes of all allocations in the garbage collector. +[^envs.bytes]: Size in bytes of all `Env` objects allocated by the Nix evaluator. +[^list.bytes]: Size in bytes of all lists allocated by the Nix evaluator. +[^sets.bytes]: Size in bytes of all attrsets allocated by the Nix evaluator. +[^symbols.bytes]: Size in bytes of all items in the Nix evaluator symbol table. +[^values.bytes]: Size in bytes of all values allocated by the Nix evaluator. +[^envs.number]: The count of all `Env` objects allocated. +[^nrAvoided]: The number of thunks avoided being created. +[^nrExprs]: The number of expression objects ever created. +[^nrFunctionCalls]: The number of function calls ever made. +[^nrLookups]: The number of lookups into an attrset ever made. +[^nrOpUpdateValuesCopied]: The number of attrset values copied in the process of merging attrsets. +[^nrOpUpdates]: The number of attrset merge operations (`//`) performed. +[^nrPrimOpCalls]: The number of function calls to primops (Nix builtins) ever made. +[^nrThunks]: The number of thunks ever made. +[^sets.number]: The number of attrsets ever made. +[^symbols.number]: The number of symbols ever added to the symbol table. +[^values.number]: The number of values ever made. +[^envs.elements]: The number of values contained within an `Env` object. +[^list.concats]: The number of list concatenation operations (`++`) performed. +[^list.elements]: The number of values contained within a list. +[^sets.elements]: The number of values contained within an attrset. +[^sizes.Attr]: Size in bytes of the `Attr` type. +[^sizes.Bindings]: Size in bytes of the `Bindings` type. +[^sizes.Env]: Size in bytes of the `Env` type. +[^sizes.Value]: Size in bytes of the `Value` type. +""" + + +@dataclass(frozen=True) +class PairwiseTestResults: + updated: pd.DataFrame + equivalent: pd.DataFrame + + @staticmethod + def tabulate(table, headers) -> str: + return tabulate( + table, headers, tablefmt="github", floatfmt=".4f", missingval="-" + ) + + def updated_to_markdown(self, explain: bool) -> str: + assert not self.updated.empty + return self.tabulate( + headers=[str(column) for column in self.updated.columns], + table=[ + [ + metric_table_name(row["metric"], explain), + *[ + None if np.isnan(val) or np.allclose(val, 0) else val + for val in row[1:] + ], + ] + for _, row in self.updated.iterrows() + ], + ) + + def equivalent_to_markdown(self, explain: bool) -> str: + assert not self.equivalent.empty + return self.tabulate( + headers=[str(column) for column in self.equivalent.columns], + table=[ + [ + metric_table_name(row["metric"], explain), + row["value"], + ] + for _, row in self.equivalent.iterrows() + ], + ) + + def to_markdown(self, explain: bool) -> str: + result = "" + + if not self.equivalent.empty: + result += "## Unchanged values\n\n" + result += self.equivalent_to_markdown(explain) + + if not self.updated.empty: + result += ("\n\n" if result else "") + "## Updated values\n\n" + result += self.updated_to_markdown(explain) + + if explain: + result += METRIC_EXPLANATION_FOOTNOTE + + return result + + +@dataclass(frozen=True) +class Equivalent: + metric: str + value: float + + +@dataclass(frozen=True) +class Comparison: + metric: str + mean_before: float + mean_after: float + mean_diff: float + mean_pct_change: float + + +@dataclass(frozen=True) +class ComparisonWithPValue(Comparison): + p_value: float + t_stat: float + + +def metric_sort_key(name: str): + if name in ("time.cpu", "time.gc", "time.gcFraction"): + return (1, name) + elif name.startswith("gc"): + return (2, name) + elif name.endswith(("bytes", "Bytes")): + return (3, name) + elif name.startswith("nr") or name.endswith("number"): + return (4, name) + else: + return (5, name) + + +def perform_pairwise_tests( + before_metrics: dict, after_metrics: dict +) -> PairwiseTestResults: + common_files = sorted(set(before_metrics) & set(after_metrics)) + all_keys = sorted( + { + metric_keys + for file_metrics in before_metrics.values() + for metric_keys in file_metrics.keys() + }, + key=metric_sort_key, + ) + + updated = [] + equivalent = [] + + for key in all_keys: + before_vals = [] + after_vals = [] + + for fname in common_files: + if key in before_metrics[fname] and key in after_metrics[fname]: + before_vals.append(before_metrics[fname][key]) + after_vals.append(after_metrics[fname][key]) + + if len(before_vals) == 0: + continue + + before_arr = np.array(before_vals) + after_arr = np.array(after_vals) + + diff = after_arr - before_arr + + if np.allclose(diff, 0): + equivalent.append(Equivalent(metric=key, value=before_vals[0])) + else: + pct_change = 100 * diff / before_arr + + result = Comparison( + metric=key, + mean_before=np.mean(before_arr), + mean_after=np.mean(after_arr), + mean_diff=np.mean(diff), + mean_pct_change=np.mean(pct_change), + ) + + if len(before_vals) > 1: + t_stat, p_val = ttest_rel(after_arr, before_arr) + result = ComparisonWithPValue( + **asdict(result), p_value=p_val, t_stat=t_stat + ) + + updated.append(result) + + return PairwiseTestResults( + updated=pd.DataFrame(map(asdict, updated)), + equivalent=pd.DataFrame(map(asdict, equivalent)), + ) + + +def main(): + parser = argparse.ArgumentParser( + description="Performance comparison of Nix evaluation statistics" + ) + parser.add_argument( + "--explain", action="store_true", help="Explain the evaluation statistics" + ) + parser.add_argument( + "before", help="File or directory containing baseline (data before)" + ) + parser.add_argument( + "after", help="File or directory containing comparison (data after)" + ) + + options = parser.parse_args() + + before_metrics = load_all_metrics(Path(options.before)) + after_metrics = load_all_metrics(Path(options.after)) + pairwise_test_results = perform_pairwise_tests(before_metrics, after_metrics) + print(pairwise_test_results.to_markdown(explain=options.explain)) + + +if __name__ == "__main__": + main() diff --git a/modules/flake/packages/cmp-stats/package.nix b/modules/flake/packages/cmp-stats/package.nix new file mode 100644 index 00000000..03fa9f4f --- /dev/null +++ b/modules/flake/packages/cmp-stats/package.nix @@ -0,0 +1,42 @@ +{ + lib, + python3, + stdenvNoCC, + makeWrapper, +}: +let + python = python3.withPackages (ps: [ + ps.numpy + ps.pandas + ps.scipy + ps.tabulate + ]); +in +stdenvNoCC.mkDerivation { + pname = "cmp-stats"; + version = "0"; + + dontUnpack = true; + + nativeBuildInputs = [ makeWrapper ]; + + installPhase = '' + runHook preInstall + + mkdir -p $out/share/cmp-stats + + cp ${./cmp-stats.py} "$out/share/cmp-stats/cmp-stats.py" + + makeWrapper ${python.interpreter} "$out/bin/cmp-stats" \ + --add-flags "$out/share/cmp-stats/cmp-stats.py" + + runHook postInstall + ''; + + meta = { + description = "Performance comparison of Nix evaluation statistics"; + license = lib.licenses.mit; + mainProgram = "cmp-stats"; + maintainers = with lib.maintainers; [ isabelroses ]; + }; +} diff --git a/modules/flake/packages/default.nix b/modules/flake/packages/default.nix index 34bdbb39..e9ac5d54 100644 --- a/modules/flake/packages/default.nix +++ b/modules/flake/packages/default.nix @@ -6,6 +6,8 @@ pkgs.lib.makeScope pkgs.newScope (self: { inherit inputs; # keep-sorted start block=yes newline_separated=yes + cmp-stats = self.callPackage ./cmp-stats/package.nix { }; + docs = self.callPackage ./docs/package.nix { }; iztaller = self.callPackage ./iztaller/package.nix { -- 2.51.2