ci(perf): add the weekly baseline tightening workflow

Plan item B4. performance-ratchet.yml measures master on three runners
(initial bytes and the journeys) and tighten-baselines.mjs lowers every
baseline that all three runs beat, opening a no-release-note PR only when
the file changed. The journeys steps move into a composite action shared
with ci.yml's Performance journeys job.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
4grayandClaude Opus 5.5 committed 2026-09-29 22:52:13 +02:00
1 parent 963a431bb7
commit a7f1ecfe49
9 files changed
+1110 -58

No files matched your search

@@ -0,0 +1,74 @@
name: Run the performance journeys
description: >-
Runs `pnpm run perf:journeys` under xvfb on an Ubuntu runner, writes the
measurements to the job summary and returns the path of the summary.json
it wrote. Callers check out, set up pnpm and Node, install dependencies
and upload what they need. Contract: docs/architecture/performance-journeys.md.
outputs:
summary:
description: Path of the journey summary.json this run wrote.
value: ${{ steps.summary.outputs.path }}
runs:
using: composite
steps:
# The journeys drive Electron through Playwright's _electron API
# and never launch a Playwright browser, so no `playwright
# install`. The runner image ships Electron's shared libraries and
# xvfb; fail fast with a clear message if an image update drops one.
# pnpm skips Electron's postinstall, so download the binary first;
# otherwise ldd sees no file and the check passes vacuously.
- name: Check Electron runtime dependencies
shell: bash
run: |
command -v xvfb-run || { echo "::error::xvfb-run is missing on the runner"; exit 1; }
node tools/testing/ensure-electron-binary.mjs || { echo "::error::Electron binary download failed"; exit 1; }
missing="$(ldd node_modules/electron/dist/electron | grep 'not found' || true)"
if [ -n "$missing" ]; then
echo "::error::Electron is missing shared libraries:"
echo "$missing"
exit 1
fi
# The Nx target builds electron-backend:build-performance first
# and playwright.journeys.config.ts starts the Xtream mock server.
- name: Run the performance journeys
shell: bash
run: xvfb-run --auto-servernum --server-args="-screen 0 1280x960x24" pnpm run perf:journeys
env:
CI: 'true'
NX_TASKS_RUNNER_DYNAMIC_OUTPUT: 'false'
# Every run writes a fresh timestamped directory, so a clean
# checkout must hold exactly one summary.
- name: Locate the journey summary
id: summary
shell: bash
run: |
set -euo pipefail
mapfile -t summaries < <(find dist/performance/journeys -mindepth 2 -maxdepth 2 -name summary.json)
if [ "${#summaries[@]}" -ne 1 ]; then
echo "::error::Expected one journey summary, found ${#summaries[@]}"
exit 1
fi
echo "path=${summaries[0]}" >> "$GITHUB_OUTPUT"
- name: Report journey measurements
shell: bash
env:
SUMMARY: ${{ steps.summary.outputs.path }}
run: |
set -euo pipefail
jq -r '
"## Performance journeys (\(.harness.platform), \(.harness.measuredIterations) measured iterations)", "",
(.journeys | to_entries[] | .key as $journey | .value as $j |
"### `\($journey)`", "",
"| Measurement | Value | Iterations |",
"| --- | ---: | --- |",
(($j.counters // {}) | to_entries[] |
($j.counterStability[.key] // {}) as $s |
"| `\(.key)` | \(.value) | \(($s.values // []) | map(tostring) | join(", "))\(if $s.stable == false then " (unstable)" else "" end) |"),
(($j.wallClock // {}) | to_entries[] | "| `\(.key)` | \(.value) | |"),
"")
' "$SUMMARY" | tee -a "$GITHUB_STEP_SUMMARY"
+3 -54
View File
@@ -321,61 +321,10 @@ jobs:
- name: Install dependencies
run: pnpm install --frozen-lockfile
# The journeys drive Electron through Playwright's _electron API
# and never launch a Playwright browser, so no `playwright
# install`. The runner image ships Electron's shared libraries and
# xvfb; fail fast with a clear message if an image update drops one.
# pnpm skips Electron's postinstall, so download the binary first;
# otherwise ldd sees no file and the check passes vacuously.
- name: Check Electron runtime dependencies
run: |
command -v xvfb-run || { echo "::error::xvfb-run is missing on the runner"; exit 1; }
node tools/testing/ensure-electron-binary.mjs || { echo "::error::Electron binary download failed"; exit 1; }
missing="$(ldd node_modules/electron/dist/electron | grep 'not found' || true)"
if [ -n "$missing" ]; then
echo "::error::Electron is missing shared libraries:"
echo "$missing"
exit 1
fi
# The Nx target builds electron-backend:build-performance first
# and playwright.journeys.config.ts starts the Xtream mock server.
# Electron dependency check, the xvfb run, the summary lookup and
# the job-summary report; shared with performance-ratchet.yml.
- name: Run the performance journeys
run: xvfb-run --auto-servernum --server-args="-screen 0 1280x960x24" pnpm run perf:journeys
env:
CI: true
NX_TASKS_RUNNER_DYNAMIC_OUTPUT: false
# Every run writes a fresh timestamped directory, so a clean
# checkout must hold exactly one summary.
- name: Locate the journey summary
id: summary
run: |
set -euo pipefail
mapfile -t summaries < <(find dist/performance/journeys -mindepth 2 -maxdepth 2 -name summary.json)
if [ "${#summaries[@]}" -ne 1 ]; then
echo "::error::Expected one journey summary, found ${#summaries[@]}"
exit 1
fi
echo "path=${summaries[0]}" >> "$GITHUB_OUTPUT"
- name: Report journey measurements
env:
SUMMARY: ${{ steps.summary.outputs.path }}
run: |
set -euo pipefail
jq -r '
"## Performance journeys (\(.harness.platform), \(.harness.measuredIterations) measured iterations)", "",
(.journeys | to_entries[] | .key as $journey | .value as $j |
"### `\($journey)`", "",
"| Measurement | Value | Iterations |",
"| --- | ---: | --- |",
(($j.counters // {}) | to_entries[] |
($j.counterStability[.key] // {}) as $s |
"| `\(.key)` | \(.value) | \(($s.values // []) | map(tostring) | join(", "))\(if $s.stable == false then " (unstable)" else "" end) |"),
(($j.wallClock // {}) | to_entries[] | "| `\(.key)` | \(.value) | |"),
"")
' "$SUMMARY" | tee -a "$GITHUB_STEP_SUMMARY"
uses: ./.github/actions/performance-journeys
- name: Upload journey summaries
if: always()
+223
View File
@@ -0,0 +1,223 @@
name: Performance ratchet
# Weekly tightening of tools/performance/journey-baselines.json (plan item B4;
# contract in the Ratchet section of docs/architecture/performance-journeys.md).
# Three runners measure the same master commit independently; a pull request
# lowers every baseline that all three beat. Nothing is ever raised, and slack
# and tolerances are left alone. A manual dispatch on another branch measures
# and reports but never opens a pull request.
on:
schedule:
- cron: '41 4 * * 1'
workflow_dispatch:
permissions:
contents: read
concurrency:
group: performance-ratchet-${{ github.ref }}
cancel-in-progress: false
jobs:
measure:
name: Measure (run ${{ matrix.run }})
if: github.repository == '4gray/iptvnator'
runs-on: ubuntu-latest
# Production web build plus the journeys (their electron-performance
# build and seven Electron processes per journey).
timeout-minutes: 60
strategy:
# Separate runners, so one machine's noise cannot pass for a
# reduction. The tightener needs all three runs.
fail-fast: true
matrix:
run: [1, 2, 3]
env:
NX_SKIP_NX_CACHE: true
steps:
- name: Checkout code
uses: actions/checkout@v7
with:
persist-credentials: false
- name: Install pnpm
uses: pnpm/action-setup@v6.0.10
- name: Setup Node.js
uses: actions/setup-node@v7
with:
node-version-file: '.nvmrc'
cache: 'pnpm'
- name: Install dependencies
run: pnpm install --frozen-lockfile
# Same build as the Initial bytes ratchet job in ci.yml. Measured
# before the journeys, whose build also writes to dist/.
- name: Build web app (production)
run: pnpm nx build web --skip-nx-cache
env:
CI: true
NX_TASKS_RUNNER_DYNAMIC_OUTPUT: false
- name: Measure renderer.initialBytes
run: node tools/performance/measure-initial-bytes.mjs --summary dist/performance/ratchet/initial-bytes.summary.json
# A failed journey run must not block tightening the initial bytes:
# its entries are then missing from this run's summary, and the
# tightener keeps any baseline that a run did not measure.
- name: Run the performance journeys
id: journeys
continue-on-error: true
uses: ./.github/actions/performance-journeys
- name: Keep the journey summary
if: steps.journeys.outputs.summary != ''
env:
SUMMARY: ${{ steps.journeys.outputs.summary }}
run: cp "$SUMMARY" dist/performance/ratchet/journeys.summary.json
- name: Upload run summaries
uses: actions/upload-artifact@v7
with:
name: performance-ratchet-run-${{ matrix.run }}
path: dist/performance/ratchet/
if-no-files-found: error
retention-days: 30
tighten:
name: Tighten baselines
needs: measure
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- name: Checkout code
uses: actions/checkout@v7
with:
persist-credentials: false
# The ratchet scripts are dependency-free Node; no pnpm install.
- name: Setup Node.js
uses: actions/setup-node@v7
with:
node-version-file: '.nvmrc'
- name: Download run summaries
uses: actions/download-artifact@v8
with:
pattern: performance-ratchet-run-*
path: ${{ runner.temp }}/runs
# Each run directory holds that runner's initial-bytes and journeys
# summaries. The direction check is the same one ci.yml runs on the
# PR, without --allow-increase: a tightening can only lower limits.
- name: Lower baselines every run beat
id: tighten
env:
RUNS_DIR: ${{ runner.temp }}/runs
REPORT: ${{ runner.temp }}/tighten-report.md
BASE_BASELINES: ${{ runner.temp }}/base-journey-baselines.json
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: |
set -euo pipefail
baselines=tools/performance/journey-baselines.json
cp "$baselines" "$BASE_BASELINES"
runs=()
for dir in "$RUNS_DIR"/performance-ratchet-run-*; do
runs+=(--run "$dir")
done
node tools/performance/tighten-baselines.mjs \
"${runs[@]}" \
--evidence-run "$RUN_URL" \
--report "$REPORT"
{
echo "## Performance ratchet"
echo
cat "$REPORT"
} >> "$GITHUB_STEP_SUMMARY"
node tools/performance/check-baseline-direction.mjs \
--base "$BASE_BASELINES" \
--head "$baselines"
if git diff --quiet -- "$baselines"; then
echo "changed=false" >> "$GITHUB_OUTPUT"
else
git diff -- "$baselines"
echo "changed=true" >> "$GITHUB_OUTPUT"
fi
- name: Report a dispatch outside master
if: steps.tighten.outputs.changed == 'true' && github.ref != 'refs/heads/master'
env:
REF_NAME: ${{ github.ref_name }}
run: echo "Dispatched on $REF_NAME, not master; the diff above is not proposed as a pull request."
# PAT, not GITHUB_TOKEN: a pull request pushed with GITHUB_TOKEN
# triggers no workflows, and this one needs ci.yml's Initial bytes
# ratchet run. Same secret as refresh-windows-embedded-mpv-runtime.
# The PR number is only known once the PR exists, so evidencePr is
# filled in afterwards and the single commit amended.
- name: Create or update the tightening pull request
if: steps.tighten.outputs.changed == 'true' && github.ref == 'refs/heads/master'
env:
GH_TOKEN: ${{ secrets.PAT }}
REPOSITORY: ${{ github.repository }}
BRANCH_NAME: automation/performance-ratchet
TITLE: 'ci(perf): tighten journey baselines'
LABEL: no-release-note
REPORT: ${{ runner.temp }}/tighten-report.md
BODY: ${{ runner.temp }}/pr-body.md
HEAD_SHA: ${{ github.sha }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: |
set -euo pipefail
if [ -z "${GH_TOKEN}" ]; then
echo "::error::The PAT secret is required so the bot-created PR triggers normal CI."
exit 1
fi
baselines=tools/performance/journey-baselines.json
gh auth setup-git
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
git switch -C "$BRANCH_NAME"
git add -- "$baselines"
git commit -m "$TITLE" -m "Measured by $RUN_URL"
git fetch origin "$BRANCH_NAME:refs/remotes/origin/$BRANCH_NAME" || true
git push --force-with-lease origin "HEAD:refs/heads/$BRANCH_NAME"
pr="$(gh api "repos/$REPOSITORY/pulls?base=master&head=${REPOSITORY%%/*}:$BRANCH_NAME&state=open" --jq '.[0].number // empty')"
if [ -z "$pr" ]; then
pr="$(gh api -X POST "repos/$REPOSITORY/pulls" \
-f title="$TITLE" -f head="$BRANCH_NAME" -f base=master \
-f body="Measurements follow." --jq '.number')"
fi
gh api -X POST "repos/$REPOSITORY/issues/$pr/labels" -f "labels[]=$LABEL" --silent
node tools/performance/tighten-baselines.mjs \
--fill-evidence-pr "$pr" --evidence-run "$RUN_URL"
git add -- "$baselines"
git commit --amend --no-edit
git push --force-with-lease origin "HEAD:refs/heads/$BRANCH_NAME"
{
echo "Automated weekly tightening of \`$baselines\` from [three runner measurements]($RUN_URL) of \`$HEAD_SHA\`."
echo
echo "Each lowered baseline was strictly below its value in all three runs and now holds the largest of the three. Slack and tolerances are unchanged; \`check-baseline-direction.mjs\` accepted the file without \`--allow-increase\`."
echo
echo "## Per-run measurements"
echo
cat "$REPORT"
echo
echo "## Diff"
echo
echo '```diff'
git diff "$HEAD_SHA" HEAD -- "$baselines"
echo '```'
echo
echo "If \`master\` moved since \`$HEAD_SHA\`, make sure the Initial bytes ratchet job passes on this PR before merging."
} > "$BODY"
gh api -X PATCH "repos/$REPOSITORY/pulls/$pr" -F "body=@$BODY" --silent
echo "Pull request: ${GITHUB_SERVER_URL}/$REPOSITORY/pull/$pr"
+41 -1
View File
@@ -525,7 +525,9 @@ the job for every PR until #1734. Treat the job as blocking before merging;
making it a required check is a maintainer decision.
The runtime counters come from the `Performance journeys` job of the same
workflow, on `ubuntu-latest` only. It runs `pnpm run perf:journeys` under
workflow, on `ubuntu-latest` only. Through the
`.github/actions/performance-journeys` composite action it runs
`pnpm run perf:journeys` under
`xvfb-run` (the Nx target builds `electron-backend:build-performance`, the
Playwright config starts the Xtream mock), writes the measurements to the job
summary and uploads `dist/performance/journeys/` as the `performance-journeys`
@@ -554,6 +556,44 @@ in all eighteen runner iterations; the `spawnToFirstCardMs` P50 ranged from
differ from a Mac (12 and 571 there, the fast path without the Linux-only
`getWindowState` call), so take J1 baseline values from the runner only.
### Weekly tightening
`.github/workflows/performance-ratchet.yml` lowers baselines without waiting
for someone to act on a "tighten" hint. Every Monday, and on
`workflow_dispatch`, three `ubuntu-latest` jobs measure the same commit
independently: the production `apps/web` build with
`measure-initial-bytes.mjs --summary`, then the journeys through the
`.github/actions/performance-journeys` composite action, which the
`Performance journeys` job above uses too. A failed journey run does not stop
its job; its entries are then unmeasured in that run. A final job runs
`tools/performance/tighten-baselines.mjs` on the three runs:
- an entry is lowered only when every run measured it and every measurement
is strictly below `value`; the new `value` is the largest of the three (for
a wall-clock entry the largest per-run summary value, which is the largest
P50 for a `.p50` entry);
- a counter marked `counterStability.<name>.stable: false` in any run is
kept, and the report says which run and which iterations disagreed;
- `value` never goes up, `slack` and `toleranceRatio` never change, and no
entry is added or removed: the result must pass
`check-baseline-direction.mjs` without `--allow-increase`, which both the
script and the job check;
- a lowered entry gets `updatedAt`, `measuredWith` and `evidenceRun` (the
workflow run URL); `evidencePr` is set to the tightening PR once it exists.
When the file changed and the run is on `master`, the job pushes
`automation/performance-ratchet` and opens (or updates) a pull request with
the per-run table, the diff and the run URL, labelled `no-release-note`. It
pushes with the existing `PAT` secret, as the Windows MPV pin refresh does,
because a pull request pushed with `GITHUB_TOKEN` starts no CI. When no
baseline was below its value in all three runs, the workflow ends without a
pull request. A dispatch on another branch measures and prints the diff but
never opens one; use `gh workflow run performance-ratchet.yml --ref <branch>`
to validate a change to the workflow. Review the pull request like a manual
tightening: if `master` moved since the measured commit, the
`Initial bytes ratchet` job on the pull request is what shows that the new
value still holds (the concurrent-merge effect above).
## Charset parse benchmark
V8 stores a string as two-byte UTF-16 once one character falls outside
+4 -1
View File
@@ -260,7 +260,10 @@ bytes on the initial path (the J1 counter `renderer.initialBytes`).
`tools/performance/journey-baselines.json`; baselines only move down. CI runs
the same check in the `Initial bytes ratchet` job of `ci.yml` for PRs that
target `master` and for `master` pushes (dispatch it with
`gh workflow run ci.yml --ref <branch>` for a stacked branch). `perf:journeys` builds the `electron-performance` configuration and runs every
`gh workflow run ci.yml --ref <branch>` for a stacked branch). The weekly
`performance-ratchet.yml` workflow lowers baselines through a bot PR; validate
a change to it with `gh workflow run performance-ratchet.yml --ref <branch>`,
which measures but opens no PR off `master`. `perf:journeys` builds the `electron-performance` configuration and runs every
journey spec against the Xtream mock: J1 launch, then J2 open-source (a
second set of launches, each followed by the click on the portal card), both
written to the same summary file; its probe specs run with
+1 -1
View File
@@ -77,7 +77,7 @@
"perf:initial-bytes:check": "node tools/performance/measure-initial-bytes.mjs --summary dist/performance/initial-bytes.summary.json && node tools/performance/check-journey-ratchet.mjs --summary dist/performance/initial-bytes.summary.json --only launch/renderer.initialBytes",
"perf:journeys": "nx run electron-backend-e2e:journeys",
"perf:ratchet:check": "node tools/performance/check-journey-ratchet.mjs --summary dist/performance/journey-summary.json",
"perf:tools:test": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs",
"perf:tools:test": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs tools/performance/tighten-baselines.test.mjs",
"agents:validate": "node tools/skills/validate-agent-guidance.mjs",
"skills:validate": "node tools/skills/validate-repository-skills.mjs",
"release:artwork:dry-run": "tsx --tsconfig tsconfig.base.json tools/release/generate-marketing-artwork.ts --dry-run",
+1 -1
View File
@@ -14,7 +14,7 @@
{ "externalDependencies": ["parse5"] }
],
"options": {
"command": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs",
"command": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs tools/performance/tighten-baselines.test.mjs",
"cwd": "{workspaceRoot}"
}
},
+367
View File
@@ -0,0 +1,367 @@
/**
* Lowers the baselines in tools/performance/journey-baselines.json that every
* measured run beat. The weekly .github/workflows/performance-ratchet.yml job
* measures `master` three times on separate runners and runs this script on
* the summaries; the contract is the Ratchet section of
* docs/architecture/performance-journeys.md.
*
* Rules:
* - At least three runs. A run is one summary file, or several (the initial
* bytes summary and the journeys summary of the same runner) merged.
* - An entry is tightened only when every run measured it and every
* measurement is strictly below `value`. The new value is the largest
* measurement: for a counter the largest count, for a wall-clock entry (one
* with `toleranceRatio`) the largest per-run summary value, which for a
* `.p50` entry is the largest P50.
* - A counter marked `counterStability.<name>.stable === false` in any run is
* skipped: its summary value hides iterations that disagreed.
* - `value` never goes up, `slack` and `toleranceRatio` never change, and no
* entry is added or removed; check-baseline-direction.mjs must accept the
* result without --allow-increase, and this script checks that itself.
* - A tightened entry gets `updatedAt`, `measuredWith` and `evidenceRun` (the
* workflow run URL). `evidencePr` is left alone; once the PR exists,
* `--fill-evidence-pr` sets it on the entries carrying that run URL.
*
* Usage:
* node tools/performance/tighten-baselines.mjs \
* --run <summary.json|run-dir> --run … --run … \
* --evidence-run <workflow run URL> \
* [--baselines tools/performance/journey-baselines.json] \
* [--report <report.md>] [--measured-with <text>] [--date YYYY-MM-DD]
* node tools/performance/tighten-baselines.mjs \
* --fill-evidence-pr <number> --evidence-run <workflow run URL> \
* [--baselines tools/performance/journey-baselines.json]
*
* A run directory contributes every `*.json` file directly inside it.
*/
import { readdir, readFile, stat, writeFile } from 'node:fs/promises';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { compareBaselineDirection } from './check-baseline-direction.mjs';
import {
DEFAULT_BASELINES_PATH,
validateBaselines,
} from './check-journey-ratchet.mjs';
export const MIN_RUNS = 3;
export const DEFAULT_MEASURED_WITH =
'.github/workflows/performance-ratchet.yml';
const SECTIONS = ['counters', 'wallClock', 'counterStability'];
function isPlainObject(value) {
return typeof value === 'object' && value !== null && !Array.isArray(value);
}
function formatNumber(value) {
return value.toLocaleString('en-US', { maximumFractionDigits: 2 });
}
/**
* Merges the summaries of one run into one `journeys` map. The same
* measurement in two files of one run is ambiguous and therefore an error.
*/
export function mergeRunSummaries(summaries, runLabel = 'run') {
const journeys = {};
for (const summary of summaries) {
if (!isPlainObject(summary?.journeys)) {
throw new Error(`${runLabel}: a summary has no "journeys" map.`);
}
for (const [journey, measured] of Object.entries(summary.journeys)) {
const target = (journeys[journey] ??= {
counters: {},
wallClock: {},
counterStability: {},
});
for (const section of SECTIONS) {
for (const [name, value] of Object.entries(
measured?.[section] ?? {}
)) {
if (Object.hasOwn(target[section], name)) {
throw new Error(
`${runLabel}: journeys.${journey}.${section}.${name} appears in two summaries of the same run.`
);
}
target[section][name] = value;
}
}
}
}
return { journeys };
}
function decide(entry, name, journey, runs) {
const counter = entry.toleranceRatio === undefined;
const section = counter ? 'counters' : 'wallClock';
const measured = runs.map(
(run) => run.journeys[journey]?.[section]?.[name]
);
for (const [index, value] of measured.entries()) {
if (value === undefined) {
return {
measured,
reason: `not measured in run ${index + 1} (journeys.${journey}.${section}).`,
};
}
if (typeof value !== 'number' || !Number.isFinite(value)) {
return {
measured,
reason: `run ${index + 1} measured ${JSON.stringify(value)}, not a finite number.`,
};
}
}
if (counter) {
for (const [index, run] of runs.entries()) {
const stability = run.journeys[journey]?.counterStability?.[name];
if (stability?.stable === false) {
const values = Array.isArray(stability.values)
? ` (iterations ${stability.values.join(', ')})`
: '';
return {
measured,
reason: `unstable in run ${index + 1}${values}; the summary value hides iterations that disagreed.`,
};
}
}
}
const highest = Math.max(...measured);
if (highest >= entry.value) {
return {
measured,
reason: `not below ${formatNumber(entry.value)} in every run (highest ${formatNumber(highest)}).`,
};
}
return { measured, newValue: highest };
}
/**
* Pure tightening. Returns the new baselines object and one row per entry so
* the CLI and the PR body can show the per-run numbers.
*/
export function tightenBaselines({
baselines,
runs,
evidenceRun,
measuredWith = DEFAULT_MEASURED_WITH,
date = new Date().toISOString().slice(0, 10),
}) {
validateBaselines(baselines);
if (runs.length < MIN_RUNS) {
throw new Error(
`Tightening needs at least ${MIN_RUNS} runs; received ${runs.length}.`
);
}
if (!evidenceRun) {
throw new Error('--evidence-run <workflow run URL> is required.');
}
const next = structuredClone(baselines);
const rows = [];
for (const [journey, entries] of Object.entries(baselines.journeys)) {
for (const [name, entry] of Object.entries(entries)) {
const decision = decide(entry, name, journey, runs);
rows.push({
label: `${journey}/${name}`,
unit: entry.unit,
value: entry.value,
...decision,
});
if (decision.newValue === undefined) continue;
next.journeys[journey][name] = {
...entry,
value: decision.newValue,
updatedAt: date,
measuredWith: `${measuredWith}, max of ${runs.length} runs`,
evidenceRun,
};
}
}
const direction = compareBaselineDirection({ base: baselines, head: next });
if (direction.failures.length > 0) {
throw new Error(
`Refusing to write a weakened baselines file: ${direction.failures.join(' ')}`
);
}
const tightened = rows.filter((row) => row.newValue !== undefined);
return { baselines: next, rows, tightened };
}
/** Sets `evidencePr` on the entries a tightening run wrote. */
export function fillEvidencePr({ baselines, evidenceRun, pr }) {
validateBaselines(baselines);
if (!Number.isInteger(pr) || pr <= 0) {
throw new Error(`--fill-evidence-pr expects a PR number, got ${pr}.`);
}
const next = structuredClone(baselines);
let filled = 0;
for (const entries of Object.values(next.journeys)) {
for (const entry of Object.values(entries)) {
if (entry.evidenceRun !== evidenceRun) continue;
entry.evidencePr = pr;
filled += 1;
}
}
if (filled === 0) {
throw new Error(`No baseline carries evidenceRun ${evidenceRun}.`);
}
return { baselines: next, filled };
}
/** Markdown: readable in a terminal, and pasted as is into the PR body. */
export function formatReport({ rows, tightened }, runNames = []) {
const runCount = Math.max(0, ...rows.map((row) => row.measured.length));
const runHeaders = Array.from(
{ length: runCount },
(_, index) => `Run ${index + 1}`
);
const lines = [
`| Baseline | Value | ${runHeaders.join(' | ')} | Result |`,
`| --- | ---: | ${runHeaders.map(() => '---:').join(' | ')} | --- |`,
];
for (const row of rows) {
const unit = row.unit ? ` ${row.unit}` : '';
const cells = row.measured.map((value) =>
typeof value === 'number' ? formatNumber(value) : '—'
);
const result =
row.newValue !== undefined
? `lowered to ${formatNumber(row.newValue)}${unit}`
: `kept: ${row.reason}`;
lines.push(
`| \`${row.label}\` | ${formatNumber(row.value)}${unit} | ${cells.join(' | ')} | ${result} |`
);
}
if (runNames.length > 0) {
lines.push('');
for (const [index, name] of runNames.entries()) {
lines.push(`- Run ${index + 1}: ${name}`);
}
}
lines.push(
'',
tightened.length > 0
? `Tightened ${tightened.length} of ${rows.length} baselines.`
: `No baseline was below its value in every run; nothing to tighten (${rows.length} checked).`
);
return lines.join('\n');
}
export function parseArgs(argv) {
const options = {
runs: [],
baselines: DEFAULT_BASELINES_PATH,
evidenceRun: null,
report: null,
measuredWith: DEFAULT_MEASURED_WITH,
date: undefined,
fillEvidencePr: null,
};
const valued = {
'--run': (value) => options.runs.push(value),
'--baselines': (value) => (options.baselines = value),
'--evidence-run': (value) => (options.evidenceRun = value),
'--report': (value) => (options.report = value),
'--measured-with': (value) => (options.measuredWith = value),
'--date': (value) => (options.date = value),
'--fill-evidence-pr': (value) =>
(options.fillEvidencePr = Number(value)),
};
for (let index = 0; index < argv.length; index += 1) {
const argument = argv[index];
if (argument === '--') continue;
const [flag, inline] = argument.split(/=(.*)/s, 2);
if (!(flag in valued)) throw new Error(`Unknown argument: ${argument}`);
const value = inline ?? argv[++index];
if (value === undefined || value === '') {
throw new Error(`Missing value for ${flag}`);
}
valued[flag](value);
}
if (
options.date !== undefined &&
!/^\d{4}-\d{2}-\d{2}$/.test(options.date)
) {
throw new Error(`--date expects YYYY-MM-DD, got "${options.date}".`);
}
if (!options.evidenceRun) {
throw new Error('--evidence-run <workflow run URL> is required.');
}
return options;
}
async function readJson(filePath) {
try {
return JSON.parse(await readFile(filePath, 'utf8'));
} catch (error) {
throw new Error(`Cannot read ${filePath}: ${error.message}`);
}
}
/** Reads one run: a summary file, or every `*.json` directly in a directory. */
export async function readRun(runPath, runLabel) {
let files = [runPath];
if ((await stat(runPath)).isDirectory()) {
files = (await readdir(runPath))
.filter((name) => name.endsWith('.json'))
.sort()
.map((name) => path.join(runPath, name));
if (files.length === 0) {
throw new Error(`${runLabel}: no summary files in ${runPath}.`);
}
}
const summaries = await Promise.all(files.map((file) => readJson(file)));
return mergeRunSummaries(summaries, runLabel);
}
async function writeJson(filePath, value) {
await writeFile(filePath, `${JSON.stringify(value, null, 4)}\n`);
}
const isMain =
process.argv[1] &&
path.resolve(process.argv[1]) ===
path.resolve(fileURLToPath(import.meta.url));
if (isMain) {
try {
const options = parseArgs(process.argv.slice(2));
const baselinesPath = path.resolve(options.baselines);
const baselines = await readJson(baselinesPath);
if (options.fillEvidencePr !== null) {
const { baselines: next, filled } = fillEvidencePr({
baselines,
evidenceRun: options.evidenceRun,
pr: options.fillEvidencePr,
});
await writeJson(baselinesPath, next);
console.log(
`Set evidencePr ${options.fillEvidencePr} on ${filled} baselines.`
);
} else {
const runs = await Promise.all(
options.runs.map((run, index) =>
readRun(path.resolve(run), `run ${index + 1}`)
)
);
const result = tightenBaselines({
baselines,
runs,
evidenceRun: options.evidenceRun,
measuredWith: options.measuredWith,
date: options.date,
});
const report = formatReport(
result,
options.runs.map((run) => path.basename(path.resolve(run)))
);
console.log(report);
if (options.report) await writeFile(options.report, `${report}\n`);
if (result.tightened.length > 0) {
await writeJson(baselinesPath, result.baselines);
}
}
} catch (error) {
console.error(`tighten-baselines: ${error.message}`);
process.exitCode = 1;
}
}
@@ -0,0 +1,396 @@
import assert from 'node:assert/strict';
import { spawnSync } from 'node:child_process';
import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises';
import os from 'node:os';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { after, before, test } from 'node:test';
import { compareBaselineDirection } from './check-baseline-direction.mjs';
import {
fillEvidencePr,
formatReport,
mergeRunSummaries,
parseArgs,
tightenBaselines,
} from './tighten-baselines.mjs';
const scriptPath = fileURLToPath(
new URL('./tighten-baselines.mjs', import.meta.url)
);
const RUN_URL = 'https://github.com/4gray/iptvnator/actions/runs/1';
const baselinesFile = (entries) => ({
version: 1,
journeys: {
launch: {
'renderer.initialBytes': {
value: 1000,
unit: 'bytes',
slack: 64,
updatedAt: '2026-09-27',
evidencePr: 1734,
measuredWith:
'pnpm nx build web && pnpm run perf:initial-bytes',
},
...entries,
},
},
});
const run = (launch) => ({ journeys: { launch } });
const bytesRun = (initialBytes) =>
run({ counters: { 'renderer.initialBytes': initialBytes } });
const tighten = (baselines, runs) =>
tightenBaselines({
baselines,
runs,
evidenceRun: RUN_URL,
date: '2026-10-05',
});
let workDir;
before(async () => {
workDir = await mkdtemp(path.join(os.tmpdir(), 'tighten-baselines-'));
});
after(async () => {
await rm(workDir, { recursive: true, force: true });
});
test('a counter below its value in every run drops to the largest run', () => {
const baselines = baselinesFile();
const result = tighten(baselines, [
bytesRun(990),
bytesRun(995),
bytesRun(980),
]);
assert.deepEqual(
result.baselines.journeys.launch['renderer.initialBytes'],
{
value: 995,
unit: 'bytes',
slack: 64,
updatedAt: '2026-10-05',
evidencePr: 1734,
measuredWith:
'.github/workflows/performance-ratchet.yml, max of 3 runs',
evidenceRun: RUN_URL,
}
);
assert.equal(result.tightened.length, 1);
assert.equal(
baselines.journeys.launch['renderer.initialBytes'].value,
1000,
'the input is not mutated'
);
const direction = compareBaselineDirection({
base: baselines,
head: result.baselines,
});
assert.deepEqual(direction.failures, []);
assert.deepEqual(direction.allowed, []);
assert.match(direction.lowered[0], /1,064 -> 1,059 bytes/);
});
test('one run at or above the value keeps the baseline untouched', () => {
for (const measured of [1000, 1010]) {
const baselines = baselinesFile();
const result = tighten(baselines, [
bytesRun(990),
bytesRun(measured),
bytesRun(980),
]);
assert.deepEqual(result.baselines, baselines);
assert.deepEqual(result.tightened, []);
assert.match(result.rows[0].reason, /not below 1,000 in every run/);
}
});
test('an entry missing or non-numeric in any run is kept', () => {
const missing = tighten(baselinesFile(), [
bytesRun(990),
run({ counters: {} }),
bytesRun(980),
]);
assert.deepEqual(missing.tightened, []);
assert.match(missing.rows[0].reason, /not measured in run 2/);
const invalid = tighten(baselinesFile(), [
bytesRun(990),
bytesRun(980),
bytesRun('980'),
]);
assert.deepEqual(invalid.tightened, []);
assert.match(invalid.rows[0].reason, /run 3 measured "980"/);
});
test('a counter marked unstable in any run is skipped with the iterations', () => {
const baselines = baselinesFile({
'renderer.ipcCallsToFirstCard': { value: 20 },
});
const ipcRun = (value, stable = true) =>
run({
counters: {
'renderer.initialBytes': 990,
'renderer.ipcCallsToFirstCard': value,
},
counterStability: {
'renderer.ipcCallsToFirstCard': {
stable,
values: stable ? [value, value] : [13, value],
},
},
});
const result = tighten(baselines, [
ipcRun(16),
ipcRun(16, false),
ipcRun(16),
]);
const row = result.rows.find(
(entry) => entry.label === 'launch/renderer.ipcCallsToFirstCard'
);
assert.match(row.reason, /unstable in run 2 \(iterations 13, 16\)/);
assert.equal(
result.baselines.journeys.launch['renderer.ipcCallsToFirstCard'].value,
20
);
assert.equal(
result.baselines.journeys.launch['renderer.initialBytes'].value,
990,
'other entries are still tightened'
);
});
test('a wall-clock entry drops to the largest P50 and keeps its tolerance', () => {
const baselines = baselinesFile({
'spawnToFirstCardMs.p50': {
value: 1600,
unit: 'ms',
toleranceRatio: 1.25,
},
});
const wallRun = (p50) =>
run({
counters: { 'renderer.initialBytes': 1000 },
wallClock: {
'spawnToFirstCardMs.p50': p50,
'spawnToFirstCardMs.p90': 5000,
},
// counterStability does not apply to wall-clock entries.
counterStability: {
'spawnToFirstCardMs.p50': { stable: false },
},
});
const result = tighten(baselines, [
wallRun(1401.5),
wallRun(1550.25),
wallRun(1480),
]);
assert.deepEqual(
result.baselines.journeys.launch['spawnToFirstCardMs.p50'],
{
value: 1550.25,
unit: 'ms',
toleranceRatio: 1.25,
updatedAt: '2026-10-05',
measuredWith:
'.github/workflows/performance-ratchet.yml, max of 3 runs',
evidenceRun: RUN_URL,
}
);
assert.deepEqual(
compareBaselineDirection({ base: baselines, head: result.baselines })
.failures,
[]
);
});
test('fewer than three runs or no run URL is refused', () => {
assert.throws(
() => tighten(baselinesFile(), [bytesRun(990), bytesRun(990)]),
/at least 3 runs; received 2/
);
assert.throws(
() =>
tightenBaselines({
baselines: baselinesFile(),
runs: [bytesRun(1), bytesRun(1), bytesRun(1)],
}),
/--evidence-run/
);
});
test('entries are never added and values are never raised', () => {
const baselines = baselinesFile();
const result = tighten(baselines, [
run({ counters: { 'renderer.initialBytes': 1200, newCounter: 1 } }),
bytesRun(1200),
bytesRun(1200),
]);
assert.deepEqual(result.baselines, baselines);
});
test('summaries of one run merge; a duplicate measurement is an error', () => {
const merged = mergeRunSummaries([
bytesRun(990),
run({
counters: { 'renderer.ipcCallsToFirstCard': 16 },
wallClock: { 'spawnToFirstCardMs.p50': 1400 },
}),
]);
assert.deepEqual(merged.journeys.launch.counters, {
'renderer.initialBytes': 990,
'renderer.ipcCallsToFirstCard': 16,
});
assert.throws(
() => mergeRunSummaries([bytesRun(990), bytesRun(991)], 'run 2'),
/run 2: journeys\.launch\.counters\.renderer\.initialBytes appears in two summaries/
);
});
test('fillEvidencePr sets the PR only on entries from that run', () => {
const baselines = baselinesFile({
other: { value: 5, evidenceRun: 'https://example.invalid/runs/0' },
});
baselines.journeys.launch['renderer.initialBytes'].evidenceRun = RUN_URL;
const { baselines: next, filled } = fillEvidencePr({
baselines,
evidenceRun: RUN_URL,
pr: 1800,
});
assert.equal(filled, 1);
assert.equal(
next.journeys.launch['renderer.initialBytes'].evidencePr,
1800
);
assert.equal(next.journeys.launch.other.evidencePr, undefined);
assert.throws(
() => fillEvidencePr({ baselines, evidenceRun: 'x', pr: 1 }),
/No baseline carries evidenceRun x/
);
});
test('the report lists every run value and the decision', () => {
const result = tighten(baselinesFile({ gone: { value: 3 } }), [
bytesRun(990),
bytesRun(995),
bytesRun(980),
]);
const report = formatReport(result, ['a', 'b', 'c']);
assert.match(
report,
/\| `launch\/renderer.initialBytes` \| 1,000 bytes \| 990 \| 995 \| 980 \| lowered to 995 bytes \|/
);
assert.match(
report,
/\| `launch\/gone` \| 3 \| — \| — \| — \| kept: not measured in run 1/
);
assert.match(report, /- Run 2: b/);
assert.match(report, /Tightened 1 of 2 baselines\./);
});
test('parseArgs reads repeated runs and requires the run URL', () => {
assert.deepEqual(
parseArgs([
'--run',
'a.json',
'--run=b',
'--evidence-run',
RUN_URL,
'--date=2026-10-05',
]).runs,
['a.json', 'b']
);
assert.throws(() => parseArgs(['--run', 'a.json']), /--evidence-run/);
assert.throws(
() => parseArgs(['--evidence-run', RUN_URL, '--date', '5.10.2026']),
/YYYY-MM-DD/
);
assert.throws(() => parseArgs(['--bogus']), /Unknown argument/);
assert.throws(() => parseArgs(['--run']), /Missing value for --run/);
});
test('CLI rewrites the file only when a baseline was tightened', async () => {
const baselinesPath = path.join(workDir, 'baselines.json');
const original = `${JSON.stringify(baselinesFile(), null, 4)}\n`;
await writeFile(baselinesPath, original);
const runArgs = [];
for (const [index, bytes] of [990, 1000, 985].entries()) {
const dir = path.join(workDir, `run-${index + 1}`);
await mkdir(dir, { recursive: true });
await writeFile(
path.join(dir, 'initial-bytes.summary.json'),
JSON.stringify(bytesRun(bytes))
);
await writeFile(
path.join(dir, 'journeys.summary.json'),
JSON.stringify(run({ counters: { 'renderer.longTasks': 2 } }))
);
runArgs.push('--run', dir);
}
const cli = (...extra) =>
spawnSync(
process.execPath,
[
scriptPath,
...runArgs,
'--baselines',
baselinesPath,
'--evidence-run',
RUN_URL,
...extra,
],
{ encoding: 'utf8' }
);
const unchanged = cli();
assert.equal(unchanged.status, 0, unchanged.stderr);
assert.match(unchanged.stdout, /nothing to tighten/);
assert.equal(await readFile(baselinesPath, 'utf8'), original);
await writeFile(
path.join(workDir, 'run-2', 'initial-bytes.summary.json'),
JSON.stringify(bytesRun(970))
);
const reportPath = path.join(workDir, 'report.md');
const tightened = cli('--report', reportPath, '--date', '2026-10-05');
assert.equal(tightened.status, 0, tightened.stderr);
const written = JSON.parse(await readFile(baselinesPath, 'utf8'));
assert.equal(written.journeys.launch['renderer.initialBytes'].value, 990);
assert.match(await readFile(reportPath, 'utf8'), /lowered to 990 bytes/);
const fill = spawnSync(
process.execPath,
[
scriptPath,
'--baselines',
baselinesPath,
'--evidence-run',
RUN_URL,
'--fill-evidence-pr',
'1800',
],
{ encoding: 'utf8' }
);
assert.equal(fill.status, 0, fill.stderr);
const filled = JSON.parse(await readFile(baselinesPath, 'utf8'));
assert.equal(
filled.journeys.launch['renderer.initialBytes'].evidencePr,
1800
);
const tooFew = spawnSync(
process.execPath,
[
scriptPath,
'--run',
path.join(workDir, 'run-1'),
'--evidence-run',
RUN_URL,
],
{ encoding: 'utf8' }
);
assert.equal(tooFew.status, 1);
assert.match(tooFew.stderr, /at least 3 runs; received 1/);
});