From ec8b931dbf8deba37272bce9b18dc43b5bbd480e Mon Sep 17 00:00:00 2001 From: 4gray <4gray@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:50:30 +0200 Subject: [PATCH] ci(perf): add the weekly baseline tightening workflow (#1760) --- .../actions/performance-journeys/action.yml | 74 ++++ .github/workflows/ci.yml | 57 +-- .github/workflows/performance-ratchet.yml | 246 +++++++++++ docs/architecture/performance-journeys.md | 49 ++- docs/architecture/validation-map.md | 7 +- package.json | 2 +- tools/performance/project.json | 2 +- tools/performance/tighten-baselines.mjs | 367 ++++++++++++++++ tools/performance/tighten-baselines.test.mjs | 396 ++++++++++++++++++ 9 files changed, 1142 insertions(+), 58 deletions(-) create mode 100644 .github/actions/performance-journeys/action.yml create mode 100644 .github/workflows/performance-ratchet.yml create mode 100644 tools/performance/tighten-baselines.mjs create mode 100644 tools/performance/tighten-baselines.test.mjs diff --git a/.github/actions/performance-journeys/action.yml b/.github/actions/performance-journeys/action.yml new file mode 100644 index 000000000..009675585 --- /dev/null +++ b/.github/actions/performance-journeys/action.yml @@ -0,0 +1,74 @@ +name: Run the performance journeys +description: >- + Runs `pnpm run perf:journeys` under xvfb on an Ubuntu runner, writes the + measurements to the job summary and returns the path of the summary.json + it wrote. Callers check out, set up pnpm and Node, install dependencies + and upload what they need. Contract: docs/architecture/performance-journeys.md. + +outputs: + summary: + description: Path of the journey summary.json this run wrote. + value: ${{ steps.summary.outputs.path }} + +runs: + using: composite + steps: + # The journeys drive Electron through Playwright's _electron API + # and never launch a Playwright browser, so no `playwright + # install`. The runner image ships Electron's shared libraries and + # xvfb; fail fast with a clear message if an image update drops one. + # pnpm skips Electron's postinstall, so download the binary first; + # otherwise ldd sees no file and the check passes vacuously. + - name: Check Electron runtime dependencies + shell: bash + run: | + command -v xvfb-run || { echo "::error::xvfb-run is missing on the runner"; exit 1; } + node tools/testing/ensure-electron-binary.mjs || { echo "::error::Electron binary download failed"; exit 1; } + missing="$(ldd node_modules/electron/dist/electron | grep 'not found' || true)" + if [ -n "$missing" ]; then + echo "::error::Electron is missing shared libraries:" + echo "$missing" + exit 1 + fi + + # The Nx target builds electron-backend:build-performance first + # and playwright.journeys.config.ts starts the Xtream mock server. + - name: Run the performance journeys + shell: bash + run: xvfb-run --auto-servernum --server-args="-screen 0 1280x960x24" pnpm run perf:journeys + env: + CI: 'true' + NX_TASKS_RUNNER_DYNAMIC_OUTPUT: 'false' + + # Every run writes a fresh timestamped directory, so a clean + # checkout must hold exactly one summary. + - name: Locate the journey summary + id: summary + shell: bash + run: | + set -euo pipefail + mapfile -t summaries < <(find dist/performance/journeys -mindepth 2 -maxdepth 2 -name summary.json) + if [ "${#summaries[@]}" -ne 1 ]; then + echo "::error::Expected one journey summary, found ${#summaries[@]}" + exit 1 + fi + echo "path=${summaries[0]}" >> "$GITHUB_OUTPUT" + + - name: Report journey measurements + shell: bash + env: + SUMMARY: ${{ steps.summary.outputs.path }} + run: | + set -euo pipefail + jq -r ' + "## Performance journeys (\(.harness.platform), \(.harness.measuredIterations) measured iterations)", "", + (.journeys | to_entries[] | .key as $journey | .value as $j | + "### `\($journey)`", "", + "| Measurement | Value | Iterations |", + "| --- | ---: | --- |", + (($j.counters // {}) | to_entries[] | + ($j.counterStability[.key] // {}) as $s | + "| `\(.key)` | \(.value) | \(($s.values // []) | map(tostring) | join(", "))\(if $s.stable == false then " (unstable)" else "" end) |"), + (($j.wallClock // {}) | to_entries[] | "| `\(.key)` | \(.value) | |"), + "") + ' "$SUMMARY" | tee -a "$GITHUB_STEP_SUMMARY" diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 07c0fd2b7..b05476d9c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -321,61 +321,10 @@ jobs: - name: Install dependencies run: pnpm install --frozen-lockfile - # The journeys drive Electron through Playwright's _electron API - # and never launch a Playwright browser, so no `playwright - # install`. The runner image ships Electron's shared libraries and - # xvfb; fail fast with a clear message if an image update drops one. - # pnpm skips Electron's postinstall, so download the binary first; - # otherwise ldd sees no file and the check passes vacuously. - - name: Check Electron runtime dependencies - run: | - command -v xvfb-run || { echo "::error::xvfb-run is missing on the runner"; exit 1; } - node tools/testing/ensure-electron-binary.mjs || { echo "::error::Electron binary download failed"; exit 1; } - missing="$(ldd node_modules/electron/dist/electron | grep 'not found' || true)" - if [ -n "$missing" ]; then - echo "::error::Electron is missing shared libraries:" - echo "$missing" - exit 1 - fi - - # The Nx target builds electron-backend:build-performance first - # and playwright.journeys.config.ts starts the Xtream mock server. + # Electron dependency check, the xvfb run, the summary lookup and + # the job-summary report; shared with performance-ratchet.yml. - name: Run the performance journeys - run: xvfb-run --auto-servernum --server-args="-screen 0 1280x960x24" pnpm run perf:journeys - env: - CI: true - NX_TASKS_RUNNER_DYNAMIC_OUTPUT: false - - # Every run writes a fresh timestamped directory, so a clean - # checkout must hold exactly one summary. - - name: Locate the journey summary - id: summary - run: | - set -euo pipefail - mapfile -t summaries < <(find dist/performance/journeys -mindepth 2 -maxdepth 2 -name summary.json) - if [ "${#summaries[@]}" -ne 1 ]; then - echo "::error::Expected one journey summary, found ${#summaries[@]}" - exit 1 - fi - echo "path=${summaries[0]}" >> "$GITHUB_OUTPUT" - - - name: Report journey measurements - env: - SUMMARY: ${{ steps.summary.outputs.path }} - run: | - set -euo pipefail - jq -r ' - "## Performance journeys (\(.harness.platform), \(.harness.measuredIterations) measured iterations)", "", - (.journeys | to_entries[] | .key as $journey | .value as $j | - "### `\($journey)`", "", - "| Measurement | Value | Iterations |", - "| --- | ---: | --- |", - (($j.counters // {}) | to_entries[] | - ($j.counterStability[.key] // {}) as $s | - "| `\(.key)` | \(.value) | \(($s.values // []) | map(tostring) | join(", "))\(if $s.stable == false then " (unstable)" else "" end) |"), - (($j.wallClock // {}) | to_entries[] | "| `\(.key)` | \(.value) | |"), - "") - ' "$SUMMARY" | tee -a "$GITHUB_STEP_SUMMARY" + uses: ./.github/actions/performance-journeys - name: Upload journey summaries if: always() diff --git a/.github/workflows/performance-ratchet.yml b/.github/workflows/performance-ratchet.yml new file mode 100644 index 000000000..39e3a0c4d --- /dev/null +++ b/.github/workflows/performance-ratchet.yml @@ -0,0 +1,246 @@ +name: Performance ratchet + +# Weekly tightening of tools/performance/journey-baselines.json (plan item B4; +# contract in the Ratchet section of docs/architecture/performance-journeys.md). +# Three runners measure the same master commit independently; a pull request +# lowers every baseline that all three beat. Nothing is ever raised, and slack +# and tolerances are left alone. A manual dispatch on another branch measures +# and reports but never opens a pull request. + +on: + schedule: + - cron: '41 4 * * 1' + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: performance-ratchet-${{ github.ref }} + cancel-in-progress: false + +jobs: + measure: + name: Measure (run ${{ matrix.run }}) + if: github.repository == '4gray/iptvnator' + runs-on: ubuntu-latest + # Production web build plus the journeys (their electron-performance + # build and seven Electron processes per journey). + timeout-minutes: 60 + strategy: + # Separate runners, so one machine's noise cannot pass for a + # reduction. The tightener needs all three runs. + fail-fast: true + matrix: + run: [1, 2, 3] + env: + NX_SKIP_NX_CACHE: true + + steps: + - name: Checkout code + uses: actions/checkout@v7 + with: + persist-credentials: false + + - name: Install pnpm + uses: pnpm/action-setup@v6.0.10 + + - name: Setup Node.js + uses: actions/setup-node@v7 + with: + node-version-file: '.nvmrc' + cache: 'pnpm' + + - name: Install dependencies + run: pnpm install --frozen-lockfile + + # Same build as the Initial bytes ratchet job in ci.yml. Measured + # before the journeys, whose build also writes to dist/. + - name: Build web app (production) + run: pnpm nx build web --skip-nx-cache + env: + CI: true + NX_TASKS_RUNNER_DYNAMIC_OUTPUT: false + + - name: Measure renderer.initialBytes + run: node tools/performance/measure-initial-bytes.mjs --summary dist/performance/ratchet/initial-bytes.summary.json + + # A failed journey run must not block tightening the initial bytes: + # its entries are then missing from this run's summary, and the + # tightener keeps any baseline that a run did not measure. + - name: Run the performance journeys + id: journeys + continue-on-error: true + uses: ./.github/actions/performance-journeys + + - name: Keep the journey summary + if: steps.journeys.outputs.summary != '' + env: + SUMMARY: ${{ steps.journeys.outputs.summary }} + run: cp "$SUMMARY" dist/performance/ratchet/journeys.summary.json + + - name: Upload run summaries + uses: actions/upload-artifact@v7 + with: + name: performance-ratchet-run-${{ matrix.run }} + path: dist/performance/ratchet/ + if-no-files-found: error + retention-days: 30 + + tighten: + name: Tighten baselines + needs: measure + runs-on: ubuntu-latest + timeout-minutes: 15 + + steps: + # This job later receives secrets.PAT, so its actions are pinned + # to reviewed commits, as in refresh-windows-embedded-mpv-runtime. + - name: Checkout code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + # The ratchet scripts are dependency-free Node; no pnpm install. + - name: Setup Node.js + uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 + with: + node-version-file: '.nvmrc' + + - name: Download run summaries + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: performance-ratchet-run-* + path: ${{ runner.temp }}/runs + + # Each run directory holds that runner's initial-bytes and journeys + # summaries. The direction check is the same one ci.yml runs on the + # PR, without --allow-increase: a tightening can only lower limits. + - name: Lower baselines every run beat + id: tighten + env: + RUNS_DIR: ${{ runner.temp }}/runs + REPORT: ${{ runner.temp }}/tighten-report.md + BASE_BASELINES: ${{ runner.temp }}/base-journey-baselines.json + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + run: | + set -euo pipefail + baselines=tools/performance/journey-baselines.json + cp "$baselines" "$BASE_BASELINES" + runs=() + for dir in "$RUNS_DIR"/performance-ratchet-run-*; do + runs+=(--run "$dir") + done + node tools/performance/tighten-baselines.mjs \ + "${runs[@]}" \ + --evidence-run "$RUN_URL" \ + --report "$REPORT" + { + echo "## Performance ratchet" + echo + cat "$REPORT" + } >> "$GITHUB_STEP_SUMMARY" + node tools/performance/check-baseline-direction.mjs \ + --base "$BASE_BASELINES" \ + --head "$baselines" + if git diff --quiet -- "$baselines"; then + echo "changed=false" >> "$GITHUB_OUTPUT" + else + git diff -- "$baselines" + echo "changed=true" >> "$GITHUB_OUTPUT" + fi + + - name: Report a dispatch outside master + if: steps.tighten.outputs.changed == 'true' && github.ref != 'refs/heads/master' + env: + REF_NAME: ${{ github.ref_name }} + run: echo "Dispatched on $REF_NAME, not master; the diff above is not proposed as a pull request." + + # PAT, not GITHUB_TOKEN: a pull request pushed with GITHUB_TOKEN + # triggers no workflows, and this one needs ci.yml's Initial bytes + # ratchet run. Same secret as refresh-windows-embedded-mpv-runtime. + # The PR number is only known once the PR exists, so evidencePr is + # filled in afterwards and the single commit amended. Each run + # replaces the branch with one fresh commit, but never over an open + # tightening PR that someone else committed to (review edits, an + # "Update branch" merge); explicit leases also refuse a push that + # lands between that check and ours. + - name: Create or update the tightening pull request + if: steps.tighten.outputs.changed == 'true' && github.ref == 'refs/heads/master' + env: + GH_TOKEN: ${{ secrets.PAT }} + REPOSITORY: ${{ github.repository }} + BRANCH_NAME: automation/performance-ratchet + TITLE: 'ci(perf): tighten journey baselines' + LABEL: no-release-note + REPORT: ${{ runner.temp }}/tighten-report.md + BODY: ${{ runner.temp }}/pr-body.md + HEAD_SHA: ${{ github.sha }} + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + run: | + set -euo pipefail + if [ -z "${GH_TOKEN}" ]; then + echo "::error::The PAT secret is required so the bot-created PR triggers normal CI." + exit 1 + fi + baselines=tools/performance/journey-baselines.json + + bot_email="41898282+github-actions[bot]@users.noreply.github.com" + + gh auth setup-git + git config user.name "github-actions[bot]" + git config user.email "$bot_email" + + pr="$(gh api "repos/$REPOSITORY/pulls?base=master&head=${REPOSITORY%%/*}:$BRANCH_NAME&state=open" --jq '.[0].number // empty')" + remote_tip="$(git ls-remote --heads origin "refs/heads/$BRANCH_NAME" | cut -f1)" + if [ -n "$pr" ] && [ -n "$remote_tip" ]; then + foreign="$(gh api "repos/$REPOSITORY/compare/master...$remote_tip" | + jq -r --arg bot "$bot_email" '.commits[] | select(.commit.author.email != $bot) | "\(.sha[0:9]) \(.commit.author.name)"')" + if [ -n "$foreign" ]; then + echo "::warning::Pull request #$pr has commits this workflow did not make; its branch is left as is. Merge or close it so the next run can refresh it. This run's numbers are in the job summary." + echo "$foreign" + exit 0 + fi + fi + if [ -n "$remote_tip" ]; then + git fetch --no-tags --depth=1 origin "$remote_tip" + fi + + git switch -C "$BRANCH_NAME" + git add -- "$baselines" + git commit -m "$TITLE" -m "Measured by $RUN_URL" + git push --force-with-lease="refs/heads/$BRANCH_NAME:$remote_tip" origin "HEAD:refs/heads/$BRANCH_NAME" + pushed="$(git rev-parse HEAD)" + + if [ -z "$pr" ]; then + pr="$(gh api -X POST "repos/$REPOSITORY/pulls" \ + -f title="$TITLE" -f head="$BRANCH_NAME" -f base=master \ + -f body="Measurements follow." --jq '.number')" + fi + gh api -X POST "repos/$REPOSITORY/issues/$pr/labels" -f "labels[]=$LABEL" --silent + + node tools/performance/tighten-baselines.mjs \ + --fill-evidence-pr "$pr" --evidence-run "$RUN_URL" + git add -- "$baselines" + git commit --amend --no-edit + git push --force-with-lease="refs/heads/$BRANCH_NAME:$pushed" origin "HEAD:refs/heads/$BRANCH_NAME" + + { + echo "Automated weekly tightening of \`$baselines\` from [three runner measurements]($RUN_URL) of \`$HEAD_SHA\`." + echo + echo "Each lowered baseline was strictly below its value in all three runs and now holds the largest of the three. Slack and tolerances are unchanged; \`check-baseline-direction.mjs\` accepted the file without \`--allow-increase\`." + echo + echo "## Per-run measurements" + echo + cat "$REPORT" + echo + echo "## Diff" + echo + echo '```diff' + git diff "$HEAD_SHA" HEAD -- "$baselines" + echo '```' + echo + echo "If \`master\` moved since \`$HEAD_SHA\`, make sure the Initial bytes ratchet job passes on this PR before merging." + } > "$BODY" + gh api -X PATCH "repos/$REPOSITORY/pulls/$pr" -F "body=@$BODY" --silent + echo "Pull request: ${GITHUB_SERVER_URL}/$REPOSITORY/pull/$pr" diff --git a/docs/architecture/performance-journeys.md b/docs/architecture/performance-journeys.md index cf3a0d660..66fc6c15d 100644 --- a/docs/architecture/performance-journeys.md +++ b/docs/architecture/performance-journeys.md @@ -628,7 +628,9 @@ the job for every PR until #1734. Treat the job as blocking before merging; making it a required check is a maintainer decision. The runtime counters come from the `Performance journeys` job of the same -workflow, on `ubuntu-latest` only. It runs `pnpm run perf:journeys` under +workflow, on `ubuntu-latest` only. Through the +`.github/actions/performance-journeys` composite action it runs +`pnpm run perf:journeys` under `xvfb-run` (the Nx target builds `electron-backend:build-performance`, the Playwright config starts the Xtream mock), writes the measurements to the job summary and uploads `dist/performance/journeys/` as the `performance-journeys` @@ -661,6 +663,51 @@ it as `stable: false` because the dashboard flicker it reports is a race there (see [Settle window](#settle-window)). Add the runner's number once that flicker is fixed and the counter is deterministic. +### Weekly tightening + +`.github/workflows/performance-ratchet.yml` lowers baselines without waiting +for someone to act on a "tighten" hint. Every Monday, and on +`workflow_dispatch`, three `ubuntu-latest` jobs measure the same commit +independently: the production `apps/web` build with +`measure-initial-bytes.mjs --summary`, then the journeys through the +`.github/actions/performance-journeys` composite action, which the +`Performance journeys` job above uses too. A failed journey run does not stop +its job; its entries are then unmeasured in that run. A final job runs +`tools/performance/tighten-baselines.mjs` on the three runs: + +- an entry is lowered only when every run measured it and every measurement + is strictly below `value`; the new `value` is the largest of the three (for + a wall-clock entry the largest per-run summary value, which is the largest + P50 for a `.p50` entry); +- a counter marked `counterStability..stable: false` in any run is + kept, and the report says which run and which iterations disagreed; +- `value` never goes up, `slack` and `toleranceRatio` never change, and no + entry is added or removed: the result must pass + `check-baseline-direction.mjs` without `--allow-increase`, which both the + script and the job check; +- a lowered entry gets `updatedAt`, `measuredWith` and `evidenceRun` (the + workflow run URL); `evidencePr` is set to the tightening PR once it exists. + +When the file changed and the run is on `master`, the job pushes +`automation/performance-ratchet` and opens (or updates) a pull request with +the per-run table, the diff and the run URL, labelled `no-release-note`. It +pushes with the existing `PAT` secret, as the Windows MPV pin refresh does, +because a pull request pushed with `GITHUB_TOKEN` starts no CI. Each run +replaces the branch with one fresh commit, except when the open tightening +pull request carries a commit the workflow did not make (a review edit, an +"Update branch" merge): then it leaves the branch alone with a warning, and +the numbers stay in the job summary. When no +baseline was below its value in all three runs, the workflow ends without a +pull request. A dispatch on another branch measures and prints the diff but +never opens one, so `gh workflow run performance-ratchet.yml --ref ` +validates a change to the workflow once the file is on `master`. GitHub only +dispatches workflows that exist on the default branch, so before the first +merge of a new or renamed workflow add a temporary `push` trigger for the +branch and drop it before review, as #1760 did. Review the pull request like a manual +tightening: if `master` moved since the measured commit, the +`Initial bytes ratchet` job on the pull request is what shows that the new +value still holds (the concurrent-merge effect above). + ## Charset parse benchmark V8 stores a string as two-byte UTF-16 once one character falls outside diff --git a/docs/architecture/validation-map.md b/docs/architecture/validation-map.md index 7c5ff1fd3..46da72dd8 100644 --- a/docs/architecture/validation-map.md +++ b/docs/architecture/validation-map.md @@ -260,7 +260,12 @@ bytes on the initial path (the J1 counter `renderer.initialBytes`). `tools/performance/journey-baselines.json`; baselines only move down. CI runs the same check in the `Initial bytes ratchet` job of `ci.yml` for PRs that target `master` and for `master` pushes (dispatch it with -`gh workflow run ci.yml --ref ` for a stacked branch). `perf:journeys` builds the `electron-performance` configuration and runs every +`gh workflow run ci.yml --ref ` for a stacked branch). The weekly +`performance-ratchet.yml` workflow lowers baselines through a bot PR; validate +a change to it with `gh workflow run performance-ratchet.yml --ref `, +which measures but opens no PR off `master`. Dispatch needs the workflow file +on `master`; before that, see the temporary-trigger note under Weekly +tightening in the performance journeys document. `perf:journeys` builds the `electron-performance` configuration and runs every journey spec against the Xtream mock: J1 launch, then J2 open-source (a second set of launches, each followed by the click on the portal card), both written to the same summary file; its probe specs run with diff --git a/package.json b/package.json index c5c2fb931..b9d00257f 100644 --- a/package.json +++ b/package.json @@ -77,7 +77,7 @@ "perf:initial-bytes:check": "node tools/performance/measure-initial-bytes.mjs --summary dist/performance/initial-bytes.summary.json && node tools/performance/check-journey-ratchet.mjs --summary dist/performance/initial-bytes.summary.json --only launch/renderer.initialBytes", "perf:journeys": "nx run electron-backend-e2e:journeys", "perf:ratchet:check": "node tools/performance/check-journey-ratchet.mjs --summary dist/performance/journey-summary.json", - "perf:tools:test": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs", + "perf:tools:test": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs tools/performance/tighten-baselines.test.mjs", "agents:validate": "node tools/skills/validate-agent-guidance.mjs", "skills:validate": "node tools/skills/validate-repository-skills.mjs", "release:artwork:dry-run": "tsx --tsconfig tsconfig.base.json tools/release/generate-marketing-artwork.ts --dry-run", diff --git a/tools/performance/project.json b/tools/performance/project.json index ac5ff1f95..106390896 100644 --- a/tools/performance/project.json +++ b/tools/performance/project.json @@ -14,7 +14,7 @@ { "externalDependencies": ["parse5"] } ], "options": { - "command": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs", + "command": "node --test tools/performance/measure-initial-bytes.test.mjs tools/performance/check-journey-ratchet.test.mjs tools/performance/check-baseline-direction.test.mjs tools/performance/tighten-baselines.test.mjs", "cwd": "{workspaceRoot}" } }, diff --git a/tools/performance/tighten-baselines.mjs b/tools/performance/tighten-baselines.mjs new file mode 100644 index 000000000..0c82bf09e --- /dev/null +++ b/tools/performance/tighten-baselines.mjs @@ -0,0 +1,367 @@ +/** + * Lowers the baselines in tools/performance/journey-baselines.json that every + * measured run beat. The weekly .github/workflows/performance-ratchet.yml job + * measures `master` three times on separate runners and runs this script on + * the summaries; the contract is the Ratchet section of + * docs/architecture/performance-journeys.md. + * + * Rules: + * - At least three runs. A run is one summary file, or several (the initial + * bytes summary and the journeys summary of the same runner) merged. + * - An entry is tightened only when every run measured it and every + * measurement is strictly below `value`. The new value is the largest + * measurement: for a counter the largest count, for a wall-clock entry (one + * with `toleranceRatio`) the largest per-run summary value, which for a + * `.p50` entry is the largest P50. + * - A counter marked `counterStability..stable === false` in any run is + * skipped: its summary value hides iterations that disagreed. + * - `value` never goes up, `slack` and `toleranceRatio` never change, and no + * entry is added or removed; check-baseline-direction.mjs must accept the + * result without --allow-increase, and this script checks that itself. + * - A tightened entry gets `updatedAt`, `measuredWith` and `evidenceRun` (the + * workflow run URL). `evidencePr` is left alone; once the PR exists, + * `--fill-evidence-pr` sets it on the entries carrying that run URL. + * + * Usage: + * node tools/performance/tighten-baselines.mjs \ + * --run --run … --run … \ + * --evidence-run \ + * [--baselines tools/performance/journey-baselines.json] \ + * [--report ] [--measured-with ] [--date YYYY-MM-DD] + * node tools/performance/tighten-baselines.mjs \ + * --fill-evidence-pr --evidence-run \ + * [--baselines tools/performance/journey-baselines.json] + * + * A run directory contributes every `*.json` file directly inside it. + */ +import { readdir, readFile, stat, writeFile } from 'node:fs/promises'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; + +import { compareBaselineDirection } from './check-baseline-direction.mjs'; +import { + DEFAULT_BASELINES_PATH, + validateBaselines, +} from './check-journey-ratchet.mjs'; + +export const MIN_RUNS = 3; +export const DEFAULT_MEASURED_WITH = + '.github/workflows/performance-ratchet.yml'; + +const SECTIONS = ['counters', 'wallClock', 'counterStability']; + +function isPlainObject(value) { + return typeof value === 'object' && value !== null && !Array.isArray(value); +} + +function formatNumber(value) { + return value.toLocaleString('en-US', { maximumFractionDigits: 2 }); +} + +/** + * Merges the summaries of one run into one `journeys` map. The same + * measurement in two files of one run is ambiguous and therefore an error. + */ +export function mergeRunSummaries(summaries, runLabel = 'run') { + const journeys = {}; + for (const summary of summaries) { + if (!isPlainObject(summary?.journeys)) { + throw new Error(`${runLabel}: a summary has no "journeys" map.`); + } + for (const [journey, measured] of Object.entries(summary.journeys)) { + const target = (journeys[journey] ??= { + counters: {}, + wallClock: {}, + counterStability: {}, + }); + for (const section of SECTIONS) { + for (const [name, value] of Object.entries( + measured?.[section] ?? {} + )) { + if (Object.hasOwn(target[section], name)) { + throw new Error( + `${runLabel}: journeys.${journey}.${section}.${name} appears in two summaries of the same run.` + ); + } + target[section][name] = value; + } + } + } + } + return { journeys }; +} + +function decide(entry, name, journey, runs) { + const counter = entry.toleranceRatio === undefined; + const section = counter ? 'counters' : 'wallClock'; + const measured = runs.map( + (run) => run.journeys[journey]?.[section]?.[name] + ); + for (const [index, value] of measured.entries()) { + if (value === undefined) { + return { + measured, + reason: `not measured in run ${index + 1} (journeys.${journey}.${section}).`, + }; + } + if (typeof value !== 'number' || !Number.isFinite(value)) { + return { + measured, + reason: `run ${index + 1} measured ${JSON.stringify(value)}, not a finite number.`, + }; + } + } + if (counter) { + for (const [index, run] of runs.entries()) { + const stability = run.journeys[journey]?.counterStability?.[name]; + if (stability?.stable === false) { + const values = Array.isArray(stability.values) + ? ` (iterations ${stability.values.join(', ')})` + : ''; + return { + measured, + reason: `unstable in run ${index + 1}${values}; the summary value hides iterations that disagreed.`, + }; + } + } + } + const highest = Math.max(...measured); + if (highest >= entry.value) { + return { + measured, + reason: `not below ${formatNumber(entry.value)} in every run (highest ${formatNumber(highest)}).`, + }; + } + return { measured, newValue: highest }; +} + +/** + * Pure tightening. Returns the new baselines object and one row per entry so + * the CLI and the PR body can show the per-run numbers. + */ +export function tightenBaselines({ + baselines, + runs, + evidenceRun, + measuredWith = DEFAULT_MEASURED_WITH, + date = new Date().toISOString().slice(0, 10), +}) { + validateBaselines(baselines); + if (runs.length < MIN_RUNS) { + throw new Error( + `Tightening needs at least ${MIN_RUNS} runs; received ${runs.length}.` + ); + } + if (!evidenceRun) { + throw new Error('--evidence-run is required.'); + } + const next = structuredClone(baselines); + const rows = []; + for (const [journey, entries] of Object.entries(baselines.journeys)) { + for (const [name, entry] of Object.entries(entries)) { + const decision = decide(entry, name, journey, runs); + rows.push({ + label: `${journey}/${name}`, + unit: entry.unit, + value: entry.value, + ...decision, + }); + if (decision.newValue === undefined) continue; + next.journeys[journey][name] = { + ...entry, + value: decision.newValue, + updatedAt: date, + measuredWith: `${measuredWith}, max of ${runs.length} runs`, + evidenceRun, + }; + } + } + const direction = compareBaselineDirection({ base: baselines, head: next }); + if (direction.failures.length > 0) { + throw new Error( + `Refusing to write a weakened baselines file: ${direction.failures.join(' ')}` + ); + } + const tightened = rows.filter((row) => row.newValue !== undefined); + return { baselines: next, rows, tightened }; +} + +/** Sets `evidencePr` on the entries a tightening run wrote. */ +export function fillEvidencePr({ baselines, evidenceRun, pr }) { + validateBaselines(baselines); + if (!Number.isInteger(pr) || pr <= 0) { + throw new Error(`--fill-evidence-pr expects a PR number, got ${pr}.`); + } + const next = structuredClone(baselines); + let filled = 0; + for (const entries of Object.values(next.journeys)) { + for (const entry of Object.values(entries)) { + if (entry.evidenceRun !== evidenceRun) continue; + entry.evidencePr = pr; + filled += 1; + } + } + if (filled === 0) { + throw new Error(`No baseline carries evidenceRun ${evidenceRun}.`); + } + return { baselines: next, filled }; +} + +/** Markdown: readable in a terminal, and pasted as is into the PR body. */ +export function formatReport({ rows, tightened }, runNames = []) { + const runCount = Math.max(0, ...rows.map((row) => row.measured.length)); + const runHeaders = Array.from( + { length: runCount }, + (_, index) => `Run ${index + 1}` + ); + const lines = [ + `| Baseline | Value | ${runHeaders.join(' | ')} | Result |`, + `| --- | ---: | ${runHeaders.map(() => '---:').join(' | ')} | --- |`, + ]; + for (const row of rows) { + const unit = row.unit ? ` ${row.unit}` : ''; + const cells = row.measured.map((value) => + typeof value === 'number' ? formatNumber(value) : '—' + ); + const result = + row.newValue !== undefined + ? `lowered to ${formatNumber(row.newValue)}${unit}` + : `kept: ${row.reason}`; + lines.push( + `| \`${row.label}\` | ${formatNumber(row.value)}${unit} | ${cells.join(' | ')} | ${result} |` + ); + } + if (runNames.length > 0) { + lines.push(''); + for (const [index, name] of runNames.entries()) { + lines.push(`- Run ${index + 1}: ${name}`); + } + } + lines.push( + '', + tightened.length > 0 + ? `Tightened ${tightened.length} of ${rows.length} baselines.` + : `No baseline was below its value in every run; nothing to tighten (${rows.length} checked).` + ); + return lines.join('\n'); +} + +export function parseArgs(argv) { + const options = { + runs: [], + baselines: DEFAULT_BASELINES_PATH, + evidenceRun: null, + report: null, + measuredWith: DEFAULT_MEASURED_WITH, + date: undefined, + fillEvidencePr: null, + }; + const valued = { + '--run': (value) => options.runs.push(value), + '--baselines': (value) => (options.baselines = value), + '--evidence-run': (value) => (options.evidenceRun = value), + '--report': (value) => (options.report = value), + '--measured-with': (value) => (options.measuredWith = value), + '--date': (value) => (options.date = value), + '--fill-evidence-pr': (value) => + (options.fillEvidencePr = Number(value)), + }; + for (let index = 0; index < argv.length; index += 1) { + const argument = argv[index]; + if (argument === '--') continue; + const [flag, inline] = argument.split(/=(.*)/s, 2); + if (!(flag in valued)) throw new Error(`Unknown argument: ${argument}`); + const value = inline ?? argv[++index]; + if (value === undefined || value === '') { + throw new Error(`Missing value for ${flag}`); + } + valued[flag](value); + } + if ( + options.date !== undefined && + !/^\d{4}-\d{2}-\d{2}$/.test(options.date) + ) { + throw new Error(`--date expects YYYY-MM-DD, got "${options.date}".`); + } + if (!options.evidenceRun) { + throw new Error('--evidence-run is required.'); + } + return options; +} + +async function readJson(filePath) { + try { + return JSON.parse(await readFile(filePath, 'utf8')); + } catch (error) { + throw new Error(`Cannot read ${filePath}: ${error.message}`); + } +} + +/** Reads one run: a summary file, or every `*.json` directly in a directory. */ +export async function readRun(runPath, runLabel) { + let files = [runPath]; + if ((await stat(runPath)).isDirectory()) { + files = (await readdir(runPath)) + .filter((name) => name.endsWith('.json')) + .sort() + .map((name) => path.join(runPath, name)); + if (files.length === 0) { + throw new Error(`${runLabel}: no summary files in ${runPath}.`); + } + } + const summaries = await Promise.all(files.map((file) => readJson(file))); + return mergeRunSummaries(summaries, runLabel); +} + +async function writeJson(filePath, value) { + await writeFile(filePath, `${JSON.stringify(value, null, 4)}\n`); +} + +const isMain = + process.argv[1] && + path.resolve(process.argv[1]) === + path.resolve(fileURLToPath(import.meta.url)); + +if (isMain) { + try { + const options = parseArgs(process.argv.slice(2)); + const baselinesPath = path.resolve(options.baselines); + const baselines = await readJson(baselinesPath); + if (options.fillEvidencePr !== null) { + const { baselines: next, filled } = fillEvidencePr({ + baselines, + evidenceRun: options.evidenceRun, + pr: options.fillEvidencePr, + }); + await writeJson(baselinesPath, next); + console.log( + `Set evidencePr ${options.fillEvidencePr} on ${filled} baselines.` + ); + } else { + const runs = await Promise.all( + options.runs.map((run, index) => + readRun(path.resolve(run), `run ${index + 1}`) + ) + ); + const result = tightenBaselines({ + baselines, + runs, + evidenceRun: options.evidenceRun, + measuredWith: options.measuredWith, + date: options.date, + }); + const report = formatReport( + result, + options.runs.map((run) => path.basename(path.resolve(run))) + ); + console.log(report); + if (options.report) await writeFile(options.report, `${report}\n`); + if (result.tightened.length > 0) { + await writeJson(baselinesPath, result.baselines); + } + } + } catch (error) { + console.error(`tighten-baselines: ${error.message}`); + process.exitCode = 1; + } +} diff --git a/tools/performance/tighten-baselines.test.mjs b/tools/performance/tighten-baselines.test.mjs new file mode 100644 index 000000000..4c3542719 --- /dev/null +++ b/tools/performance/tighten-baselines.test.mjs @@ -0,0 +1,396 @@ +import assert from 'node:assert/strict'; +import { spawnSync } from 'node:child_process'; +import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'; +import os from 'node:os'; +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { after, before, test } from 'node:test'; + +import { compareBaselineDirection } from './check-baseline-direction.mjs'; +import { + fillEvidencePr, + formatReport, + mergeRunSummaries, + parseArgs, + tightenBaselines, +} from './tighten-baselines.mjs'; + +const scriptPath = fileURLToPath( + new URL('./tighten-baselines.mjs', import.meta.url) +); +const RUN_URL = 'https://github.com/4gray/iptvnator/actions/runs/1'; + +const baselinesFile = (entries) => ({ + version: 1, + journeys: { + launch: { + 'renderer.initialBytes': { + value: 1000, + unit: 'bytes', + slack: 64, + updatedAt: '2026-09-27', + evidencePr: 1734, + measuredWith: + 'pnpm nx build web && pnpm run perf:initial-bytes', + }, + ...entries, + }, + }, +}); + +const run = (launch) => ({ journeys: { launch } }); +const bytesRun = (initialBytes) => + run({ counters: { 'renderer.initialBytes': initialBytes } }); + +const tighten = (baselines, runs) => + tightenBaselines({ + baselines, + runs, + evidenceRun: RUN_URL, + date: '2026-10-05', + }); + +let workDir; +before(async () => { + workDir = await mkdtemp(path.join(os.tmpdir(), 'tighten-baselines-')); +}); +after(async () => { + await rm(workDir, { recursive: true, force: true }); +}); + +test('a counter below its value in every run drops to the largest run', () => { + const baselines = baselinesFile(); + const result = tighten(baselines, [ + bytesRun(990), + bytesRun(995), + bytesRun(980), + ]); + assert.deepEqual( + result.baselines.journeys.launch['renderer.initialBytes'], + { + value: 995, + unit: 'bytes', + slack: 64, + updatedAt: '2026-10-05', + evidencePr: 1734, + measuredWith: + '.github/workflows/performance-ratchet.yml, max of 3 runs', + evidenceRun: RUN_URL, + } + ); + assert.equal(result.tightened.length, 1); + assert.equal( + baselines.journeys.launch['renderer.initialBytes'].value, + 1000, + 'the input is not mutated' + ); + const direction = compareBaselineDirection({ + base: baselines, + head: result.baselines, + }); + assert.deepEqual(direction.failures, []); + assert.deepEqual(direction.allowed, []); + assert.match(direction.lowered[0], /1,064 -> 1,059 bytes/); +}); + +test('one run at or above the value keeps the baseline untouched', () => { + for (const measured of [1000, 1010]) { + const baselines = baselinesFile(); + const result = tighten(baselines, [ + bytesRun(990), + bytesRun(measured), + bytesRun(980), + ]); + assert.deepEqual(result.baselines, baselines); + assert.deepEqual(result.tightened, []); + assert.match(result.rows[0].reason, /not below 1,000 in every run/); + } +}); + +test('an entry missing or non-numeric in any run is kept', () => { + const missing = tighten(baselinesFile(), [ + bytesRun(990), + run({ counters: {} }), + bytesRun(980), + ]); + assert.deepEqual(missing.tightened, []); + assert.match(missing.rows[0].reason, /not measured in run 2/); + + const invalid = tighten(baselinesFile(), [ + bytesRun(990), + bytesRun(980), + bytesRun('980'), + ]); + assert.deepEqual(invalid.tightened, []); + assert.match(invalid.rows[0].reason, /run 3 measured "980"/); +}); + +test('a counter marked unstable in any run is skipped with the iterations', () => { + const baselines = baselinesFile({ + 'renderer.ipcCallsToFirstCard': { value: 20 }, + }); + const ipcRun = (value, stable = true) => + run({ + counters: { + 'renderer.initialBytes': 990, + 'renderer.ipcCallsToFirstCard': value, + }, + counterStability: { + 'renderer.ipcCallsToFirstCard': { + stable, + values: stable ? [value, value] : [13, value], + }, + }, + }); + const result = tighten(baselines, [ + ipcRun(16), + ipcRun(16, false), + ipcRun(16), + ]); + const row = result.rows.find( + (entry) => entry.label === 'launch/renderer.ipcCallsToFirstCard' + ); + assert.match(row.reason, /unstable in run 2 \(iterations 13, 16\)/); + assert.equal( + result.baselines.journeys.launch['renderer.ipcCallsToFirstCard'].value, + 20 + ); + assert.equal( + result.baselines.journeys.launch['renderer.initialBytes'].value, + 990, + 'other entries are still tightened' + ); +}); + +test('a wall-clock entry drops to the largest P50 and keeps its tolerance', () => { + const baselines = baselinesFile({ + 'spawnToFirstCardMs.p50': { + value: 1600, + unit: 'ms', + toleranceRatio: 1.25, + }, + }); + const wallRun = (p50) => + run({ + counters: { 'renderer.initialBytes': 1000 }, + wallClock: { + 'spawnToFirstCardMs.p50': p50, + 'spawnToFirstCardMs.p90': 5000, + }, + // counterStability does not apply to wall-clock entries. + counterStability: { + 'spawnToFirstCardMs.p50': { stable: false }, + }, + }); + const result = tighten(baselines, [ + wallRun(1401.5), + wallRun(1550.25), + wallRun(1480), + ]); + assert.deepEqual( + result.baselines.journeys.launch['spawnToFirstCardMs.p50'], + { + value: 1550.25, + unit: 'ms', + toleranceRatio: 1.25, + updatedAt: '2026-10-05', + measuredWith: + '.github/workflows/performance-ratchet.yml, max of 3 runs', + evidenceRun: RUN_URL, + } + ); + assert.deepEqual( + compareBaselineDirection({ base: baselines, head: result.baselines }) + .failures, + [] + ); +}); + +test('fewer than three runs or no run URL is refused', () => { + assert.throws( + () => tighten(baselinesFile(), [bytesRun(990), bytesRun(990)]), + /at least 3 runs; received 2/ + ); + assert.throws( + () => + tightenBaselines({ + baselines: baselinesFile(), + runs: [bytesRun(1), bytesRun(1), bytesRun(1)], + }), + /--evidence-run/ + ); +}); + +test('entries are never added and values are never raised', () => { + const baselines = baselinesFile(); + const result = tighten(baselines, [ + run({ counters: { 'renderer.initialBytes': 1200, newCounter: 1 } }), + bytesRun(1200), + bytesRun(1200), + ]); + assert.deepEqual(result.baselines, baselines); +}); + +test('summaries of one run merge; a duplicate measurement is an error', () => { + const merged = mergeRunSummaries([ + bytesRun(990), + run({ + counters: { 'renderer.ipcCallsToFirstCard': 16 }, + wallClock: { 'spawnToFirstCardMs.p50': 1400 }, + }), + ]); + assert.deepEqual(merged.journeys.launch.counters, { + 'renderer.initialBytes': 990, + 'renderer.ipcCallsToFirstCard': 16, + }); + assert.throws( + () => mergeRunSummaries([bytesRun(990), bytesRun(991)], 'run 2'), + /run 2: journeys\.launch\.counters\.renderer\.initialBytes appears in two summaries/ + ); +}); + +test('fillEvidencePr sets the PR only on entries from that run', () => { + const baselines = baselinesFile({ + other: { value: 5, evidenceRun: 'https://example.invalid/runs/0' }, + }); + baselines.journeys.launch['renderer.initialBytes'].evidenceRun = RUN_URL; + const { baselines: next, filled } = fillEvidencePr({ + baselines, + evidenceRun: RUN_URL, + pr: 1800, + }); + assert.equal(filled, 1); + assert.equal( + next.journeys.launch['renderer.initialBytes'].evidencePr, + 1800 + ); + assert.equal(next.journeys.launch.other.evidencePr, undefined); + assert.throws( + () => fillEvidencePr({ baselines, evidenceRun: 'x', pr: 1 }), + /No baseline carries evidenceRun x/ + ); +}); + +test('the report lists every run value and the decision', () => { + const result = tighten(baselinesFile({ gone: { value: 3 } }), [ + bytesRun(990), + bytesRun(995), + bytesRun(980), + ]); + const report = formatReport(result, ['a', 'b', 'c']); + assert.match( + report, + /\| `launch\/renderer.initialBytes` \| 1,000 bytes \| 990 \| 995 \| 980 \| lowered to 995 bytes \|/ + ); + assert.match( + report, + /\| `launch\/gone` \| 3 \| — \| — \| — \| kept: not measured in run 1/ + ); + assert.match(report, /- Run 2: b/); + assert.match(report, /Tightened 1 of 2 baselines\./); +}); + +test('parseArgs reads repeated runs and requires the run URL', () => { + assert.deepEqual( + parseArgs([ + '--run', + 'a.json', + '--run=b', + '--evidence-run', + RUN_URL, + '--date=2026-10-05', + ]).runs, + ['a.json', 'b'] + ); + assert.throws(() => parseArgs(['--run', 'a.json']), /--evidence-run/); + assert.throws( + () => parseArgs(['--evidence-run', RUN_URL, '--date', '5.10.2026']), + /YYYY-MM-DD/ + ); + assert.throws(() => parseArgs(['--bogus']), /Unknown argument/); + assert.throws(() => parseArgs(['--run']), /Missing value for --run/); +}); + +test('CLI rewrites the file only when a baseline was tightened', async () => { + const baselinesPath = path.join(workDir, 'baselines.json'); + const original = `${JSON.stringify(baselinesFile(), null, 4)}\n`; + await writeFile(baselinesPath, original); + const runArgs = []; + for (const [index, bytes] of [990, 1000, 985].entries()) { + const dir = path.join(workDir, `run-${index + 1}`); + await mkdir(dir, { recursive: true }); + await writeFile( + path.join(dir, 'initial-bytes.summary.json'), + JSON.stringify(bytesRun(bytes)) + ); + await writeFile( + path.join(dir, 'journeys.summary.json'), + JSON.stringify(run({ counters: { 'renderer.longTasks': 2 } })) + ); + runArgs.push('--run', dir); + } + const cli = (...extra) => + spawnSync( + process.execPath, + [ + scriptPath, + ...runArgs, + '--baselines', + baselinesPath, + '--evidence-run', + RUN_URL, + ...extra, + ], + { encoding: 'utf8' } + ); + + const unchanged = cli(); + assert.equal(unchanged.status, 0, unchanged.stderr); + assert.match(unchanged.stdout, /nothing to tighten/); + assert.equal(await readFile(baselinesPath, 'utf8'), original); + + await writeFile( + path.join(workDir, 'run-2', 'initial-bytes.summary.json'), + JSON.stringify(bytesRun(970)) + ); + const reportPath = path.join(workDir, 'report.md'); + const tightened = cli('--report', reportPath, '--date', '2026-10-05'); + assert.equal(tightened.status, 0, tightened.stderr); + const written = JSON.parse(await readFile(baselinesPath, 'utf8')); + assert.equal(written.journeys.launch['renderer.initialBytes'].value, 990); + assert.match(await readFile(reportPath, 'utf8'), /lowered to 990 bytes/); + + const fill = spawnSync( + process.execPath, + [ + scriptPath, + '--baselines', + baselinesPath, + '--evidence-run', + RUN_URL, + '--fill-evidence-pr', + '1800', + ], + { encoding: 'utf8' } + ); + assert.equal(fill.status, 0, fill.stderr); + const filled = JSON.parse(await readFile(baselinesPath, 'utf8')); + assert.equal( + filled.journeys.launch['renderer.initialBytes'].evidencePr, + 1800 + ); + + const tooFew = spawnSync( + process.execPath, + [ + scriptPath, + '--run', + path.join(workDir, 'run-1'), + '--evidence-run', + RUN_URL, + ], + { encoding: 'utf8' } + ); + assert.equal(tooFew.status, 1); + assert.match(tooFew.stderr, /at least 3 runs; received 1/); +});