diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e1a028afb..08f56981e 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -327,12 +327,17 @@ jobs: id: journeys uses: ./.github/actions/performance-journeys - # Only the J1 counters shown deterministic on this runner; the - # other journey measurements stay evidence (see Ratchet in - # docs/architecture/performance-journeys.md). Here and not in - # the composite action, so the weekly tightening still measures - # a run that would fail it. - - name: Check the J1 runtime counters against the baselines + # Only the counters identical in every measured iteration of + # recent master runs; their entries say whether a counter is + # validated against wall-clock or a guard only (see Ratchet in + # docs/architecture/performance-journeys.md). Here and not in the + # composite action, so the weekly tightening still measures a run + # that would fail it. tools/performance tests keep this list equal + # to the journey entries of journey-baselines.json. It also runs + # when a later step of the action (the job-summary report) failed + # after the summary was written, so the counters are still checked. + - name: Check the journey counters against the baselines + if: ${{ !cancelled() && steps.journeys.outputs.summary != '' }} env: SUMMARY: ${{ steps.journeys.outputs.summary }} run: >- @@ -340,6 +345,14 @@ jobs: --summary "$SUMMARY" --only launch/renderer.ipcCallsToFirstCard --only launch/renderer.domMutationsToFirstCard + --only launch/main.modulesRegisteredBeforeWindow + --only launch/renderer.layoutShiftScore + --only launch/renderer.layoutShiftScoreSettled + --only open-source/main.mockHttpRequestsToSettled + --only open-source/renderer.ipcCallsToFirstPage + --only open-source/renderer.layoutShiftScore + --only playback/renderer.httpRequestsToPlaying + --only playback/renderer.layoutShiftScore - name: Upload journey summaries if: always() diff --git a/.github/workflows/performance-ratchet.yml b/.github/workflows/performance-ratchet.yml index 39e3a0c4d..2f77794bb 100644 --- a/.github/workflows/performance-ratchet.yml +++ b/.github/workflows/performance-ratchet.yml @@ -240,7 +240,7 @@ jobs: git diff "$HEAD_SHA" HEAD -- "$baselines" echo '```' echo - echo "If \`master\` moved since \`$HEAD_SHA\`, make sure the Initial bytes ratchet job passes on this PR before merging." + echo "If \`master\` moved since \`$HEAD_SHA\`, make sure the Initial bytes ratchet and Performance journeys jobs pass on this PR before merging." } > "$BODY" gh api -X PATCH "repos/$REPOSITORY/pulls/$pr" -F "body=@$BODY" --silent echo "Pull request: ${GITHUB_SERVER_URL}/$REPOSITORY/pull/$pr" diff --git a/docs/architecture/performance-journeys.md b/docs/architecture/performance-journeys.md index 984ec766a..d16fb4f50 100644 --- a/docs/architecture/performance-journeys.md +++ b/docs/architecture/performance-journeys.md @@ -627,10 +627,10 @@ runs before 2738bc28a had 30 of 36 iterations on the slow path (18 calls, numbers so `tools/performance/check-journey-ratchet.mjs` can compare them with `tools/performance/journey-baselines.json`. The summary writer checks only that every measured iteration reports the same counter names with finite -values, so a new counter needs no schema change. A J1 runtime baseline is added -once its counter is deterministic on the CI runner. Two are enforced -(`renderer.ipcCallsToFirstCard` and `renderer.domMutationsToFirstCard`, see -[Ratchet](#ratchet)); the other runtime counters are evidence only. +values, so a new counter needs no schema change. A runtime baseline is added +once its counter is deterministic on the CI runner; the enforced ones and the +reasons for the others are under +[Enforced journey counters](#enforced-journey-counters). J3 adds the `journeys.playback` entry with the same shape and no schema version change: `counters` and `wallClock` hold only plain numbers, and its @@ -864,8 +864,9 @@ mutations come from the EPG timeline rendering about 240 programme blocks from the `get_simple_data_table` response before the first frame. Whether that response and its render land before `playing` is a race on a slower machine, so check the runner's `counterStability` before trusting the -mutation and request counts. No J3 baseline exists yet; J3 counters join the -ratchet once three runner runs agree. +mutation and request counts. On the runner `renderer.httpRequestsToPlaying` +and `renderer.layoutShiftScore` are enforced as guards; see +[Enforced journey counters](#enforced-journey-counters). ## `renderer.initialBytes` @@ -943,7 +944,10 @@ that file: - a measurement below its baseline passes and prints a "tighten" hint; - a measured counter without a baseline is noted, not failed; - checking nothing fails: an empty baselines file, or `--only` naming an - entry that does not exist, cannot exit 0. + entry that does not exist, cannot exit 0; +- an optional `note` (a string) is printed with the entry's failure; journey + entries use it to mark a guard that is not validated against wall-clock + (see [Enforced journey counters](#enforced-journey-counters)). `--only /` (repeatable) restricts the check to the named baselines. A script that measures one counter writes its own summary file @@ -1041,46 +1045,85 @@ Pushes to `master` and manual dispatches always run it. The job is warn-only (`c weeks (plan item B3): a regression marks the job failed without failing the workflow. Making it required is a maintainer decision. -The job enforces two J1 runtime counters: `renderer.ipcCallsToFirstCard` -(15 calls) and `renderer.domMutationsToFirstCard` (558 mutations). After the -`Run the performance journeys` step it runs -`check-journey-ratchet.mjs --only launch/renderer.ipcCallsToFirstCard --only launch/renderer.domMutationsToFirstCard` -on the summary that step wrote. Both entries have `slack` 0 and -`evidenceRun` 37192092882, and were identical and `stable: true` in all -three dispatched runs of the fix that removed the launch race on `master` -(see -[When the window is shown](#when-the-window-is-shown)). The step is in the -job, not in the composite action, so the weekly tightening still measures a -run that would fail it. While the job is warn-only, a regression fails the -job and not the workflow. The two summaries of #1782 before that fix (18 -and 1,018) fail the check. +After the `Run the performance journeys` step, the job runs +`check-journey-ratchet.mjs --only …` on the summary that step wrote, for the +journey entries of `journey-baselines.json` (every entry except +`renderer.initialBytes`, which the `Initial bytes ratchet` job checks). A +`performance-tools` test keeps that `--only` list equal to those entries, so +a baseline cannot be added without being enforced. The step is in the job, +not in the composite action, so the weekly tightening still measures a run +that would fail it. While the job is warn-only, a regression fails the job +and not the workflow. -Until that fix, J1 had two paths on the runner and no runtime counter could -be enforced. Three dispatched runs on 2026-09-27 (36271875209, 36271879955 -and 36271884616) already showed both paths (16 calls / 939 mutations against -13 / 576 at the time), and later `master` runs mixed them more often. +#### Enforced journey counters -The other J1 counters in the same three runs: +Two J1 entries are validated (Principle 3) and carry no note: +`launch/renderer.ipcCallsToFirstCard` 15 and +`launch/renderer.domMutationsToFirstCard` 558 (#1828, `evidenceRun` +37192092882). #1828 removed the launch race (the window shown at +`did-finish-load`, see [When the window is shown](#when-the-window-is-shown)): +three dispatched runs on `master` read 15 / 558 in all 18 iterations, and +load to the first card went from about 940 ms to 466-528 ms. Slow-path +summaries (18 / 1,018 or more) fail the check. -| Counter | Value | `stable` in all three runs | -| ------------------------------------- | ----- | ------------------------------------------------------------------------- | -| `main.modulesRegisteredBeforeWindow` | 2 | yes | -| `renderer.ipcSerialDepthToFirstCard` | 6 | yes (was unstable through the race) | -| `renderer.cdTicksIdle30s` | 4 | yes (was unstable through the race) | -| `renderer.layoutShiftScore` | 0 | yes | -| `renderer.layoutShiftScoreSettled` | 0 | yes (#1782's hero fix plus this one) | -| `renderer.longTasks` | 2 | yes | -| `renderer.cdTicksToFirstCard` | 21 | no: one iteration of 36928725392 read 22 (and one of 36930457538 read 20) | -| `main.sqlStatementsBeforeReadyToShow` | 95 | no: 93 or 95 in every run | +A counter is enforced once it was identical in every measured iteration of +every recent `master` run. The other entries were identical in all 55 +measured iterations of the 11 `master` runs from 2026-10-03 08:25 to +2026-10-04 06:55 (CI runs 37109621784 to 37184230956), with `slack` 0: -`renderer.cdTicksToFirstCard` keeps the one-tick race described under -[change detection](#change-detection-ticks), which the Mac shows too (20 or -21). `main.sqlStatementsBeforeReadyToShow` keeps the download and recording -recovery racing `ready-to-show` (plan item A2). The six stable counters are -candidates for further baselines once more runs agree. Wall-clock entries -stay evidence. Runner counters still differ from a Mac (14 calls and 554 -mutations there; the missing call is the Linux-only `getWindowState`), so -take J1 baseline values from the runner only. +| Entry | Value | Week (69 runs since 2026-09-27) | +| -------------------------------------------- | ----- | ------------------------------------------------------------------------- | +| `launch/main.modulesRegisteredBeforeWindow` | 2 | identical | +| `launch/renderer.layoutShiftScore` | 0 | identical | +| `launch/renderer.layoutShiftScoreSettled` | 0 | 0.235 in some iterations of 8 runs up to 2026-10-02 (hero flicker, #1782) | +| `open-source/main.mockHttpRequestsToSettled` | 1 | identical | +| `open-source/renderer.ipcCallsToFirstPage` | 17 | identical | +| `open-source/renderer.layoutShiftScore` | 0.233 | 0.221, then 0.222; 0.233 since #1814, never mixed within a run | +| `playback/renderer.httpRequestsToPlaying` | 2 | identical | +| `playback/renderer.layoutShiftScore` | 0.001 | identical | + +`open-source/renderer.layoutShiftScore` read 0.222 in that window and 0.233 +in every iteration of every `master` run from 84aef83a6 (#1814, page Back +buttons moved into the header) on, so its value is 0.233 with `evidenceRun` +37372780064 (b78224376). The 0.011 that #1814 added is not explained yet; +lowering it back is a separate change. + +Being deterministic is not the same as being validated. Principle 3 of the +plan promotes a counter to a guardrail once a PR has shown that lowering it +lowered the journey's wall-clock. None of these counters has that evidence +yet: `main.modulesRegisteredBeforeWindow` waits for the deferred IPC +registration (plan item C4), the layout-shift scores measure visual +stability rather than time, and no PR has moved a J2 or J3 counter. Each +entry therefore carries +`"note": "guard only, not validated: …"`. A guard stops a regression of a +deterministic number, and the checker prints the note with a failure; it +says nothing about whether lowering that number makes the journey faster. +When a PR shows that link, it drops the note and names the evidence in +`evidencePr`. `renderer.initialBytes` predates the note and carries none; +this document records no wall-clock change for it either. + +Not enforced, with the reason: + +- J1 `renderer.ipcSerialDepthToFirstCard` (6): bimodal on `master` until + #1828 (9 or 6) and identical in its three dispatched runs; a candidate + once `master` runs agree. +- J1 `main.sqlStatementsBeforeReadyToShow`: 93 or 95 even without the launch + race, because the download and recording recovery races `ready-to-show` + (plan item A2). +- Every `renderer.longTasks` (J1 2 or 1, J2 0 with one 1 earlier in the + week, J3 1 or 2): a long task is a task over 50 ms, so the count follows + runner speed, not work. +- Every `cdTicks` counter: the zoneless migration (plan item C6) changes + them. +- J2 `renderer.domMutationsToFirstPage`: 1,602 or 1,603 between runs of + recent commits. +- J3 `renderer.ipcCallsToPlaying` (4 or 5) and + `renderer.domMutationsToPlaying` (6,182, 6,183 or 6,199): not identical, + and the EPG rendering work changes the mutation count. +- J4: not measured yet. + +Runner counters differ from a Mac (the Linux-only `getWindowState` call, for +one), so take every journey baseline value from the runner only. ### Weekly tightening @@ -1105,7 +1148,11 @@ its job; its entries are then unmeasured in that run. A final job runs `check-baseline-direction.mjs` without `--allow-increase`, which both the script and the job check; - a lowered entry gets `updatedAt`, `measuredWith` and `evidenceRun` (the - workflow run URL); `evidencePr` is set to the tightening PR once it exists. + workflow run URL); `evidencePr` is set to the tightening PR once it exists; + every other field, including a `note`, is kept, so a guard stays marked as + not validated after it is lowered; +- a counter already at 0 is never lowered; a layout-shift score is lowered + to the three-decimal value the summary reports. When the file changed and the run is on `master`, the job pushes `automation/performance-ratchet` and opens (or updates) a pull request with @@ -1124,8 +1171,10 @@ dispatches workflows that exist on the default branch, so before the first merge of a new or renamed workflow add a temporary `push` trigger for the branch and drop it before review, as #1760 did. Review the pull request like a manual tightening: if `master` moved since the measured commit, the -`Initial bytes ratchet` job on the pull request is what shows that the new -value still holds (the concurrent-merge effect above). +`Initial bytes ratchet` job (for `renderer.initialBytes`) and the +`Performance journeys` job (for the journey counters) on the pull request +are what show that the new values still hold (the concurrent-merge effect +above). ## Charset parse benchmark @@ -1172,7 +1221,13 @@ reports slow imports of non-Latin playlists. 3. Cover the extraction and the failure modes with `node --test` and register the test file in `tools/performance/project.json`. 4. Validate the counter before it becomes a guardrail: one PR must show that - lowering it moved wall-clock in the same journey. + lowering it moved wall-clock in the same journey. A counter that is + deterministic but not validated may be enforced as a guard: its baseline + entry carries a `note` saying so (see + [Enforced journey counters](#enforced-journey-counters)). +5. A journey counter's baseline is enforced only when it is also in the + `--only` list of the `Performance journeys` job in `ci.yml`; the + `performance-tools` tests fail when the two differ. ## Adding a journey diff --git a/docs/architecture/validation-map.md b/docs/architecture/validation-map.md index bd2fe853e..4ee3ee303 100644 --- a/docs/architecture/validation-map.md +++ b/docs/architecture/validation-map.md @@ -276,7 +276,7 @@ pnpm nx build web pnpm run perf:initial-bytes # breakdown only pnpm run perf:initial-bytes:check # measure, then compare with the committed baseline pnpm nx test performance-tools -pnpm run perf:journeys # J1 launch + J2 open-source journeys, one dist/performance/journeys//summary.json +pnpm run perf:journeys # J1 launch, J2 open-source and J3 playback journeys, one dist/performance/journeys//summary.json ``` `perf:initial-bytes` reads the built `dist/apps/web/index.html` and sums the @@ -285,7 +285,12 @@ bytes on the initial path (the J1 counter `renderer.initialBytes`). `tools/performance/journey-baselines.json`; baselines only move down. CI runs the same check in the `Initial bytes ratchet` job of `ci.yml` for PRs that target `master` and for `master` pushes (dispatch it with -`gh workflow run ci.yml --ref ` for a stacked branch). The weekly +`gh workflow run ci.yml --ref ` for a stacked branch). The +`Performance journeys` job checks the journey counters listed with `--only` +in its `Check the journey counters against the baselines` step against the +same file (warn-only); a new journey baseline +must be added to that list too, which `pnpm nx test performance-tools` +checks. The weekly `performance-ratchet.yml` workflow lowers baselines through a bot PR; validate a change to it with `gh workflow run performance-ratchet.yml --ref `, which measures but opens no PR off `master`. Dispatch needs the workflow file @@ -293,10 +298,7 @@ on `master`; before that, see the temporary-trigger note under Weekly tightening in the performance journeys document. `perf:journeys` builds the `electron-performance` configuration and runs every journey spec against the Xtream mock: J1 launch, then J2 open-source (a second set of launches, each followed by the click on the portal card), both -written to the same summary file. The `Performance journeys` job of `ci.yml` -(warn-only) checks `renderer.ipcCallsToFirstCard` and -`renderer.domMutationsToFirstCard` of that summary against the baselines; -its probe specs run with +written to the same summary file; its probe specs run with `pnpm nx run electron-backend-e2e:test-performance-harness`, which CI runs in the `Unit Tests and Typechecks` job of `ci.yml` on every run. The `electron-backend-e2e` command targets call `tsx` and `playwright` directly, diff --git a/tools/performance/check-baseline-direction.mjs b/tools/performance/check-baseline-direction.mjs index 41816767c..e5a0ada85 100644 --- a/tools/performance/check-baseline-direction.mjs +++ b/tools/performance/check-baseline-direction.mjs @@ -49,8 +49,9 @@ function entries(baselines) { return flat; } +// Three decimals: journey summaries round layout-shift scores to three. function formatNumber(value) { - return value.toLocaleString('en-US', { maximumFractionDigits: 2 }); + return value.toLocaleString('en-US', { maximumFractionDigits: 3 }); } /** diff --git a/tools/performance/check-journey-ratchet.mjs b/tools/performance/check-journey-ratchet.mjs index b03bd70eb..3459b5506 100644 --- a/tools/performance/check-journey-ratchet.mjs +++ b/tools/performance/check-journey-ratchet.mjs @@ -14,6 +14,9 @@ * are lowered by hand, with the measured output as evidence, never raised. * - Checking nothing is a failure: an empty baselines file, or `--only` * naming an entry that does not exist, must not exit 0. + * - An optional `note` (a string) says why an entry is enforced, such as a + * guard whose link to wall-clock is not validated (Principle 3). It is + * printed with a failure, so the author sees what the counter stands for. * * `--only /` (repeatable) restricts the check to the named * baselines, so a script that measures one counter can check that counter @@ -34,10 +37,11 @@ export const DEFAULT_BASELINES_PATH = /** The PR label that lets check-baseline-direction.mjs accept a weakening. */ export const BASELINE_INCREASE_LABEL = 'perf-baseline-increase'; +// Three decimals: journey summaries round layout-shift scores to three. function formatNumber(value) { return Number.isInteger(value) ? value.toLocaleString('en-US') - : value.toLocaleString('en-US', { maximumFractionDigits: 2 }); + : value.toLocaleString('en-US', { maximumFractionDigits: 3 }); } function isPlainObject(value) { @@ -82,6 +86,9 @@ export function validateBaselines(baselines) { `Baseline ${label} has an invalid "slack" (must be an integer >= 0).` ); } + if (entry.note !== undefined && typeof entry.note !== 'string') { + throw new Error(`Baseline ${label} has a non-string "note".`); + } if ( entry.slack !== undefined && entry.toleranceRatio !== undefined @@ -185,8 +192,9 @@ export function compareToBaselines({ baselines, summary, only = [] }) { const limitText = describeLimit(entry, unit); if (measured > limit) { + const note = entry.note ? ` Note: ${entry.note}` : ''; result.failures.push( - `${label}: ${formatNumber(measured)}${unit} exceeds ${limitText} by ${formatNumber(measured - limit)}${unit}. Bring the value back down; baselines only move down. If the growth is a deliberate trade-off, raise the baseline in ${DEFAULT_BASELINES_PATH}, make the case in the PR, and ask a maintainer to add the ${BASELINE_INCREASE_LABEL} label.` + `${label}: ${formatNumber(measured)}${unit} exceeds ${limitText} by ${formatNumber(measured - limit)}${unit}. Bring the value back down; baselines only move down. If the growth is a deliberate trade-off, raise the baseline in ${DEFAULT_BASELINES_PATH}, make the case in the PR, and ask a maintainer to add the ${BASELINE_INCREASE_LABEL} label.${note}` ); } else if (entry.slack && measured > entry.value) { result.passed.push( diff --git a/tools/performance/check-journey-ratchet.test.mjs b/tools/performance/check-journey-ratchet.test.mjs index dd3a44cf7..2b4a04c08 100644 --- a/tools/performance/check-journey-ratchet.test.mjs +++ b/tools/performance/check-journey-ratchet.test.mjs @@ -21,6 +21,10 @@ const committedBaselinesPath = fileURLToPath( new URL('./journey-baselines.json', import.meta.url) ); +const ciWorkflowPath = fileURLToPath( + new URL('../../.github/workflows/ci.yml', import.meta.url) +); + const baselines = { version: 1, journeys: { @@ -357,6 +361,49 @@ test('rejects malformed baseline files', () => { }), /sets both "slack" and "toleranceRatio"/ ); + assert.throws( + () => + validateBaselines({ + journeys: { launch: { x: { value: 1, note: true } } }, + }), + /non-string "note"/ + ); +}); + +test('a failing entry prints its note; a fractional score is exact', () => { + const guarded = { + journeys: { + 'open-source': { + 'renderer.layoutShiftScore': { + value: 0.222, + unit: 'score', + slack: 0, + note: 'guard only, not validated', + }, + }, + }, + }; + const check = (score) => + compareToBaselines({ + baselines: guarded, + summary: { + journeys: { + 'open-source': { + counters: { 'renderer.layoutShiftScore': score }, + }, + }, + }, + }); + + assert.deepEqual(check(0.222).failures, []); + const above = check(0.223); + assert.equal(above.failures.length, 1); + assert.match( + above.failures[0], + /0\.223 score exceeds baseline 0\.222 by 0\.001 score/ + ); + assert.match(above.failures[0], /Note: guard only, not validated$/); + assert.equal(check(0.221).tightenable.length, 1); }); test('formats a summary line for both outcomes', () => { @@ -444,6 +491,27 @@ test('the committed baselines file is valid and every entry names its evidence f } }); +test('the Performance journeys job checks every journey-run baseline', async () => { + const committed = JSON.parse( + await readFile(committedBaselinesPath, 'utf8') + ); + const workflow = await readFile(ciWorkflowPath, 'utf8'); + const step = workflow.match( + /- name: Check the journey counters against the baselines\n([\s\S]*?)(?=\n\s*- name: )/ + ); + assert.ok(step, 'ci.yml must keep the journey counter check step'); + const only = [...step[1].matchAll(/--only (\S+)/g)].map((m) => m[1]); + // renderer.initialBytes is measured from the web build by the Initial + // bytes ratchet job (perf:initial-bytes:check); every other baseline comes + // from the journeys and must be enforced by this step, or it guards nothing. + const expected = Object.entries(committed.journeys) + .flatMap(([journey, entries]) => + Object.keys(entries).map((name) => `${journey}/${name}`) + ) + .filter((label) => label !== 'launch/renderer.initialBytes'); + assert.deepEqual([...only].sort(), [...expected].sort()); +}); + async function runCli(summary, extraBaselines = baselines) { const summaryPath = path.join( workDir, diff --git a/tools/performance/journey-baselines.json b/tools/performance/journey-baselines.json index 205353e56..6b5cd01ee 100644 --- a/tools/performance/journey-baselines.json +++ b/tools/performance/journey-baselines.json @@ -27,6 +27,90 @@ "evidencePr": 1828, "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37192092882", "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in 3 runs" + }, + "main.modulesRegisteredBeforeWindow": { + "value": 2, + "unit": "phases", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-04", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37184230956", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in all 55 measured iterations of 11 master runs, 2026-10-03 to 2026-10-04" + }, + "renderer.layoutShiftScore": { + "value": 0, + "unit": "score", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-04", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37184230956", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in all 55 measured iterations of 11 master runs, 2026-10-03 to 2026-10-04" + }, + "renderer.layoutShiftScoreSettled": { + "value": 0, + "unit": "score", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-04", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37184230956", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in all 55 measured iterations of 11 master runs, 2026-10-03 to 2026-10-04" + } + }, + "open-source": { + "main.mockHttpRequestsToSettled": { + "value": 1, + "unit": "requests", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-04", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37184230956", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in all 55 measured iterations of 11 master runs, 2026-10-03 to 2026-10-04" + }, + "renderer.ipcCallsToFirstPage": { + "value": 17, + "unit": "calls", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-04", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37184230956", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in all 55 measured iterations of 11 master runs, 2026-10-03 to 2026-10-04" + }, + "renderer.layoutShiftScore": { + "value": 0.233, + "unit": "score", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-06", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37372780064", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in every measured iteration of the 7 master runs from 84aef83a6 (#1814, which moved it from 0.222) to b78224376" + } + }, + "playback": { + "renderer.httpRequestsToPlaying": { + "value": 2, + "unit": "requests", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-04", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37184230956", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in all 55 measured iterations of 11 master runs, 2026-10-03 to 2026-10-04" + }, + "renderer.layoutShiftScore": { + "value": 0.001, + "unit": "score", + "slack": 0, + "note": "guard only, not validated: no PR has yet shown that lowering this counter lowers the journey wall-clock (Principle 3)", + "updatedAt": "2026-10-04", + "evidencePr": null, + "evidenceRun": "https://github.com/4gray/iptvnator/actions/runs/37184230956", + "measuredWith": "pnpm run perf:journeys (ubuntu-latest, xvfb), identical in all 55 measured iterations of 11 master runs, 2026-10-03 to 2026-10-04" } } } diff --git a/tools/performance/project.json b/tools/performance/project.json index 106390896..9f0e3936a 100644 --- a/tools/performance/project.json +++ b/tools/performance/project.json @@ -11,6 +11,7 @@ "inputs": [ "{projectRoot}/*.mjs", "{projectRoot}/*.json", + "{workspaceRoot}/.github/workflows/ci.yml", { "externalDependencies": ["parse5"] } ], "options": { diff --git a/tools/performance/tighten-baselines.mjs b/tools/performance/tighten-baselines.mjs index 0c82bf09e..8ae7786d6 100644 --- a/tools/performance/tighten-baselines.mjs +++ b/tools/performance/tighten-baselines.mjs @@ -54,8 +54,9 @@ function isPlainObject(value) { return typeof value === 'object' && value !== null && !Array.isArray(value); } +// Three decimals: journey summaries round layout-shift scores to three. function formatNumber(value) { - return value.toLocaleString('en-US', { maximumFractionDigits: 2 }); + return value.toLocaleString('en-US', { maximumFractionDigits: 3 }); } /** diff --git a/tools/performance/tighten-baselines.test.mjs b/tools/performance/tighten-baselines.test.mjs index 4c3542719..ccd3e5b25 100644 --- a/tools/performance/tighten-baselines.test.mjs +++ b/tools/performance/tighten-baselines.test.mjs @@ -93,6 +93,66 @@ test('a counter below its value in every run drops to the largest run', () => { assert.match(direction.lowered[0], /1,064 -> 1,059 bytes/); }); +test('a journey-run entry keeps its note and is lowered to three decimals', () => { + const score = { + value: 0.222, + unit: 'score', + slack: 0, + note: 'guard only, not validated', + updatedAt: '2026-10-04', + evidencePr: 1, + measuredWith: 'pnpm run perf:journeys', + }; + const baselines = { + version: 1, + journeys: { + ...baselinesFile().journeys, + 'open-source': { 'renderer.layoutShiftScore': score }, + playback: { 'renderer.layoutShiftScore': { ...score, value: 0 } }, + }, + }; + // The weekly job merges each runner's initial-bytes and journey summaries. + const runs = [0.221, 0.22, 0.221].map((value, index) => + mergeRunSummaries( + [ + bytesRun(1000), + { + journeys: { + 'open-source': { + counters: { 'renderer.layoutShiftScore': value }, + }, + playback: { + counters: { 'renderer.layoutShiftScore': 0 }, + }, + }, + }, + ], + `run ${index + 1}` + ) + ); + const result = tighten(baselines, runs); + assert.deepEqual( + result.baselines.journeys['open-source']['renderer.layoutShiftScore'], + { + ...score, + value: 0.221, + updatedAt: '2026-10-05', + measuredWith: + '.github/workflows/performance-ratchet.yml, max of 3 runs', + evidenceRun: RUN_URL, + } + ); + assert.deepEqual( + result.baselines.journeys.playback, + baselines.journeys.playback, + 'a counter already at 0 is kept' + ); + assert.match( + formatReport(result), + /\| `open-source\/renderer\.layoutShiftScore` \| 0\.222 score \| 0\.221 \| 0\.22 \| 0\.221 \| lowered to 0\.221 score \|/ + ); +}); + test('one run at or above the value keeps the baseline untouched', () => { for (const measured of [1000, 1010]) { const baselines = baselinesFile();