diff --git a/apps/electron-backend-e2e/project.json b/apps/electron-backend-e2e/project.json index 2c65b3e85..0ff4ff951 100644 --- a/apps/electron-backend-e2e/project.json +++ b/apps/electron-backend-e2e/project.json @@ -49,6 +49,15 @@ "command": "pnpm exec playwright test --config=playwright.xtream-performance.config.ts src/xtream.performance.ts" } }, + "benchmark-charset-parse": { + "executor": "nx:run-commands", + "cache": false, + "parallelism": false, + "options": { + "cwd": "apps/electron-backend-e2e", + "command": "pnpm exec tsx --expose-gc --tsconfig tsconfig.json src/performance/charset-parse.benchmark.ts" + } + }, "journeys": { "dependsOn": ["electron-backend:build-performance"], "executor": "nx:run-commands", diff --git a/apps/electron-backend-e2e/src/performance/charset-parse-benchmark-report.spec.ts b/apps/electron-backend-e2e/src/performance/charset-parse-benchmark-report.spec.ts new file mode 100644 index 000000000..88662478c --- /dev/null +++ b/apps/electron-backend-e2e/src/performance/charset-parse-benchmark-report.spec.ts @@ -0,0 +1,154 @@ +import assert from 'node:assert/strict'; +import { describe, it } from 'node:test'; + +import { + CHARSET_BENCHMARK_VARIANTS, + type CharsetBenchmarkVariant, + type CharsetCaseMeasurement, + compareCharsets, + formatCharsetReport, + medianOf, + summarizeSelfSamples, + variantOrderForRound, +} from './charset-parse-benchmark-report'; + +function measurement( + variant: CharsetBenchmarkVariant, + cpuMs: number[], + profileSamples: number, + workload = 'm3u parse' +): CharsetCaseMeasurement { + return { + workload, + variant, + inputBytes: 10, + inputLength: 10, + hasNonLatin1Characters: variant !== 'latin1', + wallClockMs: cpuMs.map((value) => value * 2), + cpuMs, + profileSamples, + topSelfFrames: [], + }; +} + +describe('charset parse benchmark report', () => { + it('rotates the starting variant every round', () => { + assert.deepEqual(variantOrderForRound(0), [ + 'latin1', + 'latin1-bom', + 'cyrillic', + ]); + assert.deepEqual(variantOrderForRound(1), [ + 'latin1-bom', + 'cyrillic', + 'latin1', + ]); + assert.deepEqual(variantOrderForRound(2), [ + 'cyrillic', + 'latin1', + 'latin1-bom', + ]); + assert.deepEqual(variantOrderForRound(3), variantOrderForRound(0)); + for (let round = 0; round < 6; round += 1) { + assert.deepEqual( + [...variantOrderForRound(round)].sort(), + [...CHARSET_BENCHMARK_VARIANTS].sort() + ); + } + }); + + it('takes the median of odd and even samples without mutating them', () => { + const values = [5, 1, 3]; + + assert.equal(medianOf(values), 3); + assert.deepEqual(values, [5, 1, 3]); + assert.equal(medianOf([4, 1, 3, 2]), 2.5); + assert.throws(() => medianOf([]), /empty sample/); + }); + + it('counts self samples per frame, most frequent first', () => { + const summary = summarizeSelfSamples( + { + nodes: [ + { + id: 1, + callFrame: { + functionName: '', + url: '', + lineNumber: -1, + }, + }, + { + id: 2, + callFrame: { + functionName: 'scanAttributes', + url: 'file:///repo/node_modules/iptv-playlist-parser/src/index.js', + lineNumber: 42, + }, + }, + ], + samples: [2, 2, 1, 2, 9], + }, + 2 + ); + + assert.equal(summary.total, 5); + assert.deepEqual(summary.top, [ + { frame: 'scanAttributes index.js:43', samples: 3 }, + { frame: '(anonymous)', samples: 1 }, + ]); + }); + + it('compares every variant with latin1 and flags only slowdowns seen by both signals', () => { + const [comparison] = compareCharsets([ + measurement('latin1', [10, 12, 11], 100), + measurement('latin1-bom', [16, 17, 18], 140), + measurement('cyrillic', [16, 17, 18], 160), + ]); + + assert.equal(comparison.workload, 'm3u parse'); + assert.deepEqual( + comparison.variants.map((row) => row.variant), + [...CHARSET_BENCHMARK_VARIANTS] + ); + const [latin1, bom, cyrillic] = comparison.variants; + assert.equal(latin1.cpuRatio, 1); + assert.equal(latin1.exceedsThreshold, false); + assert.equal(bom.cpuP50Ms, 17); + assert.equal(bom.wallP50Ms, 34); + assert.ok(Math.abs(bom.cpuRatio - 17 / 11) < 1e-9); + // CPU time is over 1.5x, but profile samples (1.4x) are not. + assert.equal(bom.exceedsThreshold, false); + assert.equal(cyrillic.sampleRatio, 1.6); + assert.equal(cyrillic.exceedsThreshold, true); + }); + + it('fails when a variant was not measured', () => { + assert.throws( + () => + compareCharsets([ + measurement('latin1', [1], 1), + measurement('cyrillic', [1], 1), + ]), + /Missing latin1-bom measurement for m3u parse/ + ); + }); + + it('formats one markdown row per workload and variant', () => { + const report = formatCharsetReport( + compareCharsets([ + measurement('latin1', [10], 100), + measurement('latin1-bom', [11], 100), + measurement('cyrillic', [20], 200), + ]) + ); + const lines = report.split('\n'); + + assert.equal(lines.length, 5); + assert.match(lines[0], /^\| Workload \| Input \|/); + assert.equal( + lines[4], + '| m3u parse | cyrillic | 40.0 | 20.0 | 200 | 2.00x | 2.00x | 2.00x | yes |' + ); + }); +}); diff --git a/apps/electron-backend-e2e/src/performance/charset-parse-benchmark-report.ts b/apps/electron-backend-e2e/src/performance/charset-parse-benchmark-report.ts new file mode 100644 index 000000000..867c95ed2 --- /dev/null +++ b/apps/electron-backend-e2e/src/performance/charset-parse-benchmark-report.ts @@ -0,0 +1,206 @@ +/** Two-byte input slower than this ratio justifies parser changes (plan D1). */ +export const CHARSET_SLOWDOWN_THRESHOLD = 1.5; + +/** + * Input variants of the charset benchmark. `latin1-bom` is the `latin1` + * fixture behind a UTF-8 byte-order mark: same content, but V8 must store + * the decoded string as two-byte, which isolates the encoding effect from + * the content effect that Cyrillic titles have on ASCII-only regexes. + */ +export const CHARSET_BENCHMARK_VARIANTS = [ + 'latin1', + 'latin1-bom', + 'cyrillic', +] as const; + +export type CharsetBenchmarkVariant = + (typeof CHARSET_BENCHMARK_VARIANTS)[number]; + +export interface CpuProfileNode { + readonly id: number; + readonly callFrame: { + readonly functionName: string; + readonly url: string; + readonly lineNumber: number; + }; +} + +export interface CpuProfileLike { + readonly nodes: readonly CpuProfileNode[]; + readonly samples?: readonly number[]; +} + +export interface SelfSampleFrame { + readonly frame: string; + readonly samples: number; +} + +export interface CharsetCaseMeasurement { + readonly workload: string; + readonly variant: CharsetBenchmarkVariant; + readonly inputBytes: number; + readonly inputLength: number; + /** True when V8 must store the decoded input as a two-byte string. */ + readonly hasNonLatin1Characters: boolean; + readonly wallClockMs: readonly number[]; + readonly cpuMs: readonly number[]; + readonly profileSamples: number; + readonly topSelfFrames: readonly SelfSampleFrame[]; +} + +export interface CharsetVariantSummary { + readonly variant: CharsetBenchmarkVariant; + readonly wallP50Ms: number; + readonly cpuP50Ms: number; + readonly profileSamples: number; + readonly wallRatio: number; + readonly cpuRatio: number; + readonly sampleRatio: number; + /** Slower than the threshold on CPU time and on profile samples. */ + readonly exceedsThreshold: boolean; +} + +export interface CharsetComparison { + readonly workload: string; + readonly variants: readonly CharsetVariantSummary[]; +} + +/** + * Variant order for one benchmark round. The starting variant rotates every + * round so no input always runs first or last after a garbage collection. + */ +export function variantOrderForRound(round: number): CharsetBenchmarkVariant[] { + const variants = [...CHARSET_BENCHMARK_VARIANTS]; + const offset = round % variants.length; + return [...variants.slice(offset), ...variants.slice(0, offset)]; +} + +export function medianOf(values: readonly number[]): number { + if (values.length === 0) { + throw new Error('Cannot take the median of an empty sample'); + } + const sorted = [...values].sort((left, right) => left - right); + const middle = Math.floor(sorted.length / 2); + return sorted.length % 2 === 1 + ? sorted[middle] + : (sorted[middle - 1] + sorted[middle]) / 2; +} + +/** Counts self samples per frame; the top frames explain where time went. */ +export function summarizeSelfSamples( + profile: CpuProfileLike, + limit = 8 +): { total: number; top: SelfSampleFrame[] } { + const nodesById = new Map(profile.nodes.map((node) => [node.id, node])); + const counts = new Map(); + const samples = profile.samples ?? []; + + for (const nodeId of samples) { + const frame = describeFrame(nodesById.get(nodeId)); + counts.set(frame, (counts.get(frame) ?? 0) + 1); + } + + return { total: samples.length, top: topFrames(counts, limit) }; +} + +export function topFrames( + counts: ReadonlyMap, + limit: number +): SelfSampleFrame[] { + return [...counts.entries()] + .map(([frame, samples]) => ({ frame, samples })) + .sort( + (left, right) => + right.samples - left.samples || + left.frame.localeCompare(right.frame) + ) + .slice(0, limit); +} + +export function compareCharsets( + measurements: readonly CharsetCaseMeasurement[] +): CharsetComparison[] { + const workloads = [...new Set(measurements.map((m) => m.workload))]; + + return workloads.map((workload) => { + const baseline = findCase(measurements, workload, 'latin1'); + const baselineWall = medianOf(baseline.wallClockMs); + const baselineCpu = medianOf(baseline.cpuMs); + + return { + workload, + variants: CHARSET_BENCHMARK_VARIANTS.map((variant) => { + const measurement = findCase(measurements, workload, variant); + const wallP50Ms = medianOf(measurement.wallClockMs); + const cpuP50Ms = medianOf(measurement.cpuMs); + const cpuRatio = cpuP50Ms / baselineCpu; + const sampleRatio = + measurement.profileSamples / baseline.profileSamples; + return { + variant, + wallP50Ms, + cpuP50Ms, + profileSamples: measurement.profileSamples, + wallRatio: wallP50Ms / baselineWall, + cpuRatio, + sampleRatio, + exceedsThreshold: + cpuRatio > CHARSET_SLOWDOWN_THRESHOLD && + sampleRatio > CHARSET_SLOWDOWN_THRESHOLD, + }; + }), + }; + }); +} + +export function formatCharsetReport( + comparisons: readonly CharsetComparison[] +): string { + const header = + '| Workload | Input | Wall P50 ms | CPU P50 ms | Samples | Wall ratio | CPU ratio | Sample ratio | Over 1.5x |'; + const divider = + '| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | --- |'; + const rows = comparisons.flatMap((comparison) => + comparison.variants.map((row) => + [ + comparison.workload, + row.variant, + row.wallP50Ms.toFixed(1), + row.cpuP50Ms.toFixed(1), + String(row.profileSamples), + `${row.wallRatio.toFixed(2)}x`, + `${row.cpuRatio.toFixed(2)}x`, + `${row.sampleRatio.toFixed(2)}x`, + row.exceedsThreshold ? 'yes' : 'no', + ].join(' | ') + ) + ); + return [header, divider, ...rows.map((row) => `| ${row} |`)].join('\n'); +} + +function findCase( + measurements: readonly CharsetCaseMeasurement[], + workload: string, + variant: CharsetBenchmarkVariant +): CharsetCaseMeasurement { + const match = measurements.find( + (m) => m.workload === workload && m.variant === variant + ); + if (!match) { + throw new Error(`Missing ${variant} measurement for ${workload}`); + } + return match; +} + +function describeFrame(node: CpuProfileNode | undefined): string { + if (!node) { + return '(unknown)'; + } + const { functionName, url, lineNumber } = node.callFrame; + const name = functionName || '(anonymous)'; + if (!url) { + return name; + } + const file = url.slice(url.lastIndexOf('/') + 1); + return `${name} ${file}:${lineNumber + 1}`; +} diff --git a/apps/electron-backend-e2e/src/performance/charset-parse-workloads.ts b/apps/electron-backend-e2e/src/performance/charset-parse-workloads.ts new file mode 100644 index 000000000..8f0234121 --- /dev/null +++ b/apps/electron-backend-e2e/src/performance/charset-parse-workloads.ts @@ -0,0 +1,191 @@ +import { createPlaylistObject } from '@iptvnator/shared/m3u-utils'; +import { parse } from 'iptv-playlist-parser'; +import { resolve } from 'node:path'; + +import type { CharsetBenchmarkVariant } from './charset-parse-benchmark-report'; +import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset'; +import { createSyntheticM3uFixture, SYNTHETIC_M3U_SEED } from './synthetic-m3u'; +import { createSyntheticXmltvFixture } from './synthetic-xmltv'; + +export const ENTRY_COUNT = 50_000; +/** + * The EPG worker writes 64 KiB decoded chunks. The benchmark writes slices of + * one decoded string instead, so every slice keeps the input's one-byte or + * two-byte representation (decoding each chunk separately would make the + * BOM control one-byte after its first chunk) and no character is split. + */ +const XMLTV_CHUNK_CHARS = 64 * 1024; +const REPLACEMENT_CHARACTER = '\ufffd'; +const UTF8_BYTE_ORDER_MARK = Buffer.from([0xef, 0xbb, 0xbf]); + +type Variant = CharsetBenchmarkVariant; + +interface ParsedProgramme { + readonly title: readonly { readonly value: string }[]; +} + +interface StreamingEpgParserModule { + StreamingEpgParser: new ( + onChannels: (channels: unknown[]) => void, + onPrograms: (programs: ParsedProgramme[]) => void, + onProgress: (channels: number, programs: number) => void, + channelBatchSize?: number, + programBatchSize?: number + ) => { write(chunk: string): void; finish(): { totalPrograms: number } }; +} + +export interface Workload { + readonly name: string; + /** Prepares untimed input, then returns the timed operation. */ + prepare(variant: Variant): () => void; + input(variant: Variant): string; + /** Untimed check that the parser saw the fixture's exact titles. */ + verify(variant: Variant): void; +} + +export async function createWorkloads(): Promise { + const epgModule = (await import( + resolve( + __dirname, + '../../../electron-backend/src/app/workers/epg-streaming-parser.ts' + ) + )) as StreamingEpgParserModule; + const m3uBuffers = buffersPerVariant((charset) => + createSyntheticM3uFixture(ENTRY_COUNT, { charset }) + ); + const xmltvBuffers = buffersPerVariant((charset) => + createSyntheticXmltvFixture(ENTRY_COUNT, { charset }) + ); + const m3uInput = (variant: Variant) => + requireBuffer(m3uBuffers, variant).toString('utf8'); + const xmltvInput = (variant: Variant) => + requireBuffer(xmltvBuffers, variant).toString('utf8'); + + const verifyM3u = (variant: Variant) => { + const titles = titlesFor(variant); + const [first] = parse(m3uInput(variant)).items; + expectEqual( + first?.name, + `${titles.channel} ${SYNTHETIC_M3U_SEED}-000001` + ); + expectEqual(first?.group.title, `${titles.group} 001`); + }; + const parseXmltv = ( + variant: Variant, + onPrograms: (programs: ParsedProgramme[]) => void + ) => { + const chunks = sliceText(xmltvInput(variant), XMLTV_CHUNK_CHARS); + return () => { + const parser = new epgModule.StreamingEpgParser( + () => undefined, + onPrograms, + () => undefined, + 100, + 1000 + ); + for (const chunk of chunks) { + parser.write(chunk); + } + assertCount(parser.finish().totalPrograms); + }; + }; + + return [ + { + // PARSE_M3U phase of the main-process import. + name: 'm3u parse', + input: m3uInput, + verify: verifyM3u, + prepare(variant) { + const body = m3uInput(variant); + return () => assertCount(parse(body).items.length); + }, + }, + { + // NORMALIZE phase of the main-process import. + name: 'm3u normalize', + input: m3uInput, + verify: verifyM3u, + prepare(variant) { + const parsed = parse(m3uInput(variant)); + return () => + assertCount( + createPlaylistObject('benchmark', parsed, 'x', 'URL') + .count + ); + }, + }, + { + // EPG worker's saxes-based parser; UTF-8 decoding is excluded. + name: 'xmltv stream parse', + input: xmltvInput, + verify(variant) { + const titles: string[] = []; + parseXmltv(variant, (programs) => { + for (const programme of programs) { + titles.push(programme.title[0]?.value ?? ''); + } + })(); + const programme = titlesFor(variant).programme; + expectEqual(titles[0], `${programme} 000001-0001`); + expectEqual(titles.at(-1), `${programme} 000500-0100`); + if ( + titles.some((title) => + title.includes(REPLACEMENT_CHARACTER) + ) + ) { + throw new Error(`${variant} XMLTV titles contain U+FFFD`); + } + }, + prepare: (variant) => parseXmltv(variant, () => undefined), + }, + ]; +} + +function titlesFor(variant: Variant) { + return SYNTHETIC_TITLE_VOCABULARY[ + variant === 'cyrillic' ? 'cyrillic' : 'latin1' + ]; +} + +function sliceText(text: string, size: number): string[] { + const chunks: string[] = []; + for (let offset = 0; offset < text.length; offset += size) { + chunks.push(text.slice(offset, offset + size)); + } + return chunks; +} + +function expectEqual(actual: string | undefined, expected: string): void { + if (actual !== expected) { + throw new Error(`Expected "${expected}", parsed "${String(actual)}"`); + } +} + +function buffersPerVariant( + create: (charset: 'latin1' | 'cyrillic') => { body: string } +): Map { + const latin1 = Buffer.from(create('latin1').body, 'utf8'); + return new Map([ + ['latin1', latin1], + ['latin1-bom', Buffer.concat([UTF8_BYTE_ORDER_MARK, latin1])], + ['cyrillic', Buffer.from(create('cyrillic').body, 'utf8')], + ]); +} + +function requireBuffer( + buffers: Map, + variant: Variant +): Buffer { + const buffer = buffers.get(variant); + if (!buffer) { + throw new Error(`No fixture for ${variant}`); + } + return buffer; +} + +function assertCount(count: number): void { + if (count !== ENTRY_COUNT) { + throw new Error(`Expected ${ENTRY_COUNT} entries, parsed ${count}`); + } +} diff --git a/apps/electron-backend-e2e/src/performance/charset-parse.benchmark.ts b/apps/electron-backend-e2e/src/performance/charset-parse.benchmark.ts new file mode 100644 index 000000000..b524974a8 --- /dev/null +++ b/apps/electron-backend-e2e/src/performance/charset-parse.benchmark.ts @@ -0,0 +1,198 @@ +/** + * Plan D1: does two-byte (non-Latin-1) input slow down playlist and EPG + * parsing? Runs each workload on the `latin1` and `cyrillic` synthetic + * fixtures (identical layout) and on `latin1-bom`, the latin1 bytes behind a + * UTF-8 byte-order mark, which forces V8's two-byte representation without + * changing content. Reports P50 wall-clock and process CPU time over the + * timed iterations, and CPU-profile sample counts from a separate profiled + * pass (sampled in-process through the inspector, like `node --cpu-prof`). + * + * pnpm nx run electron-backend-e2e:benchmark-charset-parse \ + * [--iterations=5] [--warmup=1] [--sampling-interval-us=100] \ + * [--output=/absolute/report.json] + * + * To measure with Electron's V8 instead of the Node on PATH, run from + * apps/electron-backend-e2e: + * + * TSX_TSCONFIG_PATH=tsconfig.json ELECTRON_RUN_AS_NODE=1 \ + * "$(node -p "require('electron')")" --expose-gc --import tsx \ + * src/performance/charset-parse.benchmark.ts + */ +import { writeFile } from 'node:fs/promises'; +import { Session } from 'node:inspector/promises'; +import { resolve } from 'node:path'; +import { parseArgs } from 'node:util'; + +import { + CHARSET_BENCHMARK_VARIANTS, + type CharsetBenchmarkVariant, + type CharsetCaseMeasurement, + compareCharsets, + type CpuProfileLike, + formatCharsetReport, + summarizeSelfSamples, + topFrames, + variantOrderForRound, +} from './charset-parse-benchmark-report'; +import { + createWorkloads, + ENTRY_COUNT, + type Workload, +} from './charset-parse-workloads'; + +type Variant = CharsetBenchmarkVariant; + +async function main(): Promise { + const { values } = parseArgs({ + options: { + iterations: { type: 'string', default: '5' }, + warmup: { type: 'string', default: '1' }, + 'sampling-interval-us': { type: 'string', default: '100' }, + output: { type: 'string' }, + }, + }); + const iterations = positiveInteger(values.iterations, 'iterations'); + const warmup = nonNegativeInteger(values.warmup, 'warmup'); + const samplingIntervalUs = positiveInteger( + values['sampling-interval-us'], + 'sampling-interval-us' + ); + const workloads = await createWorkloads(); + const session = new Session(); + session.connect(); + await session.post('Profiler.enable'); + await session.post('Profiler.setSamplingInterval', { + interval: samplingIntervalUs, + }); + + const measurements: CharsetCaseMeasurement[] = []; + for (const workload of workloads) { + for (const variant of CHARSET_BENCHMARK_VARIANTS) { + workload.verify(variant); + } + const wall = new Map(); + const cpu = new Map(); + const profiles = new Map(); + // Variants alternate inside every round, starting with a different + // one each round, so JIT warm-up and heap growth favour no input. + // Rounds after the timed ones run under the profiler. + for (let round = 0; round < warmup + iterations * 2; round += 1) { + for (const variant of variantOrderForRound(round)) { + const run = workload.prepare(variant); + collectGarbage(); + const profiled = round >= warmup + iterations; + if (profiled) { + await session.post('Profiler.start'); + } + const cpuBefore = process.cpuUsage(); + const startedAt = performance.now(); + run(); + const elapsedMs = performance.now() - startedAt; + const cpuUsed = process.cpuUsage(cpuBefore); + if (profiled) { + const { profile } = await session.post('Profiler.stop'); + append(profiles, variant, profile); + } else if (round >= warmup) { + append(wall, variant, elapsedMs); + append( + cpu, + variant, + (cpuUsed.user + cpuUsed.system) / 1000 + ); + } + } + } + for (const variant of CHARSET_BENCHMARK_VARIANTS) { + measurements.push( + measurementFor(workload, variant, { wall, cpu, profiles }) + ); + } + } + session.disconnect(); + + const comparisons = compareCharsets(measurements); + process.stdout.write( + `Node ${process.version}, ${iterations} iterations after ${warmup} warm-up, ` + + `${ENTRY_COUNT} entries, sampling every ${samplingIntervalUs} µs\n\n` + + `${formatCharsetReport(comparisons)}\n\n` + ); + for (const measurement of measurements) { + process.stdout.write( + `${measurement.workload} [${measurement.variant}] non-Latin-1 input: ${ + measurement.hasNonLatin1Characters + }; top self frames: ${measurement.topSelfFrames + .slice(0, 5) + .map((frame) => `${frame.frame} (${frame.samples})`) + .join(', ')}\n` + ); + } + if (values.output) { + await writeFile( + resolve(values.output), + `${JSON.stringify({ node: process.version, iterations, warmup, samplingIntervalUs, comparisons, measurements }, null, 2)}\n` + ); + } +} + +interface CollectedSamples { + readonly wall: Map; + readonly cpu: Map; + readonly profiles: Map; +} + +function measurementFor( + workload: Workload, + variant: Variant, + collected: CollectedSamples +): CharsetCaseMeasurement { + const input = workload.input(variant); + const summaries = (collected.profiles.get(variant) ?? []).map((profile) => + summarizeSelfSamples(profile, Number.POSITIVE_INFINITY) + ); + const merged = new Map(); + for (const frame of summaries.flatMap((summary) => summary.top)) { + merged.set(frame.frame, (merged.get(frame.frame) ?? 0) + frame.samples); + } + return { + workload: workload.name, + variant, + inputBytes: Buffer.byteLength(input, 'utf8'), + inputLength: input.length, + // Latin-1 round-trip is lossless only when every code unit is <= 0xff. + hasNonLatin1Characters: + Buffer.from(input, 'latin1').toString('latin1') !== input, + wallClockMs: collected.wall.get(variant) ?? [], + cpuMs: collected.cpu.get(variant) ?? [], + profileSamples: summaries.reduce((sum, s) => sum + s.total, 0), + topSelfFrames: topFrames(merged, 10), + }; +} + +function append(map: Map, variant: Variant, value: T): void { + map.set(variant, [...(map.get(variant) ?? []), value]); +} + +function collectGarbage(): void { + (globalThis as { gc?: () => void }).gc?.(); +} + +function positiveInteger(value: string | undefined, name: string): number { + const parsed = nonNegativeInteger(value, name); + if (parsed < 1) { + throw new Error(`--${name} must be a positive integer`); + } + return parsed; +} + +function nonNegativeInteger(value: string | undefined, name: string): number { + const parsed = Number(value); + if (!Number.isSafeInteger(parsed) || parsed < 0) { + throw new Error(`--${name} must be a non-negative integer`); + } + return parsed; +} + +main().catch((error: unknown) => { + process.stderr.write(`${String(error)}\n`); + process.exitCode = 1; +}); diff --git a/apps/electron-backend-e2e/src/performance/synthetic-charset.ts b/apps/electron-backend-e2e/src/performance/synthetic-charset.ts new file mode 100644 index 000000000..12121324a --- /dev/null +++ b/apps/electron-backend-e2e/src/performance/synthetic-charset.ts @@ -0,0 +1,48 @@ +/** + * Title vocabulary for the synthetic M3U and XMLTV performance fixtures. + * + * V8 keeps a string one-byte (Latin-1) while every character fits in a byte + * and stores the whole string as two-byte UTF-16 once a single character + * outside Latin-1 appears. `cyrillic` fixtures exercise that two-byte path. + * + * Every Cyrillic entry has exactly the UTF-16 length of its Latin entry, so + * both variants have the same line count and the same character layout; only + * the display titles (channel, group and programme titles) differ. + */ +export const SYNTHETIC_CHARSETS = ['latin1', 'cyrillic'] as const; + +export type SyntheticCharset = (typeof SYNTHETIC_CHARSETS)[number]; + +export interface SyntheticTitleVocabulary { + readonly channel: string; + readonly group: string; + readonly language: string; + readonly programme: string; +} + +export const SYNTHETIC_TITLE_VOCABULARY: Readonly< + Record +> = Object.freeze({ + latin1: Object.freeze({ + channel: 'Synthetic Channel', + group: 'Synthetic Group', + language: 'en', + programme: 'Synthetic Programme', + }), + cyrillic: Object.freeze({ + channel: 'Пробный телеканал', + group: 'Пробная рубрика', + language: 'ru', + programme: 'Синтетическая серия', + }), +}); + +export function resolveSyntheticCharset( + charset: SyntheticCharset | undefined +): SyntheticCharset { + const resolved = charset ?? 'latin1'; + if (!SYNTHETIC_CHARSETS.includes(resolved)) { + throw new Error(`Unsupported synthetic charset: ${String(charset)}`); + } + return resolved; +} diff --git a/apps/electron-backend-e2e/src/performance/synthetic-m3u.spec.ts b/apps/electron-backend-e2e/src/performance/synthetic-m3u.spec.ts index 094f4f80d..aafc8063d 100644 --- a/apps/electron-backend-e2e/src/performance/synthetic-m3u.spec.ts +++ b/apps/electron-backend-e2e/src/performance/synthetic-m3u.spec.ts @@ -2,6 +2,7 @@ import assert from 'node:assert/strict'; import { createHash } from 'node:crypto'; import { describe, it } from 'node:test'; +import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset'; import { createSyntheticM3uFixture, SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS, @@ -14,6 +15,12 @@ const GOLDEN_SHA256 = { 100_000: '2ffe912f61ea106ba8de3414b0a927cdaa3264ac637d44c4b219f199eb94c353', } as const; +const CYRILLIC_GOLDEN_SHA256 = { + 10_000: 'bc4f3c2a3d507db459dff2d4a723d8fa8729fb957cc436f062e703564201a20b', + 50_000: '4a3fc713548a7ab50c70dc0751eb5b2483122af9d01f7d3975f309dc820c6a04', + 100_000: '94727a8452920c07727812f4ec78b77deac353f5a46ee096620efba2296b7e2b', +} as const; + describe('synthetic M3U performance fixture', () => { it('generates each supported catalog deterministically with exact metadata', () => { assert.deepEqual( @@ -47,6 +54,72 @@ describe('synthetic M3U performance fixture', () => { } }); + it('keeps the latin1 option identical to the default fixture', () => { + const byDefault = createSyntheticM3uFixture(10_000); + const explicit = createSyntheticM3uFixture(10_000, { + charset: 'latin1', + }); + + assert.equal(byDefault.charset, 'latin1'); + assert.equal(explicit.body, byDefault.body); + assert.equal(explicit.sha256, GOLDEN_SHA256[10_000]); + // UTF-8 byte length equals UTF-16 length only for ASCII-only text. + assert.equal(byDefault.bytes, byDefault.body.length); + }); + + it('generates each cyrillic catalog deterministically', () => { + for (const channelCount of SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS) { + const cyrillic = createSyntheticM3uFixture(channelCount, { + charset: 'cyrillic', + }); + + assert.equal(cyrillic.charset, 'cyrillic'); + assert.equal(cyrillic.channelCount, channelCount); + assert.equal(cyrillic.sha256, CYRILLIC_GOLDEN_SHA256[channelCount]); + } + }); + + it('keeps the cyrillic layout identical apart from titles', () => { + const latin = createSyntheticM3uFixture(10_000); + const cyrillic = createSyntheticM3uFixture(10_000, { + charset: 'cyrillic', + }); + + assert.equal(cyrillic.bytes, Buffer.byteLength(cyrillic.body, 'utf8')); + assert.equal( + cyrillic.sha256, + createHash('sha256').update(cyrillic.body, 'utf8').digest('hex') + ); + assert.equal(cyrillic.body.length, latin.body.length); + assert.equal(toLatinTitles(cyrillic.body), latin.body); + assertOnlyLoopbackUrls(cyrillic.body); + }); + + it('puts characters outside Latin-1 into every cyrillic channel entry', () => { + const lines = createSyntheticM3uFixture(10_000, { charset: 'cyrillic' }) + .body.trimEnd() + .split('\n'); + + for (const line of lines.filter((entry) => + entry.startsWith('#EXTINF:') + )) { + assert.ok(hasCharacterAbove(line, 0xff), line); + } + for (const line of lines.filter((entry) => entry.startsWith('http'))) { + assert.equal(hasCharacterAbove(line, 0x7f), false, line); + } + }); + + it('rejects unsupported charsets', () => { + assert.throws( + () => + createSyntheticM3uFixture(10_000, { + charset: 'greek' as never, + }), + /Unsupported synthetic charset/ + ); + }); + it('rejects unsupported channel counts', () => { assert.throws( () => createSyntheticM3uFixture(9_999), @@ -55,6 +128,29 @@ describe('synthetic M3U performance fixture', () => { }); }); +function toLatinTitles(body: string): string { + const latin = SYNTHETIC_TITLE_VOCABULARY.latin1; + const cyrillic = SYNTHETIC_TITLE_VOCABULARY.cyrillic; + return replaceEvery( + replaceEvery(body, cyrillic.group, latin.group), + cyrillic.channel, + latin.channel + ); +} + +function hasCharacterAbove(text: string, maxCodeUnit: number): boolean { + for (let index = 0; index < text.length; index += 1) { + if (text.charCodeAt(index) > maxCodeUnit) { + return true; + } + } + return false; +} + +function replaceEvery(text: string, search: string, replacement: string) { + return text.split(search).join(replacement); +} + function assertOnlyLoopbackUrls(body: string): void { const urls = body.match(/https?:\/\/[^\s"]+/g) ?? []; diff --git a/apps/electron-backend-e2e/src/performance/synthetic-m3u.ts b/apps/electron-backend-e2e/src/performance/synthetic-m3u.ts index 1cdf8d18b..3c52d2b0d 100644 --- a/apps/electron-backend-e2e/src/performance/synthetic-m3u.ts +++ b/apps/electron-backend-e2e/src/performance/synthetic-m3u.ts @@ -1,5 +1,11 @@ import { createHash } from 'node:crypto'; +import { + resolveSyntheticCharset, + type SyntheticCharset, + SYNTHETIC_TITLE_VOCABULARY, +} from './synthetic-charset'; + export const SYNTHETIC_M3U_SEED = 240_724; export const SYNTHETIC_M3U_CHANNEL_COUNT = 100_000; export const SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS = [ @@ -15,13 +21,22 @@ export interface SyntheticM3uFixture { readonly body: string; readonly bytes: number; readonly channelCount: SyntheticM3uChannelCount; + readonly charset: SyntheticCharset; readonly sha256: string; } +export interface SyntheticM3uFixtureOptions { + /** Script of channel names and group titles; defaults to `latin1`. */ + readonly charset?: SyntheticCharset; +} + export function createSyntheticM3uFixture( - channelCount: number = SYNTHETIC_M3U_CHANNEL_COUNT + channelCount: number = SYNTHETIC_M3U_CHANNEL_COUNT, + options: SyntheticM3uFixtureOptions = {} ): SyntheticM3uFixture { assertSupportedChannelCount(channelCount); + const charset = resolveSyntheticCharset(options.charset); + const titles = SYNTHETIC_TITLE_VOCABULARY[charset]; const lines = new Array(1 + channelCount * 2); lines[0] = '#EXTM3U'; @@ -31,12 +46,12 @@ export function createSyntheticM3uFixture( const lineOffset = 1 + offset * 2; const stableIndex = String(index).padStart(6, '0'); - lines[lineOffset] = `#EXTINF:-1 group-title="Synthetic Group ${String( + lines[lineOffset] = `#EXTINF:-1 group-title="${titles.group} ${String( group ).padStart( 3, '0' - )}",Synthetic Channel ${SYNTHETIC_M3U_SEED}-${stableIndex}`; + )}",${titles.channel} ${SYNTHETIC_M3U_SEED}-${stableIndex}`; lines[lineOffset + 1] = `http://127.0.0.1/stream/${SYNTHETIC_M3U_SEED}/${index}`; } @@ -46,6 +61,7 @@ export function createSyntheticM3uFixture( body, bytes: Buffer.byteLength(body, 'utf8'), channelCount, + charset, sha256: createHash('sha256').update(body, 'utf8').digest('hex'), }); } diff --git a/apps/electron-backend-e2e/src/performance/synthetic-xmltv.spec.ts b/apps/electron-backend-e2e/src/performance/synthetic-xmltv.spec.ts new file mode 100644 index 000000000..4e59ffd10 --- /dev/null +++ b/apps/electron-backend-e2e/src/performance/synthetic-xmltv.spec.ts @@ -0,0 +1,156 @@ +import assert from 'node:assert/strict'; +import { createHash } from 'node:crypto'; +import { describe, it } from 'node:test'; + +import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset'; +import { + createSyntheticXmltvFixture, + SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS, + SYNTHETIC_XMLTV_PROGRAMME_COUNT, + SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL, +} from './synthetic-xmltv'; + +const GOLDEN_SHA256 = { + latin1: { + 10_000: 'e9d0ca0f7b73efc5cdee42cb2c4ee5bb4564b274b5042b13f8cd0d965d5011bc', + 50_000: '08b955c84822b7eb0d825a73eef6a6abc88b6d372c2dd32ad25d3969d667fd49', + 100_000: + '932f0ba7089cab9e205a78142a245ebb497826ee7f1f4819603d77eba8e95a9e', + }, + cyrillic: { + 10_000: '5b0df408e90c77bebe296fe1cc7e97cec5b929ebc7ad21b7480a6849a29f912b', + 50_000: 'bb1a0ca9fe874ab847b062c768868d76da750366aaa680bc712cb4cb7bfbf809', + 100_000: + 'cd637f4a25d47592116417a64bbb8c5b48d87f5bb5b0b4693f7af6c7603a8859', + }, +} as const; + +describe('synthetic XMLTV performance fixture', () => { + it('generates each supported schedule deterministically', () => { + assert.deepEqual( + SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS, + [10_000, 50_000, 100_000] + ); + assert.equal(SYNTHETIC_XMLTV_PROGRAMME_COUNT, 50_000); + + for (const charset of ['latin1', 'cyrillic'] as const) { + for (const programmeCount of SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS) { + const fixture = createSyntheticXmltvFixture(programmeCount, { + charset, + }); + + assert.equal(fixture.charset, charset); + assert.equal(fixture.programmeCount, programmeCount); + assert.equal( + fixture.channelCount, + programmeCount / SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL + ); + assert.equal( + fixture.sha256, + GOLDEN_SHA256[charset][programmeCount] + ); + } + } + }); + + it('reports exact metadata and element counts', () => { + for (const charset of ['latin1', 'cyrillic'] as const) { + const fixture = createSyntheticXmltvFixture(10_000, { charset }); + + assert.equal( + fixture.bytes, + Buffer.byteLength(fixture.body, 'utf8') + ); + assert.equal( + fixture.sha256, + createHash('sha256').update(fixture.body, 'utf8').digest('hex') + ); + assert.equal(countOf(fixture.body, '\n')); + } + }); + + it('defaults to 50k programmes with an ASCII-only latin1 body', () => { + const fixture = createSyntheticXmltvFixture(); + + assert.equal(fixture.charset, 'latin1'); + assert.equal(fixture.programmeCount, 50_000); + // UTF-8 byte length equals UTF-16 length only for ASCII-only text. + assert.equal(fixture.bytes, fixture.body.length); + }); + + it('keeps the cyrillic layout identical apart from titles', () => { + const latin = createSyntheticXmltvFixture(10_000); + const cyrillic = createSyntheticXmltvFixture(10_000, { + charset: 'cyrillic', + }); + const latinTitles = SYNTHETIC_TITLE_VOCABULARY.latin1; + const cyrillicTitles = SYNTHETIC_TITLE_VOCABULARY.cyrillic; + + assert.equal(cyrillic.body.length, latin.body.length); + const translated = [ + [cyrillicTitles.channel, latinTitles.channel], + [cyrillicTitles.programme, latinTitles.programme], + [ + `lang="${cyrillicTitles.language}"`, + `lang="${latinTitles.language}"`, + ], + ].reduce( + (text, [search, replacement]) => + text.split(search).join(replacement), + cyrillic.body + ); + assert.equal(translated, latin.body); + const firstTitle = cyrillic.body.indexOf(''); + assert.ok(firstTitle > 0); + assert.ok( + cyrillic.body.charCodeAt(firstTitle + '<title lang="ru">'.length) > + 0xff + ); + }); + + it('schedules consecutive half-hour programmes per channel', () => { + const lines = createSyntheticXmltvFixture(10_000).body.split('\n'); + const firstProgramme = lines.find((line) => + line.includes('<programme ') + ); + + assert.equal( + firstProgramme, + ' <programme start="20260101000000 +0000" stop="20260101003000 +0000" channel="synthetic.000001"><title lang="en">Synthetic Programme 000001-0001Synthetic description for slot 1.Synthetic' + ); + assert.ok( + lines.includes( + ' Synthetic Programme 000100-0100Synthetic description for slot 100.Synthetic' + ) + ); + }); + + it('rejects unsupported programme counts and charsets', () => { + assert.throws( + () => createSyntheticXmltvFixture(12_345), + /Unsupported synthetic XMLTV programme count/ + ); + assert.throws( + () => + createSyntheticXmltvFixture(10_000, { + charset: 'arabic' as never, + }), + /Unsupported synthetic charset/ + ); + }); +}); + +function countOf(body: string, needle: string): number { + let count = 0; + for ( + let index = body.indexOf(needle); + index !== -1; + index = body.indexOf(needle, index + needle.length) + ) { + count += 1; + } + return count; +} diff --git a/apps/electron-backend-e2e/src/performance/synthetic-xmltv.ts b/apps/electron-backend-e2e/src/performance/synthetic-xmltv.ts new file mode 100644 index 000000000..e19e62771 --- /dev/null +++ b/apps/electron-backend-e2e/src/performance/synthetic-xmltv.ts @@ -0,0 +1,138 @@ +import { createHash } from 'node:crypto'; + +import { + resolveSyntheticCharset, + type SyntheticCharset, + SYNTHETIC_TITLE_VOCABULARY, +} from './synthetic-charset'; +import { SYNTHETIC_M3U_SEED } from './synthetic-m3u'; + +export const SYNTHETIC_XMLTV_PROGRAMME_COUNT = 50_000; +export const SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL = 100; +export const SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS = [ + 10_000, + SYNTHETIC_XMLTV_PROGRAMME_COUNT, + 100_000, +] as const; + +/** 2026-01-01T00:00:00Z; every programme lasts 30 minutes. */ +const SCHEDULE_START_EPOCH_MS = Date.UTC(2026, 0, 1); +const PROGRAMME_DURATION_MS = 30 * 60 * 1000; + +export type SyntheticXmltvProgrammeCount = + (typeof SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS)[number]; + +export interface SyntheticXmltvFixture { + readonly body: string; + readonly bytes: number; + readonly channelCount: number; + readonly charset: SyntheticCharset; + readonly programmeCount: SyntheticXmltvProgrammeCount; + readonly sha256: string; +} + +export interface SyntheticXmltvFixtureOptions { + /** Script of channel display names and programme titles. */ + readonly charset?: SyntheticCharset; +} + +/** + * Deterministic XMLTV document: all `` entries first, then + * `SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL` consecutive programmes per + * channel. Descriptions, categories and attributes stay ASCII in every + * charset so only the titles decide the string encoding. + */ +export function createSyntheticXmltvFixture( + programmeCount: number = SYNTHETIC_XMLTV_PROGRAMME_COUNT, + options: SyntheticXmltvFixtureOptions = {} +): SyntheticXmltvFixture { + assertSupportedProgrammeCount(programmeCount); + const charset = resolveSyntheticCharset(options.charset); + const titles = SYNTHETIC_TITLE_VOCABULARY[charset]; + const channelCount = + programmeCount / SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL; + const lines: string[] = [ + '', + '', + ]; + + for (let offset = 0; offset < channelCount; offset += 1) { + const channelId = syntheticChannelId(offset); + lines.push( + ` ${ + titles.channel + } ${SYNTHETIC_M3U_SEED}-${stableNumber(offset + 1, 6)}` + ); + } + + const slotTimestamps = Array.from( + { length: SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL + 1 }, + (_, slot) => + xmltvTimestamp( + SCHEDULE_START_EPOCH_MS + slot * PROGRAMME_DURATION_MS + ) + ); + for (let offset = 0; offset < channelCount; offset += 1) { + const channelId = syntheticChannelId(offset); + const channelNumber = stableNumber(offset + 1, 6); + for ( + let slot = 0; + slot < SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL; + slot += 1 + ) { + lines.push( + ` ${ + titles.programme + } ${channelNumber}-${stableNumber( + slot + 1, + 4 + )}Synthetic description for slot ${ + slot + 1 + }.Synthetic` + ); + } + } + + lines.push(''); + const body = `${lines.join('\n')}\n`; + return Object.freeze({ + body, + bytes: Buffer.byteLength(body, 'utf8'), + channelCount, + charset, + programmeCount, + sha256: createHash('sha256').update(body, 'utf8').digest('hex'), + }); +} + +function syntheticChannelId(offset: number): string { + return `synthetic.${stableNumber(offset + 1, 6)}`; +} + +function stableNumber(value: number, width: number): string { + return String(value).padStart(width, '0'); +} + +function xmltvTimestamp(epochMs: number): string { + const iso = new Date(epochMs).toISOString(); + return `${iso.slice(0, 4)}${iso.slice(5, 7)}${iso.slice(8, 10)}${iso.slice( + 11, + 13 + )}${iso.slice(14, 16)}${iso.slice(17, 19)} +0000`; +} + +function assertSupportedProgrammeCount( + programmeCount: number +): asserts programmeCount is SyntheticXmltvProgrammeCount { + if ( + !SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS.some( + (supported) => supported === programmeCount + ) + ) { + throw new Error( + `Unsupported synthetic XMLTV programme count: ${programmeCount}` + ); + } +} diff --git a/docs/architecture/performance-journeys.md b/docs/architecture/performance-journeys.md index 2a698b00b..2b5a77b15 100644 --- a/docs/architecture/performance-journeys.md +++ b/docs/architecture/performance-journeys.md @@ -382,6 +382,43 @@ in all eighteen runner iterations; the `spawnToFirstCardMs` P50 ranged from differ from a Mac (12 and 571 there, the fast path without the Linux-only `getWindowState` call), so take J1 baseline values from the runner only. +## Charset parse benchmark + +V8 stores a string as two-byte UTF-16 once one character falls outside +Latin-1, and substrings of such a string stay two-byte, even ASCII-only URL +lines. `src/performance/charset-parse.benchmark.ts` checks whether that slows +playlist and EPG parsing. It is a Node benchmark, not a journey, and is not +ratcheted: + +```bash +pnpm nx run electron-backend-e2e:benchmark-charset-parse --iterations=5 +``` + +It parses 50,000 M3U channels (`iptv-playlist-parser`, then +`createPlaylistObject`, the main-process `PARSE_M3U` and `NORMALIZE` phases) +and 50,000 XMLTV programmes (`StreamingEpgParser`, the EPG worker's parser). +Each workload runs on three inputs: `latin1` and `cyrillic` from the +synthetic generators (`charset` option of `synthetic-m3u.ts` and +`synthetic-xmltv.ts`, identical layout apart from titles), and `latin1-bom`, +the latin1 bytes behind a UTF-8 byte-order mark. The BOM forces two-byte +storage without changing content, which separates the encoding cost from +the effect that non-ASCII titles have on ASCII-only regexes. The XMLTV +parser receives 64 Ki-character slices of one decoded string rather than +per-chunk decoded buffers: slices keep the input's representation (a +per-chunk decode would make the BOM control one-byte after its first +chunk), and no multi-byte character is split. Before timing, an untimed +pass checks that the parsed titles match the fixture. + +The report gives P50 wall-clock and CPU time after one warm-up, plus +CPU-profile sample counts and top self frames from a separate profiled pass. +Inputs alternate within each round and the starting input rotates between +rounds. Prefer CPU time and samples on a busy machine. + +The 2026-09-27 measurement (plan item D1) found every workload under the 1.5x +threshold on Node 22 and inside Electron 43, so D2 regex prefilters were not +applied. Rerun the benchmark after changing either parser or when a user +reports slow imports of non-Latin playlists. + ## Adding a counter 1. Produce the value from the built output or from a deterministic probe, not