test(performance): measure two-byte string cost in M3U and XMLTV parsing (D1) (#1713)

This commit is contained in:
4gray authored and GitHub committed 2026-09-28 00:03:18 +02:00
1 parent 363411542f
commit 8b570e3390
11 files changed
+1252 -3

No files matched your search

+9
View File
@@ -49,6 +49,15 @@
"command": "pnpm exec playwright test --config=playwright.xtream-performance.config.ts src/xtream.performance.ts"
}
},
"benchmark-charset-parse": {
"executor": "nx:run-commands",
"cache": false,
"parallelism": false,
"options": {
"cwd": "apps/electron-backend-e2e",
"command": "pnpm exec tsx --expose-gc --tsconfig tsconfig.json src/performance/charset-parse.benchmark.ts"
}
},
"journeys": {
"dependsOn": ["electron-backend:build-performance"],
"executor": "nx:run-commands",
@@ -0,0 +1,154 @@
import assert from 'node:assert/strict';
import { describe, it } from 'node:test';
import {
CHARSET_BENCHMARK_VARIANTS,
type CharsetBenchmarkVariant,
type CharsetCaseMeasurement,
compareCharsets,
formatCharsetReport,
medianOf,
summarizeSelfSamples,
variantOrderForRound,
} from './charset-parse-benchmark-report';
function measurement(
variant: CharsetBenchmarkVariant,
cpuMs: number[],
profileSamples: number,
workload = 'm3u parse'
): CharsetCaseMeasurement {
return {
workload,
variant,
inputBytes: 10,
inputLength: 10,
hasNonLatin1Characters: variant !== 'latin1',
wallClockMs: cpuMs.map((value) => value * 2),
cpuMs,
profileSamples,
topSelfFrames: [],
};
}
describe('charset parse benchmark report', () => {
it('rotates the starting variant every round', () => {
assert.deepEqual(variantOrderForRound(0), [
'latin1',
'latin1-bom',
'cyrillic',
]);
assert.deepEqual(variantOrderForRound(1), [
'latin1-bom',
'cyrillic',
'latin1',
]);
assert.deepEqual(variantOrderForRound(2), [
'cyrillic',
'latin1',
'latin1-bom',
]);
assert.deepEqual(variantOrderForRound(3), variantOrderForRound(0));
for (let round = 0; round < 6; round += 1) {
assert.deepEqual(
[...variantOrderForRound(round)].sort(),
[...CHARSET_BENCHMARK_VARIANTS].sort()
);
}
});
it('takes the median of odd and even samples without mutating them', () => {
const values = [5, 1, 3];
assert.equal(medianOf(values), 3);
assert.deepEqual(values, [5, 1, 3]);
assert.equal(medianOf([4, 1, 3, 2]), 2.5);
assert.throws(() => medianOf([]), /empty sample/);
});
it('counts self samples per frame, most frequent first', () => {
const summary = summarizeSelfSamples(
{
nodes: [
{
id: 1,
callFrame: {
functionName: '',
url: '',
lineNumber: -1,
},
},
{
id: 2,
callFrame: {
functionName: 'scanAttributes',
url: 'file:///repo/node_modules/iptv-playlist-parser/src/index.js',
lineNumber: 42,
},
},
],
samples: [2, 2, 1, 2, 9],
},
2
);
assert.equal(summary.total, 5);
assert.deepEqual(summary.top, [
{ frame: 'scanAttributes index.js:43', samples: 3 },
{ frame: '(anonymous)', samples: 1 },
]);
});
it('compares every variant with latin1 and flags only slowdowns seen by both signals', () => {
const [comparison] = compareCharsets([
measurement('latin1', [10, 12, 11], 100),
measurement('latin1-bom', [16, 17, 18], 140),
measurement('cyrillic', [16, 17, 18], 160),
]);
assert.equal(comparison.workload, 'm3u parse');
assert.deepEqual(
comparison.variants.map((row) => row.variant),
[...CHARSET_BENCHMARK_VARIANTS]
);
const [latin1, bom, cyrillic] = comparison.variants;
assert.equal(latin1.cpuRatio, 1);
assert.equal(latin1.exceedsThreshold, false);
assert.equal(bom.cpuP50Ms, 17);
assert.equal(bom.wallP50Ms, 34);
assert.ok(Math.abs(bom.cpuRatio - 17 / 11) < 1e-9);
// CPU time is over 1.5x, but profile samples (1.4x) are not.
assert.equal(bom.exceedsThreshold, false);
assert.equal(cyrillic.sampleRatio, 1.6);
assert.equal(cyrillic.exceedsThreshold, true);
});
it('fails when a variant was not measured', () => {
assert.throws(
() =>
compareCharsets([
measurement('latin1', [1], 1),
measurement('cyrillic', [1], 1),
]),
/Missing latin1-bom measurement for m3u parse/
);
});
it('formats one markdown row per workload and variant', () => {
const report = formatCharsetReport(
compareCharsets([
measurement('latin1', [10], 100),
measurement('latin1-bom', [11], 100),
measurement('cyrillic', [20], 200),
])
);
const lines = report.split('\n');
assert.equal(lines.length, 5);
assert.match(lines[0], /^\| Workload \| Input \|/);
assert.equal(
lines[4],
'| m3u parse | cyrillic | 40.0 | 20.0 | 200 | 2.00x | 2.00x | 2.00x | yes |'
);
});
});
@@ -0,0 +1,206 @@
/** Two-byte input slower than this ratio justifies parser changes (plan D1). */
export const CHARSET_SLOWDOWN_THRESHOLD = 1.5;
/**
* Input variants of the charset benchmark. `latin1-bom` is the `latin1`
* fixture behind a UTF-8 byte-order mark: same content, but V8 must store
* the decoded string as two-byte, which isolates the encoding effect from
* the content effect that Cyrillic titles have on ASCII-only regexes.
*/
export const CHARSET_BENCHMARK_VARIANTS = [
'latin1',
'latin1-bom',
'cyrillic',
] as const;
export type CharsetBenchmarkVariant =
(typeof CHARSET_BENCHMARK_VARIANTS)[number];
export interface CpuProfileNode {
readonly id: number;
readonly callFrame: {
readonly functionName: string;
readonly url: string;
readonly lineNumber: number;
};
}
export interface CpuProfileLike {
readonly nodes: readonly CpuProfileNode[];
readonly samples?: readonly number[];
}
export interface SelfSampleFrame {
readonly frame: string;
readonly samples: number;
}
export interface CharsetCaseMeasurement {
readonly workload: string;
readonly variant: CharsetBenchmarkVariant;
readonly inputBytes: number;
readonly inputLength: number;
/** True when V8 must store the decoded input as a two-byte string. */
readonly hasNonLatin1Characters: boolean;
readonly wallClockMs: readonly number[];
readonly cpuMs: readonly number[];
readonly profileSamples: number;
readonly topSelfFrames: readonly SelfSampleFrame[];
}
export interface CharsetVariantSummary {
readonly variant: CharsetBenchmarkVariant;
readonly wallP50Ms: number;
readonly cpuP50Ms: number;
readonly profileSamples: number;
readonly wallRatio: number;
readonly cpuRatio: number;
readonly sampleRatio: number;
/** Slower than the threshold on CPU time and on profile samples. */
readonly exceedsThreshold: boolean;
}
export interface CharsetComparison {
readonly workload: string;
readonly variants: readonly CharsetVariantSummary[];
}
/**
* Variant order for one benchmark round. The starting variant rotates every
* round so no input always runs first or last after a garbage collection.
*/
export function variantOrderForRound(round: number): CharsetBenchmarkVariant[] {
const variants = [...CHARSET_BENCHMARK_VARIANTS];
const offset = round % variants.length;
return [...variants.slice(offset), ...variants.slice(0, offset)];
}
export function medianOf(values: readonly number[]): number {
if (values.length === 0) {
throw new Error('Cannot take the median of an empty sample');
}
const sorted = [...values].sort((left, right) => left - right);
const middle = Math.floor(sorted.length / 2);
return sorted.length % 2 === 1
? sorted[middle]
: (sorted[middle - 1] + sorted[middle]) / 2;
}
/** Counts self samples per frame; the top frames explain where time went. */
export function summarizeSelfSamples(
profile: CpuProfileLike,
limit = 8
): { total: number; top: SelfSampleFrame[] } {
const nodesById = new Map(profile.nodes.map((node) => [node.id, node]));
const counts = new Map<string, number>();
const samples = profile.samples ?? [];
for (const nodeId of samples) {
const frame = describeFrame(nodesById.get(nodeId));
counts.set(frame, (counts.get(frame) ?? 0) + 1);
}
return { total: samples.length, top: topFrames(counts, limit) };
}
export function topFrames(
counts: ReadonlyMap<string, number>,
limit: number
): SelfSampleFrame[] {
return [...counts.entries()]
.map(([frame, samples]) => ({ frame, samples }))
.sort(
(left, right) =>
right.samples - left.samples ||
left.frame.localeCompare(right.frame)
)
.slice(0, limit);
}
export function compareCharsets(
measurements: readonly CharsetCaseMeasurement[]
): CharsetComparison[] {
const workloads = [...new Set(measurements.map((m) => m.workload))];
return workloads.map((workload) => {
const baseline = findCase(measurements, workload, 'latin1');
const baselineWall = medianOf(baseline.wallClockMs);
const baselineCpu = medianOf(baseline.cpuMs);
return {
workload,
variants: CHARSET_BENCHMARK_VARIANTS.map((variant) => {
const measurement = findCase(measurements, workload, variant);
const wallP50Ms = medianOf(measurement.wallClockMs);
const cpuP50Ms = medianOf(measurement.cpuMs);
const cpuRatio = cpuP50Ms / baselineCpu;
const sampleRatio =
measurement.profileSamples / baseline.profileSamples;
return {
variant,
wallP50Ms,
cpuP50Ms,
profileSamples: measurement.profileSamples,
wallRatio: wallP50Ms / baselineWall,
cpuRatio,
sampleRatio,
exceedsThreshold:
cpuRatio > CHARSET_SLOWDOWN_THRESHOLD &&
sampleRatio > CHARSET_SLOWDOWN_THRESHOLD,
};
}),
};
});
}
export function formatCharsetReport(
comparisons: readonly CharsetComparison[]
): string {
const header =
'| Workload | Input | Wall P50 ms | CPU P50 ms | Samples | Wall ratio | CPU ratio | Sample ratio | Over 1.5x |';
const divider =
'| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | --- |';
const rows = comparisons.flatMap((comparison) =>
comparison.variants.map((row) =>
[
comparison.workload,
row.variant,
row.wallP50Ms.toFixed(1),
row.cpuP50Ms.toFixed(1),
String(row.profileSamples),
`${row.wallRatio.toFixed(2)}x`,
`${row.cpuRatio.toFixed(2)}x`,
`${row.sampleRatio.toFixed(2)}x`,
row.exceedsThreshold ? 'yes' : 'no',
].join(' | ')
)
);
return [header, divider, ...rows.map((row) => `| ${row} |`)].join('\n');
}
function findCase(
measurements: readonly CharsetCaseMeasurement[],
workload: string,
variant: CharsetBenchmarkVariant
): CharsetCaseMeasurement {
const match = measurements.find(
(m) => m.workload === workload && m.variant === variant
);
if (!match) {
throw new Error(`Missing ${variant} measurement for ${workload}`);
}
return match;
}
function describeFrame(node: CpuProfileNode | undefined): string {
if (!node) {
return '(unknown)';
}
const { functionName, url, lineNumber } = node.callFrame;
const name = functionName || '(anonymous)';
if (!url) {
return name;
}
const file = url.slice(url.lastIndexOf('/') + 1);
return `${name} ${file}:${lineNumber + 1}`;
}
@@ -0,0 +1,191 @@
import { createPlaylistObject } from '@iptvnator/shared/m3u-utils';
import { parse } from 'iptv-playlist-parser';
import { resolve } from 'node:path';
import type { CharsetBenchmarkVariant } from './charset-parse-benchmark-report';
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
import { createSyntheticM3uFixture, SYNTHETIC_M3U_SEED } from './synthetic-m3u';
import { createSyntheticXmltvFixture } from './synthetic-xmltv';
export const ENTRY_COUNT = 50_000;
/**
* The EPG worker writes 64 KiB decoded chunks. The benchmark writes slices of
* one decoded string instead, so every slice keeps the input's one-byte or
* two-byte representation (decoding each chunk separately would make the
* BOM control one-byte after its first chunk) and no character is split.
*/
const XMLTV_CHUNK_CHARS = 64 * 1024;
const REPLACEMENT_CHARACTER = '\ufffd';
const UTF8_BYTE_ORDER_MARK = Buffer.from([0xef, 0xbb, 0xbf]);
type Variant = CharsetBenchmarkVariant;
interface ParsedProgramme {
readonly title: readonly { readonly value: string }[];
}
interface StreamingEpgParserModule {
StreamingEpgParser: new (
onChannels: (channels: unknown[]) => void,
onPrograms: (programs: ParsedProgramme[]) => void,
onProgress: (channels: number, programs: number) => void,
channelBatchSize?: number,
programBatchSize?: number
) => { write(chunk: string): void; finish(): { totalPrograms: number } };
}
export interface Workload {
readonly name: string;
/** Prepares untimed input, then returns the timed operation. */
prepare(variant: Variant): () => void;
input(variant: Variant): string;
/** Untimed check that the parser saw the fixture's exact titles. */
verify(variant: Variant): void;
}
export async function createWorkloads(): Promise<Workload[]> {
const epgModule = (await import(
resolve(
__dirname,
'../../../electron-backend/src/app/workers/epg-streaming-parser.ts'
)
)) as StreamingEpgParserModule;
const m3uBuffers = buffersPerVariant((charset) =>
createSyntheticM3uFixture(ENTRY_COUNT, { charset })
);
const xmltvBuffers = buffersPerVariant((charset) =>
createSyntheticXmltvFixture(ENTRY_COUNT, { charset })
);
const m3uInput = (variant: Variant) =>
requireBuffer(m3uBuffers, variant).toString('utf8');
const xmltvInput = (variant: Variant) =>
requireBuffer(xmltvBuffers, variant).toString('utf8');
const verifyM3u = (variant: Variant) => {
const titles = titlesFor(variant);
const [first] = parse(m3uInput(variant)).items;
expectEqual(
first?.name,
`${titles.channel} ${SYNTHETIC_M3U_SEED}-000001`
);
expectEqual(first?.group.title, `${titles.group} 001`);
};
const parseXmltv = (
variant: Variant,
onPrograms: (programs: ParsedProgramme[]) => void
) => {
const chunks = sliceText(xmltvInput(variant), XMLTV_CHUNK_CHARS);
return () => {
const parser = new epgModule.StreamingEpgParser(
() => undefined,
onPrograms,
() => undefined,
100,
1000
);
for (const chunk of chunks) {
parser.write(chunk);
}
assertCount(parser.finish().totalPrograms);
};
};
return [
{
// PARSE_M3U phase of the main-process import.
name: 'm3u parse',
input: m3uInput,
verify: verifyM3u,
prepare(variant) {
const body = m3uInput(variant);
return () => assertCount(parse(body).items.length);
},
},
{
// NORMALIZE phase of the main-process import.
name: 'm3u normalize',
input: m3uInput,
verify: verifyM3u,
prepare(variant) {
const parsed = parse(m3uInput(variant));
return () =>
assertCount(
createPlaylistObject('benchmark', parsed, 'x', 'URL')
.count
);
},
},
{
// EPG worker's saxes-based parser; UTF-8 decoding is excluded.
name: 'xmltv stream parse',
input: xmltvInput,
verify(variant) {
const titles: string[] = [];
parseXmltv(variant, (programs) => {
for (const programme of programs) {
titles.push(programme.title[0]?.value ?? '');
}
})();
const programme = titlesFor(variant).programme;
expectEqual(titles[0], `${programme} 000001-0001`);
expectEqual(titles.at(-1), `${programme} 000500-0100`);
if (
titles.some((title) =>
title.includes(REPLACEMENT_CHARACTER)
)
) {
throw new Error(`${variant} XMLTV titles contain U+FFFD`);
}
},
prepare: (variant) => parseXmltv(variant, () => undefined),
},
];
}
function titlesFor(variant: Variant) {
return SYNTHETIC_TITLE_VOCABULARY[
variant === 'cyrillic' ? 'cyrillic' : 'latin1'
];
}
function sliceText(text: string, size: number): string[] {
const chunks: string[] = [];
for (let offset = 0; offset < text.length; offset += size) {
chunks.push(text.slice(offset, offset + size));
}
return chunks;
}
function expectEqual(actual: string | undefined, expected: string): void {
if (actual !== expected) {
throw new Error(`Expected "${expected}", parsed "${String(actual)}"`);
}
}
function buffersPerVariant(
create: (charset: 'latin1' | 'cyrillic') => { body: string }
): Map<Variant, Buffer> {
const latin1 = Buffer.from(create('latin1').body, 'utf8');
return new Map<Variant, Buffer>([
['latin1', latin1],
['latin1-bom', Buffer.concat([UTF8_BYTE_ORDER_MARK, latin1])],
['cyrillic', Buffer.from(create('cyrillic').body, 'utf8')],
]);
}
function requireBuffer(
buffers: Map<Variant, Buffer>,
variant: Variant
): Buffer {
const buffer = buffers.get(variant);
if (!buffer) {
throw new Error(`No fixture for ${variant}`);
}
return buffer;
}
function assertCount(count: number): void {
if (count !== ENTRY_COUNT) {
throw new Error(`Expected ${ENTRY_COUNT} entries, parsed ${count}`);
}
}
@@ -0,0 +1,198 @@
/**
* Plan D1: does two-byte (non-Latin-1) input slow down playlist and EPG
* parsing? Runs each workload on the `latin1` and `cyrillic` synthetic
* fixtures (identical layout) and on `latin1-bom`, the latin1 bytes behind a
* UTF-8 byte-order mark, which forces V8's two-byte representation without
* changing content. Reports P50 wall-clock and process CPU time over the
* timed iterations, and CPU-profile sample counts from a separate profiled
* pass (sampled in-process through the inspector, like `node --cpu-prof`).
*
* pnpm nx run electron-backend-e2e:benchmark-charset-parse \
* [--iterations=5] [--warmup=1] [--sampling-interval-us=100] \
* [--output=/absolute/report.json]
*
* To measure with Electron's V8 instead of the Node on PATH, run from
* apps/electron-backend-e2e:
*
* TSX_TSCONFIG_PATH=tsconfig.json ELECTRON_RUN_AS_NODE=1 \
* "$(node -p "require('electron')")" --expose-gc --import tsx \
* src/performance/charset-parse.benchmark.ts
*/
import { writeFile } from 'node:fs/promises';
import { Session } from 'node:inspector/promises';
import { resolve } from 'node:path';
import { parseArgs } from 'node:util';
import {
CHARSET_BENCHMARK_VARIANTS,
type CharsetBenchmarkVariant,
type CharsetCaseMeasurement,
compareCharsets,
type CpuProfileLike,
formatCharsetReport,
summarizeSelfSamples,
topFrames,
variantOrderForRound,
} from './charset-parse-benchmark-report';
import {
createWorkloads,
ENTRY_COUNT,
type Workload,
} from './charset-parse-workloads';
type Variant = CharsetBenchmarkVariant;
async function main(): Promise<void> {
const { values } = parseArgs({
options: {
iterations: { type: 'string', default: '5' },
warmup: { type: 'string', default: '1' },
'sampling-interval-us': { type: 'string', default: '100' },
output: { type: 'string' },
},
});
const iterations = positiveInteger(values.iterations, 'iterations');
const warmup = nonNegativeInteger(values.warmup, 'warmup');
const samplingIntervalUs = positiveInteger(
values['sampling-interval-us'],
'sampling-interval-us'
);
const workloads = await createWorkloads();
const session = new Session();
session.connect();
await session.post('Profiler.enable');
await session.post('Profiler.setSamplingInterval', {
interval: samplingIntervalUs,
});
const measurements: CharsetCaseMeasurement[] = [];
for (const workload of workloads) {
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
workload.verify(variant);
}
const wall = new Map<Variant, number[]>();
const cpu = new Map<Variant, number[]>();
const profiles = new Map<Variant, CpuProfileLike[]>();
// Variants alternate inside every round, starting with a different
// one each round, so JIT warm-up and heap growth favour no input.
// Rounds after the timed ones run under the profiler.
for (let round = 0; round < warmup + iterations * 2; round += 1) {
for (const variant of variantOrderForRound(round)) {
const run = workload.prepare(variant);
collectGarbage();
const profiled = round >= warmup + iterations;
if (profiled) {
await session.post('Profiler.start');
}
const cpuBefore = process.cpuUsage();
const startedAt = performance.now();
run();
const elapsedMs = performance.now() - startedAt;
const cpuUsed = process.cpuUsage(cpuBefore);
if (profiled) {
const { profile } = await session.post('Profiler.stop');
append(profiles, variant, profile);
} else if (round >= warmup) {
append(wall, variant, elapsedMs);
append(
cpu,
variant,
(cpuUsed.user + cpuUsed.system) / 1000
);
}
}
}
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
measurements.push(
measurementFor(workload, variant, { wall, cpu, profiles })
);
}
}
session.disconnect();
const comparisons = compareCharsets(measurements);
process.stdout.write(
`Node ${process.version}, ${iterations} iterations after ${warmup} warm-up, ` +
`${ENTRY_COUNT} entries, sampling every ${samplingIntervalUs} µs\n\n` +
`${formatCharsetReport(comparisons)}\n\n`
);
for (const measurement of measurements) {
process.stdout.write(
`${measurement.workload} [${measurement.variant}] non-Latin-1 input: ${
measurement.hasNonLatin1Characters
}; top self frames: ${measurement.topSelfFrames
.slice(0, 5)
.map((frame) => `${frame.frame} (${frame.samples})`)
.join(', ')}\n`
);
}
if (values.output) {
await writeFile(
resolve(values.output),
`${JSON.stringify({ node: process.version, iterations, warmup, samplingIntervalUs, comparisons, measurements }, null, 2)}\n`
);
}
}
interface CollectedSamples {
readonly wall: Map<Variant, number[]>;
readonly cpu: Map<Variant, number[]>;
readonly profiles: Map<Variant, CpuProfileLike[]>;
}
function measurementFor(
workload: Workload,
variant: Variant,
collected: CollectedSamples
): CharsetCaseMeasurement {
const input = workload.input(variant);
const summaries = (collected.profiles.get(variant) ?? []).map((profile) =>
summarizeSelfSamples(profile, Number.POSITIVE_INFINITY)
);
const merged = new Map<string, number>();
for (const frame of summaries.flatMap((summary) => summary.top)) {
merged.set(frame.frame, (merged.get(frame.frame) ?? 0) + frame.samples);
}
return {
workload: workload.name,
variant,
inputBytes: Buffer.byteLength(input, 'utf8'),
inputLength: input.length,
// Latin-1 round-trip is lossless only when every code unit is <= 0xff.
hasNonLatin1Characters:
Buffer.from(input, 'latin1').toString('latin1') !== input,
wallClockMs: collected.wall.get(variant) ?? [],
cpuMs: collected.cpu.get(variant) ?? [],
profileSamples: summaries.reduce((sum, s) => sum + s.total, 0),
topSelfFrames: topFrames(merged, 10),
};
}
function append<T>(map: Map<Variant, T[]>, variant: Variant, value: T): void {
map.set(variant, [...(map.get(variant) ?? []), value]);
}
function collectGarbage(): void {
(globalThis as { gc?: () => void }).gc?.();
}
function positiveInteger(value: string | undefined, name: string): number {
const parsed = nonNegativeInteger(value, name);
if (parsed < 1) {
throw new Error(`--${name} must be a positive integer`);
}
return parsed;
}
function nonNegativeInteger(value: string | undefined, name: string): number {
const parsed = Number(value);
if (!Number.isSafeInteger(parsed) || parsed < 0) {
throw new Error(`--${name} must be a non-negative integer`);
}
return parsed;
}
main().catch((error: unknown) => {
process.stderr.write(`${String(error)}\n`);
process.exitCode = 1;
});
@@ -0,0 +1,48 @@
/**
* Title vocabulary for the synthetic M3U and XMLTV performance fixtures.
*
* V8 keeps a string one-byte (Latin-1) while every character fits in a byte
* and stores the whole string as two-byte UTF-16 once a single character
* outside Latin-1 appears. `cyrillic` fixtures exercise that two-byte path.
*
* Every Cyrillic entry has exactly the UTF-16 length of its Latin entry, so
* both variants have the same line count and the same character layout; only
* the display titles (channel, group and programme titles) differ.
*/
export const SYNTHETIC_CHARSETS = ['latin1', 'cyrillic'] as const;
export type SyntheticCharset = (typeof SYNTHETIC_CHARSETS)[number];
export interface SyntheticTitleVocabulary {
readonly channel: string;
readonly group: string;
readonly language: string;
readonly programme: string;
}
export const SYNTHETIC_TITLE_VOCABULARY: Readonly<
Record<SyntheticCharset, SyntheticTitleVocabulary>
> = Object.freeze({
latin1: Object.freeze({
channel: 'Synthetic Channel',
group: 'Synthetic Group',
language: 'en',
programme: 'Synthetic Programme',
}),
cyrillic: Object.freeze({
channel: 'Пробный телеканал',
group: 'Пробная рубрика',
language: 'ru',
programme: 'Синтетическая серия',
}),
});
export function resolveSyntheticCharset(
charset: SyntheticCharset | undefined
): SyntheticCharset {
const resolved = charset ?? 'latin1';
if (!SYNTHETIC_CHARSETS.includes(resolved)) {
throw new Error(`Unsupported synthetic charset: ${String(charset)}`);
}
return resolved;
}
@@ -2,6 +2,7 @@ import assert from 'node:assert/strict';
import { createHash } from 'node:crypto';
import { describe, it } from 'node:test';
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
import {
createSyntheticM3uFixture,
SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS,
@@ -14,6 +15,12 @@ const GOLDEN_SHA256 = {
100_000: '2ffe912f61ea106ba8de3414b0a927cdaa3264ac637d44c4b219f199eb94c353',
} as const;
const CYRILLIC_GOLDEN_SHA256 = {
10_000: 'bc4f3c2a3d507db459dff2d4a723d8fa8729fb957cc436f062e703564201a20b',
50_000: '4a3fc713548a7ab50c70dc0751eb5b2483122af9d01f7d3975f309dc820c6a04',
100_000: '94727a8452920c07727812f4ec78b77deac353f5a46ee096620efba2296b7e2b',
} as const;
describe('synthetic M3U performance fixture', () => {
it('generates each supported catalog deterministically with exact metadata', () => {
assert.deepEqual(
@@ -47,6 +54,72 @@ describe('synthetic M3U performance fixture', () => {
}
});
it('keeps the latin1 option identical to the default fixture', () => {
const byDefault = createSyntheticM3uFixture(10_000);
const explicit = createSyntheticM3uFixture(10_000, {
charset: 'latin1',
});
assert.equal(byDefault.charset, 'latin1');
assert.equal(explicit.body, byDefault.body);
assert.equal(explicit.sha256, GOLDEN_SHA256[10_000]);
// UTF-8 byte length equals UTF-16 length only for ASCII-only text.
assert.equal(byDefault.bytes, byDefault.body.length);
});
it('generates each cyrillic catalog deterministically', () => {
for (const channelCount of SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS) {
const cyrillic = createSyntheticM3uFixture(channelCount, {
charset: 'cyrillic',
});
assert.equal(cyrillic.charset, 'cyrillic');
assert.equal(cyrillic.channelCount, channelCount);
assert.equal(cyrillic.sha256, CYRILLIC_GOLDEN_SHA256[channelCount]);
}
});
it('keeps the cyrillic layout identical apart from titles', () => {
const latin = createSyntheticM3uFixture(10_000);
const cyrillic = createSyntheticM3uFixture(10_000, {
charset: 'cyrillic',
});
assert.equal(cyrillic.bytes, Buffer.byteLength(cyrillic.body, 'utf8'));
assert.equal(
cyrillic.sha256,
createHash('sha256').update(cyrillic.body, 'utf8').digest('hex')
);
assert.equal(cyrillic.body.length, latin.body.length);
assert.equal(toLatinTitles(cyrillic.body), latin.body);
assertOnlyLoopbackUrls(cyrillic.body);
});
it('puts characters outside Latin-1 into every cyrillic channel entry', () => {
const lines = createSyntheticM3uFixture(10_000, { charset: 'cyrillic' })
.body.trimEnd()
.split('\n');
for (const line of lines.filter((entry) =>
entry.startsWith('#EXTINF:')
)) {
assert.ok(hasCharacterAbove(line, 0xff), line);
}
for (const line of lines.filter((entry) => entry.startsWith('http'))) {
assert.equal(hasCharacterAbove(line, 0x7f), false, line);
}
});
it('rejects unsupported charsets', () => {
assert.throws(
() =>
createSyntheticM3uFixture(10_000, {
charset: 'greek' as never,
}),
/Unsupported synthetic charset/
);
});
it('rejects unsupported channel counts', () => {
assert.throws(
() => createSyntheticM3uFixture(9_999),
@@ -55,6 +128,29 @@ describe('synthetic M3U performance fixture', () => {
});
});
function toLatinTitles(body: string): string {
const latin = SYNTHETIC_TITLE_VOCABULARY.latin1;
const cyrillic = SYNTHETIC_TITLE_VOCABULARY.cyrillic;
return replaceEvery(
replaceEvery(body, cyrillic.group, latin.group),
cyrillic.channel,
latin.channel
);
}
function hasCharacterAbove(text: string, maxCodeUnit: number): boolean {
for (let index = 0; index < text.length; index += 1) {
if (text.charCodeAt(index) > maxCodeUnit) {
return true;
}
}
return false;
}
function replaceEvery(text: string, search: string, replacement: string) {
return text.split(search).join(replacement);
}
function assertOnlyLoopbackUrls(body: string): void {
const urls = body.match(/https?:\/\/[^\s"]+/g) ?? [];
@@ -1,5 +1,11 @@
import { createHash } from 'node:crypto';
import {
resolveSyntheticCharset,
type SyntheticCharset,
SYNTHETIC_TITLE_VOCABULARY,
} from './synthetic-charset';
export const SYNTHETIC_M3U_SEED = 240_724;
export const SYNTHETIC_M3U_CHANNEL_COUNT = 100_000;
export const SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS = [
@@ -15,13 +21,22 @@ export interface SyntheticM3uFixture {
readonly body: string;
readonly bytes: number;
readonly channelCount: SyntheticM3uChannelCount;
readonly charset: SyntheticCharset;
readonly sha256: string;
}
export interface SyntheticM3uFixtureOptions {
/** Script of channel names and group titles; defaults to `latin1`. */
readonly charset?: SyntheticCharset;
}
export function createSyntheticM3uFixture(
channelCount: number = SYNTHETIC_M3U_CHANNEL_COUNT
channelCount: number = SYNTHETIC_M3U_CHANNEL_COUNT,
options: SyntheticM3uFixtureOptions = {}
): SyntheticM3uFixture {
assertSupportedChannelCount(channelCount);
const charset = resolveSyntheticCharset(options.charset);
const titles = SYNTHETIC_TITLE_VOCABULARY[charset];
const lines = new Array<string>(1 + channelCount * 2);
lines[0] = '#EXTM3U';
@@ -31,12 +46,12 @@ export function createSyntheticM3uFixture(
const lineOffset = 1 + offset * 2;
const stableIndex = String(index).padStart(6, '0');
lines[lineOffset] = `#EXTINF:-1 group-title="Synthetic Group ${String(
lines[lineOffset] = `#EXTINF:-1 group-title="${titles.group} ${String(
group
).padStart(
3,
'0'
)}",Synthetic Channel ${SYNTHETIC_M3U_SEED}-${stableIndex}`;
)}",${titles.channel} ${SYNTHETIC_M3U_SEED}-${stableIndex}`;
lines[lineOffset + 1] =
`http://127.0.0.1/stream/${SYNTHETIC_M3U_SEED}/${index}`;
}
@@ -46,6 +61,7 @@ export function createSyntheticM3uFixture(
body,
bytes: Buffer.byteLength(body, 'utf8'),
channelCount,
charset,
sha256: createHash('sha256').update(body, 'utf8').digest('hex'),
});
}
@@ -0,0 +1,156 @@
import assert from 'node:assert/strict';
import { createHash } from 'node:crypto';
import { describe, it } from 'node:test';
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
import {
createSyntheticXmltvFixture,
SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS,
SYNTHETIC_XMLTV_PROGRAMME_COUNT,
SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL,
} from './synthetic-xmltv';
const GOLDEN_SHA256 = {
latin1: {
10_000: 'e9d0ca0f7b73efc5cdee42cb2c4ee5bb4564b274b5042b13f8cd0d965d5011bc',
50_000: '08b955c84822b7eb0d825a73eef6a6abc88b6d372c2dd32ad25d3969d667fd49',
100_000:
'932f0ba7089cab9e205a78142a245ebb497826ee7f1f4819603d77eba8e95a9e',
},
cyrillic: {
10_000: '5b0df408e90c77bebe296fe1cc7e97cec5b929ebc7ad21b7480a6849a29f912b',
50_000: 'bb1a0ca9fe874ab847b062c768868d76da750366aaa680bc712cb4cb7bfbf809',
100_000:
'cd637f4a25d47592116417a64bbb8c5b48d87f5bb5b0b4693f7af6c7603a8859',
},
} as const;
describe('synthetic XMLTV performance fixture', () => {
it('generates each supported schedule deterministically', () => {
assert.deepEqual(
SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS,
[10_000, 50_000, 100_000]
);
assert.equal(SYNTHETIC_XMLTV_PROGRAMME_COUNT, 50_000);
for (const charset of ['latin1', 'cyrillic'] as const) {
for (const programmeCount of SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS) {
const fixture = createSyntheticXmltvFixture(programmeCount, {
charset,
});
assert.equal(fixture.charset, charset);
assert.equal(fixture.programmeCount, programmeCount);
assert.equal(
fixture.channelCount,
programmeCount / SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL
);
assert.equal(
fixture.sha256,
GOLDEN_SHA256[charset][programmeCount]
);
}
}
});
it('reports exact metadata and element counts', () => {
for (const charset of ['latin1', 'cyrillic'] as const) {
const fixture = createSyntheticXmltvFixture(10_000, { charset });
assert.equal(
fixture.bytes,
Buffer.byteLength(fixture.body, 'utf8')
);
assert.equal(
fixture.sha256,
createHash('sha256').update(fixture.body, 'utf8').digest('hex')
);
assert.equal(countOf(fixture.body, '<channel '), 100);
assert.equal(countOf(fixture.body, '<programme '), 10_000);
assert.ok(fixture.body.startsWith('<?xml version="1.0"'));
assert.ok(fixture.body.endsWith('</tv>\n'));
}
});
it('defaults to 50k programmes with an ASCII-only latin1 body', () => {
const fixture = createSyntheticXmltvFixture();
assert.equal(fixture.charset, 'latin1');
assert.equal(fixture.programmeCount, 50_000);
// UTF-8 byte length equals UTF-16 length only for ASCII-only text.
assert.equal(fixture.bytes, fixture.body.length);
});
it('keeps the cyrillic layout identical apart from titles', () => {
const latin = createSyntheticXmltvFixture(10_000);
const cyrillic = createSyntheticXmltvFixture(10_000, {
charset: 'cyrillic',
});
const latinTitles = SYNTHETIC_TITLE_VOCABULARY.latin1;
const cyrillicTitles = SYNTHETIC_TITLE_VOCABULARY.cyrillic;
assert.equal(cyrillic.body.length, latin.body.length);
const translated = [
[cyrillicTitles.channel, latinTitles.channel],
[cyrillicTitles.programme, latinTitles.programme],
[
`lang="${cyrillicTitles.language}"`,
`lang="${latinTitles.language}"`,
],
].reduce(
(text, [search, replacement]) =>
text.split(search).join(replacement),
cyrillic.body
);
assert.equal(translated, latin.body);
const firstTitle = cyrillic.body.indexOf('<title lang="ru">');
assert.ok(firstTitle > 0);
assert.ok(
cyrillic.body.charCodeAt(firstTitle + '<title lang="ru">'.length) >
0xff
);
});
it('schedules consecutive half-hour programmes per channel', () => {
const lines = createSyntheticXmltvFixture(10_000).body.split('\n');
const firstProgramme = lines.find((line) =>
line.includes('<programme ')
);
assert.equal(
firstProgramme,
' <programme start="20260101000000 +0000" stop="20260101003000 +0000" channel="synthetic.000001"><title lang="en">Synthetic Programme 000001-0001</title><desc lang="en">Synthetic description for slot 1.</desc><category lang="en">Synthetic</category></programme>'
);
assert.ok(
lines.includes(
' <programme start="20260103013000 +0000" stop="20260103020000 +0000" channel="synthetic.000100"><title lang="en">Synthetic Programme 000100-0100</title><desc lang="en">Synthetic description for slot 100.</desc><category lang="en">Synthetic</category></programme>'
)
);
});
it('rejects unsupported programme counts and charsets', () => {
assert.throws(
() => createSyntheticXmltvFixture(12_345),
/Unsupported synthetic XMLTV programme count/
);
assert.throws(
() =>
createSyntheticXmltvFixture(10_000, {
charset: 'arabic' as never,
}),
/Unsupported synthetic charset/
);
});
});
function countOf(body: string, needle: string): number {
let count = 0;
for (
let index = body.indexOf(needle);
index !== -1;
index = body.indexOf(needle, index + needle.length)
) {
count += 1;
}
return count;
}
@@ -0,0 +1,138 @@
import { createHash } from 'node:crypto';
import {
resolveSyntheticCharset,
type SyntheticCharset,
SYNTHETIC_TITLE_VOCABULARY,
} from './synthetic-charset';
import { SYNTHETIC_M3U_SEED } from './synthetic-m3u';
export const SYNTHETIC_XMLTV_PROGRAMME_COUNT = 50_000;
export const SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL = 100;
export const SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS = [
10_000,
SYNTHETIC_XMLTV_PROGRAMME_COUNT,
100_000,
] as const;
/** 2026-01-01T00:00:00Z; every programme lasts 30 minutes. */
const SCHEDULE_START_EPOCH_MS = Date.UTC(2026, 0, 1);
const PROGRAMME_DURATION_MS = 30 * 60 * 1000;
export type SyntheticXmltvProgrammeCount =
(typeof SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS)[number];
export interface SyntheticXmltvFixture {
readonly body: string;
readonly bytes: number;
readonly channelCount: number;
readonly charset: SyntheticCharset;
readonly programmeCount: SyntheticXmltvProgrammeCount;
readonly sha256: string;
}
export interface SyntheticXmltvFixtureOptions {
/** Script of channel display names and programme titles. */
readonly charset?: SyntheticCharset;
}
/**
* Deterministic XMLTV document: all `<channel>` entries first, then
* `SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL` consecutive programmes per
* channel. Descriptions, categories and attributes stay ASCII in every
* charset so only the titles decide the string encoding.
*/
export function createSyntheticXmltvFixture(
programmeCount: number = SYNTHETIC_XMLTV_PROGRAMME_COUNT,
options: SyntheticXmltvFixtureOptions = {}
): SyntheticXmltvFixture {
assertSupportedProgrammeCount(programmeCount);
const charset = resolveSyntheticCharset(options.charset);
const titles = SYNTHETIC_TITLE_VOCABULARY[charset];
const channelCount =
programmeCount / SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL;
const lines: string[] = [
'<?xml version="1.0" encoding="UTF-8"?>',
'<tv generator-info-name="iptvnator-synthetic">',
];
for (let offset = 0; offset < channelCount; offset += 1) {
const channelId = syntheticChannelId(offset);
lines.push(
` <channel id="${channelId}"><display-name lang="${titles.language}">${
titles.channel
} ${SYNTHETIC_M3U_SEED}-${stableNumber(offset + 1, 6)}</display-name></channel>`
);
}
const slotTimestamps = Array.from(
{ length: SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL + 1 },
(_, slot) =>
xmltvTimestamp(
SCHEDULE_START_EPOCH_MS + slot * PROGRAMME_DURATION_MS
)
);
for (let offset = 0; offset < channelCount; offset += 1) {
const channelId = syntheticChannelId(offset);
const channelNumber = stableNumber(offset + 1, 6);
for (
let slot = 0;
slot < SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL;
slot += 1
) {
lines.push(
` <programme start="${slotTimestamps[slot]}" stop="${
slotTimestamps[slot + 1]
}" channel="${channelId}"><title lang="${titles.language}">${
titles.programme
} ${channelNumber}-${stableNumber(
slot + 1,
4
)}</title><desc lang="en">Synthetic description for slot ${
slot + 1
}.</desc><category lang="en">Synthetic</category></programme>`
);
}
}
lines.push('</tv>');
const body = `${lines.join('\n')}\n`;
return Object.freeze({
body,
bytes: Buffer.byteLength(body, 'utf8'),
channelCount,
charset,
programmeCount,
sha256: createHash('sha256').update(body, 'utf8').digest('hex'),
});
}
function syntheticChannelId(offset: number): string {
return `synthetic.${stableNumber(offset + 1, 6)}`;
}
function stableNumber(value: number, width: number): string {
return String(value).padStart(width, '0');
}
function xmltvTimestamp(epochMs: number): string {
const iso = new Date(epochMs).toISOString();
return `${iso.slice(0, 4)}${iso.slice(5, 7)}${iso.slice(8, 10)}${iso.slice(
11,
13
)}${iso.slice(14, 16)}${iso.slice(17, 19)} +0000`;
}
function assertSupportedProgrammeCount(
programmeCount: number
): asserts programmeCount is SyntheticXmltvProgrammeCount {
if (
!SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS.some(
(supported) => supported === programmeCount
)
) {
throw new Error(
`Unsupported synthetic XMLTV programme count: ${programmeCount}`
);
}
}
+37
View File
@@ -382,6 +382,43 @@ in all eighteen runner iterations; the `spawnToFirstCardMs` P50 ranged from
differ from a Mac (12 and 571 there, the fast path without the Linux-only
`getWindowState` call), so take J1 baseline values from the runner only.
## Charset parse benchmark
V8 stores a string as two-byte UTF-16 once one character falls outside
Latin-1, and substrings of such a string stay two-byte, even ASCII-only URL
lines. `src/performance/charset-parse.benchmark.ts` checks whether that slows
playlist and EPG parsing. It is a Node benchmark, not a journey, and is not
ratcheted:
```bash
pnpm nx run electron-backend-e2e:benchmark-charset-parse --iterations=5
```
It parses 50,000 M3U channels (`iptv-playlist-parser`, then
`createPlaylistObject`, the main-process `PARSE_M3U` and `NORMALIZE` phases)
and 50,000 XMLTV programmes (`StreamingEpgParser`, the EPG worker's parser).
Each workload runs on three inputs: `latin1` and `cyrillic` from the
synthetic generators (`charset` option of `synthetic-m3u.ts` and
`synthetic-xmltv.ts`, identical layout apart from titles), and `latin1-bom`,
the latin1 bytes behind a UTF-8 byte-order mark. The BOM forces two-byte
storage without changing content, which separates the encoding cost from
the effect that non-ASCII titles have on ASCII-only regexes. The XMLTV
parser receives 64 Ki-character slices of one decoded string rather than
per-chunk decoded buffers: slices keep the input's representation (a
per-chunk decode would make the BOM control one-byte after its first
chunk), and no multi-byte character is split. Before timing, an untimed
pass checks that the parsed titles match the fixture.
The report gives P50 wall-clock and CPU time after one warm-up, plus
CPU-profile sample counts and top self frames from a separate profiled pass.
Inputs alternate within each round and the starting input rotates between
rounds. Prefer CPU time and samples on a busy machine.
The 2026-09-27 measurement (plan item D1) found every workload under the 1.5x
threshold on Node 22 and inside Electron 43, so D2 regex prefilters were not
applied. Rerun the benchmark after changing either parser or when a user
reports slow imports of non-Latin playlists.
## Adding a counter
1. Produce the value from the built output or from a deterministic probe, not