mirror of
https://github.com/4gray/iptvnator.git
synced 2026-10-08 09:01:03 -08:00
test(performance): measure two-byte string cost in M3U and XMLTV parsing (D1) (#1713)
This commit is contained in:
1 parent
363411542f
commit
8b570e3390
11 files changed
+1252
-3
No files matched your search
@@ -49,6 +49,15 @@
|
||||
"command": "pnpm exec playwright test --config=playwright.xtream-performance.config.ts src/xtream.performance.ts"
|
||||
}
|
||||
},
|
||||
"benchmark-charset-parse": {
|
||||
"executor": "nx:run-commands",
|
||||
"cache": false,
|
||||
"parallelism": false,
|
||||
"options": {
|
||||
"cwd": "apps/electron-backend-e2e",
|
||||
"command": "pnpm exec tsx --expose-gc --tsconfig tsconfig.json src/performance/charset-parse.benchmark.ts"
|
||||
}
|
||||
},
|
||||
"journeys": {
|
||||
"dependsOn": ["electron-backend:build-performance"],
|
||||
"executor": "nx:run-commands",
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
import assert from 'node:assert/strict';
|
||||
import { describe, it } from 'node:test';
|
||||
|
||||
import {
|
||||
CHARSET_BENCHMARK_VARIANTS,
|
||||
type CharsetBenchmarkVariant,
|
||||
type CharsetCaseMeasurement,
|
||||
compareCharsets,
|
||||
formatCharsetReport,
|
||||
medianOf,
|
||||
summarizeSelfSamples,
|
||||
variantOrderForRound,
|
||||
} from './charset-parse-benchmark-report';
|
||||
|
||||
function measurement(
|
||||
variant: CharsetBenchmarkVariant,
|
||||
cpuMs: number[],
|
||||
profileSamples: number,
|
||||
workload = 'm3u parse'
|
||||
): CharsetCaseMeasurement {
|
||||
return {
|
||||
workload,
|
||||
variant,
|
||||
inputBytes: 10,
|
||||
inputLength: 10,
|
||||
hasNonLatin1Characters: variant !== 'latin1',
|
||||
wallClockMs: cpuMs.map((value) => value * 2),
|
||||
cpuMs,
|
||||
profileSamples,
|
||||
topSelfFrames: [],
|
||||
};
|
||||
}
|
||||
|
||||
describe('charset parse benchmark report', () => {
|
||||
it('rotates the starting variant every round', () => {
|
||||
assert.deepEqual(variantOrderForRound(0), [
|
||||
'latin1',
|
||||
'latin1-bom',
|
||||
'cyrillic',
|
||||
]);
|
||||
assert.deepEqual(variantOrderForRound(1), [
|
||||
'latin1-bom',
|
||||
'cyrillic',
|
||||
'latin1',
|
||||
]);
|
||||
assert.deepEqual(variantOrderForRound(2), [
|
||||
'cyrillic',
|
||||
'latin1',
|
||||
'latin1-bom',
|
||||
]);
|
||||
assert.deepEqual(variantOrderForRound(3), variantOrderForRound(0));
|
||||
for (let round = 0; round < 6; round += 1) {
|
||||
assert.deepEqual(
|
||||
[...variantOrderForRound(round)].sort(),
|
||||
[...CHARSET_BENCHMARK_VARIANTS].sort()
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
it('takes the median of odd and even samples without mutating them', () => {
|
||||
const values = [5, 1, 3];
|
||||
|
||||
assert.equal(medianOf(values), 3);
|
||||
assert.deepEqual(values, [5, 1, 3]);
|
||||
assert.equal(medianOf([4, 1, 3, 2]), 2.5);
|
||||
assert.throws(() => medianOf([]), /empty sample/);
|
||||
});
|
||||
|
||||
it('counts self samples per frame, most frequent first', () => {
|
||||
const summary = summarizeSelfSamples(
|
||||
{
|
||||
nodes: [
|
||||
{
|
||||
id: 1,
|
||||
callFrame: {
|
||||
functionName: '',
|
||||
url: '',
|
||||
lineNumber: -1,
|
||||
},
|
||||
},
|
||||
{
|
||||
id: 2,
|
||||
callFrame: {
|
||||
functionName: 'scanAttributes',
|
||||
url: 'file:///repo/node_modules/iptv-playlist-parser/src/index.js',
|
||||
lineNumber: 42,
|
||||
},
|
||||
},
|
||||
],
|
||||
samples: [2, 2, 1, 2, 9],
|
||||
},
|
||||
2
|
||||
);
|
||||
|
||||
assert.equal(summary.total, 5);
|
||||
assert.deepEqual(summary.top, [
|
||||
{ frame: 'scanAttributes index.js:43', samples: 3 },
|
||||
{ frame: '(anonymous)', samples: 1 },
|
||||
]);
|
||||
});
|
||||
|
||||
it('compares every variant with latin1 and flags only slowdowns seen by both signals', () => {
|
||||
const [comparison] = compareCharsets([
|
||||
measurement('latin1', [10, 12, 11], 100),
|
||||
measurement('latin1-bom', [16, 17, 18], 140),
|
||||
measurement('cyrillic', [16, 17, 18], 160),
|
||||
]);
|
||||
|
||||
assert.equal(comparison.workload, 'm3u parse');
|
||||
assert.deepEqual(
|
||||
comparison.variants.map((row) => row.variant),
|
||||
[...CHARSET_BENCHMARK_VARIANTS]
|
||||
);
|
||||
const [latin1, bom, cyrillic] = comparison.variants;
|
||||
assert.equal(latin1.cpuRatio, 1);
|
||||
assert.equal(latin1.exceedsThreshold, false);
|
||||
assert.equal(bom.cpuP50Ms, 17);
|
||||
assert.equal(bom.wallP50Ms, 34);
|
||||
assert.ok(Math.abs(bom.cpuRatio - 17 / 11) < 1e-9);
|
||||
// CPU time is over 1.5x, but profile samples (1.4x) are not.
|
||||
assert.equal(bom.exceedsThreshold, false);
|
||||
assert.equal(cyrillic.sampleRatio, 1.6);
|
||||
assert.equal(cyrillic.exceedsThreshold, true);
|
||||
});
|
||||
|
||||
it('fails when a variant was not measured', () => {
|
||||
assert.throws(
|
||||
() =>
|
||||
compareCharsets([
|
||||
measurement('latin1', [1], 1),
|
||||
measurement('cyrillic', [1], 1),
|
||||
]),
|
||||
/Missing latin1-bom measurement for m3u parse/
|
||||
);
|
||||
});
|
||||
|
||||
it('formats one markdown row per workload and variant', () => {
|
||||
const report = formatCharsetReport(
|
||||
compareCharsets([
|
||||
measurement('latin1', [10], 100),
|
||||
measurement('latin1-bom', [11], 100),
|
||||
measurement('cyrillic', [20], 200),
|
||||
])
|
||||
);
|
||||
const lines = report.split('\n');
|
||||
|
||||
assert.equal(lines.length, 5);
|
||||
assert.match(lines[0], /^\| Workload \| Input \|/);
|
||||
assert.equal(
|
||||
lines[4],
|
||||
'| m3u parse | cyrillic | 40.0 | 20.0 | 200 | 2.00x | 2.00x | 2.00x | yes |'
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,206 @@
|
||||
/** Two-byte input slower than this ratio justifies parser changes (plan D1). */
|
||||
export const CHARSET_SLOWDOWN_THRESHOLD = 1.5;
|
||||
|
||||
/**
|
||||
* Input variants of the charset benchmark. `latin1-bom` is the `latin1`
|
||||
* fixture behind a UTF-8 byte-order mark: same content, but V8 must store
|
||||
* the decoded string as two-byte, which isolates the encoding effect from
|
||||
* the content effect that Cyrillic titles have on ASCII-only regexes.
|
||||
*/
|
||||
export const CHARSET_BENCHMARK_VARIANTS = [
|
||||
'latin1',
|
||||
'latin1-bom',
|
||||
'cyrillic',
|
||||
] as const;
|
||||
|
||||
export type CharsetBenchmarkVariant =
|
||||
(typeof CHARSET_BENCHMARK_VARIANTS)[number];
|
||||
|
||||
export interface CpuProfileNode {
|
||||
readonly id: number;
|
||||
readonly callFrame: {
|
||||
readonly functionName: string;
|
||||
readonly url: string;
|
||||
readonly lineNumber: number;
|
||||
};
|
||||
}
|
||||
|
||||
export interface CpuProfileLike {
|
||||
readonly nodes: readonly CpuProfileNode[];
|
||||
readonly samples?: readonly number[];
|
||||
}
|
||||
|
||||
export interface SelfSampleFrame {
|
||||
readonly frame: string;
|
||||
readonly samples: number;
|
||||
}
|
||||
|
||||
export interface CharsetCaseMeasurement {
|
||||
readonly workload: string;
|
||||
readonly variant: CharsetBenchmarkVariant;
|
||||
readonly inputBytes: number;
|
||||
readonly inputLength: number;
|
||||
/** True when V8 must store the decoded input as a two-byte string. */
|
||||
readonly hasNonLatin1Characters: boolean;
|
||||
readonly wallClockMs: readonly number[];
|
||||
readonly cpuMs: readonly number[];
|
||||
readonly profileSamples: number;
|
||||
readonly topSelfFrames: readonly SelfSampleFrame[];
|
||||
}
|
||||
|
||||
export interface CharsetVariantSummary {
|
||||
readonly variant: CharsetBenchmarkVariant;
|
||||
readonly wallP50Ms: number;
|
||||
readonly cpuP50Ms: number;
|
||||
readonly profileSamples: number;
|
||||
readonly wallRatio: number;
|
||||
readonly cpuRatio: number;
|
||||
readonly sampleRatio: number;
|
||||
/** Slower than the threshold on CPU time and on profile samples. */
|
||||
readonly exceedsThreshold: boolean;
|
||||
}
|
||||
|
||||
export interface CharsetComparison {
|
||||
readonly workload: string;
|
||||
readonly variants: readonly CharsetVariantSummary[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Variant order for one benchmark round. The starting variant rotates every
|
||||
* round so no input always runs first or last after a garbage collection.
|
||||
*/
|
||||
export function variantOrderForRound(round: number): CharsetBenchmarkVariant[] {
|
||||
const variants = [...CHARSET_BENCHMARK_VARIANTS];
|
||||
const offset = round % variants.length;
|
||||
return [...variants.slice(offset), ...variants.slice(0, offset)];
|
||||
}
|
||||
|
||||
export function medianOf(values: readonly number[]): number {
|
||||
if (values.length === 0) {
|
||||
throw new Error('Cannot take the median of an empty sample');
|
||||
}
|
||||
const sorted = [...values].sort((left, right) => left - right);
|
||||
const middle = Math.floor(sorted.length / 2);
|
||||
return sorted.length % 2 === 1
|
||||
? sorted[middle]
|
||||
: (sorted[middle - 1] + sorted[middle]) / 2;
|
||||
}
|
||||
|
||||
/** Counts self samples per frame; the top frames explain where time went. */
|
||||
export function summarizeSelfSamples(
|
||||
profile: CpuProfileLike,
|
||||
limit = 8
|
||||
): { total: number; top: SelfSampleFrame[] } {
|
||||
const nodesById = new Map(profile.nodes.map((node) => [node.id, node]));
|
||||
const counts = new Map<string, number>();
|
||||
const samples = profile.samples ?? [];
|
||||
|
||||
for (const nodeId of samples) {
|
||||
const frame = describeFrame(nodesById.get(nodeId));
|
||||
counts.set(frame, (counts.get(frame) ?? 0) + 1);
|
||||
}
|
||||
|
||||
return { total: samples.length, top: topFrames(counts, limit) };
|
||||
}
|
||||
|
||||
export function topFrames(
|
||||
counts: ReadonlyMap<string, number>,
|
||||
limit: number
|
||||
): SelfSampleFrame[] {
|
||||
return [...counts.entries()]
|
||||
.map(([frame, samples]) => ({ frame, samples }))
|
||||
.sort(
|
||||
(left, right) =>
|
||||
right.samples - left.samples ||
|
||||
left.frame.localeCompare(right.frame)
|
||||
)
|
||||
.slice(0, limit);
|
||||
}
|
||||
|
||||
export function compareCharsets(
|
||||
measurements: readonly CharsetCaseMeasurement[]
|
||||
): CharsetComparison[] {
|
||||
const workloads = [...new Set(measurements.map((m) => m.workload))];
|
||||
|
||||
return workloads.map((workload) => {
|
||||
const baseline = findCase(measurements, workload, 'latin1');
|
||||
const baselineWall = medianOf(baseline.wallClockMs);
|
||||
const baselineCpu = medianOf(baseline.cpuMs);
|
||||
|
||||
return {
|
||||
workload,
|
||||
variants: CHARSET_BENCHMARK_VARIANTS.map((variant) => {
|
||||
const measurement = findCase(measurements, workload, variant);
|
||||
const wallP50Ms = medianOf(measurement.wallClockMs);
|
||||
const cpuP50Ms = medianOf(measurement.cpuMs);
|
||||
const cpuRatio = cpuP50Ms / baselineCpu;
|
||||
const sampleRatio =
|
||||
measurement.profileSamples / baseline.profileSamples;
|
||||
return {
|
||||
variant,
|
||||
wallP50Ms,
|
||||
cpuP50Ms,
|
||||
profileSamples: measurement.profileSamples,
|
||||
wallRatio: wallP50Ms / baselineWall,
|
||||
cpuRatio,
|
||||
sampleRatio,
|
||||
exceedsThreshold:
|
||||
cpuRatio > CHARSET_SLOWDOWN_THRESHOLD &&
|
||||
sampleRatio > CHARSET_SLOWDOWN_THRESHOLD,
|
||||
};
|
||||
}),
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
export function formatCharsetReport(
|
||||
comparisons: readonly CharsetComparison[]
|
||||
): string {
|
||||
const header =
|
||||
'| Workload | Input | Wall P50 ms | CPU P50 ms | Samples | Wall ratio | CPU ratio | Sample ratio | Over 1.5x |';
|
||||
const divider =
|
||||
'| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | --- |';
|
||||
const rows = comparisons.flatMap((comparison) =>
|
||||
comparison.variants.map((row) =>
|
||||
[
|
||||
comparison.workload,
|
||||
row.variant,
|
||||
row.wallP50Ms.toFixed(1),
|
||||
row.cpuP50Ms.toFixed(1),
|
||||
String(row.profileSamples),
|
||||
`${row.wallRatio.toFixed(2)}x`,
|
||||
`${row.cpuRatio.toFixed(2)}x`,
|
||||
`${row.sampleRatio.toFixed(2)}x`,
|
||||
row.exceedsThreshold ? 'yes' : 'no',
|
||||
].join(' | ')
|
||||
)
|
||||
);
|
||||
return [header, divider, ...rows.map((row) => `| ${row} |`)].join('\n');
|
||||
}
|
||||
|
||||
function findCase(
|
||||
measurements: readonly CharsetCaseMeasurement[],
|
||||
workload: string,
|
||||
variant: CharsetBenchmarkVariant
|
||||
): CharsetCaseMeasurement {
|
||||
const match = measurements.find(
|
||||
(m) => m.workload === workload && m.variant === variant
|
||||
);
|
||||
if (!match) {
|
||||
throw new Error(`Missing ${variant} measurement for ${workload}`);
|
||||
}
|
||||
return match;
|
||||
}
|
||||
|
||||
function describeFrame(node: CpuProfileNode | undefined): string {
|
||||
if (!node) {
|
||||
return '(unknown)';
|
||||
}
|
||||
const { functionName, url, lineNumber } = node.callFrame;
|
||||
const name = functionName || '(anonymous)';
|
||||
if (!url) {
|
||||
return name;
|
||||
}
|
||||
const file = url.slice(url.lastIndexOf('/') + 1);
|
||||
return `${name} ${file}:${lineNumber + 1}`;
|
||||
}
|
||||
@@ -0,0 +1,191 @@
|
||||
import { createPlaylistObject } from '@iptvnator/shared/m3u-utils';
|
||||
import { parse } from 'iptv-playlist-parser';
|
||||
import { resolve } from 'node:path';
|
||||
|
||||
import type { CharsetBenchmarkVariant } from './charset-parse-benchmark-report';
|
||||
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
|
||||
import { createSyntheticM3uFixture, SYNTHETIC_M3U_SEED } from './synthetic-m3u';
|
||||
import { createSyntheticXmltvFixture } from './synthetic-xmltv';
|
||||
|
||||
export const ENTRY_COUNT = 50_000;
|
||||
/**
|
||||
* The EPG worker writes 64 KiB decoded chunks. The benchmark writes slices of
|
||||
* one decoded string instead, so every slice keeps the input's one-byte or
|
||||
* two-byte representation (decoding each chunk separately would make the
|
||||
* BOM control one-byte after its first chunk) and no character is split.
|
||||
*/
|
||||
const XMLTV_CHUNK_CHARS = 64 * 1024;
|
||||
const REPLACEMENT_CHARACTER = '\ufffd';
|
||||
const UTF8_BYTE_ORDER_MARK = Buffer.from([0xef, 0xbb, 0xbf]);
|
||||
|
||||
type Variant = CharsetBenchmarkVariant;
|
||||
|
||||
interface ParsedProgramme {
|
||||
readonly title: readonly { readonly value: string }[];
|
||||
}
|
||||
|
||||
interface StreamingEpgParserModule {
|
||||
StreamingEpgParser: new (
|
||||
onChannels: (channels: unknown[]) => void,
|
||||
onPrograms: (programs: ParsedProgramme[]) => void,
|
||||
onProgress: (channels: number, programs: number) => void,
|
||||
channelBatchSize?: number,
|
||||
programBatchSize?: number
|
||||
) => { write(chunk: string): void; finish(): { totalPrograms: number } };
|
||||
}
|
||||
|
||||
export interface Workload {
|
||||
readonly name: string;
|
||||
/** Prepares untimed input, then returns the timed operation. */
|
||||
prepare(variant: Variant): () => void;
|
||||
input(variant: Variant): string;
|
||||
/** Untimed check that the parser saw the fixture's exact titles. */
|
||||
verify(variant: Variant): void;
|
||||
}
|
||||
|
||||
export async function createWorkloads(): Promise<Workload[]> {
|
||||
const epgModule = (await import(
|
||||
resolve(
|
||||
__dirname,
|
||||
'../../../electron-backend/src/app/workers/epg-streaming-parser.ts'
|
||||
)
|
||||
)) as StreamingEpgParserModule;
|
||||
const m3uBuffers = buffersPerVariant((charset) =>
|
||||
createSyntheticM3uFixture(ENTRY_COUNT, { charset })
|
||||
);
|
||||
const xmltvBuffers = buffersPerVariant((charset) =>
|
||||
createSyntheticXmltvFixture(ENTRY_COUNT, { charset })
|
||||
);
|
||||
const m3uInput = (variant: Variant) =>
|
||||
requireBuffer(m3uBuffers, variant).toString('utf8');
|
||||
const xmltvInput = (variant: Variant) =>
|
||||
requireBuffer(xmltvBuffers, variant).toString('utf8');
|
||||
|
||||
const verifyM3u = (variant: Variant) => {
|
||||
const titles = titlesFor(variant);
|
||||
const [first] = parse(m3uInput(variant)).items;
|
||||
expectEqual(
|
||||
first?.name,
|
||||
`${titles.channel} ${SYNTHETIC_M3U_SEED}-000001`
|
||||
);
|
||||
expectEqual(first?.group.title, `${titles.group} 001`);
|
||||
};
|
||||
const parseXmltv = (
|
||||
variant: Variant,
|
||||
onPrograms: (programs: ParsedProgramme[]) => void
|
||||
) => {
|
||||
const chunks = sliceText(xmltvInput(variant), XMLTV_CHUNK_CHARS);
|
||||
return () => {
|
||||
const parser = new epgModule.StreamingEpgParser(
|
||||
() => undefined,
|
||||
onPrograms,
|
||||
() => undefined,
|
||||
100,
|
||||
1000
|
||||
);
|
||||
for (const chunk of chunks) {
|
||||
parser.write(chunk);
|
||||
}
|
||||
assertCount(parser.finish().totalPrograms);
|
||||
};
|
||||
};
|
||||
|
||||
return [
|
||||
{
|
||||
// PARSE_M3U phase of the main-process import.
|
||||
name: 'm3u parse',
|
||||
input: m3uInput,
|
||||
verify: verifyM3u,
|
||||
prepare(variant) {
|
||||
const body = m3uInput(variant);
|
||||
return () => assertCount(parse(body).items.length);
|
||||
},
|
||||
},
|
||||
{
|
||||
// NORMALIZE phase of the main-process import.
|
||||
name: 'm3u normalize',
|
||||
input: m3uInput,
|
||||
verify: verifyM3u,
|
||||
prepare(variant) {
|
||||
const parsed = parse(m3uInput(variant));
|
||||
return () =>
|
||||
assertCount(
|
||||
createPlaylistObject('benchmark', parsed, 'x', 'URL')
|
||||
.count
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
// EPG worker's saxes-based parser; UTF-8 decoding is excluded.
|
||||
name: 'xmltv stream parse',
|
||||
input: xmltvInput,
|
||||
verify(variant) {
|
||||
const titles: string[] = [];
|
||||
parseXmltv(variant, (programs) => {
|
||||
for (const programme of programs) {
|
||||
titles.push(programme.title[0]?.value ?? '');
|
||||
}
|
||||
})();
|
||||
const programme = titlesFor(variant).programme;
|
||||
expectEqual(titles[0], `${programme} 000001-0001`);
|
||||
expectEqual(titles.at(-1), `${programme} 000500-0100`);
|
||||
if (
|
||||
titles.some((title) =>
|
||||
title.includes(REPLACEMENT_CHARACTER)
|
||||
)
|
||||
) {
|
||||
throw new Error(`${variant} XMLTV titles contain U+FFFD`);
|
||||
}
|
||||
},
|
||||
prepare: (variant) => parseXmltv(variant, () => undefined),
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
function titlesFor(variant: Variant) {
|
||||
return SYNTHETIC_TITLE_VOCABULARY[
|
||||
variant === 'cyrillic' ? 'cyrillic' : 'latin1'
|
||||
];
|
||||
}
|
||||
|
||||
function sliceText(text: string, size: number): string[] {
|
||||
const chunks: string[] = [];
|
||||
for (let offset = 0; offset < text.length; offset += size) {
|
||||
chunks.push(text.slice(offset, offset + size));
|
||||
}
|
||||
return chunks;
|
||||
}
|
||||
|
||||
function expectEqual(actual: string | undefined, expected: string): void {
|
||||
if (actual !== expected) {
|
||||
throw new Error(`Expected "${expected}", parsed "${String(actual)}"`);
|
||||
}
|
||||
}
|
||||
|
||||
function buffersPerVariant(
|
||||
create: (charset: 'latin1' | 'cyrillic') => { body: string }
|
||||
): Map<Variant, Buffer> {
|
||||
const latin1 = Buffer.from(create('latin1').body, 'utf8');
|
||||
return new Map<Variant, Buffer>([
|
||||
['latin1', latin1],
|
||||
['latin1-bom', Buffer.concat([UTF8_BYTE_ORDER_MARK, latin1])],
|
||||
['cyrillic', Buffer.from(create('cyrillic').body, 'utf8')],
|
||||
]);
|
||||
}
|
||||
|
||||
function requireBuffer(
|
||||
buffers: Map<Variant, Buffer>,
|
||||
variant: Variant
|
||||
): Buffer {
|
||||
const buffer = buffers.get(variant);
|
||||
if (!buffer) {
|
||||
throw new Error(`No fixture for ${variant}`);
|
||||
}
|
||||
return buffer;
|
||||
}
|
||||
|
||||
function assertCount(count: number): void {
|
||||
if (count !== ENTRY_COUNT) {
|
||||
throw new Error(`Expected ${ENTRY_COUNT} entries, parsed ${count}`);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
/**
|
||||
* Plan D1: does two-byte (non-Latin-1) input slow down playlist and EPG
|
||||
* parsing? Runs each workload on the `latin1` and `cyrillic` synthetic
|
||||
* fixtures (identical layout) and on `latin1-bom`, the latin1 bytes behind a
|
||||
* UTF-8 byte-order mark, which forces V8's two-byte representation without
|
||||
* changing content. Reports P50 wall-clock and process CPU time over the
|
||||
* timed iterations, and CPU-profile sample counts from a separate profiled
|
||||
* pass (sampled in-process through the inspector, like `node --cpu-prof`).
|
||||
*
|
||||
* pnpm nx run electron-backend-e2e:benchmark-charset-parse \
|
||||
* [--iterations=5] [--warmup=1] [--sampling-interval-us=100] \
|
||||
* [--output=/absolute/report.json]
|
||||
*
|
||||
* To measure with Electron's V8 instead of the Node on PATH, run from
|
||||
* apps/electron-backend-e2e:
|
||||
*
|
||||
* TSX_TSCONFIG_PATH=tsconfig.json ELECTRON_RUN_AS_NODE=1 \
|
||||
* "$(node -p "require('electron')")" --expose-gc --import tsx \
|
||||
* src/performance/charset-parse.benchmark.ts
|
||||
*/
|
||||
import { writeFile } from 'node:fs/promises';
|
||||
import { Session } from 'node:inspector/promises';
|
||||
import { resolve } from 'node:path';
|
||||
import { parseArgs } from 'node:util';
|
||||
|
||||
import {
|
||||
CHARSET_BENCHMARK_VARIANTS,
|
||||
type CharsetBenchmarkVariant,
|
||||
type CharsetCaseMeasurement,
|
||||
compareCharsets,
|
||||
type CpuProfileLike,
|
||||
formatCharsetReport,
|
||||
summarizeSelfSamples,
|
||||
topFrames,
|
||||
variantOrderForRound,
|
||||
} from './charset-parse-benchmark-report';
|
||||
import {
|
||||
createWorkloads,
|
||||
ENTRY_COUNT,
|
||||
type Workload,
|
||||
} from './charset-parse-workloads';
|
||||
|
||||
type Variant = CharsetBenchmarkVariant;
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
options: {
|
||||
iterations: { type: 'string', default: '5' },
|
||||
warmup: { type: 'string', default: '1' },
|
||||
'sampling-interval-us': { type: 'string', default: '100' },
|
||||
output: { type: 'string' },
|
||||
},
|
||||
});
|
||||
const iterations = positiveInteger(values.iterations, 'iterations');
|
||||
const warmup = nonNegativeInteger(values.warmup, 'warmup');
|
||||
const samplingIntervalUs = positiveInteger(
|
||||
values['sampling-interval-us'],
|
||||
'sampling-interval-us'
|
||||
);
|
||||
const workloads = await createWorkloads();
|
||||
const session = new Session();
|
||||
session.connect();
|
||||
await session.post('Profiler.enable');
|
||||
await session.post('Profiler.setSamplingInterval', {
|
||||
interval: samplingIntervalUs,
|
||||
});
|
||||
|
||||
const measurements: CharsetCaseMeasurement[] = [];
|
||||
for (const workload of workloads) {
|
||||
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
|
||||
workload.verify(variant);
|
||||
}
|
||||
const wall = new Map<Variant, number[]>();
|
||||
const cpu = new Map<Variant, number[]>();
|
||||
const profiles = new Map<Variant, CpuProfileLike[]>();
|
||||
// Variants alternate inside every round, starting with a different
|
||||
// one each round, so JIT warm-up and heap growth favour no input.
|
||||
// Rounds after the timed ones run under the profiler.
|
||||
for (let round = 0; round < warmup + iterations * 2; round += 1) {
|
||||
for (const variant of variantOrderForRound(round)) {
|
||||
const run = workload.prepare(variant);
|
||||
collectGarbage();
|
||||
const profiled = round >= warmup + iterations;
|
||||
if (profiled) {
|
||||
await session.post('Profiler.start');
|
||||
}
|
||||
const cpuBefore = process.cpuUsage();
|
||||
const startedAt = performance.now();
|
||||
run();
|
||||
const elapsedMs = performance.now() - startedAt;
|
||||
const cpuUsed = process.cpuUsage(cpuBefore);
|
||||
if (profiled) {
|
||||
const { profile } = await session.post('Profiler.stop');
|
||||
append(profiles, variant, profile);
|
||||
} else if (round >= warmup) {
|
||||
append(wall, variant, elapsedMs);
|
||||
append(
|
||||
cpu,
|
||||
variant,
|
||||
(cpuUsed.user + cpuUsed.system) / 1000
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
|
||||
measurements.push(
|
||||
measurementFor(workload, variant, { wall, cpu, profiles })
|
||||
);
|
||||
}
|
||||
}
|
||||
session.disconnect();
|
||||
|
||||
const comparisons = compareCharsets(measurements);
|
||||
process.stdout.write(
|
||||
`Node ${process.version}, ${iterations} iterations after ${warmup} warm-up, ` +
|
||||
`${ENTRY_COUNT} entries, sampling every ${samplingIntervalUs} µs\n\n` +
|
||||
`${formatCharsetReport(comparisons)}\n\n`
|
||||
);
|
||||
for (const measurement of measurements) {
|
||||
process.stdout.write(
|
||||
`${measurement.workload} [${measurement.variant}] non-Latin-1 input: ${
|
||||
measurement.hasNonLatin1Characters
|
||||
}; top self frames: ${measurement.topSelfFrames
|
||||
.slice(0, 5)
|
||||
.map((frame) => `${frame.frame} (${frame.samples})`)
|
||||
.join(', ')}\n`
|
||||
);
|
||||
}
|
||||
if (values.output) {
|
||||
await writeFile(
|
||||
resolve(values.output),
|
||||
`${JSON.stringify({ node: process.version, iterations, warmup, samplingIntervalUs, comparisons, measurements }, null, 2)}\n`
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
interface CollectedSamples {
|
||||
readonly wall: Map<Variant, number[]>;
|
||||
readonly cpu: Map<Variant, number[]>;
|
||||
readonly profiles: Map<Variant, CpuProfileLike[]>;
|
||||
}
|
||||
|
||||
function measurementFor(
|
||||
workload: Workload,
|
||||
variant: Variant,
|
||||
collected: CollectedSamples
|
||||
): CharsetCaseMeasurement {
|
||||
const input = workload.input(variant);
|
||||
const summaries = (collected.profiles.get(variant) ?? []).map((profile) =>
|
||||
summarizeSelfSamples(profile, Number.POSITIVE_INFINITY)
|
||||
);
|
||||
const merged = new Map<string, number>();
|
||||
for (const frame of summaries.flatMap((summary) => summary.top)) {
|
||||
merged.set(frame.frame, (merged.get(frame.frame) ?? 0) + frame.samples);
|
||||
}
|
||||
return {
|
||||
workload: workload.name,
|
||||
variant,
|
||||
inputBytes: Buffer.byteLength(input, 'utf8'),
|
||||
inputLength: input.length,
|
||||
// Latin-1 round-trip is lossless only when every code unit is <= 0xff.
|
||||
hasNonLatin1Characters:
|
||||
Buffer.from(input, 'latin1').toString('latin1') !== input,
|
||||
wallClockMs: collected.wall.get(variant) ?? [],
|
||||
cpuMs: collected.cpu.get(variant) ?? [],
|
||||
profileSamples: summaries.reduce((sum, s) => sum + s.total, 0),
|
||||
topSelfFrames: topFrames(merged, 10),
|
||||
};
|
||||
}
|
||||
|
||||
function append<T>(map: Map<Variant, T[]>, variant: Variant, value: T): void {
|
||||
map.set(variant, [...(map.get(variant) ?? []), value]);
|
||||
}
|
||||
|
||||
function collectGarbage(): void {
|
||||
(globalThis as { gc?: () => void }).gc?.();
|
||||
}
|
||||
|
||||
function positiveInteger(value: string | undefined, name: string): number {
|
||||
const parsed = nonNegativeInteger(value, name);
|
||||
if (parsed < 1) {
|
||||
throw new Error(`--${name} must be a positive integer`);
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
function nonNegativeInteger(value: string | undefined, name: string): number {
|
||||
const parsed = Number(value);
|
||||
if (!Number.isSafeInteger(parsed) || parsed < 0) {
|
||||
throw new Error(`--${name} must be a non-negative integer`);
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
main().catch((error: unknown) => {
|
||||
process.stderr.write(`${String(error)}\n`);
|
||||
process.exitCode = 1;
|
||||
});
|
||||
@@ -0,0 +1,48 @@
|
||||
/**
|
||||
* Title vocabulary for the synthetic M3U and XMLTV performance fixtures.
|
||||
*
|
||||
* V8 keeps a string one-byte (Latin-1) while every character fits in a byte
|
||||
* and stores the whole string as two-byte UTF-16 once a single character
|
||||
* outside Latin-1 appears. `cyrillic` fixtures exercise that two-byte path.
|
||||
*
|
||||
* Every Cyrillic entry has exactly the UTF-16 length of its Latin entry, so
|
||||
* both variants have the same line count and the same character layout; only
|
||||
* the display titles (channel, group and programme titles) differ.
|
||||
*/
|
||||
export const SYNTHETIC_CHARSETS = ['latin1', 'cyrillic'] as const;
|
||||
|
||||
export type SyntheticCharset = (typeof SYNTHETIC_CHARSETS)[number];
|
||||
|
||||
export interface SyntheticTitleVocabulary {
|
||||
readonly channel: string;
|
||||
readonly group: string;
|
||||
readonly language: string;
|
||||
readonly programme: string;
|
||||
}
|
||||
|
||||
export const SYNTHETIC_TITLE_VOCABULARY: Readonly<
|
||||
Record<SyntheticCharset, SyntheticTitleVocabulary>
|
||||
> = Object.freeze({
|
||||
latin1: Object.freeze({
|
||||
channel: 'Synthetic Channel',
|
||||
group: 'Synthetic Group',
|
||||
language: 'en',
|
||||
programme: 'Synthetic Programme',
|
||||
}),
|
||||
cyrillic: Object.freeze({
|
||||
channel: 'Пробный телеканал',
|
||||
group: 'Пробная рубрика',
|
||||
language: 'ru',
|
||||
programme: 'Синтетическая серия',
|
||||
}),
|
||||
});
|
||||
|
||||
export function resolveSyntheticCharset(
|
||||
charset: SyntheticCharset | undefined
|
||||
): SyntheticCharset {
|
||||
const resolved = charset ?? 'latin1';
|
||||
if (!SYNTHETIC_CHARSETS.includes(resolved)) {
|
||||
throw new Error(`Unsupported synthetic charset: ${String(charset)}`);
|
||||
}
|
||||
return resolved;
|
||||
}
|
||||
@@ -2,6 +2,7 @@ import assert from 'node:assert/strict';
|
||||
import { createHash } from 'node:crypto';
|
||||
import { describe, it } from 'node:test';
|
||||
|
||||
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
|
||||
import {
|
||||
createSyntheticM3uFixture,
|
||||
SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS,
|
||||
@@ -14,6 +15,12 @@ const GOLDEN_SHA256 = {
|
||||
100_000: '2ffe912f61ea106ba8de3414b0a927cdaa3264ac637d44c4b219f199eb94c353',
|
||||
} as const;
|
||||
|
||||
const CYRILLIC_GOLDEN_SHA256 = {
|
||||
10_000: 'bc4f3c2a3d507db459dff2d4a723d8fa8729fb957cc436f062e703564201a20b',
|
||||
50_000: '4a3fc713548a7ab50c70dc0751eb5b2483122af9d01f7d3975f309dc820c6a04',
|
||||
100_000: '94727a8452920c07727812f4ec78b77deac353f5a46ee096620efba2296b7e2b',
|
||||
} as const;
|
||||
|
||||
describe('synthetic M3U performance fixture', () => {
|
||||
it('generates each supported catalog deterministically with exact metadata', () => {
|
||||
assert.deepEqual(
|
||||
@@ -47,6 +54,72 @@ describe('synthetic M3U performance fixture', () => {
|
||||
}
|
||||
});
|
||||
|
||||
it('keeps the latin1 option identical to the default fixture', () => {
|
||||
const byDefault = createSyntheticM3uFixture(10_000);
|
||||
const explicit = createSyntheticM3uFixture(10_000, {
|
||||
charset: 'latin1',
|
||||
});
|
||||
|
||||
assert.equal(byDefault.charset, 'latin1');
|
||||
assert.equal(explicit.body, byDefault.body);
|
||||
assert.equal(explicit.sha256, GOLDEN_SHA256[10_000]);
|
||||
// UTF-8 byte length equals UTF-16 length only for ASCII-only text.
|
||||
assert.equal(byDefault.bytes, byDefault.body.length);
|
||||
});
|
||||
|
||||
it('generates each cyrillic catalog deterministically', () => {
|
||||
for (const channelCount of SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS) {
|
||||
const cyrillic = createSyntheticM3uFixture(channelCount, {
|
||||
charset: 'cyrillic',
|
||||
});
|
||||
|
||||
assert.equal(cyrillic.charset, 'cyrillic');
|
||||
assert.equal(cyrillic.channelCount, channelCount);
|
||||
assert.equal(cyrillic.sha256, CYRILLIC_GOLDEN_SHA256[channelCount]);
|
||||
}
|
||||
});
|
||||
|
||||
it('keeps the cyrillic layout identical apart from titles', () => {
|
||||
const latin = createSyntheticM3uFixture(10_000);
|
||||
const cyrillic = createSyntheticM3uFixture(10_000, {
|
||||
charset: 'cyrillic',
|
||||
});
|
||||
|
||||
assert.equal(cyrillic.bytes, Buffer.byteLength(cyrillic.body, 'utf8'));
|
||||
assert.equal(
|
||||
cyrillic.sha256,
|
||||
createHash('sha256').update(cyrillic.body, 'utf8').digest('hex')
|
||||
);
|
||||
assert.equal(cyrillic.body.length, latin.body.length);
|
||||
assert.equal(toLatinTitles(cyrillic.body), latin.body);
|
||||
assertOnlyLoopbackUrls(cyrillic.body);
|
||||
});
|
||||
|
||||
it('puts characters outside Latin-1 into every cyrillic channel entry', () => {
|
||||
const lines = createSyntheticM3uFixture(10_000, { charset: 'cyrillic' })
|
||||
.body.trimEnd()
|
||||
.split('\n');
|
||||
|
||||
for (const line of lines.filter((entry) =>
|
||||
entry.startsWith('#EXTINF:')
|
||||
)) {
|
||||
assert.ok(hasCharacterAbove(line, 0xff), line);
|
||||
}
|
||||
for (const line of lines.filter((entry) => entry.startsWith('http'))) {
|
||||
assert.equal(hasCharacterAbove(line, 0x7f), false, line);
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects unsupported charsets', () => {
|
||||
assert.throws(
|
||||
() =>
|
||||
createSyntheticM3uFixture(10_000, {
|
||||
charset: 'greek' as never,
|
||||
}),
|
||||
/Unsupported synthetic charset/
|
||||
);
|
||||
});
|
||||
|
||||
it('rejects unsupported channel counts', () => {
|
||||
assert.throws(
|
||||
() => createSyntheticM3uFixture(9_999),
|
||||
@@ -55,6 +128,29 @@ describe('synthetic M3U performance fixture', () => {
|
||||
});
|
||||
});
|
||||
|
||||
function toLatinTitles(body: string): string {
|
||||
const latin = SYNTHETIC_TITLE_VOCABULARY.latin1;
|
||||
const cyrillic = SYNTHETIC_TITLE_VOCABULARY.cyrillic;
|
||||
return replaceEvery(
|
||||
replaceEvery(body, cyrillic.group, latin.group),
|
||||
cyrillic.channel,
|
||||
latin.channel
|
||||
);
|
||||
}
|
||||
|
||||
function hasCharacterAbove(text: string, maxCodeUnit: number): boolean {
|
||||
for (let index = 0; index < text.length; index += 1) {
|
||||
if (text.charCodeAt(index) > maxCodeUnit) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function replaceEvery(text: string, search: string, replacement: string) {
|
||||
return text.split(search).join(replacement);
|
||||
}
|
||||
|
||||
function assertOnlyLoopbackUrls(body: string): void {
|
||||
const urls = body.match(/https?:\/\/[^\s"]+/g) ?? [];
|
||||
|
||||
|
||||
@@ -1,5 +1,11 @@
|
||||
import { createHash } from 'node:crypto';
|
||||
|
||||
import {
|
||||
resolveSyntheticCharset,
|
||||
type SyntheticCharset,
|
||||
SYNTHETIC_TITLE_VOCABULARY,
|
||||
} from './synthetic-charset';
|
||||
|
||||
export const SYNTHETIC_M3U_SEED = 240_724;
|
||||
export const SYNTHETIC_M3U_CHANNEL_COUNT = 100_000;
|
||||
export const SUPPORTED_SYNTHETIC_M3U_CHANNEL_COUNTS = [
|
||||
@@ -15,13 +21,22 @@ export interface SyntheticM3uFixture {
|
||||
readonly body: string;
|
||||
readonly bytes: number;
|
||||
readonly channelCount: SyntheticM3uChannelCount;
|
||||
readonly charset: SyntheticCharset;
|
||||
readonly sha256: string;
|
||||
}
|
||||
|
||||
export interface SyntheticM3uFixtureOptions {
|
||||
/** Script of channel names and group titles; defaults to `latin1`. */
|
||||
readonly charset?: SyntheticCharset;
|
||||
}
|
||||
|
||||
export function createSyntheticM3uFixture(
|
||||
channelCount: number = SYNTHETIC_M3U_CHANNEL_COUNT
|
||||
channelCount: number = SYNTHETIC_M3U_CHANNEL_COUNT,
|
||||
options: SyntheticM3uFixtureOptions = {}
|
||||
): SyntheticM3uFixture {
|
||||
assertSupportedChannelCount(channelCount);
|
||||
const charset = resolveSyntheticCharset(options.charset);
|
||||
const titles = SYNTHETIC_TITLE_VOCABULARY[charset];
|
||||
const lines = new Array<string>(1 + channelCount * 2);
|
||||
lines[0] = '#EXTM3U';
|
||||
|
||||
@@ -31,12 +46,12 @@ export function createSyntheticM3uFixture(
|
||||
const lineOffset = 1 + offset * 2;
|
||||
const stableIndex = String(index).padStart(6, '0');
|
||||
|
||||
lines[lineOffset] = `#EXTINF:-1 group-title="Synthetic Group ${String(
|
||||
lines[lineOffset] = `#EXTINF:-1 group-title="${titles.group} ${String(
|
||||
group
|
||||
).padStart(
|
||||
3,
|
||||
'0'
|
||||
)}",Synthetic Channel ${SYNTHETIC_M3U_SEED}-${stableIndex}`;
|
||||
)}",${titles.channel} ${SYNTHETIC_M3U_SEED}-${stableIndex}`;
|
||||
lines[lineOffset + 1] =
|
||||
`http://127.0.0.1/stream/${SYNTHETIC_M3U_SEED}/${index}`;
|
||||
}
|
||||
@@ -46,6 +61,7 @@ export function createSyntheticM3uFixture(
|
||||
body,
|
||||
bytes: Buffer.byteLength(body, 'utf8'),
|
||||
channelCount,
|
||||
charset,
|
||||
sha256: createHash('sha256').update(body, 'utf8').digest('hex'),
|
||||
});
|
||||
}
|
||||
|
||||
@@ -0,0 +1,156 @@
|
||||
import assert from 'node:assert/strict';
|
||||
import { createHash } from 'node:crypto';
|
||||
import { describe, it } from 'node:test';
|
||||
|
||||
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
|
||||
import {
|
||||
createSyntheticXmltvFixture,
|
||||
SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS,
|
||||
SYNTHETIC_XMLTV_PROGRAMME_COUNT,
|
||||
SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL,
|
||||
} from './synthetic-xmltv';
|
||||
|
||||
const GOLDEN_SHA256 = {
|
||||
latin1: {
|
||||
10_000: 'e9d0ca0f7b73efc5cdee42cb2c4ee5bb4564b274b5042b13f8cd0d965d5011bc',
|
||||
50_000: '08b955c84822b7eb0d825a73eef6a6abc88b6d372c2dd32ad25d3969d667fd49',
|
||||
100_000:
|
||||
'932f0ba7089cab9e205a78142a245ebb497826ee7f1f4819603d77eba8e95a9e',
|
||||
},
|
||||
cyrillic: {
|
||||
10_000: '5b0df408e90c77bebe296fe1cc7e97cec5b929ebc7ad21b7480a6849a29f912b',
|
||||
50_000: 'bb1a0ca9fe874ab847b062c768868d76da750366aaa680bc712cb4cb7bfbf809',
|
||||
100_000:
|
||||
'cd637f4a25d47592116417a64bbb8c5b48d87f5bb5b0b4693f7af6c7603a8859',
|
||||
},
|
||||
} as const;
|
||||
|
||||
describe('synthetic XMLTV performance fixture', () => {
|
||||
it('generates each supported schedule deterministically', () => {
|
||||
assert.deepEqual(
|
||||
SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS,
|
||||
[10_000, 50_000, 100_000]
|
||||
);
|
||||
assert.equal(SYNTHETIC_XMLTV_PROGRAMME_COUNT, 50_000);
|
||||
|
||||
for (const charset of ['latin1', 'cyrillic'] as const) {
|
||||
for (const programmeCount of SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS) {
|
||||
const fixture = createSyntheticXmltvFixture(programmeCount, {
|
||||
charset,
|
||||
});
|
||||
|
||||
assert.equal(fixture.charset, charset);
|
||||
assert.equal(fixture.programmeCount, programmeCount);
|
||||
assert.equal(
|
||||
fixture.channelCount,
|
||||
programmeCount / SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL
|
||||
);
|
||||
assert.equal(
|
||||
fixture.sha256,
|
||||
GOLDEN_SHA256[charset][programmeCount]
|
||||
);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it('reports exact metadata and element counts', () => {
|
||||
for (const charset of ['latin1', 'cyrillic'] as const) {
|
||||
const fixture = createSyntheticXmltvFixture(10_000, { charset });
|
||||
|
||||
assert.equal(
|
||||
fixture.bytes,
|
||||
Buffer.byteLength(fixture.body, 'utf8')
|
||||
);
|
||||
assert.equal(
|
||||
fixture.sha256,
|
||||
createHash('sha256').update(fixture.body, 'utf8').digest('hex')
|
||||
);
|
||||
assert.equal(countOf(fixture.body, '<channel '), 100);
|
||||
assert.equal(countOf(fixture.body, '<programme '), 10_000);
|
||||
assert.ok(fixture.body.startsWith('<?xml version="1.0"'));
|
||||
assert.ok(fixture.body.endsWith('</tv>\n'));
|
||||
}
|
||||
});
|
||||
|
||||
it('defaults to 50k programmes with an ASCII-only latin1 body', () => {
|
||||
const fixture = createSyntheticXmltvFixture();
|
||||
|
||||
assert.equal(fixture.charset, 'latin1');
|
||||
assert.equal(fixture.programmeCount, 50_000);
|
||||
// UTF-8 byte length equals UTF-16 length only for ASCII-only text.
|
||||
assert.equal(fixture.bytes, fixture.body.length);
|
||||
});
|
||||
|
||||
it('keeps the cyrillic layout identical apart from titles', () => {
|
||||
const latin = createSyntheticXmltvFixture(10_000);
|
||||
const cyrillic = createSyntheticXmltvFixture(10_000, {
|
||||
charset: 'cyrillic',
|
||||
});
|
||||
const latinTitles = SYNTHETIC_TITLE_VOCABULARY.latin1;
|
||||
const cyrillicTitles = SYNTHETIC_TITLE_VOCABULARY.cyrillic;
|
||||
|
||||
assert.equal(cyrillic.body.length, latin.body.length);
|
||||
const translated = [
|
||||
[cyrillicTitles.channel, latinTitles.channel],
|
||||
[cyrillicTitles.programme, latinTitles.programme],
|
||||
[
|
||||
`lang="${cyrillicTitles.language}"`,
|
||||
`lang="${latinTitles.language}"`,
|
||||
],
|
||||
].reduce(
|
||||
(text, [search, replacement]) =>
|
||||
text.split(search).join(replacement),
|
||||
cyrillic.body
|
||||
);
|
||||
assert.equal(translated, latin.body);
|
||||
const firstTitle = cyrillic.body.indexOf('<title lang="ru">');
|
||||
assert.ok(firstTitle > 0);
|
||||
assert.ok(
|
||||
cyrillic.body.charCodeAt(firstTitle + '<title lang="ru">'.length) >
|
||||
0xff
|
||||
);
|
||||
});
|
||||
|
||||
it('schedules consecutive half-hour programmes per channel', () => {
|
||||
const lines = createSyntheticXmltvFixture(10_000).body.split('\n');
|
||||
const firstProgramme = lines.find((line) =>
|
||||
line.includes('<programme ')
|
||||
);
|
||||
|
||||
assert.equal(
|
||||
firstProgramme,
|
||||
' <programme start="20260101000000 +0000" stop="20260101003000 +0000" channel="synthetic.000001"><title lang="en">Synthetic Programme 000001-0001</title><desc lang="en">Synthetic description for slot 1.</desc><category lang="en">Synthetic</category></programme>'
|
||||
);
|
||||
assert.ok(
|
||||
lines.includes(
|
||||
' <programme start="20260103013000 +0000" stop="20260103020000 +0000" channel="synthetic.000100"><title lang="en">Synthetic Programme 000100-0100</title><desc lang="en">Synthetic description for slot 100.</desc><category lang="en">Synthetic</category></programme>'
|
||||
)
|
||||
);
|
||||
});
|
||||
|
||||
it('rejects unsupported programme counts and charsets', () => {
|
||||
assert.throws(
|
||||
() => createSyntheticXmltvFixture(12_345),
|
||||
/Unsupported synthetic XMLTV programme count/
|
||||
);
|
||||
assert.throws(
|
||||
() =>
|
||||
createSyntheticXmltvFixture(10_000, {
|
||||
charset: 'arabic' as never,
|
||||
}),
|
||||
/Unsupported synthetic charset/
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
function countOf(body: string, needle: string): number {
|
||||
let count = 0;
|
||||
for (
|
||||
let index = body.indexOf(needle);
|
||||
index !== -1;
|
||||
index = body.indexOf(needle, index + needle.length)
|
||||
) {
|
||||
count += 1;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
@@ -0,0 +1,138 @@
|
||||
import { createHash } from 'node:crypto';
|
||||
|
||||
import {
|
||||
resolveSyntheticCharset,
|
||||
type SyntheticCharset,
|
||||
SYNTHETIC_TITLE_VOCABULARY,
|
||||
} from './synthetic-charset';
|
||||
import { SYNTHETIC_M3U_SEED } from './synthetic-m3u';
|
||||
|
||||
export const SYNTHETIC_XMLTV_PROGRAMME_COUNT = 50_000;
|
||||
export const SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL = 100;
|
||||
export const SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS = [
|
||||
10_000,
|
||||
SYNTHETIC_XMLTV_PROGRAMME_COUNT,
|
||||
100_000,
|
||||
] as const;
|
||||
|
||||
/** 2026-01-01T00:00:00Z; every programme lasts 30 minutes. */
|
||||
const SCHEDULE_START_EPOCH_MS = Date.UTC(2026, 0, 1);
|
||||
const PROGRAMME_DURATION_MS = 30 * 60 * 1000;
|
||||
|
||||
export type SyntheticXmltvProgrammeCount =
|
||||
(typeof SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS)[number];
|
||||
|
||||
export interface SyntheticXmltvFixture {
|
||||
readonly body: string;
|
||||
readonly bytes: number;
|
||||
readonly channelCount: number;
|
||||
readonly charset: SyntheticCharset;
|
||||
readonly programmeCount: SyntheticXmltvProgrammeCount;
|
||||
readonly sha256: string;
|
||||
}
|
||||
|
||||
export interface SyntheticXmltvFixtureOptions {
|
||||
/** Script of channel display names and programme titles. */
|
||||
readonly charset?: SyntheticCharset;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deterministic XMLTV document: all `<channel>` entries first, then
|
||||
* `SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL` consecutive programmes per
|
||||
* channel. Descriptions, categories and attributes stay ASCII in every
|
||||
* charset so only the titles decide the string encoding.
|
||||
*/
|
||||
export function createSyntheticXmltvFixture(
|
||||
programmeCount: number = SYNTHETIC_XMLTV_PROGRAMME_COUNT,
|
||||
options: SyntheticXmltvFixtureOptions = {}
|
||||
): SyntheticXmltvFixture {
|
||||
assertSupportedProgrammeCount(programmeCount);
|
||||
const charset = resolveSyntheticCharset(options.charset);
|
||||
const titles = SYNTHETIC_TITLE_VOCABULARY[charset];
|
||||
const channelCount =
|
||||
programmeCount / SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL;
|
||||
const lines: string[] = [
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<tv generator-info-name="iptvnator-synthetic">',
|
||||
];
|
||||
|
||||
for (let offset = 0; offset < channelCount; offset += 1) {
|
||||
const channelId = syntheticChannelId(offset);
|
||||
lines.push(
|
||||
` <channel id="${channelId}"><display-name lang="${titles.language}">${
|
||||
titles.channel
|
||||
} ${SYNTHETIC_M3U_SEED}-${stableNumber(offset + 1, 6)}</display-name></channel>`
|
||||
);
|
||||
}
|
||||
|
||||
const slotTimestamps = Array.from(
|
||||
{ length: SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL + 1 },
|
||||
(_, slot) =>
|
||||
xmltvTimestamp(
|
||||
SCHEDULE_START_EPOCH_MS + slot * PROGRAMME_DURATION_MS
|
||||
)
|
||||
);
|
||||
for (let offset = 0; offset < channelCount; offset += 1) {
|
||||
const channelId = syntheticChannelId(offset);
|
||||
const channelNumber = stableNumber(offset + 1, 6);
|
||||
for (
|
||||
let slot = 0;
|
||||
slot < SYNTHETIC_XMLTV_PROGRAMMES_PER_CHANNEL;
|
||||
slot += 1
|
||||
) {
|
||||
lines.push(
|
||||
` <programme start="${slotTimestamps[slot]}" stop="${
|
||||
slotTimestamps[slot + 1]
|
||||
}" channel="${channelId}"><title lang="${titles.language}">${
|
||||
titles.programme
|
||||
} ${channelNumber}-${stableNumber(
|
||||
slot + 1,
|
||||
4
|
||||
)}</title><desc lang="en">Synthetic description for slot ${
|
||||
slot + 1
|
||||
}.</desc><category lang="en">Synthetic</category></programme>`
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
lines.push('</tv>');
|
||||
const body = `${lines.join('\n')}\n`;
|
||||
return Object.freeze({
|
||||
body,
|
||||
bytes: Buffer.byteLength(body, 'utf8'),
|
||||
channelCount,
|
||||
charset,
|
||||
programmeCount,
|
||||
sha256: createHash('sha256').update(body, 'utf8').digest('hex'),
|
||||
});
|
||||
}
|
||||
|
||||
function syntheticChannelId(offset: number): string {
|
||||
return `synthetic.${stableNumber(offset + 1, 6)}`;
|
||||
}
|
||||
|
||||
function stableNumber(value: number, width: number): string {
|
||||
return String(value).padStart(width, '0');
|
||||
}
|
||||
|
||||
function xmltvTimestamp(epochMs: number): string {
|
||||
const iso = new Date(epochMs).toISOString();
|
||||
return `${iso.slice(0, 4)}${iso.slice(5, 7)}${iso.slice(8, 10)}${iso.slice(
|
||||
11,
|
||||
13
|
||||
)}${iso.slice(14, 16)}${iso.slice(17, 19)} +0000`;
|
||||
}
|
||||
|
||||
function assertSupportedProgrammeCount(
|
||||
programmeCount: number
|
||||
): asserts programmeCount is SyntheticXmltvProgrammeCount {
|
||||
if (
|
||||
!SUPPORTED_SYNTHETIC_XMLTV_PROGRAMME_COUNTS.some(
|
||||
(supported) => supported === programmeCount
|
||||
)
|
||||
) {
|
||||
throw new Error(
|
||||
`Unsupported synthetic XMLTV programme count: ${programmeCount}`
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -382,6 +382,43 @@ in all eighteen runner iterations; the `spawnToFirstCardMs` P50 ranged from
|
||||
differ from a Mac (12 and 571 there, the fast path without the Linux-only
|
||||
`getWindowState` call), so take J1 baseline values from the runner only.
|
||||
|
||||
## Charset parse benchmark
|
||||
|
||||
V8 stores a string as two-byte UTF-16 once one character falls outside
|
||||
Latin-1, and substrings of such a string stay two-byte, even ASCII-only URL
|
||||
lines. `src/performance/charset-parse.benchmark.ts` checks whether that slows
|
||||
playlist and EPG parsing. It is a Node benchmark, not a journey, and is not
|
||||
ratcheted:
|
||||
|
||||
```bash
|
||||
pnpm nx run electron-backend-e2e:benchmark-charset-parse --iterations=5
|
||||
```
|
||||
|
||||
It parses 50,000 M3U channels (`iptv-playlist-parser`, then
|
||||
`createPlaylistObject`, the main-process `PARSE_M3U` and `NORMALIZE` phases)
|
||||
and 50,000 XMLTV programmes (`StreamingEpgParser`, the EPG worker's parser).
|
||||
Each workload runs on three inputs: `latin1` and `cyrillic` from the
|
||||
synthetic generators (`charset` option of `synthetic-m3u.ts` and
|
||||
`synthetic-xmltv.ts`, identical layout apart from titles), and `latin1-bom`,
|
||||
the latin1 bytes behind a UTF-8 byte-order mark. The BOM forces two-byte
|
||||
storage without changing content, which separates the encoding cost from
|
||||
the effect that non-ASCII titles have on ASCII-only regexes. The XMLTV
|
||||
parser receives 64 Ki-character slices of one decoded string rather than
|
||||
per-chunk decoded buffers: slices keep the input's representation (a
|
||||
per-chunk decode would make the BOM control one-byte after its first
|
||||
chunk), and no multi-byte character is split. Before timing, an untimed
|
||||
pass checks that the parsed titles match the fixture.
|
||||
|
||||
The report gives P50 wall-clock and CPU time after one warm-up, plus
|
||||
CPU-profile sample counts and top self frames from a separate profiled pass.
|
||||
Inputs alternate within each round and the starting input rotates between
|
||||
rounds. Prefer CPU time and samples on a busy machine.
|
||||
|
||||
The 2026-09-27 measurement (plan item D1) found every workload under the 1.5x
|
||||
threshold on Node 22 and inside Electron 43, so D2 regex prefilters were not
|
||||
applied. Rerun the benchmark after changing either parser or when a user
|
||||
reports slow imports of non-Latin playlists.
|
||||
|
||||
## Adding a counter
|
||||
|
||||
1. Produce the value from the built output or from a deterministic probe, not
|
||||
|
||||
Reference in new issue
Block a user