test(performance): keep charset benchmark inputs in their representation

Review follow-up for the D1 charset benchmark.

- XMLTV: feed the parser 64 Ki-character slices of one decoded string.
  Decoding each 64 KiB buffer separately left the BOM control one-byte
  after its first chunk and could split Cyrillic characters into U+FFFD.
- Verify parsed M3U names/groups and XMLTV titles against the fixture in
  an untimed pass, including that no title contains U+FFFD.
- Rotate the starting input every round so no variant always runs first.
- Move workload definitions to charset-parse-workloads.ts to keep the
  runner under the line target.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
4grayandClaude Opus 5.5 committed 2026-09-27 20:56:05 +02:00
1 parent 6aaf22ccec
commit 284b7748be
5 files changed
+255 -139

No files matched your search

@@ -9,6 +9,7 @@ import {
formatCharsetReport,
medianOf,
summarizeSelfSamples,
variantOrderForRound,
} from './charset-parse-benchmark-report';
function measurement(
@@ -31,6 +32,31 @@ function measurement(
}
describe('charset parse benchmark report', () => {
it('rotates the starting variant every round', () => {
assert.deepEqual(variantOrderForRound(0), [
'latin1',
'latin1-bom',
'cyrillic',
]);
assert.deepEqual(variantOrderForRound(1), [
'latin1-bom',
'cyrillic',
'latin1',
]);
assert.deepEqual(variantOrderForRound(2), [
'cyrillic',
'latin1',
'latin1-bom',
]);
assert.deepEqual(variantOrderForRound(3), variantOrderForRound(0));
for (let round = 0; round < 6; round += 1) {
assert.deepEqual(
[...variantOrderForRound(round)].sort(),
[...CHARSET_BENCHMARK_VARIANTS].sort()
);
}
});
it('takes the median of odd and even samples without mutating them', () => {
const values = [5, 1, 3];
@@ -65,6 +65,16 @@ export interface CharsetComparison {
readonly variants: readonly CharsetVariantSummary[];
}
/**
* Variant order for one benchmark round. The starting variant rotates every
* round so no input always runs first or last after a garbage collection.
*/
export function variantOrderForRound(round: number): CharsetBenchmarkVariant[] {
const variants = [...CHARSET_BENCHMARK_VARIANTS];
const offset = round % variants.length;
return [...variants.slice(offset), ...variants.slice(0, offset)];
}
export function medianOf(values: readonly number[]): number {
if (values.length === 0) {
throw new Error('Cannot take the median of an empty sample');
@@ -0,0 +1,191 @@
import { createPlaylistObject } from '@iptvnator/shared/m3u-utils';
import { parse } from 'iptv-playlist-parser';
import { resolve } from 'node:path';
import type { CharsetBenchmarkVariant } from './charset-parse-benchmark-report';
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
import { createSyntheticM3uFixture, SYNTHETIC_M3U_SEED } from './synthetic-m3u';
import { createSyntheticXmltvFixture } from './synthetic-xmltv';
export const ENTRY_COUNT = 50_000;
/**
* The EPG worker writes 64 KiB decoded chunks. The benchmark writes slices of
* one decoded string instead, so every slice keeps the input's one-byte or
* two-byte representation (decoding each chunk separately would make the
* BOM control one-byte after its first chunk) and no character is split.
*/
const XMLTV_CHUNK_CHARS = 64 * 1024;
const REPLACEMENT_CHARACTER = '\ufffd';
const UTF8_BYTE_ORDER_MARK = Buffer.from([0xef, 0xbb, 0xbf]);
type Variant = CharsetBenchmarkVariant;
interface ParsedProgramme {
readonly title: readonly { readonly value: string }[];
}
interface StreamingEpgParserModule {
StreamingEpgParser: new (
onChannels: (channels: unknown[]) => void,
onPrograms: (programs: ParsedProgramme[]) => void,
onProgress: (channels: number, programs: number) => void,
channelBatchSize?: number,
programBatchSize?: number
) => { write(chunk: string): void; finish(): { totalPrograms: number } };
}
export interface Workload {
readonly name: string;
/** Prepares untimed input, then returns the timed operation. */
prepare(variant: Variant): () => void;
input(variant: Variant): string;
/** Untimed check that the parser saw the fixture's exact titles. */
verify(variant: Variant): void;
}
export async function createWorkloads(): Promise<Workload[]> {
const epgModule = (await import(
resolve(
__dirname,
'../../../electron-backend/src/app/workers/epg-streaming-parser.ts'
)
)) as StreamingEpgParserModule;
const m3uBuffers = buffersPerVariant((charset) =>
createSyntheticM3uFixture(ENTRY_COUNT, { charset })
);
const xmltvBuffers = buffersPerVariant((charset) =>
createSyntheticXmltvFixture(ENTRY_COUNT, { charset })
);
const m3uInput = (variant: Variant) =>
requireBuffer(m3uBuffers, variant).toString('utf8');
const xmltvInput = (variant: Variant) =>
requireBuffer(xmltvBuffers, variant).toString('utf8');
const verifyM3u = (variant: Variant) => {
const titles = titlesFor(variant);
const [first] = parse(m3uInput(variant)).items;
expectEqual(
first?.name,
`${titles.channel} ${SYNTHETIC_M3U_SEED}-000001`
);
expectEqual(first?.group.title, `${titles.group} 001`);
};
const parseXmltv = (
variant: Variant,
onPrograms: (programs: ParsedProgramme[]) => void
) => {
const chunks = sliceText(xmltvInput(variant), XMLTV_CHUNK_CHARS);
return () => {
const parser = new epgModule.StreamingEpgParser(
() => undefined,
onPrograms,
() => undefined,
100,
1000
);
for (const chunk of chunks) {
parser.write(chunk);
}
assertCount(parser.finish().totalPrograms);
};
};
return [
{
// PARSE_M3U phase of the main-process import.
name: 'm3u parse',
input: m3uInput,
verify: verifyM3u,
prepare(variant) {
const body = m3uInput(variant);
return () => assertCount(parse(body).items.length);
},
},
{
// NORMALIZE phase of the main-process import.
name: 'm3u normalize',
input: m3uInput,
verify: verifyM3u,
prepare(variant) {
const parsed = parse(m3uInput(variant));
return () =>
assertCount(
createPlaylistObject('benchmark', parsed, 'x', 'URL')
.count
);
},
},
{
// EPG worker's saxes-based parser; UTF-8 decoding is excluded.
name: 'xmltv stream parse',
input: xmltvInput,
verify(variant) {
const titles: string[] = [];
parseXmltv(variant, (programs) => {
for (const programme of programs) {
titles.push(programme.title[0]?.value ?? '');
}
})();
const programme = titlesFor(variant).programme;
expectEqual(titles[0], `${programme} 000001-0001`);
expectEqual(titles.at(-1), `${programme} 000500-0100`);
if (
titles.some((title) =>
title.includes(REPLACEMENT_CHARACTER)
)
) {
throw new Error(`${variant} XMLTV titles contain U+FFFD`);
}
},
prepare: (variant) => parseXmltv(variant, () => undefined),
},
];
}
function titlesFor(variant: Variant) {
return SYNTHETIC_TITLE_VOCABULARY[
variant === 'cyrillic' ? 'cyrillic' : 'latin1'
];
}
function sliceText(text: string, size: number): string[] {
const chunks: string[] = [];
for (let offset = 0; offset < text.length; offset += size) {
chunks.push(text.slice(offset, offset + size));
}
return chunks;
}
function expectEqual(actual: string | undefined, expected: string): void {
if (actual !== expected) {
throw new Error(`Expected "${expected}", parsed "${String(actual)}"`);
}
}
function buffersPerVariant(
create: (charset: 'latin1' | 'cyrillic') => { body: string }
): Map<Variant, Buffer> {
const latin1 = Buffer.from(create('latin1').body, 'utf8');
return new Map<Variant, Buffer>([
['latin1', latin1],
['latin1-bom', Buffer.concat([UTF8_BYTE_ORDER_MARK, latin1])],
['cyrillic', Buffer.from(create('cyrillic').body, 'utf8')],
]);
}
function requireBuffer(
buffers: Map<Variant, Buffer>,
variant: Variant
): Buffer {
const buffer = buffers.get(variant);
if (!buffer) {
throw new Error(`No fixture for ${variant}`);
}
return buffer;
}
function assertCount(count: number): void {
if (count !== ENTRY_COUNT) {
throw new Error(`Expected ${ENTRY_COUNT} entries, parsed ${count}`);
}
}
@@ -18,8 +18,6 @@
* "$(node -p "require('electron')")" --expose-gc --import tsx \
* src/performance/charset-parse.benchmark.ts
*/
import { createPlaylistObject } from '@iptvnator/shared/m3u-utils';
import { parse } from 'iptv-playlist-parser';
import { writeFile } from 'node:fs/promises';
import { Session } from 'node:inspector/promises';
import { resolve } from 'node:path';
@@ -34,34 +32,16 @@ import {
formatCharsetReport,
summarizeSelfSamples,
topFrames,
variantOrderForRound,
} from './charset-parse-benchmark-report';
import { createSyntheticM3uFixture } from './synthetic-m3u';
import { createSyntheticXmltvFixture } from './synthetic-xmltv';
const ENTRY_COUNT = 50_000;
/** Matches the default highWaterMark of the worker's HTTP/decoder stream. */
const XMLTV_CHUNK_BYTES = 64 * 1024;
const UTF8_BYTE_ORDER_MARK = Buffer.from([0xef, 0xbb, 0xbf]);
import {
createWorkloads,
ENTRY_COUNT,
type Workload,
} from './charset-parse-workloads';
type Variant = CharsetBenchmarkVariant;
interface StreamingEpgParserModule {
StreamingEpgParser: new (
onChannels: (channels: unknown[]) => void,
onPrograms: (programs: unknown[]) => void,
onProgress: (channels: number, programs: number) => void,
channelBatchSize?: number,
programBatchSize?: number
) => { write(chunk: string): void; finish(): { totalPrograms: number } };
}
interface Workload {
readonly name: string;
/** Prepares untimed input, then returns the timed operation. */
prepare(variant: Variant): () => void;
input(variant: Variant): string;
}
async function main(): Promise<void> {
const { values } = parseArgs({
options: {
@@ -87,14 +67,17 @@ async function main(): Promise<void> {
const measurements: CharsetCaseMeasurement[] = [];
for (const workload of workloads) {
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
workload.verify(variant);
}
const wall = new Map<Variant, number[]>();
const cpu = new Map<Variant, number[]>();
const profiles = new Map<Variant, CpuProfileLike[]>();
// Variants alternate inside every round so JIT warm-up and heap growth
// do not favour whichever variant runs last. Rounds after the timed
// ones run under the profiler, which perturbs timing.
// Variants alternate inside every round, starting with a different
// one each round, so JIT warm-up and heap growth favour no input.
// Rounds after the timed ones run under the profiler.
for (let round = 0; round < warmup + iterations * 2; round += 1) {
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
for (const variant of variantOrderForRound(round)) {
const run = workload.prepare(variant);
collectGarbage();
const profiled = round >= warmup + iterations;
@@ -151,79 +134,6 @@ async function main(): Promise<void> {
}
}
async function createWorkloads(): Promise<Workload[]> {
const epgModule = (await import(
resolve(
__dirname,
'../../../electron-backend/src/app/workers/epg-streaming-parser.ts'
)
)) as StreamingEpgParserModule;
const m3uBuffers = buffersPerVariant((charset) =>
createSyntheticM3uFixture(ENTRY_COUNT, { charset })
);
const xmltvBuffers = buffersPerVariant((charset) =>
createSyntheticXmltvFixture(ENTRY_COUNT, { charset })
);
const m3uInput = (variant: Variant) =>
requireBuffer(m3uBuffers, variant).toString('utf8');
const xmltvInput = (variant: Variant) =>
requireBuffer(xmltvBuffers, variant).toString('utf8');
return [
{
// PARSE_M3U phase of the main-process import.
name: 'm3u parse',
input: m3uInput,
prepare(variant) {
const body = m3uInput(variant);
return () => assertCount(parse(body).items.length);
},
},
{
// NORMALIZE phase of the main-process import.
name: 'm3u normalize',
input: m3uInput,
prepare(variant) {
const parsed = parse(m3uInput(variant));
return () =>
assertCount(
createPlaylistObject('benchmark', parsed, 'x', 'URL')
.count
);
},
},
{
// EPG worker: per-chunk UTF-8 decode plus the saxes-based parser.
name: 'xmltv stream parse',
input: xmltvInput,
prepare(variant) {
const buffer = requireBuffer(xmltvBuffers, variant);
return () => {
const parser = new epgModule.StreamingEpgParser(
() => undefined,
() => undefined,
() => undefined,
100,
1000
);
for (
let offset = 0;
offset < buffer.length;
offset += XMLTV_CHUNK_BYTES
) {
parser.write(
buffer
.subarray(offset, offset + XMLTV_CHUNK_BYTES)
.toString('utf-8')
);
}
assertCount(parser.finish().totalPrograms);
};
},
},
];
}
interface CollectedSamples {
readonly wall: Map<Variant, number[]>;
readonly cpu: Map<Variant, number[]>;
@@ -258,38 +168,10 @@ function measurementFor(
};
}
function buffersPerVariant(
create: (charset: 'latin1' | 'cyrillic') => { body: string }
): Map<Variant, Buffer> {
const latin1 = Buffer.from(create('latin1').body, 'utf8');
return new Map<Variant, Buffer>([
['latin1', latin1],
['latin1-bom', Buffer.concat([UTF8_BYTE_ORDER_MARK, latin1])],
['cyrillic', Buffer.from(create('cyrillic').body, 'utf8')],
]);
}
function requireBuffer(
buffers: Map<Variant, Buffer>,
variant: Variant
): Buffer {
const buffer = buffers.get(variant);
if (!buffer) {
throw new Error(`No fixture for ${variant}`);
}
return buffer;
}
function append<T>(map: Map<Variant, T[]>, variant: Variant, value: T): void {
map.set(variant, [...(map.get(variant) ?? []), value]);
}
function assertCount(count: number): void {
if (count !== ENTRY_COUNT) {
throw new Error(`Expected ${ENTRY_COUNT} entries, parsed ${count}`);
}
}
function collectGarbage(): void {
(globalThis as { gc?: () => void }).gc?.();
}
+15 -8
View File
@@ -343,18 +343,25 @@ pnpm nx run electron-backend-e2e:benchmark-charset-parse --iterations=5
It parses 50,000 M3U channels (`iptv-playlist-parser`, then
`createPlaylistObject`, the main-process `PARSE_M3U` and `NORMALIZE` phases)
and 50,000 XMLTV programmes (`StreamingEpgParser` fed 64 KiB chunks, as in the
EPG worker). Each workload runs on three inputs: `latin1` and `cyrillic`
from the synthetic generators (`charset` option of `synthetic-m3u.ts` and
and 50,000 XMLTV programmes (`StreamingEpgParser`, the EPG worker's parser).
Each workload runs on three inputs: `latin1` and `cyrillic` from the
synthetic generators (`charset` option of `synthetic-m3u.ts` and
`synthetic-xmltv.ts`, identical layout apart from titles), and `latin1-bom`,
the latin1 bytes behind a UTF-8 byte-order mark. The BOM forces two-byte
storage without changing content, which separates the encoding cost from
the effect that non-ASCII titles have on ASCII-only regexes. The report gives
P50 wall-clock and CPU time after one warm-up, plus CPU-profile sample counts
and top self frames from a separate profiled pass. Prefer CPU time and
samples on a busy machine.
the effect that non-ASCII titles have on ASCII-only regexes. The XMLTV
parser receives 64 Ki-character slices of one decoded string rather than
per-chunk decoded buffers: slices keep the input's representation (a
per-chunk decode would make the BOM control one-byte after its first
chunk), and no multi-byte character is split. Before timing, an untimed
pass checks that the parsed titles match the fixture.
The 2026-09-26 measurement (plan item D1) found every workload under the 1.5x
The report gives P50 wall-clock and CPU time after one warm-up, plus
CPU-profile sample counts and top self frames from a separate profiled pass.
Inputs alternate within each round and the starting input rotates between
rounds. Prefer CPU time and samples on a busy machine.
The 2026-09-27 measurement (plan item D1) found every workload under the 1.5x
threshold on Node 22 and inside Electron 43, so D2 regex prefilters were not
applied. Rerun the benchmark after changing either parser or when a user
reports slow imports of non-Latin playlists.