mirror of
https://github.com/4gray/iptvnator.git
synced 2026-10-08 09:01:03 -08:00
test(performance): keep charset benchmark inputs in their representation
Review follow-up for the D1 charset benchmark. - XMLTV: feed the parser 64 Ki-character slices of one decoded string. Decoding each 64 KiB buffer separately left the BOM control one-byte after its first chunk and could split Cyrillic characters into U+FFFD. - Verify parsed M3U names/groups and XMLTV titles against the fixture in an untimed pass, including that no title contains U+FFFD. - Rotate the starting input every round so no variant always runs first. - Move workload definitions to charset-parse-workloads.ts to keep the runner under the line target. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
6aaf22ccec
commit
284b7748be
5 files changed
+255
-139
No files matched your search
@@ -9,6 +9,7 @@ import {
|
||||
formatCharsetReport,
|
||||
medianOf,
|
||||
summarizeSelfSamples,
|
||||
variantOrderForRound,
|
||||
} from './charset-parse-benchmark-report';
|
||||
|
||||
function measurement(
|
||||
@@ -31,6 +32,31 @@ function measurement(
|
||||
}
|
||||
|
||||
describe('charset parse benchmark report', () => {
|
||||
it('rotates the starting variant every round', () => {
|
||||
assert.deepEqual(variantOrderForRound(0), [
|
||||
'latin1',
|
||||
'latin1-bom',
|
||||
'cyrillic',
|
||||
]);
|
||||
assert.deepEqual(variantOrderForRound(1), [
|
||||
'latin1-bom',
|
||||
'cyrillic',
|
||||
'latin1',
|
||||
]);
|
||||
assert.deepEqual(variantOrderForRound(2), [
|
||||
'cyrillic',
|
||||
'latin1',
|
||||
'latin1-bom',
|
||||
]);
|
||||
assert.deepEqual(variantOrderForRound(3), variantOrderForRound(0));
|
||||
for (let round = 0; round < 6; round += 1) {
|
||||
assert.deepEqual(
|
||||
[...variantOrderForRound(round)].sort(),
|
||||
[...CHARSET_BENCHMARK_VARIANTS].sort()
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
it('takes the median of odd and even samples without mutating them', () => {
|
||||
const values = [5, 1, 3];
|
||||
|
||||
|
||||
@@ -65,6 +65,16 @@ export interface CharsetComparison {
|
||||
readonly variants: readonly CharsetVariantSummary[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Variant order for one benchmark round. The starting variant rotates every
|
||||
* round so no input always runs first or last after a garbage collection.
|
||||
*/
|
||||
export function variantOrderForRound(round: number): CharsetBenchmarkVariant[] {
|
||||
const variants = [...CHARSET_BENCHMARK_VARIANTS];
|
||||
const offset = round % variants.length;
|
||||
return [...variants.slice(offset), ...variants.slice(0, offset)];
|
||||
}
|
||||
|
||||
export function medianOf(values: readonly number[]): number {
|
||||
if (values.length === 0) {
|
||||
throw new Error('Cannot take the median of an empty sample');
|
||||
|
||||
@@ -0,0 +1,191 @@
|
||||
import { createPlaylistObject } from '@iptvnator/shared/m3u-utils';
|
||||
import { parse } from 'iptv-playlist-parser';
|
||||
import { resolve } from 'node:path';
|
||||
|
||||
import type { CharsetBenchmarkVariant } from './charset-parse-benchmark-report';
|
||||
import { SYNTHETIC_TITLE_VOCABULARY } from './synthetic-charset';
|
||||
import { createSyntheticM3uFixture, SYNTHETIC_M3U_SEED } from './synthetic-m3u';
|
||||
import { createSyntheticXmltvFixture } from './synthetic-xmltv';
|
||||
|
||||
export const ENTRY_COUNT = 50_000;
|
||||
/**
|
||||
* The EPG worker writes 64 KiB decoded chunks. The benchmark writes slices of
|
||||
* one decoded string instead, so every slice keeps the input's one-byte or
|
||||
* two-byte representation (decoding each chunk separately would make the
|
||||
* BOM control one-byte after its first chunk) and no character is split.
|
||||
*/
|
||||
const XMLTV_CHUNK_CHARS = 64 * 1024;
|
||||
const REPLACEMENT_CHARACTER = '\ufffd';
|
||||
const UTF8_BYTE_ORDER_MARK = Buffer.from([0xef, 0xbb, 0xbf]);
|
||||
|
||||
type Variant = CharsetBenchmarkVariant;
|
||||
|
||||
interface ParsedProgramme {
|
||||
readonly title: readonly { readonly value: string }[];
|
||||
}
|
||||
|
||||
interface StreamingEpgParserModule {
|
||||
StreamingEpgParser: new (
|
||||
onChannels: (channels: unknown[]) => void,
|
||||
onPrograms: (programs: ParsedProgramme[]) => void,
|
||||
onProgress: (channels: number, programs: number) => void,
|
||||
channelBatchSize?: number,
|
||||
programBatchSize?: number
|
||||
) => { write(chunk: string): void; finish(): { totalPrograms: number } };
|
||||
}
|
||||
|
||||
export interface Workload {
|
||||
readonly name: string;
|
||||
/** Prepares untimed input, then returns the timed operation. */
|
||||
prepare(variant: Variant): () => void;
|
||||
input(variant: Variant): string;
|
||||
/** Untimed check that the parser saw the fixture's exact titles. */
|
||||
verify(variant: Variant): void;
|
||||
}
|
||||
|
||||
export async function createWorkloads(): Promise<Workload[]> {
|
||||
const epgModule = (await import(
|
||||
resolve(
|
||||
__dirname,
|
||||
'../../../electron-backend/src/app/workers/epg-streaming-parser.ts'
|
||||
)
|
||||
)) as StreamingEpgParserModule;
|
||||
const m3uBuffers = buffersPerVariant((charset) =>
|
||||
createSyntheticM3uFixture(ENTRY_COUNT, { charset })
|
||||
);
|
||||
const xmltvBuffers = buffersPerVariant((charset) =>
|
||||
createSyntheticXmltvFixture(ENTRY_COUNT, { charset })
|
||||
);
|
||||
const m3uInput = (variant: Variant) =>
|
||||
requireBuffer(m3uBuffers, variant).toString('utf8');
|
||||
const xmltvInput = (variant: Variant) =>
|
||||
requireBuffer(xmltvBuffers, variant).toString('utf8');
|
||||
|
||||
const verifyM3u = (variant: Variant) => {
|
||||
const titles = titlesFor(variant);
|
||||
const [first] = parse(m3uInput(variant)).items;
|
||||
expectEqual(
|
||||
first?.name,
|
||||
`${titles.channel} ${SYNTHETIC_M3U_SEED}-000001`
|
||||
);
|
||||
expectEqual(first?.group.title, `${titles.group} 001`);
|
||||
};
|
||||
const parseXmltv = (
|
||||
variant: Variant,
|
||||
onPrograms: (programs: ParsedProgramme[]) => void
|
||||
) => {
|
||||
const chunks = sliceText(xmltvInput(variant), XMLTV_CHUNK_CHARS);
|
||||
return () => {
|
||||
const parser = new epgModule.StreamingEpgParser(
|
||||
() => undefined,
|
||||
onPrograms,
|
||||
() => undefined,
|
||||
100,
|
||||
1000
|
||||
);
|
||||
for (const chunk of chunks) {
|
||||
parser.write(chunk);
|
||||
}
|
||||
assertCount(parser.finish().totalPrograms);
|
||||
};
|
||||
};
|
||||
|
||||
return [
|
||||
{
|
||||
// PARSE_M3U phase of the main-process import.
|
||||
name: 'm3u parse',
|
||||
input: m3uInput,
|
||||
verify: verifyM3u,
|
||||
prepare(variant) {
|
||||
const body = m3uInput(variant);
|
||||
return () => assertCount(parse(body).items.length);
|
||||
},
|
||||
},
|
||||
{
|
||||
// NORMALIZE phase of the main-process import.
|
||||
name: 'm3u normalize',
|
||||
input: m3uInput,
|
||||
verify: verifyM3u,
|
||||
prepare(variant) {
|
||||
const parsed = parse(m3uInput(variant));
|
||||
return () =>
|
||||
assertCount(
|
||||
createPlaylistObject('benchmark', parsed, 'x', 'URL')
|
||||
.count
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
// EPG worker's saxes-based parser; UTF-8 decoding is excluded.
|
||||
name: 'xmltv stream parse',
|
||||
input: xmltvInput,
|
||||
verify(variant) {
|
||||
const titles: string[] = [];
|
||||
parseXmltv(variant, (programs) => {
|
||||
for (const programme of programs) {
|
||||
titles.push(programme.title[0]?.value ?? '');
|
||||
}
|
||||
})();
|
||||
const programme = titlesFor(variant).programme;
|
||||
expectEqual(titles[0], `${programme} 000001-0001`);
|
||||
expectEqual(titles.at(-1), `${programme} 000500-0100`);
|
||||
if (
|
||||
titles.some((title) =>
|
||||
title.includes(REPLACEMENT_CHARACTER)
|
||||
)
|
||||
) {
|
||||
throw new Error(`${variant} XMLTV titles contain U+FFFD`);
|
||||
}
|
||||
},
|
||||
prepare: (variant) => parseXmltv(variant, () => undefined),
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
function titlesFor(variant: Variant) {
|
||||
return SYNTHETIC_TITLE_VOCABULARY[
|
||||
variant === 'cyrillic' ? 'cyrillic' : 'latin1'
|
||||
];
|
||||
}
|
||||
|
||||
function sliceText(text: string, size: number): string[] {
|
||||
const chunks: string[] = [];
|
||||
for (let offset = 0; offset < text.length; offset += size) {
|
||||
chunks.push(text.slice(offset, offset + size));
|
||||
}
|
||||
return chunks;
|
||||
}
|
||||
|
||||
function expectEqual(actual: string | undefined, expected: string): void {
|
||||
if (actual !== expected) {
|
||||
throw new Error(`Expected "${expected}", parsed "${String(actual)}"`);
|
||||
}
|
||||
}
|
||||
|
||||
function buffersPerVariant(
|
||||
create: (charset: 'latin1' | 'cyrillic') => { body: string }
|
||||
): Map<Variant, Buffer> {
|
||||
const latin1 = Buffer.from(create('latin1').body, 'utf8');
|
||||
return new Map<Variant, Buffer>([
|
||||
['latin1', latin1],
|
||||
['latin1-bom', Buffer.concat([UTF8_BYTE_ORDER_MARK, latin1])],
|
||||
['cyrillic', Buffer.from(create('cyrillic').body, 'utf8')],
|
||||
]);
|
||||
}
|
||||
|
||||
function requireBuffer(
|
||||
buffers: Map<Variant, Buffer>,
|
||||
variant: Variant
|
||||
): Buffer {
|
||||
const buffer = buffers.get(variant);
|
||||
if (!buffer) {
|
||||
throw new Error(`No fixture for ${variant}`);
|
||||
}
|
||||
return buffer;
|
||||
}
|
||||
|
||||
function assertCount(count: number): void {
|
||||
if (count !== ENTRY_COUNT) {
|
||||
throw new Error(`Expected ${ENTRY_COUNT} entries, parsed ${count}`);
|
||||
}
|
||||
}
|
||||
@@ -18,8 +18,6 @@
|
||||
* "$(node -p "require('electron')")" --expose-gc --import tsx \
|
||||
* src/performance/charset-parse.benchmark.ts
|
||||
*/
|
||||
import { createPlaylistObject } from '@iptvnator/shared/m3u-utils';
|
||||
import { parse } from 'iptv-playlist-parser';
|
||||
import { writeFile } from 'node:fs/promises';
|
||||
import { Session } from 'node:inspector/promises';
|
||||
import { resolve } from 'node:path';
|
||||
@@ -34,34 +32,16 @@ import {
|
||||
formatCharsetReport,
|
||||
summarizeSelfSamples,
|
||||
topFrames,
|
||||
variantOrderForRound,
|
||||
} from './charset-parse-benchmark-report';
|
||||
import { createSyntheticM3uFixture } from './synthetic-m3u';
|
||||
import { createSyntheticXmltvFixture } from './synthetic-xmltv';
|
||||
|
||||
const ENTRY_COUNT = 50_000;
|
||||
/** Matches the default highWaterMark of the worker's HTTP/decoder stream. */
|
||||
const XMLTV_CHUNK_BYTES = 64 * 1024;
|
||||
const UTF8_BYTE_ORDER_MARK = Buffer.from([0xef, 0xbb, 0xbf]);
|
||||
import {
|
||||
createWorkloads,
|
||||
ENTRY_COUNT,
|
||||
type Workload,
|
||||
} from './charset-parse-workloads';
|
||||
|
||||
type Variant = CharsetBenchmarkVariant;
|
||||
|
||||
interface StreamingEpgParserModule {
|
||||
StreamingEpgParser: new (
|
||||
onChannels: (channels: unknown[]) => void,
|
||||
onPrograms: (programs: unknown[]) => void,
|
||||
onProgress: (channels: number, programs: number) => void,
|
||||
channelBatchSize?: number,
|
||||
programBatchSize?: number
|
||||
) => { write(chunk: string): void; finish(): { totalPrograms: number } };
|
||||
}
|
||||
|
||||
interface Workload {
|
||||
readonly name: string;
|
||||
/** Prepares untimed input, then returns the timed operation. */
|
||||
prepare(variant: Variant): () => void;
|
||||
input(variant: Variant): string;
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
const { values } = parseArgs({
|
||||
options: {
|
||||
@@ -87,14 +67,17 @@ async function main(): Promise<void> {
|
||||
|
||||
const measurements: CharsetCaseMeasurement[] = [];
|
||||
for (const workload of workloads) {
|
||||
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
|
||||
workload.verify(variant);
|
||||
}
|
||||
const wall = new Map<Variant, number[]>();
|
||||
const cpu = new Map<Variant, number[]>();
|
||||
const profiles = new Map<Variant, CpuProfileLike[]>();
|
||||
// Variants alternate inside every round so JIT warm-up and heap growth
|
||||
// do not favour whichever variant runs last. Rounds after the timed
|
||||
// ones run under the profiler, which perturbs timing.
|
||||
// Variants alternate inside every round, starting with a different
|
||||
// one each round, so JIT warm-up and heap growth favour no input.
|
||||
// Rounds after the timed ones run under the profiler.
|
||||
for (let round = 0; round < warmup + iterations * 2; round += 1) {
|
||||
for (const variant of CHARSET_BENCHMARK_VARIANTS) {
|
||||
for (const variant of variantOrderForRound(round)) {
|
||||
const run = workload.prepare(variant);
|
||||
collectGarbage();
|
||||
const profiled = round >= warmup + iterations;
|
||||
@@ -151,79 +134,6 @@ async function main(): Promise<void> {
|
||||
}
|
||||
}
|
||||
|
||||
async function createWorkloads(): Promise<Workload[]> {
|
||||
const epgModule = (await import(
|
||||
resolve(
|
||||
__dirname,
|
||||
'../../../electron-backend/src/app/workers/epg-streaming-parser.ts'
|
||||
)
|
||||
)) as StreamingEpgParserModule;
|
||||
const m3uBuffers = buffersPerVariant((charset) =>
|
||||
createSyntheticM3uFixture(ENTRY_COUNT, { charset })
|
||||
);
|
||||
const xmltvBuffers = buffersPerVariant((charset) =>
|
||||
createSyntheticXmltvFixture(ENTRY_COUNT, { charset })
|
||||
);
|
||||
const m3uInput = (variant: Variant) =>
|
||||
requireBuffer(m3uBuffers, variant).toString('utf8');
|
||||
const xmltvInput = (variant: Variant) =>
|
||||
requireBuffer(xmltvBuffers, variant).toString('utf8');
|
||||
|
||||
return [
|
||||
{
|
||||
// PARSE_M3U phase of the main-process import.
|
||||
name: 'm3u parse',
|
||||
input: m3uInput,
|
||||
prepare(variant) {
|
||||
const body = m3uInput(variant);
|
||||
return () => assertCount(parse(body).items.length);
|
||||
},
|
||||
},
|
||||
{
|
||||
// NORMALIZE phase of the main-process import.
|
||||
name: 'm3u normalize',
|
||||
input: m3uInput,
|
||||
prepare(variant) {
|
||||
const parsed = parse(m3uInput(variant));
|
||||
return () =>
|
||||
assertCount(
|
||||
createPlaylistObject('benchmark', parsed, 'x', 'URL')
|
||||
.count
|
||||
);
|
||||
},
|
||||
},
|
||||
{
|
||||
// EPG worker: per-chunk UTF-8 decode plus the saxes-based parser.
|
||||
name: 'xmltv stream parse',
|
||||
input: xmltvInput,
|
||||
prepare(variant) {
|
||||
const buffer = requireBuffer(xmltvBuffers, variant);
|
||||
return () => {
|
||||
const parser = new epgModule.StreamingEpgParser(
|
||||
() => undefined,
|
||||
() => undefined,
|
||||
() => undefined,
|
||||
100,
|
||||
1000
|
||||
);
|
||||
for (
|
||||
let offset = 0;
|
||||
offset < buffer.length;
|
||||
offset += XMLTV_CHUNK_BYTES
|
||||
) {
|
||||
parser.write(
|
||||
buffer
|
||||
.subarray(offset, offset + XMLTV_CHUNK_BYTES)
|
||||
.toString('utf-8')
|
||||
);
|
||||
}
|
||||
assertCount(parser.finish().totalPrograms);
|
||||
};
|
||||
},
|
||||
},
|
||||
];
|
||||
}
|
||||
|
||||
interface CollectedSamples {
|
||||
readonly wall: Map<Variant, number[]>;
|
||||
readonly cpu: Map<Variant, number[]>;
|
||||
@@ -258,38 +168,10 @@ function measurementFor(
|
||||
};
|
||||
}
|
||||
|
||||
function buffersPerVariant(
|
||||
create: (charset: 'latin1' | 'cyrillic') => { body: string }
|
||||
): Map<Variant, Buffer> {
|
||||
const latin1 = Buffer.from(create('latin1').body, 'utf8');
|
||||
return new Map<Variant, Buffer>([
|
||||
['latin1', latin1],
|
||||
['latin1-bom', Buffer.concat([UTF8_BYTE_ORDER_MARK, latin1])],
|
||||
['cyrillic', Buffer.from(create('cyrillic').body, 'utf8')],
|
||||
]);
|
||||
}
|
||||
|
||||
function requireBuffer(
|
||||
buffers: Map<Variant, Buffer>,
|
||||
variant: Variant
|
||||
): Buffer {
|
||||
const buffer = buffers.get(variant);
|
||||
if (!buffer) {
|
||||
throw new Error(`No fixture for ${variant}`);
|
||||
}
|
||||
return buffer;
|
||||
}
|
||||
|
||||
function append<T>(map: Map<Variant, T[]>, variant: Variant, value: T): void {
|
||||
map.set(variant, [...(map.get(variant) ?? []), value]);
|
||||
}
|
||||
|
||||
function assertCount(count: number): void {
|
||||
if (count !== ENTRY_COUNT) {
|
||||
throw new Error(`Expected ${ENTRY_COUNT} entries, parsed ${count}`);
|
||||
}
|
||||
}
|
||||
|
||||
function collectGarbage(): void {
|
||||
(globalThis as { gc?: () => void }).gc?.();
|
||||
}
|
||||
|
||||
@@ -343,18 +343,25 @@ pnpm nx run electron-backend-e2e:benchmark-charset-parse --iterations=5
|
||||
|
||||
It parses 50,000 M3U channels (`iptv-playlist-parser`, then
|
||||
`createPlaylistObject`, the main-process `PARSE_M3U` and `NORMALIZE` phases)
|
||||
and 50,000 XMLTV programmes (`StreamingEpgParser` fed 64 KiB chunks, as in the
|
||||
EPG worker). Each workload runs on three inputs: `latin1` and `cyrillic`
|
||||
from the synthetic generators (`charset` option of `synthetic-m3u.ts` and
|
||||
and 50,000 XMLTV programmes (`StreamingEpgParser`, the EPG worker's parser).
|
||||
Each workload runs on three inputs: `latin1` and `cyrillic` from the
|
||||
synthetic generators (`charset` option of `synthetic-m3u.ts` and
|
||||
`synthetic-xmltv.ts`, identical layout apart from titles), and `latin1-bom`,
|
||||
the latin1 bytes behind a UTF-8 byte-order mark. The BOM forces two-byte
|
||||
storage without changing content, which separates the encoding cost from
|
||||
the effect that non-ASCII titles have on ASCII-only regexes. The report gives
|
||||
P50 wall-clock and CPU time after one warm-up, plus CPU-profile sample counts
|
||||
and top self frames from a separate profiled pass. Prefer CPU time and
|
||||
samples on a busy machine.
|
||||
the effect that non-ASCII titles have on ASCII-only regexes. The XMLTV
|
||||
parser receives 64 Ki-character slices of one decoded string rather than
|
||||
per-chunk decoded buffers: slices keep the input's representation (a
|
||||
per-chunk decode would make the BOM control one-byte after its first
|
||||
chunk), and no multi-byte character is split. Before timing, an untimed
|
||||
pass checks that the parsed titles match the fixture.
|
||||
|
||||
The 2026-09-26 measurement (plan item D1) found every workload under the 1.5x
|
||||
The report gives P50 wall-clock and CPU time after one warm-up, plus
|
||||
CPU-profile sample counts and top self frames from a separate profiled pass.
|
||||
Inputs alternate within each round and the starting input rotates between
|
||||
rounds. Prefer CPU time and samples on a busy machine.
|
||||
|
||||
The 2026-09-27 measurement (plan item D1) found every workload under the 1.5x
|
||||
threshold on Node 22 and inside Electron 43, so D2 regex prefilters were not
|
||||
applied. Rerun the benchmark after changing either parser or when a user
|
||||
reports slow imports of non-Latin playlists.
|
||||
|
||||
Reference in new issue
Block a user