fix(playback): detect CP1251 subtitles by letter bytes, not whole-file ratio

Codex review on #1471: short-dialogue CP1251 SRT files are dominated by
ASCII cue counters and timing lines, so a whole-file high-byte ratio stayed
under the threshold and the Cyrillic text decoded as Windows-1252 mojibake.
The discriminator now compares high bytes against ASCII letters only —
timing scaffolding is identical in every candidate encoding and no longer
dilutes the signal.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
4grayandClaude Fable 5 committed 2026-08-23 09:05:08 +02:00
1 parent 174e4cf887
commit 79c0557d1f
2 files changed
+31 -1

No files matched your search

@@ -40,6 +40,25 @@ describe('external-subtitle-cues.util', () => {
);
});
it('detects CP1251 in a short-dialogue SRT where ASCII timing bytes dominate', () => {
// "Да" / "Нет" in CP1251 among full SRT timing scaffolding: only
// 5 of ~70 bytes are high, but ALL letter bytes are — the
// discriminator must ignore the timing lines.
const ascii = (text: string) =>
Array.from(text).map((c) => c.charCodeAt(0));
const srt = [
...ascii('1\n00:00:01,000 --> 00:00:02,000\n'),
0xc4, 0xe0, // Да
...ascii('\n\n2\n00:00:03,000 --> 00:00:04,000\n'),
0xcd, 0xe5, 0xf2, // Нет
...ascii('\n'),
];
expect(decodeExternalSubtitleBytes(toBuffer(srt))).toContain('Да');
expect(decodeExternalSubtitleBytes(toBuffer(srt))).toContain(
'Нет'
);
});
it('decodes mostly-ASCII Windows-1252 bytes (sparse accents)', () => {
// "resume: cafe" with two accented letters among ASCII.
const cp1252 = [
@@ -64,14 +64,25 @@ export function decodeExternalSubtitleBytes(buffer: ArrayBuffer): string {
// Not valid UTF-8: a legacy single-byte encoding.
}
// Compare only letter-carrying bytes: cue counters and timing lines are
// ASCII digits/punctuation and identical in every candidate encoding, so
// ratios over the whole file would drown short dialogue ("Да"/"Нет") in
// timing bytes. In a non-Latin-script file virtually every letter is a
// high byte; Latin text with accents stays dominated by ASCII letters.
let highBytes = 0;
let asciiLetters = 0;
for (const byte of bytes) {
if (byte >= 0x80) {
highBytes += 1;
} else if (
(byte >= 0x41 && byte <= 0x5a) ||
(byte >= 0x61 && byte <= 0x7a)
) {
asciiLetters += 1;
}
}
const encoding =
highBytes / Math.max(1, bytes.length) > 0.15
highBytes / Math.max(1, highBytes + asciiLetters) > 0.3
? 'windows-1251'
: 'windows-1252';
try {