From 79c0557d1f3187f4e8a1113888aa0e678684dd20 Mon Sep 17 00:00:00 2001 From: 4gray Date: Sun, 23 Aug 2026 09:05:08 +0200 Subject: [PATCH] fix(playback): detect CP1251 subtitles by letter bytes, not whole-file ratio MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex review on #1471: short-dialogue CP1251 SRT files are dominated by ASCII cue counters and timing lines, so a whole-file high-byte ratio stayed under the threshold and the Cyrillic text decoded as Windows-1252 mojibake. The discriminator now compares high bytes against ASCII letters only — timing scaffolding is identical in every candidate encoding and no longer dilutes the signal. Co-Authored-By: Claude Fable 5 --- .../external-subtitle-cues.util.spec.ts | 19 +++++++++++++++++++ .../external-subtitle-cues.util.ts | 13 ++++++++++++- 2 files changed, 31 insertions(+), 1 deletion(-) diff --git a/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.spec.ts b/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.spec.ts index abc45ce55..dc0477b91 100644 --- a/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.spec.ts +++ b/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.spec.ts @@ -40,6 +40,25 @@ describe('external-subtitle-cues.util', () => { ); }); + it('detects CP1251 in a short-dialogue SRT where ASCII timing bytes dominate', () => { + // "Да" / "Нет" in CP1251 among full SRT timing scaffolding: only + // 5 of ~70 bytes are high, but ALL letter bytes are — the + // discriminator must ignore the timing lines. + const ascii = (text: string) => + Array.from(text).map((c) => c.charCodeAt(0)); + const srt = [ + ...ascii('1\n00:00:01,000 --> 00:00:02,000\n'), + 0xc4, 0xe0, // Да + ...ascii('\n\n2\n00:00:03,000 --> 00:00:04,000\n'), + 0xcd, 0xe5, 0xf2, // Нет + ...ascii('\n'), + ]; + expect(decodeExternalSubtitleBytes(toBuffer(srt))).toContain('Да'); + expect(decodeExternalSubtitleBytes(toBuffer(srt))).toContain( + 'Нет' + ); + }); + it('decodes mostly-ASCII Windows-1252 bytes (sparse accents)', () => { // "resume: cafe" with two accented letters among ASCII. const cp1252 = [ diff --git a/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.ts b/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.ts index 3837b2019..1e822056c 100644 --- a/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.ts +++ b/libs/ui/playback/src/lib/web-video-support/external-subtitle-cues.util.ts @@ -64,14 +64,25 @@ export function decodeExternalSubtitleBytes(buffer: ArrayBuffer): string { // Not valid UTF-8: a legacy single-byte encoding. } + // Compare only letter-carrying bytes: cue counters and timing lines are + // ASCII digits/punctuation and identical in every candidate encoding, so + // ratios over the whole file would drown short dialogue ("Да"/"Нет") in + // timing bytes. In a non-Latin-script file virtually every letter is a + // high byte; Latin text with accents stays dominated by ASCII letters. let highBytes = 0; + let asciiLetters = 0; for (const byte of bytes) { if (byte >= 0x80) { highBytes += 1; + } else if ( + (byte >= 0x41 && byte <= 0x5a) || + (byte >= 0x61 && byte <= 0x7a) + ) { + asciiLetters += 1; } } const encoding = - highBytes / Math.max(1, bytes.length) > 0.15 + highBytes / Math.max(1, highBytes + asciiLetters) > 0.3 ? 'windows-1251' : 'windows-1252'; try {