mirror of
https://github.com/4gray/iptvnator.git
synced 2026-10-11 11:06:16 -08:00
fix(playback): detect CP1251 subtitles by letter bytes, not whole-file ratio
Codex review on #1471: short-dialogue CP1251 SRT files are dominated by ASCII cue counters and timing lines, so a whole-file high-byte ratio stayed under the threshold and the Cyrillic text decoded as Windows-1252 mojibake. The discriminator now compares high bytes against ASCII letters only — timing scaffolding is identical in every candidate encoding and no longer dilutes the signal. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
174e4cf887
commit
79c0557d1f
2 files changed
+31
-1
No files matched your search
@@ -40,6 +40,25 @@ describe('external-subtitle-cues.util', () => {
|
||||
);
|
||||
});
|
||||
|
||||
it('detects CP1251 in a short-dialogue SRT where ASCII timing bytes dominate', () => {
|
||||
// "Да" / "Нет" in CP1251 among full SRT timing scaffolding: only
|
||||
// 5 of ~70 bytes are high, but ALL letter bytes are — the
|
||||
// discriminator must ignore the timing lines.
|
||||
const ascii = (text: string) =>
|
||||
Array.from(text).map((c) => c.charCodeAt(0));
|
||||
const srt = [
|
||||
...ascii('1\n00:00:01,000 --> 00:00:02,000\n'),
|
||||
0xc4, 0xe0, // Да
|
||||
...ascii('\n\n2\n00:00:03,000 --> 00:00:04,000\n'),
|
||||
0xcd, 0xe5, 0xf2, // Нет
|
||||
...ascii('\n'),
|
||||
];
|
||||
expect(decodeExternalSubtitleBytes(toBuffer(srt))).toContain('Да');
|
||||
expect(decodeExternalSubtitleBytes(toBuffer(srt))).toContain(
|
||||
'Нет'
|
||||
);
|
||||
});
|
||||
|
||||
it('decodes mostly-ASCII Windows-1252 bytes (sparse accents)', () => {
|
||||
// "resume: cafe" with two accented letters among ASCII.
|
||||
const cp1252 = [
|
||||
|
||||
@@ -64,14 +64,25 @@ export function decodeExternalSubtitleBytes(buffer: ArrayBuffer): string {
|
||||
// Not valid UTF-8: a legacy single-byte encoding.
|
||||
}
|
||||
|
||||
// Compare only letter-carrying bytes: cue counters and timing lines are
|
||||
// ASCII digits/punctuation and identical in every candidate encoding, so
|
||||
// ratios over the whole file would drown short dialogue ("Да"/"Нет") in
|
||||
// timing bytes. In a non-Latin-script file virtually every letter is a
|
||||
// high byte; Latin text with accents stays dominated by ASCII letters.
|
||||
let highBytes = 0;
|
||||
let asciiLetters = 0;
|
||||
for (const byte of bytes) {
|
||||
if (byte >= 0x80) {
|
||||
highBytes += 1;
|
||||
} else if (
|
||||
(byte >= 0x41 && byte <= 0x5a) ||
|
||||
(byte >= 0x61 && byte <= 0x7a)
|
||||
) {
|
||||
asciiLetters += 1;
|
||||
}
|
||||
}
|
||||
const encoding =
|
||||
highBytes / Math.max(1, bytes.length) > 0.15
|
||||
highBytes / Math.max(1, highBytes + asciiLetters) > 0.3
|
||||
? 'windows-1251'
|
||||
: 'windows-1252';
|
||||
try {
|
||||
|
||||
Reference in new issue
Block a user