mirror of
https://github.com/4gray/iptvnator.git
synced 2026-10-09 01:16:15 -08:00
fix(playback): require a meaningful Cyrillic letter share before choosing CP1251
Codex review round 6 on #1471: an isolated accented CP1252 word decodes under CP1251 to a tiny pure-Cyrillic word ("À table" -> "А table") whose single vote flipped the file to Cyrillic. The chooser now also requires Cyrillic letters to carry a meaningful share of all letters — genuinely Cyrillic dialogue dominates its own letter count even with embedded Latin names, while isolated accents never do. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
6ad9278230
commit
333d62fc1f
3 files changed
+40
-7
No files matched your search
@@ -695,9 +695,11 @@ Per-engine implementations:
|
||||
BOMs, strict UTF-8, then `chooseLegacySingleByteDecode`, which picks
|
||||
between Windows-1251 and Windows-1252 by the plausibility of the 1251
|
||||
candidate's decoded words — pure-Cyrillic words vote for 1251, words
|
||||
mixing Cyrillic with ASCII letters vote against, since misread Latin text
|
||||
like "était" decodes to the mixed-script "йtait" that real subtitles never
|
||||
contain), because `Blob.text()`'s silent UTF-8 substitution turns common
|
||||
mixing Cyrillic with ASCII letters vote against (misread Latin text like
|
||||
"était" decodes to the mixed-script "йtait" that real subtitles never
|
||||
contain), and Cyrillic must also carry a meaningful share of all letters
|
||||
so an isolated accented CP1252 word ("À table" → "А table") cannot flip
|
||||
the file), because `Blob.text()`'s silent UTF-8 substitution turns common
|
||||
legacy-encoded SRT files into mojibake. `WebVideoExternalSubtitles` parses
|
||||
the file (`external-subtitle-cues.util.ts`) and renders it through a native
|
||||
`TextTrack` on the video element, so it works under every source kind. The
|
||||
|
||||
@@ -76,6 +76,21 @@ describe('external-subtitle-cues.util', () => {
|
||||
expect(decoded).toContain('À table');
|
||||
});
|
||||
|
||||
it('keeps an isolated accented CP1252 word Latin ("À table")', () => {
|
||||
// À=0xC0 decodes under CP1251 to the pure-Cyrillic one-letter
|
||||
// word "А"; without the letter-share guard that single vote
|
||||
// flips the whole file to Cyrillic.
|
||||
const cp1252 = [
|
||||
0xc0, // À
|
||||
...Array.from(
|
||||
' table !\n1\n00:00:01,000 --> 00:00:02,000\n'
|
||||
).map((c) => c.charCodeAt(0)),
|
||||
];
|
||||
expect(decodeExternalSubtitleBytes(toBuffer(cp1252))).toContain(
|
||||
'À table'
|
||||
);
|
||||
});
|
||||
|
||||
it('decodes mostly-ASCII Windows-1252 bytes (sparse accents)', () => {
|
||||
// "resume: cafe" with two accented letters among ASCII.
|
||||
const cp1252 = [
|
||||
|
||||
@@ -82,14 +82,30 @@ export function decodeExternalSubtitleBytes(buffer: ArrayBuffer): string {
|
||||
*/
|
||||
function chooseLegacySingleByteDecode(buffer: ArrayBuffer): string {
|
||||
const decoded1251 = new TextDecoder('windows-1251').decode(buffer);
|
||||
let score = 0;
|
||||
let cyrillicLetters = 0;
|
||||
let mixedLetters = 0;
|
||||
let asciiLetters = 0;
|
||||
for (const word of decoded1251.split(/[^\p{L}]+/u)) {
|
||||
if (!/[\u0400-\u04ff]/.test(word)) {
|
||||
if (!word) {
|
||||
continue;
|
||||
}
|
||||
score += /[A-Za-z]/.test(word) ? -word.length : word.length;
|
||||
if (!/[\u0400-\u04ff]/.test(word)) {
|
||||
asciiLetters += word.length;
|
||||
} else if (/[A-Za-z]/.test(word)) {
|
||||
mixedLetters += word.length;
|
||||
} else {
|
||||
cyrillicLetters += word.length;
|
||||
}
|
||||
}
|
||||
return score > 0
|
||||
// Two guards: mixed-script words are strong evidence of misread Latin
|
||||
// text, and isolated accented CP1252 words ("\u00c0 table" \u2192 "\u0410 table")
|
||||
// masquerade as tiny pure-Cyrillic words \u2014 so Cyrillic must also carry a
|
||||
// meaningful share of all letters before 1251 wins. Genuinely Cyrillic
|
||||
// dialogue dominates its own letter count even with embedded Latin names.
|
||||
const plausiblyCyrillic =
|
||||
cyrillicLetters > mixedLetters * 2 &&
|
||||
cyrillicLetters * 4 > asciiLetters;
|
||||
return plausiblyCyrillic
|
||||
? decoded1251
|
||||
: new TextDecoder('windows-1252').decode(buffer);
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user