diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts index 2cc08901e..6566ae95c 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts @@ -83,6 +83,11 @@ describe('provider tag stripping', () => { ); }); + it('keeps bare 4-5 char words before a spaced dash (real titles)', () => { + expect(normalizeTitle('DUNE - Part Two')).toBe('dune part two'); + expect(normalizeTitle('ALIEN - Covenant')).toBe('alien covenant'); + }); + it('strips underscore and double-dash suffix tags', () => { expect(normalizeTitle('Fallout_eng')).toBe('fallout'); expect(normalizeTitle('Breaking Bad (US)_msub')).toBe('breaking bad'); @@ -94,6 +99,12 @@ describe('provider tag stripping', () => { expect(normalizeTitle('The_Last_of_Us')).toBe('the last of us'); }); + it('keeps sole-underscore titles whose tail is not a known tag', () => { + expect(normalizeTitle('Mr_Robot')).toBe('mr robot'); + expect(normalizeTitle('Cowboy_Bebop')).toBe('cowboy bebop'); + expect(normalizeTitle('Mrs_Davis')).toBe('mrs davis'); + }); + it('strips joined dash tags only for case-uniform vocabulary tokens', () => { expect(normalizeTitle('Breaking Bad-eng')).toBe('breaking bad'); expect(normalizeTitle('The Last of Us-DE')).toBe('the last of us'); @@ -102,6 +113,11 @@ describe('provider tag stripping', () => { expect(normalizeTitle('Kick-It')).toBe('kick it'); }); + it('keeps English hyphenated word endings that collide with codes', () => { + expect(normalizeTitle('drive-in')).toBe('drive in'); + expect(normalizeTitle('Plug-in')).toBe('plug in'); + }); + it('strips bare trailing UPPERCASE vocabulary tags', () => { expect(normalizeTitle('The Pitt (2025) DE')).toBe('the pitt'); expect(normalizeTitle('Breaking Bad ES')).toBe('breaking bad'); diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.ts b/libs/shared/interfaces/src/lib/title-normalization.util.ts index 68b690c0c..b06c73694 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.ts @@ -37,15 +37,27 @@ const WRAPPED_TAG_PREFIX = /^\s*\|(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}\|\s*/; /** * Leading channel/language prefix like "EN - ", "DE| ", "FR: ", including * compound provider/quality forms ("4K-DE - ", "AR-SUBS - ", "4K-OSN+ - ") - * and longer tags ("EXYU| ", "MULTI| "). + * and longer pipe-tagged forms ("EXYU| ", "MULTI| "). * UPPERCASE-only on purpose: a case-insensitive match would amputate real * title words ("It: Chapter Two" → "Chapter Two"). Every segment must * contain a letter so numeric titles ("1917 - ...") are never tags. - * Colon separators stay limited to 2–3 chars — longer acronyms before a - * colon are franchise titles ("NCIS: LA"), not tags. + * + * Separator strength gates how wide a single segment may be: + * - dash ("EN - "): compound OR 2–3 chars — a bare 4–5 char word before + * a spaced dash is a real title ("DUNE - Part Two", "ALIEN - Covenant") + * - pipe ("EXYU| "): compound OR 2–5 chars — a pipe is a strong tag signal + * - colon ("EN: "): 2–3 chars — longer acronyms are franchise titles + * ("NCIS: LA") */ -const LANGUAGE_PREFIX = - /^(?:(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}(?:-(?=[0-9+]*[A-Z])[A-Z0-9+]{2,6}){0,2}\s*[-|]\s+|(?=[0-9+]*[A-Z])[A-Z0-9+]{2,3}\s*:\s+)/; +const SEG = '(?=[0-9+]*[A-Z])[A-Z0-9+]'; +const COMPOUND_TAG = `${SEG}{2,5}(?:-${SEG}{2,6}){1,2}`; +const LANGUAGE_PREFIX = new RegExp( + '^(?:' + + `(?:${COMPOUND_TAG}|${SEG}{2,3})\\s*-\\s+` + + `|(?:${COMPOUND_TAG}|${SEG}{2,5})\\s*\\|\\s+` + + `|${SEG}{2,3}\\s*:\\s+` + + ')' +); /** * Curated whitelist for TRAILING language/subtitle tags ("Fallout_eng", @@ -61,8 +73,16 @@ const TRAILING_TAG_VOCABULARY = new Set([ 'SC', 'SE', 'SUB', 'SUBS', 'SW', 'TA', 'TL', 'TR', 'TUR', ]); +/** + * Vocabulary tags that are also common English hyphenated-word endings + * ("drive-in", "plug-in"). Accepted only after a STRONG separator (bare + * spaced/uppercase form), never in the weak joined-dash/underscore forms + * where "X-in"/"X_in" reads as a title, not a tag. + */ +const WEAK_JOIN_EXCLUSIONS = new Set(['IN']); + const DOUBLE_DASH_SUFFIX = /[-–]{2}[A-Za-z]{2,5}\s*$/; -const UNDERSCORE_SUFFIX = /_[A-Za-z]{2,5}\s*$/; +const UNDERSCORE_SUFFIX = /_([A-Za-z]{2,5})\s*$/; const JOINED_DASH_SUFFIX = /-([A-Za-z]{2,5})\s*$/; const TRAILING_TAG_SUFFIX = /\s([A-Z]{2,5})\s*$/; @@ -71,6 +91,21 @@ function isCaseUniform(token: string): boolean { return token === token.toLowerCase() || token === token.toUpperCase(); } +/** + * A captured joined-dash/underscore token is provider metadata only when + * it is a known vocabulary tag, case-uniform, and not one of the English + * word-forming exclusions. This keeps "Mr_Robot", "Cowboy_Bebop", + * "drive-in", and "Plug-in" intact while stripping "_eng", "-DE", "-it". + */ +function isJoinedTag(token: string): boolean { + const upper = token.toUpperCase(); + return ( + TRAILING_TAG_VOCABULARY.has(upper) && + !WEAK_JOIN_EXCLUSIONS.has(upper) && + isCaseUniform(token) + ); +} + /** ES2015-safe trailing-whitespace trim (the lib target predates trimEnd). */ function trimRight(value: string): string { return value.replace(/\s+$/, ''); @@ -103,22 +138,22 @@ function stripTrailingTagOnce(value: string): string { return trimRight(result.replace(DOUBLE_DASH_SUFFIX, '')); } - // "Fallout_eng" — but not "The_Last_of_Us", where underscores are - // space substitutes (only strip when this is the sole underscore). + // "Fallout_eng" — but not "The_Last_of_Us" (underscores as spaces, only + // strip a sole underscore) and not "Mr_Robot"/"Cowboy_Bebop" (the tail + // must be a known tag, so the segment is real-title evidence otherwise). + const underscore = result.match(UNDERSCORE_SUFFIX); if ( - UNDERSCORE_SUFFIX.test(result) && - result.indexOf('_') === result.lastIndexOf('_') + underscore && + result.indexOf('_') === result.lastIndexOf('_') && + isJoinedTag(underscore[1]) ) { return trimRight(result.replace(UNDERSCORE_SUFFIX, '')); } + // "Breaking Bad-eng", "The Last of Us-DE" — but not "drive-in"/"Plug-in". + // hasLowercase gates ALL-CAPS titles out (no casing signal to trust). const joined = result.match(JOINED_DASH_SUFFIX); - if ( - joined && - TRAILING_TAG_VOCABULARY.has(joined[1].toUpperCase()) && - isCaseUniform(joined[1]) && - (joined[1] !== joined[1].toUpperCase() || hasLowercase) - ) { + if (joined && hasLowercase && isJoinedTag(joined[1])) { return trimRight(result.replace(JOINED_DASH_SUFFIX, '')); } diff --git a/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.spec.ts b/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.spec.ts index a91d4f9ee..3633d4e2e 100644 --- a/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.spec.ts +++ b/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.spec.ts @@ -53,9 +53,9 @@ describe('stripCountryPrefix', () => { ); }); - it('strips longer uppercase tags', () => { - expect(stripCountryPrefix('EXYU - News')).toBe('News'); - expect(stripCountryPrefix('MULTI - Movies')).toBe('Movies'); + it('strips longer pipe-tagged prefixes', () => { + expect(stripCountryPrefix('EXYU| News')).toBe('News'); + expect(stripCountryPrefix('MULTI| Movies')).toBe('Movies'); }); it('never treats numeric fragments as tags', () => { @@ -64,6 +64,15 @@ describe('stripCountryPrefix', () => { ); }); + it('keeps bare 4-5 char words before a spaced dash (real titles)', () => { + expect(stripCountryPrefix('DUNE - Part Two')).toBe( + 'DUNE - Part Two' + ); + expect(stripCountryPrefix('ALIEN - Covenant')).toBe( + 'ALIEN - Covenant' + ); + }); + it('only strips the first tag segment', () => { expect(stripCountryPrefix('ES - A3 - Sports')).toBe('A3 - Sports'); }); diff --git a/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.ts b/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.ts index 8d382f82a..d0b8b74dc 100644 --- a/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.ts +++ b/libs/shared/m3u-utils/src/lib/strip-country-prefix.util.ts @@ -4,16 +4,17 @@ * ("Sky - Sports F1", "Discovery - Science"). Pipes are conventionally * used only as tag separators, so they always count. * - * A tag is 1–3 short UPPERCASE alphanumeric segments ("US", "4K-DE", - * "AR-SUBS", "OSN+"). Every segment must contain a letter so numeric - * titles ("1917 - ...") are never treated as tags. Colon separators stay - * limited to plain 2–3 char tags — longer acronyms before a colon are - * franchise titles ("NCIS: LA"), not tags. + * Before a dash a tag is either a compound ("4K-DE", "AR-SUBS", "4K-OSN+" + * — the inner hyphen is the tag signal) or a plain 2–3 char code ("US", + * "4K"). A bare 4–5 char word before a spaced dash is a real title + * ("DUNE - Part Two", "ALIEN - Covenant"), so it is not a tag. Colon tags + * stay 2–3 chars — longer acronyms are franchise titles ("NCIS: LA"). + * Every segment must contain a letter so numbers ("1917 - ...") are safe. */ const DASH_SEPARATORS = [' - ', '- ', ' -']; const COLON_SEPARATOR = ': '; const TAG_PREFIX_PATTERN = - /^(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}(?:-(?=[0-9+]*[A-Z])[A-Z0-9+]{2,6}){0,2}$/; + /^(?:(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}(?:-(?=[0-9+]*[A-Z])[A-Z0-9+]{2,6}){1,2}|(?=[0-9+]*[A-Z])[A-Z0-9+]{2,3})$/; const COLON_TAG_PATTERN = /^(?=[0-9+]*[A-Z])[A-Z0-9+]{2,3}$/; interface SeparatorMatch {