/** * Provider-title normalization shared by the renderer (TMDB matching, * catalog indexes) and the Electron DB worker (cross-playlist title * matching). Pure functions — no Angular/Node dependencies. */ import { SEASON_WORD_ALTERNATIVES } from './season-marker.util'; const QUALITY_TAGS = new Set([ '4k', 'uhd', 'fhd', 'hd', 'sd', 'hdr', 'hevc', 'h264', 'h265', 'x264', 'x265', '480p', '720p', '1080p', '2160p', 'multi', 'multisub', 'vostfr', 'vf', 'dubbed', ]); /** * Wrapped tag at the very start of a provider title: "|DE| ARD", * "|MULTI| Fallout". The lookahead requires a letter in the tag so a * numeric fragment can never be treated as one. */ const WRAPPED_TAG_PREFIX = /^\s*\|(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}\|\s*/; /** * Leading channel/language prefix like "EN - ", "DE| ", "FR: ", including * compound provider/quality forms ("4K-DE - ", "AR-SUBS - ", "4K-OSN+ - ") * and longer pipe-tagged forms ("EXYU| ", "MULTI| "). * UPPERCASE-only on purpose: a case-insensitive match would amputate real * title words ("It: Chapter Two" → "Chapter Two"). Every segment must * contain a letter so numeric titles ("1917 - ...") are never tags. * * Separator strength gates how wide a single segment may be: * - dash ("EN - "): compound OR 2–3 chars — a bare 4–5 char word before * a spaced dash is a real title ("DUNE - Part Two", "ALIEN - Covenant") * - pipe ("EXYU| "): compound OR 2–5 chars — a pipe is a strong tag signal * - colon ("EN: "): 2–3 chars — longer acronyms are franchise titles * ("NCIS: LA") */ const SEG = '(?=[0-9+]*[A-Z])[A-Z0-9+]'; const COMPOUND_TAG = `${SEG}{2,5}(?:-${SEG}{2,6}){1,2}`; const LANGUAGE_PREFIX = new RegExp( '^(?:' + `(?:${COMPOUND_TAG}|${SEG}{2,3})\\s*-\\s+` + `|(?:${COMPOUND_TAG}|${SEG}{2,5})\\s*\\|\\s+` + `|${SEG}{2,3}\\s*:\\s+` + ')' ); /** * Curated whitelist for TRAILING language/subtitle tags ("Fallout_eng", * "Breaking Bad-DE", "The Pitt (2025) ES"). Trailing stripping must be * vocabulary-gated: a pattern-only rule would amputate real endings — * roman numerals ("Rocky II"), acronyms ("Made in USA"), franchise * suffixes ("NCIS: LA"). US/USA/UK/LA are deliberately absent. */ const TRAILING_TAG_VOCABULARY = new Set([ 'AF', 'AL', 'ALB', 'AR', 'BY', 'DE', 'DUB', 'EN', 'ENG', 'ES', 'ESP', 'EXYU', 'FR', 'FRA', 'GE', 'GR', 'HU', 'IN', 'IR', 'IS', 'IT', 'ITA', 'KA', 'KU', 'LAT', 'ML', 'MSUB', 'MULTI', 'NL', 'PL', 'PT', 'RO', 'RU', 'SC', 'SE', 'SUB', 'SUBS', 'SW', 'TA', 'TL', 'TR', 'TUR', ]); /** * Vocabulary tags that are also common English hyphenated-word endings * ("drive-in", "plug-in"). Accepted only after a STRONG separator (bare * spaced/uppercase form), never in the weak joined-dash/underscore forms * where "X-in"/"X_in" reads as a title, not a tag. */ const WEAK_JOIN_EXCLUSIONS = new Set(['IN']); const DOUBLE_DASH_SUFFIX = /[-–]{2}[A-Za-z]{2,5}\s*$/; const UNDERSCORE_SUFFIX = /_([A-Za-z]{2,5})\s*$/; const JOINED_DASH_SUFFIX = /-([A-Za-z]{2,5})\s*$/; const TRAILING_TAG_SUFFIX = /\s([A-Z]{2,5})\s*$/; /** Tag tokens are case-uniform; real title words are Capitalized. */ function isCaseUniform(token: string): boolean { return token === token.toLowerCase() || token === token.toUpperCase(); } /** * A captured joined-dash/underscore token is provider metadata only when * it is a known vocabulary tag, case-uniform, and not one of the English * word-forming exclusions. This keeps "Mr_Robot", "Cowboy_Bebop", * "drive-in", and "Plug-in" intact while stripping "_eng", "-DE", "-it". */ function isJoinedTag(token: string): boolean { const upper = token.toUpperCase(); return ( TRAILING_TAG_VOCABULARY.has(upper) && !WEAK_JOIN_EXCLUSIONS.has(upper) && isCaseUniform(token) ); } /** ES2015-safe trailing-whitespace trim (the lib target predates trimEnd). */ function trimRight(value: string): string { return value.replace(/\s+$/, ''); } /** * Strip appended language/subtitle tags. Runs BEFORE lowercasing — * casing is the main false-positive guard: ALL-CAPS titles carry no * casing signal ("THE LAST OF US" must keep its "US"), so caps-gated * rules are skipped for them, and Capitalized endings ("Making It", * "Kick-It") never look like tags. Compound tags ("FR-EN") shed one * token per pass, so stripping repeats to a fixpoint. */ function stripTrailingTags(value: string): string { let result = value; for (let pass = 0; pass < 3; pass++) { const next = stripTrailingTagOnce(result); if (next === result) break; result = next; } return result; } function stripTrailingTagOnce(value: string): string { const result = trimRight(value); const hasLowercase = /\p{Ll}/u.test(result); // "The Last of Us--esp": no real title contains a double dash. if (DOUBLE_DASH_SUFFIX.test(result)) { return trimRight(result.replace(DOUBLE_DASH_SUFFIX, '')); } // "Fallout_eng" — but not "The_Last_of_Us" (underscores as spaces, only // strip a sole underscore) and not "Mr_Robot"/"Cowboy_Bebop" (the tail // must be a known tag, so the segment is real-title evidence otherwise). const underscore = result.match(UNDERSCORE_SUFFIX); if ( underscore && result.indexOf('_') === result.lastIndexOf('_') && isJoinedTag(underscore[1]) ) { return trimRight(result.replace(UNDERSCORE_SUFFIX, '')); } // "Breaking Bad-eng", "The Last of Us-DE" — but not "drive-in"/"Plug-in". // hasLowercase gates ALL-CAPS titles out (no casing signal to trust). const joined = result.match(JOINED_DASH_SUFFIX); if (joined && hasLowercase && isJoinedTag(joined[1])) { return trimRight(result.replace(JOINED_DASH_SUFFIX, '')); } const trailing = result.match(TRAILING_TAG_SUFFIX); if (trailing && hasLowercase && TRAILING_TAG_VOCABULARY.has(trailing[1])) { return trimRight(result.replace(TRAILING_TAG_SUFFIX, '')); } return result; } const YEAR_PATTERN = /\b(19\d{2}|20\d{2})\b/; /** * Release-year tag at the very end of a title ("The Matrix 1999"). Only * trailing years are stripped — an unanchored pattern would eat years that * are part of the title ("2001: A Space Odyssey" → "a space odyssey"). */ const TRAILING_YEAR_PATTERN = /(?:^|\s)(19\d{2}|20\d{2})$/; /** * Trailing season markers on series titles: "The Boys s05", "сезон 2", * plus number-first forms ("2 season", "Пацаны 2 сезон", "2nd Season" — * "2-й" normalizes to "2 й", hence the optional ordinal token). The * ordinal list carries NFD-decomposed forms too: this pattern runs after * diacritics stripping, which turns "й" into "и" ("2-й сезон" → "2 и * сезон"). Uses (?:^|\s) instead of \b — JS word boundaries are * ASCII-only and never fire next to Cyrillic letters. */ const SEASON_SUFFIX_PATTERN = new RegExp( '(?:^|\\s)(?:' + 's\\d{1,2}' + `|(?:${SEASON_WORD_ALTERNATIVES})\\s*\\d{1,2}` + `|\\d{1,2}\\s*(?:st|nd|rd|th|й|и|я|ой|ои)?\\s+(?:${SEASON_WORD_ALTERNATIVES})` + ')$', 'iu' ); /** * A provider title normalized on two tiers. Trailing years on provider * titles are ambiguous — usually a release tag ("The Matrix 1999") but * sometimes part of the title itself ("Blade Runner 2049") — so matching * must try the exact form first and only fall back to the year-stripped * form when the stripped year does not contradict the other side's year. */ export interface NormalizedTitleKeys { /** Fully normalized, trailing year KEPT ("blade runner 2049") */ exact: string; /** Trailing year stripped ("blade runner"); equals `exact` if none */ base: string; /** The trailing year removed in `base`, when there was one */ trailingYear: number | null; } export function normalizeTitleKeys( raw: string | null | undefined ): NormalizedTitleKeys { if (!raw) { return { exact: '', base: '', trailingYear: null }; } const cleaned = stripTrailingTags( raw .replace(WRAPPED_TAG_PREFIX, '') // Inner classes exclude the opening delimiter too, so runaway // inputs like "[[[[[..." backtrack linearly (CodeQL js/polynomial-redos) .replace(/\[[^\][]*\]|\([^()]*\)|\{[^{}]*\}/g, ' ') .replace(LANGUAGE_PREFIX, '') ) .normalize('NFD') .replace(/[\u0300-\u036F]/g, '') .toLowerCase() // Greek \u03A3 has two lowercase forms and `toLowerCase` picks by position: // "\u0391\u03A3" becomes "\u03B1\u03C2" while an already-lowercase "\u03B1\u03C3" stays medial, so // the same word reaches this line spelled two ways. Both SQL tiers // fold them together \u2014 SQLite's trigram tokenizer does it natively, // and the scan's GLOB classes do it in `caseInsensitiveGlobPattern` \u2014 // so without this the candidate is admitted by the query and then // thrown away by the confirmation. Folding to the medial form is what // Unicode case folding does. .replace(/\u03C2/g, '\u03C3') .replace(/[^\p{L}\p{N}]+/gu, ' ') .split(' ') .filter((token) => token !== '' && !QUALITY_TAGS.has(token)) .join(' ') .trim(); // Portal series list titles carry season suffixes ("The Boys s05"); // TMDB knows only the show title. const stripSeason = (value: string) => value.replace(SEASON_SUFFIX_PATTERN, '').trim() || value; const exact = stripSeason(cleaned); // Trailing years are release tags ("The Matrix 1999"), but a year can // also BE the title ("2012") — never normalize down to an empty string. const yearMatch = cleaned.match(TRAILING_YEAR_PATTERN); const withoutYear = cleaned .replace(TRAILING_YEAR_PATTERN, '') .replace(/\s+/g, ' ') .trim(); if (!yearMatch || !withoutYear) { return { exact, base: exact, trailingYear: null }; } return { exact, base: stripSeason(withoutYear), trailingYear: Number(yearMatch[1]), }; } export function normalizeTitle(raw: string | null | undefined): string { return normalizeTitleKeys(raw).base; } /** * True when a base-tier (year-stripped) title match is not contradicted by * the known years of both sides. Unknown years never block a match. */ export function titleYearsCompatible( a: number | null | undefined, b: number | null | undefined ): boolean { return ( a === null || a === undefined || b === null || b === undefined || Math.abs(a - b) <= 1 ); } /** Extract a release year from a date string or from tags in a raw title */ export function extractYear( releaseDate: string | null | undefined, rawTitle?: string | null ): number | null { const fromDate = releaseDate?.match(YEAR_PATTERN)?.[0]; if (fromDate) { return Number(fromDate); } const fromTitle = rawTitle?.match(YEAR_PATTERN)?.[0]; return fromTitle ? Number(fromTitle) : null; } /** Bracketed release tag: "Dune (2021)", "Dune [2021]". */ const BRACKETED_YEAR_PATTERN = /[([{]\s*(19\d{2}|20\d{2})\s*[)\]}]/; /** * The year a provider title states as a release TAG, never one that belongs to * the film's name. * * `extractYear` reads any year anywhere in the title, which is the right answer * where a year is only a search hint that scoring will confirm. It is the wrong * answer wherever the year becomes part of an IDENTITY: "2001: A Space Odyssey" * is not a 2001 film, and calling it one makes every genuine 1968 copy fail the * year gate — the movie then has no alternative sources at all — while its pin * key moves the moment enrichment supplies the real year. * * Only bracketed and trailing forms count. The trailing form stays ambiguous on * purpose ("Blade Runner 2049" is a title, not a tag); that is what the * `exact`/`base` two-tier match is for, and nothing here can settle it. */ export function releaseTagYear( rawTitle: string | null | undefined ): number | null { if (!rawTitle) { return null; } // Before `normalizeTitleKeys`, which strips bracketed segments wholesale // and would take the tag with them. const bracketed = rawTitle.match(BRACKETED_YEAR_PATTERN); return bracketed ? Number(bracketed[1]) : normalizeTitleKeys(rawTitle).trailingYear; }