Files
iptvnator/libs/shared/interfaces/src/lib/title-normalization.util.ts
T
4gray 78df3e7dbb fix(portals): match Greek titles whichever sigma the provider typed (#1310)
Greek Σ has two lowercase forms — medial σ and word-final ς — and neither the
candidate query nor the confirmation treated them as one letter.

The GLOB scan built each character's class from a one-way reach that only
arrived at ς when it started from ς, so a request for "ΑΣ" never admitted a
stored "Ας". Classes are now built from a fold group — every character sharing
an uppercase form — derived by scanning the cased ranges at module load the way
ACCENTED_BY_BASE already is. It generalises past sigma on its own: dotless ı
folds with i, long ſ with s, historic Cyrillic letterforms with В Д О С Т Ъ Ѣ.
Only the 24 groups of 767 that a per-character fold would miss are kept.

Admitting the row was only half of it. normalizeTitleKeys then compared "ασ"
against "ας" and discarded it, because toLowerCase picks the sigma form by
position. Both SQL tiers already folded them together — SQLite's trigram
tokenizer does full Unicode folding natively, unlike LOWER() — so the JS
confirmation was the only tier that did not, making this a pre-existing gap on
the FTS path as well. Normalization now rewrites ς to σ after lowercasing,
which is what Unicode case folding does.

Guards unchanged: a case mapping that changes length (ß → SS, İ) or a GLOB
metacharacter still returns null rather than a partial pattern.
2026-07-29 23:14:59 +02:00

338 lines
12 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Provider-title normalization shared by the renderer (TMDB matching,
* catalog indexes) and the Electron DB worker (cross-playlist title
* matching). Pure functions — no Angular/Node dependencies.
*/
import { SEASON_WORD_ALTERNATIVES } from './season-marker.util';
const QUALITY_TAGS = new Set([
'4k',
'uhd',
'fhd',
'hd',
'sd',
'hdr',
'hevc',
'h264',
'h265',
'x264',
'x265',
'480p',
'720p',
'1080p',
'2160p',
'multi',
'multisub',
'vostfr',
'vf',
'dubbed',
]);
/**
* Wrapped tag at the very start of a provider title: "|DE| ARD",
* "|MULTI| Fallout". The lookahead requires a letter in the tag so a
* numeric fragment can never be treated as one.
*/
const WRAPPED_TAG_PREFIX = /^\s*\|(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}\|\s*/;
/**
* Leading channel/language prefix like "EN - ", "DE| ", "FR: ", including
* compound provider/quality forms ("4K-DE - ", "AR-SUBS - ", "4K-OSN+ - ")
* and longer pipe-tagged forms ("EXYU| ", "MULTI| ").
* UPPERCASE-only on purpose: a case-insensitive match would amputate real
* title words ("It: Chapter Two" → "Chapter Two"). Every segment must
* contain a letter so numeric titles ("1917 - ...") are never tags.
*
* Separator strength gates how wide a single segment may be:
* - dash ("EN - "): compound OR 2–3 chars — a bare 4–5 char word before
* a spaced dash is a real title ("DUNE - Part Two", "ALIEN - Covenant")
* - pipe ("EXYU| "): compound OR 2–5 chars — a pipe is a strong tag signal
* - colon ("EN: "): 2–3 chars — longer acronyms are franchise titles
* ("NCIS: LA")
*/
const SEG = '(?=[0-9+]*[A-Z])[A-Z0-9+]';
const COMPOUND_TAG = `${SEG}{2,5}(?:-${SEG}{2,6}){1,2}`;
const LANGUAGE_PREFIX = new RegExp(
'^(?:' +
`(?:${COMPOUND_TAG}|${SEG}{2,3})\\s*-\\s+` +
`|(?:${COMPOUND_TAG}|${SEG}{2,5})\\s*\\|\\s+` +
`|${SEG}{2,3}\\s*:\\s+` +
')'
);
/**
* Curated whitelist for TRAILING language/subtitle tags ("Fallout_eng",
* "Breaking Bad-DE", "The Pitt (2025) ES"). Trailing stripping must be
* vocabulary-gated: a pattern-only rule would amputate real endings —
* roman numerals ("Rocky II"), acronyms ("Made in USA"), franchise
* suffixes ("NCIS: LA"). US/USA/UK/LA are deliberately absent.
*/
const TRAILING_TAG_VOCABULARY = new Set([
'AF', 'AL', 'ALB', 'AR', 'BY', 'DE', 'DUB', 'EN', 'ENG', 'ES', 'ESP',
'EXYU', 'FR', 'FRA', 'GE', 'GR', 'HU', 'IN', 'IR', 'IS', 'IT', 'ITA',
'KA', 'KU', 'LAT', 'ML', 'MSUB', 'MULTI', 'NL', 'PL', 'PT', 'RO', 'RU',
'SC', 'SE', 'SUB', 'SUBS', 'SW', 'TA', 'TL', 'TR', 'TUR',
]);
/**
* Vocabulary tags that are also common English hyphenated-word endings
* ("drive-in", "plug-in"). Accepted only after a STRONG separator (bare
* spaced/uppercase form), never in the weak joined-dash/underscore forms
* where "X-in"/"X_in" reads as a title, not a tag.
*/
const WEAK_JOIN_EXCLUSIONS = new Set(['IN']);
const DOUBLE_DASH_SUFFIX = /[-–]{2}[A-Za-z]{2,5}\s*$/;
const UNDERSCORE_SUFFIX = /_([A-Za-z]{2,5})\s*$/;
const JOINED_DASH_SUFFIX = /-([A-Za-z]{2,5})\s*$/;
const TRAILING_TAG_SUFFIX = /\s([A-Z]{2,5})\s*$/;
/** Tag tokens are case-uniform; real title words are Capitalized. */
function isCaseUniform(token: string): boolean {
return token === token.toLowerCase() || token === token.toUpperCase();
}
/**
* A captured joined-dash/underscore token is provider metadata only when
* it is a known vocabulary tag, case-uniform, and not one of the English
* word-forming exclusions. This keeps "Mr_Robot", "Cowboy_Bebop",
* "drive-in", and "Plug-in" intact while stripping "_eng", "-DE", "-it".
*/
function isJoinedTag(token: string): boolean {
const upper = token.toUpperCase();
return (
TRAILING_TAG_VOCABULARY.has(upper) &&
!WEAK_JOIN_EXCLUSIONS.has(upper) &&
isCaseUniform(token)
);
}
/** ES2015-safe trailing-whitespace trim (the lib target predates trimEnd). */
function trimRight(value: string): string {
return value.replace(/\s+$/, '');
}
/**
* Strip appended language/subtitle tags. Runs BEFORE lowercasing —
* casing is the main false-positive guard: ALL-CAPS titles carry no
* casing signal ("THE LAST OF US" must keep its "US"), so caps-gated
* rules are skipped for them, and Capitalized endings ("Making It",
* "Kick-It") never look like tags. Compound tags ("FR-EN") shed one
* token per pass, so stripping repeats to a fixpoint.
*/
function stripTrailingTags(value: string): string {
let result = value;
for (let pass = 0; pass < 3; pass++) {
const next = stripTrailingTagOnce(result);
if (next === result) break;
result = next;
}
return result;
}
function stripTrailingTagOnce(value: string): string {
const result = trimRight(value);
const hasLowercase = /\p{Ll}/u.test(result);
// "The Last of Us--esp": no real title contains a double dash.
if (DOUBLE_DASH_SUFFIX.test(result)) {
return trimRight(result.replace(DOUBLE_DASH_SUFFIX, ''));
}
// "Fallout_eng" — but not "The_Last_of_Us" (underscores as spaces, only
// strip a sole underscore) and not "Mr_Robot"/"Cowboy_Bebop" (the tail
// must be a known tag, so the segment is real-title evidence otherwise).
const underscore = result.match(UNDERSCORE_SUFFIX);
if (
underscore &&
result.indexOf('_') === result.lastIndexOf('_') &&
isJoinedTag(underscore[1])
) {
return trimRight(result.replace(UNDERSCORE_SUFFIX, ''));
}
// "Breaking Bad-eng", "The Last of Us-DE" — but not "drive-in"/"Plug-in".
// hasLowercase gates ALL-CAPS titles out (no casing signal to trust).
const joined = result.match(JOINED_DASH_SUFFIX);
if (joined && hasLowercase && isJoinedTag(joined[1])) {
return trimRight(result.replace(JOINED_DASH_SUFFIX, ''));
}
const trailing = result.match(TRAILING_TAG_SUFFIX);
if (trailing && hasLowercase && TRAILING_TAG_VOCABULARY.has(trailing[1])) {
return trimRight(result.replace(TRAILING_TAG_SUFFIX, ''));
}
return result;
}
const YEAR_PATTERN = /\b(19\d{2}|20\d{2})\b/;
/**
* Release-year tag at the very end of a title ("The Matrix 1999"). Only
* trailing years are stripped — an unanchored pattern would eat years that
* are part of the title ("2001: A Space Odyssey" → "a space odyssey").
*/
const TRAILING_YEAR_PATTERN = /(?:^|\s)(19\d{2}|20\d{2})$/;
/**
* Trailing season markers on series titles: "The Boys s05", "сезон 2",
* plus number-first forms ("2 season", "Пацаны 2 сезон", "2nd Season" —
* "2-й" normalizes to "2 й", hence the optional ordinal token). The
* ordinal list carries NFD-decomposed forms too: this pattern runs after
* diacritics stripping, which turns "й" into "и" ("2-й сезон" → "2 и
* сезон"). Uses (?:^|\s) instead of \b — JS word boundaries are
* ASCII-only and never fire next to Cyrillic letters.
*/
const SEASON_SUFFIX_PATTERN = new RegExp(
'(?:^|\\s)(?:' +
's\\d{1,2}' +
`|(?:${SEASON_WORD_ALTERNATIVES})\\s*\\d{1,2}` +
`|\\d{1,2}\\s*(?:st|nd|rd|th|й|и|я|ой|ои)?\\s+(?:${SEASON_WORD_ALTERNATIVES})` +
')$',
'iu'
);
/**
* A provider title normalized on two tiers. Trailing years on provider
* titles are ambiguous — usually a release tag ("The Matrix 1999") but
* sometimes part of the title itself ("Blade Runner 2049") — so matching
* must try the exact form first and only fall back to the year-stripped
* form when the stripped year does not contradict the other side's year.
*/
export interface NormalizedTitleKeys {
/** Fully normalized, trailing year KEPT ("blade runner 2049") */
exact: string;
/** Trailing year stripped ("blade runner"); equals `exact` if none */
base: string;
/** The trailing year removed in `base`, when there was one */
trailingYear: number | null;
}
export function normalizeTitleKeys(
raw: string | null | undefined
): NormalizedTitleKeys {
if (!raw) {
return { exact: '', base: '', trailingYear: null };
}
const cleaned = stripTrailingTags(
raw
.replace(WRAPPED_TAG_PREFIX, '')
// Inner classes exclude the opening delimiter too, so runaway
// inputs like "[[[[[..." backtrack linearly (CodeQL js/polynomial-redos)
.replace(/\[[^\][]*\]|\([^()]*\)|\{[^{}]*\}/g, ' ')
.replace(LANGUAGE_PREFIX, '')
)
.normalize('NFD')
.replace(/[\u0300-\u036F]/g, '')
.toLowerCase()
// Greek \u03A3 has two lowercase forms and `toLowerCase` picks by position:
// "\u0391\u03A3" becomes "\u03B1\u03C2" while an already-lowercase "\u03B1\u03C3" stays medial, so
// the same word reaches this line spelled two ways. Both SQL tiers
// fold them together \u2014 SQLite's trigram tokenizer does it natively,
// and the scan's GLOB classes do it in `caseInsensitiveGlobPattern` \u2014
// so without this the candidate is admitted by the query and then
// thrown away by the confirmation. Folding to the medial form is what
// Unicode case folding does.
.replace(/\u03C2/g, '\u03C3')
.replace(/[^\p{L}\p{N}]+/gu, ' ')
.split(' ')
.filter((token) => token !== '' && !QUALITY_TAGS.has(token))
.join(' ')
.trim();
// Portal series list titles carry season suffixes ("The Boys s05");
// TMDB knows only the show title.
const stripSeason = (value: string) =>
value.replace(SEASON_SUFFIX_PATTERN, '').trim() || value;
const exact = stripSeason(cleaned);
// Trailing years are release tags ("The Matrix 1999"), but a year can
// also BE the title ("2012") — never normalize down to an empty string.
const yearMatch = cleaned.match(TRAILING_YEAR_PATTERN);
const withoutYear = cleaned
.replace(TRAILING_YEAR_PATTERN, '')
.replace(/\s+/g, ' ')
.trim();
if (!yearMatch || !withoutYear) {
return { exact, base: exact, trailingYear: null };
}
return {
exact,
base: stripSeason(withoutYear),
trailingYear: Number(yearMatch[1]),
};
}
export function normalizeTitle(raw: string | null | undefined): string {
return normalizeTitleKeys(raw).base;
}
/**
* True when a base-tier (year-stripped) title match is not contradicted by
* the known years of both sides. Unknown years never block a match.
*/
export function titleYearsCompatible(
a: number | null | undefined,
b: number | null | undefined
): boolean {
return (
a === null ||
a === undefined ||
b === null ||
b === undefined ||
Math.abs(a - b) <= 1
);
}
/** Extract a release year from a date string or from tags in a raw title */
export function extractYear(
releaseDate: string | null | undefined,
rawTitle?: string | null
): number | null {
const fromDate = releaseDate?.match(YEAR_PATTERN)?.[0];
if (fromDate) {
return Number(fromDate);
}
const fromTitle = rawTitle?.match(YEAR_PATTERN)?.[0];
return fromTitle ? Number(fromTitle) : null;
}
/** Bracketed release tag: "Dune (2021)", "Dune [2021]". */
const BRACKETED_YEAR_PATTERN = /[([{]\s*(19\d{2}|20\d{2})\s*[)\]}]/;
/**
* The year a provider title states as a release TAG, never one that belongs to
* the film's name.
*
* `extractYear` reads any year anywhere in the title, which is the right answer
* where a year is only a search hint that scoring will confirm. It is the wrong
* answer wherever the year becomes part of an IDENTITY: "2001: A Space Odyssey"
* is not a 2001 film, and calling it one makes every genuine 1968 copy fail the
* year gate — the movie then has no alternative sources at all — while its pin
* key moves the moment enrichment supplies the real year.
*
* Only bracketed and trailing forms count. The trailing form stays ambiguous on
* purpose ("Blade Runner 2049" is a title, not a tag); that is what the
* `exact`/`base` two-tier match is for, and nothing here can settle it.
*/
export function releaseTagYear(
rawTitle: string | null | undefined
): number | null {
if (!rawTitle) {
return null;
}
// Before `normalizeTitleKeys`, which strips bracketed segments wholesale
// and would take the tag with them.
const bracketed = rawTitle.match(BRACKETED_YEAR_PATTERN);
return bracketed
? Number(bracketed[1])
: normalizeTitleKeys(rawTitle).trailingYear;
}