diff --git a/docs/architecture/tmdb-metadata-enrichment.md b/docs/architecture/tmdb-metadata-enrichment.md index 99abf701a..94f64cea8 100644 --- a/docs/architecture/tmdb-metadata-enrichment.md +++ b/docs/architecture/tmdb-metadata-enrichment.md @@ -537,8 +537,9 @@ identity is the query with only its case removed (`searchQueryIdentity`): variants are deduplicated by it and every attempted variant is cached under its own key (`title:|year:|v4`, in the language that variant was searched in), never by the folded key and never only under the first -variant — "Леика" and "Лейка" fold to one key but are different searches with -different answers, so a verdict for one must not be read back for the other; +variant — "Леика" and "Лейка", an illustrative pair, fold to one key while +staying two different searches that can get different answers, so a verdict +for one must not be read back for the other; a misspelled original title must not swallow the display title that TMDB actually knows; and two items that share an original title but not a display title walk different variant lists, so a row keyed on the first variant alone @@ -547,11 +548,13 @@ splits Cyrillic "й" into "и" + a combining breve and "ё" into "е" + a diaeresis, and Arabic hamza forms ("أ") into a bare alef + a combining hamza that the punctuation step then turns into a space inside the word. The key drops or splits on those marks, and TMDB's `/search` does not fold them the -same way — a query of `леика` returns zero results while `Лейка` returns the -show. Under the old single-form design every Russian title with "й"/"ё" -("Лейка (10 серий)", "Тестовый Сериал", "Пробный Выпуск") was searched -folded, missed, and cached as missing for the 7-day negative TTL. Compare -results only through `normalized`; never send it over the wire. +same way, so a folded query matches nothing there. Under the old single-form +design that hit every Russian title carrying "й" or "ё" and every Arabic +title carrying a hamza form: each was searched folded, missed, and cached as +missing for the 7-day negative TTL. The Cyrillic and Arabic strings used +throughout this section are illustrative stand-ins chosen to fold the same +way, not the titles the failures were observed on. Compare results only +through `normalized`; never send it over the wire. Electron IPC path (follows the standard DB worker contract, see [SQLite DB Worker](./sqlite-db-worker.md)): diff --git a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts index 3669196ad..fec3034e9 100644 --- a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts +++ b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts @@ -9,7 +9,8 @@ import { TmdbSearchResult } from './tmdb.types'; * Regression coverage for the search wire format. The comparison key folds * diacritics — Cyrillic "й" becomes "и" — and that key used to be sent as * the TMDB query, which matched nothing for any title carrying "й"/"ё" and - * cached the miss for a week ("Лейка (10 серий)", "Тестовый Сериал"). + * cached the miss for a week. The titles below are illustrative stand-ins + * that fold the same way, not the ones the failures were observed on. */ describe('TmdbIdResolverService.resolveBySearch', () => { const series2026: TmdbSearchResult = { diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts index 49159bf70..7a1ac086f 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts @@ -160,9 +160,9 @@ describe('buildSearchTitleVariants', () => { }); it('keeps spellings that fold to one key but differ on the wire', () => { - // A misspelled original title must not swallow the display title: - // TMDB knows "Лейка" and not "Леика", and only the second variant - // would find it. + // A misspelled original title must not swallow the display title. + // Take a pair where only the display spelling is the one TMDB + // indexes: only the second variant can find it. expect(buildSearchTitleVariants('Лейка', 'Леика')).toEqual([ { query: 'Леика', normalized: 'леика' }, { query: 'Лейка', normalized: 'леика' }, diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.ts b/libs/services/src/lib/tmdb/tmdb-matcher.ts index 57ef57088..e7c862292 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.ts @@ -53,7 +53,8 @@ export interface SearchTitleVariant { /** * The identity of one search on the wire: the query with only the case * removed, since TMDB matches case-insensitively and nothing else about - * the spelling may be folded away — "Леика" and "Лейка" are different + * the spelling may be folded away — "Леика" and "Лейка" (illustrative) are + * different * searches with different answers, however alike their comparison keys. * Both the variant deduplication and the cache row use this, so a cached * verdict can never be read back for a search that was never sent. @@ -116,11 +117,11 @@ export function buildSearchLookupKey( // v2: normalizeTitleKeys learned to strip appended language/quality // tags; the version suffix invalidates cached (incl. negative) match // resolutions keyed on the old polluted titles. - // v3: the search query stopped being the folded key ("леика" for - // "Лейка"), which TMDB answered with nothing; every negative row recorded - // under v2 for a title with "й"/"ё" is that bug, not a missing title, and - // must not block the retry for its 7-day TTL. Rows are keyed by the - // query since then. + // v3: the search query stopped being the folded key — the fold rewrites + // "й" as "и" ("леика" for "Лейка", to illustrate) and TMDB answers that + // spelling with nothing; every negative row recorded under v2 for a title + // with "й"/"ё" is that bug, not a missing title, and must not block the + // retry for its 7-day TTL. Rows are keyed by the query since then. // v4: year evidence is tiered (see `yearEvidenceTier`), so every v3 row // resolved by popularity across tiers may name the wrong show — and a // positive row stays fresh for 30 days. diff --git a/libs/shared/database/src/lib/connection.ts b/libs/shared/database/src/lib/connection.ts index f19806d3d..efc7a43f7 100644 --- a/libs/shared/database/src/lib/connection.ts +++ b/libs/shared/database/src/lib/connection.ts @@ -976,9 +976,9 @@ function widenTmdbMetadataMediaTypeCheck(sqliteDb: Database.Database): void { * - unversioned → v2: title normalization learned to strip appended * language/quality tags. * - v2 → v3: the search query stopped being the folded comparison key. Under - * v2 every title with a Cyrillic "й"/"ё" was searched folded ("леика" for - * "Лейка", "ежик" for "Ёжик"), got no answer, and was cached as missing for - * 7 days. + * v2 every title with a Cyrillic "й"/"ё" was searched folded — the fold + * spells them "и" and "е", as in the illustrative "леика" for "Лейка" — + * got no answer, and was cached as missing for 7 days. * - v3 → v4: year evidence became tiered. Under v3 a series admitted only by * the "premiered earlier" tolerance competed with an exact-year match on * popularity alone, so a new series resolved to its older, better-known diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts index 08e9c9a47..40a96cab0 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts @@ -434,8 +434,9 @@ describe('titleYearsCompatible', () => { describe('cleanTitleForSearch', () => { it('keeps Cyrillic letters that folding would rewrite', () => { // NFD splits "й" into "и" + a breve and "ё" into "е" + a diaeresis; - // the comparison key drops both marks, and TMDB answers the folded - // spelling with nothing (issue: "Лейка (10 серий)" never matched). + // the comparison key drops both marks, and TMDB answers that folded + // spelling with nothing. The titles here are illustrative stand-ins + // chosen to fold the same way. expect(normalizeTitle('Лейка (10 серий)')).toBe('леика'); expect(cleanTitleForSearch('Лейка (10 серий)')).toBe('Лейка'); expect(cleanTitleForSearch('Ёжик 2010')).toBe('Ёжик'); @@ -446,8 +447,8 @@ describe('cleanTitleForSearch', () => { it('keeps Arabic hamza forms that folding splits into two words', () => { // "أ" decomposes into a bare alef + U+0654, which is outside the - // stripped mark range and so becomes a SPACE in the key. TMDB finds - // "أمثلة تجريبية" and nothing for "ا مثلة تجريبية". + // stripped mark range and so becomes a SPACE in the key — splitting + // the word in two, which is not a spelling TMDB indexes. expect(normalizeTitle('AR| أمثلة تجريبية')).toBe('ا مثلة تجريبية'); expect(cleanTitleForSearch('AR| أمثلة تجريبية')).toBe('أمثلة تجريبية'); }); diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.ts b/libs/shared/interfaces/src/lib/title-normalization.util.ts index 19ca113b5..0c3759376 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.ts @@ -520,7 +520,8 @@ export function normalizeTitleKeys( * * Comparison keys must fold so two spellings of one film meet; a search * query must not, because the search engine folds by its own rules and a - * pre-folded Cyrillic query ("леика" for "Лейка") matches nothing there. + * pre-folded Cyrillic query matches nothing there ("леика" for "Лейка" + * illustrates the rewrite). * Compare the results with `normalizeTitle`, never with this. */ export function cleanTitleForSearch(raw: string | null | undefined): string {