From f0096c6437eb56ca8c481cdf5e0ebedae5479f54 Mon Sep 17 00:00:00 2001 From: 4gray Date: Mon, 21 Sep 2026 20:07:37 +0200 Subject: [PATCH] docs(tmdb): label the folding stand-ins as illustrative, not as history MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review caught that the substitution left synthetic titles inside sentences that assert observed fact: the architecture doc claimed a particular folded query "returns zero results while `Лейка` returns the show" and named the stand-ins as the titles that were searched, missed and negatively cached, and several comments read the same way. The titles never existed, so the reading is wrong in the one direction that matters — a future reader debugging a regression would take them for production evidence. Keep the claim that is actually established, which is a class-level one: under the old single-form design every Cyrillic title carrying "й"/"ё" and every Arabic title carrying a hamza form was searched folded, missed, and cached as missing for the negative TTL. Present the strings themselves as illustrative stand-ins chosen to fold the same way. The verified invariant — what NFD does to those letters, and what the two tiers therefore produce — is unchanged and still assertable, because it is a property of the text rather than of any title. Co-Authored-By: Claude Opus 5 --- docs/architecture/tmdb-metadata-enrichment.md | 17 ++++++++++------- .../lib/tmdb/tmdb-id-resolver.service.spec.ts | 3 ++- libs/services/src/lib/tmdb/tmdb-matcher.spec.ts | 6 +++--- libs/services/src/lib/tmdb/tmdb-matcher.ts | 13 +++++++------ libs/shared/database/src/lib/connection.ts | 6 +++--- .../src/lib/title-normalization.util.spec.ts | 9 +++++---- .../src/lib/title-normalization.util.ts | 3 ++- 7 files changed, 32 insertions(+), 25 deletions(-) diff --git a/docs/architecture/tmdb-metadata-enrichment.md b/docs/architecture/tmdb-metadata-enrichment.md index 99abf701a..94f64cea8 100644 --- a/docs/architecture/tmdb-metadata-enrichment.md +++ b/docs/architecture/tmdb-metadata-enrichment.md @@ -537,8 +537,9 @@ identity is the query with only its case removed (`searchQueryIdentity`): variants are deduplicated by it and every attempted variant is cached under its own key (`title:|year:|v4`, in the language that variant was searched in), never by the folded key and never only under the first -variant — "Леика" and "Лейка" fold to one key but are different searches with -different answers, so a verdict for one must not be read back for the other; +variant — "Леика" and "Лейка", an illustrative pair, fold to one key while +staying two different searches that can get different answers, so a verdict +for one must not be read back for the other; a misspelled original title must not swallow the display title that TMDB actually knows; and two items that share an original title but not a display title walk different variant lists, so a row keyed on the first variant alone @@ -547,11 +548,13 @@ splits Cyrillic "й" into "и" + a combining breve and "ё" into "е" + a diaeresis, and Arabic hamza forms ("أ") into a bare alef + a combining hamza that the punctuation step then turns into a space inside the word. The key drops or splits on those marks, and TMDB's `/search` does not fold them the -same way — a query of `леика` returns zero results while `Лейка` returns the -show. Under the old single-form design every Russian title with "й"/"ё" -("Лейка (10 серий)", "Тестовый Сериал", "Пробный Выпуск") was searched -folded, missed, and cached as missing for the 7-day negative TTL. Compare -results only through `normalized`; never send it over the wire. +same way, so a folded query matches nothing there. Under the old single-form +design that hit every Russian title carrying "й" or "ё" and every Arabic +title carrying a hamza form: each was searched folded, missed, and cached as +missing for the 7-day negative TTL. The Cyrillic and Arabic strings used +throughout this section are illustrative stand-ins chosen to fold the same +way, not the titles the failures were observed on. Compare results only +through `normalized`; never send it over the wire. Electron IPC path (follows the standard DB worker contract, see [SQLite DB Worker](./sqlite-db-worker.md)): diff --git a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts index 3669196ad..fec3034e9 100644 --- a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts +++ b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts @@ -9,7 +9,8 @@ import { TmdbSearchResult } from './tmdb.types'; * Regression coverage for the search wire format. The comparison key folds * diacritics — Cyrillic "й" becomes "и" — and that key used to be sent as * the TMDB query, which matched nothing for any title carrying "й"/"ё" and - * cached the miss for a week ("Лейка (10 серий)", "Тестовый Сериал"). + * cached the miss for a week. The titles below are illustrative stand-ins + * that fold the same way, not the ones the failures were observed on. */ describe('TmdbIdResolverService.resolveBySearch', () => { const series2026: TmdbSearchResult = { diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts index 49159bf70..7a1ac086f 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts @@ -160,9 +160,9 @@ describe('buildSearchTitleVariants', () => { }); it('keeps spellings that fold to one key but differ on the wire', () => { - // A misspelled original title must not swallow the display title: - // TMDB knows "Лейка" and not "Леика", and only the second variant - // would find it. + // A misspelled original title must not swallow the display title. + // Take a pair where only the display spelling is the one TMDB + // indexes: only the second variant can find it. expect(buildSearchTitleVariants('Лейка', 'Леика')).toEqual([ { query: 'Леика', normalized: 'леика' }, { query: 'Лейка', normalized: 'леика' }, diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.ts b/libs/services/src/lib/tmdb/tmdb-matcher.ts index 57ef57088..e7c862292 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.ts @@ -53,7 +53,8 @@ export interface SearchTitleVariant { /** * The identity of one search on the wire: the query with only the case * removed, since TMDB matches case-insensitively and nothing else about - * the spelling may be folded away — "Леика" and "Лейка" are different + * the spelling may be folded away — "Леика" and "Лейка" (illustrative) are + * different * searches with different answers, however alike their comparison keys. * Both the variant deduplication and the cache row use this, so a cached * verdict can never be read back for a search that was never sent. @@ -116,11 +117,11 @@ export function buildSearchLookupKey( // v2: normalizeTitleKeys learned to strip appended language/quality // tags; the version suffix invalidates cached (incl. negative) match // resolutions keyed on the old polluted titles. - // v3: the search query stopped being the folded key ("леика" for - // "Лейка"), which TMDB answered with nothing; every negative row recorded - // under v2 for a title with "й"/"ё" is that bug, not a missing title, and - // must not block the retry for its 7-day TTL. Rows are keyed by the - // query since then. + // v3: the search query stopped being the folded key — the fold rewrites + // "й" as "и" ("леика" for "Лейка", to illustrate) and TMDB answers that + // spelling with nothing; every negative row recorded under v2 for a title + // with "й"/"ё" is that bug, not a missing title, and must not block the + // retry for its 7-day TTL. Rows are keyed by the query since then. // v4: year evidence is tiered (see `yearEvidenceTier`), so every v3 row // resolved by popularity across tiers may name the wrong show — and a // positive row stays fresh for 30 days. diff --git a/libs/shared/database/src/lib/connection.ts b/libs/shared/database/src/lib/connection.ts index f19806d3d..efc7a43f7 100644 --- a/libs/shared/database/src/lib/connection.ts +++ b/libs/shared/database/src/lib/connection.ts @@ -976,9 +976,9 @@ function widenTmdbMetadataMediaTypeCheck(sqliteDb: Database.Database): void { * - unversioned → v2: title normalization learned to strip appended * language/quality tags. * - v2 → v3: the search query stopped being the folded comparison key. Under - * v2 every title with a Cyrillic "й"/"ё" was searched folded ("леика" for - * "Лейка", "ежик" for "Ёжик"), got no answer, and was cached as missing for - * 7 days. + * v2 every title with a Cyrillic "й"/"ё" was searched folded — the fold + * spells them "и" and "е", as in the illustrative "леика" for "Лейка" — + * got no answer, and was cached as missing for 7 days. * - v3 → v4: year evidence became tiered. Under v3 a series admitted only by * the "premiered earlier" tolerance competed with an exact-year match on * popularity alone, so a new series resolved to its older, better-known diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts index 08e9c9a47..40a96cab0 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts @@ -434,8 +434,9 @@ describe('titleYearsCompatible', () => { describe('cleanTitleForSearch', () => { it('keeps Cyrillic letters that folding would rewrite', () => { // NFD splits "й" into "и" + a breve and "ё" into "е" + a diaeresis; - // the comparison key drops both marks, and TMDB answers the folded - // spelling with nothing (issue: "Лейка (10 серий)" never matched). + // the comparison key drops both marks, and TMDB answers that folded + // spelling with nothing. The titles here are illustrative stand-ins + // chosen to fold the same way. expect(normalizeTitle('Лейка (10 серий)')).toBe('леика'); expect(cleanTitleForSearch('Лейка (10 серий)')).toBe('Лейка'); expect(cleanTitleForSearch('Ёжик 2010')).toBe('Ёжик'); @@ -446,8 +447,8 @@ describe('cleanTitleForSearch', () => { it('keeps Arabic hamza forms that folding splits into two words', () => { // "أ" decomposes into a bare alef + U+0654, which is outside the - // stripped mark range and so becomes a SPACE in the key. TMDB finds - // "أمثلة تجريبية" and nothing for "ا مثلة تجريبية". + // stripped mark range and so becomes a SPACE in the key — splitting + // the word in two, which is not a spelling TMDB indexes. expect(normalizeTitle('AR| أمثلة تجريبية')).toBe('ا مثلة تجريبية'); expect(cleanTitleForSearch('AR| أمثلة تجريبية')).toBe('أمثلة تجريبية'); }); diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.ts b/libs/shared/interfaces/src/lib/title-normalization.util.ts index 19ca113b5..0c3759376 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.ts @@ -520,7 +520,8 @@ export function normalizeTitleKeys( * * Comparison keys must fold so two spellings of one film meet; a search * query must not, because the search engine folds by its own rules and a - * pre-folded Cyrillic query ("леика" for "Лейка") matches nothing there. + * pre-folded Cyrillic query matches nothing there ("леика" for "Лейка" + * illustrates the rewrite). * Compare the results with `normalizeTitle`, never with this. */ export function cleanTitleForSearch(raw: string | null | undefined): string {