diff --git a/.changes/tmdb-cyrillic-search-spelling.md b/.changes/tmdb-cyrillic-search-spelling.md new file mode 100644 index 000000000..bd7fb6d05 --- /dev/null +++ b/.changes/tmdb-cyrillic-search-spelling.md @@ -0,0 +1,10 @@ +--- +type: fix +area: tmdb +--- + +Russian titles containing "й" or "ё" ("Фейк", "Волшебный участок", "Молодой +Шерлок") and Arabic titles with hamza letters ("أطرق بابي") now match on TMDB. +The search used to send a folded spelling ("феик") that TMDB never recognised +and cached the miss for a week; those cached misses are cleared on the next +start. diff --git a/CLAUDE.md b/CLAUDE.md index 8900c0793..d14547d51 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1617,7 +1617,7 @@ stream_id`); it drops `series_id`/`movie_id`, so the builder pins the - Opt-in via `Settings > Metadata (TMDB)` (sends titles to TMDB); the section also has a "check key" button and a cache panel (row count + payload size, with a clear button). Distributed builds ship without a shared key; users supply their own. `DEFAULT_TMDB_API_KEY` in `libs/services/src/lib/tmdb/tmdb-config.ts` is empty by default; `tools/tmdb/inject-tmdb-key.mjs` still supports optional CI injection from `TMDB_API_KEY`, but is a no-op when it is unset. A user key takes precedence over any injected default; without either key, enrichment stays inactive even when enabled. Requiring a personal key is project policy, not a categorical TMDB terms restriction; see the canonical "Settings and API Key" section in `docs/architecture/tmdb-metadata-enrichment.md`. - Match confidence: a provider `tmdb_id` is a strong hint, not gospel — its payload is weighed against the item (`assessProviderId`: title or year agrees → use it; both years known and incompatible → the search may take over; title-only mismatch → keep it, since TMDB localizes titles). A 404 marks the id dead (`badProviderId:` row); transient failures never do. Without a usable id: normalized-title + year (±1) search with a strict gate — no confident match means no enrichment - Detail views render provider data immediately; enrichment patches the selection asynchronously (staleness-guarded) -- Cached in SQLite `tmdb_metadata` (Electron, via DB worker ops `DB_GET/SET_TMDB_METADATA`, plus `DB_GET_TMDB_CACHE_STATS` / `DB_CLEAR_TMDB_METADATA` behind the settings cache panel) or in-memory (PWA); localized via the app language setting. Search-match lookup keys are versioned, and connection startup removes obsolete unversioned rows once through the `migration:tmdb-search-lookup-v2-cache-cleanup:v1` app-state marker. +- Cached in SQLite `tmdb_metadata` (Electron, via DB worker ops `DB_GET/SET_TMDB_METADATA`, plus `DB_GET_TMDB_CACHE_STATS` / `DB_CLEAR_TMDB_METADATA` behind the settings cache panel) or in-memory (PWA); localized via the app language setting. Search-match lookup keys are versioned (`|v3`), and connection startup removes the rows of every retired generation once, each under its own app-state marker (`migration:tmdb-search-lookup-v2-cache-cleanup:v1` for unversioned rows, `migration:tmdb-search-lookup-v3-cache-cleanup:v1` for `|v2` rows). The TMDB search query is `cleanTitleForSearch` (provider spelling, tags/brackets/season/year stripped), NOT the folded `normalizeTitle` key used by `pickConfidentMatch`; variant deduplication uses the lowercased query (`searchQueryIdentity`) and every attempted variant is cached under its own key, since "Феик"/"Фейк" fold to one key but are different searches and two items can share an original title while walking different variant lists: NFD folding rewrites Cyrillic "й"→"и" and "ё"→"е" and splits Arabic hamza forms, and TMDB answers such a query with nothing ("Фейк (10 серий)" never matched). Never send the folded key over the wire. - Service layer: `libs/services/src/lib/tmdb/`; store glue: `libs/portal/xtream/data-access/src/lib/stores/xtream-tmdb-enrichment.ts` and `libs/portal/stalker/data-access/src/lib/stores/stalker-tmdb-enrichment.ts` (hooked in `withStalkerSelection().setSelectedItem`) - TMDB attribution (logo + disclaimer) is required and shown in the settings TMDB section and About - See `docs/architecture/tmdb-metadata-enrichment.md` diff --git a/docs/architecture/tmdb-metadata-enrichment.md b/docs/architecture/tmdb-metadata-enrichment.md index 9bbf515b3..9390b9db7 100644 --- a/docs/architecture/tmdb-metadata-enrichment.md +++ b/docs/architecture/tmdb-metadata-enrichment.md @@ -456,7 +456,7 @@ tmdb_metadata ( media_type 'movie' | 'tv' | 'person', lookup_key 'id:|v2' -- details payload row 'id:|season:' -- season payload row - 'title:|year:|v2' -- search resolution row + 'title:|year:|v3' -- search resolution row 'person:' -- person payload row 'trending:week' -- trending list row 'badProviderId:' -- id confirmed 404 by TMDB @@ -471,14 +471,46 @@ tmdb_metadata ( TTLs (enforced at read time in `TmdbCacheService.isFresh`): details and positive matches 30 days, negative matches 7 days. -Search and details keys carry a `|v2` version suffix (`buildDetailsLookupKey` -in `tmdb-matcher.ts`): for search rows so normalization changes cannot reuse -stale positive or negative resolutions, for details rows because payloads now -include videos via `append_to_response` and pre-videos cache rows had to be -invalidated. Database startup deletes the obsolete -unversioned search rows once and records -`migration:tmdb-search-lookup-v2-cache-cleanup:v1` in `app_state`; details and -person cache rows are unaffected. +Search keys carry a `|v3` and details keys a `|v2` version suffix +(`buildSearchLookupKey` / `buildDetailsLookupKey` in `tmdb-matcher.ts`): for +search rows so normalization or query changes cannot reuse stale positive or +negative resolutions, for details rows because payloads now include videos +via `append_to_response` and pre-videos cache rows had to be invalidated. +Search v2 → v3 retired the rows written while the folded comparison key was +also the wire query (see "Search query vs. comparison key" below). Database +startup deletes the rows of every retired search-key generation once, each +under its own `app_state` marker so a skipped release still runs the cleanups +it missed (`migration:tmdb-search-lookup-v2-cache-cleanup:v1` for the +unversioned rows, `migration:tmdb-search-lookup-v3-cache-cleanup:v1` for the +`|v2` rows; `LEGACY_TMDB_SEARCH_CACHE_CLEANUPS` in `connection.ts`); details +and person cache rows are unaffected. + +### Search query vs. comparison key + +`buildSearchTitleVariants` yields `{ query, normalized }` pairs. `normalized` +is `normalizeTitle` — the folded key used for `pickConfidentMatch`, where +both sides fold the same way. `query` is `cleanTitleForSearch` +(`libs/shared/interfaces`): the same tag, bracket, season and trailing-year +stripping, but the letters left as the provider wrote them. The search's +identity is the query with only its case removed (`searchQueryIdentity`): +variants are deduplicated by it and every attempted variant is cached under +its own key (`title:|year:|v3`, in the language that variant +was searched in), never by the folded key and never only under the first +variant — "Феик" and "Фейк" fold to one key but are different searches with +different answers, so a verdict for one must not be read back for the other; +a misspelled original title must not swallow the display title that TMDB +actually knows; and two items that share an original title but not a display +title walk different variant lists, so a row keyed on the first variant alone +would hand the second item the first one's answer, or its cached miss. The two must differ because folding is lossy outside Latin: NFD +splits Cyrillic "й" into "и" + a combining breve and "ё" into "е" + a +diaeresis, and Arabic hamza forms ("أ") into a bare alef + a combining hamza +that the punctuation step then turns into a space inside the word. The key +drops or splits on those marks, and TMDB's `/search` does not fold them the +same way — a query of `феик` returns zero results while `Фейк` returns the +show. Under the old single-form design every Russian title with "й"/"ё" +("Фейк (10 серий)", "Волшебный участок", "Молодой Шерлок") was searched +folded, missed, and cached as missing for the 7-day negative TTL. Compare +results only through `normalized`; never send it over the wire. Electron IPC path (follows the standard DB worker contract, see [SQLite DB Worker](./sqlite-db-worker.md)): diff --git a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts new file mode 100644 index 000000000..aa9549eef --- /dev/null +++ b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts @@ -0,0 +1,220 @@ +import { Injector, runInInjectionContext } from '@angular/core'; +import { TmdbApiService } from './tmdb-api.service'; +import { TmdbCacheService } from './tmdb-cache.service'; +import { TmdbIdResolverService } from './tmdb-id-resolver.service'; +import { TmdbRuntimeService } from './tmdb-runtime.service'; +import { TmdbSearchResult } from './tmdb.types'; + +/** + * Regression coverage for the search wire format. The comparison key folds + * diacritics — Cyrillic "й" becomes "и" — and that key used to be sent as + * the TMDB query, which matched nothing for any title carrying "й"/"ё" and + * cached the miss for a week ("Фейк (10 серий)", "Волшебный участок"). + */ +describe('TmdbIdResolverService.resolveBySearch', () => { + const fake2026: TmdbSearchResult = { + id: 317869, + name: 'Фейк', + original_name: 'Фейк', + first_air_date: '2026-07-16', + vote_count: 3, + }; + const fake2024: TmdbSearchResult = { + id: 322696, + name: 'Фейк', + original_name: 'Фейк', + first_air_date: '2024-12-05', + vote_count: 0, + }; + + let searchTv: jest.Mock; + let searchMovie: jest.Mock; + let cacheGet: jest.Mock; + let cacheSet: jest.Mock; + + // The services Jest target has no @angular/core/testing — build the + // service in a plain injection context instead of TestBed. + function createService(): TmdbIdResolverService { + const injector = Injector.create({ + providers: [ + { + provide: TmdbRuntimeService, + useValue: { + apiKey: () => 'key', + appLanguage: () => 'en', + }, + }, + { + provide: TmdbApiService, + useValue: { searchTv, searchMovie }, + }, + { + provide: TmdbCacheService, + useValue: { + get: cacheGet, + set: cacheSet, + isFresh: () => false, + }, + }, + ], + }); + return runInInjectionContext( + injector, + () => new TmdbIdResolverService() + ); + } + + beforeEach(() => { + searchTv = jest.fn(async (query: string) => + // TMDB does not fold Cyrillic: only the provider spelling hits + query === 'Фейк' ? [fake2026, fake2024] : [] + ); + searchMovie = jest.fn().mockResolvedValue([]); + cacheGet = jest.fn().mockResolvedValue(null); + cacheSet = jest.fn().mockResolvedValue(undefined); + }); + + it('sends the provider spelling to TMDB and matches on the folded key', async () => { + const service = createService(); + + const id = await service.resolveBySearch('tv', { + title: 'Фейк (10 серий)', + year: 2026, + }); + + expect(id).toBe(317869); + expect(searchTv).toHaveBeenCalledTimes(1); + expect(searchTv).toHaveBeenCalledWith('Фейк', null, 'ru-RU', 'key'); + expect(cacheSet).toHaveBeenCalledWith({ + mediaType: 'tv', + lookupKey: 'title:фейк|year:2026|v3', + language: 'ru-RU', + tmdbId: 317869, + payload: null, + }); + }); + + it('caches the miss under the folded v3 key', async () => { + searchTv.mockResolvedValue([]); + const service = createService(); + + const id = await service.resolveBySearch('tv', { + title: 'Молодой Шерлок', + year: 2026, + }); + + expect(id).toBeNull(); + expect(searchTv).toHaveBeenCalledWith( + 'Молодой Шерлок', + null, + 'ru-RU', + 'key' + ); + expect(cacheSet).toHaveBeenCalledWith( + expect.objectContaining({ + lookupKey: 'title:молодой шерлок|year:2026|v3', + tmdbId: null, + }) + ); + }); + + it('reads the cache before searching', async () => { + cacheGet.mockResolvedValue({ tmdbId: 317869 }); + const service = createService(); + const cacheService = (service as unknown as { cache: TmdbCacheService }) + .cache; + jest.spyOn(cacheService, 'isFresh').mockReturnValue(true); + + const id = await service.resolveBySearch('tv', { + title: 'Фейк (10 серий)', + year: 2026, + }); + + expect(id).toBe(317869); + expect(cacheGet).toHaveBeenCalledWith( + 'tv', + 'title:фейк|year:2026|v3', + 'ru-RU' + ); + expect(searchTv).not.toHaveBeenCalled(); + }); + + it('searches a display spelling that a misspelled original title folds onto', async () => { + const service = createService(); + + const id = await service.resolveBySearch('tv', { + title: 'Фейк', + originalTitle: 'Феик', + year: 2026, + }); + + expect(id).toBe(317869); + expect(searchTv.mock.calls.map(([query]) => query)).toEqual([ + 'Феик', + 'Фейк', + ]); + // Each attempted variant records its own verdict + expect( + cacheSet.mock.calls.map(([row]) => [row.lookupKey, row.tmdbId]) + ).toEqual([ + ['title:феик|year:2026|v3', null], + ['title:фейк|year:2026|v3', 317869], + ]); + }); + + it("does not let one variant's cached verdict answer for another", async () => { + // Item A (original "Феик", display "Феик") cached a miss under + // "феик". Item B shares the original title but displays "Фейк": + // the cached miss must not suppress B's own second variant. + cacheGet.mockImplementation(async (_type: string, key: string) => + key === 'title:феик|year:2026|v3' ? { tmdbId: null } : null + ); + const service = createService(); + const cacheService = (service as unknown as { cache: TmdbCacheService }) + .cache; + jest.spyOn(cacheService, 'isFresh').mockImplementation( + (row) => row !== null && row !== undefined + ); + + const id = await service.resolveBySearch('tv', { + title: 'Фейк', + originalTitle: 'Феик', + year: 2026, + }); + + expect(id).toBe(317869); + expect(searchTv).toHaveBeenCalledTimes(1); + expect(searchTv).toHaveBeenCalledWith('Фейк', null, 'ru-RU', 'key'); + expect(cacheGet.mock.calls.map(([, key]) => key)).toEqual([ + 'title:феик|year:2026|v3', + 'title:фейк|year:2026|v3', + ]); + }); + + it('tries the language-prefix-stripped fallback with its own spelling', async () => { + searchMovie = jest.fn(async (query: string) => + query === 'Amélie' + ? [ + { + id: 194, + title: 'Amélie', + original_title: 'Le Fabuleux Destin d’Amélie Poulain', + release_date: '2001-04-25', + }, + ] + : [] + ); + const service = createService(); + + const id = await service.resolveBySearch('movie', { + title: 'FR Amélie', + year: 2001, + }); + + expect(id).toBe(194); + expect(searchMovie.mock.calls.map(([query]) => query)).toEqual([ + 'FR Amélie', + 'Amélie', + ]); + }); +}); diff --git a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.ts b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.ts index d8e90051a..d6b37cf19 100644 --- a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.ts +++ b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.ts @@ -8,6 +8,7 @@ import { tmdbSearchLanguageForTitle, } from './tmdb-config'; import { + SearchTitleVariant, buildBadProviderIdLookupKey, buildSearchLookupKey, buildSearchTitleVariants, @@ -37,7 +38,12 @@ export class TmdbIdResolverService { /** * Resolve a title/year to a TMDB id via /search with the confidence - * gate. Both hits and misses are cached; misses use a shorter TTL. + * gate. Every attempted variant is cached under its own key — hits for + * 30 days, misses for 7 — because a verdict belongs to the search that + * produced it, not to the item that asked: two items sharing an + * original title but not a display title walk different variant lists, + * and a row keyed on the first variant alone would hand the second item + * the first one's answer, or its cached miss. */ async resolveBySearch( mediaType: TmdbMediaType, @@ -49,22 +55,39 @@ export class TmdbIdResolverService { query.title, query.originalTitle ); - if (variants.length === 0) { - return null; + const year = query.year ?? extractYear(null, query.title); + + for (const variant of variants) { + const resolved = await this.resolveVariant( + mediaType, + variant, + year + ); + if (resolved !== null) { + return resolved; + } } - const year = query.year ?? extractYear(null, query.title); - const cacheLanguage = tmdbSearchLanguageForTitle( - variants[0], + return null; + } + + /** One variant's cached or freshly searched verdict; null on a miss. */ + private async resolveVariant( + mediaType: TmdbMediaType, + variant: SearchTitleVariant, + year: number | null + ): Promise { + // Cyrillic (and other non-app-script) titles search in their own + // language so TMDB returns comparable titles — see + // tmdbSearchLanguageForTitle. The cache row carries the same + // language, since the answer depends on it. + const language = tmdbSearchLanguageForTitle( + variant.normalized, this.runtime.appLanguage() ); - const lookupKey = buildSearchLookupKey(variants[0], year); + const lookupKey = buildSearchLookupKey(variant.query, year); - const cached = await this.cache.get( - mediaType, - lookupKey, - cacheLanguage - ); + const cached = await this.cache.get(mediaType, lookupKey, language); const ttl = cached?.tmdbId !== null && cached?.tmdbId !== undefined ? TMDB_MATCH_CACHE_TTL_MS @@ -73,46 +96,36 @@ export class TmdbIdResolverService { return cached?.tmdbId ?? null; } - let match = null; - for (const variant of variants) { - // Cyrillic (and other non-app-script) titles search in their - // own language so TMDB returns comparable titles — see - // tmdbSearchLanguageForTitle. Search by title only: TMDB's - // year params filter strictly; the ±1/season tolerance lives - // in pickConfidentMatch instead. - const language = tmdbSearchLanguageForTitle( - variant, - this.runtime.appLanguage() - ); - const results = - mediaType === 'movie' - ? await this.api.searchMovie( - variant, - null, - language, - this.runtime.apiKey() - ) - : await this.api.searchTv( - variant, - null, - language, - this.runtime.apiKey() - ); + // Search by title only: TMDB's year params filter strictly; the + // ±1/season tolerance lives in pickConfidentMatch instead. The wire + // query is the provider's own spelling (`variant.query`), never the + // folded comparison key: TMDB does not fold Cyrillic "й" the way + // the key does, and a folded query finds nothing. + const results = + mediaType === 'movie' + ? await this.api.searchMovie( + variant.query, + null, + language, + this.runtime.apiKey() + ) + : await this.api.searchTv( + variant.query, + null, + language, + this.runtime.apiKey() + ); - match = pickConfidentMatch( - results, - { title: variant, year }, - mediaType - ); - if (match) { - break; - } - } + const match = pickConfidentMatch( + results, + { title: variant.normalized, year }, + mediaType + ); await this.cache.set({ mediaType, lookupKey, - language: cacheLanguage, + language, tmdbId: match?.id ?? null, payload: null, }); diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts index 7a3b737a1..4bc8860a1 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts @@ -79,11 +79,26 @@ describe('extractYear', () => { describe('lookup keys', () => { it('builds stable search and details keys', () => { - expect(buildSearchLookupKey('the matrix', 1999)).toBe( - 'title:the matrix|year:1999|v2' + expect(buildSearchLookupKey('The Matrix', 1999)).toBe( + 'title:the matrix|year:1999|v3' ); - expect(buildSearchLookupKey('the matrix', null)).toBe( - 'title:the matrix|year:|v2' + expect(buildSearchLookupKey('The Matrix', null)).toBe( + 'title:the matrix|year:|v3' + ); + }); + + it('keys search rows by the wire spelling, not the folded key', () => { + // "Феик" and "Фейк" fold to one comparison key but are different + // searches with different answers; a verdict cached for one must + // never be read back for the other. + expect(buildSearchLookupKey('Фейк', 2026)).toBe( + 'title:фейк|year:2026|v3' + ); + expect(buildSearchLookupKey('Феик', 2026)).not.toBe( + buildSearchLookupKey('Фейк', 2026) + ); + expect(buildSearchLookupKey('THE BOYS', 2019)).toBe( + buildSearchLookupKey('The Boys', 2019) ); expect(buildDetailsLookupKey(603)).toBe('id:603|v2'); }); @@ -92,35 +107,71 @@ describe('lookup keys', () => { describe('buildSearchTitleVariants', () => { it('orders original title before display title', () => { expect(buildSearchTitleVariants('Пацаны', 'The Boys')).toEqual([ - 'the boys', - 'пацаны', + { query: 'The Boys', normalized: 'the boys' }, + { query: 'Пацаны', normalized: 'пацаны' }, ]); }); it('adds a language-prefix-stripped fallback variant', () => { expect(buildSearchTitleVariants('DE Batman', null)).toEqual([ - 'de batman', - 'batman', + { query: 'DE Batman', normalized: 'de batman' }, + { query: 'Batman', normalized: 'batman' }, ]); - expect( - buildSearchTitleVariants('English The Godfather', null) - ).toEqual(['english the godfather', 'the godfather']); + expect(buildSearchTitleVariants('English The Godfather', null)).toEqual( + [ + { + query: 'English The Godfather', + normalized: 'english the godfather', + }, + { query: 'The Godfather', normalized: 'the godfather' }, + ] + ); }); it('keeps titles that merely look like prefixed ones as the primary variant', () => { // "It Follows" must be searched as-is first; the stripped variant // is only a fallback - expect(buildSearchTitleVariants('It Follows', null)[0]).toBe( - 'it follows' - ); + expect(buildSearchTitleVariants('It Follows', null)[0]).toEqual({ + query: 'It Follows', + normalized: 'it follows', + }); }); - it('deduplicates and drops empty values', () => { + it('sends the provider spelling to the search but compares on the folded key', () => { + // The folded key rewrites "й" as "и"; TMDB finds nothing for it. + expect(buildSearchTitleVariants('Фейк (10 серий)', null)).toEqual([ + { query: 'Фейк', normalized: 'феик' }, + ]); + // Arabic hamza forms fold into a space inside the word + expect(buildSearchTitleVariants('إيمان', null)).toEqual([ + { query: 'إيمان', normalized: 'ا يمان' }, + ]); + }); + + it('deduplicates by the wire spelling and drops empty values', () => { expect(buildSearchTitleVariants('The Boys', 'The Boys')).toEqual([ - 'the boys', + { query: 'The Boys', normalized: 'the boys' }, + ]); + // Case alone is not a different search + expect(buildSearchTitleVariants('the boys', 'The Boys')).toEqual([ + { query: 'The Boys', normalized: 'the boys' }, ]); expect(buildSearchTitleVariants('', null)).toEqual([]); }); + + it('keeps spellings that fold to one key but differ on the wire', () => { + // A misspelled original title must not swallow the display title: + // TMDB knows "Фейк" and not "Феик", and only the second variant + // would find it. + expect(buildSearchTitleVariants('Фейк', 'Феик')).toEqual([ + { query: 'Феик', normalized: 'феик' }, + { query: 'Фейк', normalized: 'феик' }, + ]); + expect(buildSearchTitleVariants('Amelie', 'Amélie')).toEqual([ + { query: 'Amélie', normalized: 'amelie' }, + { query: 'Amelie', normalized: 'amelie' }, + ]); + }); }); describe('parseProviderTmdbId', () => { diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.ts b/libs/services/src/lib/tmdb/tmdb-matcher.ts index 390e373a9..98977beb0 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.ts @@ -1,5 +1,6 @@ import { TmdbMediaType, + cleanTitleForSearch, extractYear, normalizeTitle, } from '@iptvnator/shared/interfaces'; @@ -36,22 +37,58 @@ function stripLeadingLanguageToken(raw: string): string | null { return null; } +/** + * One search candidate: what to SEND to TMDB and what to COMPARE its + * answers against. The two differ on purpose — see `cleanTitleForSearch`: + * a folded query ("феик") finds nothing on TMDB while the folded key is + * exactly what the confidence gate and the cache need. + */ +export interface SearchTitleVariant { + /** Provider spelling with tags/brackets/season/year stripped */ + query: string; + /** `normalizeTitle` of the same text; cache key and comparison form */ + normalized: string; +} + +/** + * The identity of one search on the wire: the query with only the case + * removed, since TMDB matches case-insensitively and nothing else about + * the spelling may be folded away — "Феик" and "Фейк" are different + * searches with different answers, however alike their comparison keys. + * Both the variant deduplication and the cache row use this, so a cached + * verdict can never be read back for a search that was never sent. + */ +export function searchQueryIdentity(query: string): string { + return query.toLowerCase(); +} + /** * Ordered search-title candidates for one provider item: the original * title, the display title, then the same values with a leading * language-looking token dropped. The confidence gate still applies to * every variant, so extra candidates cannot produce wrong matches — only - * extra searches on misses. + * extra searches on misses. Deduplicated by the wire identity, never by + * the folded comparison key: two spellings that fold to one key can still + * be different searches, and dropping the second would silently skip the + * one TMDB actually knows. */ export function buildSearchTitleVariants( title: string | null | undefined, originalTitle?: string | null -): string[] { - const variants: string[] = []; +): SearchTitleVariant[] { + const variants: SearchTitleVariant[] = []; const push = (raw: string | null | undefined) => { const normalized = normalizeTitle(raw); - if (normalized && !variants.includes(normalized)) { - variants.push(normalized); + const query = cleanTitleForSearch(raw); + const identity = searchQueryIdentity(query); + if ( + normalized && + query && + !variants.some( + (variant) => searchQueryIdentity(variant.query) === identity + ) + ) { + variants.push({ query, normalized }); } }; @@ -66,14 +103,25 @@ export function buildSearchTitleVariants( return variants; } +/** + * Cache row for one search verdict, keyed by the wire query's identity + * (`searchQueryIdentity`), not by the folded comparison key: the verdict + * depends on what was sent, and two spellings sharing a folded key ("Все" + * / "Всё") may get different answers. + */ export function buildSearchLookupKey( - normalizedTitle: string, + query: string, year: number | null ): string { // v2: normalizeTitleKeys learned to strip appended language/quality // tags; the version suffix invalidates cached (incl. negative) match - // resolutions keyed on the old polluted titles - return `title:${normalizedTitle}|year:${year ?? ''}|v2`; + // resolutions keyed on the old polluted titles. + // v3: the search query stopped being the folded key ("феик" for + // "Фейк"), which TMDB answered with nothing; every negative row recorded + // under v2 for a title with "й"/"ё" is that bug, not a missing title, and + // must not block the retry for its 7-day TTL. Rows are keyed by the + // query since then. + return `title:${searchQueryIdentity(query)}|year:${year ?? ''}|v3`; } export function buildDetailsLookupKey(tmdbId: number): string { @@ -108,9 +156,7 @@ export function parseProviderTmdbId( * name mismatch alone says more about our own inputs than about the id. */ export type ProviderIdVerdict = - | 'corroborated' - | 'contradicted' - | 'inconclusive'; + 'corroborated' | 'contradicted' | 'inconclusive'; /** The same tolerance the search gate uses, applied in reverse */ function yearsAgree( @@ -135,7 +181,11 @@ export function assessProviderId( original_name?: string; first_air_date?: string; }, - query: { title?: string | null; originalTitle?: string | null; year?: number | null }, + query: { + title?: string | null; + originalTitle?: string | null; + year?: number | null; + }, mediaType: TmdbMediaType ): ProviderIdVerdict { // Same effective year the search would use, so the two agree on what @@ -171,7 +221,9 @@ export function detailsMatchProviderTitle( query: { title?: string | null; originalTitle?: string | null } ): boolean { const variants = new Set( - buildSearchTitleVariants(query.title, query.originalTitle) + buildSearchTitleVariants(query.title, query.originalTitle).map( + (variant) => variant.normalized + ) ); if (variants.size === 0) { // Nothing to compare against — never call that a mismatch diff --git a/libs/shared/database/src/lib/connection-migrations.spec.ts b/libs/shared/database/src/lib/connection-migrations.spec.ts index c89ba825b..5a6c115bb 100644 --- a/libs/shared/database/src/lib/connection-migrations.spec.ts +++ b/libs/shared/database/src/lib/connection-migrations.spec.ts @@ -139,8 +139,16 @@ describe('runMigrations error tolerance', () => { }); }); -describe('TMDB search lookup v2 cache cleanup', () => { - it('deletes legacy search rows and records the migration atomically', () => { +describe('TMDB search lookup cache cleanup', () => { + function deleteStatements(sqlite: SqliteHandle): string[] { + return (sqlite.prepare as jest.Mock).mock.calls + .map(([statement]) => compactSql(statement)) + .filter((statement) => + statement.includes('DELETE FROM tmdb_metadata') + ); + } + + it('deletes every retired key generation and records each migration atomically', () => { const deleteRun = jest.fn(); const markerRun = jest.fn(); const { sqlite, transaction } = createSqliteMock([ @@ -151,22 +159,51 @@ describe('TMDB search lookup v2 cache cleanup', () => { cleanupLegacyTmdbSearchCache(sqlite); - expect(transaction).toHaveBeenCalledTimes(1); - expect(deleteRun).toHaveBeenCalledTimes(1); - expect(markerRun).toHaveBeenCalledWith( - 'migration:tmdb-search-lookup-v2-cache-cleanup:v1' - ); - const deleteSql = compactSql( - (sqlite.prepare as jest.Mock).mock.calls.find(([statement]) => - statement.includes('DELETE FROM tmdb_metadata') - )?.[0] - ); - expect(deleteSql).toContain( + expect(transaction).toHaveBeenCalledTimes(2); + expect(deleteRun).toHaveBeenCalledTimes(2); + expect(markerRun.mock.calls.map(([key]) => key)).toEqual([ + 'migration:tmdb-search-lookup-v2-cache-cleanup:v1', + 'migration:tmdb-search-lookup-v3-cache-cleanup:v1', + ]); + const [unversionedDelete, v2Delete] = deleteStatements(sqlite); + expect(unversionedDelete).toContain( "lookup_key LIKE 'title:%|year:%' AND lookup_key NOT LIKE 'title:%|year:%|v%'" ); + // Only v2 search rows: the v3 rows the resolver writes now, and the + // `id:`/`person:`/`badProviderId:` rows, must survive. + expect(v2Delete).toContain("WHERE lookup_key LIKE 'title:%|year:%|v2'"); + expect(v2Delete).not.toContain('v3'); }); - it('does nothing after the migration has completed', () => { + it('runs only the generations that have not completed yet', () => { + const markerRun = jest.fn(); + const { sqlite, transaction } = createSqliteMock([ + [ + 'SELECT value FROM app_state', + { + get: (key: unknown) => + key === + 'migration:tmdb-search-lookup-v2-cache-cleanup:v1' + ? { value: 'done' } + : undefined, + }, + ], + ['INSERT INTO app_state', { run: markerRun }], + ]); + + cleanupLegacyTmdbSearchCache(sqlite); + + expect(transaction).toHaveBeenCalledTimes(1); + expect(markerRun).toHaveBeenCalledTimes(1); + expect(markerRun).toHaveBeenCalledWith( + 'migration:tmdb-search-lookup-v3-cache-cleanup:v1' + ); + expect(deleteStatements(sqlite)).toEqual([ + expect.stringContaining("lookup_key LIKE 'title:%|year:%|v2'"), + ]); + }); + + it('does nothing after every migration has completed', () => { const { sqlite, prepare, transaction } = createSqliteMock([ completedMigrationStateRule, ]); @@ -174,7 +211,8 @@ describe('TMDB search lookup v2 cache cleanup', () => { cleanupLegacyTmdbSearchCache(sqlite); expect(transaction).not.toHaveBeenCalled(); - expect(prepare).toHaveBeenCalledTimes(1); + // One marker read per retired generation, nothing else + expect(prepare).toHaveBeenCalledTimes(2); }); }); diff --git a/libs/shared/database/src/lib/connection.ts b/libs/shared/database/src/lib/connection.ts index a2c5a4158..ff040e9ea 100644 --- a/libs/shared/database/src/lib/connection.ts +++ b/libs/shared/database/src/lib/connection.ts @@ -43,6 +43,8 @@ const EPG_PROGRAM_SOURCE_URL_BACKFILL_MIGRATION_KEY = 'migration:epg-program-source-url-backfill:v1'; const TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY = 'migration:tmdb-search-lookup-v2-cache-cleanup:v1'; +const TMDB_SEARCH_LOOKUP_V3_CACHE_CLEANUP_MIGRATION_KEY = + 'migration:tmdb-search-lookup-v3-cache-cleanup:v1'; const EPG_PROGRAM_SOURCE_URL_BACKFILL_BATCH_SIZE = 50_000; function readTraceFlag(name: string): boolean { @@ -963,50 +965,82 @@ function widenTmdbMetadataMediaTypeCheck(sqliteDb: Database.Database): void { } /** - * Search-match cache keys gained a v2 suffix when title normalization changed. - * Remove the now-unreachable unversioned rows once rather than leaving negative - * resolutions and other legacy search matches in long-lived installations. + * Every search-match cache key generation that has been retired, oldest + * first, each with the predicate selecting exactly the rows written under + * it. A retired generation's rows are unreachable — the resolver only ever + * reads the current key — so they would otherwise sit in long-lived + * installations forever, negative resolutions included. + * + * - unversioned → v2: title normalization learned to strip appended + * language/quality tags. + * - v2 → v3: the search query stopped being the folded comparison key. Under + * v2 every title with a Cyrillic "й"/"ё" was searched folded ("феик" for + * "Фейк", "елки" for "Ёлки"), got no answer, and was cached as missing for + * 7 days. + */ +const LEGACY_TMDB_SEARCH_CACHE_CLEANUPS: ReadonlyArray<{ + migrationKey: string; + rowPredicate: string; +}> = [ + { + migrationKey: TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY, + rowPredicate: `lookup_key LIKE 'title:%|year:%' + AND lookup_key NOT LIKE 'title:%|year:%|v%'`, + }, + { + migrationKey: TMDB_SEARCH_LOOKUP_V3_CACHE_CLEANUP_MIGRATION_KEY, + rowPredicate: `lookup_key LIKE 'title:%|year:%|v2'`, + }, +]; + +/** + * Remove search-match rows written under a retired key generation, once per + * generation, recorded in `app_state`. Each generation is its own marker so + * an installation that skipped a release still runs every cleanup it missed, + * in order. */ function cleanupLegacyTmdbSearchCache(sqliteDb: Database.Database): void { - try { - const migrationState = sqliteDb - .prepare(`SELECT value FROM app_state WHERE key = ?`) - .get(TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY) as - { value?: unknown } | undefined; + for (const cleanup of LEGACY_TMDB_SEARCH_CACHE_CLEANUPS) { + try { + const migrationState = sqliteDb + .prepare(`SELECT value FROM app_state WHERE key = ?`) + .get(cleanup.migrationKey) as { value?: unknown } | undefined; - if (migrationState?.value === 'done') { - return; + if (migrationState?.value === 'done') { + continue; + } + + const executeCleanup = sqliteDb.transaction(() => { + sqliteDb + .prepare( + `DELETE FROM tmdb_metadata + WHERE ${cleanup.rowPredicate}` + ) + .run(); + sqliteDb + .prepare( + `INSERT INTO app_state (key, value, updated_at) + VALUES (?, 'done', datetime('now')) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at` + ) + .run(cleanup.migrationKey); + }); + + executeCleanup(); + } catch (error) { + const message = + typeof error === 'object' && + error !== null && + 'message' in error + ? String((error as { message?: unknown }).message ?? error) + : String(error); + + console.warn( + `Legacy TMDB search cache cleanup failed (continuing): ${message}` + ); } - - const executeCleanup = sqliteDb.transaction(() => { - sqliteDb - .prepare( - `DELETE FROM tmdb_metadata - WHERE lookup_key LIKE 'title:%|year:%' - AND lookup_key NOT LIKE 'title:%|year:%|v%'` - ) - .run(); - sqliteDb - .prepare( - `INSERT INTO app_state (key, value, updated_at) - VALUES (?, 'done', datetime('now')) - ON CONFLICT(key) DO UPDATE SET - value = excluded.value, - updated_at = excluded.updated_at` - ) - .run(TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY); - }); - - executeCleanup(); - } catch (error) { - const message = - typeof error === 'object' && error !== null && 'message' in error - ? String((error as { message?: unknown }).message ?? error) - : String(error); - - console.warn( - `Legacy TMDB search cache cleanup failed (continuing): ${message}` - ); } } diff --git a/libs/shared/database/src/lib/tmdb-search-cache-cleanup.spec.ts b/libs/shared/database/src/lib/tmdb-search-cache-cleanup.spec.ts new file mode 100644 index 000000000..137f453a5 --- /dev/null +++ b/libs/shared/database/src/lib/tmdb-search-cache-cleanup.spec.ts @@ -0,0 +1,157 @@ +import { execFileSync } from 'node:child_process'; +import { createRequire } from 'node:module'; +import { resolve } from 'node:path'; +import { pathToFileURL } from 'node:url'; + +/** + * Real-SQLite coverage for the retired search-key cleanups, run inside + * Electron so better-sqlite3 links against the ABI the app ships with. The + * mock-based spec proves the SQL shape; this one proves what a persisted + * database looks like after an upgrade — from a release that skipped every + * cleanup, from the previous release, on a fresh database, and on the next + * startup after that. + */ +it('drops only retired search rows across skipped, previous, pre-person, fresh and repeated startups', () => { + const electron = createRequire(__filename)('electron') as string; + const connectionUrl = pathToFileURL( + resolve(__dirname, 'connection.ts') + ).href; + const result = execFileSync( + electron, + [ + '--import', + 'tsx', + '--eval', + ` + const { default: Database } = await import('better-sqlite3'); + const { __databaseConnectionTestHooks: hooks } = await import(${JSON.stringify(connectionUrl)}); + console.log = () => undefined; + const V2_MARKER = 'migration:tmdb-search-lookup-v2-cache-cleanup:v1'; + const V3_MARKER = 'migration:tmdb-search-lookup-v3-cache-cleanup:v1'; + const ROWS = [ + ['tv', 'title:феик|year:2026', 'ru-RU', null], + ['tv', 'title:феик|year:2026|v2', 'ru-RU', null], + ['tv', 'title:the boys|year:2019|v2', 'en-US', 76479], + ['tv', 'title:фейк|year:2026|v3', 'ru-RU', 317869], + ['tv', 'id:317869|v2', 'ru-RU', 317869], + ['tv', 'id:317869|season:1', 'ru-RU', 317869], + ['person', 'person:287', 'en-US', 287], + ['movie', 'badProviderId:999', 'any', null], + ['movie', 'trending:week', 'en-US', null], + ]; + function openDb(markers) { + const db = new Database(':memory:'); + hooks.createTables(db); + const insert = db.prepare('INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id, payload) VALUES (?, ?, ?, ?, ?)'); + for (const [type, key, lang, id] of ROWS) insert.run(type, key, lang, id, id === null ? null : '{"id":' + id + '}'); + const mark = db.prepare("INSERT INTO app_state (key, value, updated_at) VALUES (?, 'done', datetime('now'))"); + for (const marker of markers) mark.run(marker); + return db; + } + function snapshot(db) { + return { + keys: db.prepare('SELECT lookup_key FROM tmdb_metadata ORDER BY lookup_key').all().map((r) => r.lookup_key), + payloads: db.prepare("SELECT payload FROM tmdb_metadata WHERE lookup_key LIKE 'id:%' ORDER BY lookup_key").all().map((r) => r.payload), + markers: db.prepare("SELECT key FROM app_state WHERE key LIKE 'migration:tmdb-search-lookup-%' AND value = 'done' ORDER BY key").all().map((r) => r.key), + }; + } + // Skipped every release since the unversioned keys: both cleanups run + // through the real initialization path + const skipped = openDb([]); + hooks.runMigrations(skipped); + const skippedAfter = snapshot(skipped); + // Next startup: a row written meanwhile under the current key survives + skipped.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:гудовы|year:2026|v3', 'ru-RU', 318894)").run(); + hooks.runMigrations(skipped); + const repeated = snapshot(skipped); + // Previous release: the unversioned cleanup already ran; only v2 rows go + const previous = openDb([V2_MARKER]); + hooks.runMigrations(previous); + const previousAfter = snapshot(previous); + hooks.runMigrations(previous); + const previousRepeated = snapshot(previous); + // Oldest historical schema: the pre-'person' CHECK. That migration + // rebuilds the pure-cache table empty by design; the cleanups must + // still record their markers on the rebuilt table and stay idempotent. + const prePerson = new Database(':memory:'); + hooks.createTables(prePerson); + prePerson.exec("DROP TABLE tmdb_metadata; CREATE TABLE tmdb_metadata (id INTEGER PRIMARY KEY AUTOINCREMENT, media_type TEXT NOT NULL CHECK (media_type IN ('movie', 'tv')), lookup_key TEXT NOT NULL, language TEXT NOT NULL, tmdb_id INTEGER, payload TEXT, fetched_at TEXT DEFAULT (datetime('now'))); CREATE UNIQUE INDEX tmdb_metadata_lookup_unique ON tmdb_metadata(media_type, lookup_key, language)"); + prePerson.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:феик|year:2026', 'ru-RU', NULL)").run(); + hooks.runMigrations(prePerson); + const prePersonAfter = { ...snapshot(prePerson), check: prePerson.prepare("SELECT sql FROM sqlite_master WHERE name = 'tmdb_metadata'").get().sql.includes("'person'") }; + hooks.runMigrations(prePerson); + const prePersonRepeated = snapshot(prePerson); + // Fresh database through the real initialization path + const fresh = new Database(':memory:'); + hooks.createTables(fresh); + hooks.runMigrations(fresh); + const freshAfter = snapshot(fresh); + hooks.runMigrations(fresh); + const freshRepeated = snapshot(fresh); + process.stdout.write(JSON.stringify({ skippedAfter, repeated, previousAfter, previousRepeated, prePersonAfter, prePersonRepeated, freshAfter, freshRepeated })); + `, + ], + { + cwd: process.cwd(), + encoding: 'utf8', + env: { + ...process.env, + ELECTRON_RUN_AS_NODE: '1', + TSX_TSCONFIG_PATH: resolve(process.cwd(), 'tsconfig.base.json'), + }, + } + ); + + const V2_MARKER = 'migration:tmdb-search-lookup-v2-cache-cleanup:v1'; + const V3_MARKER = 'migration:tmdb-search-lookup-v3-cache-cleanup:v1'; + const survivors = [ + 'badProviderId:999', + 'id:317869|season:1', + 'id:317869|v2', + 'person:287', + 'title:фейк|year:2026|v3', + 'trending:week', + ]; + const detailsPayloads = ['{"id":317869}', '{"id":317869}']; + + expect(JSON.parse(result)).toEqual({ + skippedAfter: { + keys: survivors, + payloads: detailsPayloads, + markers: [V2_MARKER, V3_MARKER], + }, + repeated: { + keys: [...survivors, 'title:гудовы|year:2026|v3'].sort(), + payloads: detailsPayloads, + markers: [V2_MARKER, V3_MARKER], + }, + previousAfter: { + // The unversioned row is that generation's business, already done + keys: [...survivors, 'title:феик|year:2026'].sort(), + payloads: detailsPayloads, + markers: [V2_MARKER, V3_MARKER], + }, + previousRepeated: { + keys: [...survivors, 'title:феик|year:2026'].sort(), + payloads: detailsPayloads, + markers: [V2_MARKER, V3_MARKER], + }, + prePersonAfter: { + keys: [], + payloads: [], + markers: [V2_MARKER, V3_MARKER], + check: true, + }, + prePersonRepeated: { + keys: [], + payloads: [], + markers: [V2_MARKER, V3_MARKER], + }, + freshAfter: { keys: [], payloads: [], markers: [V2_MARKER, V3_MARKER] }, + freshRepeated: { + keys: [], + payloads: [], + markers: [V2_MARKER, V3_MARKER], + }, + }); +}); diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts index 1345f10c3..ee31eef2f 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts @@ -1,4 +1,5 @@ import { + cleanTitleForSearch, normalizeTitle, normalizeTitleKeys, titleYearsCompatible, @@ -429,3 +430,80 @@ describe('titleYearsCompatible', () => { expect(titleYearsCompatible(1982, 2049)).toBe(false); }); }); + +describe('cleanTitleForSearch', () => { + it('keeps Cyrillic letters that folding would rewrite', () => { + // NFD splits "й" into "и" + a breve and "ё" into "е" + a diaeresis; + // the comparison key drops both marks, and TMDB answers the folded + // spelling with nothing (issue: "Фейк (10 серий)" never matched). + expect(normalizeTitle('Фейк (10 серий)')).toBe('феик'); + expect(cleanTitleForSearch('Фейк (10 серий)')).toBe('Фейк'); + expect(cleanTitleForSearch('Ёлки 2010')).toBe('Ёлки'); + expect(cleanTitleForSearch('Волшебный участок s02')).toBe( + 'Волшебный участок' + ); + }); + + it('keeps Arabic hamza forms that folding splits into two words', () => { + // "أ" decomposes into a bare alef + U+0654, which is outside the + // stripped mark range and so becomes a SPACE in the key. TMDB finds + // "أطرق بابي" and nothing for "ا طرق بابي". + expect(normalizeTitle('AR| أطرق بابي')).toBe('ا طرق بابي'); + expect(cleanTitleForSearch('AR| أطرق بابي')).toBe('أطرق بابي'); + }); + + it('keeps Latin diacritics and casing', () => { + expect(cleanTitleForSearch('Amélie (2001)')).toBe('Amélie'); + expect(cleanTitleForSearch('THE LAST OF US')).toBe('THE LAST OF US'); + }); + + it('recomposes decomposed input and keeps marks attached to their letter', () => { + // Providers ship "Dönüş" as "o" + U+0308; the key folds it away but + // the query must not split the word on the combining mark. + expect( + cleanTitleForSearch( + '1942\u2032ye Do\u0308nu\u0308s\u0327 (2012) TR' + ) + ).toBe('1942 ye Dönüş'); + expect( + normalizeTitle('1942\u2032ye Do\u0308nu\u0308s\u0327 (2012) TR') + ).toBe('1942 ye donus'); + // No precomposed form exists for "ọ̀": the mark stays on the letter + expect(cleanTitleForSearch('AR - Ìjọ̀gbọ̀n (2023)')).toBe('Ìjọ̀gbọ̀n'); + expect(normalizeTitle('AR - Ìjọ̀gbọ̀n (2023)')).toBe('ijogbon'); + }); + + it('applies the same tag, bracket, season and year stripping as the key', () => { + expect(cleanTitleForSearch('EN - The Matrix (1999) 4K')).toBe( + 'The Matrix' + ); + expect(cleanTitleForSearch('|DE| Breaking Bad-DE')).toBe( + 'Breaking Bad' + ); + expect(cleanTitleForSearch('The Boys s05')).toBe('The Boys'); + expect(cleanTitleForSearch('Пацаны 2 сезон')).toBe('Пацаны'); + expect(cleanTitleForSearch('The Matrix 1999')).toBe('The Matrix'); + // A leading token that is the film's own name survives on both tiers + expect(cleanTitleForSearch('AKA - 2023')).toBe('AKA'); + expect(normalizeTitle('AKA - 2023')).toBe('aka'); + }); + + it('normalizes to the same key as the text it was derived from', () => { + for (const raw of [ + 'Фейк (10 серий)', + 'EN - The Matrix (1999) 4K', + 'Ёлки 2010', + 'Amélie (2001)', + '|FR|VO|Le dernier empereur', + ]) { + expect(normalizeTitle(cleanTitleForSearch(raw))).toBe( + normalizeTitle(raw) + ); + } + }); + + it('returns an empty string for empty input', () => { + expect(cleanTitleForSearch('')).toBe(''); + expect(cleanTitleForSearch(null)).toBe(''); + }); +}); diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.ts b/libs/shared/interfaces/src/lib/title-normalization.util.ts index ef1ba7522..89b0b71bb 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.ts @@ -302,13 +302,13 @@ const HAS_LETTER = /\p{L}/u; * "2023" and were offered to each other as alternative sources. A miss beats * a wrong match, so an unknown token keeps its title. */ -function normalizeAfterLeadingTag(value: string): string { +function normalizeAfterLeadingTag(value: string, fold: boolean): string { const match = value.match(LANGUAGE_PREFIX); if (!match) { - return normalizeRest(value); + return normalizeRest(value, fold); } - const stripped = normalizeRest(value.replace(LANGUAGE_PREFIX, '')); + const stripped = normalizeRest(value.replace(LANGUAGE_PREFIX, ''), fold); // Deliberately not `stripSeason`: its "never return empty" fallback would // report a lone season marker as a surviving word. if (HAS_LETTER.test(stripped.replace(SEASON_SUFFIX_PATTERN, ''))) { @@ -316,7 +316,7 @@ function normalizeAfterLeadingTag(value: string): string { } const token = match[0].replace(PREFIX_SEPARATOR_TAIL, ''); - return isKnownPrefixTag(token) ? stripped : normalizeRest(value); + return isKnownPrefixTag(token) ? stripped : normalizeRest(value, fold); } const DOUBLE_DASH_SUFFIX = /[-–]{2}[A-Za-z]{2,5}\s*$/; @@ -438,28 +438,49 @@ const SEASON_SUFFIX_PATTERN = new RegExp( * actually produce instead of predicting it. Cheap enough to run twice, * because the second run only happens for a title whose stripped form came * out with no word in it at all. + * + * `fold` selects the letter folding (diacritics, case, sigma). It is ON for + * every comparison key and OFF for the text sent to a remote search — see + * `cleanTitleForSearch`. Folding is lossy for scripts whose "diacritics" are + * distinct letters: NFD turns Cyrillic "й" into "и" + a combining breve, and + * dropping the breve rewrites "Фейк" as "феик", "ё" as "е". Two provider + * copies of a title still meet on that key, which is all a comparison needs, + * but TMDB's search does not fold Cyrillic the same way and answers a folded + * query with nothing at all. */ -function normalizeRest(value: string): string { - return ( - stripTrailingTags(value) - .normalize('NFD') - .replace(/[̀-ͯ]/g, '') - .toLowerCase() - // Greek Σ has two lowercase forms and `toLowerCase` picks by - // position: "ΑΣ" becomes "ας" while an already-lowercase "ασ" - // stays medial, so the same word reaches this line spelled two - // ways. Both SQL tiers fold them together — SQLite's trigram - // tokenizer does it natively, and the scan's GLOB classes do it in - // `caseInsensitiveGlobPattern` — so without this the candidate is - // admitted by the query and then thrown away by the confirmation. - // Folding to the medial form is what Unicode case folding does. - .replace(/ς/g, 'σ') - .replace(/[^\p{L}\p{N}]+/gu, ' ') - .split(' ') - .filter((token) => token !== '' && !QUALITY_TAGS.has(token)) - .join(' ') - .trim() - ); +function normalizeRest(value: string, fold: boolean): string { + const withoutTrailingTags = stripTrailingTags(value); + const folded = fold + ? withoutTrailingTags + .normalize('NFD') + .replace(/[̀-ͯ]/g, '') + .toLowerCase() + // Greek Σ has two lowercase forms and `toLowerCase` picks by + // position: "ΑΣ" becomes "ας" while an already-lowercase "ασ" + // stays medial, so the same word reaches this line spelled two + // ways. Both SQL tiers fold them together — SQLite's trigram + // tokenizer does it natively, and the scan's GLOB classes do + // it in `caseInsensitiveGlobPattern` — so without this the + // candidate is admitted by the query and then thrown away by + // the confirmation. Folding to the medial form is what Unicode + // case folding does. + .replace(/ς/g, 'σ') + : // Providers ship some titles decomposed ("o" + U+0308 for "ö"). + // Recompose so the query reads as the provider meant it, and + // keep any mark that has no precomposed form ("ọ̀") attached to + // its letter instead of letting the word-splitting step below + // turn it into a space inside the word. + withoutTrailingTags.normalize('NFC'); + const nonWord = fold ? /[^\p{L}\p{N}]+/gu : /[^\p{L}\p{N}\p{M}]+/gu; + + return folded + .replace(nonWord, ' ') + .split(' ') + .filter( + (token) => token !== '' && !QUALITY_TAGS.has(token.toLowerCase()) + ) + .join(' ') + .trim(); } /** @@ -488,6 +509,27 @@ export interface NormalizedTitleKeys { export function normalizeTitleKeys( raw: string | null | undefined +): NormalizedTitleKeys { + return buildTitleKeys(raw, true); +} + +/** + * The title to SEND to a remote search such as TMDB: the same tag, bracket, + * season and year stripping as `normalizeTitle`, but with the letters left + * exactly as the provider wrote them — no diacritic folding, no lowercasing. + * + * Comparison keys must fold so two spellings of one film meet; a search + * query must not, because the search engine folds by its own rules and a + * pre-folded Cyrillic query ("феик" for "Фейк") matches nothing there. + * Compare the results with `normalizeTitle`, never with this. + */ +export function cleanTitleForSearch(raw: string | null | undefined): string { + return buildTitleKeys(raw, false).base; +} + +function buildTitleKeys( + raw: string | null | undefined, + fold: boolean ): NormalizedTitleKeys { if (!raw) { return { exact: '', base: '', trailingYear: null }; @@ -498,7 +540,8 @@ export function normalizeTitleKeys( .replace(WRAPPED_TAG_PREFIX, '') // Inner classes exclude the opening delimiter too, so runaway // inputs like "[[[[[..." backtrack linearly (CodeQL js/polynomial-redos) - .replace(/\[[^\][]*\]|\([^()]*\)|\{[^{}]*\}/g, ' ') + .replace(/\[[^\][]*\]|\([^()]*\)|\{[^{}]*\}/g, ' '), + fold ); const exact = stripSeason(cleaned);