diff --git a/apps/electron-backend/src/app/database/operations/content-search.util.sqlite.spec.ts b/apps/electron-backend/src/app/database/operations/content-search.util.sqlite.spec.ts index 24115f65c..b732fd191 100644 --- a/apps/electron-backend/src/app/database/operations/content-search.util.sqlite.spec.ts +++ b/apps/electron-backend/src/app/database/operations/content-search.util.sqlite.spec.ts @@ -30,7 +30,7 @@ const TITLES = [ 'Şan Ve Şeref', 'Первый канал HD', 'ПЕРВЫЙ КАНАЛ', - 'Ёлки 1914', + 'Ёжик ТВ', 'Йога для всех', 'Матч ТВ', 'Ελλάδα Σήμερα', @@ -179,7 +179,7 @@ describe('content-search.util against SQLite', () => { it.each([ ['Первый', 'первый', ['Первый канал HD', 'ПЕРВЫЙ КАНАЛ']], ['ПЕРВ', 'перв', ['Первый канал HD', 'ПЕРВЫЙ КАНАЛ']], - ['Ёлки', 'ёлки', ['Ёлки 1914']], + ['Ёжик', 'ёжик', ['Ёжик ТВ']], ['Йога', 'йога', ['Йога для всех']], ['Ма', 'ма', ['Матч ТВ']], ['ΕΛΛ', 'ελλ', ['Ελλάδα Σήμερα']], diff --git a/docs/architecture/tmdb-metadata-enrichment.md b/docs/architecture/tmdb-metadata-enrichment.md index e4772b125..94f64cea8 100644 --- a/docs/architecture/tmdb-metadata-enrichment.md +++ b/docs/architecture/tmdb-metadata-enrichment.md @@ -537,8 +537,9 @@ identity is the query with only its case removed (`searchQueryIdentity`): variants are deduplicated by it and every attempted variant is cached under its own key (`title:|year:|v4`, in the language that variant was searched in), never by the folded key and never only under the first -variant — "Феик" and "Фейк" fold to one key but are different searches with -different answers, so a verdict for one must not be read back for the other; +variant — "Леика" and "Лейка", an illustrative pair, fold to one key while +staying two different searches that can get different answers, so a verdict +for one must not be read back for the other; a misspelled original title must not swallow the display title that TMDB actually knows; and two items that share an original title but not a display title walk different variant lists, so a row keyed on the first variant alone @@ -547,11 +548,13 @@ splits Cyrillic "й" into "и" + a combining breve and "ё" into "е" + a diaeresis, and Arabic hamza forms ("أ") into a bare alef + a combining hamza that the punctuation step then turns into a space inside the word. The key drops or splits on those marks, and TMDB's `/search` does not fold them the -same way — a query of `феик` returns zero results while `Фейк` returns the -show. Under the old single-form design every Russian title with "й"/"ё" -("Фейк (10 серий)", "Волшебный участок", "Молодой Шерлок") was searched -folded, missed, and cached as missing for the 7-day negative TTL. Compare -results only through `normalized`; never send it over the wire. +same way, so a folded query matches nothing there. Under the old single-form +design that hit every Russian title carrying "й" or "ё" and every Arabic +title carrying a hamza form: each was searched folded, missed, and cached as +missing for the 7-day negative TTL. The Cyrillic and Arabic strings used +throughout this section are illustrative stand-ins chosen to fold the same +way, not the titles the failures were observed on. Compare results only +through `normalized`; never send it over the wire. Electron IPC path (follows the standard DB worker contract, see [SQLite DB Worker](./sqlite-db-worker.md)): diff --git a/libs/portal/stalker/data-access/src/lib/stalker-series.adapters.spec.ts b/libs/portal/stalker/data-access/src/lib/stalker-series.adapters.spec.ts index 3fe42a96f..88d73c2da 100644 --- a/libs/portal/stalker/data-access/src/lib/stalker-series.adapters.spec.ts +++ b/libs/portal/stalker/data-access/src/lib/stalker-series.adapters.spec.ts @@ -21,10 +21,10 @@ describe('stalker-series.adapters', () => { 'The Gentlemen season2', 'The Gentlemen (s02)', 'The Gentlemen (season 2)', - 'Олдскул (2 сезон)', - 'Олдскул (сезон 2)', - 'Олдскул s02', - 'Олдскул season2', + 'Пример (2 сезон)', + 'Пример (сезон 2)', + 'Пример s02', + 'Пример season2', ])( 'uses the title season for lazy VOD %s without changing tracking IDs', (title) => { diff --git a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts index 6b8908004..fec3034e9 100644 --- a/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts +++ b/libs/services/src/lib/tmdb/tmdb-id-resolver.service.spec.ts @@ -9,20 +9,21 @@ import { TmdbSearchResult } from './tmdb.types'; * Regression coverage for the search wire format. The comparison key folds * diacritics — Cyrillic "й" becomes "и" — and that key used to be sent as * the TMDB query, which matched nothing for any title carrying "й"/"ё" and - * cached the miss for a week ("Фейк (10 серий)", "Волшебный участок"). + * cached the miss for a week. The titles below are illustrative stand-ins + * that fold the same way, not the ones the failures were observed on. */ describe('TmdbIdResolverService.resolveBySearch', () => { - const fake2026: TmdbSearchResult = { - id: 317869, - name: 'Фейк', - original_name: 'Фейк', + const series2026: TmdbSearchResult = { + id: 101101, + name: 'Лейка', + original_name: 'Лейка', first_air_date: '2026-07-16', vote_count: 3, }; - const fake2024: TmdbSearchResult = { - id: 322696, - name: 'Фейк', - original_name: 'Фейк', + const series2024: TmdbSearchResult = { + id: 101102, + name: 'Лейка', + original_name: 'Лейка', first_air_date: '2024-12-05', vote_count: 0, }; @@ -67,7 +68,7 @@ describe('TmdbIdResolverService.resolveBySearch', () => { beforeEach(() => { searchTv = jest.fn(async (query: string) => // TMDB does not fold Cyrillic: only the provider spelling hits - query === 'Фейк' ? [fake2026, fake2024] : [] + query === 'Лейка' ? [series2026, series2024] : [] ); searchMovie = jest.fn().mockResolvedValue([]); cacheGet = jest.fn().mockResolvedValue(null); @@ -78,18 +79,18 @@ describe('TmdbIdResolverService.resolveBySearch', () => { const service = createService(); const id = await service.resolveBySearch('tv', { - title: 'Фейк (10 серий)', + title: 'Лейка (10 серий)', year: 2026, }); - expect(id).toBe(317869); + expect(id).toBe(101101); expect(searchTv).toHaveBeenCalledTimes(1); - expect(searchTv).toHaveBeenCalledWith('Фейк', null, 'ru-RU', 'key'); + expect(searchTv).toHaveBeenCalledWith('Лейка', null, 'ru-RU', 'key'); expect(cacheSet).toHaveBeenCalledWith({ mediaType: 'tv', - lookupKey: 'title:фейк|year:2026|v4', + lookupKey: 'title:лейка|year:2026|v4', language: 'ru-RU', - tmdbId: 317869, + tmdbId: 101101, payload: null, }); }); @@ -99,41 +100,41 @@ describe('TmdbIdResolverService.resolveBySearch', () => { const service = createService(); const id = await service.resolveBySearch('tv', { - title: 'Молодой Шерлок', + title: 'Пробный Выпуск', year: 2026, }); expect(id).toBeNull(); expect(searchTv).toHaveBeenCalledWith( - 'Молодой Шерлок', + 'Пробный Выпуск', null, 'ru-RU', 'key' ); expect(cacheSet).toHaveBeenCalledWith( expect.objectContaining({ - lookupKey: 'title:молодой шерлок|year:2026|v4', + lookupKey: 'title:пробный выпуск|year:2026|v4', tmdbId: null, }) ); }); it('reads the cache before searching', async () => { - cacheGet.mockResolvedValue({ tmdbId: 317869 }); + cacheGet.mockResolvedValue({ tmdbId: 101101 }); const service = createService(); const cacheService = (service as unknown as { cache: TmdbCacheService }) .cache; jest.spyOn(cacheService, 'isFresh').mockReturnValue(true); const id = await service.resolveBySearch('tv', { - title: 'Фейк (10 серий)', + title: 'Лейка (10 серий)', year: 2026, }); - expect(id).toBe(317869); + expect(id).toBe(101101); expect(cacheGet).toHaveBeenCalledWith( 'tv', - 'title:фейк|year:2026|v4', + 'title:лейка|year:2026|v4', 'ru-RU' ); expect(searchTv).not.toHaveBeenCalled(); @@ -143,31 +144,31 @@ describe('TmdbIdResolverService.resolveBySearch', () => { const service = createService(); const id = await service.resolveBySearch('tv', { - title: 'Фейк', - originalTitle: 'Феик', + title: 'Лейка', + originalTitle: 'Леика', year: 2026, }); - expect(id).toBe(317869); + expect(id).toBe(101101); expect(searchTv.mock.calls.map(([query]) => query)).toEqual([ - 'Феик', - 'Фейк', + 'Леика', + 'Лейка', ]); // Each attempted variant records its own verdict expect( cacheSet.mock.calls.map(([row]) => [row.lookupKey, row.tmdbId]) ).toEqual([ - ['title:феик|year:2026|v4', null], - ['title:фейк|year:2026|v4', 317869], + ['title:леика|year:2026|v4', null], + ['title:лейка|year:2026|v4', 101101], ]); }); it("does not let one variant's cached verdict answer for another", async () => { - // Item A (original "Феик", display "Феик") cached a miss under - // "феик". Item B shares the original title but displays "Фейк": + // Item A (original "Леика", display "Леика") cached a miss under + // "леика". Item B shares the original title but displays "Лейка": // the cached miss must not suppress B's own second variant. cacheGet.mockImplementation(async (_type: string, key: string) => - key === 'title:феик|year:2026|v4' ? { tmdbId: null } : null + key === 'title:леика|year:2026|v4' ? { tmdbId: null } : null ); const service = createService(); const cacheService = (service as unknown as { cache: TmdbCacheService }) @@ -177,17 +178,17 @@ describe('TmdbIdResolverService.resolveBySearch', () => { ); const id = await service.resolveBySearch('tv', { - title: 'Фейк', - originalTitle: 'Феик', + title: 'Лейка', + originalTitle: 'Леика', year: 2026, }); - expect(id).toBe(317869); + expect(id).toBe(101101); expect(searchTv).toHaveBeenCalledTimes(1); - expect(searchTv).toHaveBeenCalledWith('Фейк', null, 'ru-RU', 'key'); + expect(searchTv).toHaveBeenCalledWith('Лейка', null, 'ru-RU', 'key'); expect(cacheGet.mock.calls.map(([, key]) => key)).toEqual([ - 'title:феик|year:2026|v4', - 'title:фейк|year:2026|v4', + 'title:леика|year:2026|v4', + 'title:лейка|year:2026|v4', ]); }); diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts index d70d232bd..7a1ac086f 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.spec.ts @@ -88,14 +88,14 @@ describe('lookup keys', () => { }); it('keys search rows by the wire spelling, not the folded key', () => { - // "Феик" and "Фейк" fold to one comparison key but are different + // "Леика" and "Лейка" fold to one comparison key but are different // searches with different answers; a verdict cached for one must // never be read back for the other. - expect(buildSearchLookupKey('Фейк', 2026)).toBe( - 'title:фейк|year:2026|v4' + expect(buildSearchLookupKey('Лейка', 2026)).toBe( + 'title:лейка|year:2026|v4' ); - expect(buildSearchLookupKey('Феик', 2026)).not.toBe( - buildSearchLookupKey('Фейк', 2026) + expect(buildSearchLookupKey('Леика', 2026)).not.toBe( + buildSearchLookupKey('Лейка', 2026) ); expect(buildSearchLookupKey('THE BOYS', 2019)).toBe( buildSearchLookupKey('The Boys', 2019) @@ -139,8 +139,8 @@ describe('buildSearchTitleVariants', () => { it('sends the provider spelling to the search but compares on the folded key', () => { // The folded key rewrites "й" as "и"; TMDB finds nothing for it. - expect(buildSearchTitleVariants('Фейк (10 серий)', null)).toEqual([ - { query: 'Фейк', normalized: 'феик' }, + expect(buildSearchTitleVariants('Лейка (10 серий)', null)).toEqual([ + { query: 'Лейка', normalized: 'леика' }, ]); // Arabic hamza forms fold into a space inside the word expect(buildSearchTitleVariants('إيمان', null)).toEqual([ @@ -160,12 +160,12 @@ describe('buildSearchTitleVariants', () => { }); it('keeps spellings that fold to one key but differ on the wire', () => { - // A misspelled original title must not swallow the display title: - // TMDB knows "Фейк" and not "Феик", and only the second variant - // would find it. - expect(buildSearchTitleVariants('Фейк', 'Феик')).toEqual([ - { query: 'Феик', normalized: 'феик' }, - { query: 'Фейк', normalized: 'феик' }, + // A misspelled original title must not swallow the display title. + // Take a pair where only the display spelling is the one TMDB + // indexes: only the second variant can find it. + expect(buildSearchTitleVariants('Лейка', 'Леика')).toEqual([ + { query: 'Леика', normalized: 'леика' }, + { query: 'Лейка', normalized: 'леика' }, ]); expect(buildSearchTitleVariants('Amelie', 'Amélie')).toEqual([ { query: 'Amélie', normalized: 'amelie' }, diff --git a/libs/services/src/lib/tmdb/tmdb-matcher.ts b/libs/services/src/lib/tmdb/tmdb-matcher.ts index e21fb0cee..6c966bb5b 100644 --- a/libs/services/src/lib/tmdb/tmdb-matcher.ts +++ b/libs/services/src/lib/tmdb/tmdb-matcher.ts @@ -40,8 +40,8 @@ function stripLeadingLanguageToken(raw: string): string | null { /** * One search candidate: what to SEND to TMDB and what to COMPARE its * answers against. The two differ on purpose — see `cleanTitleForSearch`: - * a folded query ("феик") finds nothing on TMDB while the folded key is - * exactly what the confidence gate and the cache need. + * folding rewrites letters TMDB matches on, so the folded form finds nothing + * there, while it is exactly what the confidence gate and the cache need. */ export interface SearchTitleVariant { /** Provider spelling with tags/brackets/season/year stripped */ @@ -53,7 +53,8 @@ export interface SearchTitleVariant { /** * The identity of one search on the wire: the query with only the case * removed, since TMDB matches case-insensitively and nothing else about - * the spelling may be folded away — "Феик" and "Фейк" are different + * the spelling may be folded away — "Леика" and "Лейка" (illustrative) are + * different * searches with different answers, however alike their comparison keys. * Both the variant deduplication and the cache row use this, so a cached * verdict can never be read back for a search that was never sent. @@ -116,11 +117,11 @@ export function buildSearchLookupKey( // v2: normalizeTitleKeys learned to strip appended language/quality // tags; the version suffix invalidates cached (incl. negative) match // resolutions keyed on the old polluted titles. - // v3: the search query stopped being the folded key ("феик" for - // "Фейк"), which TMDB answered with nothing; every negative row recorded - // under v2 for a title with "й"/"ё" is that bug, not a missing title, and - // must not block the retry for its 7-day TTL. Rows are keyed by the - // query since then. + // v3: the search query stopped being the folded key — the fold rewrites + // "й" as "и" ("леика" for "Лейка", to illustrate) and TMDB answers that + // spelling with nothing; every negative row recorded under v2 for a title + // with "й"/"ё" is that bug, not a missing title, and must not block the + // retry for its 7-day TTL. Rows are keyed by the query since then. // v4: year evidence is tiered (see `yearEvidenceTier`), so every v3 row // resolved by popularity across tiers may name the wrong show — and a // positive row stays fresh for 30 days. diff --git a/libs/shared/database/src/lib/connection.ts b/libs/shared/database/src/lib/connection.ts index 543f92296..efc7a43f7 100644 --- a/libs/shared/database/src/lib/connection.ts +++ b/libs/shared/database/src/lib/connection.ts @@ -976,9 +976,9 @@ function widenTmdbMetadataMediaTypeCheck(sqliteDb: Database.Database): void { * - unversioned → v2: title normalization learned to strip appended * language/quality tags. * - v2 → v3: the search query stopped being the folded comparison key. Under - * v2 every title with a Cyrillic "й"/"ё" was searched folded ("феик" for - * "Фейк", "елки" for "Ёлки"), got no answer, and was cached as missing for - * 7 days. + * v2 every title with a Cyrillic "й"/"ё" was searched folded — the fold + * spells them "и" and "е", as in the illustrative "леика" for "Лейка" — + * got no answer, and was cached as missing for 7 days. * - v3 → v4: year evidence became tiered. Under v3 a series admitted only by * the "premiered earlier" tolerance competed with an exact-year match on * popularity alone, so a new series resolved to its older, better-known diff --git a/libs/shared/database/src/lib/tmdb-search-cache-cleanup.spec.ts b/libs/shared/database/src/lib/tmdb-search-cache-cleanup.spec.ts index 753984808..44352764e 100644 --- a/libs/shared/database/src/lib/tmdb-search-cache-cleanup.spec.ts +++ b/libs/shared/database/src/lib/tmdb-search-cache-cleanup.spec.ts @@ -30,13 +30,13 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a const V3_MARKER = 'migration:tmdb-search-lookup-v3-cache-cleanup:v1'; const V4_MARKER = 'migration:tmdb-search-lookup-v4-cache-cleanup:v1'; const ROWS = [ - ['tv', 'title:феик|year:2026', 'ru-RU', null], - ['tv', 'title:феик|year:2026|v2', 'ru-RU', null], + ['tv', 'title:леика|year:2026', 'ru-RU', null], + ['tv', 'title:леика|year:2026|v2', 'ru-RU', null], ['tv', 'title:the boys|year:2019|v2', 'en-US', 76479], - ['tv', 'title:фейк|year:2026|v3', 'ru-RU', 317869], + ['tv', 'title:лейка|year:2026|v3', 'ru-RU', 101101], ['tv', 'title:nightfall|year:2026|v4', 'ru-RU', 424242], - ['tv', 'id:317869|v2', 'ru-RU', 317869], - ['tv', 'id:317869|season:1', 'ru-RU', 317869], + ['tv', 'id:101101|v2', 'ru-RU', 101101], + ['tv', 'id:101101|season:1', 'ru-RU', 101101], ['person', 'person:287', 'en-US', 287], ['movie', 'badProviderId:999', 'any', null], ['movie', 'trending:week', 'en-US', null], @@ -63,7 +63,7 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a hooks.runMigrations(skipped); const skippedAfter = snapshot(skipped); // Next startup: a row written meanwhile under the current key survives - skipped.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:гудовы|year:2026|v4', 'ru-RU', 318894)").run(); + skipped.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:сосенка|year:2026|v4', 'ru-RU', 101103)").run(); hooks.runMigrations(skipped); const repeated = snapshot(skipped); // Previous release: the earlier cleanups already ran; only v3 rows go @@ -78,7 +78,7 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a const prePerson = new Database(':memory:'); hooks.createTables(prePerson); prePerson.exec("DROP TABLE tmdb_metadata; CREATE TABLE tmdb_metadata (id INTEGER PRIMARY KEY AUTOINCREMENT, media_type TEXT NOT NULL CHECK (media_type IN ('movie', 'tv')), lookup_key TEXT NOT NULL, language TEXT NOT NULL, tmdb_id INTEGER, payload TEXT, fetched_at TEXT DEFAULT (datetime('now'))); CREATE UNIQUE INDEX tmdb_metadata_lookup_unique ON tmdb_metadata(media_type, lookup_key, language)"); - prePerson.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:феик|year:2026', 'ru-RU', NULL)").run(); + prePerson.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:леика|year:2026', 'ru-RU', NULL)").run(); hooks.runMigrations(prePerson); const prePersonAfter = { ...snapshot(prePerson), check: prePerson.prepare("SELECT sql FROM sqlite_master WHERE name = 'tmdb_metadata'").get().sql.includes("'person'") }; hooks.runMigrations(prePerson); @@ -109,13 +109,13 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a const V4_MARKER = 'migration:tmdb-search-lookup-v4-cache-cleanup:v1'; const survivors = [ 'badProviderId:999', - 'id:317869|season:1', - 'id:317869|v2', + 'id:101101|season:1', + 'id:101101|v2', 'person:287', 'title:nightfall|year:2026|v4', 'trending:week', ]; - const detailsPayloads = ['{"id":317869}', '{"id":317869}']; + const detailsPayloads = ['{"id":101101}', '{"id":101101}']; expect(JSON.parse(result)).toEqual({ skippedAfter: { @@ -124,7 +124,7 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a markers: [V2_MARKER, V3_MARKER, V4_MARKER], }, repeated: { - keys: [...survivors, 'title:гудовы|year:2026|v4'].sort(), + keys: [...survivors, 'title:сосенка|year:2026|v4'].sort(), payloads: detailsPayloads, markers: [V2_MARKER, V3_MARKER, V4_MARKER], }, @@ -132,8 +132,8 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a // The older generations' rows are their business, already done keys: [ ...survivors, - 'title:феик|year:2026', - 'title:феик|year:2026|v2', + 'title:леика|year:2026', + 'title:леика|year:2026|v2', 'title:the boys|year:2019|v2', ].sort(), payloads: detailsPayloads, @@ -142,8 +142,8 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a previousRepeated: { keys: [ ...survivors, - 'title:феик|year:2026', - 'title:феик|year:2026|v2', + 'title:леика|year:2026', + 'title:леика|year:2026|v2', 'title:the boys|year:2019|v2', ].sort(), payloads: detailsPayloads, diff --git a/libs/shared/interfaces/src/lib/search-text-fold.util.spec.ts b/libs/shared/interfaces/src/lib/search-text-fold.util.spec.ts index 5a5604000..84f7c7c8e 100644 --- a/libs/shared/interfaces/src/lib/search-text-fold.util.spec.ts +++ b/libs/shared/interfaces/src/lib/search-text-fold.util.spec.ts @@ -13,7 +13,7 @@ describe('foldSearchText', () => { ['Ünlü Şef', 'ünlü şef'], ['ÇANAKKALE', 'çanakkale'], ['Первый канал HD', 'первый канал hd'], - ['Ёлки', 'ёлки'], + ['Ёжик', 'ёжик'], ['Йога для всех', 'йога для всех'], ['Ελλάδα Σήμερα', 'ελλάδα σήμερα'], ['Amélie', 'amélie'], @@ -28,7 +28,7 @@ describe('foldSearchText', () => { it.each([ ['Amélie', 'Amélie'], ['İnşaat', 'İnşaat'], - ['Ёлки', 'Ёлки'], + ['Ёжик', 'Ёжик'], ['Йога', 'Йога'], ['Ünlü', 'Ünlü'], ['Ά', 'Ά'], diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts index ee31eef2f..40a96cab0 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts @@ -434,22 +434,23 @@ describe('titleYearsCompatible', () => { describe('cleanTitleForSearch', () => { it('keeps Cyrillic letters that folding would rewrite', () => { // NFD splits "й" into "и" + a breve and "ё" into "е" + a diaeresis; - // the comparison key drops both marks, and TMDB answers the folded - // spelling with nothing (issue: "Фейк (10 серий)" never matched). - expect(normalizeTitle('Фейк (10 серий)')).toBe('феик'); - expect(cleanTitleForSearch('Фейк (10 серий)')).toBe('Фейк'); - expect(cleanTitleForSearch('Ёлки 2010')).toBe('Ёлки'); - expect(cleanTitleForSearch('Волшебный участок s02')).toBe( - 'Волшебный участок' + // the comparison key drops both marks, and TMDB answers that folded + // spelling with nothing. The titles here are illustrative stand-ins + // chosen to fold the same way. + expect(normalizeTitle('Лейка (10 серий)')).toBe('леика'); + expect(cleanTitleForSearch('Лейка (10 серий)')).toBe('Лейка'); + expect(cleanTitleForSearch('Ёжик 2010')).toBe('Ёжик'); + expect(cleanTitleForSearch('Тестовый Сериал s02')).toBe( + 'Тестовый Сериал' ); }); it('keeps Arabic hamza forms that folding splits into two words', () => { // "أ" decomposes into a bare alef + U+0654, which is outside the - // stripped mark range and so becomes a SPACE in the key. TMDB finds - // "أطرق بابي" and nothing for "ا طرق بابي". - expect(normalizeTitle('AR| أطرق بابي')).toBe('ا طرق بابي'); - expect(cleanTitleForSearch('AR| أطرق بابي')).toBe('أطرق بابي'); + // stripped mark range and so becomes a SPACE in the key — splitting + // the word in two, which is not a spelling TMDB indexes. + expect(normalizeTitle('AR| أمثلة تجريبية')).toBe('ا مثلة تجريبية'); + expect(cleanTitleForSearch('AR| أمثلة تجريبية')).toBe('أمثلة تجريبية'); }); it('keeps Latin diacritics and casing', () => { @@ -490,9 +491,9 @@ describe('cleanTitleForSearch', () => { it('normalizes to the same key as the text it was derived from', () => { for (const raw of [ - 'Фейк (10 серий)', + 'Лейка (10 серий)', 'EN - The Matrix (1999) 4K', - 'Ёлки 2010', + 'Ёжик 2010', 'Amélie (2001)', '|FR|VO|Le dernier empereur', ]) { diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.ts b/libs/shared/interfaces/src/lib/title-normalization.util.ts index 89b0b71bb..0c3759376 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.ts @@ -443,7 +443,7 @@ const SEASON_SUFFIX_PATTERN = new RegExp( * every comparison key and OFF for the text sent to a remote search — see * `cleanTitleForSearch`. Folding is lossy for scripts whose "diacritics" are * distinct letters: NFD turns Cyrillic "й" into "и" + a combining breve, and - * dropping the breve rewrites "Фейк" as "феик", "ё" as "е". Two provider + * dropping the breve rewrites "Лейка" as "леика", "ё" as "е". Two provider * copies of a title still meet on that key, which is all a comparison needs, * but TMDB's search does not fold Cyrillic the same way and answers a folded * query with nothing at all. @@ -520,7 +520,8 @@ export function normalizeTitleKeys( * * Comparison keys must fold so two spellings of one film meet; a search * query must not, because the search engine folds by its own rules and a - * pre-folded Cyrillic query ("феик" for "Фейк") matches nothing there. + * pre-folded Cyrillic query matches nothing there ("леика" for "Лейка" + * illustrates the rewrite). * Compare the results with `normalizeTitle`, never with this. */ export function cleanTitleForSearch(raw: string | null | undefined): string {