fix(tmdb): a new series no longer matches its older, better-known namesake

TMDB returns titles in the REQUEST language, so a Cyrillic query issued with
`ru-RU` collapses unrelated foreign shows onto ordinary local nouns: an older
foreign series comes back under exactly the same localized name as a recent
local-language one. `pickConfidentMatch` admitted the older row through the
series "premiered earlier" tolerance — portals report the running season's
year while TMDB reports the premiere — and then let `pickMostPopular` decide
across every admitted candidate, discarding the year evidence that had just
admitted them. 26 votes beat 4, and the 2026 series rendered the 2018 show's
poster, cast, genres and rating.

Rank admitted candidates by year evidence first (`yearEvidenceTier`: the
provider's exact year, then a year off by one, then the series tolerance) and
let popularity break ties only inside the strongest tier any candidate
reached. The tolerance stays — three of eight real lookups from one install
depend on it (asked 2026 / premiere 2025; asked 2025 / premiere 2023; asked
2026 / premiere 2023) — but it is a last resort, not an equal. Measured over
400 Cyrillic series titles sampled from a real catalog, 20 normalized keys
had a same-titled older series and 16 of those were the more popular row.

The mirror case is accepted knowingly: a long-running show whose stated
season year happens to BE another same-titled show's premiere year now
resolves to the newer show. Only the older show's season air dates could
separate the two and a search response does not carry them, while that shape
needs three coincidences at once against one that needs none.

Search cache keys move to `|v4` with a matching startup cleanup, because a
positive row naming the wrong show stays fresh for 30 days.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
4grayandClaude Opus 5 committed 2026-09-21 07:44:46 +02:00
1 parent f80a21beef
commit b87ea9a178
9 files changed
+267 -65

No files matched your search

@@ -159,20 +159,24 @@ describe('TMDB search lookup cache cleanup', () => {
cleanupLegacyTmdbSearchCache(sqlite);
expect(transaction).toHaveBeenCalledTimes(2);
expect(deleteRun).toHaveBeenCalledTimes(2);
expect(transaction).toHaveBeenCalledTimes(3);
expect(deleteRun).toHaveBeenCalledTimes(3);
expect(markerRun.mock.calls.map(([key]) => key)).toEqual([
'migration:tmdb-search-lookup-v2-cache-cleanup:v1',
'migration:tmdb-search-lookup-v3-cache-cleanup:v1',
'migration:tmdb-search-lookup-v4-cache-cleanup:v1',
]);
const [unversionedDelete, v2Delete] = deleteStatements(sqlite);
const [unversionedDelete, v2Delete, v3Delete] =
deleteStatements(sqlite);
expect(unversionedDelete).toContain(
"lookup_key LIKE 'title:%|year:%' AND lookup_key NOT LIKE 'title:%|year:%|v%'"
);
// Only v2 search rows: the v3 rows the resolver writes now, and the
// `id:`/`person:`/`badProviderId:` rows, must survive.
// Each generation deletes only its own rows: the v4 rows the resolver
// writes now, and the `id:`/`person:`/`badProviderId:` rows, survive.
expect(v2Delete).toContain("WHERE lookup_key LIKE 'title:%|year:%|v2'");
expect(v2Delete).not.toContain('v3');
expect(v3Delete).toContain("WHERE lookup_key LIKE 'title:%|year:%|v3'");
expect(v3Delete).not.toContain('v4');
});
it('runs only the generations that have not completed yet', () => {
@@ -193,13 +197,14 @@ describe('TMDB search lookup cache cleanup', () => {
cleanupLegacyTmdbSearchCache(sqlite);
expect(transaction).toHaveBeenCalledTimes(1);
expect(markerRun).toHaveBeenCalledTimes(1);
expect(markerRun).toHaveBeenCalledWith(
'migration:tmdb-search-lookup-v3-cache-cleanup:v1'
);
expect(transaction).toHaveBeenCalledTimes(2);
expect(markerRun.mock.calls.map(([key]) => key)).toEqual([
'migration:tmdb-search-lookup-v3-cache-cleanup:v1',
'migration:tmdb-search-lookup-v4-cache-cleanup:v1',
]);
expect(deleteStatements(sqlite)).toEqual([
expect.stringContaining("lookup_key LIKE 'title:%|year:%|v2'"),
expect.stringContaining("lookup_key LIKE 'title:%|year:%|v3'"),
]);
});
@@ -212,7 +217,7 @@ describe('TMDB search lookup cache cleanup', () => {
expect(transaction).not.toHaveBeenCalled();
// One marker read per retired generation, nothing else
expect(prepare).toHaveBeenCalledTimes(2);
expect(prepare).toHaveBeenCalledTimes(3);
});
});
@@ -45,6 +45,8 @@ const TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY =
'migration:tmdb-search-lookup-v2-cache-cleanup:v1';
const TMDB_SEARCH_LOOKUP_V3_CACHE_CLEANUP_MIGRATION_KEY =
'migration:tmdb-search-lookup-v3-cache-cleanup:v1';
const TMDB_SEARCH_LOOKUP_V4_CACHE_CLEANUP_MIGRATION_KEY =
'migration:tmdb-search-lookup-v4-cache-cleanup:v1';
const EPG_PROGRAM_SOURCE_URL_BACKFILL_BATCH_SIZE = 50_000;
function readTraceFlag(name: string): boolean {
@@ -977,6 +979,10 @@ function widenTmdbMetadataMediaTypeCheck(sqliteDb: Database.Database): void {
* v2 every title with a Cyrillic "й"/"ё" was searched folded ("феик" for
* "Фейк", "елки" for "Ёлки"), got no answer, and was cached as missing for
* 7 days.
* - v3 → v4: year evidence became tiered. Under v3 a series admitted only by
* the "premiered earlier" tolerance competed with an exact-year match on
* popularity alone, so a new series resolved to its older, better-known
* namesake — and that positive row stays fresh for 30 days.
*/
const LEGACY_TMDB_SEARCH_CACHE_CLEANUPS: ReadonlyArray<{
migrationKey: string;
@@ -991,6 +997,10 @@ const LEGACY_TMDB_SEARCH_CACHE_CLEANUPS: ReadonlyArray<{
migrationKey: TMDB_SEARCH_LOOKUP_V3_CACHE_CLEANUP_MIGRATION_KEY,
rowPredicate: `lookup_key LIKE 'title:%|year:%|v2'`,
},
{
migrationKey: TMDB_SEARCH_LOOKUP_V4_CACHE_CLEANUP_MIGRATION_KEY,
rowPredicate: `lookup_key LIKE 'title:%|year:%|v3'`,
},
];
/**
@@ -28,11 +28,13 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a
console.log = () => undefined;
const V2_MARKER = 'migration:tmdb-search-lookup-v2-cache-cleanup:v1';
const V3_MARKER = 'migration:tmdb-search-lookup-v3-cache-cleanup:v1';
const V4_MARKER = 'migration:tmdb-search-lookup-v4-cache-cleanup:v1';
const ROWS = [
['tv', 'title:феик|year:2026', 'ru-RU', null],
['tv', 'title:феик|year:2026|v2', 'ru-RU', null],
['tv', 'title:the boys|year:2019|v2', 'en-US', 76479],
['tv', 'title:фейк|year:2026|v3', 'ru-RU', 317869],
['tv', 'title:nightfall|year:2026|v4', 'ru-RU', 424242],
['tv', 'id:317869|v2', 'ru-RU', 317869],
['tv', 'id:317869|season:1', 'ru-RU', 317869],
['person', 'person:287', 'en-US', 287],
@@ -61,11 +63,11 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a
hooks.runMigrations(skipped);
const skippedAfter = snapshot(skipped);
// Next startup: a row written meanwhile under the current key survives
skipped.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:гудовы|year:2026|v3', 'ru-RU', 318894)").run();
skipped.prepare("INSERT INTO tmdb_metadata (media_type, lookup_key, language, tmdb_id) VALUES ('tv', 'title:гудовы|year:2026|v4', 'ru-RU', 318894)").run();
hooks.runMigrations(skipped);
const repeated = snapshot(skipped);
// Previous release: the unversioned cleanup already ran; only v2 rows go
const previous = openDb([V2_MARKER]);
// Previous release: the earlier cleanups already ran; only v3 rows go
const previous = openDb([V2_MARKER, V3_MARKER]);
hooks.runMigrations(previous);
const previousAfter = snapshot(previous);
hooks.runMigrations(previous);
@@ -104,12 +106,13 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a
const V2_MARKER = 'migration:tmdb-search-lookup-v2-cache-cleanup:v1';
const V3_MARKER = 'migration:tmdb-search-lookup-v3-cache-cleanup:v1';
const V4_MARKER = 'migration:tmdb-search-lookup-v4-cache-cleanup:v1';
const survivors = [
'badProviderId:999',
'id:317869|season:1',
'id:317869|v2',
'person:287',
'title:фейк|year:2026|v3',
'title:nightfall|year:2026|v4',
'trending:week',
];
const detailsPayloads = ['{"id":317869}', '{"id":317869}'];
@@ -118,40 +121,54 @@ it('drops only retired search rows across skipped, previous, pre-person, fresh a
skippedAfter: {
keys: survivors,
payloads: detailsPayloads,
markers: [V2_MARKER, V3_MARKER],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
},
repeated: {
keys: [...survivors, 'title:гудовы|year:2026|v3'].sort(),
keys: [...survivors, 'title:гудовы|year:2026|v4'].sort(),
payloads: detailsPayloads,
markers: [V2_MARKER, V3_MARKER],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
},
previousAfter: {
// The unversioned row is that generation's business, already done
keys: [...survivors, 'title:феик|year:2026'].sort(),
// The older generations' rows are their business, already done
keys: [
...survivors,
'title:феик|year:2026',
'title:феик|year:2026|v2',
'title:the boys|year:2019|v2',
].sort(),
payloads: detailsPayloads,
markers: [V2_MARKER, V3_MARKER],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
},
previousRepeated: {
keys: [...survivors, 'title:феик|year:2026'].sort(),
keys: [
...survivors,
'title:феик|year:2026',
'title:феик|year:2026|v2',
'title:the boys|year:2019|v2',
].sort(),
payloads: detailsPayloads,
markers: [V2_MARKER, V3_MARKER],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
},
prePersonAfter: {
keys: [],
payloads: [],
markers: [V2_MARKER, V3_MARKER],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
check: true,
},
prePersonRepeated: {
keys: [],
payloads: [],
markers: [V2_MARKER, V3_MARKER],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
},
freshAfter: {
keys: [],
payloads: [],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
},
freshAfter: { keys: [], payloads: [], markers: [V2_MARKER, V3_MARKER] },
freshRepeated: {
keys: [],
payloads: [],
markers: [V2_MARKER, V3_MARKER],
markers: [V2_MARKER, V3_MARKER, V4_MARKER],
},
});
});