From 0b46aadee0e749b810a0cfdf4f05db5992970c33 Mon Sep 17 00:00:00 2001 From: 4gray Date: Wed, 29 Jul 2026 08:15:30 +0200 Subject: [PATCH] fix(portals): fold diacritics in the title index MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Cross-playlist matching compares normalized titles ("Amélie" -> "amelie") against an index built from the raw title, and the trigram tokenizer does not fold diacritics by default. Every accented title was therefore invisible to the FTS path: two identical `Amélie` entries produced no candidates at all. That is the broadest of the Unicode gaps review found, and it predates the short-title work. The tokenizer is fixed at CREATE time, so existing databases recreate and rebuild the index once behind a migration marker. `remove_diacritics` needs SQLite 3.45+, so support is probed on a temp table first: an older runtime keeps its working index untouched and the migration is not recorded as done, leaving a later version free to upgrade it. Case folding for non-ASCII remains impossible in stock SQLite — "ОН" cannot find "Он" by any available predicate — and is documented as the known limit rather than patched around again. Co-Authored-By: Claude Opus 5 --- docs/architecture/vod-multi-source.md | 20 ++++ .../src/lib/connection-migrations.spec.ts | 80 +++++++++++++ libs/shared/database/src/lib/connection.ts | 110 +++++++++++++++++- 3 files changed, 209 insertions(+), 1 deletion(-) diff --git a/docs/architecture/vod-multi-source.md b/docs/architecture/vod-multi-source.md index ce82bb830..c57e55e2c 100644 --- a/docs/architecture/vod-multi-source.md +++ b/docs/architecture/vod-multi-source.md @@ -481,6 +481,26 @@ both sources running and Stop owning only the newer one. ## Short titles and Unicode +The title index folds diacritics. `content_title_fts` is created with +`tokenize='trigram remove_diacritics 1'`, because matching compares NORMALIZED +titles ("Amélie" → "amelie") against an index built from the raw one — without +folding, every accented title was invisible to the FTS path, and two identical +`Amélie` entries produced no candidates at all. The tokenizer is fixed at +CREATE time, so existing databases are recreated and rebuilt once behind +`migration:content-title-fts-remove-diacritics:v1`. + +`remove_diacritics` needs SQLite 3.45+. The migration probes support on a temp +table first and does nothing when the runtime rejects it, leaving the working +index in place — and does NOT record itself as done, so a later app version +shipping a newer SQLite upgrades then. + +**Case folding for non-ASCII is still not possible.** `LOWER()`, GLOB classes +and the trigram tokenizer all fold ASCII only, so a Cyrillic title stored with +different capitalisation in two playlists ("ОН" vs "Он") cannot be matched by +any predicate available in stock SQLite. Closing that needs a stored +normalized-title column, which is deliberately out of scope here. + + SQLite's `LOWER()` and GLOB character classes are ASCII-only, so a short non-ASCII title could not be folded or word-bounded and simply never matched — the film stayed absent from the chip. The ASCII/Unicode branch is decided from the RAW token, not the normalized diff --git a/libs/shared/database/src/lib/connection-migrations.spec.ts b/libs/shared/database/src/lib/connection-migrations.spec.ts index ed067e7a9..d28810379 100644 --- a/libs/shared/database/src/lib/connection-migrations.spec.ts +++ b/libs/shared/database/src/lib/connection-migrations.spec.ts @@ -7,6 +7,8 @@ const { indexMigrationStatements, runMigrations, cleanupLegacyTmdbSearchCache, + upgradeContentTitleFtsTokenizer, + contentTitleFtsStatement, } = __databaseConnectionTestHooks; type SqliteHandle = Parameters[0]; @@ -252,3 +254,81 @@ describe('runMigrations Xtream cache deduplication', () => { expect(deleteContentRun).toHaveBeenCalledWith(11); }); }); + +/** + * The title index folds diacritics, or the cross-playlist matcher cannot see + * accented titles at all: it compares normalized titles ("amelie") against an + * index built from the raw one ("Amélie"). + */ +describe('content title FTS tokenizer upgrade', () => { + it('asks for diacritic folding in the statement it creates', () => { + expect(compactSql(contentTitleFtsStatement(true))).toContain( + "tokenize='trigram remove_diacritics 1'" + ); + expect(compactSql(contentTitleFtsStatement(false))).toContain( + "tokenize='trigram'" + ); + }); + + it('recreates and rebuilds the index, then records the migration', () => { + const rebuild = { run: jest.fn() }; + const marker = { run: jest.fn() }; + const { sqlite, exec, transaction } = createSqliteMock([ + ['SELECT value FROM app_state', { get: () => undefined }], + ['INSERT INTO content_title_fts(content_title_fts)', rebuild], + ['INSERT INTO app_state', marker], + ]); + + upgradeContentTitleFtsTokenizer(sqlite); + + // The tokenizer is fixed at CREATE time, so the table has to go. + const statements = exec.mock.calls.map(([sql]) => compactSql(sql)); + expect(statements).toContainEqual( + expect.stringContaining('DROP TABLE IF EXISTS content_title_fts') + ); + expect(statements).toContainEqual( + expect.stringContaining("tokenize='trigram remove_diacritics 1'") + ); + expect(rebuild.run).toHaveBeenCalled(); + expect(marker.run).toHaveBeenCalled(); + // One transaction, so a rejected CREATE rolls the drop back and the + // working index survives. + expect(transaction).toHaveBeenCalled(); + }); + + it('does nothing once the migration has completed', () => { + const { sqlite, exec } = createSqliteMock([ + completedMigrationStateRule, + ]); + + upgradeContentTitleFtsTokenizer(sqlite); + + expect(exec).not.toHaveBeenCalled(); + }); + + it('leaves the index alone when the runtime rejects the tokenizer', () => { + const marker = { run: jest.fn() }; + const exec = jest.fn((statement: string) => { + if (statement.includes('remove_diacritics')) { + throw new Error('unknown tokenizer option'); + } + }); + const { sqlite } = createSqliteMock( + [ + ['SELECT value FROM app_state', { get: () => undefined }], + ['INSERT INTO app_state', marker], + ], + exec + ); + + upgradeContentTitleFtsTokenizer(sqlite); + + // The probe failed, so the real table was never dropped — and the + // migration is NOT marked done, so a newer SQLite retries it. + const statements = exec.mock.calls.map(([sql]) => compactSql(sql)); + expect(statements).not.toContainEqual( + expect.stringContaining('DROP TABLE IF EXISTS content_title_fts') + ); + expect(marker.run).not.toHaveBeenCalled(); + }); +}); diff --git a/libs/shared/database/src/lib/connection.ts b/libs/shared/database/src/lib/connection.ts index 1665103e6..c139ef488 100644 --- a/libs/shared/database/src/lib/connection.ts +++ b/libs/shared/database/src/lib/connection.ts @@ -28,6 +28,8 @@ const XTREAM_ADDED_EPOCH_SECONDS_MIGRATION_KEY = 'migration:xtream-content-added-epoch-seconds:v1'; const CONTENT_TITLE_FTS_MIGRATION_KEY = 'migration:content-title-fts-trigram:v1'; +const CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY = + 'migration:content-title-fts-remove-diacritics:v1'; const EPG_PROGRAM_SOURCE_URL_BACKFILL_MIGRATION_KEY = 'migration:epg-program-source-url-backfill:v1'; const TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY = @@ -395,6 +397,8 @@ export const __databaseConnectionTestHooks = { ensureDownloadsPauseResumeSchema, normalizeXtreamContentAddedEpochs, ensureContentTitleFts, + upgradeContentTitleFtsTokenizer, + contentTitleFtsStatement, backfillEpgProgramSourceUrls, cleanupLegacyTmdbSearchCache, runMigrations, @@ -618,6 +622,106 @@ function normalizeXtreamContentAddedEpochs(sqliteDb: Database.Database): void { } } +/** + * The title index, folding diacritics when the runtime can. + * + * Cross-playlist matching compares NORMALIZED titles ("Amélie" -> "amelie"), + * but the index holds the raw title, and the trigram tokenizer does not fold + * diacritics by default. Every accented title was therefore invisible to it: + * two identical `Amélie` entries produced no candidates at all. + * + * `remove_diacritics` needs SQLite 3.45+, so an older runtime keeps the plain + * tokenizer rather than losing the index — accented titles stay unmatched + * there, which is exactly the behaviour it had before. + */ +function contentTitleFtsStatement(removeDiacritics: boolean): string { + const tokenize = removeDiacritics + ? `'trigram remove_diacritics 1'` + : `'trigram'`; + return `CREATE VIRTUAL TABLE IF NOT EXISTS content_title_fts USING fts5( + title, + content='content', + content_rowid='id', + tokenize=${tokenize} + )`; +} + +/** Whether this SQLite accepts the folding tokenizer, asked without risk. */ +function supportsTrigramDiacriticFolding(sqliteDb: Database.Database): boolean { + try { + sqliteDb.exec( + `CREATE VIRTUAL TABLE temp.content_title_fts_probe USING fts5( + title, tokenize='trigram remove_diacritics 1' + )` + ); + sqliteDb.exec(`DROP TABLE temp.content_title_fts_probe`); + return true; + } catch { + return false; + } +} + +/** + * Rebuild the title index with diacritic folding. + * + * The tokenizer is fixed at CREATE time, so an existing database keeps the + * old one until the table is recreated. Wrapped in a transaction: if the + * CREATE is rejected the drop rolls back and the working index survives. + */ +function upgradeContentTitleFtsTokenizer(sqliteDb: Database.Database): boolean { + try { + const migrationState = sqliteDb + .prepare(`SELECT value FROM app_state WHERE key = ?`) + .get(CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY) as + | { value?: unknown } + | undefined; + + if (migrationState?.value === 'done') { + return false; + } + + if (!supportsTrigramDiacriticFolding(sqliteDb)) { + // Not marked done: a later app version ships a newer SQLite, and + // this should upgrade itself then rather than stay degraded. + return false; + } + + const executeMigration = sqliteDb.transaction(() => { + sqliteDb.exec(`DROP TABLE IF EXISTS content_title_fts`); + sqliteDb.exec(contentTitleFtsStatement(true)); + sqliteDb + .prepare( + `INSERT INTO content_title_fts(content_title_fts) + VALUES ('rebuild')` + ) + .run(); + + sqliteDb + .prepare( + `INSERT INTO app_state (key, value, updated_at) + VALUES (?, 'done', datetime('now')) + ON CONFLICT(key) DO UPDATE SET + value = excluded.value, + updated_at = excluded.updated_at` + ) + .run(CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY); + }); + + executeMigration(); + return true; + } catch (error) { + const message = + typeof error === 'object' && error !== null && 'message' in error + ? String((error as { message?: unknown }).message ?? error) + : String(error); + + console.warn( + `Content title FTS tokenizer upgrade failed (continuing): ${message}` + ); + return false; + } +} + function ensureContentTitleFts(sqliteDb: Database.Database): void { try { const migrationState = sqliteDb @@ -951,7 +1055,11 @@ function runMigrations(sqliteDb: Database.Database): void { cleanupLegacyTmdbSearchCache(sqliteDb); ensureDownloadsPauseResumeSchema(sqliteDb); runMigrationStatements(sqliteDb, COLUMN_MIGRATION_STATEMENTS); - ensureContentTitleFts(sqliteDb); + // The tokenizer upgrade recreates and rebuilds the index itself, so the + // plain rebuild below would only repeat work it just did. + if (!upgradeContentTitleFtsTokenizer(sqliteDb)) { + ensureContentTitleFts(sqliteDb); + } deduplicateXtreamCache(sqliteDb); normalizeXtreamContentAddedEpochs(sqliteDb); runMigrationStatements(sqliteDb, INDEX_MIGRATION_STATEMENTS);