fix(portals): fold diacritics in the title index

Cross-playlist matching compares normalized titles ("Amélie" -> "amelie")
against an index built from the raw title, and the trigram tokenizer does not
fold diacritics by default. Every accented title was therefore invisible to
the FTS path: two identical `Amélie` entries produced no candidates at all.
That is the broadest of the Unicode gaps review found, and it predates the
short-title work.

The tokenizer is fixed at CREATE time, so existing databases recreate and
rebuild the index once behind a migration marker. `remove_diacritics` needs
SQLite 3.45+, so support is probed on a temp table first: an older runtime
keeps its working index untouched and the migration is not recorded as done,
leaving a later version free to upgrade it.

Case folding for non-ASCII remains impossible in stock SQLite — "ОН" cannot
find "Он" by any available predicate — and is documented as the known limit
rather than patched around again.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
4grayandClaude Opus 5 committed 2026-07-29 08:15:30 +02:00
1 parent a0eb68196d
commit 0b46aadee0
3 files changed
+209 -1

No files matched your search

+20
View File
@@ -481,6 +481,26 @@ both sources running and Stop owning only the newer one.
## Short titles and Unicode
The title index folds diacritics. `content_title_fts` is created with
`tokenize='trigram remove_diacritics 1'`, because matching compares NORMALIZED
titles ("Amélie" → "amelie") against an index built from the raw one — without
folding, every accented title was invisible to the FTS path, and two identical
`Amélie` entries produced no candidates at all. The tokenizer is fixed at
CREATE time, so existing databases are recreated and rebuilt once behind
`migration:content-title-fts-remove-diacritics:v1`.
`remove_diacritics` needs SQLite 3.45+. The migration probes support on a temp
table first and does nothing when the runtime rejects it, leaving the working
index in place — and does NOT record itself as done, so a later app version
shipping a newer SQLite upgrades then.
**Case folding for non-ASCII is still not possible.** `LOWER()`, GLOB classes
and the trigram tokenizer all fold ASCII only, so a Cyrillic title stored with
different capitalisation in two playlists ("ОН" vs "Он") cannot be matched by
any predicate available in stock SQLite. Closing that needs a stored
normalized-title column, which is deliberately out of scope here.
SQLite's `LOWER()` and GLOB character classes are ASCII-only, so a short
non-ASCII title could not be folded or word-bounded and simply never matched —
the film stayed absent from the chip. The ASCII/Unicode branch is decided from the RAW token, not the normalized
@@ -7,6 +7,8 @@ const {
indexMigrationStatements,
runMigrations,
cleanupLegacyTmdbSearchCache,
upgradeContentTitleFtsTokenizer,
contentTitleFtsStatement,
} = __databaseConnectionTestHooks;
type SqliteHandle = Parameters<typeof runMigrations>[0];
@@ -252,3 +254,81 @@ describe('runMigrations Xtream cache deduplication', () => {
expect(deleteContentRun).toHaveBeenCalledWith(11);
});
});
/**
* The title index folds diacritics, or the cross-playlist matcher cannot see
* accented titles at all: it compares normalized titles ("amelie") against an
* index built from the raw one ("Amélie").
*/
describe('content title FTS tokenizer upgrade', () => {
it('asks for diacritic folding in the statement it creates', () => {
expect(compactSql(contentTitleFtsStatement(true))).toContain(
"tokenize='trigram remove_diacritics 1'"
);
expect(compactSql(contentTitleFtsStatement(false))).toContain(
"tokenize='trigram'"
);
});
it('recreates and rebuilds the index, then records the migration', () => {
const rebuild = { run: jest.fn() };
const marker = { run: jest.fn() };
const { sqlite, exec, transaction } = createSqliteMock([
['SELECT value FROM app_state', { get: () => undefined }],
['INSERT INTO content_title_fts(content_title_fts)', rebuild],
['INSERT INTO app_state', marker],
]);
upgradeContentTitleFtsTokenizer(sqlite);
// The tokenizer is fixed at CREATE time, so the table has to go.
const statements = exec.mock.calls.map(([sql]) => compactSql(sql));
expect(statements).toContainEqual(
expect.stringContaining('DROP TABLE IF EXISTS content_title_fts')
);
expect(statements).toContainEqual(
expect.stringContaining("tokenize='trigram remove_diacritics 1'")
);
expect(rebuild.run).toHaveBeenCalled();
expect(marker.run).toHaveBeenCalled();
// One transaction, so a rejected CREATE rolls the drop back and the
// working index survives.
expect(transaction).toHaveBeenCalled();
});
it('does nothing once the migration has completed', () => {
const { sqlite, exec } = createSqliteMock([
completedMigrationStateRule,
]);
upgradeContentTitleFtsTokenizer(sqlite);
expect(exec).not.toHaveBeenCalled();
});
it('leaves the index alone when the runtime rejects the tokenizer', () => {
const marker = { run: jest.fn() };
const exec = jest.fn((statement: string) => {
if (statement.includes('remove_diacritics')) {
throw new Error('unknown tokenizer option');
}
});
const { sqlite } = createSqliteMock(
[
['SELECT value FROM app_state', { get: () => undefined }],
['INSERT INTO app_state', marker],
],
exec
);
upgradeContentTitleFtsTokenizer(sqlite);
// The probe failed, so the real table was never dropped — and the
// migration is NOT marked done, so a newer SQLite retries it.
const statements = exec.mock.calls.map(([sql]) => compactSql(sql));
expect(statements).not.toContainEqual(
expect.stringContaining('DROP TABLE IF EXISTS content_title_fts')
);
expect(marker.run).not.toHaveBeenCalled();
});
});
+109 -1
View File
@@ -28,6 +28,8 @@ const XTREAM_ADDED_EPOCH_SECONDS_MIGRATION_KEY =
'migration:xtream-content-added-epoch-seconds:v1';
const CONTENT_TITLE_FTS_MIGRATION_KEY =
'migration:content-title-fts-trigram:v1';
const CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY =
'migration:content-title-fts-remove-diacritics:v1';
const EPG_PROGRAM_SOURCE_URL_BACKFILL_MIGRATION_KEY =
'migration:epg-program-source-url-backfill:v1';
const TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY =
@@ -395,6 +397,8 @@ export const __databaseConnectionTestHooks = {
ensureDownloadsPauseResumeSchema,
normalizeXtreamContentAddedEpochs,
ensureContentTitleFts,
upgradeContentTitleFtsTokenizer,
contentTitleFtsStatement,
backfillEpgProgramSourceUrls,
cleanupLegacyTmdbSearchCache,
runMigrations,
@@ -618,6 +622,106 @@ function normalizeXtreamContentAddedEpochs(sqliteDb: Database.Database): void {
}
}
/**
* The title index, folding diacritics when the runtime can.
*
* Cross-playlist matching compares NORMALIZED titles ("Amélie" -> "amelie"),
* but the index holds the raw title, and the trigram tokenizer does not fold
* diacritics by default. Every accented title was therefore invisible to it:
* two identical `Amélie` entries produced no candidates at all.
*
* `remove_diacritics` needs SQLite 3.45+, so an older runtime keeps the plain
* tokenizer rather than losing the index — accented titles stay unmatched
* there, which is exactly the behaviour it had before.
*/
function contentTitleFtsStatement(removeDiacritics: boolean): string {
const tokenize = removeDiacritics
? `'trigram remove_diacritics 1'`
: `'trigram'`;
return `CREATE VIRTUAL TABLE IF NOT EXISTS content_title_fts USING fts5(
title,
content='content',
content_rowid='id',
tokenize=${tokenize}
)`;
}
/** Whether this SQLite accepts the folding tokenizer, asked without risk. */
function supportsTrigramDiacriticFolding(sqliteDb: Database.Database): boolean {
try {
sqliteDb.exec(
`CREATE VIRTUAL TABLE temp.content_title_fts_probe USING fts5(
title, tokenize='trigram remove_diacritics 1'
)`
);
sqliteDb.exec(`DROP TABLE temp.content_title_fts_probe`);
return true;
} catch {
return false;
}
}
/**
* Rebuild the title index with diacritic folding.
*
* The tokenizer is fixed at CREATE time, so an existing database keeps the
* old one until the table is recreated. Wrapped in a transaction: if the
* CREATE is rejected the drop rolls back and the working index survives.
*/
function upgradeContentTitleFtsTokenizer(sqliteDb: Database.Database): boolean {
try {
const migrationState = sqliteDb
.prepare(`SELECT value FROM app_state WHERE key = ?`)
.get(CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY) as
| { value?: unknown }
| undefined;
if (migrationState?.value === 'done') {
return false;
}
if (!supportsTrigramDiacriticFolding(sqliteDb)) {
// Not marked done: a later app version ships a newer SQLite, and
// this should upgrade itself then rather than stay degraded.
return false;
}
const executeMigration = sqliteDb.transaction(() => {
sqliteDb.exec(`DROP TABLE IF EXISTS content_title_fts`);
sqliteDb.exec(contentTitleFtsStatement(true));
sqliteDb
.prepare(
`INSERT INTO content_title_fts(content_title_fts)
VALUES ('rebuild')`
)
.run();
sqliteDb
.prepare(
`INSERT INTO app_state (key, value, updated_at)
VALUES (?, 'done', datetime('now'))
ON CONFLICT(key) DO UPDATE SET
value = excluded.value,
updated_at = excluded.updated_at`
)
.run(CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY);
});
executeMigration();
return true;
} catch (error) {
const message =
typeof error === 'object' && error !== null && 'message' in error
? String((error as { message?: unknown }).message ?? error)
: String(error);
console.warn(
`Content title FTS tokenizer upgrade failed (continuing): ${message}`
);
return false;
}
}
function ensureContentTitleFts(sqliteDb: Database.Database): void {
try {
const migrationState = sqliteDb
@@ -951,7 +1055,11 @@ function runMigrations(sqliteDb: Database.Database): void {
cleanupLegacyTmdbSearchCache(sqliteDb);
ensureDownloadsPauseResumeSchema(sqliteDb);
runMigrationStatements(sqliteDb, COLUMN_MIGRATION_STATEMENTS);
ensureContentTitleFts(sqliteDb);
// The tokenizer upgrade recreates and rebuilds the index itself, so the
// plain rebuild below would only repeat work it just did.
if (!upgradeContentTitleFtsTokenizer(sqliteDb)) {
ensureContentTitleFts(sqliteDb);
}
deduplicateXtreamCache(sqliteDb);
normalizeXtreamContentAddedEpochs(sqliteDb);
runMigrationStatements(sqliteDb, INDEX_MIGRATION_STATEMENTS);