mirror of
https://github.com/4gray/iptvnator.git
synced 2026-10-11 02:46:16 -08:00
fix(portals): fold diacritics in the title index
Cross-playlist matching compares normalized titles ("Amélie" -> "amelie")
against an index built from the raw title, and the trigram tokenizer does not
fold diacritics by default. Every accented title was therefore invisible to
the FTS path: two identical `Amélie` entries produced no candidates at all.
That is the broadest of the Unicode gaps review found, and it predates the
short-title work.
The tokenizer is fixed at CREATE time, so existing databases recreate and
rebuild the index once behind a migration marker. `remove_diacritics` needs
SQLite 3.45+, so support is probed on a temp table first: an older runtime
keeps its working index untouched and the migration is not recorded as done,
leaving a later version free to upgrade it.
Case folding for non-ASCII remains impossible in stock SQLite — "ОН" cannot
find "Он" by any available predicate — and is documented as the known limit
rather than patched around again.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
a0eb68196d
commit
0b46aadee0
3 files changed
+209
-1
No files matched your search
@@ -481,6 +481,26 @@ both sources running and Stop owning only the newer one.
|
||||
|
||||
## Short titles and Unicode
|
||||
|
||||
The title index folds diacritics. `content_title_fts` is created with
|
||||
`tokenize='trigram remove_diacritics 1'`, because matching compares NORMALIZED
|
||||
titles ("Amélie" → "amelie") against an index built from the raw one — without
|
||||
folding, every accented title was invisible to the FTS path, and two identical
|
||||
`Amélie` entries produced no candidates at all. The tokenizer is fixed at
|
||||
CREATE time, so existing databases are recreated and rebuilt once behind
|
||||
`migration:content-title-fts-remove-diacritics:v1`.
|
||||
|
||||
`remove_diacritics` needs SQLite 3.45+. The migration probes support on a temp
|
||||
table first and does nothing when the runtime rejects it, leaving the working
|
||||
index in place — and does NOT record itself as done, so a later app version
|
||||
shipping a newer SQLite upgrades then.
|
||||
|
||||
**Case folding for non-ASCII is still not possible.** `LOWER()`, GLOB classes
|
||||
and the trigram tokenizer all fold ASCII only, so a Cyrillic title stored with
|
||||
different capitalisation in two playlists ("ОН" vs "Он") cannot be matched by
|
||||
any predicate available in stock SQLite. Closing that needs a stored
|
||||
normalized-title column, which is deliberately out of scope here.
|
||||
|
||||
|
||||
SQLite's `LOWER()` and GLOB character classes are ASCII-only, so a short
|
||||
non-ASCII title could not be folded or word-bounded and simply never matched —
|
||||
the film stayed absent from the chip. The ASCII/Unicode branch is decided from the RAW token, not the normalized
|
||||
|
||||
@@ -7,6 +7,8 @@ const {
|
||||
indexMigrationStatements,
|
||||
runMigrations,
|
||||
cleanupLegacyTmdbSearchCache,
|
||||
upgradeContentTitleFtsTokenizer,
|
||||
contentTitleFtsStatement,
|
||||
} = __databaseConnectionTestHooks;
|
||||
|
||||
type SqliteHandle = Parameters<typeof runMigrations>[0];
|
||||
@@ -252,3 +254,81 @@ describe('runMigrations Xtream cache deduplication', () => {
|
||||
expect(deleteContentRun).toHaveBeenCalledWith(11);
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* The title index folds diacritics, or the cross-playlist matcher cannot see
|
||||
* accented titles at all: it compares normalized titles ("amelie") against an
|
||||
* index built from the raw one ("Amélie").
|
||||
*/
|
||||
describe('content title FTS tokenizer upgrade', () => {
|
||||
it('asks for diacritic folding in the statement it creates', () => {
|
||||
expect(compactSql(contentTitleFtsStatement(true))).toContain(
|
||||
"tokenize='trigram remove_diacritics 1'"
|
||||
);
|
||||
expect(compactSql(contentTitleFtsStatement(false))).toContain(
|
||||
"tokenize='trigram'"
|
||||
);
|
||||
});
|
||||
|
||||
it('recreates and rebuilds the index, then records the migration', () => {
|
||||
const rebuild = { run: jest.fn() };
|
||||
const marker = { run: jest.fn() };
|
||||
const { sqlite, exec, transaction } = createSqliteMock([
|
||||
['SELECT value FROM app_state', { get: () => undefined }],
|
||||
['INSERT INTO content_title_fts(content_title_fts)', rebuild],
|
||||
['INSERT INTO app_state', marker],
|
||||
]);
|
||||
|
||||
upgradeContentTitleFtsTokenizer(sqlite);
|
||||
|
||||
// The tokenizer is fixed at CREATE time, so the table has to go.
|
||||
const statements = exec.mock.calls.map(([sql]) => compactSql(sql));
|
||||
expect(statements).toContainEqual(
|
||||
expect.stringContaining('DROP TABLE IF EXISTS content_title_fts')
|
||||
);
|
||||
expect(statements).toContainEqual(
|
||||
expect.stringContaining("tokenize='trigram remove_diacritics 1'")
|
||||
);
|
||||
expect(rebuild.run).toHaveBeenCalled();
|
||||
expect(marker.run).toHaveBeenCalled();
|
||||
// One transaction, so a rejected CREATE rolls the drop back and the
|
||||
// working index survives.
|
||||
expect(transaction).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('does nothing once the migration has completed', () => {
|
||||
const { sqlite, exec } = createSqliteMock([
|
||||
completedMigrationStateRule,
|
||||
]);
|
||||
|
||||
upgradeContentTitleFtsTokenizer(sqlite);
|
||||
|
||||
expect(exec).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('leaves the index alone when the runtime rejects the tokenizer', () => {
|
||||
const marker = { run: jest.fn() };
|
||||
const exec = jest.fn((statement: string) => {
|
||||
if (statement.includes('remove_diacritics')) {
|
||||
throw new Error('unknown tokenizer option');
|
||||
}
|
||||
});
|
||||
const { sqlite } = createSqliteMock(
|
||||
[
|
||||
['SELECT value FROM app_state', { get: () => undefined }],
|
||||
['INSERT INTO app_state', marker],
|
||||
],
|
||||
exec
|
||||
);
|
||||
|
||||
upgradeContentTitleFtsTokenizer(sqlite);
|
||||
|
||||
// The probe failed, so the real table was never dropped — and the
|
||||
// migration is NOT marked done, so a newer SQLite retries it.
|
||||
const statements = exec.mock.calls.map(([sql]) => compactSql(sql));
|
||||
expect(statements).not.toContainEqual(
|
||||
expect.stringContaining('DROP TABLE IF EXISTS content_title_fts')
|
||||
);
|
||||
expect(marker.run).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
@@ -28,6 +28,8 @@ const XTREAM_ADDED_EPOCH_SECONDS_MIGRATION_KEY =
|
||||
'migration:xtream-content-added-epoch-seconds:v1';
|
||||
const CONTENT_TITLE_FTS_MIGRATION_KEY =
|
||||
'migration:content-title-fts-trigram:v1';
|
||||
const CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY =
|
||||
'migration:content-title-fts-remove-diacritics:v1';
|
||||
const EPG_PROGRAM_SOURCE_URL_BACKFILL_MIGRATION_KEY =
|
||||
'migration:epg-program-source-url-backfill:v1';
|
||||
const TMDB_SEARCH_LOOKUP_V2_CACHE_CLEANUP_MIGRATION_KEY =
|
||||
@@ -395,6 +397,8 @@ export const __databaseConnectionTestHooks = {
|
||||
ensureDownloadsPauseResumeSchema,
|
||||
normalizeXtreamContentAddedEpochs,
|
||||
ensureContentTitleFts,
|
||||
upgradeContentTitleFtsTokenizer,
|
||||
contentTitleFtsStatement,
|
||||
backfillEpgProgramSourceUrls,
|
||||
cleanupLegacyTmdbSearchCache,
|
||||
runMigrations,
|
||||
@@ -618,6 +622,106 @@ function normalizeXtreamContentAddedEpochs(sqliteDb: Database.Database): void {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The title index, folding diacritics when the runtime can.
|
||||
*
|
||||
* Cross-playlist matching compares NORMALIZED titles ("Amélie" -> "amelie"),
|
||||
* but the index holds the raw title, and the trigram tokenizer does not fold
|
||||
* diacritics by default. Every accented title was therefore invisible to it:
|
||||
* two identical `Amélie` entries produced no candidates at all.
|
||||
*
|
||||
* `remove_diacritics` needs SQLite 3.45+, so an older runtime keeps the plain
|
||||
* tokenizer rather than losing the index — accented titles stay unmatched
|
||||
* there, which is exactly the behaviour it had before.
|
||||
*/
|
||||
function contentTitleFtsStatement(removeDiacritics: boolean): string {
|
||||
const tokenize = removeDiacritics
|
||||
? `'trigram remove_diacritics 1'`
|
||||
: `'trigram'`;
|
||||
return `CREATE VIRTUAL TABLE IF NOT EXISTS content_title_fts USING fts5(
|
||||
title,
|
||||
content='content',
|
||||
content_rowid='id',
|
||||
tokenize=${tokenize}
|
||||
)`;
|
||||
}
|
||||
|
||||
/** Whether this SQLite accepts the folding tokenizer, asked without risk. */
|
||||
function supportsTrigramDiacriticFolding(sqliteDb: Database.Database): boolean {
|
||||
try {
|
||||
sqliteDb.exec(
|
||||
`CREATE VIRTUAL TABLE temp.content_title_fts_probe USING fts5(
|
||||
title, tokenize='trigram remove_diacritics 1'
|
||||
)`
|
||||
);
|
||||
sqliteDb.exec(`DROP TABLE temp.content_title_fts_probe`);
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild the title index with diacritic folding.
|
||||
*
|
||||
* The tokenizer is fixed at CREATE time, so an existing database keeps the
|
||||
* old one until the table is recreated. Wrapped in a transaction: if the
|
||||
* CREATE is rejected the drop rolls back and the working index survives.
|
||||
*/
|
||||
function upgradeContentTitleFtsTokenizer(sqliteDb: Database.Database): boolean {
|
||||
try {
|
||||
const migrationState = sqliteDb
|
||||
.prepare(`SELECT value FROM app_state WHERE key = ?`)
|
||||
.get(CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY) as
|
||||
| { value?: unknown }
|
||||
| undefined;
|
||||
|
||||
if (migrationState?.value === 'done') {
|
||||
return false;
|
||||
}
|
||||
|
||||
if (!supportsTrigramDiacriticFolding(sqliteDb)) {
|
||||
// Not marked done: a later app version ships a newer SQLite, and
|
||||
// this should upgrade itself then rather than stay degraded.
|
||||
return false;
|
||||
}
|
||||
|
||||
const executeMigration = sqliteDb.transaction(() => {
|
||||
sqliteDb.exec(`DROP TABLE IF EXISTS content_title_fts`);
|
||||
sqliteDb.exec(contentTitleFtsStatement(true));
|
||||
sqliteDb
|
||||
.prepare(
|
||||
`INSERT INTO content_title_fts(content_title_fts)
|
||||
VALUES ('rebuild')`
|
||||
)
|
||||
.run();
|
||||
|
||||
sqliteDb
|
||||
.prepare(
|
||||
`INSERT INTO app_state (key, value, updated_at)
|
||||
VALUES (?, 'done', datetime('now'))
|
||||
ON CONFLICT(key) DO UPDATE SET
|
||||
value = excluded.value,
|
||||
updated_at = excluded.updated_at`
|
||||
)
|
||||
.run(CONTENT_TITLE_FTS_DIACRITICS_MIGRATION_KEY);
|
||||
});
|
||||
|
||||
executeMigration();
|
||||
return true;
|
||||
} catch (error) {
|
||||
const message =
|
||||
typeof error === 'object' && error !== null && 'message' in error
|
||||
? String((error as { message?: unknown }).message ?? error)
|
||||
: String(error);
|
||||
|
||||
console.warn(
|
||||
`Content title FTS tokenizer upgrade failed (continuing): ${message}`
|
||||
);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function ensureContentTitleFts(sqliteDb: Database.Database): void {
|
||||
try {
|
||||
const migrationState = sqliteDb
|
||||
@@ -951,7 +1055,11 @@ function runMigrations(sqliteDb: Database.Database): void {
|
||||
cleanupLegacyTmdbSearchCache(sqliteDb);
|
||||
ensureDownloadsPauseResumeSchema(sqliteDb);
|
||||
runMigrationStatements(sqliteDb, COLUMN_MIGRATION_STATEMENTS);
|
||||
ensureContentTitleFts(sqliteDb);
|
||||
// The tokenizer upgrade recreates and rebuilds the index itself, so the
|
||||
// plain rebuild below would only repeat work it just did.
|
||||
if (!upgradeContentTitleFtsTokenizer(sqliteDb)) {
|
||||
ensureContentTitleFts(sqliteDb);
|
||||
}
|
||||
deduplicateXtreamCache(sqliteDb);
|
||||
normalizeXtreamContentAddedEpochs(sqliteDb);
|
||||
runMigrationStatements(sqliteDb, INDEX_MIGRATION_STATEMENTS);
|
||||
|
||||
Reference in new issue
Block a user