From 257a53d9e7af169c9dfda4e82d685f27aea270a8 Mon Sep 17 00:00:00 2001 From: 4gray Date: Wed, 29 Jul 2026 19:53:56 +0200 Subject: [PATCH] fix(portals): cover a letter spelled two ways in lower case MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Greek Σ lowercases to σ, but a word-final sigma is written ς and is equally a lowercase of it, so a class built only from the character in hand knew one spelling of two. Each class now also carries the uppercase form's own lowercase, which reaches the other one. One-way on purpose: σ → Σ → σ never arrives at ς. Left so because ς is only correct at the end of a word, which is exactly where the request's last character sits — the pair that occurs in real titles is covered, and closing the other direction needs a fold table. Co-Authored-By: Claude Opus 5 --- .../title-sources.operations.spec.ts | 2 +- .../operations/title-token-glob.spec.ts | 29 +++++++++++++++++-- .../database/operations/title-token-glob.ts | 23 ++++++++++----- docs/architecture/vod-multi-source.md | 18 +++++++++--- 4 files changed, 58 insertions(+), 14 deletions(-) diff --git a/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts b/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts index fa7ea54a3..1f517b2ef 100644 --- a/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts +++ b/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts @@ -76,7 +76,7 @@ describe('title-sources.operations', () => { // code points — so the case is folded in JavaScript and handed // over one class per character. Without this a title stored in // any case other than the two below is simply never found. - expect(scanQuery.params).toContain('*[оО][нН]*'); + expect(scanQuery.params).toContain('*[Оо][нН]*'); expect(scanQuery.sql).toContain('instr'); // Both the folded and the as-typed form, since the class is built // from the raw token and cannot cover a diacritic difference. diff --git a/apps/electron-backend/src/app/database/operations/title-token-glob.spec.ts b/apps/electron-backend/src/app/database/operations/title-token-glob.spec.ts index 08242e890..9b1bb261f 100644 --- a/apps/electron-backend/src/app/database/operations/title-token-glob.spec.ts +++ b/apps/electron-backend/src/app/database/operations/title-token-glob.spec.ts @@ -39,14 +39,30 @@ function hasSqlite(): boolean { const describeWithSqlite = hasSqlite() ? describe : describe.skip; describe('caseInsensitiveGlobPattern', () => { - it('builds a two-case class for each cased character', () => { - expect(caseInsensitiveGlobPattern('Он')).toBe('*[оО][нН]*'); + it('builds a case class for each cased character', () => { + expect(caseInsensitiveGlobPattern('Он')).toBe('*[Оо][нН]*'); }); it('leaves uncased characters alone', () => { expect(caseInsensitiveGlobPattern('7')).toBe('*7*'); }); + it('covers a letter with two lowercase spellings', () => { + // Greek Σ lowercases to σ, but a word-final sigma is written ς and is + // just as much a lowercase of it. Going back down from the uppercase + // form reaches the spelling the character in hand does not name. + expect(caseInsensitiveGlobPattern('ς')).toBe('*[ςΣσ]*'); + }); + + it('reaches the second spelling in one direction only', () => { + // σ → Σ → σ never arrives at ς, so a request spelled with a medial + // sigma does not match a stored final one. Left as is: ς is only ever + // correct at the end of a word, which is exactly where the request's + // own last character sits, so the pair that occurs in real titles is + // the one above. Closing the other direction needs a fold table. + expect(caseInsensitiveGlobPattern('σ')).toBe('*[σΣ]*'); + }); + it('refuses a token holding a GLOB metacharacter', () => { // SQLite GLOB has no escape character, so an unescaped `*` here would // silently become a wildcard and match every row in the table. @@ -95,6 +111,15 @@ describeWithSqlite('caseInsensitiveGlobPattern against SQLite', () => { expect(sqliteGlob('Дюна', pattern)).toBe(false); }); + it('matches a Greek word written with either sigma', () => { + const pattern = caseInsensitiveGlobPattern('ος') as string; + + expect(sqliteGlob('ος', pattern)).toBe(true); + expect(sqliteGlob('ΟΣ', pattern)).toBe(true); + // The spelling a class built from the request alone would miss. + expect(sqliteGlob('οσ', pattern)).toBe(true); + }); + it('matches Greek and accented Latin in any case', () => { expect( sqliteGlob('ΟΙ', caseInsensitiveGlobPattern('οι') as string) diff --git a/apps/electron-backend/src/app/database/operations/title-token-glob.ts b/apps/electron-backend/src/app/database/operations/title-token-glob.ts index fb4b82d54..4add8f63e 100644 --- a/apps/electron-backend/src/app/database/operations/title-token-glob.ts +++ b/apps/electron-backend/src/app/database/operations/title-token-glob.ts @@ -33,24 +33,33 @@ export function caseInsensitiveGlobPattern(token: string): string | null { let pattern = '*'; for (const character of token) { - const lower = character.toLowerCase(); const upper = character.toUpperCase(); + // Going back down from the uppercase form catches letters with more + // than one lowercase spelling: Greek Σ lowercases to σ, but ς is an + // equally valid lowercase of it, and a class built only from the + // character in hand would know just one of the two. + const forms = new Set([ + character, + character.toLowerCase(), + upper, + upper.toLowerCase(), + ]); - if (lower === upper) { + if (forms.size === 1) { pattern += character; continue; } if ( - [...lower].length !== 1 || - [...upper].length !== 1 || - GLOB_METACHARACTERS.test(lower) || - GLOB_METACHARACTERS.test(upper) + [...forms].some( + (form) => + [...form].length !== 1 || GLOB_METACHARACTERS.test(form) + ) ) { return null; } - pattern += `[${lower}${upper}]`; + pattern += `[${[...forms].join('')}]`; } return `${pattern}*`; diff --git a/docs/architecture/vod-multi-source.md b/docs/architecture/vod-multi-source.md index 568f5ae8f..7896a8f34 100644 --- a/docs/architecture/vod-multi-source.md +++ b/docs/architecture/vod-multi-source.md @@ -533,10 +533,20 @@ The **scan** tier does fold it, because GLOB character classes are *not* ASCII-only — `patternCompare` reads them as UTF-8 code points, and `'Он' GLOB '*[Оо][Нн]*'` is true. `caseInsensitiveGlobPattern` therefore folds the case in JavaScript, where Unicode case mapping is real, and hands SQLite one -`[lowerUpper]` class per character. It returns `null` — leaving the substring -tests as the whole answer — for a token holding a GLOB metacharacter (SQLite -GLOB has no escape character, so an unescaped `*` would become a wildcard) or a -case mapping that changes length (`ß` uppercases to `SS`). +class per character. Each class also carries the uppercase form's OWN lowercase, +which is what covers a letter spelled two ways in lower case: Greek `Σ` +lowercases to `σ`, but a word-final sigma is written `ς` and is equally a +lowercase of it. That reach is one-way — `σ → Σ → σ` never arrives at `ς` — and +left so deliberately, because `ς` is only correct at the end of a word, exactly +where the request's own last character sits; closing the other direction needs a +fold table. + +It returns `null` — leaving the substring tests as the whole answer — for a +token holding a GLOB metacharacter (SQLite GLOB has no escape character, so an +unescaped `*` would become a wildcard matching every row) or a case mapping that +changes length (`ß` uppercases to `SS`, `İ` lowercases to two code points). +Neither has a single-character class that means the same thing, and a wrong +pattern is worse than an absent one. The ASCII/Unicode branch is decided from the RAW token, not the normalized one: normalization folds diacritics, so "Ça" arrives as "ca" and looks like plain