fix(portals): cover a letter spelled two ways in lower case

Greek Σ lowercases to σ, but a word-final sigma is written ς and is
equally a lowercase of it, so a class built only from the character in
hand knew one spelling of two. Each class now also carries the uppercase
form's own lowercase, which reaches the other one.

One-way on purpose: σ → Σ → σ never arrives at ς. Left so because ς is
only correct at the end of a word, which is exactly where the request's
last character sits — the pair that occurs in real titles is covered, and
closing the other direction needs a fold table.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
4grayandClaude Opus 5 committed 2026-07-29 19:53:56 +02:00
1 parent ddb299e237
commit 257a53d9e7
4 files changed
+58 -14

No files matched your search

@@ -76,7 +76,7 @@ describe('title-sources.operations', () => {
// code points — so the case is folded in JavaScript and handed
// over one class per character. Without this a title stored in
// any case other than the two below is simply never found.
expect(scanQuery.params).toContain('*[оО][нН]*');
expect(scanQuery.params).toContain('*[Оо][нН]*');
expect(scanQuery.sql).toContain('instr');
// Both the folded and the as-typed form, since the class is built
// from the raw token and cannot cover a diacritic difference.
@@ -39,14 +39,30 @@ function hasSqlite(): boolean {
const describeWithSqlite = hasSqlite() ? describe : describe.skip;
describe('caseInsensitiveGlobPattern', () => {
it('builds a two-case class for each cased character', () => {
expect(caseInsensitiveGlobPattern('Он')).toBe('*[оО][нН]*');
it('builds a case class for each cased character', () => {
expect(caseInsensitiveGlobPattern('Он')).toBe('*[Оо][нН]*');
});
it('leaves uncased characters alone', () => {
expect(caseInsensitiveGlobPattern('7')).toBe('*7*');
});
it('covers a letter with two lowercase spellings', () => {
// Greek Σ lowercases to σ, but a word-final sigma is written ς and is
// just as much a lowercase of it. Going back down from the uppercase
// form reaches the spelling the character in hand does not name.
expect(caseInsensitiveGlobPattern('ς')).toBe('*[ςΣσ]*');
});
it('reaches the second spelling in one direction only', () => {
// σ → Σ → σ never arrives at ς, so a request spelled with a medial
// sigma does not match a stored final one. Left as is: ς is only ever
// correct at the end of a word, which is exactly where the request's
// own last character sits, so the pair that occurs in real titles is
// the one above. Closing the other direction needs a fold table.
expect(caseInsensitiveGlobPattern('σ')).toBe('*[σΣ]*');
});
it('refuses a token holding a GLOB metacharacter', () => {
// SQLite GLOB has no escape character, so an unescaped `*` here would
// silently become a wildcard and match every row in the table.
@@ -95,6 +111,15 @@ describeWithSqlite('caseInsensitiveGlobPattern against SQLite', () => {
expect(sqliteGlob('Дюна', pattern)).toBe(false);
});
it('matches a Greek word written with either sigma', () => {
const pattern = caseInsensitiveGlobPattern('ος') as string;
expect(sqliteGlob('ος', pattern)).toBe(true);
expect(sqliteGlob('ΟΣ', pattern)).toBe(true);
// The spelling a class built from the request alone would miss.
expect(sqliteGlob('οσ', pattern)).toBe(true);
});
it('matches Greek and accented Latin in any case', () => {
expect(
sqliteGlob('ΟΙ', caseInsensitiveGlobPattern('οι') as string)
@@ -33,24 +33,33 @@ export function caseInsensitiveGlobPattern(token: string): string | null {
let pattern = '*';
for (const character of token) {
const lower = character.toLowerCase();
const upper = character.toUpperCase();
// Going back down from the uppercase form catches letters with more
// than one lowercase spelling: Greek Σ lowercases to σ, but ς is an
// equally valid lowercase of it, and a class built only from the
// character in hand would know just one of the two.
const forms = new Set([
character,
character.toLowerCase(),
upper,
upper.toLowerCase(),
]);
if (lower === upper) {
if (forms.size === 1) {
pattern += character;
continue;
}
if (
[...lower].length !== 1 ||
[...upper].length !== 1 ||
GLOB_METACHARACTERS.test(lower) ||
GLOB_METACHARACTERS.test(upper)
[...forms].some(
(form) =>
[...form].length !== 1 || GLOB_METACHARACTERS.test(form)
)
) {
return null;
}
pattern += `[${lower}${upper}]`;
pattern += `[${[...forms].join('')}]`;
}
return `${pattern}*`;
+14 -4
View File
@@ -533,10 +533,20 @@ The **scan** tier does fold it, because GLOB character classes are *not*
ASCII-only — `patternCompare` reads them as UTF-8 code points, and
`'Он' GLOB '*[Оо][Нн]*'` is true. `caseInsensitiveGlobPattern` therefore folds
the case in JavaScript, where Unicode case mapping is real, and hands SQLite one
`[lowerUpper]` class per character. It returns `null` — leaving the substring
tests as the whole answer — for a token holding a GLOB metacharacter (SQLite
GLOB has no escape character, so an unescaped `*` would become a wildcard) or a
case mapping that changes length (`ß` uppercases to `SS`).
class per character. Each class also carries the uppercase form's OWN lowercase,
which is what covers a letter spelled two ways in lower case: Greek `Σ`
lowercases to `σ`, but a word-final sigma is written `ς` and is equally a
lowercase of it. That reach is one-way — `σ → Σ → σ` never arrives at `ς` — and
left so deliberately, because `ς` is only correct at the end of a word, exactly
where the request's own last character sits; closing the other direction needs a
fold table.
It returns `null` — leaving the substring tests as the whole answer — for a
token holding a GLOB metacharacter (SQLite GLOB has no escape character, so an
unescaped `*` would become a wildcard matching every row) or a case mapping that
changes length (`ß` uppercases to `SS`, `İ` lowercases to two code points).
Neither has a single-character class that means the same thing, and a wrong
pattern is worse than an absent one.
The ASCII/Unicode branch is decided from the RAW token, not the normalized one:
normalization folds diacritics, so "Ça" arrives as "ca" and looks like plain