mirror of
https://github.com/4gray/iptvnator.git
synced 2026-10-10 01:56:16 -08:00
fix(portals): cover a letter spelled two ways in lower case
Greek Σ lowercases to σ, but a word-final sigma is written ς and is equally a lowercase of it, so a class built only from the character in hand knew one spelling of two. Each class now also carries the uppercase form's own lowercase, which reaches the other one. One-way on purpose: σ → Σ → σ never arrives at ς. Left so because ς is only correct at the end of a word, which is exactly where the request's last character sits — the pair that occurs in real titles is covered, and closing the other direction needs a fold table. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
1 parent
ddb299e237
commit
257a53d9e7
4 files changed
+58
-14
No files matched your search
@@ -76,7 +76,7 @@ describe('title-sources.operations', () => {
|
||||
// code points — so the case is folded in JavaScript and handed
|
||||
// over one class per character. Without this a title stored in
|
||||
// any case other than the two below is simply never found.
|
||||
expect(scanQuery.params).toContain('*[оО][нН]*');
|
||||
expect(scanQuery.params).toContain('*[Оо][нН]*');
|
||||
expect(scanQuery.sql).toContain('instr');
|
||||
// Both the folded and the as-typed form, since the class is built
|
||||
// from the raw token and cannot cover a diacritic difference.
|
||||
|
||||
@@ -39,14 +39,30 @@ function hasSqlite(): boolean {
|
||||
const describeWithSqlite = hasSqlite() ? describe : describe.skip;
|
||||
|
||||
describe('caseInsensitiveGlobPattern', () => {
|
||||
it('builds a two-case class for each cased character', () => {
|
||||
expect(caseInsensitiveGlobPattern('Он')).toBe('*[оО][нН]*');
|
||||
it('builds a case class for each cased character', () => {
|
||||
expect(caseInsensitiveGlobPattern('Он')).toBe('*[Оо][нН]*');
|
||||
});
|
||||
|
||||
it('leaves uncased characters alone', () => {
|
||||
expect(caseInsensitiveGlobPattern('7')).toBe('*7*');
|
||||
});
|
||||
|
||||
it('covers a letter with two lowercase spellings', () => {
|
||||
// Greek Σ lowercases to σ, but a word-final sigma is written ς and is
|
||||
// just as much a lowercase of it. Going back down from the uppercase
|
||||
// form reaches the spelling the character in hand does not name.
|
||||
expect(caseInsensitiveGlobPattern('ς')).toBe('*[ςΣσ]*');
|
||||
});
|
||||
|
||||
it('reaches the second spelling in one direction only', () => {
|
||||
// σ → Σ → σ never arrives at ς, so a request spelled with a medial
|
||||
// sigma does not match a stored final one. Left as is: ς is only ever
|
||||
// correct at the end of a word, which is exactly where the request's
|
||||
// own last character sits, so the pair that occurs in real titles is
|
||||
// the one above. Closing the other direction needs a fold table.
|
||||
expect(caseInsensitiveGlobPattern('σ')).toBe('*[σΣ]*');
|
||||
});
|
||||
|
||||
it('refuses a token holding a GLOB metacharacter', () => {
|
||||
// SQLite GLOB has no escape character, so an unescaped `*` here would
|
||||
// silently become a wildcard and match every row in the table.
|
||||
@@ -95,6 +111,15 @@ describeWithSqlite('caseInsensitiveGlobPattern against SQLite', () => {
|
||||
expect(sqliteGlob('Дюна', pattern)).toBe(false);
|
||||
});
|
||||
|
||||
it('matches a Greek word written with either sigma', () => {
|
||||
const pattern = caseInsensitiveGlobPattern('ος') as string;
|
||||
|
||||
expect(sqliteGlob('ος', pattern)).toBe(true);
|
||||
expect(sqliteGlob('ΟΣ', pattern)).toBe(true);
|
||||
// The spelling a class built from the request alone would miss.
|
||||
expect(sqliteGlob('οσ', pattern)).toBe(true);
|
||||
});
|
||||
|
||||
it('matches Greek and accented Latin in any case', () => {
|
||||
expect(
|
||||
sqliteGlob('ΟΙ', caseInsensitiveGlobPattern('οι') as string)
|
||||
|
||||
@@ -33,24 +33,33 @@ export function caseInsensitiveGlobPattern(token: string): string | null {
|
||||
|
||||
let pattern = '*';
|
||||
for (const character of token) {
|
||||
const lower = character.toLowerCase();
|
||||
const upper = character.toUpperCase();
|
||||
// Going back down from the uppercase form catches letters with more
|
||||
// than one lowercase spelling: Greek Σ lowercases to σ, but ς is an
|
||||
// equally valid lowercase of it, and a class built only from the
|
||||
// character in hand would know just one of the two.
|
||||
const forms = new Set([
|
||||
character,
|
||||
character.toLowerCase(),
|
||||
upper,
|
||||
upper.toLowerCase(),
|
||||
]);
|
||||
|
||||
if (lower === upper) {
|
||||
if (forms.size === 1) {
|
||||
pattern += character;
|
||||
continue;
|
||||
}
|
||||
|
||||
if (
|
||||
[...lower].length !== 1 ||
|
||||
[...upper].length !== 1 ||
|
||||
GLOB_METACHARACTERS.test(lower) ||
|
||||
GLOB_METACHARACTERS.test(upper)
|
||||
[...forms].some(
|
||||
(form) =>
|
||||
[...form].length !== 1 || GLOB_METACHARACTERS.test(form)
|
||||
)
|
||||
) {
|
||||
return null;
|
||||
}
|
||||
|
||||
pattern += `[${lower}${upper}]`;
|
||||
pattern += `[${[...forms].join('')}]`;
|
||||
}
|
||||
|
||||
return `${pattern}*`;
|
||||
|
||||
@@ -533,10 +533,20 @@ The **scan** tier does fold it, because GLOB character classes are *not*
|
||||
ASCII-only — `patternCompare` reads them as UTF-8 code points, and
|
||||
`'Он' GLOB '*[Оо][Нн]*'` is true. `caseInsensitiveGlobPattern` therefore folds
|
||||
the case in JavaScript, where Unicode case mapping is real, and hands SQLite one
|
||||
`[lowerUpper]` class per character. It returns `null` — leaving the substring
|
||||
tests as the whole answer — for a token holding a GLOB metacharacter (SQLite
|
||||
GLOB has no escape character, so an unescaped `*` would become a wildcard) or a
|
||||
case mapping that changes length (`ß` uppercases to `SS`).
|
||||
class per character. Each class also carries the uppercase form's OWN lowercase,
|
||||
which is what covers a letter spelled two ways in lower case: Greek `Σ`
|
||||
lowercases to `σ`, but a word-final sigma is written `ς` and is equally a
|
||||
lowercase of it. That reach is one-way — `σ → Σ → σ` never arrives at `ς` — and
|
||||
left so deliberately, because `ς` is only correct at the end of a word, exactly
|
||||
where the request's own last character sits; closing the other direction needs a
|
||||
fold table.
|
||||
|
||||
It returns `null` — leaving the substring tests as the whole answer — for a
|
||||
token holding a GLOB metacharacter (SQLite GLOB has no escape character, so an
|
||||
unescaped `*` would become a wildcard matching every row) or a case mapping that
|
||||
changes length (`ß` uppercases to `SS`, `İ` lowercases to two code points).
|
||||
Neither has a single-character class that means the same thing, and a wrong
|
||||
pattern is worse than an absent one.
|
||||
|
||||
The ASCII/Unicode branch is decided from the RAW token, not the normalized one:
|
||||
normalization folds diacritics, so "Ça" arrives as "ca" and looks like plain
|
||||
|
||||
Reference in new issue
Block a user