fix(search): locale-invariant Turkish case folding in every search path (#1640)

Turkish upper and lower case queries now return the same results everywhere a
title can be searched. Lower-casing the dotted capital "İ" (U+0130) leaves a
combining dot behind, so "İnş" and "inş" reached different search arms and
different results.

- Case folding is locale-invariant: `toLowerCase()`, never `toLocaleLowerCase()`,
  which under a Turkish or Azeri OS locale maps ASCII "I" to the dotless "ı".
- The Electron content search composes to NFC and drops the leftover combining
  marks before tokenizing, and its LIKE/GLOB pattern builders additionally spell
  the `'tr'`-locale İ forms, since SQLite LIKE folds only ASCII.
- A shared `foldSearchText` covers every in-memory filter: channel lists, the
  Xtream and Stalker catalogs, category filters, collections, the EPG guide, the
  command palette, sources, the playlist switcher and the download lists.
- Composing before the strip keeps canonically equivalent spellings equal while
  the fold stays accent-sensitive; the Turkish I/ı pair is deliberately left
  alone, as the FTS index does not fold it either.

Covered by a SQLite-backed spec over the real trigram index plus regression
cases in the affected renderer specs.

Closes #609.

Co-Authored-By: Justin Willhite <5132924+thejdubb02@users.noreply.github.com>
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
authored and GitHub committed 2026-09-20 21:14:02 +02:00
1 parent 975b1f7f74
commit b30c783e85
38 files changed
+736 -111

No files matched your search

+1
View File
@@ -60,6 +60,7 @@ export * from './lib/stream-format.enum';
export * from './lib/catalog-title-match.interface';
export * from './lib/theme.enum';
export * from './lib/season-marker.util';
export * from './lib/search-text-fold.util';
export * from './lib/stalker-account-info-dialog-data.interface';
export * from './lib/title-normalization.util';
export * from './lib/tmdb.interface';
@@ -0,0 +1,58 @@
import { foldSearchText } from './search-text-fold.util';
describe('foldSearchText', () => {
it('folds the Turkish dotted capital I onto a plain lower-case i (issue #609)', () => {
expect(foldSearchText('İnşaat Kanalı')).toBe('inşaat kanalı');
expect(
foldSearchText('İnşaat Kanalı').includes(foldSearchText('inş'))
).toBe(true);
expect(foldSearchText('İnş')).toBe(foldSearchText('inş'));
});
it.each([
['Ünlü Şef', 'ünlü şef'],
['ÇANAKKALE', 'çanakkale'],
['Первый канал HD', 'первый канал hd'],
['Ёлки', 'ёлки'],
['Йога для всех', 'йога для всех'],
['Ελλάδα Σήμερα', 'ελλάδα σήμερα'],
['Amélie', 'amélie'],
['Straße', 'straße'],
])(
'keeps precomposed letters of other scripts intact: %s',
(input, expected) => {
expect(foldSearchText(input)).toBe(expected);
}
);
it.each([
['Amélie', 'Amélie'],
['İnşaat', 'İnşaat'],
['Ёлки', 'Ёлки'],
['Йога', 'Йога'],
['Ünlü', 'Ünlü'],
['Ά', 'Ά'],
])(
'folds canonically equivalent spellings of %s to one string',
(precomposed, decomposed) => {
expect(foldSearchText(decomposed)).toBe(
foldSearchText(precomposed)
);
}
);
it('stays accent-sensitive, unlike the diacritic-folding SQL index', () => {
// The renderer filters were accent-sensitive before this helper and
// stay so: only marks that cannot compose are dropped. Accent-blind
// matching is the FTS index's own behaviour, not this fold's.
expect(foldSearchText('Amélie')).not.toBe(foldSearchText('Amelie'));
});
it('does not fold the dotless ı onto i', () => {
// The Turkish I/ı pair is a separate letter, not a case form of i;
// it is deliberately left alone (see the FTS index, which does not
// fold it either).
expect(foldSearchText('Işık')).toBe('işık');
expect(foldSearchText('ışık')).toBe('ışık');
});
});
@@ -0,0 +1,35 @@
/**
* Locale-invariant case fold for in-memory search filters (channel lists,
* catalog and category filters, source and download lists). Pure function,
* no Angular/Node dependencies.
*
* `toLowerCase()` alone is not enough for one visible input: the Turkish
* dotted capital "İ" (U+0130) lower-cases to "i" plus a combining dot above
* (U+0307), so `"İnşaat".toLowerCase().includes("inş")` is `false` even
* though the two spellings are the same word (issue #609).
*
* The fold therefore runs `toLowerCase()`, re-composes the result to NFC and
* drops whatever combining marks are left. Composing first is what makes the
* fold agree on canonically equivalent text: a provider title stored
* decomposed ("e" + U+0301) and a query typed precomposed ("é") reach the
* same string, and a Greek "Ά" lower-cases to a form NFC maps onto the same
* "ά" the user types. Only marks that cannot compose survive to be stripped —
* the dotted I's leftover dot among them — so precomposed letters (ç, ş, ü,
* é, ё, й) are preserved and the fold stays accent-sensitive, exactly as the
* filters were before. A lower-cased string of printable ASCII is already NFC
* and carries no marks, which is the fast path taken by most titles.
*
* `toLocaleLowerCase()` is deliberately not used: under a Turkish or Azeri
* OS locale it maps ASCII "I" to the dotless "ı", so the same list would
* filter differently per machine.
*/
const COMBINING_MARKS_REGEXP = /[̀-ͯ]/g;
const NON_PRINTABLE_ASCII_REGEXP = /[^ -~]/;
export function foldSearchText(value: string): string {
const lowered = value.toLowerCase();
return NON_PRINTABLE_ASCII_REGEXP.test(lowered)
? lowered.normalize('NFC').replace(COMBINING_MARKS_REGEXP, '')
: lowered;
}