diff --git a/.changes/xtream-vod-source-language-detection.md b/.changes/xtream-vod-source-language-detection.md new file mode 100644 index 000000000..42850cb27 --- /dev/null +++ b/.changes/xtream-vod-source-language-detection.md @@ -0,0 +1,10 @@ +--- +type: feature +area: xtream +--- + +The movie sources popover now recognizes more language tags: prefixes with +Unicode pipes, brackets or dashes ("EN │ …", "[EN] …", "EN - …"), Cyrillic +tags ("РУС | …") and MULTI. It also reads the language off category names +("EN | Netflix"), and a tag welded to a title ("EN|Movie") no longer hides +that copy from the other playlists' copies of the same movie. diff --git a/CLAUDE.md b/CLAUDE.md index 6d96f2fa6..2494b0d73 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -1224,7 +1224,7 @@ engine` (restart required) or **VOD Multi-Source** (alternative sources for a movie): -- Finds the same movie in the user's other imported playlists and adds a "Sources N" chip to the Xtream VOD action row (only when ≥1 alternative exists), plus a `.source-caption` line reporting where playback is coming from. The chip opens a 660px anchored CDK-overlay popover (`libs/ui/components/src/lib/vod-sources/`; not `MatMenu`, which caps its width at 280px), reused unchanged in the inline player's now-playing bar and on the playback-error screen. It opens ABOVE the chip (right edges aligned, pressed state on the chip while open), height-capped by the overlay's flexible bounding box so only the source list scrolls, and flips below when less than the overlay `minHeight` remains above; filter chips (All / Available / HD+ / language-prefix select) compose with the host search, "Available" auto-runs check-all when no verdicts exist, and expanded copy rows show a parsed language chip + raw stream title with diff-only tags ("same as above" for the parent's copy). Checks run through a 4-slot queue and settled verdicts are cached 10 min per movie+source (`VodSourceProbeCacheService`). Both chips are handed the same `matchKind` and `vodAutoFailover` and both write the setting back. The details-page chip badge counts TOTAL **copies** across all playlists (the in-player chip still counts alternatives); the caption ("also found in N other playlists") counts distinct **playlists** via `alternativePlaylistCount`, because the popover groups one portal's copies under that portal. The action row's Favorites and Download buttons are icon-only 64px squares: filled red heart when favorited, and a download idle icon → progress ring (real percent, indeterminate spin, paused-resume) → green done-checkmark whose click reveals the file (state read from the download manager; the labeled "Play from source" secondary is gone — provider playback for a downloaded movie goes through the Sources popover). +- Finds the same movie in the user's other imported playlists and adds a "Sources N" chip to the Xtream VOD action row (only when ≥1 alternative exists), plus a `.source-caption` line reporting where playback is coming from. The chip opens a 660px anchored CDK-overlay popover (`libs/ui/components/src/lib/vod-sources/`; not `MatMenu`, which caps its width at 280px), reused unchanged in the inline player's now-playing bar and on the playback-error screen. It opens ABOVE the chip (right edges aligned, pressed state on the chip while open), height-capped by the overlay's flexible bounding box so only the source list scrolls, and flips below when less than the overlay `minHeight` remains above; filter chips (All / Available / HD+ / language select) compose with the host search, "Available" auto-runs check-all when no verdicts exist, and expanded copy rows show a parsed language chip + raw stream title with diff-only tags ("same as above" for the parent's copy). A row's language is `vodSourceLanguage` (`libs/shared/interfaces/src/lib/vod-source-language.util.ts`): the title's own prefix (pipe incl. Unicode lookalikes, bracketed, or ALL-CAPS spaced-dash form; Latin/Cyrillic 2–4 letters + `MULTI`; only the legacy pipe form is permissive — bracket/dash matches must also pass `isKnownLanguageTag`, since those positions carry quality/rip tags like `[HD]`) wins, else the language the stream's visible categories unambiguously carry ("EN | Netflix" — discovery returns all category names — the FTS tier joins them with `group_concat(cat.name, char(31))` under the GROUP BY it already needs, the scan tier must NOT group (per-category uniqueness means sibling rows can carry different titles and grouping would drop a matching one) and its names merge in TypeScript, prefixed categories must agree, and category prefixes must pass `isKnownLanguageTag`, since `new`/`top`/`hot` are real ISO 639-3 codes but everyday category words; the route's own row reads the one category the route arrived through, overlaid late by the host's same-key `refreshRouteFacts` since cold/direct routes load categories after discovery). Both forms are parsed guesses: browse filter and chips only, never ranking/failover/dub-warning inputs. Recognition alone is not enough — `normalizeTitleKeys` must STRIP the same tag or the copy is never discovered, so its leading-tag rule shares `PROVIDER_PIPE_CLASS` and drops the required space after a pipe. It goes no further on purpose: a wrong guess costs a filter option, a wrong strip corrupts identity, and on 1.27M real titles a case-insensitive/Cyrillic pipe rule corrupts 349 keys ("Akira | 1988", "Момо | Momo" — the name sits in the tag position) while `–`/`—` on the dash branch amputates 14 subtitled titles. Verify such widenings against the real catalog before shipping them. Checks run through a 4-slot queue and settled verdicts are cached 10 min per movie+source (`VodSourceProbeCacheService`). Both chips are handed the same `matchKind` and `vodAutoFailover` and both write the setting back. The details-page chip badge counts TOTAL **copies** across all playlists (the in-player chip still counts alternatives); the caption ("also found in N other playlists") counts distinct **playlists** via `alternativePlaylistCount`, because the popover groups one portal's copies under that portal. The action row's Favorites and Download buttons are icon-only 64px squares: filled red heart when favorited, and a download idle icon → progress ring (real percent, indeterminate spin, paused-resume) → green done-checkmark whose click reveals the file (state read from the download manager; the labeled "Play from source" secondary is gone — provider playback for a downloaded movie goes through the Sources popover). - Scope v1 is **Xtream ↔ Xtream, movies only, Electron only**. Stalker never reaches the `content` table and M3U is a JSON blob whose search forces `content_type:'live'`; both are additive later since `VodSourceCandidate.portalType` already carries all three. In the PWA every entry point is gated off by a bridge `typeof` check and the chip renders nothing. - **Metadata provenance is the core contract.** Every field is `{value, provenance}` where `api`/`probe` are facts (plain tag), `parsed` is a title-regex guess (tag prefixed `~`, warn colour), and absent renders **no tag at all** plus a `check` chip. `factualOnly()` in `vod-source-metadata.util.ts` is the only accessor allowed for ranking/failover, so guesses are structurally unable to influence a decision. `VodSourceProbeStatus` separates `fail` (contacted and refused) from `unknown` (timed out / blocked / no capability) — an unchecked source is never shown as offline. Quality is derived from pixel **width** because letterboxing crops height — but a known height vetoes the answer on every tier, since cropping only removes lines: a taller frame is a different shape (1440×1080 anamorphic or 1600×900 are not 720p, 960×540 is not 576p) and gets no tag rather than a wrong one carrying `api` provenance. The route's OWN row is never resolved, so it takes its facts from the `get_vod_info` the page already loaded (`providerVodMetadataOf`, shared with the resolver) and picks them up via `refreshRouteFacts()` even when they arrive without changing the movie identity — otherwise `audioDiffersFactually` has nothing on one side and the dub warning cannot fire on a route-to-alternative switch. - Discovery (`DB_FIND_TITLE_SOURCES`, trigram FTS over `content_title_fts`) is lazy and returns only what the `content` table can prove; titles whose tokens are all shorter than three characters ("Up", "It") fall back to a scan, since the trigram tokenizer cannot index them at all. A source that is never read looks exactly like one that does not exist, so: the current playlist is excluded **in SQL** and duplicates collapse there too (`GROUP BY cat.playlist_id, c.xtream_id` before the limit — one playlist's dozens of identically ranked category rows would otherwise crowd out every alternative), and the scan matches an ASCII token as a whole word (`' ' || LOWER(title) || ' ' GLOB '*[^a-z0-9]it[^a-z0-9]*'`) ordered by title length **with no row limit** — FTS keeps its 60-row window because it ranks by relevance, while a scan cannot rank, and the GLOB reads every row regardless so a limit would only truncate the answer. The year gate covers BOTH match tiers: `normalizeTitleKeys` strips bracketed segments, so "Dune (1984)" normalizes identically to "Dune" and would otherwise be an _exact_ match for the 2021 film; a bracketed year is read out of the raw title and a stated disagreement rejects the row — but the two tiers read different forms: the base tier accepts bracketed or trailing (it just stripped a trailing year, the only thing separating "Dune 1984" from "Dune 2021"), while the exact tier reads bracketed ONLY, since reaching it means both titles are the same string and a trailing number is then part of the NAME ("Blade Runner 2049" against a metadata year of 2017 would otherwise vanish once enrichment lands). A non-ASCII token cannot be folded by `LOWER()` (ASCII-only) but CAN be by a GLOB character class (UTF-8 code points), so `caseInsensitiveGlobPattern` folds the case in JS and emits one `[lowerUpper]` class per character — returning `null`, leaving the two substring tests alone, for a GLOB metacharacter or a length-changing case map (`ß`→`SS`). The movie's own year comes from `releaseTagYear` (bracketed or trailing only), never `extractYear`: a year inside the NAME ("2001: A Space Odyssey") would fail every genuine 1968 copy at the year gate and move the pin key once enrichment lands. One row inside the excluded playlist is kept when the caller names it (`keepContentId`), because a pin can point at another copy in the playlist being viewed — the host reads the pin before discovery for exactly this. Resolution is deferred to click/pin/check because `content` stores no `container_extension` and `constructVodUrl` returns `''` without one — each alternative costs a live `get_vod_info` against the foreign playlist's credentials. diff --git a/apps/electron-backend/src/app/database/operations/title-sources-matching.spec.ts b/apps/electron-backend/src/app/database/operations/title-sources-matching.spec.ts index 18af42c84..c1c1bcfad 100644 --- a/apps/electron-backend/src/app/database/operations/title-sources-matching.spec.ts +++ b/apps/electron-backend/src/app/database/operations/title-sources-matching.spec.ts @@ -26,10 +26,55 @@ describe('title-sources.operations — confirmation and scoping', () => { posterUrl: 'https://cdn.example.com/dune.jpg', matchConfidence: 'exact', year: null, + categoryNames: [], }, ]); }); + it('keeps a stream whose sibling row carries a different title', async () => { + // `content` is unique per (category, type, stream), so one stream + // in two categories is two rows and nothing forces their titles to + // agree. The scan tier must therefore not GROUP: given a free + // choice SQLite can keep "Dune Part Two", the normalized + // confirmation rejects it, and the source disappears even though + // its sibling row says "Dune". "Up" routes to the scan tier. + const { db } = createDbMock([ + { ...duneRow, title: 'Up Above', category_names: 'DE | Kino' }, + { ...duneRow, title: 'Up', category_names: 'EN | Movies' }, + ]); + + const matches = await findTitleSources(db, { title: 'Up' }); + + expect(matches).toHaveLength(1); + expect(matches[0].title).toBe('Up'); + // ...and the rejected sibling's category still describes the same + // stream, so its name is merged in rather than dropped. + expect(matches[0].categoryNames).toEqual([ + 'DE | Kino', + 'EN | Movies', + ]); + }); + + it('splits aggregated category names on the unit separator', async () => { + // `group_concat`'s default `,` appears in real category names, so + // the queries aggregate with char(31) and the split must read + // exactly that — a comma inside a name stays part of the name. + const { db } = createDbMock([ + { + ...duneRow, + category_names: + 'EN | Netflix\u001fAction, Adventure\u001fEN | Netflix', + }, + ]); + + const matches = await findTitleSources(db, { title: 'Dune' }); + + expect(matches[0].categoryNames).toEqual([ + 'EN | Netflix', + 'Action, Adventure', + ]); + }); + it('confirms a year-stripped candidate as a fuzzy match', async () => { const { db } = createDbMock([{ ...duneRow, title: 'Dune 1984' }]); @@ -162,7 +207,9 @@ describe('title-sources.operations — confirmation and scoping', () => { expect( `${requested} confirms ${matches.length} of ${rows.length}` - ).toBe(`${requested} confirms ${rows.length} of ${rows.length}`); + ).toBe( + `${requested} confirms ${rows.length} of ${rows.length}` + ); } }); diff --git a/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts b/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts index 1cf2da513..9230ab3c2 100644 --- a/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts +++ b/apps/electron-backend/src/app/database/operations/title-sources.operations.spec.ts @@ -75,6 +75,29 @@ describe('title-sources.operations', () => { expect(compiledQuery(fts.all).sql).toContain('LIMIT'); }); + it('returns category names without grouping the scan tier', async () => { + // The renderer reads a language prefix off category names + // ("EN | Netflix"), so both tiers must return them — but only the + // FTS tier may GROUP: `content` is unique per (category, type, + // stream), so grouping the scan would let SQLite keep an + // arbitrary row's title and reject a stream a sibling row would + // have confirmed. + const scan = createDbMock([]); + await findTitleSources(scan.db, { title: 'It' }); + const scanQuery = compiledQuery(scan.all); + expect(scanQuery.sql).toContain('cat.name AS category_names'); + expect(scanQuery.sql).not.toContain('GROUP BY'); + + const fts = createDbMock([]); + await findTitleSources(fts.db, { title: 'Dune' }); + const ftsQuery = compiledQuery(fts.all); + // char(31): `group_concat`'s default `,` occurs inside real names. + expect(ftsQuery.sql).toContain('group_concat(cat.name, char(31))'); + expect(ftsQuery.sql).toContain( + 'GROUP BY cat.playlist_id, c.xtream_id' + ); + }); + it('can still find a short non-ASCII title', async () => { // SQLite's LOWER() is ASCII-only, so folding "Он" to "он" never // happened and the film stayed invisible in the Sources chip. diff --git a/apps/electron-backend/src/app/database/operations/title-sources.operations.ts b/apps/electron-backend/src/app/database/operations/title-sources.operations.ts index 932661e1f..ecc458d16 100644 --- a/apps/electron-backend/src/app/database/operations/title-sources.operations.ts +++ b/apps/electron-backend/src/app/database/operations/title-sources.operations.ts @@ -35,6 +35,42 @@ interface TitleSourceRow { category_xtream_id: number; playlist_id: string; playlist_name: string; + /** + * Visible category names for this row: `group_concat`-joined on the FTS + * tier, which returns one row per stream, and a single name on the scan + * tier, which returns one row per category. Both are read through + * `splitCategoryNames`, and the caller merges them per stream. + */ + category_names: string | null; +} + +/** + * The separator for aggregated category names. `group_concat`'s default `,` + * appears in real category names ("Action, Adventure"), so splitting on it + * would shred them; the ASCII unit separator cannot. Written as `char(31)` in + * SQL and `\u001f` in the split. + */ +const CATEGORY_NAME_SEPARATOR = '\u001f'; + +/** + * Aggregated category names back into a list. Duplicates are dropped here + * rather than with `DISTINCT` in SQL, because SQLite refuses a custom + * separator on a DISTINCT aggregate — and duplicate names change nothing for + * a reader that only compares language prefixes. + */ +function splitCategoryNames(aggregated: string | null): string[] { + if (!aggregated) { + return []; + } + + const names: string[] = []; + for (const raw of aggregated.split(CATEGORY_NAME_SEPARATOR)) { + const name = raw.trim(); + if (name && !names.includes(name)) { + names.push(name); + } + } + return names; } export interface FindTitleSourcesRequest { @@ -197,6 +233,15 @@ function scanCandidateQuery( ), sql` AND ` ); + // Deliberately NOT grouped, unlike the FTS tier. `content` is unique per + // (category, type, stream), so one stream sitting in several categories + // is several rows and nothing forces their titles to agree. Grouping + // would hand SQLite a free choice of which title to keep, and the + // normalized confirmation below would then reject the whole stream on a + // title that a sibling row would have matched — the source vanishes. + // The FTS tier can afford the grouping because its window makes it + // necessary; this tier takes no window at all, so it keeps every row and + // merges their category names afterwards. return sql` SELECT c.id AS content_id, @@ -205,7 +250,8 @@ function scanCandidateQuery( c.poster_url AS poster_url, cat.xtream_id AS category_xtream_id, cat.playlist_id AS playlist_id, - p.name AS playlist_name + p.name AS playlist_name, + cat.name AS category_names FROM content AS c INNER JOIN categories AS cat ON c.category_id = cat.id INNER JOIN playlists AS p ON cat.playlist_id = p.id @@ -236,7 +282,8 @@ function ftsCandidateQuery(matchQuery: string, excludePlaylist: SQL) { c.poster_url AS poster_url, cat.xtream_id AS category_xtream_id, cat.playlist_id AS playlist_id, - p.name AS playlist_name + p.name AS playlist_name, + group_concat(cat.name, char(31)) AS category_names FROM content_title_fts INNER JOIN content AS c ON c.id = content_title_fts.rowid INNER JOIN categories AS cat ON c.category_id = cat.id @@ -298,6 +345,35 @@ export async function findTitleSources( // One playlist can list the same film in several categories; the user // thinks of that as one source. const seen = new Set(); + // Every category the stream sits in AMONG THE ROWS THE QUERY MATCHED, + // merged. The FTS tier already joined those into one row, so this is a + // no-op there; the scan tier returns a row per category and must not + // group (see `scanCandidateQuery`), so this is where its names come + // together. Rows the confirmation later rejects still contribute — a + // differently titled sibling row is the same stream, and its category + // says the same thing about the language. + // + // A sibling row whose title did NOT match is invisible here, so a stream + // listed under a localized title in another category can look + // unanimous when it is not. That is deliberate. The field is a guess + // feeding a chip and a browse filter — ranking and failover cannot reach + // it — and completing it costs real latency: measured on a 3.9 GB + // catalog, resolving every category through a correlated subquery takes + // the discovery query from 0.74 s to 2.0 s, and a second bounded lookup + // for the same 60 streams takes 19.7 s. A detail page pays that on every + // open, to correct a cosmetic guess in a shape that occurs 0 times in + // 2.7M rows. + const categoryNames = new Map(); + for (const row of rows) { + const key = `${row.playlist_id}:${row.xtream_id}`; + const names = categoryNames.get(key) ?? []; + for (const name of splitCategoryNames(row.category_names)) { + if (!names.includes(name)) { + names.push(name); + } + } + categoryNames.set(key, names); + } for (const row of rows) { // SQL already excluded these; this keeps the guarantee a property of @@ -357,6 +433,7 @@ export async function findTitleSources( posterUrl: row.poster_url, matchConfidence: exactMatch ? 'exact' : 'fuzzy', year: rowYear, + categoryNames: categoryNames.get(dedupeKey) ?? [], }); } diff --git a/docs/architecture/vod-multi-source.md b/docs/architecture/vod-multi-source.md index 834b70262..96acd7d21 100644 --- a/docs/architecture/vod-multi-source.md +++ b/docs/architecture/vod-multi-source.md @@ -156,9 +156,8 @@ fit contract: The chip row composes with the host search (AND): **All (N)** resets the chip filters and states the total copy count, **Available** keeps only sources whose probe verified them, **HD+** keeps sources whose quality tag reads -1080p or better, and the language select is built from the `EN|`-style -prefixes actually present in the raw titles. Two of these encode a decision -worth writing down: +1080p or better, and the language select is built from the languages actually +present in the list. Two of these encode a decision worth writing down: - "Available" is strict: only `probe.status === 'ok'` passes. An unchecked source must never pass a filter with that name — and because checks are @@ -174,14 +173,98 @@ worth writing down: When search or filters reduce the list, a muted "X of N" counter appears, and groups with no matching copy disappear entirely. +### Where a row's language comes from + +The language the select and the copy chips read is `vodSourceLanguage` +(`libs/shared/interfaces/src/lib/vod-source-language.util.ts`): the stream +title's own prefix when it has one, else the language the stream's categories +agree on. Both are parsed guesses — they feed browsing only, and neither +ranking, failover nor the dub warning can reach them (`factualOnly` and +`audioDiffersFactually` read other fields entirely). + +`titleLanguagePrefix` accepts the three shapes panels actually write: a 2–4 +letter Latin or Cyrillic tag (plus the five-letter `MULTI` marker) before a +pipe **or any of its Unicode lookalikes** (`¦`, `│`, `|`, …— visually +identical to `|`, invisible to a literal match), a bracketed tag at the very +start (`[EN] Movie`), and an ALL-uppercase tag before a **spaced** dash +(`EN - Movie`). The dash form is stricter on purpose: dashes are ordinary +title punctuation, and "Up - the movie" or "X-Men" must not read as a +language. Strictness is tiered per form: only the legacy pipe form is taken +at its word, while a bracket or dash match must ALSO pass +`isKnownLanguageTag` — those positions are where quality and rip tags live +("[HD]", "[CAM]", "NEW - "), and since a title prefix outranks the category +language, a fabricated one would mask a real category-derived language and +get the row excluded by the very filter meant to find it. + +Recognizing a prefix is only half the job: the same tag also has to be +STRIPPED by `normalizeTitleKeys`, or the tagged copy and the bare one never +match and the row is never discovered at all. Its leading-tag rule therefore +shares this file's pipe set (`PROVIDER_PIPE_CLASS`) and, on the pipe branch, +needs no space after the separator — so "EN │ Fallout", "EN|Fallout" and +"|FR|VO|Le dernier empereur" reach the same key as the bare title. + +It stops there, and the asymmetry with the reader above is the point. A wrong +GUESS costs a junk option in a filter; a wrong STRIP corrupts a film's +identity everywhere the key is used. Measured against 1.27M real catalog +titles: making the pipe branch case-insensitive or Cyrillic corrupts 349 keys +and rescues none, because "Akira | 1988" and "Момо | Momo" put the film's own +name in the tag position; widening the dash branch to `–`/`—` amputates 14 +subtitled titles ("1918 – A Batalha de Kruty"); and Cyrillic before a dash +does not occur at all. So normalization keeps its uppercase-Latin, +space-required form everywhere except the pipe separator itself, and the +reader is free to be permissive because the gate in front of its riskier +forms — and the fact that a row only appears once it HAS matched — keeps a +bad guess cosmetic. + +The category path exists because many panels tag the CATEGORY ("EN | Netflix", +"DE | Apple TV") and leave stream titles bare. Discovery returns every visible +category name a stream sits in, and the two query tiers get there +differently: the FTS tier already groups per `(playlist_id, xtream_id)` for +its window, so it joins them with `group_concat(cat.name, char(31))` — +`char(31)` because the default `,` appears inside real category names — while +the scan tier must NOT group. `content` is unique per +`(category, type, stream)`, so one stream in several categories is several +rows whose titles need not agree; grouping would let SQLite keep an arbitrary +one and the normalized confirmation would then reject a stream that a sibling +row would have matched. The scan therefore returns a row per category and the +names are merged per stream in TypeScript. + +Either tier only ever sees the categories of rows the query MATCHED, so a +stream also listed under a localized title in another category can look +unanimous when it is not. Deliberate: the field is a guess feeding a chip and +a browse filter, and completing it is expensive — on a 3.9 GB catalog, +resolving every category through a correlated subquery takes discovery from +0.74 s to 2.0 s (a cost every detail-page open pays), and a second bounded +lookup for the same 60 streams takes 19.7 s, to correct a cosmetic guess in a +shape that occurs 0 times in 2.7M rows. + +Either way +`unambiguousCategoryLanguage` reduces them: categories without a recognized +language prefix abstain, all prefixed ones must agree, and a conflict yields +nothing. Category prefixes must additionally pass `isKnownLanguageTag`. +Measured on a real catalog, that gate is what keeps "VOD" (5,245 movies), +"KIDS" (1,010), "SHOW" and "WWE" out of a list of LANGUAGES; "NEW |", +"TOP |" and "VIP |" are everyday shapes too, and `new`, `top` and `hot` are +even assigned ISO 639-3 codes — which is why the gate is an +`Intl.DisplayNames` check for two-letter codes plus a curated list for +longer tags rather than a registry lookup. Only the pipe title form stays +permissive: a tag before a pipe in a movie title is overwhelmingly a +language, and tightening the legacy form would drop filter options that work +today. The route's own row reads the one category the route arrived through +(`VodMultiSourceMovie.categoryName`), which is the visible one; it loads late +on cold/direct routes, so the host's same-key refresh +(`refreshRouteFacts`) overlays it — and provider facts — onto the existing +route row without a rediscovery. + ### The popover: copy rows Expanding "N copies in this playlist" lists EVERY copy of the group — including the one the parent row already shows — as compact `app-vod-source-copy-row`s indented under the parent's text column. The playlist's name and monogram are not repeated: the primary text is the -provider's own raw stream title (mono), with the parsed `EN|`/`RU|` language -prefix promoted to a chip before it. The tag row shows only values that +provider's own raw stream title (mono), with the copy's parsed language +(`vodSourceLanguage` — title prefix first, category fallback) promoted to a +chip before it. The tag row shows only values that DIFFER from the parent's copy (identical container/codec are omitted), so what distinguishes a copy is the only thing on the line; the copy identical to the parent shows a muted "same as above" note instead. The fuzzy-match diff --git a/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.spec.ts b/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.spec.ts index 05f3fc68b..b71d9e418 100644 --- a/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.spec.ts +++ b/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.spec.ts @@ -68,3 +68,46 @@ describe('VodSourceDiscoveryService — failure logging', () => { .join('\n'); } }); + +describe('VodSourceDiscoveryService — candidate mapping', () => { + afterEach(() => { + delete (window as { electron?: unknown }).electron; + }); + + it('derives the category language, leaving conflicts and noise empty', async () => { + (window as { electron?: unknown }).electron = { + dbFindTitleSources: jest + .fn() + .mockResolvedValue([ + row(1, ['EN | Netflix', 'EN | Action']), + row(2, ['EN | Netflix', 'DE | Cinema']), + row(3, ['TOP | 250']), + row(4, undefined), + ]), + }; + const service = new VodSourceDiscoveryService(); + + const result = await service.discover({ + title: 'Dune', + currentPlaylistId: 'playlist-0', + }); + + expect(result.sources.map((source) => source.categoryLanguage)).toEqual( + ['EN', null, null, null] + ); + }); + + function row(id: number, categoryNames: string[] | undefined) { + return { + playlistId: `playlist-${id}`, + playlistName: `Portal ${id}`, + categoryId: id, + xtreamId: 100 + id, + title: 'Dune', + posterUrl: null, + matchConfidence: 'exact' as const, + year: null, + categoryNames, + }; + } +}); diff --git a/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.ts b/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.ts index 78fd6384f..6e85c7165 100644 --- a/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.ts +++ b/libs/portal/shared/data-access/src/lib/multi-source/vod-source-discovery.service.ts @@ -1,8 +1,9 @@ import { Injectable } from '@angular/core'; -import type { - VodSourceCandidate, - VodSourceCandidateRow, - VodSourceMatchKind, +import { + unambiguousCategoryLanguage, + type VodSourceCandidate, + type VodSourceCandidateRow, + type VodSourceMatchKind, } from '@iptvnator/shared/interfaces'; import { createLogger } from '@iptvnator/portal/shared/util'; import { parseTitleMetadata } from './vod-source-metadata.util'; @@ -99,5 +100,8 @@ function toCandidate(row: VodSourceCandidateRow): VodSourceCandidate { quality: parsed.quality, codec: parsed.codec, audio: parsed.audio, + // A guess like everything else here: the language the stream's + // categories agree on, standing in when the title has no prefix. + categoryLanguage: unambiguousCategoryLanguage(row.categoryNames), }; } diff --git a/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.component.ts b/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.component.ts index 51c26dc8c..8c011c0be 100644 --- a/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.component.ts +++ b/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.component.ts @@ -259,8 +259,15 @@ export class VodDetailsRouteComponent implements OnInit, OnDestroy { ); }); /** Movie identity for multi-source discovery; null until a title exists */ - private readonly multiSourceMovie = computed(() => - resolveVodMultiSourceMovie({ + private readonly multiSourceMovie = computed(() => { + // Electron stores categories under `name`, the live API under + // `category_name` — the same duality the fallback view reads. + const category = this.selectedCategory() as { + name?: string; + category_name?: string; + } | null; + + return resolveVodMultiSourceMovie({ playlistId: this.xtreamStore.currentPlaylist()?.id, // `title` is the alias the Xtream data source actually writes // (createPlaylist maps name -> title), so reading only `name` @@ -273,8 +280,9 @@ export class VodDetailsRouteComponent implements OnInit, OnDestroy { catalogItem: this.selectedCatalogItem(), containerExtension: this.selectedItem()?.movie_data?.container_extension, - }) - ); + categoryName: category?.name ?? category?.category_name ?? null, + }); + }); readonly selectedVodInfo = computed(() => { const item = this.selectedItem(); return item && hasUsableXtreamVodMetadata(item) diff --git a/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.harness.ts b/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.harness.ts index 3340916dd..95198a26f 100644 --- a/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.harness.ts +++ b/libs/portal/xtream/feature/src/lib/vod-details/vod-details-route.harness.ts @@ -58,6 +58,7 @@ export function createVodDetailsRouteStubs() { addRecentItem: jest.fn(), cancelDetailsRequest: jest.fn(), vodStreamsPlaylistId: signal(null), + vodCategoriesPlaylistId: signal(null), downloadsAvailable: signal(false), downloads: signal([]), isDownloaded: jest.fn().mockReturnValue(false), @@ -198,6 +199,7 @@ export async function configureVodDetailsRouteTestBed( addRecentItem: stubs.addRecentItem, cancelDetailsRequest: stubs.cancelDetailsRequest, vodStreamsPlaylistId: stubs.vodStreamsPlaylistId, + vodCategoriesPlaylistId: stubs.vodCategoriesPlaylistId, }, }, { diff --git a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.spec.ts b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.spec.ts index 9bd08742b..1a605d9c5 100644 --- a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.spec.ts +++ b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.spec.ts @@ -85,6 +85,33 @@ describe('currentSourceRow', () => { expect(audioDiffersFactually(row, alternative('ru'))).toBe(false); }); + it('reads the language off the route category, title prefix first', () => { + const viaCategory = resolveVodMultiSourceMovie({ + ...MOVIE, + vodInfo: null, + categoryName: 'EN | Netflix', + }); + expect(currentSourceRow(viaCategory as never).categoryLanguage).toBe( + 'EN' + ); + + // A category prefix that is not a language stays out entirely. + const viaNoise = resolveVodMultiSourceMovie({ + ...MOVIE, + vodInfo: null, + categoryName: 'TOP | 250', + }); + expect(currentSourceRow(viaNoise as never).categoryLanguage).toBeNull(); + + const withoutCategory = resolveVodMultiSourceMovie({ + ...MOVIE, + vodInfo: null, + }); + expect( + currentSourceRow(withoutCategory as never).categoryLanguage + ).toBeNull(); + }); + it('does not call a codec change a dub change', () => { const movie = resolveVodMultiSourceMovie({ ...MOVIE, diff --git a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.ts b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.ts index 84a7aa313..aaa640d15 100644 --- a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.ts +++ b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-current-row.ts @@ -1,5 +1,8 @@ import { applyApiMetadata } from '@iptvnator/portal/shared/data-access'; -import type { VodSourceCandidate } from '@iptvnator/shared/interfaces'; +import { + unambiguousCategoryLanguage, + type VodSourceCandidate, +} from '@iptvnator/shared/interfaces'; import type { VodMultiSourceMovie } from './vod-multi-source-identity'; /** @@ -30,7 +33,23 @@ export function currentSourceRow( rawTitle: movie.title, matchConfidence: 'exact', year: movie.year ?? null, + categoryLanguage: routeCategoryLanguage(movie), }; return movie.metadata ? applyApiMetadata(row, movie.metadata) : row; } + +/** + * The route row's category-derived language: what the one category the route + * arrived through states, if that is a language. Alternatives read this off + * every category the DB knows; the route only knows the visible one. Shared + * with the host's refresh path so the two cannot drift — the category loads + * late on cold/direct routes and arrives without changing the movie key. + */ +export function routeCategoryLanguage( + movie: VodMultiSourceMovie +): string | null { + return unambiguousCategoryLanguage( + movie.categoryName ? [movie.categoryName] : null + ); +} diff --git a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host-session.spec.ts b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host-session.spec.ts index c3d8a7fc7..ea3e3793e 100644 --- a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host-session.spec.ts +++ b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host-session.spec.ts @@ -144,6 +144,46 @@ describe('VodMultiSourceHostService — session lifecycle', () => { expect(discovery.discover).toHaveBeenCalledTimes(2); }); + it('overlays a late-arriving category language on the route row', async () => { + movie.set(MOVIE_A); + await flushEffects(); + expect(rowFor(CURRENT_A_ID)?.categoryLanguage ?? null).toBeNull(); + + // Cold/direct routes load categories after discovery ran, and the + // category name is outside the movie key on purpose — the same-key + // refresh is the only path that can deliver it to the route row. + movie.set({ ...MOVIE_A, categoryName: 'EN | Netflix' }); + await flushEffects(); + expect(rowFor(CURRENT_A_ID)?.categoryLanguage).toBe('EN'); + expect(discovery.discover).toHaveBeenCalledTimes(1); + + // A category that names no language never fabricates one. + movie.set({ ...MOVIE_A, categoryName: 'TOP | 250' }); + await flushEffects(); + expect(rowFor(CURRENT_A_ID)?.categoryLanguage ?? null).toBeNull(); + }); + + it('keeps a category that lands while discovery is in flight', async () => { + const first = createDeferred(); + discovery.discover.mockReturnValueOnce(first.promise); + + movie.set(MOVIE_A); + while (discovery.discover.mock.calls.length === 0) { + TestBed.tick(); + await Promise.resolve(); + } + + // The route row does not exist until discovery answers, so this + // same-key emission has nothing to refresh — it must not be lost. + movie.set({ ...MOVIE_A, categoryName: 'EN | Netflix' }); + await flushEffects(); + + first.resolve({ sources: [], matchKind: 'title-year' }); + await flushEffects(); + + expect(rowFor(CURRENT_A_ID)?.categoryLanguage).toBe('EN'); + }); + it('does not burn a source the user only selected', async () => { // One alternative, so the route copy is the ONLY fallback left and // the outcome cannot depend on how candidates are ranked. diff --git a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host.service.ts b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host.service.ts index 1dc149c4f..9001e6046 100644 --- a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host.service.ts +++ b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-host.service.ts @@ -33,7 +33,10 @@ import { import type { VodMultiSourceSwitchNotice } from './vod-multi-source-notice'; import { createVodSourceCounts } from './vod-multi-source-counts'; import { createCheckQueue } from './vod-multi-source-check-queue'; -import { currentSourceRow } from './vod-multi-source-current-row'; +import { + currentSourceRow, + routeCategoryLanguage, +} from './vod-multi-source-current-row'; import { probeSource } from './vod-multi-source-probe'; import { pinnedSourceAwaitingPlay, @@ -194,37 +197,56 @@ export class VodMultiSourceHostService { } /** - * Overlay the provider's facts onto the route's own row, in place. + * Overlay what arrived late onto the route's own row, in place. * - * The row is built when discovery runs, which on a sparse panel happens - * before `get_vod_info` answers — and if that answer adds no year and no - * TMDB id, the movie key does not change, so nothing rebuilds the row and - * it keeps stating nothing. Every comparison against it is then one-sided: - * the dub warning in particular cannot fire at all. + * The row is built when discovery runs, and two things routinely land + * AFTER that without changing the movie key: the provider's facts (a + * sparse panel's `get_vod_info` adds no year and no TMDB id) and the + * route category (cold/direct routes load categories late). Nothing + * rebuilds the row for either, so both are refreshed here — otherwise + * the dub warning stays one-sided and the row's category-derived + * language never appears, letting the language filter hide the very + * source that is playing. * * Merged onto the existing row rather than rebuilt from the movie, so a * probe result already sitting on it survives. + * + * A same-key emission can also arrive while discovery is still in + * flight, before the route row exists to refresh. That is not lost, + * and the reason is the `findSource` call below: it reads the + * controller's sources SIGNAL inside the `bind()` effect, so the + * publish that finally creates the row re-runs the effect, which lands + * here again with the movie's latest reading. Wrapping this read in + * `untracked()` would silently break that redelivery — a session spec + * pins it. */ private refreshRouteFacts(movie: VodMultiSourceMovie): void { - const facts = movie.metadata; const routeSourceId = this.routeSourceId; - if (!facts || !routeSourceId) { - return; - } - - const factsKey = JSON.stringify(facts); - if (this.routeFactsKey === factsKey) { - return; - } - - const existing = this.controller.findSource(routeSourceId); + const existing = routeSourceId + ? this.controller.findSource(routeSourceId) + : undefined; if (!existing) { return; } - this.routeFactsKey = factsKey; - this.controller.updateSource(applyApiMetadata(existing, facts)); - this.publish(); + let next = existing; + + const categoryLanguage = routeCategoryLanguage(movie); + if ((existing.categoryLanguage ?? null) !== categoryLanguage) { + next = { ...next, categoryLanguage }; + } + + const facts = movie.metadata; + const factsKey = facts ? JSON.stringify(facts) : null; + if (facts && this.routeFactsKey !== factsKey) { + this.routeFactsKey = factsKey; + next = applyApiMetadata(next, facts); + } + + if (next !== existing) { + this.controller.updateSource(next); + this.publish(); + } } /** diff --git a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-identity.ts b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-identity.ts index 72e413b5d..72718f1ce 100644 --- a/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-identity.ts +++ b/libs/portal/xtream/feature/src/lib/vod-details/vod-multi-source-identity.ts @@ -29,12 +29,17 @@ export interface VodMultiSourceMovie { * would restart the search for no gain. */ metadata?: ProviderVodMetadata; + /** + * Name of the category the route arrived through ("EN | Netflix"), so the + * route's own row can carry a category-derived language like discovered + * alternatives do. Outside the movie key for the same reason as + * `metadata`: it describes presentation, not which film this is. + */ + categoryName?: string | null; } type CatalogItem = - | (Partial & { title?: string }) - | null - | undefined; + (Partial & { title?: string }) | null | undefined; /** * Resolve the movie identity, or null while it is not yet knowable. @@ -52,6 +57,8 @@ export function resolveVodMultiSourceMovie(input: { catalogItem: CatalogItem; /** From `movie_data`, which sits beside `info` rather than inside it. */ containerExtension?: string | null; + /** Name of the category the route arrived through, when known. */ + categoryName?: string | null; }): VodMultiSourceMovie | null { const { playlistId, vodId, vodInfo, catalogItem } = input; @@ -79,6 +86,7 @@ export function resolveVodMultiSourceMovie(input: { // supplies the real one. year: extractYear(vodInfo?.releasedate) ?? releaseTagYear(title), tmdbId: vodInfo?.tmdb_id, + categoryName: input.categoryName ?? null, // Only once `get_vod_info` has landed. Before that the row simply // states nothing, which is the honest reading of "not known yet". metadata: vodInfo diff --git a/libs/shared/interfaces/src/index.ts b/libs/shared/interfaces/src/index.ts index 1328de28b..1abab7749 100644 --- a/libs/shared/interfaces/src/index.ts +++ b/libs/shared/interfaces/src/index.ts @@ -52,6 +52,7 @@ export * from './lib/stalker-account-info-dialog-data.interface'; export * from './lib/title-normalization.util'; export * from './lib/tmdb.interface'; export * from './lib/vod-source.interface'; +export * from './lib/vod-source-language.util'; export * from './lib/vod-source-match-key.util'; export * from './lib/xtream-account-info-dialog-data.interface'; export * from './lib/xtream-category.interface'; diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts index f365e454c..0191c4269 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.spec.ts @@ -100,12 +100,8 @@ describe('provider tag stripping', () => { it('strips long and compound leading tags', () => { expect(normalizeTitle('EXYU| Fallout')).toBe('fallout'); expect(normalizeTitle('MULTI| Breaking Bad')).toBe('breaking bad'); - expect(normalizeTitle('4K-DE - The Pitt (2025) (US)')).toBe( - 'the pitt' - ); - expect(normalizeTitle('AR-SUBS - Fallout (2024) (US)')).toBe( - 'fallout' - ); + expect(normalizeTitle('4K-DE - The Pitt (2025) (US)')).toBe('the pitt'); + expect(normalizeTitle('AR-SUBS - Fallout (2024) (US)')).toBe('fallout'); expect(normalizeTitle('4K-OSN+ - The Last of Us (2023)')).toBe( 'the last of us' ); @@ -117,6 +113,40 @@ describe('provider tag stripping', () => { ); }); + it('reads pipe lookalikes as the pipe they look like', () => { + // `│`, `¦` and `|` are indistinguishable from `|` in a catalog, so + // the same tag must not survive in one playlist and vanish in + // another — the two copies would never match as the same film. + expect(normalizeTitle('EN │ Fallout')).toBe('fallout'); + expect(normalizeTitle('DE ¦ Fallout')).toBe('fallout'); + expect(normalizeTitle('MULTI|Fallout')).toBe('fallout'); + }); + + it('strips a pipe tag welded to the title', () => { + // "|FR|VO|Le dernier empereur" — the wrapped tag goes first, then + // "VO|" with no space after it. + expect(normalizeTitle('|FR|VO|Le dernier empereur')).toBe( + 'le dernier empereur' + ); + expect(normalizeTitle('EN|Fallout')).toBe('fallout'); + }); + + it('keeps a name that only looks like a tag before a pipe', () => { + // Pins the UPPERCASE rule on the pipe branch. Relaxing it there is + // tempting — nothing but a tag precedes a pipe, surely — but measured + // against 1.27M real catalog titles a case-insensitive (or Cyrillic) + // pipe rule corrupted 349 keys and rescued none: "name | year" and + // the Russian "localized | original" convention both put the film's + // own name in the tag position. + // The base tier drops the trailing year, so the name is what must + // survive; the exact tier shows the whole string it came from. + expect(normalizeTitle('Akira | 1988')).toBe('akira'); + expect(normalizeTitleKeys('Akira | 1988').exact).toBe('akira 1988'); + expect(normalizeTitle('Coco | 2017')).toBe('coco'); + expect(normalizeTitle('Момо | Momo')).toBe('момо momo'); + expect(normalizeTitle('Мумия | The Mummy')).toBe('мумия the mummy'); + }); + it('keeps bare 4-5 char words before a spaced dash (real titles)', () => { expect(normalizeTitle('DUNE - Part Two')).toBe('dune part two'); expect(normalizeTitle('ALIEN - Covenant')).toBe('alien covenant'); @@ -170,40 +200,88 @@ describe('provider tag stripping', () => { }); const pittCorpus = [ - 'The Pitt (2025)_sub', 'The Pitt (2025)-it', 'The Pitt (2025)', - 'The Pitt (Hindi)', 'The Pitt (2025) 4K', 'The Pitt (2025) DE', - 'The Pitt (2025) ES', 'The Pitt (2025) FR', 'The Pitt (2025)_eng', - 'The Pitt [MULTI-SUB]', 'The Pitt (2025) (4K DV)', 'GR - The Pitt', - '4K-DE - The Pitt (2025) (US)', '4K-TR - The Pitt (2025) (US)', - 'AR-SUBS - The Pitt (2025) (US)', 'DE - The Pitt (2025) (US)', - 'ALB| The Pitt', 'EXYU| The Pitt', '|ALB| The Pitt', '|DE| The Pitt', + 'The Pitt (2025)_sub', + 'The Pitt (2025)-it', + 'The Pitt (2025)', + 'The Pitt (Hindi)', + 'The Pitt (2025) 4K', + 'The Pitt (2025) DE', + 'The Pitt (2025) ES', + 'The Pitt (2025) FR', + 'The Pitt (2025)_eng', + 'The Pitt [MULTI-SUB]', + 'The Pitt (2025) (4K DV)', + 'GR - The Pitt', + '4K-DE - The Pitt (2025) (US)', + '4K-TR - The Pitt (2025) (US)', + 'AR-SUBS - The Pitt (2025) (US)', + 'DE - The Pitt (2025) (US)', + 'ALB| The Pitt', + 'EXYU| The Pitt', + '|ALB| The Pitt', + '|DE| The Pitt', ]; const falloutCorpus = [ - 'Fallout', 'DE - Fallout (2024)', 'Fallout (2024) - 4K', - 'Fallout (2024) FR-EN', 'Fallout (2024) Multi', 'Fallout (2024)_fr', - 'Fallout_esp', 'Fallout (4K)', '4K-AMZ - Fallout (2024) (US)', - 'AL - Fallout (2024)', 'AMZ - Fallout (2024) (US)', - 'AR-DE - Fallout (US)', 'LA - Fallout', 'EN| Fallout - 4K', - 'MULTI| Fallout - 4K', 'Fallout ( مدبلج )', 'Fallout (Telugu)', - '|EN| Fallout - 4K', '|MULTI| Fallout', '|TR| Fallout', + 'Fallout', + 'DE - Fallout (2024)', + 'Fallout (2024) - 4K', + 'Fallout (2024) FR-EN', + 'Fallout (2024) Multi', + 'Fallout (2024)_fr', + 'Fallout_esp', + 'Fallout (4K)', + '4K-AMZ - Fallout (2024) (US)', + 'AL - Fallout (2024)', + 'AMZ - Fallout (2024) (US)', + 'AR-DE - Fallout (US)', + 'LA - Fallout', + 'EN| Fallout - 4K', + 'MULTI| Fallout - 4K', + 'Fallout ( مدبلج )', + 'Fallout (Telugu)', + '|EN| Fallout - 4K', + '|MULTI| Fallout', + '|TR| Fallout', ]; const lastOfUsCorpus = [ - 'The Last of Us', 'The Last Of Us', 'The Last of Us (2023) 4K', - 'The Last of Us (2023) AF', 'The Last of Us_tr', - 'The Last of Us--esp', 'The Last of Us-DE', 'The Last of Us-esp', - 'The Last of Us [L]', 'The Last of Us ( HD )', - '4K-OSN+ - The Last of Us (2023)', 'IS - The Last of Us (2023) (US)', - 'RU - The Last of Us', 'ALB| The Last of Us', + 'The Last of Us', + 'The Last Of Us', + 'The Last of Us (2023) 4K', + 'The Last of Us (2023) AF', + 'The Last of Us_tr', + 'The Last of Us--esp', + 'The Last of Us-DE', + 'The Last of Us-esp', + 'The Last of Us [L]', + 'The Last of Us ( HD )', + '4K-OSN+ - The Last of Us (2023)', + 'IS - The Last of Us (2023) (US)', + 'RU - The Last of Us', + 'ALB| The Last of Us', ]; const breakingBadCorpus = [ - 'Breaking Bad', 'Breaking Bad (2008)_fr', 'Breaking Bad (US)_msub', - 'Breaking Bad_it', 'Breaking Bad-DE', 'Breaking Bad-eng', - 'Breaking Bad ( عائلي )', 'Breaking Bad (Pure)', - 'Breaking Bad - Multi', 'Breaking Bad ES', 'AR-DE - Breaking Bad', - 'EN| Breaking Bad SUB', 'MULTI| Breaking Bad', 'AR| Breaking Bad', + 'Breaking Bad', + 'Breaking Bad (2008)_fr', + 'Breaking Bad (US)_msub', + 'Breaking Bad_it', + 'Breaking Bad-DE', + 'Breaking Bad-eng', + 'Breaking Bad ( عائلي )', + 'Breaking Bad (Pure)', + 'Breaking Bad - Multi', + 'Breaking Bad ES', + 'AR-DE - Breaking Bad', + 'EN| Breaking Bad SUB', + 'MULTI| Breaking Bad', + 'AR| Breaking Bad', + // Pipe lookalikes and a lowercase tag: identical on screen to the + // forms above, so they have to reach the same key. + 'EN │ Breaking Bad', + 'DE ¦ Breaking Bad', + 'FR|Breaking Bad', ]; it.each([ diff --git a/libs/shared/interfaces/src/lib/title-normalization.util.ts b/libs/shared/interfaces/src/lib/title-normalization.util.ts index b77104612..e48657567 100644 --- a/libs/shared/interfaces/src/lib/title-normalization.util.ts +++ b/libs/shared/interfaces/src/lib/title-normalization.util.ts @@ -29,12 +29,27 @@ const QUALITY_TAGS = new Set([ 'dubbed', ]); +/** + * The pipe and the display lookalikes providers use interchangeably with it + * (`¦`, `│`, fullwidth `|`, …). They are visually identical to `|` in a + * catalog, so a rule that reads only U+007C leaves the same tag stripped in + * one playlist and welded to the title in another — and the two copies then + * never match as the same film. + * + * Exported because `vod-source-language.util.ts` reads the same separator to + * decide a row's language: one set, so the "is this a tag" answer cannot + * differ between matching and display. + */ +export const PROVIDER_PIPE_CLASS = '[|¦│┃❘∣⏐⎪︱︳丨|]'; + /** * Wrapped tag at the very start of a provider title: "|DE| ARD", * "|MULTI| Fallout". The lookahead requires a letter in the tag so a * numeric fragment can never be treated as one. */ -const WRAPPED_TAG_PREFIX = /^\s*\|(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}\|\s*/; +const WRAPPED_TAG_PREFIX = new RegExp( + `^\\s*${PROVIDER_PIPE_CLASS}(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}${PROVIDER_PIPE_CLASS}\\s*` +); /** * Leading channel/language prefix like "EN - ", "DE| ", "FR: ", including @@ -50,13 +65,27 @@ const WRAPPED_TAG_PREFIX = /^\s*\|(?=[0-9+]*[A-Z])[A-Z0-9+]{2,5}\|\s*/; * - pipe ("EXYU| "): compound OR 2–5 chars — a pipe is a strong tag signal * - colon ("EN: "): 2–3 chars — longer acronyms are franchise titles * ("NCIS: LA") + * + * The pipe branch alone does not require a space after the separator: + * "|FR|VO|Le dernier empereur" welds the tag to the title, and no real title + * begins with a short word immediately followed by a pipe. + * + * The UPPERCASE-only restriction stays on every branch, pipe included. It is + * tempting to drop it there on the theory that nothing but a tag precedes a + * pipe — 1.27M real catalog titles say otherwise, and in two ways at once: + * "Akira | 1988" and "Coco | 2017" put the film's NAME before the pipe and + * the year after it, and Russian catalogs write "Момо | Momo", + * "Мумия (2026) | Lee Cronin's The Mummy" — the localized title, then the + * original. Case-insensitivity (or a Cyrillic alphabet) turns every one of + * those names into a "tag" and strips it; measured against that corpus it + * corrupted 349 keys and rescued none. */ const SEG = '(?=[0-9+]*[A-Z])[A-Z0-9+]'; const COMPOUND_TAG = `${SEG}{2,5}(?:-${SEG}{2,6}){1,2}`; const LANGUAGE_PREFIX = new RegExp( '^(?:' + `(?:${COMPOUND_TAG}|${SEG}{2,3})\\s*-\\s+` + - `|(?:${COMPOUND_TAG}|${SEG}{2,5})\\s*\\|\\s+` + + `|(?:${COMPOUND_TAG}|${SEG}{2,5})\\s*${PROVIDER_PIPE_CLASS}\\s*` + `|${SEG}{2,3}\\s*:\\s+` + ')' ); @@ -69,10 +98,48 @@ const LANGUAGE_PREFIX = new RegExp( * suffixes ("NCIS: LA"). US/USA/UK/LA are deliberately absent. */ const TRAILING_TAG_VOCABULARY = new Set([ - 'AF', 'AL', 'ALB', 'AR', 'BY', 'DE', 'DUB', 'EN', 'ENG', 'ES', 'ESP', - 'EXYU', 'FR', 'FRA', 'GE', 'GR', 'HU', 'IN', 'IR', 'IS', 'IT', 'ITA', - 'KA', 'KU', 'LAT', 'ML', 'MSUB', 'MULTI', 'NL', 'PL', 'PT', 'RO', 'RU', - 'SC', 'SE', 'SUB', 'SUBS', 'SW', 'TA', 'TL', 'TR', 'TUR', + 'AF', + 'AL', + 'ALB', + 'AR', + 'BY', + 'DE', + 'DUB', + 'EN', + 'ENG', + 'ES', + 'ESP', + 'EXYU', + 'FR', + 'FRA', + 'GE', + 'GR', + 'HU', + 'IN', + 'IR', + 'IS', + 'IT', + 'ITA', + 'KA', + 'KU', + 'LAT', + 'ML', + 'MSUB', + 'MULTI', + 'NL', + 'PL', + 'PT', + 'RO', + 'RU', + 'SC', + 'SE', + 'SUB', + 'SUBS', + 'SW', + 'TA', + 'TL', + 'TR', + 'TUR', ]); /** diff --git a/libs/shared/interfaces/src/lib/vod-source-language.util.spec.ts b/libs/shared/interfaces/src/lib/vod-source-language.util.spec.ts new file mode 100644 index 000000000..7e3d52011 --- /dev/null +++ b/libs/shared/interfaces/src/lib/vod-source-language.util.spec.ts @@ -0,0 +1,174 @@ +import { + isKnownLanguageTag, + titleLanguagePrefix, + unambiguousCategoryLanguage, + vodSourceLanguage, +} from './vod-source-language.util'; + +describe('titleLanguagePrefix', () => { + it('reads the tag before a pipe, any case', () => { + expect(titleLanguagePrefix('EN| Night of the Living Dead')).toBe('EN'); + expect(titleLanguagePrefix('ALB |Some Movie')).toBe('ALB'); + expect(titleLanguagePrefix(' ru| Ночь')).toBe('RU'); + }); + + it('reads Cyrillic tags', () => { + expect(titleLanguagePrefix('РУС | Фильм')).toBe('РУС'); + expect(titleLanguagePrefix('укр| Фільм')).toBe('УКР'); + }); + + it('reads Unicode pipe lookalikes as the separator', () => { + expect(titleLanguagePrefix('EN │ Movie')).toBe('EN'); + expect(titleLanguagePrefix('DE ¦ Der Film')).toBe('DE'); + expect(titleLanguagePrefix('FR|Le Film')).toBe('FR'); + }); + + it('reads a bracketed tag at the start', () => { + expect(titleLanguagePrefix('[EN] Movie')).toBe('EN'); + expect(titleLanguagePrefix('(ru) Ночь')).toBe('RU'); + // A bracket deeper in the title is not a prefix. + expect(titleLanguagePrefix('Movie [EN]')).toBeNull(); + }); + + it('requires the brackets to pair', () => { + // Openers and closers matched independently would accept these. + expect(titleLanguagePrefix('[EN) Movie')).toBeNull(); + expect(titleLanguagePrefix('(DE] Film')).toBeNull(); + }); + + it('reads an uppercase tag before a spaced dash, and only that form', () => { + expect(titleLanguagePrefix('EN - Movie')).toBe('EN'); + expect(titleLanguagePrefix('РУС - Фильм')).toBe('РУС'); + // Lowercase before a dash is a title word ("Up - the movie"). + expect(titleLanguagePrefix('Up - the movie')).toBeNull(); + // Unspaced dashes are ordinary punctuation. + expect(titleLanguagePrefix('X-Men')).toBeNull(); + expect(titleLanguagePrefix('EN-Movie')).toBeNull(); + }); + + it('gates bracket and dash tags to known languages', () => { + // Brackets and dashes are where quality and rip tags live; taking + // them at their word would fabricate an "HD" language that outranks + // and masks a real category-derived one. + expect(titleLanguagePrefix('[HD] Dune')).toBeNull(); + expect(titleLanguagePrefix('[UHD] Dune')).toBeNull(); + expect(titleLanguagePrefix('[CAM] Dune')).toBeNull(); + expect(titleLanguagePrefix('NEW - Dune')).toBeNull(); + expect(titleLanguagePrefix('VIP - Dune')).toBeNull(); + // The legacy pipe form stays permissive — tightening it would drop + // filter options that work today. + expect(titleLanguagePrefix('SNF| Dune')).toBe('SNF'); + }); + + it('accepts the MULTI marker despite its five letters', () => { + expect(titleLanguagePrefix('MULTI | Movie')).toBe('MULTI'); + expect(titleLanguagePrefix('Multi| Movie')).toBe('MULTI'); + }); + + it('rejects titles whose separator is not a language marker', () => { + // Five letters is a word, not a language code. + expect(titleLanguagePrefix('NIGHT| of something')).toBeNull(); + expect(titleLanguagePrefix('Night of the Living Dead')).toBeNull(); + expect(titleLanguagePrefix('4K| Movie')).toBeNull(); + expect(titleLanguagePrefix(undefined)).toBeNull(); + expect(titleLanguagePrefix('')).toBeNull(); + }); +}); + +describe('isKnownLanguageTag', () => { + it('accepts assigned two-letter ISO codes in any case', () => { + expect(isKnownLanguageTag('EN')).toBe(true); + expect(isKnownLanguageTag('de')).toBe(true); + expect(isKnownLanguageTag('uk')).toBe(true); + }); + + it('rejects unassigned two-letter tokens', () => { + expect(isKnownLanguageTag('HD')).toBe(false); + expect(isKnownLanguageTag('XX')).toBe(false); + }); + + it('accepts curated long tags, Latin and Cyrillic', () => { + expect(isKnownLanguageTag('ENG')).toBe(true); + expect(isKnownLanguageTag('DEU')).toBe(true); + expect(isKnownLanguageTag('LAT')).toBe(true); + expect(isKnownLanguageTag('РУС')).toBe(true); + expect(isKnownLanguageTag('MULTI')).toBe(true); + }); + + it('rejects the content-type prefixes categories actually use', () => { + // The four the gate turns away most often on a real catalog; without + // it the language select offers "VOD" and "KIDS". + for (const tag of ['VOD', 'KIDS', 'SHOW', 'WWE']) { + expect(isKnownLanguageTag(tag)).toBe(false); + } + }); + + it('rejects everyday category words that are real ISO 639-3 codes', () => { + // `new`, `top` and `hot` are assigned in ISO 639-3, which is exactly + // why validation is curated instead of registry-driven. + expect(isKnownLanguageTag('NEW')).toBe(false); + expect(isKnownLanguageTag('TOP')).toBe(false); + expect(isKnownLanguageTag('HOT')).toBe(false); + expect(isKnownLanguageTag('VIP')).toBe(false); + expect(isKnownLanguageTag('KIDS')).toBe(false); + }); +}); + +describe('unambiguousCategoryLanguage', () => { + it('reads the language every prefixed category agrees on', () => { + expect( + unambiguousCategoryLanguage(['EN | Netflix', 'EN | Action']) + ).toBe('EN'); + }); + + it('lets unprefixed categories abstain rather than veto', () => { + expect( + unambiguousCategoryLanguage(['EN | Netflix', 'Netflix 4K']) + ).toBe('EN'); + }); + + it('yields nothing on a conflict', () => { + expect( + unambiguousCategoryLanguage(['EN | Netflix', 'DE | Cinema']) + ).toBeNull(); + }); + + it('rejects category prefixes that are not languages', () => { + expect(unambiguousCategoryLanguage(['NEW | 2024'])).toBeNull(); + expect(unambiguousCategoryLanguage(['TOP | 250'])).toBeNull(); + expect(unambiguousCategoryLanguage(['VIP | Cinema'])).toBeNull(); + // ...while a real language beside them still reads through. + expect( + unambiguousCategoryLanguage(['NEW | 2024', 'DE | Apple TV']) + ).toBe('DE'); + }); + + it('handles empty and missing input', () => { + expect(unambiguousCategoryLanguage([])).toBeNull(); + expect(unambiguousCategoryLanguage(null)).toBeNull(); + expect(unambiguousCategoryLanguage(undefined)).toBeNull(); + expect(unambiguousCategoryLanguage([null, ''])).toBeNull(); + }); +}); + +describe('vodSourceLanguage', () => { + it('prefers the title prefix over the category language', () => { + expect( + vodSourceLanguage({ + rawTitle: 'RU| Movie', + categoryLanguage: 'EN', + }) + ).toBe('RU'); + }); + + it('falls back to the category language when the title says nothing', () => { + expect( + vodSourceLanguage({ rawTitle: 'Movie', categoryLanguage: 'EN' }) + ).toBe('EN'); + }); + + it('yields nothing when neither side knows', () => { + expect(vodSourceLanguage({ rawTitle: 'Movie' })).toBeNull(); + expect(vodSourceLanguage({})).toBeNull(); + }); +}); diff --git a/libs/shared/interfaces/src/lib/vod-source-language.util.ts b/libs/shared/interfaces/src/lib/vod-source-language.util.ts new file mode 100644 index 000000000..c7908d9b4 --- /dev/null +++ b/libs/shared/interfaces/src/lib/vod-source-language.util.ts @@ -0,0 +1,303 @@ +import { PROVIDER_PIPE_CLASS } from './title-normalization.util'; + +/** + * Language prefixes for VOD multi-source rows. + * + * Panels rarely state a spoken language as a fact; what they do instead is + * prefix stream titles ("EN| Movie") and category names ("EN | Netflix") with + * a short tag. Everything here is therefore a GUESS by construction: it feeds + * the browse filter and the copy-row chips, and is structurally excluded from + * ranking and failover (`factualOnly` never reads it). + * + * Three tiers of strictness, matched to each form's noise profile: + * + * - The PIPE title form is permissive, as it always has been. A tag before a + * pipe in a MOVIE title is overwhelmingly a language — titles do not start + * with "VIP |" — and tightening the legacy form would drop filter options + * that work today. + * - The BRACKET and DASH title forms are gated by `isKnownLanguageTag`. They + * are new (nothing to regress) and their prefixes skew toward quality and + * rip tags — "[HD] Dune", "NEW - Dune" — which would not only pollute the + * select but, since a title prefix outranks the category language, mask a + * real one and get the row excluded by the very filter meant to find it. + * - Category names are noisiest, so a prefix read off a category always + * passes `isKnownLanguageTag`. Measured on a real catalog, the gate is what + * keeps "VOD" (5,245 movies), "KIDS" (1,010), "SHOW" and "WWE" out of a + * list of LANGUAGES; "NEW | 2024" and "TOP | 250" are everyday shapes too, + * and `new`, `top` and `hot` are even assigned ISO 639-3 codes — which is + * why the gate is curated rather than a registry lookup. Empty beats wrong: + * an unrecognized tag yields no language rather than a wrong filter option. + */ + +/** + * The pipe and its display lookalikes, shared with title normalization so + * that what counts as a tag separator here is exactly what counts as one + * when two copies are matched as the same film. + */ +const PIPE = PROVIDER_PIPE_CLASS; + +/** + * A candidate language token: 2–4 letters of ONE script, or the `MULTI` + * marker panels use for multi-audio releases. Single-script on purpose — + * a mixed-script "word" before a pipe is decoration, not a tag. Digits are + * excluded, so "4K |" never reads as a language. + */ +const TOKEN = '(?:[A-Za-z]{2,4}|[А-Яа-яЁё]{2,4}|[Mm][Uu][Ll][Tt][Ii])'; + +/** `EN| Movie`, `ru │ Фильм`, `MULTI ¦ Movie` — any case before a pipe. */ +const PIPE_FORM = new RegExp(`^\\s*(${TOKEN})\\s*${PIPE}`); + +/** + * `[EN] Movie`, `(RU) Фильм` — a bracketed tag at the very start. + * + * The two bracket styles are separate alternatives rather than one class of + * openers and one of closers: the latter pairs them independently, so a + * malformed `[EN)` would be read as a tag. + */ +const BRACKET_FORM = new RegExp( + `^\\s*(?:\\[\\s*(${TOKEN})\\s*\\]|\\(\\s*(${TOKEN})\\s*\\))` +); + +/** + * `EN - Movie` — dash-separated, and deliberately stricter than the pipe + * form: the tag must be ALL uppercase and the dash spaced on both sides. + * Dashes are ordinary title punctuation ("X-Men", "Up - the movie"), so a + * lowercase or unspaced form is a title, not a tag. + */ +const DASH_FORM = new RegExp( + `^\\s*((?:[A-Z]{2,4}|[А-ЯЁ]{2,4}|MULTI))\\s+[-–—]\\s+` +); + +/** + * `EN| Night of the Living Dead` → `EN`. + * + * The provider's language convention for stream titles, in the three shapes + * seen in the wild: tag-before-pipe (including Unicode pipe lookalikes), + * bracketed tag, and uppercase tag before a spaced dash. Anything longer than + * four letters is a word that happens to precede a separator, not a language. + * + * Only the pipe form is taken at its word; a bracket or dash match must also + * name a KNOWN language, because those positions are where quality and rip + * tags live ("[HD]", "[CAM]") — see the tier rationale in the file header. + */ +export function titleLanguagePrefix( + rawTitle: string | null | undefined +): string | null { + const title = rawTitle ?? ''; + + const pipe = capturedTag(PIPE_FORM.exec(title)); + if (pipe) { + return pipe.toUpperCase(); + } + + const gated = + capturedTag(BRACKET_FORM.exec(title)) ?? + capturedTag(DASH_FORM.exec(title)); + return gated && isKnownLanguageTag(gated) ? gated.toUpperCase() : null; +} + +/** + * The tag out of whichever alternative matched. Read positionally rather + * than as group 1, because a form with several alternatives (brackets) has + * one group per alternative and only one of them is filled. + */ +function capturedTag(match: RegExpExecArray | null): string | null { + return match?.slice(1).find((group) => group !== undefined) ?? null; +} + +/** + * `Intl.DisplayNames` is ES2021 and this workspace compiles against the + * es2018 lib, so it is reached through a narrow shim (the same pattern + * `Intl.Locale` uses in the metadata util). Absent — which no supported + * runtime actually is — two-letter validation declines rather than guesses. + */ +const DisplayNamesCtor = ( + Intl as unknown as { + DisplayNames?: new ( + locales: string[], + options: { type: string; fallback: string } + ) => { of(code: string): string | undefined }; + } +).DisplayNames; + +let languageNames: { of(code: string): string | undefined } | null | undefined; + +/** + * Whether a two-letter token is an assigned ISO 639-1 code. + * + * With `fallback: 'code'`, `DisplayNames.of()` answers a real language with + * its name ("en" → "English") and echoes an unassigned code back unchanged + * ("hd" → "hd") — so "name differs from code" is precisely "this language + * exists". The two-letter space is safe for this trick; the three-letter + * space is NOT (ISO 639-3 assigns `new`, `top` and `hot`), which is why + * longer tokens go through the curated list instead. + */ +function isAssignedTwoLetterCode(lower: string): boolean { + if (!DisplayNamesCtor) { + return false; + } + + try { + languageNames ??= new DisplayNamesCtor(['en'], { + type: 'language', + fallback: 'code', + }); + const name = languageNames.of(lower); + return typeof name === 'string' && name.toLowerCase() !== lower; + } catch { + // Structurally invalid tag, or a runtime without language data — + // declining beats guessing. + return false; + } +} + +/** + * Three-and-more-letter tags panels actually use, plus the Cyrillic + * shorthands `Intl` cannot validate. Curated rather than derived from ISO + * 639-2/3: those registries assign codes to `new`, `top` and `hot`, so + * validating against them would turn everyday category prefixes into + * languages. + */ +const KNOWN_LONG_TAGS = new Set([ + // ISO 639-2 pairs (B/T) and common panel spellings, Latin script + 'eng', + 'rus', + 'ukr', + 'bel', + 'kaz', + 'ger', + 'deu', + 'fra', + 'fre', + 'spa', + 'esp', + 'lat', + 'ita', + 'por', + 'tur', + 'ara', + 'pol', + 'nld', + 'dut', + 'swe', + 'nor', + 'dan', + 'fin', + 'gre', + 'ell', + 'hun', + 'cze', + 'ces', + 'svk', + 'slo', + 'srb', + 'srp', + 'hrv', + 'cro', + 'bul', + 'ron', + 'rum', + 'alb', + 'sqi', + 'mkd', + 'bos', + 'heb', + 'hin', + 'vie', + 'tha', + 'kor', + 'jpn', + 'chi', + 'zho', + 'per', + 'fas', + 'aze', + 'kat', + 'geo', + 'hye', + 'arm', + 'uzb', + 'lit', + 'lav', + 'est', + 'multi', + // Cyrillic shorthands + 'ру', + 'уа', + 'рус', + 'укр', + 'бел', + 'каз', + 'анг', + 'англ', + 'нем', + 'фра', + 'исп', + 'ита', + 'пол', + 'тур', + 'узб', + 'арм', + 'груз', + 'азе', +]); + +/** + * Whether a parsed prefix names a language, as opposed to any short word a + * category happens to start with. + */ +export function isKnownLanguageTag(tag: string): boolean { + const lower = tag.toLowerCase(); + if (KNOWN_LONG_TAGS.has(lower)) { + return true; + } + return /^[a-z]{2}$/.test(lower) && isAssignedTwoLetterCode(lower); +} + +/** + * The language a stream's categories agree on, or null. + * + * A stream usually sits in several categories of one playlist ("EN | Netflix" + * and "EN | Action"), and the aggregation is what makes the answer honest: + * every prefixed category must name the SAME language, and that language must + * pass `isKnownLanguageTag`. Categories without a recognized language prefix + * abstain rather than veto — "EN | Netflix" plus "Netflix 4K" still reads EN, + * while "EN | Netflix" plus "DE | Cinema" is a conflict and yields nothing. + */ +export function unambiguousCategoryLanguage( + categoryNames: readonly (string | null | undefined)[] | null | undefined +): string | null { + if (!categoryNames?.length) { + return null; + } + + let language: string | null = null; + for (const name of categoryNames) { + const prefix = titleLanguagePrefix(name); + if (!prefix || !isKnownLanguageTag(prefix)) { + continue; + } + if (language !== null && language !== prefix) { + return null; + } + language = prefix; + } + return language; +} + +/** + * The language shown and filtered on for one source row. + * + * The stream's own title prefix is the more specific signal and wins; the + * category-derived language stands in only when the title says nothing. Both + * are guesses — this feeds the filter select and the copy-row chip, never a + * ranking decision. + */ +export function vodSourceLanguage(source: { + rawTitle?: string | null; + categoryLanguage?: string | null; +}): string | null { + return ( + titleLanguagePrefix(source.rawTitle) ?? source.categoryLanguage ?? null + ); +} diff --git a/libs/shared/interfaces/src/lib/vod-source.interface.ts b/libs/shared/interfaces/src/lib/vod-source.interface.ts index 46745e19c..37c4f84f8 100644 --- a/libs/shared/interfaces/src/lib/vod-source.interface.ts +++ b/libs/shared/interfaces/src/lib/vod-source.interface.ts @@ -90,6 +90,12 @@ export interface VodSourceCandidateRow { posterUrl: string | null; matchConfidence: VodSourceMatchConfidence; year: number | null; + /** + * Names of every visible category this stream sits in within its + * playlist. Carried so the renderer can read a language prefix off them + * ("EN | Netflix") when the stream title itself states none. + */ + categoryNames?: string[]; } /** @@ -132,6 +138,15 @@ export interface VodSourceCandidate { * the dub change". */ audioLanguage?: VodSourceField; + /** + * Language read off the stream's category names ("EN | Netflix"), when + * every prefixed category agrees (`unambiguousCategoryLanguage`). A + * guess by nature — it stands in for a missing title prefix in the + * browse filter and chips, and is never read by ranking, failover or + * the dub warning (`factualOnly` and `audioDiffersFactually` cannot + * reach it). + */ + categoryLanguage?: string | null; /** ISO timestamp of the last failed playback attempt, if any. */ lastFailedAt?: string; diff --git a/libs/ui/components/src/lib/vod-sources/vod-source-copy-row.component.ts b/libs/ui/components/src/lib/vod-sources/vod-source-copy-row.component.ts index 59db2ec8c..e7a13b385 100644 --- a/libs/ui/components/src/lib/vod-sources/vod-source-copy-row.component.ts +++ b/libs/ui/components/src/lib/vod-sources/vod-source-copy-row.component.ts @@ -9,7 +9,7 @@ import { MatIcon } from '@angular/material/icon'; import { MatTooltip } from '@angular/material/tooltip'; import { TranslatePipe } from '@ngx-translate/core'; import { VodSourceDescriptor } from '@iptvnator/shared/interfaces'; -import { titleLanguagePrefix } from './vod-source-filtering.util'; +import { vodSourceLanguage } from './vod-source-filtering.util'; import { VodSourceTagListComponent } from './vod-source-tag-list.component'; import { buildVodSourceCopyTags } from './vod-source-tags'; @@ -39,9 +39,7 @@ export class VodSourceCopyRowComponent { readonly playRequested = output(); readonly checkRequested = output(); - readonly language = computed(() => - titleLanguagePrefix(this.source().rawTitle) - ); + readonly language = computed(() => vodSourceLanguage(this.source())); readonly title = computed( () => this.source().rawTitle || `#${this.source().contentId}` diff --git a/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.spec.ts b/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.spec.ts index a4b7cf0d3..baaeb6655 100644 --- a/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.spec.ts +++ b/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.spec.ts @@ -30,10 +30,13 @@ function source( } describe('titleLanguagePrefix', () => { + // The full parsing matrix lives with the util in shared/interfaces; + // this asserts the re-export still reads the classic pipe form. it('reads the uppercase prefix before a pipe', () => { expect(titleLanguagePrefix('EN| Night of the Living Dead')).toBe('EN'); expect(titleLanguagePrefix('ALB |Some Movie')).toBe('ALB'); expect(titleLanguagePrefix(' ru| Ночь')).toBe('RU'); + expect(titleLanguagePrefix('РУС | Фильм')).toBe('РУС'); }); it('rejects titles whose pipe is not a language marker', () => { @@ -56,6 +59,15 @@ describe('collectLanguagePrefixes', () => { ]); expect(prefixes).toEqual(['EN', 'RU']); }); + + it('includes the category-derived language of untagged titles', () => { + const prefixes = collectLanguagePrefixes([ + source({ rawTitle: 'Plain title', categoryLanguage: 'DE' }), + // The title prefix outranks a conflicting category language. + source({ rawTitle: 'EN| Movie', categoryLanguage: 'RU' }), + ]); + expect(prefixes).toEqual(['DE', 'EN']); + }); }); describe('qualityPixels', () => { @@ -71,9 +83,9 @@ describe('qualityPixels', () => { describe('sourceMatchesFilters', () => { it('passes everything with no filters active', () => { expect(hasActiveVodSourceFilters(EMPTY_VOD_SOURCE_FILTERS)).toBe(false); - expect( - sourceMatchesFilters(source(), EMPTY_VOD_SOURCE_FILTERS) - ).toBe(true); + expect(sourceMatchesFilters(source(), EMPTY_VOD_SOURCE_FILTERS)).toBe( + true + ); }); it('available keeps only probe-verified sources', () => { @@ -129,6 +141,23 @@ describe('sourceMatchesFilters', () => { ).toBe(false); }); + it('language falls back to the category-derived language', () => { + const filters = { ...EMPTY_VOD_SOURCE_FILTERS, language: 'EN' }; + expect( + sourceMatchesFilters( + source({ rawTitle: 'Movie', categoryLanguage: 'EN' }), + filters + ) + ).toBe(true); + // A title prefix always outranks the category. + expect( + sourceMatchesFilters( + source({ rawTitle: 'RU| Movie', categoryLanguage: 'EN' }), + filters + ) + ).toBe(false); + }); + it('filters compose with AND', () => { const filters = { availableOnly: true, @@ -158,8 +187,8 @@ describe('noChecksRunYet', () => { expect( noChecksRunYet([source(), source({ probe: { status: 'probing' } })]) ).toBe(false); - expect( - noChecksRunYet([source({ probe: { status: 'fail' } })]) - ).toBe(false); + expect(noChecksRunYet([source({ probe: { status: 'fail' } })])).toBe( + false + ); }); }); diff --git a/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.ts b/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.ts index 9a3f2b662..c81476e54 100644 --- a/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.ts +++ b/libs/ui/components/src/lib/vod-sources/vod-source-filtering.util.ts @@ -1,4 +1,8 @@ -import type { VodSourceDescriptor } from '@iptvnator/shared/interfaces'; +import { + titleLanguagePrefix, + vodSourceLanguage, + type VodSourceDescriptor, +} from '@iptvnator/shared/interfaces'; /** * Filter logic for the sources popover's chip row. @@ -6,14 +10,20 @@ import type { VodSourceDescriptor } from '@iptvnator/shared/interfaces'; * Pure functions over descriptors so the menu component stays a thin view and * the composition rules (filters AND each other and the host search) can be * tested without a fixture. + * + * The language a row filters on is `vodSourceLanguage`: the stream title's + * own prefix ("EN| Movie") when it has one, else the language its categories + * unambiguously carry ("EN | Netflix") — both parsed guesses, both browse-only. */ +export { titleLanguagePrefix, vodSourceLanguage }; + export interface VodSourceFilterState { /** Keep only copies whose probe verified them reachable. */ availableOnly: boolean; /** Keep only copies whose stated quality is 1080p or better. */ hdOnly: boolean; - /** Keep only copies carrying this title language prefix; null = all. */ + /** Keep only copies carrying this language (title or category); null = all. */ language: string | null; } @@ -29,28 +39,13 @@ export function hasActiveVodSourceFilters( return filters.availableOnly || filters.hdOnly || filters.language !== null; } -/** - * `EN| Night of the Living Dead` → `EN`. - * - * Providers prefix the raw stream title with a short uppercase language tag - * before a pipe; that convention is the only language signal most panels - * emit. Anything longer than four letters is a title that happens to contain - * a pipe, not a language. - */ -export function titleLanguagePrefix( - rawTitle: string | null | undefined -): string | null { - const match = /^\s*([A-Za-z]{2,4})\s*\|/.exec(rawTitle ?? ''); - return match ? match[1].toUpperCase() : null; -} - -/** Every language prefix present in the list, in first-seen order. */ +/** Every language present in the list, in first-seen order. */ export function collectLanguagePrefixes( sources: readonly VodSourceDescriptor[] ): string[] { const languages: string[] = []; for (const source of sources) { - const language = titleLanguagePrefix(source.rawTitle); + const language = vodSourceLanguage(source); if (language && !languages.includes(language)) { languages.push(language); } @@ -93,7 +88,7 @@ export function sourceMatchesFilters( if ( filters.language !== null && - titleLanguagePrefix(source.rawTitle) !== filters.language + vodSourceLanguage(source) !== filters.language ) { return false; } diff --git a/libs/ui/components/src/lib/vod-sources/vod-sources-menu.component.spec.ts b/libs/ui/components/src/lib/vod-sources/vod-sources-menu.component.spec.ts index d5d42e5dc..8ee678fe0 100644 --- a/libs/ui/components/src/lib/vod-sources/vod-sources-menu.component.spec.ts +++ b/libs/ui/components/src/lib/vod-sources/vod-sources-menu.component.spec.ts @@ -403,6 +403,34 @@ describe('VodSourcesMenuComponent', () => { expect(rowNames()).toEqual(['Russian Portal']); }); + it('offers a category-derived language when the title has no prefix', () => { + render([ + createSource({ + id: 'a:xtream:1', + playlistId: 'a', + playlistName: 'German Portal', + rawTitle: 'Der Film (2021)', + categoryLanguage: 'DE', + }), + createSource({ + id: 'b:xtream:2', + playlistId: 'b', + playlistName: 'Bare Portal', + rawTitle: 'Der Film', + }), + ]); + + const select = fixture.debugElement.query( + By.css('.sources-menu__lang') + ); + expect(select.nativeElement.textContent).toContain('DE'); + + fixture.componentInstance.setLanguageFilter('DE'); + fixture.detectChanges(); + + expect(rowNames()).toEqual(['German Portal']); + }); + it('the All chip resets every filter and restores the full list', () => { render([ createSource({