mirror of
https://github.com/4gray/iptvnator.git
synced 2026-10-08 09:01:03 -08:00
* docs(agents): compact root guidance and preserve task-specific knowledge * fix(agents): parse guidance navigation with Markdown tokens * fix(agents): validate generic literal repository paths * fix(agents): distinguish code symbols and shortcut images * fix(agents): recognize SCSS filename literals * fix(agents): handle fenced imports and encoded paths * fix(agents): parse prose and rendered HTML anchors * fix(agents): validate rendered HTML navigation * fix(agents): use GitHub-compatible heading slugs * fix(agents): require standalone top-level Claude import * fix(agents): exclude HTML-contained guidance imports * fix(agents): handle image fragments and quoted imports * fix(agents): validate visible HTML and image source sets * fix(agents): recognize package scopes and route source work * fix(agents): parse JSONC and constrain package exemptions * fix(agents): decode link entities and allow package subpaths * fix(agents): route source work and check extensionless files * fix(agents): support package versions and source fragments * fix(agents): accept qualified package prose * fix(agents): retain rendered context for Markdown references * fix(agents): validate visible headings and spaced paths * fix(agents): validate media and hyphenated literal paths * fix(agents): decode full HTML entities and media assets * fix(agents): recognize possessive package mentions * fix(agents): validate extensionless imports and version comparators * fix(agents): retain visible backticks and explicit path punctuation * fix(agents): validate image-map navigation targets * fix(agents): count all Markdown line endings in budgets * fix(agents): delimit package prose at Unicode punctuation * fix(agents): normalize punctuation for extensionless imports * fix(agents): preserve filenames across prose punctuation * fix(agents): validate iframe document references * fix(agents): inspect document suffix before URL fragments * fix(agents): unify Markdown suffix and encoded import guards * fix(agents): handle wildcard versions and alternate documents * fix(agents): validate document formats and trim HTML URLs * fix(agents): cover document families and guidance basenames * fix(agents): require files for media references * fix(agents): preserve block boundaries and validate embeds * fix(agents): normalize internal HTML URL whitespace * fix(agents): reject empty media and ignore URL at-signs * fix(agents): validate srcdoc references and empty srcset * fix(agents): honor HTML bases and preserve adjacent imports * fix(agents): convert base file URLs to native paths * fix(agents): preserve imports after bare URL punctuation * fix(agents): exclude opaque URI prose from import scans * fix(agents): keep import tokens outside URI scheme matches * fix(agents): restrict opaque URI exemptions to parsed links * fix(agents): handle opening prose delimiters * fix(agents): scan nested imports and share document suffixes * fix(agents): reject pathless media and direct file URLs * fix(agents): reject file bases and preserve quoted URL boundaries * fix(agents): distinguish URL quotes and cover guidance variants * fix(agents): validate SVG images and conventional guides * fix(agents): handle declared package names handles and SVG use * fix(agents): normalize closing punctuation on federated handles * fix(agents): normalize Unicode punctuation on handles * fix(agents): normalize possessive federated handles * fix(agents): separate parenthetical prose from handles * fix(agents): exclude www autolinks from import scanning * ci: allow manual CodeQL validation of PR branches * fix(agents): reject nonportable Windows drive links
393 lines
14 KiB
JavaScript
393 lines
14 KiB
JavaScript
import { randomUUID } from 'node:crypto';
|
||
import parseSrcset from 'parse-srcset';
|
||
import GithubSlugger from 'github-slugger';
|
||
import { Marked, Tokenizer } from 'marked';
|
||
import { parseFragment } from 'parse5';
|
||
|
||
export const DOCUMENT_EXTENSION =
|
||
/\.(?:md|markdown|mdown|mkd|mdx|txt|json|ya?ml|html?|rst|rest|adoc|asciidoc|pdf|doc[xm]?|dot[xm]?|od[tspgfbm]|ot[tspg]|fod[tspg]|rtf|org|tex|latex)$/iu;
|
||
|
||
// Inspection only: generated HTML is parsed in memory, never executed or emitted.
|
||
const markdownLexer = new Marked({
|
||
tokenizer: {
|
||
reflink(source, links) {
|
||
const token = Tokenizer.prototype.reflink.call(this, source, links);
|
||
if (token?.type !== 'text') return token;
|
||
|
||
// Marked otherwise turns unresolved references into ordinary text.
|
||
// Retain explicit full/collapsed forms and shortcut images;
|
||
// a bare [word] without a definition remains ordinary prose.
|
||
const full = this.rules.inline.reflink.exec(source);
|
||
const collapsed = this.rules.inline.nolink.exec(source);
|
||
const match =
|
||
full ??
|
||
(collapsed?.[0].endsWith('[]') ||
|
||
collapsed?.[0].startsWith('![')
|
||
? collapsed
|
||
: undefined);
|
||
if (!match) return token;
|
||
return {
|
||
type: 'unresolved-reference',
|
||
raw: match[0],
|
||
text: match[0],
|
||
label: match[2] || match[1],
|
||
};
|
||
},
|
||
},
|
||
});
|
||
|
||
function inlineText(tokens) {
|
||
return tokens
|
||
.map((token) => {
|
||
if (token.type === 'html') return '';
|
||
if (token.tokens) return inlineText(token.tokens);
|
||
return token.type === 'text'
|
||
? decodeEntities(token.text ?? '')
|
||
: (token.text ?? '');
|
||
})
|
||
.join('');
|
||
}
|
||
|
||
function decodeEntities(text, attribute = false) {
|
||
if (attribute) {
|
||
const html = `<a href="${text.replace(/"/gu, '"')}"></a>`;
|
||
return parseFragment(html).childNodes[0].attrs[0].value;
|
||
}
|
||
// RCDATA decodes the full HTML character-reference grammar without
|
||
// interpreting literal tags. The prefix preserves an initial newline.
|
||
const html = `<textarea>x${text.replace(/</gu, '<')}</textarea>`;
|
||
return parseFragment(html).childNodes[0].childNodes[0].value.slice(1);
|
||
}
|
||
|
||
function htmlNavigation(html, inspect = () => {}) {
|
||
const anchors = [];
|
||
const references = [];
|
||
let baseHref;
|
||
function visit(node) {
|
||
if (['script', 'style', 'template'].includes(node.tagName)) return;
|
||
inspect(node);
|
||
if (node.tagName === 'base' && baseHref === undefined)
|
||
baseHref = node.attrs?.find(
|
||
(attribute) => attribute.name === 'href'
|
||
)?.value;
|
||
for (const attribute of node.attrs ?? []) {
|
||
if (
|
||
attribute.name === 'id' ||
|
||
(node.tagName === 'a' && attribute.name === 'name')
|
||
)
|
||
anchors.push(attribute.value);
|
||
if (
|
||
(['a', 'area', 'image', 'use'].includes(node.tagName) &&
|
||
attribute.name === 'href') ||
|
||
([
|
||
'img',
|
||
'video',
|
||
'audio',
|
||
'source',
|
||
'track',
|
||
'iframe',
|
||
'embed',
|
||
].includes(node.tagName) &&
|
||
attribute.name === 'src') ||
|
||
(node.tagName === 'video' && attribute.name === 'poster') ||
|
||
(node.tagName === 'object' && attribute.name === 'data') ||
|
||
(node.tagName === 'input' &&
|
||
attribute.name === 'src' &&
|
||
node.attrs.some(
|
||
(attr) =>
|
||
attr.name === 'type' &&
|
||
attr.value.toLowerCase() === 'image'
|
||
))
|
||
)
|
||
references.push({
|
||
svgUse: node.tagName === 'use',
|
||
target: attribute.value
|
||
.replace(/[\t\n\r]/gu, '')
|
||
.replace(/^[\u0000-\u0020]+|[\u0000-\u0020]+$/gu, ''),
|
||
image: !['a', 'area', 'iframe', 'object', 'embed'].includes(
|
||
node.tagName
|
||
),
|
||
});
|
||
if (node.tagName === 'iframe' && attribute.name === 'srcdoc') {
|
||
const embedded = htmlNavigation(attribute.value);
|
||
for (const reference of embedded.references)
|
||
references.push(
|
||
reference.target.startsWith('#') &&
|
||
!reference.embeddedAnchors &&
|
||
!reference.bases?.length
|
||
? {
|
||
...reference,
|
||
embeddedAnchors: embedded.anchors,
|
||
}
|
||
: reference
|
||
);
|
||
}
|
||
if (
|
||
['img', 'source'].includes(node.tagName) &&
|
||
attribute.name === 'srcset'
|
||
) {
|
||
const candidates = parseSrcset(attribute.value);
|
||
if (!candidates.length)
|
||
references.push({ target: '', image: true });
|
||
for (const candidate of candidates)
|
||
references.push({ target: candidate.url, image: true });
|
||
}
|
||
}
|
||
for (const child of node.childNodes ?? []) visit(child);
|
||
}
|
||
visit(parseFragment(html));
|
||
for (const reference of references) {
|
||
if (
|
||
reference.svgUse &&
|
||
!reference.embeddedAnchors &&
|
||
reference.target.startsWith('#')
|
||
) {
|
||
reference.image = false;
|
||
reference.embeddedAnchors = anchors;
|
||
}
|
||
}
|
||
return {
|
||
anchors,
|
||
references:
|
||
baseHref === undefined
|
||
? references
|
||
: references.map((reference) => ({
|
||
...reference,
|
||
bases: [baseHref, ...(reference.bases ?? [])],
|
||
})),
|
||
};
|
||
}
|
||
|
||
export function guidanceProse(markdown) {
|
||
function text(node, preceding = '') {
|
||
if (
|
||
['script', 'style', 'template', 'pre', 'code'].includes(
|
||
node.tagName
|
||
)
|
||
)
|
||
return ' ';
|
||
if (node.tagName === 'a') {
|
||
const href = node.attrs?.find(
|
||
(attribute) => attribute.name === 'href'
|
||
)?.value;
|
||
if (
|
||
href &&
|
||
/^[a-z][a-z\d+.-]*:(?!\/\/)/iu.test(href) &&
|
||
node.childNodes?.length === 1 &&
|
||
node.childNodes[0].nodeName === '#text' &&
|
||
node.childNodes[0].value === href
|
||
)
|
||
return ' ';
|
||
}
|
||
if (node.nodeName === '#text')
|
||
return node.value.replace(
|
||
/(?:\b[a-z][a-z\d+.-]*:\/\/|\/\/|\bwww\.)[^\s]*?(?=[)\]}>][.,;:!?]*@|\s|$)/giu,
|
||
(url, offset) => {
|
||
const opening = (
|
||
preceding + node.value.slice(0, offset)
|
||
).at(-1);
|
||
const closing = {
|
||
'"': '"',
|
||
"'": "'",
|
||
'“': '”',
|
||
'”': '”',
|
||
'‘': '’',
|
||
'’': '’',
|
||
}[opening];
|
||
const boundary = closing ? url.indexOf(closing) : -1;
|
||
return boundary >= 0 &&
|
||
/^[.,;:!?]*@/u.test(url.slice(boundary + 1))
|
||
? ' ' + url.slice(boundary)
|
||
: ' ';
|
||
}
|
||
);
|
||
let content = '';
|
||
for (const child of node.childNodes ?? [])
|
||
content += text(child, preceding + content);
|
||
return [
|
||
'address',
|
||
'article',
|
||
'aside',
|
||
'details',
|
||
'summary',
|
||
'dialog',
|
||
'dl',
|
||
'dt',
|
||
'dd',
|
||
'fieldset',
|
||
'legend',
|
||
'figure',
|
||
'figcaption',
|
||
'footer',
|
||
'form',
|
||
'header',
|
||
'hgroup',
|
||
'hr',
|
||
'main',
|
||
'nav',
|
||
'ol',
|
||
'ul',
|
||
'section',
|
||
'table',
|
||
'caption',
|
||
'thead',
|
||
'tbody',
|
||
'tfoot',
|
||
'tr',
|
||
'td',
|
||
'th',
|
||
'p',
|
||
'li',
|
||
'blockquote',
|
||
'div',
|
||
'br',
|
||
'h1',
|
||
'h2',
|
||
'h3',
|
||
'h4',
|
||
'h5',
|
||
'h6',
|
||
].includes(node.tagName)
|
||
? '\n' + content + '\n'
|
||
: content;
|
||
}
|
||
return text(parseFragment(new Marked().parse(markdown)));
|
||
}
|
||
|
||
export function guidanceStandaloneImports(markdown) {
|
||
const candidates = markdownLexer
|
||
.lexer(markdown)
|
||
.filter((token) => token.type === 'paragraph')
|
||
.flatMap((token) => [
|
||
...token.raw.matchAll(/^ {0,3}@([^\s]+)[\t ]*$/gmu),
|
||
])
|
||
.map((match) => match[1]);
|
||
// Markdown can split an HTML container across several top-level tokens.
|
||
// Check the parsed output tree as well as raw source formatting. Only text
|
||
// directly inside a root paragraph can supply the standalone directive.
|
||
const document = parseFragment(new Marked().parse(markdown));
|
||
const visible = new Map();
|
||
for (const node of document.childNodes) {
|
||
if (node.tagName !== 'p') continue;
|
||
const text = node.childNodes
|
||
.map((child) =>
|
||
child.nodeName === '#text' ? child.value : '\uFFFC'
|
||
)
|
||
.join('');
|
||
for (const match of text.matchAll(/^ {0,3}@([^\s]+)[\t ]*$/gmu))
|
||
visible.set(match[1], (visible.get(match[1]) ?? 0) + 1);
|
||
}
|
||
return candidates.filter((candidate) => {
|
||
const count = visible.get(candidate) ?? 0;
|
||
if (!count) return false;
|
||
visible.set(candidate, count - 1);
|
||
return true;
|
||
});
|
||
}
|
||
|
||
export function guidanceAnchors(markdown) {
|
||
const slugger = new GithubSlugger();
|
||
const found = new Set();
|
||
const headings = [];
|
||
const marker = `data-guidance-${randomUUID()}`;
|
||
const renderer = new Marked({
|
||
renderer: {
|
||
heading(token) {
|
||
const index = headings.push(inlineText(token.tokens)) - 1;
|
||
return `<h${token.depth} ${marker}="${index}">${this.parser.parseInline(token.tokens)}</h${token.depth}>\n`;
|
||
},
|
||
},
|
||
});
|
||
const navigation = htmlNavigation(renderer.parse(markdown), (node) => {
|
||
const attribute = node.attrs?.find((attr) => attr.name === marker);
|
||
if (attribute)
|
||
found.add(slugger.slug(headings[Number(attribute.value)]));
|
||
});
|
||
return new Set([...found, ...navigation.anchors]);
|
||
}
|
||
|
||
function isLiteralRepositoryPath(token) {
|
||
// A typo in the directory or a new root filename must still be checked.
|
||
// Exclude recognizable prose/code forms instead of allowlisting paths.
|
||
if (/^(?:@|--|[a-z][a-z\d+.-]*:|\/\/)/iu.test(token)) return false;
|
||
const explicitRelative = /^(?:\.\/|\.\.\/)/u.test(token);
|
||
if (!explicitRelative && /[^\p{L}\p{N}_./#-]/u.test(token)) return false;
|
||
if (token.includes('YYYY-MM-DD') || /(?:^|\/)\.\.\.(?:\/|$)/u.test(token))
|
||
return false;
|
||
const path = token.split('#')[0];
|
||
// Bare dotted identifiers are ambiguous. Recognize conventional file
|
||
// suffixes; other filenames can be made explicit with ./ or a Markdown link.
|
||
// This applies to user-defined symbols as well as JavaScript globals.
|
||
if (
|
||
/^[\p{L}_][\p{L}\p{N}_]*(?:\.[\p{L}_][\p{L}\p{N}_]*)+$/u.test(path) &&
|
||
!DOCUMENT_EXTENSION.test(path) &&
|
||
!/\.(?:md|mdx|json|jsonc|ya?ml|[cm]?[jt]sx?|html?|css|scss|sass|less|toml|xml|txt|sh|py|sql|svg|png|jpe?g|webp|gif|m3u8?|conf|ini|lock)$/iu.test(
|
||
path
|
||
)
|
||
)
|
||
return false;
|
||
return (
|
||
path.includes('/') ||
|
||
/^(?:Dockerfile|Containerfile|Makefile|GNUmakefile|Justfile|Procfile|Gemfile|Rakefile|Vagrantfile|LICENSE|LICENCE|NOTICE|COPYING|AUTHORS|CONTRIBUTORS|README|CHANGELOG)$/u.test(
|
||
path
|
||
) ||
|
||
/^(?:\.[\p{L}\p{N}_-][\p{L}\p{N}_.-]*|[\p{L}\p{N}_-][\p{L}\p{N}_.-]*\.[\p{L}][\p{L}\p{N}_-]*)$/u.test(
|
||
path
|
||
)
|
||
);
|
||
}
|
||
|
||
export function guidanceReferences(markdown, includeLiterals) {
|
||
const tokens = markdownLexer.lexer(markdown);
|
||
const markerTag = `guidance-reference-${randomUUID()}`;
|
||
const metadata = [];
|
||
function mark(token, reference) {
|
||
const index = metadata.push(reference) - 1;
|
||
token.type = 'html';
|
||
token.raw = `<${markerTag} data-index="${index}"></${markerTag}>`;
|
||
token.text = token.raw;
|
||
}
|
||
markdownLexer.walkTokens(tokens, (token) => {
|
||
if (token.type === 'def')
|
||
mark(token, {
|
||
target: decodeEntities(token.href, true),
|
||
definition: true,
|
||
});
|
||
else if (token.type === 'unresolved-reference')
|
||
mark(token, { unresolvedReference: token.label });
|
||
else if (
|
||
includeLiterals &&
|
||
token.type === 'codespan' &&
|
||
isLiteralRepositoryPath(token.text)
|
||
)
|
||
mark(token, { target: token.text, literal: true });
|
||
});
|
||
// Let Markdown rendering and HTML tree construction retain container context
|
||
// for ordinary links and for metadata that has no rendered navigation node.
|
||
const visible = [];
|
||
const navigation = htmlNavigation(new Marked().parser(tokens), (node) => {
|
||
if (node.tagName !== markerTag) return;
|
||
const index = Number(
|
||
node.attrs.find((attr) => attr.name === 'data-index')?.value
|
||
);
|
||
if (metadata[index]) visible.push(metadata[index]);
|
||
});
|
||
const result = [];
|
||
const seen = new Set();
|
||
function add(reference) {
|
||
const key = JSON.stringify(reference);
|
||
if (!seen.has(key)) {
|
||
seen.add(key);
|
||
result.push(reference);
|
||
}
|
||
}
|
||
for (const reference of navigation.references) add(reference);
|
||
const usedTargets = new Set(
|
||
navigation.references.map((reference) => reference.target)
|
||
);
|
||
for (const { definition, ...reference } of visible) {
|
||
if (!definition || !usedTargets.has(reference.target)) add(reference);
|
||
}
|
||
return result;
|
||
}
|