Files
iptvnator/tools/skills/agent-guidance-markdown.mjs
T
4gray faad8fd8fd docs(agents): compact root guidance and preserve task-specific knowledge (#1645)
* docs(agents): compact root guidance and preserve task-specific knowledge

* fix(agents): parse guidance navigation with Markdown tokens

* fix(agents): validate generic literal repository paths

* fix(agents): distinguish code symbols and shortcut images

* fix(agents): recognize SCSS filename literals

* fix(agents): handle fenced imports and encoded paths

* fix(agents): parse prose and rendered HTML anchors

* fix(agents): validate rendered HTML navigation

* fix(agents): use GitHub-compatible heading slugs

* fix(agents): require standalone top-level Claude import

* fix(agents): exclude HTML-contained guidance imports

* fix(agents): handle image fragments and quoted imports

* fix(agents): validate visible HTML and image source sets

* fix(agents): recognize package scopes and route source work

* fix(agents): parse JSONC and constrain package exemptions

* fix(agents): decode link entities and allow package subpaths

* fix(agents): route source work and check extensionless files

* fix(agents): support package versions and source fragments

* fix(agents): accept qualified package prose

* fix(agents): retain rendered context for Markdown references

* fix(agents): validate visible headings and spaced paths

* fix(agents): validate media and hyphenated literal paths

* fix(agents): decode full HTML entities and media assets

* fix(agents): recognize possessive package mentions

* fix(agents): validate extensionless imports and version comparators

* fix(agents): retain visible backticks and explicit path punctuation

* fix(agents): validate image-map navigation targets

* fix(agents): count all Markdown line endings in budgets

* fix(agents): delimit package prose at Unicode punctuation

* fix(agents): normalize punctuation for extensionless imports

* fix(agents): preserve filenames across prose punctuation

* fix(agents): validate iframe document references

* fix(agents): inspect document suffix before URL fragments

* fix(agents): unify Markdown suffix and encoded import guards

* fix(agents): handle wildcard versions and alternate documents

* fix(agents): validate document formats and trim HTML URLs

* fix(agents): cover document families and guidance basenames

* fix(agents): require files for media references

* fix(agents): preserve block boundaries and validate embeds

* fix(agents): normalize internal HTML URL whitespace

* fix(agents): reject empty media and ignore URL at-signs

* fix(agents): validate srcdoc references and empty srcset

* fix(agents): honor HTML bases and preserve adjacent imports

* fix(agents): convert base file URLs to native paths

* fix(agents): preserve imports after bare URL punctuation

* fix(agents): exclude opaque URI prose from import scans

* fix(agents): keep import tokens outside URI scheme matches

* fix(agents): restrict opaque URI exemptions to parsed links

* fix(agents): handle opening prose delimiters

* fix(agents): scan nested imports and share document suffixes

* fix(agents): reject pathless media and direct file URLs

* fix(agents): reject file bases and preserve quoted URL boundaries

* fix(agents): distinguish URL quotes and cover guidance variants

* fix(agents): validate SVG images and conventional guides

* fix(agents): handle declared package names handles and SVG use

* fix(agents): normalize closing punctuation on federated handles

* fix(agents): normalize Unicode punctuation on handles

* fix(agents): normalize possessive federated handles

* fix(agents): separate parenthetical prose from handles

* fix(agents): exclude www autolinks from import scanning

* ci: allow manual CodeQL validation of PR branches

* fix(agents): reject nonportable Windows drive links
2026-09-21 18:07:14 +02:00

393 lines
14 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import { randomUUID } from 'node:crypto';
import parseSrcset from 'parse-srcset';
import GithubSlugger from 'github-slugger';
import { Marked, Tokenizer } from 'marked';
import { parseFragment } from 'parse5';
export const DOCUMENT_EXTENSION =
/\.(?:md|markdown|mdown|mkd|mdx|txt|json|ya?ml|html?|rst|rest|adoc|asciidoc|pdf|doc[xm]?|dot[xm]?|od[tspgfbm]|ot[tspg]|fod[tspg]|rtf|org|tex|latex)$/iu;
// Inspection only: generated HTML is parsed in memory, never executed or emitted.
const markdownLexer = new Marked({
tokenizer: {
reflink(source, links) {
const token = Tokenizer.prototype.reflink.call(this, source, links);
if (token?.type !== 'text') return token;
// Marked otherwise turns unresolved references into ordinary text.
// Retain explicit full/collapsed forms and shortcut images;
// a bare [word] without a definition remains ordinary prose.
const full = this.rules.inline.reflink.exec(source);
const collapsed = this.rules.inline.nolink.exec(source);
const match =
full ??
(collapsed?.[0].endsWith('[]') ||
collapsed?.[0].startsWith('![')
? collapsed
: undefined);
if (!match) return token;
return {
type: 'unresolved-reference',
raw: match[0],
text: match[0],
label: match[2] || match[1],
};
},
},
});
function inlineText(tokens) {
return tokens
.map((token) => {
if (token.type === 'html') return '';
if (token.tokens) return inlineText(token.tokens);
return token.type === 'text'
? decodeEntities(token.text ?? '')
: (token.text ?? '');
})
.join('');
}
function decodeEntities(text, attribute = false) {
if (attribute) {
const html = `<a href="${text.replace(/"/gu, '&quot;')}"></a>`;
return parseFragment(html).childNodes[0].attrs[0].value;
}
// RCDATA decodes the full HTML character-reference grammar without
// interpreting literal tags. The prefix preserves an initial newline.
const html = `<textarea>x${text.replace(/</gu, '&lt;')}</textarea>`;
return parseFragment(html).childNodes[0].childNodes[0].value.slice(1);
}
function htmlNavigation(html, inspect = () => {}) {
const anchors = [];
const references = [];
let baseHref;
function visit(node) {
if (['script', 'style', 'template'].includes(node.tagName)) return;
inspect(node);
if (node.tagName === 'base' && baseHref === undefined)
baseHref = node.attrs?.find(
(attribute) => attribute.name === 'href'
)?.value;
for (const attribute of node.attrs ?? []) {
if (
attribute.name === 'id' ||
(node.tagName === 'a' && attribute.name === 'name')
)
anchors.push(attribute.value);
if (
(['a', 'area', 'image', 'use'].includes(node.tagName) &&
attribute.name === 'href') ||
([
'img',
'video',
'audio',
'source',
'track',
'iframe',
'embed',
].includes(node.tagName) &&
attribute.name === 'src') ||
(node.tagName === 'video' && attribute.name === 'poster') ||
(node.tagName === 'object' && attribute.name === 'data') ||
(node.tagName === 'input' &&
attribute.name === 'src' &&
node.attrs.some(
(attr) =>
attr.name === 'type' &&
attr.value.toLowerCase() === 'image'
))
)
references.push({
svgUse: node.tagName === 'use',
target: attribute.value
.replace(/[\t\n\r]/gu, '')
.replace(/^[\u0000-\u0020]+|[\u0000-\u0020]+$/gu, ''),
image: !['a', 'area', 'iframe', 'object', 'embed'].includes(
node.tagName
),
});
if (node.tagName === 'iframe' && attribute.name === 'srcdoc') {
const embedded = htmlNavigation(attribute.value);
for (const reference of embedded.references)
references.push(
reference.target.startsWith('#') &&
!reference.embeddedAnchors &&
!reference.bases?.length
? {
...reference,
embeddedAnchors: embedded.anchors,
}
: reference
);
}
if (
['img', 'source'].includes(node.tagName) &&
attribute.name === 'srcset'
) {
const candidates = parseSrcset(attribute.value);
if (!candidates.length)
references.push({ target: '', image: true });
for (const candidate of candidates)
references.push({ target: candidate.url, image: true });
}
}
for (const child of node.childNodes ?? []) visit(child);
}
visit(parseFragment(html));
for (const reference of references) {
if (
reference.svgUse &&
!reference.embeddedAnchors &&
reference.target.startsWith('#')
) {
reference.image = false;
reference.embeddedAnchors = anchors;
}
}
return {
anchors,
references:
baseHref === undefined
? references
: references.map((reference) => ({
...reference,
bases: [baseHref, ...(reference.bases ?? [])],
})),
};
}
export function guidanceProse(markdown) {
function text(node, preceding = '') {
if (
['script', 'style', 'template', 'pre', 'code'].includes(
node.tagName
)
)
return ' ';
if (node.tagName === 'a') {
const href = node.attrs?.find(
(attribute) => attribute.name === 'href'
)?.value;
if (
href &&
/^[a-z][a-z\d+.-]*:(?!\/\/)/iu.test(href) &&
node.childNodes?.length === 1 &&
node.childNodes[0].nodeName === '#text' &&
node.childNodes[0].value === href
)
return ' ';
}
if (node.nodeName === '#text')
return node.value.replace(
/(?:\b[a-z][a-z\d+.-]*:\/\/|\/\/|\bwww\.)[^\s]*?(?=[)\]}>][.,;:!?]*@|\s|$)/giu,
(url, offset) => {
const opening = (
preceding + node.value.slice(0, offset)
).at(-1);
const closing = {
'"': '"',
"'": "'",
'“': '”',
'”': '”',
'‘': '’',
'’': '’',
}[opening];
const boundary = closing ? url.indexOf(closing) : -1;
return boundary >= 0 &&
/^[.,;:!?]*@/u.test(url.slice(boundary + 1))
? ' ' + url.slice(boundary)
: ' ';
}
);
let content = '';
for (const child of node.childNodes ?? [])
content += text(child, preceding + content);
return [
'address',
'article',
'aside',
'details',
'summary',
'dialog',
'dl',
'dt',
'dd',
'fieldset',
'legend',
'figure',
'figcaption',
'footer',
'form',
'header',
'hgroup',
'hr',
'main',
'nav',
'ol',
'ul',
'section',
'table',
'caption',
'thead',
'tbody',
'tfoot',
'tr',
'td',
'th',
'p',
'li',
'blockquote',
'div',
'br',
'h1',
'h2',
'h3',
'h4',
'h5',
'h6',
].includes(node.tagName)
? '\n' + content + '\n'
: content;
}
return text(parseFragment(new Marked().parse(markdown)));
}
export function guidanceStandaloneImports(markdown) {
const candidates = markdownLexer
.lexer(markdown)
.filter((token) => token.type === 'paragraph')
.flatMap((token) => [
...token.raw.matchAll(/^ {0,3}@([^\s]+)[\t ]*$/gmu),
])
.map((match) => match[1]);
// Markdown can split an HTML container across several top-level tokens.
// Check the parsed output tree as well as raw source formatting. Only text
// directly inside a root paragraph can supply the standalone directive.
const document = parseFragment(new Marked().parse(markdown));
const visible = new Map();
for (const node of document.childNodes) {
if (node.tagName !== 'p') continue;
const text = node.childNodes
.map((child) =>
child.nodeName === '#text' ? child.value : '\uFFFC'
)
.join('');
for (const match of text.matchAll(/^ {0,3}@([^\s]+)[\t ]*$/gmu))
visible.set(match[1], (visible.get(match[1]) ?? 0) + 1);
}
return candidates.filter((candidate) => {
const count = visible.get(candidate) ?? 0;
if (!count) return false;
visible.set(candidate, count - 1);
return true;
});
}
export function guidanceAnchors(markdown) {
const slugger = new GithubSlugger();
const found = new Set();
const headings = [];
const marker = `data-guidance-${randomUUID()}`;
const renderer = new Marked({
renderer: {
heading(token) {
const index = headings.push(inlineText(token.tokens)) - 1;
return `<h${token.depth} ${marker}="${index}">${this.parser.parseInline(token.tokens)}</h${token.depth}>\n`;
},
},
});
const navigation = htmlNavigation(renderer.parse(markdown), (node) => {
const attribute = node.attrs?.find((attr) => attr.name === marker);
if (attribute)
found.add(slugger.slug(headings[Number(attribute.value)]));
});
return new Set([...found, ...navigation.anchors]);
}
function isLiteralRepositoryPath(token) {
// A typo in the directory or a new root filename must still be checked.
// Exclude recognizable prose/code forms instead of allowlisting paths.
if (/^(?:@|--|[a-z][a-z\d+.-]*:|\/\/)/iu.test(token)) return false;
const explicitRelative = /^(?:\.\/|\.\.\/)/u.test(token);
if (!explicitRelative && /[^\p{L}\p{N}_./#-]/u.test(token)) return false;
if (token.includes('YYYY-MM-DD') || /(?:^|\/)\.\.\.(?:\/|$)/u.test(token))
return false;
const path = token.split('#')[0];
// Bare dotted identifiers are ambiguous. Recognize conventional file
// suffixes; other filenames can be made explicit with ./ or a Markdown link.
// This applies to user-defined symbols as well as JavaScript globals.
if (
/^[\p{L}_][\p{L}\p{N}_]*(?:\.[\p{L}_][\p{L}\p{N}_]*)+$/u.test(path) &&
!DOCUMENT_EXTENSION.test(path) &&
!/\.(?:md|mdx|json|jsonc|ya?ml|[cm]?[jt]sx?|html?|css|scss|sass|less|toml|xml|txt|sh|py|sql|svg|png|jpe?g|webp|gif|m3u8?|conf|ini|lock)$/iu.test(
path
)
)
return false;
return (
path.includes('/') ||
/^(?:Dockerfile|Containerfile|Makefile|GNUmakefile|Justfile|Procfile|Gemfile|Rakefile|Vagrantfile|LICENSE|LICENCE|NOTICE|COPYING|AUTHORS|CONTRIBUTORS|README|CHANGELOG)$/u.test(
path
) ||
/^(?:\.[\p{L}\p{N}_-][\p{L}\p{N}_.-]*|[\p{L}\p{N}_-][\p{L}\p{N}_.-]*\.[\p{L}][\p{L}\p{N}_-]*)$/u.test(
path
)
);
}
export function guidanceReferences(markdown, includeLiterals) {
const tokens = markdownLexer.lexer(markdown);
const markerTag = `guidance-reference-${randomUUID()}`;
const metadata = [];
function mark(token, reference) {
const index = metadata.push(reference) - 1;
token.type = 'html';
token.raw = `<${markerTag} data-index="${index}"></${markerTag}>`;
token.text = token.raw;
}
markdownLexer.walkTokens(tokens, (token) => {
if (token.type === 'def')
mark(token, {
target: decodeEntities(token.href, true),
definition: true,
});
else if (token.type === 'unresolved-reference')
mark(token, { unresolvedReference: token.label });
else if (
includeLiterals &&
token.type === 'codespan' &&
isLiteralRepositoryPath(token.text)
)
mark(token, { target: token.text, literal: true });
});
// Let Markdown rendering and HTML tree construction retain container context
// for ordinary links and for metadata that has no rendered navigation node.
const visible = [];
const navigation = htmlNavigation(new Marked().parser(tokens), (node) => {
if (node.tagName !== markerTag) return;
const index = Number(
node.attrs.find((attr) => attr.name === 'data-index')?.value
);
if (metadata[index]) visible.push(metadata[index]);
});
const result = [];
const seen = new Set();
function add(reference) {
const key = JSON.stringify(reference);
if (!seen.has(key)) {
seen.add(key);
result.push(reference);
}
}
for (const reference of navigation.references) add(reference);
const usedTargets = new Set(
navigation.references.map((reference) => reference.target)
);
for (const { definition, ...reference } of visible) {
if (!definition || !usedTargets.has(reference.target)) add(reference);
}
return result;
}