Files
iptvnator/tools/skills/validate-agent-guidance.mjs
4gray faad8fd8fd docs(agents): compact root guidance and preserve task-specific knowledge (#1645)
* docs(agents): compact root guidance and preserve task-specific knowledge

* fix(agents): parse guidance navigation with Markdown tokens

* fix(agents): validate generic literal repository paths

* fix(agents): distinguish code symbols and shortcut images

* fix(agents): recognize SCSS filename literals

* fix(agents): handle fenced imports and encoded paths

* fix(agents): parse prose and rendered HTML anchors

* fix(agents): validate rendered HTML navigation

* fix(agents): use GitHub-compatible heading slugs

* fix(agents): require standalone top-level Claude import

* fix(agents): exclude HTML-contained guidance imports

* fix(agents): handle image fragments and quoted imports

* fix(agents): validate visible HTML and image source sets

* fix(agents): recognize package scopes and route source work

* fix(agents): parse JSONC and constrain package exemptions

* fix(agents): decode link entities and allow package subpaths

* fix(agents): route source work and check extensionless files

* fix(agents): support package versions and source fragments

* fix(agents): accept qualified package prose

* fix(agents): retain rendered context for Markdown references

* fix(agents): validate visible headings and spaced paths

* fix(agents): validate media and hyphenated literal paths

* fix(agents): decode full HTML entities and media assets

* fix(agents): recognize possessive package mentions

* fix(agents): validate extensionless imports and version comparators

* fix(agents): retain visible backticks and explicit path punctuation

* fix(agents): validate image-map navigation targets

* fix(agents): count all Markdown line endings in budgets

* fix(agents): delimit package prose at Unicode punctuation

* fix(agents): normalize punctuation for extensionless imports

* fix(agents): preserve filenames across prose punctuation

* fix(agents): validate iframe document references

* fix(agents): inspect document suffix before URL fragments

* fix(agents): unify Markdown suffix and encoded import guards

* fix(agents): handle wildcard versions and alternate documents

* fix(agents): validate document formats and trim HTML URLs

* fix(agents): cover document families and guidance basenames

* fix(agents): require files for media references

* fix(agents): preserve block boundaries and validate embeds

* fix(agents): normalize internal HTML URL whitespace

* fix(agents): reject empty media and ignore URL at-signs

* fix(agents): validate srcdoc references and empty srcset

* fix(agents): honor HTML bases and preserve adjacent imports

* fix(agents): convert base file URLs to native paths

* fix(agents): preserve imports after bare URL punctuation

* fix(agents): exclude opaque URI prose from import scans

* fix(agents): keep import tokens outside URI scheme matches

* fix(agents): restrict opaque URI exemptions to parsed links

* fix(agents): handle opening prose delimiters

* fix(agents): scan nested imports and share document suffixes

* fix(agents): reject pathless media and direct file URLs

* fix(agents): reject file bases and preserve quoted URL boundaries

* fix(agents): distinguish URL quotes and cover guidance variants

* fix(agents): validate SVG images and conventional guides

* fix(agents): handle declared package names handles and SVG use

* fix(agents): normalize closing punctuation on federated handles

* fix(agents): normalize Unicode punctuation on handles

* fix(agents): normalize possessive federated handles

* fix(agents): separate parenthetical prose from handles

* fix(agents): exclude www autolinks from import scanning

* ci: allow manual CodeQL validation of PR branches

* fix(agents): reject nonportable Windows drive links
2026-09-21 18:07:14 +02:00

322 lines
12 KiB
JavaScript
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import ts from 'typescript';
import { readFile, realpath, stat } from 'node:fs/promises';
import {
dirname,
extname,
isAbsolute,
relative,
resolve,
sep,
} from 'node:path';
import { fileURLToPath, pathToFileURL } from 'node:url';
import {
DOCUMENT_EXTENSION,
guidanceAnchors as anchors,
guidanceProse,
guidanceStandaloneImports,
guidanceReferences as references,
} from './agent-guidance-markdown.mjs';
const MARKDOWN_EXTENSION = /\.(?:md|markdown|mdown|mkd|mdx)$/iu;
const SURFACES = [
'AGENTS.md',
'CLAUDE.md',
'docs/maintenance/agent-context-map.md',
'docs/maintenance/agent-guidance-migration.md',
];
const LIMITS = { 'AGENTS.md': [200, 16384], 'CLAUDE.md': [30, 2048] };
function within(root, path) {
const local = relative(root, path);
return (
!isAbsolute(local) && local !== '..' && !local.startsWith(`..${sep}`)
);
}
async function validateReference(
rootDir,
source,
{ target, literal, image, unresolvedReference, embeddedAnchors, bases }
) {
if (unresolvedReference !== undefined)
return `${source}: unresolved Markdown reference "${unresolvedReference}"`;
if (/^[a-z]:[\\/]/iu.test(target))
return `${source}: use a repository-relative path instead of a Windows drive path: ${target}`;
if (/^file:/iu.test(target))
return `${source}: use a repository-relative path instead of a file URL: ${target}`;
if (image && !target) return `${source}: empty media target`;
let resolvedBasePath;
if (bases?.length) {
try {
let base = pathToFileURL(resolve(rootDir, source));
for (const href of bases) {
if (
/^(?:file:|[a-z]:[\\/])/iu.test(
href.replace(/[\t\n\r]/gu, '').trimStart()
)
)
return `${source}: use a repository-relative HTML base instead of a file URL or Windows drive path`;
base = new URL(href, base);
}
const url = new URL(target, base);
if (url.protocol !== 'file:') return;
resolvedBasePath = fileURLToPath(url);
target = url.pathname + url.search + url.hash;
embeddedAnchors = undefined;
} catch {
return `${source}: malformed HTML base or target: ${target}`;
}
}
if (/^(?:[a-z][a-z\d+.-]*:|\/\/)/iu.test(target)) return;
let path;
let anchor;
try {
const hash = target.indexOf('#');
const pathAndQuery = hash < 0 ? target : target.slice(0, hash);
path = decodeURIComponent(pathAndQuery.split('?')[0]);
anchor =
hash < 0 ? undefined : decodeURIComponent(target.slice(hash + 1));
} catch {
return `${source}: malformed local link: ${target}`;
}
if (image && !path && !resolvedBasePath)
return `${source}: media target requires a path: ${target}`;
if (embeddedAnchors && !path && !image)
return !anchor || embeddedAnchors.includes(anchor)
? undefined
: `${source}: missing anchor "${anchor}" in iframe srcdoc`;
const absolute =
resolvedBasePath ??
(path
? resolve(
literal ? rootDir : dirname(resolve(rootDir, source)),
path
)
: resolve(rootDir, source));
if (!within(rootDir, absolute))
return `${source}: referenced path escapes repository root: ${target}`;
try {
const actual = await realpath(absolute);
if (!within(rootDir, actual))
return `${source}: referenced path escapes repository root: ${target}`;
if (image && !(await stat(actual)).isFile())
return `${source}: media target is not a file: ${target}`;
if (
anchor &&
!image &&
MARKDOWN_EXTENSION.test(extname(actual)) &&
!anchors(await readFile(actual, 'utf8')).has(anchor)
) {
return `${source}: missing anchor "${anchor}" in ${target}`;
}
} catch (error) {
if (['ENOENT', 'ENOTDIR'].includes(error.code))
return `${source}: referenced path does not exist: ${target}`;
if (error.code === 'EISDIR')
return `${source}: anchor target is a directory: ${target}`;
throw error;
}
}
async function packageMentions(rootDir) {
async function readJson(path) {
try {
const text = await readFile(resolve(rootDir, path), 'utf8');
if (path !== 'tsconfig.base.json') return JSON.parse(text);
const parsed = ts.parseConfigFileTextToJson(path, text);
if (parsed.error)
throw new Error(
ts.flattenDiagnosticMessageText(
parsed.error.messageText,
'\n'
)
);
return parsed.config;
} catch (error) {
if (error.code === 'ENOENT') return {};
throw error;
}
}
const manifest = await readJson('package.json');
const config = await readJson('tsconfig.base.json');
const packages = [
...Object.keys(manifest.dependencies ?? {}),
...Object.keys(manifest.devDependencies ?? {}),
...Object.keys(manifest.optionalDependencies ?? {}),
...Object.keys(manifest.peerDependencies ?? {}),
];
const names = [
...packages,
...Object.keys(config.compilerOptions?.paths ?? {}),
];
const declared = names
.filter((name) => /^@[^/]+\//u.test(name))
.map((name) => name.slice(1));
const scopes = new Set(declared.map((name) => name.split('/')[0]));
return (raw) => {
if (packages.includes(`@${raw}`) || packages.includes(raw)) return true;
// ASCII punctuation also belongs to package names and version ranges.
let token = raw.split(/[,;:!?([{]|(?=[^\x00-\x7f])\p{P}/u, 1)[0];
token = token.replace(/[?!.,;:)"'\]}]+$/u, '');
token = token.replace(/['’]s$/iu, '');
token = token.replace(
/^([^/@]+(?:\/[^/@]+)?)@(?:(?:[~^]|[<>]=?|=)?\d[\w.*+-]*|\*|[a-z][\w-]*)$/iu,
'$1'
);
if (
token.split(/[\/\\]/u).some((part) => part === '.' || part === '..')
)
return false;
if (packages.includes(token) || packages.includes(`@${token}`))
return true;
const path = token.split(/[?#]/u, 1)[0];
if (
/%[\da-f]{2}/iu.test(token) ||
MARKDOWN_EXTENSION.test(path) ||
/(?:^|\/)(?:AGENTS|CLAUDE|INSTRUCTIONS|README|LICENSE|LICENCE|NOTICE|COPYING|AUTHORS|CONTRIBUTORS|CHANGELOG|CONTRIBUTING|SECURITY|CODE_OF_CONDUCT|SUPPORT)$/iu.test(
path
) ||
DOCUMENT_EXTENSION.test(path)
)
return false;
if (packages.includes(token)) return true;
if (token.endsWith('/*') && scopes.has(token.slice(0, -2))) return true;
return declared.some((name) => {
const star = name.indexOf('*');
return star < 0
? token === name ||
(packages.includes(`@${name}`) &&
token.startsWith(`${name}/`))
: token.startsWith(name.slice(0, star)) &&
token.endsWith(name.slice(star + 1));
});
};
}
export async function validateAgentGuidance({ rootDir }) {
rootDir = await realpath(rootDir);
const diagnostics = [];
const isPackageMention = await packageMentions(rootDir);
for (const source of SURFACES) {
let markdown;
try {
markdown = await readFile(resolve(rootDir, source), 'utf8');
} catch (error) {
if (error.code !== 'ENOENT') throw error;
diagnostics.push(`${source}: required guidance file is missing`);
continue;
}
if (LIMITS[source]) {
const [maxLines, maxBytes] = LIMITS[source];
const lines =
markdown === ''
? 0
: markdown
.replace(/(?:\r\n|[\r\n])$/u, '')
.split(/\r\n|[\r\n]/u).length;
const bytes = Buffer.byteLength(markdown, 'utf8');
if (lines > maxLines)
diagnostics.push(
`${source}: at most ${maxLines} lines allowed (received ${lines})`
);
if (bytes > maxBytes)
diagnostics.push(
`${source}: at most ${maxBytes} UTF-8 bytes allowed (received ${bytes})`
);
const prose = guidanceProse(markdown);
const imports = guidanceStandaloneImports(markdown);
const inlineImports = [];
for (const match of prose.matchAll(
/(?=(?:^|[^\p{L}\p{N}_@])@([^\s]+))/gu
)) {
const token = match[1];
if (isPackageMention(token)) continue;
if (
/^[\w.-]+@(?:[a-z\d](?:[a-z\d-]*[a-z\d])?\.)+[a-z]{2,}$/iu.test(
token
.split(/[([{]/u, 1)[0]
.replace(/\p{P}+$/gu, '')
.replace(/['’]s$/iu, '')
)
)
continue;
if (
/[./\\]/u.test(token) ||
/^(?:LICENSE|Makefile|Dockerfile|AGENTS|CLAUDE)(?:$|[.,;)])/u.test(
token
)
) {
inlineImports.push(token);
continue;
}
// Check real filenames before interpreting punctuation as prose.
const candidates = new Set([
token,
token.replace(/[?!.,;:)"'\]}]+$/u, ''),
]);
for (const boundary of token.matchAll(
/[,;:!?([{]|(?=[^\x00-\x7f])\p{P}/gu
))
candidates.add(token.slice(0, boundary.index));
for (const candidate of candidates) {
try {
if (
(await stat(resolve(rootDir, candidate))).isFile()
) {
inlineImports.push(token);
break;
}
} catch (error) {
if (error.code !== 'ENOENT' && error.code !== 'ENOTDIR')
throw error;
}
}
}
if (
inlineImports.some((token) => token !== 'AGENTS.md') ||
(source === 'AGENTS.md' && inlineImports.length) ||
(source === 'CLAUDE.md' &&
inlineImports.filter((token) => token === 'AGENTS.md')
.length !== 1)
) {
diagnostics.push(
`${source}: additional or inline guidance imports are not allowed`
);
}
if (source === 'CLAUDE.md') {
if (imports.length !== 1 || imports[0] !== 'AGENTS.md')
diagnostics.push(
`${source}: exactly one standalone @AGENTS.md import is required; no other imports are allowed`
);
} else if (imports.length)
diagnostics.push(`${source}: imports are not allowed`);
}
for (const reference of references(
markdown,
!source.endsWith('agent-guidance-migration.md')
)) {
const diagnostic = await validateReference(
rootDir,
source,
reference
);
if (diagnostic) diagnostics.push(diagnostic);
}
}
return { checkedFiles: SURFACES.length, diagnostics };
}
if (
process.argv[1] &&
resolve(process.argv[1]) === fileURLToPath(import.meta.url)
) {
const { checkedFiles, diagnostics } = await validateAgentGuidance({
rootDir: process.cwd(),
});
if (diagnostics.length) {
for (const diagnostic of diagnostics) console.error(diagnostic);
process.exitCode = 1;
} else console.log(`Validated ${checkedFiles} agent guidance files.`);
}