From 3f8b5be06d34c94bce131cbdbf70549bcd99125c Mon Sep 17 00:00:00 2001 From: 4gray Date: Mon, 21 Sep 2026 02:21:11 +0200 Subject: [PATCH] fix(agents): preserve block boundaries and validate embeds --- docs/development/agent-workflow.md | 4 +- tools/skills/agent-guidance-markdown.mjs | 41 +++++++++++++++++-- tools/skills/validate-agent-guidance.test.mjs | 36 ++++++++++++++++ 3 files changed, 76 insertions(+), 5 deletions(-) diff --git a/docs/development/agent-workflow.md b/docs/development/agent-workflow.md index 459d081db..9ad64eb05 100644 --- a/docs/development/agent-workflow.md +++ b/docs/development/agent-workflow.md @@ -58,7 +58,7 @@ Explicit HTML anchors use `parse5`, excluding comments, scripts, styles and temp Rendered HTML anchor and image-map area hrefs and image sources use the same local-reference checks as Markdown links, including decoded attributes and fragment validation. URL attributes discard surrounding ASCII control/space characters before resolution. -Iframe sources are document references and retain Markdown-target anchor checks. +Iframe/embed sources and object data attributes are document references and retain Markdown-target anchor checks. Image and media references must resolve to files, not directories. Image references check file existence without interpreting image fragments as Markdown headings; document links keep anchor checks even when sharing a target. @@ -85,7 +85,7 @@ separate package mentions from adjacent prose. Markdown destinations decode HTML entities before URI parsing, matching rendered links. Heading-anchor lookup is limited to Markdown targets. Source-file line fragments, PDF page fragments and other non-Markdown fragments retain file-existence checks. -The import scan includes visible HTML text and literal backticks; it excludes parsed code nodes and non-rendered containers. +The import scan separates HTML block/table elements and includes visible text and literal backticks; it excludes parsed code nodes and non-rendered containers. Navigation uses the parsed rendered tree too. Temporary in-memory markers retain definition, unresolved-reference and literal-path metadata, so Markdown inside inert templates is excluded consistently with raw HTML navigation. diff --git a/tools/skills/agent-guidance-markdown.mjs b/tools/skills/agent-guidance-markdown.mjs index 997b96476..166c608f5 100644 --- a/tools/skills/agent-guidance-markdown.mjs +++ b/tools/skills/agent-guidance-markdown.mjs @@ -78,16 +78,20 @@ function htmlNavigation(html, inspect = () => {}) { 'source', 'track', 'iframe', + 'embed', ].includes(node.tagName) && attribute.name === 'src') || - (node.tagName === 'video' && attribute.name === 'poster') + (node.tagName === 'video' && attribute.name === 'poster') || + (node.tagName === 'object' && attribute.name === 'data') ) references.push({ target: attribute.value.replace( /^[\u0000-\u0020]+|[\u0000-\u0020]+$/gu, '' ), - image: !['a', 'area', 'iframe'].includes(node.tagName), + image: !['a', 'area', 'iframe', 'object', 'embed'].includes( + node.tagName + ), }); if ( ['img', 'source'].includes(node.tagName) && @@ -113,6 +117,37 @@ export function guidanceProse(markdown) { if (node.nodeName === '#text') return node.value; const content = (node.childNodes ?? []).map(text).join(''); return [ + 'address', + 'article', + 'aside', + 'details', + 'summary', + 'dialog', + 'dl', + 'dt', + 'dd', + 'fieldset', + 'legend', + 'figure', + 'figcaption', + 'footer', + 'form', + 'header', + 'hgroup', + 'hr', + 'main', + 'nav', + 'ol', + 'ul', + 'section', + 'table', + 'caption', + 'thead', + 'tbody', + 'tfoot', + 'tr', + 'td', + 'th', 'p', 'li', 'blockquote', @@ -125,7 +160,7 @@ export function guidanceProse(markdown) { 'h5', 'h6', ].includes(node.tagName) - ? content + '\n' + ? '\n' + content + '\n' : content; } return text(parseFragment(new Marked().parse(markdown))); diff --git a/tools/skills/validate-agent-guidance.test.mjs b/tools/skills/validate-agent-guidance.test.mjs index 63e36d28f..909594c2a 100644 --- a/tools/skills/validate-agent-guidance.test.mjs +++ b/tools/skills/validate-agent-guidance.test.mjs @@ -1334,3 +1334,39 @@ for (const reference of [ ); }); } + +for (const html of [ + '
Use @angular/core

for Angular

', + '
Use @angular/corefor Angular
', + '
Use @angular/core
', +]) { + test(`HTML blocks separate package prose: ${html}`, async (t) => { + assert.deepEqual( + await diagnostics(t, { + 'AGENTS.md': html, + 'package.json': JSON.stringify({ + dependencies: { '@angular/core': '*' }, + }), + }), + [] + ); + }); +} +for (const tag of ['object', 'embed']) { + test(`embedded document targets are checked: ${tag}`, async (t) => { + const markup = (target) => + `<${tag} ${tag === 'object' ? 'data' : 'src'}="${target}">`; + for (const target of ['docs/missing.pdf', 'docs/example.md#missing']) { + assert.ok( + (await diagnostics(t, { 'AGENTS.md': markup(target) })).length > + 0 + ); + } + assert.deepEqual( + await diagnostics(t, { + 'AGENTS.md': markup('docs/example.md#repeat'), + }), + [] + ); + }); +}