diff --git a/.gitignore b/.gitignore index fc3c011..1c0069b 100644 --- a/.gitignore +++ b/.gitignore @@ -6,6 +6,7 @@ docs/.vitepress/cache/ docs/deploy .DS_Store .claude/ +.vscode/ # Raw output from working-notes/run-parity-baseline.sh (snapshots go stale; keep the tables in the notes instead) parity-results/ diff --git a/docs/checks/content-discoverability.md b/docs/checks/content-discoverability.md index 8578f09..ef1f0ab 100644 --- a/docs/checks/content-discoverability.md +++ b/docs/checks/content-discoverability.md @@ -84,6 +84,12 @@ Whether your `llms.txt` follows the [llmstxt.org](https://llmstxt.org/) structur A well-structured `llms.txt` gives agents a reliable map of the documentation. Inconsistent implementations reduce its value, though even a non-standard file with useful links is better than nothing. +### What is measured + +AFDocs recognizes both inline links (`[Guide](/guide)`) and reference links (`[Guide][intro]` with an `[intro]: /guide` definition). It uses CommonMark parsing, including links in Markdown tables. Links inside code examples or HTML comments do not count. Standalone images do not count as navigation links, but a linked image or badge contributes its enclosing link. + +Discovery, coverage, and the llms.txt link checks use this same extraction. Destinations with balanced parentheses, such as `/guide_(intro)`, are preserved. Markdown escapes and HTML entities are decoded once, so `/search?a=1&b=2` is checked as `/search?a=1&b=2`. Plain URL text and angle-bracket autolinks such as `` are not included; use inline or reference links for index entries. + ### Results | Result | Condition | diff --git a/docs/checks/content-structure.md b/docs/checks/content-structure.md index a4db483..4c9e34f 100644 --- a/docs/checks/content-structure.md +++ b/docs/checks/content-structure.md @@ -121,7 +121,9 @@ The check reads the same markdown responses the other markdown checks already fe A link whose address is not a valid URL at all fails the page rather than being counted as a well-formed absolute link. -Same-document fragment links (`#anchor` with no path) are exempt: they resolve within the content the agent already holds, and rewriting them to absolute URLs adds nothing. Links with a non-HTTP scheme (`mailto:`, `tel:`) are exempt for the same reason. Links inside fenced code blocks and inline code are ignored, so documentation that shows example markdown is not graded on its examples. Image references are classified and reported but never affect the result: they point at assets rather than at documentation an agent navigates to. +Same-document fragment links (`#anchor` with no path) are exempt: they resolve within the content the agent already holds, and rewriting them to absolute URLs adds nothing. Links with a non-HTTP scheme (`mailto:`, `tel:`) are exempt for the same reason. Links inside fenced or indented code blocks, inline code, and HTML comments are ignored, so documentation that shows example markdown is not graded on its examples. Image references are classified and reported but never affect the result: they point at assets rather than at documentation an agent navigates to. + +Link destinations follow CommonMark rules: escaped punctuation and HTML entities are decoded once before resolving the URL. Plain URL text and angle-bracket autolinks are not included in the portability tally. Relative links are resolved against the URL that served the markdown, not the page URL. For a site serving `/docs/api` as `/md/docs/api.md`, `guide.md` means `/md/docs/guide.md`. diff --git a/src/checks/content-discoverability/llms-txt-valid.ts b/src/checks/content-discoverability/llms-txt-valid.ts index d53dea7..45d53c4 100644 --- a/src/checks/content-discoverability/llms-txt-valid.ts +++ b/src/checks/content-discoverability/llms-txt-valid.ts @@ -1,5 +1,6 @@ import { registerCheck } from '../registry.js'; import { getLlmsTxtFilesForAnalysis } from '../../helpers/llms-txt.js'; +import { scanRawLinks } from '../../helpers/classify-markdown-links.js'; import type { CheckContext, CheckResult } from '../../types.js'; interface ValidationResult { @@ -11,15 +12,11 @@ interface ValidationResult { issues: string[]; } -/** Extract markdown links from text: [name](url) */ +/** Extract inline and reference links, excluding images, code and comments. */ export function extractMarkdownLinks(content: string): Array<{ name: string; url: string }> { - const linkRegex = /\[([^\]]+)\]\(([^\s)]+)(?:\s+["'][^"']*["'])?\)/g; - const links: Array<{ name: string; url: string }> = []; - let match; - while ((match = linkRegex.exec(content)) !== null) { - links.push({ name: match[1], url: match[2] }); - } - return links; + return scanRawLinks(content) + .links.filter((link) => !link.isImage && link.destination !== '') + .map((link) => ({ name: link.text, url: link.destination })); } function validateLlmsTxt(content: string, url: string): ValidationResult { diff --git a/src/helpers/classify-markdown-links.ts b/src/helpers/classify-markdown-links.ts index 3c91f04..a37ba38 100644 --- a/src/helpers/classify-markdown-links.ts +++ b/src/helpers/classify-markdown-links.ts @@ -9,6 +9,9 @@ * points once resolved. */ +import type { Nodes } from 'mdast'; +import { parseMarkdown } from './parse-markdown.js'; + /** * How much of the base URL a link needs to be reconstructable. * @@ -30,7 +33,7 @@ export type LinkClass = | 'other-scheme'; export interface MarkdownLink { - /** The URL exactly as written in the markdown. */ + /** The destination after CommonMark escape and entity decoding. */ url: string; /** The link text, collapsed and shortened. */ text: string; @@ -57,7 +60,7 @@ export interface MarkdownLink { } export interface MarkdownLinkScan { - /** Links an agent would follow, in document order, deduplicated by raw URL. */ + /** Links an agent would follow, in document order, deduplicated by decoded URL. */ links: MarkdownLink[]; /** * Image references, classified the same way but kept separate: they point @@ -134,122 +137,6 @@ export function promisesMarkdownUrl(url: string): boolean { } } -function blankRun(text: string): string { - return text.replace(/[^\n]/g, ' '); -} - -/** - * Replace fenced code blocks and inline code spans with spaces of the same - * length, so that example links inside code samples are not read as the - * page's own links. Length is preserved so the result lines up with the - * original. - * - * Fences are matched by scanning lines rather than with a backreference, - * because CommonMark lets a closing fence be longer than its opener, and - * because a fence nested inside a list item carries the list's indent. - * Deeply indented fences are ordinary in tutorial documentation, and a fence - * this function fails to recognize puts every example link inside it into - * the scan. An opener is therefore accepted at any indent: within served - * markdown, a line that is nothing but three or more backticks or tildes is - * a fence, and erring toward blanking costs recall while erring the other - * way invents broken links. - */ -function blankCode(text: string): string { - const lines = text.split('\n'); - let open: { char: string; length: number } | null = null; - - for (let i = 0; i < lines.length; i++) { - const line = lines[i]; - const match = /^[ \t]*(`{3,}|~{3,})(.*)$/.exec(line); - - if (open) { - // Everything inside an open fence is blanked, whatever it contains. - // The table-cell guard below must not reach here: a fenced shell - // example such as `cat x | grep '[a](../b.md)'` would otherwise stay - // in the scan and fail the page on a link that is sample code. - lines[i] = blankRun(line); - if ( - match && - match[1][0] === open.char && - match[1].length >= open.length && - match[2].trim() === '' - ) { - open = null; - } - continue; - } - - if (!match) continue; - // Fences inside markdown table cells are a vendor extension, not real - // CommonMark fences; `markdown-code-fence-validity` skips them too. This - // only prevents *opening* a fence. - if (line.includes('|')) continue; - // A backtick fence's info string may not itself contain a backtick. - if (match[1][0] === '`' && match[2].includes('`')) continue; - open = { char: match[1][0], length: match[1].length }; - lines[i] = blankRun(line); - } - - // An unclosed fence runs to the end of the document, per CommonMark, so - // whatever followed it has already been blanked. - return blankCodeSpans(lines.join('\n')); -} - -/** - * Blank inline code spans the way CommonMark delimits them: a backtick - * string of length N opens a span, the next backtick string of exactly - * length N closes it, and the span may run across lines. A backtick run with - * no matching closer is literal text, not a delimiter. - * - * A regex over single backticks on one line gets both directions wrong: a - * double-backtick span wrapping a link on the following line stays visible - * to the scanner and fails the page on sample code, while a backslash-escaped - * backtick reads as a delimiter and can hide a real link. - */ -function blankCodeSpans(text: string): string { - const out = text.split(''); - let i = 0; - - while (i < text.length) { - if (text[i] !== '`' || isEscapedAt(text, i)) { - i++; - continue; - } - const start = i; - while (i < text.length && text[i] === '`') i++; - const runLength = i - start; - - let j = i; - let closed = false; - while (j < text.length) { - if (text[j] !== '`' || isEscapedAt(text, j)) { - j++; - continue; - } - const runStart = j; - while (j < text.length && text[j] === '`') j++; - if (j - runStart === runLength) { - for (let k = start; k < j; k++) { - if (out[k] !== '\n') out[k] = ' '; - } - i = j; - closed = true; - break; - } - } - if (!closed) i = start + runLength; - } - - return out.join(''); -} - -/** True when the character at `i` is preceded by an odd run of backslashes. */ -function isEscapedAt(text: string, i: number): boolean { - let count = 0; - for (let j = i - 1; j >= 0 && text[j] === '\\'; j--) count++; - return count % 2 === 1; -} - function shorten(text: string): string { const flat = text.replace(/\s+/g, ' ').trim(); return flat.length > MAX_TEXT ? `${flat.slice(0, MAX_TEXT - 1)}…` : flat; @@ -258,6 +145,7 @@ function shorten(text: string): string { /** A link as written, before classification: where it sits and what it points at. */ export interface RawMarkdownLink { text: string; + /** Destination decoded once by CommonMark, not a slice of the source. */ destination: string; isImage: boolean; /** Offset of the opening `[` (or the `!` of an image) in the scanned text. */ @@ -266,232 +154,71 @@ export interface RawMarkdownLink { end: number; } -type RawLink = RawMarkdownLink; - -/** - * A CommonMark link reference definition on its own line: - * `[label]: destination "optional title"`. - */ -const REFERENCE_DEFINITION = - /^[ \t]{0,3}\[((?:\\.|[^\\[\]\n])+)\]:[ \t]*(?:<([^>\n]*)>|(\S+))(?:[ \t]+(?:"[^"\n]*"|'[^'\n]*'|\([^)\n]*\)))?[ \t]*$/gm; - -function normalizeLabel(label: string): string { - return label.trim().replace(/\s+/g, ' ').toLowerCase(); -} - -/** - * Collect link reference definitions and blank their lines, so a definition - * is never read as a shortcut reference to itself. The first definition of a - * label wins, per CommonMark. - */ -function collectReferenceDefinitions(text: string): { defs: Map; text: string } { - const defs = new Map(); - const out = text.replace( - REFERENCE_DEFINITION, - (m, label: string, angled: string | undefined, bare: string | undefined) => { - const key = normalizeLabel(label); - if (!defs.has(key)) defs.set(key, angled ?? bare ?? ''); - return blankRun(m); - }, - ); - return { defs, text: out }; -} - -function isSpace(ch: string): boolean { - return ch === ' ' || ch === '\t' || ch === '\n' || ch === '\r'; +function linkText(node: Nodes): string { + if (node.type === 'text' || node.type === 'inlineCode') return node.value; + if (node.type === 'image' || node.type === 'imageReference') return node.alt ?? ''; + if (node.type === 'break') return '\n'; + return 'children' in node ? node.children.map(linkText).join('') : ''; } /** - * CommonMark backslash escapes apply to ASCII punctuation only. `\w` is a - * literal backslash followed by `w`, not an escaped `w`, which is what keeps - * a regex literal in an MDX preamble from unescaping into a plausible URL. - */ -const ASCII_PUNCTUATION = /[!"#$%&'()*+,\-./:;<=>?@[\\\]^_`{|}~]/; - -function isEscape(text: string, i: number): boolean { - return text[i] === '\\' && i + 1 < text.length && ASCII_PUNCTUATION.test(text[i + 1]); -} - -/** - * Read a CommonMark inline-link destination starting just after the `(`. - * - * Two forms: an angle-bracket destination, which may contain spaces, and a - * bare destination, which runs to whitespace or to the closing paren and may - * contain balanced parentheses. Getting the second right matters: a regex - * that stops at the first `)` turns - * `https://host/chapter_(draft).md` into `https://host/chapter_(draft`, and - * this check would then fetch that truncated URL and report a broken link - * that does not exist. - */ -function readDestination(text: string, start: number): { value: string; end: number } | null { - let i = start; - while (i < text.length && isSpace(text[i])) i++; - - let value = ''; - if (text[i] === '<') { - i++; - while (i < text.length && text[i] !== '>' && text[i] !== '\n') { - if (isEscape(text, i)) { - value += text[i + 1]; - i += 2; - continue; - } - value += text[i++]; - } - if (text[i] !== '>') return null; - i++; - } else { - let depth = 0; - while (i < text.length) { - const ch = text[i]; - if (isSpace(ch)) break; - if (isEscape(text, i)) { - value += text[i + 1]; - i += 2; - continue; - } - if (ch === '(') depth++; - else if (ch === ')') { - if (depth === 0) break; - depth--; - } - value += ch; - i++; - } - // An unbalanced `(` means the destination was never closed; treating what - // was read as a URL is how a truncated target becomes a phantom 404. - if (depth !== 0) return null; - } - - while (i < text.length && isSpace(text[i])) i++; - const quote = text[i]; - if (quote === '"' || quote === "'" || quote === '(') { - const closer = quote === '(' ? ')' : quote; - i++; - while (i < text.length && text[i] !== closer) { - if (text[i] === '\\') { - i += 2; - continue; - } - i++; - } - if (text[i] !== closer) return null; - i++; - while (i < text.length && isSpace(text[i])) i++; - } - - if (text[i] !== ')') return null; - return { value, end: i + 1 }; -} - -/** - * Read the reference part of a link whose bracketed text has just closed at - * `j`: `[text][label]`, `[text][]` (collapsed), or `[text]` alone (shortcut). - * Returns the destination when the label is defined, else null. - */ -function readReference( - text: string, - label: string, - j: number, - defs: Map, -): { value: string; end: number } | null { - if (defs.size === 0) return null; - if (text[j] === '[') { - const close = text.indexOf(']', j + 1); - if (close === -1) return null; - const ref = text.slice(j + 1, close); - if (ref.includes('[') || ref.includes('\n\n')) return null; - const value = defs.get(normalizeLabel(ref === '' ? label : ref)); - return value === undefined ? null : { value, end: close + 1 }; - } - const value = defs.get(normalizeLabel(label)); - return value === undefined ? null : { value, end: j }; -} - -/** - * Collect links and images in document order: inline (`[text](dest)`, - * `![alt](src)`) and reference-style (`[text][label]`, `[text][]`, `[label]`) - * with `defs` holding the reference definitions. Autolinks and bare URLs are - * deliberately not collected: see `scanMarkdownLinks`. + * Scan a markdown document for the links an agent could follow, without + * classifying them. Code, HTML comments and reference definitions are + * blanked in a same-length copy, preserving CR/LF and original UTF-16 offsets. + * Destinations and labels use CommonMark decoding; autolinks are excluded. */ -function scanInlineLinks(text: string, defs: Map, base = 0): RawLink[] { - const found: RawLink[] = []; - let i = 0; - - while (i < text.length) { - const open = text.indexOf('[', i); - if (open === -1) break; - // CommonMark renders `\[example](../x)` as literal text, so an escaped - // bracket opens nothing. Without this, documentation that shows escaped - // markdown syntax is graded on its own examples. - if (isEscapedAt(text, open)) { - i = open + 1; - continue; +export function scanRawLinks(content: string): { links: RawMarkdownLink[]; blanked: string } { + const tree = parseMarkdown(content); + const definitions = new Map(); + const links: RawMarkdownLink[] = []; + const blanked = content.split(''); + + const collectDefinitions = (node: Nodes): void => { + if (node.type === 'definition' && !definitions.has(node.identifier)) { + definitions.set(node.identifier, node.url); } - - let depth = 1; - let j = open + 1; - while (j < text.length && depth > 0) { - const ch = text[j]; - if (isEscape(text, j)) { - j += 2; - continue; + if ('children' in node) node.children.forEach(collectDefinitions); + }; + collectDefinitions(tree); + + const visit = (node: Nodes): void => { + const offset = node.position?.start.offset; + const end = node.position?.end.offset; + if (offset === undefined || end === undefined) return; + + if ( + node.type === 'code' || + node.type === 'inlineCode' || + node.type === 'definition' || + (node.type === 'html' && node.value.trimStart().startsWith(''; + ctx.previousResults.set('llms-txt-exists', { + id: 'llms-txt-exists', + category: 'content-discoverability', + status: 'pass', + message: 'Found', + details: { + discoveredFiles: [{ url: `${origin}/llms.txt`, content, status: 200, redirected: false }], + }, + }); + server.use( + http.all(`${origin}/*`, ({ request }) => { + requests.push(`${request.method} ${request.url}`); + if (request.url === `${origin}/robots.txt`) + return new HttpResponse(`Sitemap: ${origin}/sitemap.xml`); + if (request.url === `${origin}/sitemap.xml`) + return new HttpResponse( + makeSitemap([ + `${origin}/docs/guide_(intro)`, + `${origin}/docs/api`, + `${origin}/docs/excluded`, + `${origin}/blog/outside`, + ]), + { headers: { 'Content-Type': 'application/xml' } }, + ); + return new HttpResponse(null, { status: 404 }); + }), + ); + const result = await check.run(ctx); + expect(result.status).toBe('pass'); + expect(result.details).toMatchObject({ coverageRate: 100, missingCount: 0, unmatchedCount: 0 }); + expect(requests).toEqual([`GET ${origin}/robots.txt`, `GET ${origin}/sitemap.xml`]); + }); + test('passes when llms.txt fully covers sitemap', async () => { const host = 'cov-pass.local'; const pages = [ diff --git a/test/unit/checks/llms-txt-links-markdown.test.ts b/test/unit/checks/llms-txt-links-markdown.test.ts index 12e84d4..59adf65 100644 --- a/test/unit/checks/llms-txt-links-markdown.test.ts +++ b/test/unit/checks/llms-txt-links-markdown.test.ts @@ -68,6 +68,22 @@ Just text, no links here. expect(result.details?.markdownRate).toBe(100); }); + it('checks real references for Markdown without requesting hidden examples', async () => { + const requests: string[] = []; + server.use( + http.all('http://test.local/*', ({ request }) => { + requests.push(`${request.method} ${request.url}`); + return new HttpResponse(null, { headers: { 'Content-Type': 'text/markdown' } }); + }), + ); + const content = + '[guide][ref]\n\n[ref]: http://test.local/guide_(intro)\n\n [code](http://test.local/code)\n\n'; + const result = await check.run(makeCtx(content)); + expect(result.status).toBe('pass'); + expect(result.details?.markdownRate).toBe(100); + expect(requests).toEqual(['HEAD http://test.local/guide_(intro)']); + }); + it('fails when same-origin links are HTML with no markdown alternatives', async () => { server.use( http.head( diff --git a/test/unit/checks/llms-txt-links-resolve.test.ts b/test/unit/checks/llms-txt-links-resolve.test.ts index 934b810..01b4bd9 100644 --- a/test/unit/checks/llms-txt-links-resolve.test.ts +++ b/test/unit/checks/llms-txt-links-resolve.test.ts @@ -57,6 +57,28 @@ describe('llms-txt-links-resolve', () => { expect(result.details?.resolved).toBe(1); }); + it.each([200, 404])( + 'checks reference links normally (HTTP %i), never code or comments', + async (status) => { + const url = 'http://test.local/guide_(intro)?x=&y'; + const requests: string[] = []; + server.use( + http.all('http://test.local/*', ({ request }) => { + requests.push(`${request.method} ${request.url}`); + return new HttpResponse(null, { status }); + }), + ); + const content = `# Docs\n\n[guide][ref]\n\n[ref]: http://test.local/guide_(intro)?x=&amp;y\n\n [code](http://test.local/code)\n\n\n\n![image](http://test.local/image.png)`; + const result = await check.run(makeCtx(content)); + expect(result.status).toBe(status === 200 ? 'pass' : 'fail'); + expect(result.details?.sameOrigin).toMatchObject({ + tested: 1, + resolved: status === 200 ? 1 : 0, + }); + expect(requests).toEqual([`HEAD ${url}`]); + }, + ); + it('warns when resolve rate is above threshold but not 100%', async () => { // 10 resolve, 1 broken = ~91% resolve rate (> 0.9 threshold → warn) const urls: string[] = []; diff --git a/test/unit/checks/llms-txt-valid.test.ts b/test/unit/checks/llms-txt-valid.test.ts index 720ffd2..a0c74ab 100644 --- a/test/unit/checks/llms-txt-valid.test.ts +++ b/test/unit/checks/llms-txt-valid.test.ts @@ -10,6 +10,30 @@ import type { DiscoveredFile } from '../../../src/types.js'; const FIXTURES = resolve(import.meta.dirname, '../../fixtures/llms-txt'); describe('extractMarkdownLinks', () => { + it('keeps duplicate navigation links but excludes standalone images and autolinks', () => { + const content = + '![logo](/logo.svg) [![badge](/badge.svg)](/guide) [guide](/guide) [empty]()'; + expect(extractMarkdownLinks(content)).toEqual([ + { name: 'badge', url: '/guide' }, + { name: 'guide', url: '/guide' }, + ]); + }); + + it('preserves nested destinations and resolves reference links', () => { + expect(extractMarkdownLinks('[guide](/guide_(intro))\n\n[API][ref]\n\n[ref]: /api')).toEqual([ + { name: 'guide', url: '/guide_(intro)' }, + { name: 'API', url: '/api' }, + ]); + }); + + it('ignores indented code and HTML comments', () => { + expect( + extractMarkdownLinks( + ' [example](/example)\n\n\n\n[real](/real)', + ), + ).toEqual([{ name: 'real', url: '/real' }]); + }); + it('extracts links from markdown', () => { const content = '- [Foo](https://example.com/foo): A foo\n- [Bar](https://example.com/bar)'; const links = extractMarkdownLinks(content); @@ -68,6 +92,22 @@ describe('llms-txt-valid', () => { expect(result.status).toBe('pass'); }); + it('counts reference links without changing structural validation', async () => { + const result = await runWithContent( + '# Docs\n\n> Summary\n\n## Guides\n\n[guide][ref]\n\n[ref]: /guide_(intro)', + ); + expect(result.status).toBe('pass'); + expect(result.details?.validations).toMatchObject([{ linkCount: 1, issues: [] }]); + }); + + it('fails when the only apparent links are examples, comments or images', async () => { + const result = await runWithContent( + '# Docs\n\n> Summary\n\n## Guides\n\n [code](/code)\n\n\n\n![logo](/logo.svg)', + ); + expect(result.status).toBe('fail'); + expect(result.details?.validations).toMatchObject([{ linkCount: 0 }]); + }); + it('warns for llms.txt without H1', async () => { const content = await readFile(resolve(FIXTURES, 'no-h1.txt'), 'utf-8'); const result = await runWithContent(content); diff --git a/test/unit/checks/markdown-link-portability.test.ts b/test/unit/checks/markdown-link-portability.test.ts index b0ca378..6ed88b9 100644 --- a/test/unit/checks/markdown-link-portability.test.ts +++ b/test/unit/checks/markdown-link-portability.test.ts @@ -99,6 +99,25 @@ describe('markdown-link-portability', () => { expect(result.message).toBe('No links found in 1 markdown pages'); }); + it('verifies only rendered navigation links, preserving decoded targets', async () => { + const destination = `${ORIGIN}/docs/guide_(intro).md?x=&y`; + const requests: string[] = []; + server.use( + http.all(`${ORIGIN}/*`, ({ request }) => { + requests.push(`${request.method} ${request.url}`); + return new HttpResponse('# Guide\n\nContent.', { + headers: { 'Content-Type': 'text/markdown' }, + }); + }), + ); + const content = `[guide][ref]\n\n[ref]: ${ORIGIN}/docs/guide_(intro).md?x=&amp;y\n\n [code](../broken.md)\n\n\n\n![image](../image.png)`; + const result = await check.run(cachedCtx([{ url: `${ORIGIN}/docs/a`, content }])); + expect(result.status).toBe('pass'); + expect(pageResults(result)[0].links).toMatchObject({ absolute: 1, pathRelative: 0, total: 1 }); + expect(pageResults(result)[0].samples).toMatchObject([{ url: destination, outcome: 'ok' }]); + expect(requests).toEqual([`GET ${destination}`]); + }); + it('warns on root-relative links that still resolve', async () => { server.use(http.get(`${ORIGIN}/docs/b.md`, markdown('# B\n\nContent.\n'))); const content = '# A\n\nSee [B](/docs/b.md).\n'; diff --git a/test/unit/helpers/classify-markdown-links.test.ts b/test/unit/helpers/classify-markdown-links.test.ts index 991aaea..3744c1b 100644 --- a/test/unit/helpers/classify-markdown-links.test.ts +++ b/test/unit/helpers/classify-markdown-links.test.ts @@ -3,10 +3,158 @@ import { classifyLink, countByClass, scanMarkdownLinks, + scanRawLinks, } from '../../../src/helpers/classify-markdown-links.js'; const BASE = 'https://docs.example.com/md/guide/api.md'; +describe('scanRawLinks', () => { + it.each(['\n', '\r\n'])('excludes code and comments in nested containers (%j)', (newline) => { + const content = [ + ' [indented](/indented)', + '', + '> - ```md', + '> [fenced](/fenced)', + '> ````', + '', + '- item', + '', + ' [nested code](/nested-code)', + '', + 'Inline `[code](/inline)` and [real](/real)', + '', + ' ', + '', + ' ', + '', + ' ', + '', + '> ', + '', + '- ', + '', + '[last](/last)', + '', + ' \n\n[real](/real)'; + expect(scanRawLinks(content).links.map((link) => link.destination)).toEqual(['/real']); + }); +}); + describe('classifyLink', () => { it('classifies by how much of the base URL the link needs', () => { expect(classifyLink('https://docs.example.com/a')).toBe('absolute'); diff --git a/test/unit/helpers/detect-pagination.test.ts b/test/unit/helpers/detect-pagination.test.ts index ac43dc5..535e5e9 100644 --- a/test/unit/helpers/detect-pagination.test.ts +++ b/test/unit/helpers/detect-pagination.test.ts @@ -24,6 +24,19 @@ describe('parseLinkHeaderNext', () => { }); describe('detectPagination', () => { + it('uses decoded reference destinations once and keeps original continuation offsets', () => { + const content = + 'Pr\u00e9face \ud83d\ude80\r\n\r\n\r\n\r\n[**Next page**][ref]\r\n\r\n[ref]: /models.md?x=&amp;y&page=2'; + const result = detectPagination(content, { baseUrl: BASE }); + expect(result.continuation).toMatchObject({ + url: '/models.md?x=&y&page=2', + resolvedUrl: 'https://docs.example.com/models.md?x=&y&page=2', + offset: content.indexOf('[**Next page**]'), + declaredIn: 'content', + }); + expect(result.signals.every((signal) => signal.url !== '/hidden?page=2')).toBe(true); + }); + describe('complete content', () => { it('finds nothing in ordinary documentation', () => { const content = `# Install\n\nRun the installer, then [configure](./configure.md) the client.\n\n## Step 2 of 5\n\nPart 1 of 3 covers setup.`; diff --git a/test/unit/helpers/get-markdown-content.test.ts b/test/unit/helpers/get-markdown-content.test.ts index 8e68ebe..65cb0ec 100644 --- a/test/unit/helpers/get-markdown-content.test.ts +++ b/test/unit/helpers/get-markdown-content.test.ts @@ -432,6 +432,32 @@ describe('fetchLlmsTxtLinkedMarkdown', () => { expect(await fetchLlmsTxtLinkedMarkdown(ctx)).toEqual([]); }); + it('fetches parsed references within the existing limit and memoizes them', async () => { + const requests: Array<{ url: string; method: string; accept: string | null }> = []; + server.use( + http.all('http://test.local/*', ({ request }) => { + requests.push({ + url: request.url, + method: request.method, + accept: request.headers.get('Accept'), + }); + return new HttpResponse('# Guide\n\nContent.', { + headers: { 'Content-Type': 'text/markdown' }, + }); + }), + ); + const content = + '[guide][ref]\n\n[ref]: http://test.local/guide_(intro).md?x=&amp;y\n\n[second](http://test.local/second.md)\n\n[overflow](http://test.local/overflow.md)\n\n [code](http://test.local/code.md)\n\n\n\n[external](https://other.example/guide.md)'; + const ctx = ctxWithLlmsTxt(content, { maxLinksToTest: 2 }); + const pages = await fetchLlmsTxtLinkedMarkdown(ctx); + expect(pages.map((page) => page.url)).toEqual([ + 'http://test.local/guide_(intro).md?x=&y', + 'http://test.local/second.md', + ]); + expect(await fetchLlmsTxtLinkedMarkdown(ctx)).toBe(pages); + expect(requests).toEqual(pages.map((page) => ({ url: page.url, method: 'GET', accept: null }))); + }); + it('fetches same-origin links that serve markdown and records the Link header', async () => { server.use( http.get( diff --git a/test/unit/helpers/get-page-urls.test.ts b/test/unit/helpers/get-page-urls.test.ts index 5b7670a..e2b0e54 100644 --- a/test/unit/helpers/get-page-urls.test.ts +++ b/test/unit/helpers/get-page-urls.test.ts @@ -808,6 +808,64 @@ describe('getPageUrls', () => { expect(result.sources).toContain('llms-txt'); }); + it('discovers real reference links without probing pages or walking example indexes', async () => { + const origin = 'http://parsed-discovery.local'; + const requests: string[] = []; + const content = [ + '# Docs', + '', + '[guide][guide]', + '[nested][index]', + '[outside](/blog/post)', + '[external](https://other.example/docs/page)', + '', + '[guide]: /docs/guide_(intro).md', + '[index]: /docs/nested.txt', + '', + ' [example](/docs/code.txt)', + '', + '', + ].join('\n'); + server.use( + http.all(`${origin}/*`, ({ request }) => { + requests.push(`${request.method} ${request.url}`); + if (request.url === `${origin}/docs/nested.txt`) { + return new HttpResponse( + '[API][ref]\n\n[ref]: /docs/api.md\n\n[deeper](/docs/deeper.txt)\n\n [code](/docs/code-page)\n\n', + ); + } + if (request.url === `${origin}/robots.txt`) + return new HttpResponse(`Sitemap: ${origin}/sitemap.xml`); + if (request.url === `${origin}/sitemap.xml`) + return new HttpResponse('', { + headers: { 'Content-Type': 'application/xml' }, + }); + return new HttpResponse(null, { status: 404 }); + }), + ); + const ctx = createContext(`${origin}/docs`, { requestDelay: 0 }); + ctx.previousResults.set('llms-txt-exists', { + id: 'llms-txt-exists', + category: 'content-discoverability', + status: 'pass', + message: 'Found', + details: { + discoveredFiles: [{ url: `${origin}/llms.txt`, content, status: 200, redirected: false }], + }, + }); + const result = await getPageUrls(ctx); + expect(result.urls).toEqual([`${origin}/docs/guide_(intro)`, `${origin}/docs/api`]); + expect(result.originalMdUrls).toEqual({ + [`${origin}/docs/guide_(intro)`]: `${origin}/docs/guide_(intro).md`, + [`${origin}/docs/api`]: `${origin}/docs/api.md`, + }); + expect(requests).toEqual([ + `GET ${origin}/docs/nested.txt`, + `GET ${origin}/robots.txt`, + `GET ${origin}/sitemap.xml`, + ]); + }); + it.each(['llms-txt', 'sitemap'])( 'preserves version leaves discovered through %s (#135)', async (source) => { @@ -2669,6 +2727,43 @@ describe('discoverAndSamplePages', () => { expect(result.sources).toContain('llms-txt'); }); + it.each(['deterministic', 'random', 'curated'])( + 'preserves %s sampling and verification limits for references', + async (samplingStrategy) => { + const origin = `http://parsed-sampling-${samplingStrategy}.local`; + const urls = Array.from({ length: 10 }, (_, index) => `${origin}/page-${index}`); + const references = urls + .map((url, index) => `[page ${index}][ref${index}]\n\n[ref${index}]: ${url}.md`) + .join('\n\n'); + const content = `${references}\n\n [code](${origin}/code.md)\n\n`; + const ctx = makeCtx(origin, content, { + samplingStrategy, + maxLinksToTest: 3, + curatedPages: samplingStrategy === 'curated' ? urls : undefined, + }); + const requests: string[] = []; + server.use( + http.all(`${origin}/*`, ({ request }) => { + requests.push(`${request.method} ${request.url}`); + return new HttpResponse('Guide', { + headers: { 'Content-Type': 'text/html' }, + }); + }), + ); + const result = await discoverAndSamplePages(ctx); + expect(result.totalPages).toBe(10); + expect(result.urls).toHaveLength(samplingStrategy === 'curated' ? 10 : 3); + expect(result.sampled).toBe(samplingStrategy !== 'curated'); + expect(result.urls.every((url) => urls.includes(url))).toBe(true); + expect(requests).toEqual( + samplingStrategy === 'curated' ? [] : result.urls.map((url) => `GET ${url}`), + ); + if (samplingStrategy === 'deterministic') + expect(result.urls).toEqual([urls[0], urls[3], urls[6]]); + expect(await discoverAndSamplePages(ctx)).toBe(result); + }, + ); + it('samples down to maxLinksToTest when over limit', async () => { const links = Array.from( { length: 10 }, diff --git a/working-notes/page-discovery-notes.md b/working-notes/page-discovery-notes.md index 05e938c..f263d79 100644 --- a/working-notes/page-discovery-notes.md +++ b/working-notes/page-discovery-notes.md @@ -304,6 +304,62 @@ across roots and indexes, empty/error responses, raw coverage mode, fallback warning propagation, normal termination, and the strict 1% boundary. The initial locale and unbounded-walk tests both failed before the fix. +### September 2026: shared CommonMark link extraction (issue #151) + +The llms.txt link regex truncated `/guide_(intro)` and missed reference +links. Both it and the portability scanner treated indented code and HTML +comments as navigation. Discovery, coverage, llms.txt link checks, and +Markdown fetching now use the same parser-backed scan as portability and +pagination. The shared parser retains the GFM table extension from #150. + +Extraction decisions and intentional corrections: + +- Collect inline and resolved full, collapsed, and shortcut references in + document order. The first definition wins. Raw scanning and the exported + `extractMarkdownLinks()` retain duplicates; portability deduplicates by + decoded destination as before. Link labels now use rendered text, + including inline code and image alt text, rather than Markdown syntax. +- Exclude inline, fenced, and indented code and HTML comments, including + nested containers and unclosed comments. Markdown inside raw HTML blocks + is not parsed as navigation; HTML anchors are not Markdown links. +- Standalone images no longer count as llms.txt navigation links. A linked + badge contributes its enclosing destination, while portability continues + to report images separately without fetching or scoring them. +- Use parser-decoded destinations directly. CommonMark unescapes ASCII + punctuation and resolves the full entity set once. In particular, + `&amp;` becomes `&`, not `&`. The legacy exported + `decodeCharacterReferences()` remains available but is not applied to + parser output. Portability's unsafe-destination guard for raw MDX regexes + and templates remains, as do diagnostics for malformed URLs such as + `https://[`. +- Preserve UTF-16 offsets and exclusive end positions against the original + source, including non-ASCII characters and CRLF. The `blanked` result is + the original length, with code, comments, and definitions replaced by + spaces while preserving CR and LF. +- Keep autolinks and bare URLs out of shared navigation extraction. + Pagination still has its own bare-URL/path and declaration policies; + those are not discovery rules. It shares parser setup, not text analysis, + with the other consumers. +- Retain the vendor table-cell fence exception. A parsed code block whose + opening fence line contains a pipe has its opening marker masked with + same-length spaces and is reparsed. Repeat if this exposes another such + opener. The existing multiline table fixture otherwise swallows a real + following link. Pipes inside genuine code blocks remain code. + +Scope filters, URL mapping, traversal depth, and verification limits are +unchanged. No per-discovered-URL content probes were added. Real references +can now enter the existing sample and trigger normal checks; example links +no longer trigger requests or false stale/broken-link results. Correcting +the candidate set can change scores and request totals without changing +scoring thresholds or budgets. Random and deterministic samples retain +their caps, and curated input remains an uncapped explicit list. + +Regressions pin original-source spans, one-pass decoding, check outcomes, +coverage exclusions, nested-index traversal, and exact request lists for +discovery, sampling, fetching, and link verification. Heading extraction, +content-parity analysis, fence validity, and llms.txt structural validation +remain separate work. + ## Documentation at scale: design considerations Drafting `docs/documentation-at-scale.md` after the #120 live reproduction