From d8f7499fb11e85981bb51f5f4268aac1b313a9eb Mon Sep 17 00:00:00 2001 From: dacharyc Date: Mon, 28 Sep 2026 09:45:35 -0400 Subject: [PATCH 1/2] fix: use CommonMark parsing for Markdown content parity Replace Markdown text stripping and repeated-item parsing with the shared CommonMark/GFM AST. Preserve literal code, historical placeholders, and existing HTML extraction and comparison thresholds. Add 20 regressions covering rendered text, code containers, metadata exclusions, multiline catalogs, nested lists, and GFM tables. Document extraction semantics and keep vendor-fence normalization unchanged for discovery consumers. Refs #152. --- docs/checks/observability.md | 6 +- .../observability/markdown-content-parity.ts | 352 +++--------------- src/helpers/parse-markdown.ts | 6 +- .../checks/markdown-content-parity.test.ts | 332 +++++++++++++++++ working-notes/parity-check-notes.md | 68 +++- 5 files changed, 453 insertions(+), 311 deletions(-) diff --git a/docs/checks/observability.md b/docs/checks/observability.md index 7489727..4b9f64c 100644 --- a/docs/checks/observability.md +++ b/docs/checks/observability.md @@ -124,6 +124,8 @@ However, content divergence is sometimes intentional. Some sites serve different Based on the percentage of HTML content segments missing from the markdown version, after normalization. Thresholds are configurable. +Markdown is parsed as CommonMark with GitHub-style tables. Formatting is compared as rendered text, while code blocks and inline code retain their literal content. Image descriptions (alt text) are retained; comments, link destinations, and reference definitions are not treated as page text. Embedded HTML contributes its text, not its markup. Literal placeholders such as `` remain comparable. + | Result | Condition | | ------ | ------------------------------------------------------------- | | Pass | Under pass threshold (default 5%) of content segments missing | @@ -183,7 +185,9 @@ For pages with repeated structure, the check therefore also compares item counts - **`pagination`**: the markdown lists fewer, and `single-fetch-completeness` found the page paginated. The markdown is windowed, not stale. - **`staleness`**: the markdown lists fewer with no pagination in sight, most often one representation generated from older data. -Each has a different owner and fix. Only top-level items are counted on both sides, so entries that carry nested bullets are not counted several times over. Entries in the compared markdown list that repeat an earlier entry verbatim are counted as duplicates and excluded from the comparison, so a generator that emits every entry twice shows up as duplication rather than as a catalog twice the size. The comparison never changes the check's result; it feeds the [dynamic content rendered statically](/interaction-diagnostics#dynamic-content-rendered-statically) diagnostic, and is skipped when one side has fewer than five items, which more often means the check could not find the list on that side (a navigation-only list that was stripped, or a table drawn by JavaScript) than a real difference. +Each has a different owner and fix. Each list is measured by its direct items, including lists nested inside other lists. Nested bullets are not added to their parent list's count. Continuation paragraphs belong to the same item; code examples do not count as lists or tables. Table headers and separator rows are excluded. + +Entries in the compared markdown list with the same complete rendered text, after normalization, are counted as duplicates and excluded from the comparison. This includes continuation paragraphs and nested content; wrapping and formatting alone do not make an entry distinct. A generator that emits every entry twice therefore shows up as duplication rather than as a catalog twice the size. The comparison never changes the check's result; it feeds the [dynamic content rendered statically](/interaction-diagnostics#dynamic-content-rendered-statically) diagnostic, and is skipped when one side has fewer than five items, which more often means the check could not find the list on that side (a navigation-only list that was stripped, or a table drawn by JavaScript) than a real difference. ### How to fix diff --git a/src/checks/observability/markdown-content-parity.ts b/src/checks/observability/markdown-content-parity.ts index f56b61a..6781dfa 100644 --- a/src/checks/observability/markdown-content-parity.ts +++ b/src/checks/observability/markdown-content-parity.ts @@ -1,6 +1,8 @@ import { parse, NodeType, type HTMLElement, type Node } from 'node-html-parser'; +import type { Nodes, Root } from 'mdast'; import { registerCheck } from '../registry.js'; import { fetchPage } from '../../helpers/fetch-page.js'; +import { parseMarkdown } from '../../helpers/parse-markdown.js'; import { toHtmlUrl } from '../../helpers/to-md-urls.js'; import { DEFAULT_PARITY_PASS_THRESHOLD, DEFAULT_PARITY_WARN_THRESHOLD } from '../../constants.js'; import type { CheckContext, CheckResult, CheckStatus } from '../../types.js'; @@ -85,11 +87,11 @@ export interface ItemCounts { structure: 'list' | 'table'; /** Items in the HTML content container (list items or table data rows). */ html: number; - /** Items in the markdown (list-item lines or pipe-table data rows). */ + /** Items in the markdown (direct list items or GFM table data rows). */ markdown: number; /** Distinct markdown items; the comparison uses this when entries repeat. */ markdownUnique: number; - /** Entries in that markdown structure that repeat an earlier entry verbatim. */ + /** Entries in that markdown structure with the same normalized rendered text. */ duplicates: number; /** True when the counts differ by more than the tolerance. */ diverges: boolean; @@ -168,6 +170,18 @@ const BLOCK_TAGS = new Set([ 'hr', ]); +const HTML_TAG_NAMES = new Set( + ( + 'a abbr address area article aside audio b base bdi bdo blockquote body br button canvas caption ' + + 'cite code col colgroup data datalist dd del details dfn dialog div dl dt em embed fieldset ' + + 'figcaption figure footer form h1 h2 h3 h4 h5 h6 head header hgroup hr html i iframe img input ' + + 'ins kbd label legend li link main map mark menu meta meter nav noscript object ol optgroup ' + + 'option output p picture pre progress q rp rt ruby s samp script search section select slot ' + + 'small source span strong style sub summary sup svg table tbody td template textarea tfoot ' + + 'th thead time title tr track u ul var video wbr' + ).split(' '), +); + /** * Minimum link density (0–1) and minimum link count for an element to be * classified as navigation chrome. Navigation panels are structurally @@ -244,84 +258,35 @@ function countHtmlItems(content: HTMLElement): HtmlItemCounts { } interface MarkdownItemCounts { - /** Top-level items in the largest contiguous list (nested bullets excluded). */ + /** Direct items in the largest list (nested bullets not added to its count). */ listItems: number; - /** Distinct top-level items in that list. */ + /** Distinct rendered items in that list. */ uniqueListItems: number; /** Data rows in the largest pipe table. */ tableRows: number; } -const MD_FENCE = /^\s*(`{3,}|~{3,})/; -const MD_LIST_ITEM = /^(\s*)(?:[-*+]|\d+[.)])\s+(.*)$/; -const MD_TABLE_DELIMITER = /^\s*\|?\s*:?-+:?\s*(?:\|\s*:?-+:?\s*)*\|?\s*$/; - /** - * Measure the largest contiguous list (blank lines between items allowed, - * any other line ends it) and the largest pipe table in markdown, outside - * fenced code. Only the block's top-level items count, mirroring the HTML - * side's direct `li` children, so a catalog whose entries carry nested - * bullets is not counted three times over. Items are deduplicated by - * normalized text within the block, so a generator that emits every entry - * twice shows up as duplicates rather than as twice the catalog. + * Measure the largest list and GFM table, mirroring the HTML side's direct + * `li` children and data rows. Deduplicate complete rendered item text within + * the selected list, including continuation paragraphs and nested content. */ -function countMarkdownItems(markdown: string): MarkdownItemCounts { - const lines = markdown.split('\n'); - let inFence: { char: string; length: number } | null = null; +function countMarkdownItems(tree: Root): MarkdownItemCounts { let listItems = 0; let uniqueListItems = 0; let tableRows = 0; - let block: Array<{ indent: number; key: string }> = []; - - const closeBlock = () => { - if (block.length > 0) { - const base = Math.min(...block.map((b) => b.indent)); - const top = block.filter((b) => b.indent === base).map((b) => b.key); - if (top.length > listItems) { - listItems = top.length; - uniqueListItems = new Set(top).size; - } - } - block = []; - }; - for (let i = 0; i < lines.length; i++) { - const line = lines[i]; - const fence = MD_FENCE.exec(line); - if (fence) { - if (inFence === null) { - closeBlock(); - inFence = { char: fence[1][0], length: fence[1].length }; - continue; - } - // CommonMark: the closer uses the same character, is at least as long - // as the opener, and carries nothing else on the line. - if ( - fence[1][0] === inFence.char && - fence[1].length >= inFence.length && - line.trim() === fence[1] - ) { - inFence = null; - } - continue; - } - if (inFence !== null) continue; - const item = MD_LIST_ITEM.exec(line); - if (item) { - block.push({ indent: item[1].replace(/\t/g, ' ').length, key: normalize(item[2]) }); - continue; - } - if (line.trim() === '') continue; - closeBlock(); - if (line.includes('|') && i + 1 < lines.length && MD_TABLE_DELIMITER.test(lines[i + 1])) { - let j = i + 2; - while (j < lines.length && lines[j].trim() !== '' && lines[j].includes('|')) j++; - const rows = j - (i + 2); - if (rows > tableRows) tableRows = rows; - i = j - 1; + const visit = (node: Nodes): void => { + if (node.type === 'list' && node.children.length > listItems) { + listItems = node.children.length; + uniqueListItems = new Set(node.children.map((item) => normalize(extractMarkdownText(item)))) + .size; + } else if (node.type === 'table') { + tableRows = Math.max(tableRows, node.children.length - 1); } - } - closeBlock(); + if ('children' in node) node.children.forEach(visit); + }; + visit(tree); return { listItems, uniqueListItems, tableRows }; } @@ -520,228 +485,34 @@ function walkNode(node: Node): string { return walkContent(el); } -/** ASCII punctuation that a backslash escapes (CommonMark §2.4). */ -const ESCAPABLE_PUNCTUATION = new Set('!"#$%&\'()*+,-./:;<=>?@[\\]^_`{|}~'); - -/** - * Find the start of the next backtick run of exactly `runLength` backticks - * at or after `from`, or -1 if none precedes a blank line. Per CommonMark - * §6.1 a code span closes only on a run of the same length, and inline - * content never crosses a paragraph boundary, so a stray backtick cannot - * pair with one hundreds of lines later and swallow the text between. - */ -function findClosingBacktickRun(text: string, from: number, runLength: number): number { - const n = text.length; - let i = from; - while (i < n) { - const ch = text[i]; - if (ch === '`') { - let j = i; - while (j < n && text[j] === '`') j++; - if (j - i === runLength) return i; - i = j; - continue; +function extractMarkdownText(node: Nodes): string { + switch (node.type) { + case 'text': + case 'code': + case 'inlineCode': + return node.value; + case 'image': + case 'imageReference': + return node.alt ?? ''; + case 'html': { + const tag = /^<([a-z][a-z0-9-]*)(?:\s[^<>]*)?>$/i.exec(node.value)?.[1]; + if (tag && !HTML_TAG_NAMES.has(tag.toLowerCase())) return node.value; + return walkContent(parse(node.value)); } - if (ch === '\n') { - let j = i + 1; - while (j < n && (text[j] === ' ' || text[j] === '\t')) j++; - if (j >= n || text[j] === '\n') return -1; + case 'break': + return '\n'; + case 'definition': + return ''; + default: { + if (!('children' in node)) return ''; + const separator = ['root', 'blockquote', 'list', 'listItem', 'table', 'tableRow'].includes( + node.type, + ) + ? '\n' + : ''; + return node.children.map(extractMarkdownText).join(separator); } - i++; } - return -1; -} - -/** - * Single left-to-right pass that replaces inline code spans with - * \x00CODE{n}\x00 placeholders and backslash-escaped punctuation with - * \x00ESC{n}\x00 placeholders, following CommonMark's inline rules: - * - * - A backtick run of length N opens a code span closed by the next run of - * exactly N backticks (so `` `a` `` contains a literal backtick and a - * bare ``` in prose with no partner is just text). Content keeps - * backslashes verbatim, as does in HTML, and gets the spec's - * one-space trim when it both starts and ends with a space. - * - Outside code spans, a backslash followed by ASCII punctuation is that - * punctuation as literal text: snake\_case renders as snake_case, - * string\[] as string[], \*not bold\* keeps its asterisks (issue #110). - * The escaped character is stashed so the heading/list/link/emphasis - * regexes never see it as syntax, and restored after they run. - * - * Fenced blocks must already be placeholder-protected; their content is - * opaque here. - */ -function protectCodeSpansAndEscapes(text: string, codeSpans: string[], escapes: string[]): string { - const n = text.length; - const out: string[] = []; - let i = 0; - let plainStart = 0; - - while (i < n) { - const ch = text[i]; - - if (ch === '\\' && i + 1 < n && ESCAPABLE_PUNCTUATION.has(text[i + 1])) { - out.push(text.slice(plainStart, i)); - const idx = escapes.length; - escapes.push(text[i + 1]); - out.push(`\x00ESC${idx}\x00`); - i += 2; - plainStart = i; - continue; - } - - if (ch === '`') { - let j = i; - while (j < n && text[j] === '`') j++; - const runLength = j - i; - const close = findClosingBacktickRun(text, j, runLength); - if (close === -1) { - // Unmatched run: literal text. Skip past it so its backticks are - // not re-examined as potential openers. - i = j; - continue; - } - out.push(text.slice(plainStart, i)); - let content = text.slice(j, close); - if (content.startsWith(' ') && content.endsWith(' ') && content.trim().length > 0) { - content = content.slice(1, -1); - } - const idx = codeSpans.length; - codeSpans.push(content); - out.push(`\x00CODE${idx}\x00`); - i = close + runLength; - plainStart = i; - continue; - } - - i++; - } - - out.push(text.slice(plainStart)); - return out.join(''); -} - -/** - * Extract plain text from markdown by stripping all formatting. - * - * Code content (both fenced blocks and inline spans) is protected from - * stripping via placeholders. Without this, content like `# Heading` or - * `[link](url)` inside code blocks/spans would have its markdown syntax - * stripped (headings, links, blockquotes, emphasis), while the HTML side - * preserves the literal text inside
 and  tags. The
- * placeholder approach hides code content from the stripping regexes,
- * then restores it after all stripping is done.
- *
- * Heading lines are also placeholder-protected: a heading like
- * "### 1. How well..." has the "1. " stripped by the numbered-list regex
- * if processed normally, even though that "1. " is part of the heading
- * text on the HTML side. Protecting heading content keeps the bullet/
- * numbered-list passes from touching it.
- *
- * Backslash-escaped punctuation is likewise placeholder-protected and then
- * restored as the bare character, so "snake\_case" in markdown matches
- * "snake_case" in HTML and "\*literal\*" is not stripped as emphasis.
- */
-function extractMarkdownText(markdown: string): string {
-  let text = markdown;
-
-  // Step 1: Protect fenced code block content from subsequent stripping.
-  // Replace entire fenced blocks (``` ... ```) with placeholders so
-  // heading/link/emphasis/blockquote regexes don't modify literal content
-  // that the HTML side preserves as-is inside 
 tags.
-  //
-  // Per CommonMark §4.5, a fence opens with N>=3 backticks and closes only
-  // on a run of >=N. Capture the opener so the close-side backreference
-  // matches; otherwise nested example fences (4-backtick outer, 3-backtick
-  // inner) get mis-paired and inner markers leak out as text.
-  //
-  // Fences inside list items are indented by the list marker width (e.g.
-  // Turndown indents them 4 spaces under "1.  item"), so the opener may not
-  // sit at column 0. Capture the opener's indentation and require the closer
-  // at the same indent plus the 0-3 spaces of slack CommonMark allows, so a
-  // more deeply indented literal ``` inside the block can't close it early.
-  const codeBlocks: string[] = [];
-  text = text.replace(
-    /^( *)(`{3,})[^`\n]*\n([\s\S]*?)^\1 {0,3}\2`*\s*$/gm,
-    (_match, _indent, _opener, content) => {
-      const idx = codeBlocks.length;
-      codeBlocks.push(content);
-      return `\x00BLOCK${idx}\x00`;
-    },
-  );
-
-  // Step 2: Protect inline code spans and backslash escapes from
-  // subsequent stripping. Both are placeholder-protected in one
-  // left-to-right pass (see protectCodeSpansAndEscapes) because CommonMark
-  // resolves them positionally: a backslash before a backtick consumes it
-  // (\`literal\` is prose, not a code span), while a backslash inside an
-  // open code span is literal content (`C:\Users\` is one span). Neither
-  // ordering of two independent regex passes gets both cases right.
-  const codeSpans: string[] = [];
-  const escapes: string[] = [];
-  text = protectCodeSpansAndEscapes(text, codeSpans, escapes);
-
-  // Step 3: Protect heading lines from list-marker stripping. Headings
-  // like "### 1. How well are X supported?" survive into the HTML as
-  // "

1. How well are X supported?

", so the leading "1. " is - // part of the heading text — not a list marker. Without this, the - // numbered-list regex would strip it and the markdown side wouldn't - // contain the HTML segment. - const headings: string[] = []; - text = text.replace(/^#{1,6}\s+(.*)$/gm, (_match, content) => { - const idx = headings.length; - headings.push(content); - return `\x00HEAD${idx}\x00`; - }); - - // Step 4: Strip list markers and setext underlines while heading lines - // are still placeholder-protected. These are the passes that would - // misinterpret heading text — e.g., the numbered-list regex stripping - // "1. " from "### 1. How well..." (issue #91). - text = text - // Remove setext-style heading underlines - .replace(/^[=-]+$/gm, '') - // Remove reference-style link definitions - .replace(/^\[.*?\]:\s+.*$/gm, '') - // Remove list bullets/numbers (before emphasis, so leading * isn't - // misinterpreted as an emphasis marker) - .replace(/^[\s]*[-*+]\s+/gm, '') - .replace(/^[\s]*\d+\.\s+/gm, ''); - - // Step 5: Restore heading text. From here on, heading content is - // processed like any other body text — emphasis, links, etc. inside - // heading text gets the same treatment so it matches the HTML side - // (where

Foo

renders as "Foo"). - // eslint-disable-next-line no-control-regex - text = text.replace(/\x00HEAD(\d+)\x00/g, (_match, idxStr) => headings[parseInt(idxStr, 10)]); - - // Step 6: Strip remaining markdown formatting on body and heading text. - text = text - // Remove link/image URLs, keep text: [text](url) → text - .replace(/!?\[([^\]]*)\]\([^)]*\)/g, '$1') - // Remove emphasis markers. * emphasis is stripped unconditionally. - // _ emphasis is stripped only at word boundaries (per CommonMark, - // _text_ is emphasis only when _ is not adjacent to an alphanumeric). - // This preserves code identifiers like mongoc_client_get_database - // that appear as plain text (not inside backticks). - .replace(/(\*{1,3})(.*?)\1/g, '$2') - .replace(/(?\s?/gm, '') - // Remove horizontal rules - .replace(/^[-*_]{3,}$/gm, ''); - - // Step 7: Restore escaped characters as their literal selves, then code - // content (without backticks/fence markers). Escapes are restored after - // formatting is stripped so a restored * or _ is never re-read as emphasis. - // eslint-disable-next-line no-control-regex - text = text.replace(/\x00ESC(\d+)\x00/g, (_match, idxStr) => escapes[parseInt(idxStr, 10)]); - // eslint-disable-next-line no-control-regex - text = text.replace(/\x00CODE(\d+)\x00/g, (_match, idxStr) => codeSpans[parseInt(idxStr, 10)]); - // eslint-disable-next-line no-control-regex - text = text.replace(/\x00BLOCK(\d+)\x00/g, (_match, idxStr) => codeBlocks[parseInt(idxStr, 10)]); - - return text; } /** @@ -819,7 +590,7 @@ function toSegments(text: string): string[] { */ function computeParity( htmlText: string, - markdownText: string, + markdownTree: Root, warnThreshold: number, failThreshold: number, ): Omit { @@ -858,7 +629,7 @@ function computeParity( }; } - const normalizedMd = normalize(extractMarkdownText(markdownText)); + const normalizedMd = normalize(extractMarkdownText(markdownTree)); const sampleDiffs: string[] = []; let missingCount = 0; @@ -991,10 +762,11 @@ async function check(ctx: CheckContext): Promise { items, } = extractHtmlText(page.body, parityExclusions); totalSegmentationStripped += segmentationStripped; - const parity = computeParity(htmlText, markdownContent, warnThreshold, failThreshold); + const markdownTree = parseMarkdown(markdownContent, { normalizeVendorFences: false }); + const parity = computeParity(htmlText, markdownTree, warnThreshold, failThreshold); const itemCounts = compareItemCounts( items, - countMarkdownItems(markdownContent), + countMarkdownItems(markdownTree), paginatedPages.has(url), ); diff --git a/src/helpers/parse-markdown.ts b/src/helpers/parse-markdown.ts index 79e525f..be94976 100644 --- a/src/helpers/parse-markdown.ts +++ b/src/helpers/parse-markdown.ts @@ -3,13 +3,17 @@ import { gfmTableFromMarkdown } from 'mdast-util-gfm-table'; import { gfmTable } from 'micromark-extension-gfm-table'; import type { Nodes, Root } from 'mdast'; -export function parseMarkdown(content: string): Root { +export function parseMarkdown( + content: string, + { normalizeVendorFences = true }: { normalizeVendorFences?: boolean } = {}, +): Root { let source = content; while (true) { const tree = fromMarkdown(source, { extensions: [gfmTable()], mdastExtensions: [gfmTableFromMarkdown()], }); + if (!normalizeVendorFences) return tree; const vendorFences: Array<{ offset: number; length: number }> = []; const visit = (node: Nodes): void => { const offset = node.position?.start.offset; diff --git a/test/unit/checks/markdown-content-parity.test.ts b/test/unit/checks/markdown-content-parity.test.ts index 13b2f11..a6b6628 100644 --- a/test/unit/checks/markdown-content-parity.test.ts +++ b/test/unit/checks/markdown-content-parity.test.ts @@ -41,6 +41,204 @@ describe('markdown-content-parity', () => { return ctx; } + describe('CommonMark extraction', () => { + const codeLines = [ + '# This heading marker is literal example code', + '- This bullet marker is literal example code', + '1. This numbered item is literal example code', + '> This blockquote marker is literal example code', + '[Example link label](https://example.com/target)', + '**These emphasis markers remain literal code**', + 'const path = "C:\\Users\\example\\documents";', + 'const placeholder = "";', + 'const entities = "& # ©";', + '', + '| Example table cell | Another table cell |', + 'const identifier = mongoc_client_get_database;', + ]; + const htmlCode = codeLines + .join('\n') + .replaceAll('&', '&') + .replaceAll('<', '<') + .replaceAll('>', '>'); + + it.each([ + ['backtick fence', '```text\n' + codeLines.join('\n') + '\n```'], + ['tilde fence', '~~~text\n' + codeLines.join('\n') + '\n~~~'], + [ + 'backtick fence with pipe metadata', + '```text title="A|B"\n' + codeLines.join('\n') + '\n```', + ], + ['tilde fence with pipe metadata', '~~~text title="A|B"\n' + codeLines.join('\n') + '\n~~~'], + ['indented code', codeLines.map((line) => ' ' + line).join('\n')], + ['blockquote fence', ['~~~text', ...codeLines, '~~~'].map((line) => '> ' + line).join('\n')], + [ + 'list fence', + '- Example:\n\n' + ['~~~text', ...codeLines, '~~~'].map((line) => ' ' + line).join('\n'), + ], + ['unclosed fence', '~~~text\n' + codeLines.join('\n')], + ])('preserves literal code in %s with LF and CRLF', async (_name, source) => { + const url = 'http://mcp-commonmark-code.local/docs/page'; + const html = `
${htmlCode}
`; + let requests = 0; + server.use( + http.get(url, () => { + requests++; + return new HttpResponse(html, { headers: { 'Content-Type': 'text/html' } }); + }), + ); + + for (const newline of ['\n', '\r\n']) { + const markdown = source.replaceAll('\n', newline); + const ctx = makeCtx([{ url, markdown, htmlBody: html }], 'mcp-commonmark-code.local'); + const result = await check.run(ctx); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ + status: 'pass', + totalSegments: codeLines.length, + missingSegments: 0, + sampleDiffs: [], + }), + ]); + await check.run(ctx); + } + expect(requests).toBe(2); + }); + + const prose = [ + [ + '

2. Configure the deployment client

', + '## 2. **Configure** the deployment client ##', + ], + [ + '

Understanding configuration options

', + 'Understanding _configuration_ options\n---', + ], + [ + '

Install the client before configuring authentication.

', + '[Install **the client**][setup] before configuring authentication.', + ], + [ + '

Follow the nested link to learn about configuration.

', + 'Follow [the nested link](https://example.com/guide_(advanced)) to learn about configuration.', + ], + [ + '

The shortcut and collapsed references describe configuration.

', + 'The [shortcut] and [collapsed][] references describe configuration.', + ], + [ + '

A nested quotation provides additional deployment context.

', + '> > A nested quotation provides additional deployment context.', + ], + [ + '
  • Nested list entries describe configuration in detail.
', + '- Parent\n 1) Nested list entries describe configuration in detail.', + ], + [ + '

The snake_case and *literal* options support [brackets].

', + 'The snake\\_case and \\*literal\\* options support \\[brackets\\].', + ], + [ + '

Use A & B and the # symbol in configuration values.

', + 'Use A & B and the # symbol in configuration values.', + ], + [ + '

The `literal` C:\\Users\\ value remains unchanged.

', + 'The `` `literal` C:\\Users\\ `` value remains unchanged.', + ], + [ + '

Diagram: Deployment nodes connected by secure channels.

', + 'Diagram: ![Deployment nodes connected by secure channels.](diagram.png)', + ], + [ + '

Overview: Service components connected through a gateway.

', + 'Overview: ![Service components connected through a gateway.][diagram]', + ], + [ + '

The transaction commits without additional configuration.

', + 'The transaction commits without additional configuration.', + ], + [ + '

Embedded HTML preserves rendered text and & entities.

', + '

Embedded HTML preserves rendered text and & entities.

', + ], + [ + '

Inline comments do not interrupt transaction processing.

', + 'Inline comments do not interrupt transaction processing.', + ], + [ + '

Hard breaks preserve the first explanatory sentence.
Another sentence follows on a new line.

', + 'Hard breaks preserve the first explanatory sentence.\\\nAnother sentence follows on a new line.', + ], + ]; + + it.each([false, true])( + 'matches rendered prose with wrapping=%s and LF/CRLF', + async (wrapped) => { + const html = `
${prose.map(([html]) => html).join('\n')}
`; + const source = prose.map(([, markdown]) => markdown).join('\n\n'); + const definitions = + '\n\n[setup]: /setup\n[shortcut]: /shortcut\n[collapsed]: /collapsed\n[diagram]: diagram.png'; + const url = 'http://mcp-commonmark-prose.local/docs/page'; + server.use( + http.get(url, () => new HttpResponse(html, { headers: { 'Content-Type': 'text/html' } })), + ); + + for (const newline of ['\n', '\r\n']) { + const markdown = ( + (wrapped + ? source + .replaceAll('configuration', 'configuration') + .replaceAll(' before ', '\nbefore ') + : source) + definitions + ).replaceAll('\n', newline); + const result = await check.run( + makeCtx([{ url, markdown, htmlBody: html }], 'mcp-commonmark-prose.local'), + ); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ status: 'pass', missingSegments: 0, totalSegments: 17 }), + ]); + } + }, + ); + + it.each(['comments', 'definitions', 'link metadata'])( + 'does not count substantive text present only in %s', + async (location) => { + const lines = Array.from( + { length: 10 }, + (_, index) => + `Required deployment instruction number ${index} explains a distinct configuration step.`, + ); + const html = `
${lines.map((line) => `

${line}

`).join('')}
`; + const hidden = lines + .slice(5) + .map((line, index) => { + if (location === 'comments') return `> `; + if (location === 'definitions') return `[unused-${index}]: /target\n "${line}"`; + return `[short label]( "${line}")`; + }) + .join('\n\n'); + const markdown = lines.slice(0, 5).join('\n\n') + '\n\n' + hidden; + const url = 'http://mcp-commonmark-hidden.local/docs/page'; + server.use( + http.get(url, () => new HttpResponse(html, { headers: { 'Content-Type': 'text/html' } })), + ); + const result = await check.run( + makeCtx([{ url, markdown, htmlBody: html }], 'mcp-commonmark-hidden.local'), + ); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ + status: 'fail', + missingSegments: 5, + missingPercent: 50, + totalSegments: 10, + }), + ]); + }, + ); + }); + it('passes when markdown and HTML have equivalent content', async () => { const html = `

Getting Started

@@ -3351,6 +3549,105 @@ See the API reference for the full list of built-in plugins and options.`; expect(page.itemCounts).toMatchObject({ html: 24, markdown: 24, diverges: false }); }); + it.each([false, true])( + 'counts and deduplicates complete multiline items inside blockquote=%s', + async (quoted) => { + const descriptions = Array.from( + { length: 24 }, + (_, index) => `model-${index}: a model that does useful things number ${index}`, + ); + const html = `
    ${descriptions.map((text) => `
  • Model configuration details

    ${text}

  • `).join('')}
`; + const first = descriptions.map((text) => `- **Model configuration details**\n\n ${text}`); + const repeated = descriptions.map( + (text) => + `- Model configuration details\n\n ${text.replace(': a model', ': a\n model')}`, + ); + const source = [...first, ...repeated].join('\n\n'); + const markdown = quoted + ? source + .split('\n') + .map((line) => '> ' + line) + .join('\n') + : source; + const result = await check.run(catalogCtx(html, markdown, 'cat-multiline.local')); + expect(result.status).toBe('pass'); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ + itemCounts: { + structure: 'list', + html: 24, + markdown: 48, + markdownUnique: 24, + duplicates: 24, + diverges: false, + }, + }), + ]); + }, + ); + + it('compares the largest nested list by its own direct children', async () => { + const html = catalogHtml(30) + .replace('
    ', '
    • Catalog
        ') + .replace('
      ', '
  • Appendix
'); + const markdown = + '- Catalog\n' + + catalogMd(30) + .split('\n') + .filter((line) => line.startsWith('- ')) + .map((line) => ' ' + line) + .join('\n') + + '\n- Appendix'; + const result = await check.run(catalogCtx(html, markdown, 'cat-largest-nested.local')); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ + itemCounts: expect.objectContaining({ + html: 30, + markdown: 30, + markdownUnique: 30, + diverges: false, + }), + }), + ]); + }); + + it.each(['indented', 'quoted fence', 'tilde fence'])( + 'ignores list and table examples in %s code', + async (kind) => { + const example = + catalogMd(40) + + '\n| Column | Value |\n| --- | --- |\n' + + Array.from({ length: 40 }, (_, index) => `| ${index} | value |`).join('\n'); + const code = + kind === 'indented' + ? example + .split('\n') + .map((line) => ' ' + line) + .join('\n') + : kind === 'quoted fence' + ? ['~~~markdown', ...example.split('\n'), '~~~'].map((line) => '> ' + line).join('\n') + : '~~~markdown\n' + example + '\n~~~'; + const result = await check.run( + catalogCtx( + catalogHtml(24), + catalogMd(24) + '\n## Code examples\n\n' + code, + 'cat-code-examples.local', + ), + ); + expect(result.status).toBe('pass'); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ + itemCounts: expect.objectContaining({ + html: 24, + markdown: 24, + markdownUnique: 24, + diverges: false, + }), + }), + ]); + }, + ); + it('does not read list lines inside a longer fence closed by a shorter marker', async () => { const inner = Array.from({ length: 40 }, (_, i) => `- fenced item ${i}`).join('\n'); const md = @@ -3400,6 +3697,41 @@ See the API reference for the full list of built-in plugins and options.`; }); }); + it('counts GFM rows with escaped pipes, code cells, and missing cells across LF/CRLF', async () => { + const labels = Array.from( + { length: 24 }, + (_, index) => `Compatibility entry ${index} for hosted deployments`, + ); + const html = `
${labels.map((label, index) => ``).join('')}
EntryValue
${label}${index % 2 ? '' : 'alpha|beta or a literal | pipe'}
`; + const source = + 'Entry | Value\n--- | ---\n' + + labels + .map((label, index) => + index % 2 ? label : label + ' | `alpha\\|beta` or a literal \\| pipe', + ) + .join('\n') + + '\n\n## Smaller table\n\nName | Value\n--- | ---\none | small\ntwo | small\n'; + for (const newline of ['\n', '\r\n']) { + const result = await check.run( + catalogCtx(html, source.replaceAll('\n', newline), 'cat-gfm.local'), + ); + expect(result.status).toBe('pass'); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ + missingSegments: 0, + itemCounts: { + structure: 'table', + html: 24, + markdown: 24, + markdownUnique: 24, + duplicates: 0, + diverges: false, + }, + }), + ]); + } + }); + it('counts table data rows, not header rows, and picks the larger structure', async () => { const rows = Array.from( { length: 30 }, diff --git a/working-notes/parity-check-notes.md b/working-notes/parity-check-notes.md index 14993be..f704647 100644 --- a/working-notes/parity-check-notes.md +++ b/working-notes/parity-check-notes.md @@ -14,8 +14,8 @@ version for comparison. ## Current state of the code - Implementation: `src/checks/observability/markdown-content-parity.ts` -- Tests: `test/unit/checks/markdown-content-parity.test.ts` (41 tests, all passing) -- Full suite: 1216 tests passing, lint clean +- Tests: `test/unit/checks/markdown-content-parity.test.ts` +- Markdown text and repeated structures use the shared CommonMark/GFM parser (issue #152). - The `diff` dependency has been removed; comparison now uses containment checking - No unused imports or cleanup needed @@ -128,23 +128,53 @@ The known tag set covers all standard HTML elements. The regex and checks it against the set. Tags with attributes (like ``) have their first word extracted as the tag name. -### Markdown text extraction ordering - -The `extractMarkdownText()` function strips markdown formatting in a specific -order: - -1. Code fences (keep content) -2. Heading markers -3. Setext heading underlines -4. Link/image URLs (keep text) -5. Reference-style link definitions -6. List bullets/numbers (before emphasis, so leading `*` isn't misinterpreted) -7. **Inline code backticks** (before emphasis, so underscores in code - identifiers like `mongoc_client_get_database` aren't mangled) -8. Emphasis markers (`*` only, not `_` — underscores are too common in code - identifiers and cause false mismatches when stripped as emphasis) -9. Blockquote markers -10. Horizontal rules +### Markdown extraction with CommonMark (issue #152) + +Parse each compared page once with the shared `parseMarkdown()` helper and +reuse its tree for text and repeated-item extraction. Parity opts out of the +helper's vendor-table fence normalization: a pipe in valid fence metadata +must not cause literal code to become prose. Discovery and pagination retain +their existing default normalization. Link-discovery projections are not +parity inputs because they intentionally exclude code. + +- Text, fenced/indented code, and inline code use their parsed values. Code + keeps literal Markdown syntax, backslashes, entities, and placeholders. + CommonMark handles delimiter lengths, containers, LF/CRLF, and unclosed + fences. The separate fence-validity check remains unchanged. +- Headings, emphasis, links (including references), lists, and blockquotes + contribute their rendered text. Block and table-cell boundaries separate + text; inline formatting does not split words. Escapes and character + references are decoded by the parser, not another stripping pass. +- Markdown images retain alt text, extending the existing inline-image + behavior to reference images. Destinations, titles, definitions, and + non-rendered comments cannot satisfy missing-content comparisons. +- Embedded HTML uses the existing DOM text walker, including block breaks + and its non-content tag exclusions. Standalone unknown opening tags remain + literal placeholders (for example ``), preserving the historical + field regression. This is not a general MDX renderer; HTML-side container, + chrome, audience, and custom-selector handling is unchanged. +- Each list is counted by direct children, and the largest list wins (first + in document order on ties), including nested lists considered independently. + Nested bullets are not added to the parent's count. Deduplication uses + complete normalized rendered item text, including continuation paragraphs + and nested content, rather than the old first source line. Items that differ + only in formatting, wrapping, or link destinations have the same text key. +- The largest GFM table contributes its data rows only. Header/delimiter + rows are excluded; missing cells and escaped pipes follow GFM parsing. + Lists/tables inside code are not structures. Raw HTML lists/tables in + Markdown remain outside this Markdown-syntax counter. + +Characterization tests reproduced loss of literal code in tilde, indented, +nested, and unclosed fences before replacing text extraction. Catalog tests +reproduced fragmented multiline lists, missing blockquoted/nested lists, +indented examples counted as items, and truncated tables. Historical field +regressions remain covered. Request assertions verify one HTML fetch per +uncached page and cache reuse; no discovery or sampling changes were made. + +The ordered placeholder/regex passes and line-based list/table parser were +removed. Normalization, scoring, thresholds, public result shapes, and +informational-only item-count diagnostics are unchanged. The session notes +below describe earlier implementations and why their regressions matter. ### Thresholds From 84c0a0e4e279553995181c1cbb27016cb71c7e6f Mon Sep 17 00:00:00 2001 From: dacharyc Date: Mon, 28 Sep 2026 09:57:51 -0400 Subject: [PATCH 2/2] fix: preserve embedded HTML context in Markdown parity Extract Markdown text and embedded markup through a shared HTML fragment so paired SVG, MathML, and custom elements are not mistaken for literal placeholders. Escape Markdown text to preserve code and retain standalone unknown placeholders using parsed source ranges. Add 10 regressions covering inline markup, visible text, standalone placeholders, and non-rendered SVG titles. Document the extraction semantics. Addresses review feedback on #155. --- .../observability/markdown-content-parity.ts | 32 +++++++---- .../checks/markdown-content-parity.test.ts | 54 ++++++++++++++++++- working-notes/parity-check-notes.md | 14 +++-- 3 files changed, 86 insertions(+), 14 deletions(-) diff --git a/src/checks/observability/markdown-content-parity.ts b/src/checks/observability/markdown-content-parity.ts index 6781dfa..6cd734c 100644 --- a/src/checks/observability/markdown-content-parity.ts +++ b/src/checks/observability/markdown-content-parity.ts @@ -485,20 +485,21 @@ function walkNode(node: Node): string { return walkContent(el); } -function extractMarkdownText(node: Nodes): string { +function escapeHtmlText(text: string): string { + return text.replaceAll('&', '&').replaceAll('<', '<').replaceAll('>', '>'); +} + +function markdownTextFragment(node: Nodes): string { switch (node.type) { case 'text': case 'code': case 'inlineCode': - return node.value; + return escapeHtmlText(node.value); case 'image': case 'imageReference': - return node.alt ?? ''; - case 'html': { - const tag = /^<([a-z][a-z0-9-]*)(?:\s[^<>]*)?>$/i.exec(node.value)?.[1]; - if (tag && !HTML_TAG_NAMES.has(tag.toLowerCase())) return node.value; - return walkContent(parse(node.value)); - } + return escapeHtmlText(node.alt ?? ''); + case 'html': + return node.value; case 'break': return '\n'; case 'definition': @@ -510,9 +511,22 @@ function extractMarkdownText(node: Nodes): string { ) ? '\n' : ''; - return node.children.map(extractMarkdownText).join(separator); + return node.children.map(markdownTextFragment).join(separator); + } + } +} + +function extractMarkdownText(node: Nodes): string { + const fragment = markdownTextFragment(node); + const root = parse(fragment, { parseNoneClosedTags: true, lowerCaseTagName: true }); + for (const element of root.querySelectorAll('*')) { + if (HTML_TAG_NAMES.has(element.rawTagName)) continue; + const source = fragment.slice(...element.range); + if (/^<[a-z][a-z0-9-]*(?:\s[^<>]*)?>$/i.test(source) && !source.endsWith('/>')) { + element.replaceWith(escapeHtmlText(source), ...element.childNodes); } } + return walkContent(root); } /** diff --git a/test/unit/checks/markdown-content-parity.test.ts b/test/unit/checks/markdown-content-parity.test.ts index a6b6628..25ce02e 100644 --- a/test/unit/checks/markdown-content-parity.test.ts +++ b/test/unit/checks/markdown-content-parity.test.ts @@ -105,6 +105,57 @@ describe('markdown-content-parity', () => { expect(requests).toBe(2); }); + it.each([ + ['inline SVG', ''], + ['self-closing SVG', ''], + ['SVG with non-content text', 'Next actionIcon label'], + ['paired custom element', ''], + ['paired uppercase custom element', ''], + ['standalone self-closing element', ''], + ['MathML with visible text', 'x+1'], + ['nested custom elements', 'visible text'], + ])('does not leak %s markup into parity text', async (_name, markup) => { + const lines = Array.from( + { length: 10 }, + (_, index) => `Click ${markup} now to complete deployment instruction number ${index}.`, + ); + const html = `
${lines.map((line) => `

${line}

`).join('')}
`; + const url = 'http://mcp-inline-markup.local/docs/page'; + server.use( + http.get(url, () => new HttpResponse(html, { headers: { 'Content-Type': 'text/html' } })), + ); + for (const newline of ['\n', '\r\n']) { + const result = await check.run( + makeCtx( + [{ url, markdown: lines.join(newline + newline), htmlBody: html }], + 'mcp-inline-markup.local', + ), + ); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ status: 'pass', totalSegments: 10, missingSegments: 0 }), + ]); + } + }); + + it('preserves Markdown formatting and standalone placeholders beside paired markup', async () => { + const lines = Array.from({ length: 10 }, (_, index) => ({ + html: `Choose <REGION> or <field name> for deployment ${index}, then confirm now.`, + markdown: `Choose or for **deployment ${index}**, then confirm now.`, + })); + const html = `
${lines.map((line) => `

${line.html}

`).join('')}
`; + const markdown = lines.map((line) => line.markdown).join('\n\n'); + const url = 'http://mcp-markup-placeholders.local/docs/page'; + server.use( + http.get(url, () => new HttpResponse(html, { headers: { 'Content-Type': 'text/html' } })), + ); + const result = await check.run( + makeCtx([{ url, markdown, htmlBody: html }], 'mcp-markup-placeholders.local'), + ); + expect(result.details?.pageResults).toEqual([ + expect.objectContaining({ status: 'pass', totalSegments: 10, missingSegments: 0 }), + ]); + }); + const prose = [ [ '

2. Configure the deployment client

', @@ -202,7 +253,7 @@ describe('markdown-content-parity', () => { }, ); - it.each(['comments', 'definitions', 'link metadata'])( + it.each(['comments', 'definitions', 'link metadata', 'SVG titles'])( 'does not count substantive text present only in %s', async (location) => { const lines = Array.from( @@ -216,6 +267,7 @@ describe('markdown-content-parity', () => { .map((line, index) => { if (location === 'comments') return `> `; if (location === 'definitions') return `[unused-${index}]: /target\n "${line}"`; + if (location === 'SVG titles') return `Icon: ${line}`; return `[short label]( "${line}")`; }) .join('\n\n'); diff --git a/working-notes/parity-check-notes.md b/working-notes/parity-check-notes.md index f704647..606f389 100644 --- a/working-notes/parity-check-notes.md +++ b/working-notes/parity-check-notes.md @@ -149,10 +149,16 @@ parity inputs because they intentionally exclude code. behavior to reference images. Destinations, titles, definitions, and non-rendered comments cannot satisfy missing-content comparisons. - Embedded HTML uses the existing DOM text walker, including block breaks - and its non-content tag exclusions. Standalone unknown opening tags remain - literal placeholders (for example ``), preserving the historical - field regression. This is not a general MDX renderer; HTML-side container, - chrome, audience, and custom-selector handling is unchanged. + and its non-content tag exclusions. Markdown text is HTML-escaped and raw + HTML nodes are kept together in one fragment, preserving ancestor context + across inline tags. Paired SVG, MathML, and custom elements are markup, not + placeholders; visible text is retained unless an ancestor is excluded by + the existing walker (such as SVG). Only unclosed, non-self-closing unknown + tags remain literal placeholders (for example ``), identified by + their parsed source ranges. This preserves the historical field regression + without leaking inline SVG children into prose (PR #155 review). This is + not a general MDX renderer; HTML-side container, chrome, audience, and + custom-selector handling is unchanged. - Each list is counted by direct children, and the largest list wins (first in document order on ties), including nested lists considered independently. Nested bullets are not added to the parent's count. Deduplication uses