From e463ae1ccedef0217a6d05466590d4b99f7d38fe Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 31 Aug 2026 23:57:28 +0000 Subject: [PATCH] fix(scripts): collect TypeScript fences opened inside a blockquote MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `scanFences` anchored a fence opener on leading spaces and tabs only, so a fence opened inside a Markdown blockquote carried a `> ` prefix the anchor never matched. The block was never collected and the gate compiled nothing for it, with no diagnostic: an uncollected block appears in no count and its page still reports as covered. The opener now tolerates a blockquote prefix and carries the opener's quote depth through the rest of the walk — the search for the closing fence reads candidates at that same depth, and body lines are stripped of that many markers before reaching the compiler. Depth 0 takes an identity path that returns the line unchanged byte for byte, so every unquoted fence in the corpus scans exactly as before. Measured over the gate's own 224-document population: 773 -> 774 collected blocks, nothing dropped. The one newly-visible block compiles. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_012wwHa4aaFybxXrfmfHioDM --- .../7086-blockquoted-fence-collector.md | 32 +++++++++ .../__tests__/check-doc-snippet-types.test.ts | 67 +++++++++++++++++++ scripts/check-doc-snippet-types.mjs | 41 +++++++++++- 3 files changed, 137 insertions(+), 3 deletions(-) create mode 100644 .changeset/7086-blockquoted-fence-collector.md diff --git a/.changeset/7086-blockquoted-fence-collector.md b/.changeset/7086-blockquoted-fence-collector.md new file mode 100644 index 0000000000..2676771491 --- /dev/null +++ b/.changeset/7086-blockquoted-fence-collector.md @@ -0,0 +1,32 @@ +--- +--- + +Doc-snippet gate tooling only, no published package source changed. + +`check-doc-snippet-types` collects blocks with `scanFences`, whose fence-opening +anchor accepted a run of leading spaces and tabs and nothing else. A fence +opened inside a Markdown blockquote carries a `> ` prefix, so the anchor never +matched, the block was never collected, and the gate compiled nothing for it. +There was no diagnostic: an uncollected block appears in no count, and its page +still reports as covered. A callout is a natural home for an import example, +which is exactly the snippet class that rots when an export is renamed — the one +class this gate exists to catch. + +The opener now tolerates a blockquote prefix and carries the opener's quote +DEPTH through the rest of the walk: the search for the closing fence reads +candidate lines at that same depth, and every body line is stripped of that many +markers before it reaches the compiler. Depth 0 — every unquoted fence in the +corpus — takes an identity path that returns the line unchanged byte for byte, +so the other 773 collected blocks scan exactly as they did before. Stripping +consumes at most one space after each marker, per CommonMark, so indentation +belonging to the snippet survives. + +Carrying the depth to the CLOSING fence is what makes this safe in both +directions. Without it a blockquoted fence would find no close and swallow the +rest of the file; and a plain fence would be closed early by any `> ` + backticks +line quoted inside it as prose. Both directions are pinned. + +Measured over the gate's own 224-document population: collected blocks 773 → 774. +The one newly-collected block is the import callout at +`content/docs/api/schema-reference.md` line 12, and it compiles — the gate's +semantic phase judges 272 of 272 with 0 failures. Nothing left the population. diff --git a/scripts/__tests__/check-doc-snippet-types.test.ts b/scripts/__tests__/check-doc-snippet-types.test.ts index 3a85b12411..1e5de57ae5 100644 --- a/scripts/__tests__/check-doc-snippet-types.test.ts +++ b/scripts/__tests__/check-doc-snippet-types.test.ts @@ -116,6 +116,73 @@ describe('fence scanning', () => { expect(blocks).toHaveLength(0); }); + it('collects a fence opened inside a blockquote and strips the quoting from its body', () => { + // objectui#7086: the opening anchor allowed leading spaces and tabs only, so a + // fence inside a callout was never collected and the gate compiled nothing for + // it. Silently — an uncollected block appears in no count, and the page still + // reports as covered. + const { blocks } = scanFences( + [ + '> **Import:** All types are available from `@object-ui/types`.', + '>', + `> ${FENCE}typescript`, + "> import type { PageNodeSchema } from '@object-ui/types';", + `> ${FENCE}`, + '', + 'Prose after the callout.', + ].join('\n'), + ); + expect(blocks).toHaveLength(1); + expect(blocks[0].language).toBe('typescript'); + expect(blocks[0].body).toBe("import type { PageNodeSchema } from '@object-ui/types';"); + }); + + it('closes a blockquoted fence at its own depth rather than running to end of file', () => { + const { blocks } = scanFences( + [ + `> ${FENCE}ts`, + '> export const a = 1;', + `> ${FENCE}`, + '', + 'export const notPartOfTheBlock = true;', + ].join('\n'), + ); + expect(blocks).toHaveLength(1); + expect(blocks[0].body).toBe('export const a = 1;'); + }); + + it('strips the opening depth only, so nesting and the snippet own indentation survive', () => { + const { blocks } = scanFences( + [ + `> > ${FENCE}ts`, + '> > export const nested = {', + '> > deep: true,', + '> > };', + `> > ${FENCE}`, + ].join('\n'), + ); + expect(blocks).toHaveLength(1); + expect(blocks[0].body).toBe(['export const nested = {', ' deep: true,', '};'].join('\n')); + }); + + it('does not let a quoted backtick line close an unquoted fence', () => { + // Depth 0 takes the identity path. This is what keeps every unquoted fence in + // the corpus collecting exactly as it did before blockquotes were recognised. + const { blocks } = scanFences( + [ + `${FENCE}ts`, + '// a quoted fence, as prose inside a snippet:', + `> ${FENCE}`, + 'export const a = 1;', + FENCE, + ].join('\n'), + ); + expect(blocks).toHaveLength(1); + expect(blocks[0].body).toBe( + ['// a quoted fence, as prose inside a snippet:', `> ${FENCE}`, 'export const a = 1;'].join('\n'), + ); + }); + it('attaches a fragment marker only to the fence directly beneath it', () => { const { blocks } = scanFences( [ diff --git a/scripts/check-doc-snippet-types.mjs b/scripts/check-doc-snippet-types.mjs index 1b3391b4c5..c1fef0705f 100644 --- a/scripts/check-doc-snippet-types.mjs +++ b/scripts/check-doc-snippet-types.mjs @@ -554,6 +554,31 @@ const UNGATED_DOCS = { // ── Fence scanning ─────────────────────────────────────────────────────────── +/** + * One level of blockquote marker: the `>` a Markdown blockquote puts in front of + * every line it contains — the fences and the code between them alike. At most one + * space after the marker is consumed, per CommonMark, so indentation that belongs + * to the snippet survives the strip. + */ +const QUOTE_MARKER = /^[ \t]*>[ \t]?/; + +/** + * `line` with `depth` levels of blockquote marker removed. `depth === 0` returns + * the line unchanged, byte for byte; that identity path is what keeps every + * unquoted fence in the corpus collecting exactly as it did before. A line that + * runs out of markers early is returned as far as it stripped, which bounds a + * lazily-continued blockquote instead of letting its fence run to end of file. + */ +function stripQuotePrefix(line, depth) { + let out = line; + for (let d = 0; d < depth; d++) { + const m = QUOTE_MARKER.exec(out); + if (!m) break; + out = out.slice(m[0].length); + } + return out; +} + /** The declaration a fragment carries; see FRAGMENT_MARKER_EXAMPLES. */ const FRAGMENT_MARKER = /^[ \t]*(?:\{\/\*|)[ \t]*$/; @@ -576,6 +601,12 @@ export const FRAGMENT_MARKER_EXAMPLES = [ * Every fenced block in one document, with the ts/tsx ones marked. Fences are * matched by their own run length so a ```` ```` ```` wrapper containing ``` does * not confuse the walk, and a block's opening info string is kept verbatim. + * + * A fence opened inside a blockquote is collected too. Its opener's quote depth + * is carried to the search for its closing fence and stripped from every body + * line, so the compiler sees the snippet the reader sees and not the `>` around + * it. Depth 0 — every unquoted fence — takes the identity path and scans exactly + * as it did before blockquotes were recognised. */ export function scanFences(source) { const lines = source.split('\n'); @@ -584,12 +615,13 @@ export function scanFences(source) { for (let i = 0; i < lines.length; i++) { const marker = FRAGMENT_MARKER.exec(lines[i]); if (marker) markers.push({ line: i + 1, reason: marker[1].trim(), consumed: false }); - const open = /^([ \t]*)(`{3,})(.*)$/.exec(lines[i]); + const open = /^([ \t]*(?:>[ \t]*)*)(`{3,})(.*)$/.exec(lines[i]); if (!open) continue; const ticks = open[2]; + const depth = (open[1].match(/>/g) ?? []).length; let close = lines.length; for (let j = i + 1; j < lines.length; j++) { - const c = /^[ \t]*(`{3,})[ \t]*$/.exec(lines[j]); + const c = /^[ \t]*(`{3,})[ \t]*$/.exec(stripQuotePrefix(lines[j], depth)); if (c && c[1].length >= ticks.length) { close = j; break; @@ -606,7 +638,10 @@ export function scanFences(source) { blocks.push({ fenceLine: i + 1, language, - body: lines.slice(i + 1, close).join('\n'), + body: lines + .slice(i + 1, close) + .map((line) => stripQuotePrefix(line, depth)) + .join('\n'), fragmentReason: above ? above.reason : null, }); }