refactor(#1396): T5 — migrate uat-predicate + uat onto markdown-sectionizer seam (#1397)

- uat-predicate.cts: replace local _stripFencedBlocks (and its private
  FenceState/StripFencedResult types) with a call to stripFencedCode from
  markdown-sectionizer.cjs (ADR-1372 T5). Both stripFalsePositiveContexts
  step (c) and analyzeMarkdown now route through the seam. The three other
  passes in stripFalsePositiveContexts — frontmatter strip, HTML-comment
  strip, blockquote-line filter — remain caller-side (seam does not do these).
  The unterminatedFence signal consumed by analyzeMarkdown is preserved; it
  is now returned by stripFencedCode (same machine, same contract).

- uat.cts: migrate the ## Current Test, ## Tests, and ## Human Verification
  section-collect patterns onto collectSection/tokenizeHeadings from the seam.
  UAT-specific item parsing (### N. Name blocks, expected/result fields,
  categorization logic) stays caller-side. The HTML-comment strip within the
  Current Test body remains caller-side (UAT document structure, not seam scope).

- tests/markdown-sectionizer.test.cjs: remove the 18-case tautological parity
  guard (DEFECT.GENERATIVE-FIX). Once uat-predicate imports the seam the guard
  compares the seam to itself — removing it is the T5 commitment per ADR-1372.

4-space-indent behavior change (CommonMark correctness improvement): the seam
uses /^( {0,3})/ (CommonMark §4.5 ≤3-space indent); the retired
_stripFencedBlocks used /^(\s*)/ (any indent). A 4-space-indented ``` is no
longer treated as a fence opener (it is an indented code block per CommonMark).
Head-to-head over 9 corpus inputs: 0 diffs on all standard cases; 2 diffs only
on the synthetic 4-space-indent edge cases. No UAT fixture or test in the suite
exercises 4-space-indented fences. The change is a correctness improvement.

Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Tom Boucher
2026-06-17 15:44:11 -04:00
committed by GitHub
parent 6e9f8bf50b
commit 308c7be481
3 changed files with 68 additions and 193 deletions

View File

@@ -16,6 +16,9 @@ import path from 'node:path';
// eslint-disable-next-line @typescript-eslint/no-require-imports
import frontmatter = require('./frontmatter.cjs');
const { extractFrontmatter } = frontmatter;
// eslint-disable-next-line @typescript-eslint/no-require-imports
import markdownSectionizer = require('./markdown-sectionizer.cjs');
const { stripFencedCode } = markdownSectionizer;
// ─── Types ────────────────────────────────────────────────────────────────────
@@ -82,8 +85,8 @@ function stripFalsePositiveContexts(content: string): string {
// Step (b): remove HTML comments anywhere; unterminated comment swallows to EOF
stripped = stripped.replace(/<!--[\s\S]*?(?:-->|$)/g, '');
// Step (c): remove fenced code blocks via CommonMark-style state machine (handles CRLF + indented fences)
stripped = _stripFencedBlocks(stripped).text;
// Step (c): remove fenced code blocks via the canonical seam (ADR-1372 T5)
stripped = stripFencedCode(stripped).text;
// Step (d): remove blockquote lines
stripped = stripped
@@ -94,61 +97,6 @@ function stripFalsePositiveContexts(content: string): string {
return stripped;
}
interface FenceState {
char: '`' | '~';
len: number;
}
interface StripFencedResult {
text: string;
unterminatedFence: boolean;
}
/**
* CommonMark-style fenced-code-block stripper.
* Tracks the opening delimiter char and length so that a ~~~ line inside a
* ``` fence is correctly treated as fence content, not a closing delimiter.
*
* Opening rule: first delimiter line with char+len sets openFence.
* Closing rule: delimiter line with SAME char, run length >= openFence.len,
* and NO trailing non-whitespace text closes the fence.
* All delimiter and content lines are dropped; non-fence lines are kept.
* Returns the kept text plus unterminatedFence:true if EOF inside a fence.
*/
function _stripFencedBlocks(content: string): StripFencedResult {
const lines = content.split('\n');
const kept: string[] = [];
let openFence: FenceState | null = null;
const delimRe = /^(\s*)(`{3,}|~{3,})(.*)$/;
for (const rawLine of lines) {
// Tolerate CRLF: strip trailing \r for matching, but we work on split-by-\n lines
// (the outer caller joined by \n already; we just handle a stray \r in the last char)
const line = rawLine.replace(/\r$/, '');
const m = delimRe.exec(line);
if (m) {
const char = m[2][0] as '`' | '~';
const len = m[2].length;
const trailing = m[3];
if (openFence === null) {
// Opening delimiter — drop this line and record the fence
openFence = { char, len };
} else if (char === openFence.char && len >= openFence.len && /^\s*$/.test(trailing)) {
// Closing delimiter (same char, sufficient length, no trailing text) — drop and close
openFence = null;
}
// else: mismatched delimiter inside fence (e.g. ~~~ inside ```) — drop as content
continue; // delimiter lines are always dropped
}
if (openFence === null) {
kept.push(rawLine);
}
// Lines inside fence are dropped
}
return { text: kept.join('\n'), unterminatedFence: openFence !== null };
}
/**
* Analyse raw markdown for structural anomalies (unterminated fence / comment).
* Exported for unit-testability and used by evaluateUatPassed for per-file malformed detection.
@@ -170,8 +118,8 @@ function analyzeMarkdown(raw: string): { unterminatedFence: boolean; unterminate
i = close + 3;
}
// Fence state machine gives the accurate unterminated-fence signal.
const { unterminatedFence } = _stripFencedBlocks(raw);
// Fence state machine gives the accurate unterminated-fence signal (seam, ADR-1372 T5).
const { unterminatedFence } = stripFencedCode(raw);
return { unterminatedFence, unterminatedComment };
}

View File

@@ -15,6 +15,9 @@ import path from 'node:path';
import io = require('./io.cjs');
const { output, error } = io;
// eslint-disable-next-line @typescript-eslint/no-require-imports
import markdownSectionizer = require('./markdown-sectionizer.cjs');
const { collectSection, tokenizeHeadings } = markdownSectionizer;
// eslint-disable-next-line @typescript-eslint/no-require-imports
import roadmapParser = require('./roadmap-parser.cjs');
const { getMilestonePhaseFilter } = roadmapParser;
// eslint-disable-next-line @typescript-eslint/no-require-imports
@@ -179,12 +182,21 @@ function cmdRenderCheckpoint(cwd: string, options: { file?: string } = {}, raw:
// ─── parseCurrentTest ─────────────────────────────────────────────────────────
function parseCurrentTest(content: string): CurrentTest {
const currentTestMatch = content.match(/##\s*Current Test\s*(?:\n<!--[\s\S]*?-->)?\n([\s\S]*?)(?=\n##\s|$)/i);
if (!currentTestMatch) {
// Use the seam to locate the ## Current Test section (ADR-1372 T5).
// HTML-comment stripping within the section body is UAT-specific, so we keep
// the comment removal caller-side after extracting the body.
const currentTestSection = collectSection(
content,
(h) => /^current\s+test$/i.test(h.text) && h.level === 2,
{ levelBounded: true },
);
if (!currentTestSection) {
error('UAT file is missing a Current Test section');
}
const section = currentTestMatch![1].trimEnd();
// Remove any leading HTML comment block (UAT-specific document structure)
const rawBody = currentTestSection!.body.replace(/^<!--[\s\S]*?-->\s*\n?/, '');
const section = rawBody.trimEnd();
if (!section.trim()) {
error('Current Test section is empty');
}
@@ -230,40 +242,53 @@ function parseCurrentTest(content: string): CurrentTest {
}
function parseFirstPendingTest(content: string): CurrentTest | null {
const testsMatch = content.match(/##\s*Tests\s*\n([\s\S]*?)(?=\n##\s|$)/i);
if (!testsMatch) {
// Use the seam to locate the ## Tests section (ADR-1372 T5).
const testsSection = collectSection(
content,
(h) => /^tests$/i.test(h.text) && h.level === 2,
{ levelBounded: true },
);
if (!testsSection) {
return null;
}
const testsSection = testsMatch[1];
const headingPattern = /^###\s*(\d+)\.\s*([^\n]+)\s*$/gm;
const headings: Array<{ index: number; number: number; name: string }> = [];
let headingMatch: RegExpExecArray | null;
while ((headingMatch = headingPattern.exec(testsSection)) !== null) {
headings.push({
index: headingMatch.index,
number: parseInt(headingMatch[1], 10),
name: headingMatch[2].trim(),
});
}
const sectionBody = testsSection.body;
// Within the Tests section body, find ### N. Name sub-headings.
// tokenizeHeadings operates on the section body as a standalone document,
// filtering to level-3 headings matching the UAT-specific "N. Name" pattern.
// The UAT-specific item parsing (number extraction, result parsing) stays caller-side.
const subHeadings = tokenizeHeadings(sectionBody).filter(
(h) => h.level === 3 && /^\d+\.\s+/.test(h.text),
);
for (let i = 0; i < subHeadings.length; i += 1) {
const current = subHeadings[i];
const next = subHeadings[i + 1];
// Slice the block for this sub-test from the section body text
const block = next
? sectionBody.slice(current.offset, next.offset)
: sectionBody.slice(current.offset);
for (let i = 0; i < headings.length; i += 1) {
const current = headings[i];
const next = headings[i + 1];
const block = testsSection.slice(current.index, next ? next.index : undefined);
if (!/^result:\s*\[?pending\]?\s*$/im.test(block)) {
continue;
}
// Extract the UAT-specific number and name from the heading text
const headingParts = current.text.match(/^(\d+)\.\s+(.+)$/);
if (!headingParts) continue;
const testNumber = parseInt(headingParts[1], 10);
const testName = headingParts[2].trim();
const expected = parseExpectedFromTestBlock(block);
if (!expected) {
error(`Pending UAT test ${current.number} is missing an expected field`);
error(`Pending UAT test ${testNumber} is missing an expected field`);
}
return {
complete: false,
number: current.number,
name: sanitizeForDisplay(current.name),
number: testNumber,
name: sanitizeForDisplay(testName),
expected: sanitizeForDisplay(expected),
};
}
@@ -342,10 +367,14 @@ function parseUatItems(content: string): UatItem[] {
function parseVerificationItems(content: string, status: string): UatItem[] {
const items: UatItem[] = [];
if (status === 'human_needed') {
// Extract from human_verification section — look for numbered items or table rows
const hvSection = content.match(/##\s*Human Verification.*?\n([\s\S]*?)(?=\n##\s|\n---\s|$)/i);
// Use the seam to locate the ## Human Verification section (ADR-1372 T5).
const hvSection = collectSection(
content,
(h) => /^human\s+verification/i.test(h.text) && h.level === 2,
{ levelBounded: true },
);
if (hvSection) {
const lines = hvSection[1].split('\n');
const lines = hvSection.body.split('\n');
for (const line of lines) {
// Match table rows: | N | description | ... |
const tableMatch = line.match(/\|\s*(\d+)\s*\|\s*([^|]+)/);

View File

@@ -17,8 +17,9 @@
* - Empty/whitespace/non-string input
*
* Includes a fast-check property test (stripFencedCode idempotence invariant).
* Includes a parity guard for the tracked duplication between stripFencedCode
* and uat-predicate's _stripFencedBlocks (DEFECT.GENERATIVE-FIX — removed in T5).
* The parity guard for the T0-era tracked duplication (stripFencedCode vs
* uat-predicate _stripFencedBlocks) was removed in T5: uat-predicate now imports
* the seam directly, so the guard would compare the seam to itself.
*/
const { test, describe } = require('node:test');
@@ -35,17 +36,6 @@ const {
replaceSection,
} = require('../gsd-core/bin/lib/markdown-sectionizer.cjs');
// uat-predicate's _stripFencedBlocks is not directly exported.
// The closest public surface is stripFalsePositiveContexts, which applies:
// (a) frontmatter strip, (b) HTML comment strip, (c) _stripFencedBlocks, (d) blockquote strip.
// For the parity corpus we use inputs with NO frontmatter, NO HTML comments, and NO blockquotes,
// so the only transformation applied is the fence stripping in step (c).
// We also use analyzeMarkdown, which calls _stripFencedBlocks directly for unterminatedFence.
const {
stripFalsePositiveContexts,
analyzeMarkdown,
} = require('../gsd-core/bin/lib/uat-predicate.cjs');
// ─── stripFencedCode ──────────────────────────────────────────────────────────
describe('stripFencedCode', () => {
@@ -1045,98 +1035,6 @@ describe('extractTaggedBlocks: nested same-name tag behavior (non-greedy limitat
});
});
// ─── Parity guard: stripFencedCode vs uat-predicate _stripFencedBlocks ────────
//
// DEFECT.GENERATIVE-FIX: stripFencedCode in the seam is a tracked duplication of
// _stripFencedBlocks in uat-predicate.cts until tier T5 deduplicates them.
// This test MUST FAIL if the two implementations diverge on any corpus input.
// Remove this describe block in T5 when uat-predicate imports the seam directly.
//
// Approach: feed a shared fence-input corpus through:
// (A) stripFencedCode (seam — direct export)
// (B) stripFalsePositiveContexts (uat-predicate public surface)
// Input must have NO frontmatter (not starting with ---), NO HTML comments,
// and NO blockquote lines, so that steps (a)(b)(d) in stripFalsePositiveContexts
// are no-ops and only the fence-stripping step (c) differs between them.
// (C) analyzeMarkdown.unterminatedFence (uat-predicate — calls _stripFencedBlocks directly)
//
// Limitation: _stripFencedBlocks is not directly exported from uat-predicate.cjs,
// so we test through the closest public surface and document the boundary.
describe('parity guard: stripFencedCode vs uat-predicate fence-stripping', () => {
// Shared corpus of fence inputs for parity testing.
// All inputs have no frontmatter, no HTML comments, no blockquotes — only fences.
const FENCE_CORPUS = [
{
label: 'no fences',
input: '## Heading\n\nSome text.\n\n- bullet',
},
{
label: 'backtick fence',
input: 'before\n```js\nconst x = 1;\n```\nafter',
},
{
label: 'tilde fence',
input: 'before\n~~~\nsome code\n~~~\nafter',
},
{
label: 'CRLF fence',
input: 'before\r\n```\r\ncode\r\n```\r\nafter',
},
{
label: 'unterminated fence',
input: 'before\n```\nsome code without closing fence',
},
{
label: 'tilde inside backtick fence (mismatched delimiter)',
input: '```\n~~~\nstill inside\n```\noutside',
},
{
label: 'backtick inside tilde fence (mismatched delimiter)',
input: '~~~\n```\nstill inside\n~~~\noutside',
},
{
label: 'longer closing fence run',
input: 'text\n```\nbody\n`````\nafter',
},
{
label: 'multiple successive fenced blocks',
input: 'a\n```\ncode1\n```\nb\n```\ncode2\n```\nc',
},
// NOTE: '4-space indent' case is intentionally excluded from the parity corpus.
// The seam uses /^( {0,3})/ (CommonMark §4.5: ≤3 leading spaces tolerated),
// while uat-predicate._stripFencedBlocks uses /^(\s*)/ (any whitespace).
// A 4-space-indented ``` is NOT a fence opener per CommonMark but IS treated
// as one by uat-predicate. This is a known pre-existing divergence; the seam
// is the more-correct implementation. The divergence is documented here so that
// T5 (which will remove uat-predicate's local copy) is aware of the fix needed.
];
for (const { label, input } of FENCE_CORPUS) {
test(`text output parity: ${label}`, () => {
const seamResult = stripFencedCode(input).text;
// stripFalsePositiveContexts with no-frontmatter/no-comment/no-blockquote input
// reduces to exactly _stripFencedBlocks (step c only).
const uatResult = stripFalsePositiveContexts(input);
assert.equal(
seamResult,
uatResult,
`stripFencedCode and uat-predicate _stripFencedBlocks diverged on: ${JSON.stringify(label)}\n` +
`seam: ${JSON.stringify(seamResult.slice(0, 120))}\n` +
`uat: ${JSON.stringify(uatResult.slice(0, 120))}`,
);
});
test(`unterminatedFence parity: ${label}`, () => {
const seamUnterminated = stripFencedCode(input).unterminatedFence;
// analyzeMarkdown calls _stripFencedBlocks directly for unterminatedFence.
const uatUnterminated = analyzeMarkdown(input).unterminatedFence;
assert.equal(
seamUnterminated,
uatUnterminated,
`unterminatedFence diverged on: ${JSON.stringify(label)}\n` +
`seam: ${seamUnterminated}, uat: ${uatUnterminated}`,
);
});
}
});
// Parity guard removed in T5 (ADR-1372): uat-predicate now imports stripFencedCode
// from the seam directly, so comparing the seam to itself is tautological.
// The seam's stripFencedCode correctness is already covered by the tests above.