Files
msd-core/src/uat-predicate.cts
Tom Boucher 9223f2f4c8 feat(#247): runtime-neutral phase uat-passed predicate from HUMAN-UAT results (#1063)
* feat(#247): runtime-neutral phase uat-passed predicate from HUMAN-UAT results

Wire the already-reserved `phase.uat-passed` alias (subcommand `uat-passed`,
mutation:false) into the phase command router with a new markdown-aware
predicate that evaluates HUMAN-UAT results and reports pass only when every
required check passes. Post-SDK-retirement (ADR-0174/#174) successor to the
SDK-framed #70, with no SDK-specific API surface.

New pure module src/uat-predicate.cts:
- stripFalsePositiveContexts: frontmatter -> HTML-comment -> CommonMark-style
  fenced-block state machine (tracks delimiter char+length) -> blockquote,
  each a small composable step, so a `result: passed` inside frontmatter, a
  fenced/~~~ block (incl. ~~~ nested in a ``` fence), a comment, or a
  blockquote is never counted.
- parseUatResultItems: heading-block parser, column-0-anchored same-line
  result; a heading with no result -> `missing` (fail-closed).
- analyzeMarkdown: unterminated fence/comment detection (malformed -> blocker).
- evaluateUatPassed: allowlist pass/verification semantics; passed = no
  blockers && >=1 check && all passing; no_uat_artifacts discriminator (no
  vacuous pass); optional requireVerification policy hook.

Thin cmdPhaseUatPassed handler in phase.cts; router closure rejects unknown
flags via makeInvalidArgs. Hardened across two Codex adversarial passes
(vacuous pass, dropped failing tests, permissive verification status,
nested-fence escape, cross-line result value, masked unterminated comment) —
all fixed fail-closed. New unit + CLI-integration suites incl. a fast-check
property test; docs, CONTEXT glossary, inventory, and changeset updated.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

* chore(#247): backfill changeset PR number (#1063)

* fix(#247): indexOf paired-scan for unterminated-comment detection

CodeQL js/incomplete-multi-character-sanitization (high) flagged the
`raw.replace(/<!--[\s\S]*?-->/g,'')`-then-`.includes('<!--')` detection in
analyzeMarkdown as incomplete sanitization (a single regex pass can leave a
residual `<!--`). Replace it with a paired left-to-right indexOf scan that
contains no `.replace()` of the comment token — CodeQL-clean and strictly
more correct (a closed earlier comment can never mask a later unterminated
one). Behaviour unchanged; 98 predicate tests + scoped docker run green.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

---------

Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
2026-06-11 16:37:17 -04:00

395 lines
15 KiB
TypeScript

/**
* UAT Predicate — Pure-computation UAT pass/fail evaluation
*
* Evaluates all *-UAT.md and *-VERIFICATION.md files in a phase directory and
* returns a typed report. Used by `phase uat-passed` to harden against the
* naive whole-file regex in cmdPhaseComplete which false-matches `result:` lines
* inside frontmatter, fenced code blocks, blockquotes, and HTML comments.
*
* Issue #247 — phase uat-passed predicate
*
* ADR-457 build-at-publish: compiled by tsc to gsd-core/bin/lib/uat-predicate.cjs.
*/
import fs from 'node:fs';
import path from 'node:path';
// eslint-disable-next-line @typescript-eslint/no-require-imports
import frontmatter = require('./frontmatter.cjs');
const { extractFrontmatter } = frontmatter;
// ─── Types ────────────────────────────────────────────────────────────────────
interface UatCheckItem {
file: string;
test: number;
name: string;
result: string;
passing: boolean;
}
interface UatPassedReport {
passed: boolean;
uat_files: string[];
verification_files: string[];
checks: UatCheckItem[];
blockers: string[];
no_uat_artifacts: boolean;
policy: {
require_verification: boolean;
};
}
// ─── Blocking state sets (documented for maintainability) ─────────────────────
// UAT file frontmatter `status` values that indicate the file is not fully done
const BLOCKING_UAT_FM_STATUSES = new Set([
'partial', 'diagnosed', 'pending', 'blocked', 'in_progress', 'failed',
]);
// UAT file frontmatter `result` values that indicate failure
const BLOCKING_UAT_FM_RESULTS = new Set(['pending', 'blocked', 'failed']);
// VERIFICATION file frontmatter `status` values that indicate passing
const PASSING_VERIFICATION_STATUSES = new Set([
'complete', 'verified', 'passed', 'human_passed',
]);
// VERIFICATION file frontmatter `status` values that explicitly block
const BLOCKING_VERIFICATION_FM_STATUSES = new Set([
'human_needed', 'gaps_found', 'pending', 'blocked', 'partial',
'failed', 'in_progress',
]);
// UAT test-item `result` values that count as passing
const PASSING_RESULTS = new Set(['passed', 'pass']);
// ─── stripFalsePositiveContexts ───────────────────────────────────────────────
/**
* Remove contexts that can contain `result: ...` lines that are NOT real test results:
* (a) leading frontmatter block at byte 0
* (b) HTML comments (unterminated comments swallow to EOF — fail-closed)
* (c) fenced code blocks (backtick and tilde, indented too) via CommonMark state machine
* (d) blockquote lines
*
* Each step is a composable function (Kernighan's Law — independently testable).
* Returns surviving lines joined by '\n'. Robust to CRLF input.
*/
function stripFalsePositiveContexts(content: string): string {
// Step (a): strip leading frontmatter block only at byte 0
let stripped = content.replace(/^---\r?\n[\s\S]*?\r?\n---[ \t]*(\r?\n|$)/, '');
// Step (b): remove HTML comments anywhere; unterminated comment swallows to EOF
stripped = stripped.replace(/<!--[\s\S]*?(?:-->|$)/g, '');
// Step (c): remove fenced code blocks via CommonMark-style state machine (handles CRLF + indented fences)
stripped = _stripFencedBlocks(stripped).text;
// Step (d): remove blockquote lines
stripped = stripped
.split('\n')
.filter(line => !/^\s*>/.test(line))
.join('\n');
return stripped;
}
interface FenceState {
char: '`' | '~';
len: number;
}
interface StripFencedResult {
text: string;
unterminatedFence: boolean;
}
/**
* CommonMark-style fenced-code-block stripper.
* Tracks the opening delimiter char and length so that a ~~~ line inside a
* ``` fence is correctly treated as fence content, not a closing delimiter.
*
* Opening rule: first delimiter line with char+len sets openFence.
* Closing rule: delimiter line with SAME char, run length >= openFence.len,
* and NO trailing non-whitespace text closes the fence.
* All delimiter and content lines are dropped; non-fence lines are kept.
* Returns the kept text plus unterminatedFence:true if EOF inside a fence.
*/
function _stripFencedBlocks(content: string): StripFencedResult {
const lines = content.split('\n');
const kept: string[] = [];
let openFence: FenceState | null = null;
const delimRe = /^(\s*)(`{3,}|~{3,})(.*)$/;
for (const rawLine of lines) {
// Tolerate CRLF: strip trailing \r for matching, but we work on split-by-\n lines
// (the outer caller joined by \n already; we just handle a stray \r in the last char)
const line = rawLine.replace(/\r$/, '');
const m = delimRe.exec(line);
if (m) {
const char = m[2][0] as '`' | '~';
const len = m[2].length;
const trailing = m[3];
if (openFence === null) {
// Opening delimiter — drop this line and record the fence
openFence = { char, len };
} else if (char === openFence.char && len >= openFence.len && /^\s*$/.test(trailing)) {
// Closing delimiter (same char, sufficient length, no trailing text) — drop and close
openFence = null;
}
// else: mismatched delimiter inside fence (e.g. ~~~ inside ```) — drop as content
continue; // delimiter lines are always dropped
}
if (openFence === null) {
kept.push(rawLine);
}
// Lines inside fence are dropped
}
return { text: kept.join('\n'), unterminatedFence: openFence !== null };
}
/**
* Analyse raw markdown for structural anomalies (unterminated fence / comment).
* Exported for unit-testability and used by evaluateUatPassed for per-file malformed detection.
*
* FIX C: properly balanced comments are stripped before checking for a dangling <!--,
* so an earlier closed comment does not mask a later unterminated one.
*/
function analyzeMarkdown(raw: string): { unterminatedFence: boolean; unterminatedComment: boolean } {
// Detect an unterminated HTML comment via a paired scan: every `<!--` must
// have a following `-->`. Using indexOf (not a regex .replace of the comment
// token) avoids the js/incomplete-multi-character-sanitization pattern — and
// is exact: a closed earlier comment never masks a later unterminated one.
let unterminatedComment = false;
for (let i = 0; ; ) {
const open = raw.indexOf('<!--', i);
if (open === -1) break;
const close = raw.indexOf('-->', open + 4);
if (close === -1) { unterminatedComment = true; break; }
i = close + 3;
}
// Fence state machine gives the accurate unterminated-fence signal.
const { unterminatedFence } = _stripFencedBlocks(raw);
return { unterminatedFence, unterminatedComment };
}
// ─── parseUatResultItems ──────────────────────────────────────────────────────
/**
* HEADING-BLOCK parser: scan the CLEANED body (after stripFalsePositiveContexts)
* for UAT test blocks.
*
* For each ### N. Name heading, the block spans until the next ### heading or EOF.
* Within each block, find a column-0 anchored result line (rejects indented YAML
* block-scalar bodies and inline/quoted fakes).
*
* - If a heading block has NO column-0 result line → emit result:'missing' (blocker).
* - Support bracketed [passed] and bare passed (#2273).
* - Returns ALL items (both passing and non-passing).
*/
function parseUatResultItems(cleanContent: string): Array<{ test: number; name: string; result: string }> {
const items: Array<{ test: number; name: string; result: string }> = [];
// Find all ### N. Name headings (line-anchored)
const headingPattern = /^###\s*(\d+)\.\s*(.+)$/gm;
const headings: Array<{ index: number; test: number; name: string }> = [];
let hMatch: RegExpExecArray | null;
while ((hMatch = headingPattern.exec(cleanContent)) !== null) {
headings.push({
index: hMatch.index + hMatch[0].length,
test: parseInt(hMatch[1], 10),
name: hMatch[2].trim(),
});
}
for (let i = 0; i < headings.length; i++) {
const h = headings[i];
const blockStart = h.index;
// More precise: find next heading's position in the original string
// We'll slice from current heading end to the position just before next heading's "###"
const nextHeadingMatch = i + 1 < headings.length
? cleanContent.lastIndexOf('\n###', headings[i + 1].index)
: -1;
const blockContent = nextHeadingMatch >= blockStart
? cleanContent.slice(blockStart, nextHeadingMatch)
: cleanContent.slice(blockStart);
// Column-0 anchored result line: /^result:[ \t]*\[?([\w-]+)\]?/mi
// Uses [ \t]* (not \s*) so the captured value must sit on the SAME line as result:.
// A result: key with the value on a subsequent line yields no match → 'missing' (blocker).
const resultMatch = /^result:[ \t]*\[?([\w-]+)\]?/mi.exec(blockContent);
if (resultMatch) {
items.push({
test: h.test,
name: h.name,
result: resultMatch[1].toLowerCase(),
});
} else {
// No column-0 result line → emit 'missing' (a non-passing state)
items.push({
test: h.test,
name: h.name,
result: 'missing',
});
}
}
return items;
}
// ─── evaluateUatPassed ────────────────────────────────────────────────────────
/**
* Evaluate all UAT/VERIFICATION files in a phase directory.
* Returns a UatPassedReport with the locked, stable shape defined by the interface.
*
* FAIL-CLOSED: any absence/ambiguity/malformed input → NOT passed.
* Pass requires at least one real passing check AND no blockers.
*/
function evaluateUatPassed(
phaseFullDir: string,
opts?: { policy?: { requireVerification?: boolean } },
): UatPassedReport {
const requireVerification = opts?.policy?.requireVerification === true;
const blockers: string[] = [];
const checks: UatCheckItem[] = [];
const uatFiles: string[] = [];
const verificationFiles: string[] = [];
// Read the directory — if unreadable, treat as no files (fail-closed: no artifacts → not passed)
let dirEntries: string[] = [];
try {
dirEntries = fs.readdirSync(phaseFullDir);
} catch {
// Unreadable dir — no_uat_artifacts:true, passed:false
const no_uat_artifacts = true;
if (requireVerification) {
blockers.push('policy: verification required but no passing *-VERIFICATION.md found');
}
return {
passed: false,
uat_files: [],
verification_files: [],
checks: [],
blockers,
no_uat_artifacts,
policy: { require_verification: requireVerification },
};
}
// Filter UAT and VERIFICATION files using the same filter as cmdPhaseComplete
const uatFileNames = dirEntries.filter(f => f.includes('-UAT') && f.endsWith('.md'));
const verFileNames = dirEntries.filter(f => f.includes('-VERIFICATION') && f.endsWith('.md'));
// ── Process UAT files ──────────────────────────────────────────────────────
for (const file of uatFileNames) {
uatFiles.push(file);
let raw = '';
try {
raw = fs.readFileSync(path.join(phaseFullDir, file), 'utf-8');
} catch {
blockers.push(`${file}: could not read file`);
continue;
}
// ── Per-file malformed markdown guard ──────────────────────────────────
// FIX D: use accurate signals from analyzeMarkdown instead of heuristics.
// unterminatedFence: CommonMark state machine detects a genuinely unclosed fence.
// unterminatedComment: strips balanced comments first, then checks for leftover <!--.
const { unterminatedFence, unterminatedComment } = analyzeMarkdown(raw);
if (unterminatedFence || unterminatedComment) {
blockers.push(`${file}: malformed markdown (unterminated fence or comment)`);
}
const fm = extractFrontmatter(raw) as Record<string, unknown>;
// File-level frontmatter status check
if (fm['status'] && BLOCKING_UAT_FM_STATUSES.has(fm['status'] as string)) {
blockers.push(`${file}: frontmatter status=${fm['status'] as string}`);
}
// File-level frontmatter result check
if (fm['result'] && BLOCKING_UAT_FM_RESULTS.has(fm['result'] as string)) {
blockers.push(`${file}: frontmatter result=${fm['result'] as string}`);
}
// Parse test items from the cleaned body (hardened against false positives)
const cleanContent = stripFalsePositiveContexts(raw);
const items = parseUatResultItems(cleanContent);
for (const item of items) {
const passing = PASSING_RESULTS.has(item.result);
checks.push({
file,
test: item.test,
name: item.name,
result: item.result,
passing,
});
if (!passing) {
blockers.push(`${file}: test ${item.test} (${item.result})`);
}
}
}
// ── Process VERIFICATION files ─────────────────────────────────────────────
let hasPassingVerification = false;
for (const file of verFileNames) {
verificationFiles.push(file);
let raw = '';
try {
raw = fs.readFileSync(path.join(phaseFullDir, file), 'utf-8');
} catch {
blockers.push(`${file}: could not read verification file`);
continue;
}
const vfm = extractFrontmatter(raw) as Record<string, unknown>;
const vStatus = vfm['status'] as string | undefined;
if (vStatus && BLOCKING_VERIFICATION_FM_STATUSES.has(vStatus)) {
blockers.push(`${file}: verification status=${vStatus}`);
} else if (vStatus && PASSING_VERIFICATION_STATUSES.has(vStatus)) {
// Allowlist: only explicitly-passing statuses count
hasPassingVerification = true;
}
// Missing or unknown status: does NOT count as passing, does NOT push a blocker
// (handled by the requireVerification policy check below if needed)
}
// ── Policy: requireVerification ───────────────────────────────────────────
if (requireVerification && !hasPassingVerification) {
blockers.push('policy: verification required but no passing *-VERIFICATION.md found');
}
// ── Determine no_uat_artifacts and passed ─────────────────────────────────
// no_uat_artifacts: true when no real UAT test items were parsed from any file
const no_uat_artifacts = checks.length === 0;
// FIX 1: require positive passing evidence; no vacuous pass
// passed = no blockers AND at least one check AND all checks passing
const passed = blockers.length === 0 && checks.length > 0 && checks.every(c => c.passing);
return {
passed,
uat_files: uatFiles,
verification_files: verificationFiles,
checks,
blockers,
no_uat_artifacts,
policy: {
require_verification: requireVerification,
},
};
}
export = {
stripFalsePositiveContexts,
parseUatResultItems,
analyzeMarkdown,
evaluateUatPassed,
};