* feat(#247): runtime-neutral phase uat-passed predicate from HUMAN-UAT results Wire the already-reserved `phase.uat-passed` alias (subcommand `uat-passed`, mutation:false) into the phase command router with a new markdown-aware predicate that evaluates HUMAN-UAT results and reports pass only when every required check passes. Post-SDK-retirement (ADR-0174/#174) successor to the SDK-framed #70, with no SDK-specific API surface. New pure module src/uat-predicate.cts: - stripFalsePositiveContexts: frontmatter -> HTML-comment -> CommonMark-style fenced-block state machine (tracks delimiter char+length) -> blockquote, each a small composable step, so a `result: passed` inside frontmatter, a fenced/~~~ block (incl. ~~~ nested in a ``` fence), a comment, or a blockquote is never counted. - parseUatResultItems: heading-block parser, column-0-anchored same-line result; a heading with no result -> `missing` (fail-closed). - analyzeMarkdown: unterminated fence/comment detection (malformed -> blocker). - evaluateUatPassed: allowlist pass/verification semantics; passed = no blockers && >=1 check && all passing; no_uat_artifacts discriminator (no vacuous pass); optional requireVerification policy hook. Thin cmdPhaseUatPassed handler in phase.cts; router closure rejects unknown flags via makeInvalidArgs. Hardened across two Codex adversarial passes (vacuous pass, dropped failing tests, permissive verification status, nested-fence escape, cross-line result value, masked unterminated comment) — all fixed fail-closed. New unit + CLI-integration suites incl. a fast-check property test; docs, CONTEXT glossary, inventory, and changeset updated. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> * chore(#247): backfill changeset PR number (#1063) * fix(#247): indexOf paired-scan for unterminated-comment detection CodeQL js/incomplete-multi-character-sanitization (high) flagged the `raw.replace(/<!--[\s\S]*?-->/g,'')`-then-`.includes('<!--')` detection in analyzeMarkdown as incomplete sanitization (a single regex pass can leave a residual `<!--`). Replace it with a paired left-to-right indexOf scan that contains no `.replace()` of the comment token — CodeQL-clean and strictly more correct (a closed earlier comment can never mask a later unterminated one). Behaviour unchanged; 98 predicate tests + scoped docker run green. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> --------- Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
395 lines
15 KiB
TypeScript
395 lines
15 KiB
TypeScript
/**
|
|
* UAT Predicate — Pure-computation UAT pass/fail evaluation
|
|
*
|
|
* Evaluates all *-UAT.md and *-VERIFICATION.md files in a phase directory and
|
|
* returns a typed report. Used by `phase uat-passed` to harden against the
|
|
* naive whole-file regex in cmdPhaseComplete which false-matches `result:` lines
|
|
* inside frontmatter, fenced code blocks, blockquotes, and HTML comments.
|
|
*
|
|
* Issue #247 — phase uat-passed predicate
|
|
*
|
|
* ADR-457 build-at-publish: compiled by tsc to gsd-core/bin/lib/uat-predicate.cjs.
|
|
*/
|
|
|
|
import fs from 'node:fs';
|
|
import path from 'node:path';
|
|
// eslint-disable-next-line @typescript-eslint/no-require-imports
|
|
import frontmatter = require('./frontmatter.cjs');
|
|
const { extractFrontmatter } = frontmatter;
|
|
|
|
// ─── Types ────────────────────────────────────────────────────────────────────
|
|
|
|
interface UatCheckItem {
|
|
file: string;
|
|
test: number;
|
|
name: string;
|
|
result: string;
|
|
passing: boolean;
|
|
}
|
|
|
|
interface UatPassedReport {
|
|
passed: boolean;
|
|
uat_files: string[];
|
|
verification_files: string[];
|
|
checks: UatCheckItem[];
|
|
blockers: string[];
|
|
no_uat_artifacts: boolean;
|
|
policy: {
|
|
require_verification: boolean;
|
|
};
|
|
}
|
|
|
|
// ─── Blocking state sets (documented for maintainability) ─────────────────────
|
|
|
|
// UAT file frontmatter `status` values that indicate the file is not fully done
|
|
const BLOCKING_UAT_FM_STATUSES = new Set([
|
|
'partial', 'diagnosed', 'pending', 'blocked', 'in_progress', 'failed',
|
|
]);
|
|
|
|
// UAT file frontmatter `result` values that indicate failure
|
|
const BLOCKING_UAT_FM_RESULTS = new Set(['pending', 'blocked', 'failed']);
|
|
|
|
// VERIFICATION file frontmatter `status` values that indicate passing
|
|
const PASSING_VERIFICATION_STATUSES = new Set([
|
|
'complete', 'verified', 'passed', 'human_passed',
|
|
]);
|
|
|
|
// VERIFICATION file frontmatter `status` values that explicitly block
|
|
const BLOCKING_VERIFICATION_FM_STATUSES = new Set([
|
|
'human_needed', 'gaps_found', 'pending', 'blocked', 'partial',
|
|
'failed', 'in_progress',
|
|
]);
|
|
|
|
// UAT test-item `result` values that count as passing
|
|
const PASSING_RESULTS = new Set(['passed', 'pass']);
|
|
|
|
// ─── stripFalsePositiveContexts ───────────────────────────────────────────────
|
|
|
|
/**
|
|
* Remove contexts that can contain `result: ...` lines that are NOT real test results:
|
|
* (a) leading frontmatter block at byte 0
|
|
* (b) HTML comments (unterminated comments swallow to EOF — fail-closed)
|
|
* (c) fenced code blocks (backtick and tilde, indented too) via CommonMark state machine
|
|
* (d) blockquote lines
|
|
*
|
|
* Each step is a composable function (Kernighan's Law — independently testable).
|
|
* Returns surviving lines joined by '\n'. Robust to CRLF input.
|
|
*/
|
|
function stripFalsePositiveContexts(content: string): string {
|
|
// Step (a): strip leading frontmatter block only at byte 0
|
|
let stripped = content.replace(/^---\r?\n[\s\S]*?\r?\n---[ \t]*(\r?\n|$)/, '');
|
|
|
|
// Step (b): remove HTML comments anywhere; unterminated comment swallows to EOF
|
|
stripped = stripped.replace(/<!--[\s\S]*?(?:-->|$)/g, '');
|
|
|
|
// Step (c): remove fenced code blocks via CommonMark-style state machine (handles CRLF + indented fences)
|
|
stripped = _stripFencedBlocks(stripped).text;
|
|
|
|
// Step (d): remove blockquote lines
|
|
stripped = stripped
|
|
.split('\n')
|
|
.filter(line => !/^\s*>/.test(line))
|
|
.join('\n');
|
|
|
|
return stripped;
|
|
}
|
|
|
|
interface FenceState {
|
|
char: '`' | '~';
|
|
len: number;
|
|
}
|
|
|
|
interface StripFencedResult {
|
|
text: string;
|
|
unterminatedFence: boolean;
|
|
}
|
|
|
|
/**
|
|
* CommonMark-style fenced-code-block stripper.
|
|
* Tracks the opening delimiter char and length so that a ~~~ line inside a
|
|
* ``` fence is correctly treated as fence content, not a closing delimiter.
|
|
*
|
|
* Opening rule: first delimiter line with char+len sets openFence.
|
|
* Closing rule: delimiter line with SAME char, run length >= openFence.len,
|
|
* and NO trailing non-whitespace text closes the fence.
|
|
* All delimiter and content lines are dropped; non-fence lines are kept.
|
|
* Returns the kept text plus unterminatedFence:true if EOF inside a fence.
|
|
*/
|
|
function _stripFencedBlocks(content: string): StripFencedResult {
|
|
const lines = content.split('\n');
|
|
const kept: string[] = [];
|
|
let openFence: FenceState | null = null;
|
|
const delimRe = /^(\s*)(`{3,}|~{3,})(.*)$/;
|
|
|
|
for (const rawLine of lines) {
|
|
// Tolerate CRLF: strip trailing \r for matching, but we work on split-by-\n lines
|
|
// (the outer caller joined by \n already; we just handle a stray \r in the last char)
|
|
const line = rawLine.replace(/\r$/, '');
|
|
const m = delimRe.exec(line);
|
|
if (m) {
|
|
const char = m[2][0] as '`' | '~';
|
|
const len = m[2].length;
|
|
const trailing = m[3];
|
|
if (openFence === null) {
|
|
// Opening delimiter — drop this line and record the fence
|
|
openFence = { char, len };
|
|
} else if (char === openFence.char && len >= openFence.len && /^\s*$/.test(trailing)) {
|
|
// Closing delimiter (same char, sufficient length, no trailing text) — drop and close
|
|
openFence = null;
|
|
}
|
|
// else: mismatched delimiter inside fence (e.g. ~~~ inside ```) — drop as content
|
|
continue; // delimiter lines are always dropped
|
|
}
|
|
if (openFence === null) {
|
|
kept.push(rawLine);
|
|
}
|
|
// Lines inside fence are dropped
|
|
}
|
|
|
|
return { text: kept.join('\n'), unterminatedFence: openFence !== null };
|
|
}
|
|
|
|
/**
|
|
* Analyse raw markdown for structural anomalies (unterminated fence / comment).
|
|
* Exported for unit-testability and used by evaluateUatPassed for per-file malformed detection.
|
|
*
|
|
* FIX C: properly balanced comments are stripped before checking for a dangling <!--,
|
|
* so an earlier closed comment does not mask a later unterminated one.
|
|
*/
|
|
function analyzeMarkdown(raw: string): { unterminatedFence: boolean; unterminatedComment: boolean } {
|
|
// Detect an unterminated HTML comment via a paired scan: every `<!--` must
|
|
// have a following `-->`. Using indexOf (not a regex .replace of the comment
|
|
// token) avoids the js/incomplete-multi-character-sanitization pattern — and
|
|
// is exact: a closed earlier comment never masks a later unterminated one.
|
|
let unterminatedComment = false;
|
|
for (let i = 0; ; ) {
|
|
const open = raw.indexOf('<!--', i);
|
|
if (open === -1) break;
|
|
const close = raw.indexOf('-->', open + 4);
|
|
if (close === -1) { unterminatedComment = true; break; }
|
|
i = close + 3;
|
|
}
|
|
|
|
// Fence state machine gives the accurate unterminated-fence signal.
|
|
const { unterminatedFence } = _stripFencedBlocks(raw);
|
|
|
|
return { unterminatedFence, unterminatedComment };
|
|
}
|
|
|
|
// ─── parseUatResultItems ──────────────────────────────────────────────────────
|
|
|
|
/**
|
|
* HEADING-BLOCK parser: scan the CLEANED body (after stripFalsePositiveContexts)
|
|
* for UAT test blocks.
|
|
*
|
|
* For each ### N. Name heading, the block spans until the next ### heading or EOF.
|
|
* Within each block, find a column-0 anchored result line (rejects indented YAML
|
|
* block-scalar bodies and inline/quoted fakes).
|
|
*
|
|
* - If a heading block has NO column-0 result line → emit result:'missing' (blocker).
|
|
* - Support bracketed [passed] and bare passed (#2273).
|
|
* - Returns ALL items (both passing and non-passing).
|
|
*/
|
|
function parseUatResultItems(cleanContent: string): Array<{ test: number; name: string; result: string }> {
|
|
const items: Array<{ test: number; name: string; result: string }> = [];
|
|
|
|
// Find all ### N. Name headings (line-anchored)
|
|
const headingPattern = /^###\s*(\d+)\.\s*(.+)$/gm;
|
|
const headings: Array<{ index: number; test: number; name: string }> = [];
|
|
let hMatch: RegExpExecArray | null;
|
|
while ((hMatch = headingPattern.exec(cleanContent)) !== null) {
|
|
headings.push({
|
|
index: hMatch.index + hMatch[0].length,
|
|
test: parseInt(hMatch[1], 10),
|
|
name: hMatch[2].trim(),
|
|
});
|
|
}
|
|
|
|
for (let i = 0; i < headings.length; i++) {
|
|
const h = headings[i];
|
|
const blockStart = h.index;
|
|
// More precise: find next heading's position in the original string
|
|
// We'll slice from current heading end to the position just before next heading's "###"
|
|
const nextHeadingMatch = i + 1 < headings.length
|
|
? cleanContent.lastIndexOf('\n###', headings[i + 1].index)
|
|
: -1;
|
|
const blockContent = nextHeadingMatch >= blockStart
|
|
? cleanContent.slice(blockStart, nextHeadingMatch)
|
|
: cleanContent.slice(blockStart);
|
|
|
|
// Column-0 anchored result line: /^result:[ \t]*\[?([\w-]+)\]?/mi
|
|
// Uses [ \t]* (not \s*) so the captured value must sit on the SAME line as result:.
|
|
// A result: key with the value on a subsequent line yields no match → 'missing' (blocker).
|
|
const resultMatch = /^result:[ \t]*\[?([\w-]+)\]?/mi.exec(blockContent);
|
|
if (resultMatch) {
|
|
items.push({
|
|
test: h.test,
|
|
name: h.name,
|
|
result: resultMatch[1].toLowerCase(),
|
|
});
|
|
} else {
|
|
// No column-0 result line → emit 'missing' (a non-passing state)
|
|
items.push({
|
|
test: h.test,
|
|
name: h.name,
|
|
result: 'missing',
|
|
});
|
|
}
|
|
}
|
|
|
|
return items;
|
|
}
|
|
|
|
// ─── evaluateUatPassed ────────────────────────────────────────────────────────
|
|
|
|
/**
|
|
* Evaluate all UAT/VERIFICATION files in a phase directory.
|
|
* Returns a UatPassedReport with the locked, stable shape defined by the interface.
|
|
*
|
|
* FAIL-CLOSED: any absence/ambiguity/malformed input → NOT passed.
|
|
* Pass requires at least one real passing check AND no blockers.
|
|
*/
|
|
function evaluateUatPassed(
|
|
phaseFullDir: string,
|
|
opts?: { policy?: { requireVerification?: boolean } },
|
|
): UatPassedReport {
|
|
const requireVerification = opts?.policy?.requireVerification === true;
|
|
|
|
const blockers: string[] = [];
|
|
const checks: UatCheckItem[] = [];
|
|
const uatFiles: string[] = [];
|
|
const verificationFiles: string[] = [];
|
|
|
|
// Read the directory — if unreadable, treat as no files (fail-closed: no artifacts → not passed)
|
|
let dirEntries: string[] = [];
|
|
try {
|
|
dirEntries = fs.readdirSync(phaseFullDir);
|
|
} catch {
|
|
// Unreadable dir — no_uat_artifacts:true, passed:false
|
|
const no_uat_artifacts = true;
|
|
if (requireVerification) {
|
|
blockers.push('policy: verification required but no passing *-VERIFICATION.md found');
|
|
}
|
|
return {
|
|
passed: false,
|
|
uat_files: [],
|
|
verification_files: [],
|
|
checks: [],
|
|
blockers,
|
|
no_uat_artifacts,
|
|
policy: { require_verification: requireVerification },
|
|
};
|
|
}
|
|
|
|
// Filter UAT and VERIFICATION files using the same filter as cmdPhaseComplete
|
|
const uatFileNames = dirEntries.filter(f => f.includes('-UAT') && f.endsWith('.md'));
|
|
const verFileNames = dirEntries.filter(f => f.includes('-VERIFICATION') && f.endsWith('.md'));
|
|
|
|
// ── Process UAT files ──────────────────────────────────────────────────────
|
|
for (const file of uatFileNames) {
|
|
uatFiles.push(file);
|
|
let raw = '';
|
|
try {
|
|
raw = fs.readFileSync(path.join(phaseFullDir, file), 'utf-8');
|
|
} catch {
|
|
blockers.push(`${file}: could not read file`);
|
|
continue;
|
|
}
|
|
|
|
// ── Per-file malformed markdown guard ──────────────────────────────────
|
|
// FIX D: use accurate signals from analyzeMarkdown instead of heuristics.
|
|
// unterminatedFence: CommonMark state machine detects a genuinely unclosed fence.
|
|
// unterminatedComment: strips balanced comments first, then checks for leftover <!--.
|
|
const { unterminatedFence, unterminatedComment } = analyzeMarkdown(raw);
|
|
if (unterminatedFence || unterminatedComment) {
|
|
blockers.push(`${file}: malformed markdown (unterminated fence or comment)`);
|
|
}
|
|
|
|
const fm = extractFrontmatter(raw) as Record<string, unknown>;
|
|
|
|
// File-level frontmatter status check
|
|
if (fm['status'] && BLOCKING_UAT_FM_STATUSES.has(fm['status'] as string)) {
|
|
blockers.push(`${file}: frontmatter status=${fm['status'] as string}`);
|
|
}
|
|
// File-level frontmatter result check
|
|
if (fm['result'] && BLOCKING_UAT_FM_RESULTS.has(fm['result'] as string)) {
|
|
blockers.push(`${file}: frontmatter result=${fm['result'] as string}`);
|
|
}
|
|
|
|
// Parse test items from the cleaned body (hardened against false positives)
|
|
const cleanContent = stripFalsePositiveContexts(raw);
|
|
const items = parseUatResultItems(cleanContent);
|
|
|
|
for (const item of items) {
|
|
const passing = PASSING_RESULTS.has(item.result);
|
|
checks.push({
|
|
file,
|
|
test: item.test,
|
|
name: item.name,
|
|
result: item.result,
|
|
passing,
|
|
});
|
|
if (!passing) {
|
|
blockers.push(`${file}: test ${item.test} (${item.result})`);
|
|
}
|
|
}
|
|
}
|
|
|
|
// ── Process VERIFICATION files ─────────────────────────────────────────────
|
|
let hasPassingVerification = false;
|
|
for (const file of verFileNames) {
|
|
verificationFiles.push(file);
|
|
let raw = '';
|
|
try {
|
|
raw = fs.readFileSync(path.join(phaseFullDir, file), 'utf-8');
|
|
} catch {
|
|
blockers.push(`${file}: could not read verification file`);
|
|
continue;
|
|
}
|
|
|
|
const vfm = extractFrontmatter(raw) as Record<string, unknown>;
|
|
const vStatus = vfm['status'] as string | undefined;
|
|
|
|
if (vStatus && BLOCKING_VERIFICATION_FM_STATUSES.has(vStatus)) {
|
|
blockers.push(`${file}: verification status=${vStatus}`);
|
|
} else if (vStatus && PASSING_VERIFICATION_STATUSES.has(vStatus)) {
|
|
// Allowlist: only explicitly-passing statuses count
|
|
hasPassingVerification = true;
|
|
}
|
|
// Missing or unknown status: does NOT count as passing, does NOT push a blocker
|
|
// (handled by the requireVerification policy check below if needed)
|
|
}
|
|
|
|
// ── Policy: requireVerification ───────────────────────────────────────────
|
|
if (requireVerification && !hasPassingVerification) {
|
|
blockers.push('policy: verification required but no passing *-VERIFICATION.md found');
|
|
}
|
|
|
|
// ── Determine no_uat_artifacts and passed ─────────────────────────────────
|
|
// no_uat_artifacts: true when no real UAT test items were parsed from any file
|
|
const no_uat_artifacts = checks.length === 0;
|
|
|
|
// FIX 1: require positive passing evidence; no vacuous pass
|
|
// passed = no blockers AND at least one check AND all checks passing
|
|
const passed = blockers.length === 0 && checks.length > 0 && checks.every(c => c.passing);
|
|
|
|
return {
|
|
passed,
|
|
uat_files: uatFiles,
|
|
verification_files: verificationFiles,
|
|
checks,
|
|
blockers,
|
|
no_uat_artifacts,
|
|
policy: {
|
|
require_verification: requireVerification,
|
|
},
|
|
};
|
|
}
|
|
|
|
export = {
|
|
stripFalsePositiveContexts,
|
|
parseUatResultItems,
|
|
analyzeMarkdown,
|
|
evaluateUatPassed,
|
|
};
|