* test(3770): add failing tests for intentional RED evidence gate RED: classifyRedEvidence / buildRedEvidenceRecord / check tdd-red-evidence do not exist yet; every row fails on require. Per #3770 only an intentional target-test failure may authorize GREEN; zero-test discovery, fixture crashes, unrelated failures, and unexpected green are INVALID_RED. * fix(3770): require intentional RED evidence before GREEN Only an intentional failure of the TARGET test (distinctly named, TAP-reported assertion failure) classifies as RED_EVIDENCE_OK and authorizes GREEN. Zero-test discovery, fixture/load crashes (file-named failures), nonzero exits without a failing test, unrelated failures, unexpected greens, and malformed/missing records are INVALID_RED and block GREEN. - src/tdd-red-evidence.cts: pure classifier + persisted record builder (reuses the prohibition-enforcement TAP primitives; fail-closed, never throws) - check tdd-red-evidence <record.json>: validates the persisted record (command, exit code, failing test, expected, actual) - gsd-executor.md / references/tdd.md / references/execute-mvp-tdd.md: RED now requires the evidence record + gate verdict, not a nonzero exit or a RED: tag * chore(3770): regenerate inventory manifest for tdd-red-evidence.cjs * fix(3770): fit executor fail-fast under size cap, fix unrelated-failure fixture, ignore generated lib - gsd-executor.md: compress the #3770 fail-fast rule to one line (49149 B < 49152 cap; line-count parity keeps the #2751 PROSE_ALLOWLIST line 816 valid) - tests: the row-6 fixture used String.replace (first-occurrence), so the `not ok` line still named the target test and the classifier was right to accept it; replaceAll makes the failure genuinely unrelated - eslint.config.mjs: ignore tsc-generated bin/lib/tdd-red-evidence.cjs (lint the src/*.cts source, per ADR-457 migration rule) Emitted-Drift-Ack-Growth: gsd-executor.md — the #3770 fail-fast rule now requires intentional RED evidence (check tdd-red-evidence) before GREEN; +172 bytes, kept under the LARGE cap and on one line * chore(3770): add changeset * chore(3770): backfill PR number in changeset --------- Co-authored-by: sim <sim@local>
This commit is contained in:
@@ -37,6 +37,7 @@ const { getRoadmapPhaseWithFallback } = roadmapModule;
|
||||
import gapCheckerModule = require('./gap-checker.cjs');
|
||||
const { runGapAnalysis } = gapCheckerModule;
|
||||
import { routeProhibitionEnforcement } from './prohibition-enforcement.cjs';
|
||||
import { classifyRedEvidence, buildRedEvidenceRecord } from './tdd-red-evidence.cjs';
|
||||
// eslint-disable-next-line @typescript-eslint/no-require-imports
|
||||
import gatePredicateEval = require('./gate-predicate-evaluator.cjs');
|
||||
const { evaluatePredicate } = gatePredicateEval;
|
||||
@@ -935,6 +936,83 @@ function cmdTddReviewCheckpoint(projectDir: string, args: string[], raw: boolean
|
||||
output(result, raw, undefined);
|
||||
}
|
||||
|
||||
// ─── tdd-red-evidence (#3770) ──────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* tdd-red-evidence: validates a persisted RED-phase test-run record for a
|
||||
* `type: tdd` plan (#3770). Only an INTENTIONAL failure of the target test
|
||||
* (verdict RED_EVIDENCE_OK) may authorize GREEN; zero-test discovery, fixture/
|
||||
* load crashes, nonzero exits without a failing test, unrelated failures, and
|
||||
* unexpected greens are INVALID_RED and block GREEN.
|
||||
*
|
||||
* The record is the JSON the executor persists after running the RED command:
|
||||
* { command, exitCode, output, targetTest, targetFile?, expected?, actual? }
|
||||
* Fail-closed: a missing/unreadable/unparseable record is INVALID_RED
|
||||
* (reason unreadable_record), never a pass.
|
||||
*
|
||||
* Args: check tdd-red-evidence <record.json>
|
||||
*/
|
||||
function cmdTddRedEvidence(_projectDir: string, args: string[], raw: boolean): void {
|
||||
const recordPath = typeof args[2] === 'string' ? args[2] : '';
|
||||
if (!recordPath) {
|
||||
error('tdd-red-evidence requires a record path: check tdd-red-evidence <record.json>', ERROR_REASON.SDK_MISSING_ARG);
|
||||
return;
|
||||
}
|
||||
const resolved = path.resolve(recordPath);
|
||||
const text = readIfExists(resolved);
|
||||
const input = ((): Record<string, unknown> | null => {
|
||||
if (!text) return null;
|
||||
try {
|
||||
return (JSON.parse(text) ?? {}) as Record<string, unknown>;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
})();
|
||||
if (!input) {
|
||||
output(
|
||||
{
|
||||
passed: false,
|
||||
block: true,
|
||||
verdict: 'INVALID_RED',
|
||||
reason: 'unreadable_record',
|
||||
record: resolved,
|
||||
readError: text ? `record is not valid JSON: ${resolved}` : `record not found or unreadable: ${resolved}`,
|
||||
},
|
||||
raw,
|
||||
undefined,
|
||||
);
|
||||
return;
|
||||
}
|
||||
const evidenceInput = {
|
||||
command: input['command'],
|
||||
exitCode: input['exitCode'],
|
||||
output: input['output'],
|
||||
targetTest: input['targetTest'],
|
||||
targetFile: input['targetFile'],
|
||||
expected: input['expected'],
|
||||
actual: input['actual'],
|
||||
};
|
||||
const result = classifyRedEvidence(evidenceInput);
|
||||
const record = buildRedEvidenceRecord(evidenceInput, result);
|
||||
output(
|
||||
{
|
||||
// Uniform gate contract: block = !passed. INVALID_RED blocks GREEN.
|
||||
passed: result.verdict === 'RED_EVIDENCE_OK',
|
||||
block: result.verdict !== 'RED_EVIDENCE_OK',
|
||||
verdict: result.verdict,
|
||||
reason: result.reason,
|
||||
evidence: result.evidence,
|
||||
record,
|
||||
message:
|
||||
result.verdict === 'RED_EVIDENCE_OK'
|
||||
? `RED evidence verified: target test "${result.evidence.target_test}" failed as expected (exit ${result.evidence.exit_code}). GREEN authorized.`
|
||||
: `INVALID_RED (${result.reason}): GREEN blocked. Fix the RED phase — only an intentional failure of target test "${result.evidence.target_test}" authorizes production edits.`,
|
||||
},
|
||||
raw,
|
||||
undefined,
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a phase argument to an absolute phase directory, or '' when it
|
||||
* cannot be resolved. Shared by every `check` arm that probes a phase's
|
||||
@@ -1701,6 +1779,12 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void {
|
||||
cmdTddReviewCheckpoint(cwd, args, raw);
|
||||
return;
|
||||
}
|
||||
if (subcommand === 'tdd-red-evidence') {
|
||||
// #3770: intentional-RED evidence gate — only a target-test failure may
|
||||
// authorize GREEN. Validates the persisted record; never executes anything.
|
||||
cmdTddRedEvidence(cwd, args, raw);
|
||||
return;
|
||||
}
|
||||
if (subcommand === 'ui-safety-gate') {
|
||||
cmdUiSafetyGate(cwd, args, raw);
|
||||
return;
|
||||
@@ -1744,7 +1828,7 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void {
|
||||
routeProhibitionEnforcement(args, raw);
|
||||
return;
|
||||
}
|
||||
error('Unknown check subcommand. Available: api-coverage-verify-pre, auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, predicate, prohibition-enforcement, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-command-paths, verify-failure-directions, verify-schema-drift, verify-codebase-drift, verify-context-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND);
|
||||
error('Unknown check subcommand. Available: api-coverage-verify-pre, auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, predicate, prohibition-enforcement, tdd-red-evidence, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-command-paths, verify-failure-directions, verify-schema-drift, verify-codebase-drift, verify-context-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND);
|
||||
}
|
||||
|
||||
export = {
|
||||
@@ -1757,6 +1841,7 @@ export = {
|
||||
cmdVerifyCommandPaths,
|
||||
cmdVerifyFailureDirections,
|
||||
cmdTddReviewCheckpoint,
|
||||
cmdTddRedEvidence,
|
||||
cmdCheckPredicate,
|
||||
buildPredicateDeps,
|
||||
parsePredicateFlags,
|
||||
|
||||
198
src/tdd-red-evidence.cts
Normal file
198
src/tdd-red-evidence.cts
Normal file
@@ -0,0 +1,198 @@
|
||||
/**
|
||||
* TDD RED-evidence classification (#3770).
|
||||
*
|
||||
* The `type: tdd` executor gate previously accepted ANY nonzero test command as
|
||||
* RED: syntax errors, zero-test discovery, fixture crashes, parser errors, and
|
||||
* unrelated assertions all authorized production edits (GREEN). This module
|
||||
* defines the compact RED evidence the gate now requires — the TARGET test's
|
||||
* identity plus a matching assertion failure — and classifies a persisted test
|
||||
* run into exactly one verdict:
|
||||
*
|
||||
* RED_EVIDENCE_OK — nonzero exit AND the target test failed as a REAL test
|
||||
* (distinctly named, TAP-reported failure). The ONLY
|
||||
* verdict that may advance to GREEN.
|
||||
* INVALID_RED — everything else, with a machine-readable reason:
|
||||
* unexpected_green | zero_tests_discovered |
|
||||
* nonzero_exit_without_test_failure |
|
||||
* fixture_or_load_failure | no_target_test_failure |
|
||||
* invalid_record | unreadable_record (router arm).
|
||||
*
|
||||
* Everything here is PURE — no fs, no spawn, no clock — so the executor can
|
||||
* persist the record (command, exit code, failing test, expected, actual) and
|
||||
* validate it via `gsd_run check tdd-red-evidence <record.json>`.
|
||||
*
|
||||
* TAP parsing reuses the proven primitives from prohibition-enforcement
|
||||
* (`parseNodeTestSummary`, `tapFailedTestNames`) — the same contract the
|
||||
* prohibition probe's fail-first prover already relies on (#1259).
|
||||
*/
|
||||
|
||||
import { parseNodeTestSummary, tapFailedTestNames } from './prohibition-enforcement.cjs';
|
||||
|
||||
export type RedEvidenceVerdict = 'RED_EVIDENCE_OK' | 'INVALID_RED';
|
||||
|
||||
export type RedEvidenceReason =
|
||||
| 'target_test_failed'
|
||||
| 'unexpected_green'
|
||||
| 'zero_tests_discovered'
|
||||
| 'nonzero_exit_without_test_failure'
|
||||
| 'fixture_or_load_failure'
|
||||
| 'no_target_test_failure'
|
||||
| 'invalid_record'
|
||||
| 'unreadable_record';
|
||||
|
||||
/** The raw run record the executor persists after the RED-phase test command. */
|
||||
export interface RedEvidenceInput {
|
||||
/** The exact test command that was run (persisted verbatim). */
|
||||
command: unknown;
|
||||
/** The command's exit code. */
|
||||
exitCode: unknown;
|
||||
/** The command's combined stdout (TAP for node --test). */
|
||||
output: unknown;
|
||||
/** Identity of the target test named by the plan (its `test('...')` name). */
|
||||
targetTest: unknown;
|
||||
/** Path of the test file the target test lives in (file-named failures are crashes). */
|
||||
targetFile?: unknown;
|
||||
/** Expected result stated by the plan's <behavior> (persisted verbatim). */
|
||||
expected?: unknown;
|
||||
/** Actual result observed in the failing assertion (persisted verbatim). */
|
||||
actual?: unknown;
|
||||
}
|
||||
|
||||
/** The classification verdict plus the compact evidence it was decided on. */
|
||||
export interface RedEvidenceResult {
|
||||
verdict: RedEvidenceVerdict;
|
||||
reason: RedEvidenceReason;
|
||||
evidence: {
|
||||
command: string;
|
||||
exit_code: number | null;
|
||||
target_test: string;
|
||||
tests: number;
|
||||
pass: number;
|
||||
fail: number;
|
||||
failing_tests: string[];
|
||||
};
|
||||
}
|
||||
|
||||
/** The persisted RED evidence record (acceptance: command, exit code, failing test, expected, actual). */
|
||||
export interface RedEvidenceRecord {
|
||||
command: string;
|
||||
exit_code: number | null;
|
||||
failing_test: string | null;
|
||||
target_test: string;
|
||||
expected: string | null;
|
||||
actual: string | null;
|
||||
verdict: RedEvidenceVerdict;
|
||||
reason: RedEvidenceReason;
|
||||
}
|
||||
|
||||
/** Basename of a path-like string ('' for non-strings) — separators `/` and `\`. */
|
||||
function baseOf(p: unknown): string {
|
||||
return typeof p === 'string' ? (p.split(/[\\/]/).pop() ?? p) : '';
|
||||
}
|
||||
|
||||
/** Coerce and validate the raw record's scalar fields. Returns null exit_code only when absent/non-numeric. */
|
||||
function readInput(input: RedEvidenceInput): {
|
||||
command: string;
|
||||
exitCode: number | null;
|
||||
output: string;
|
||||
targetTest: string;
|
||||
} | null {
|
||||
const command = typeof input?.command === 'string' ? input.command : '';
|
||||
const output = typeof input?.output === 'string' ? input.output : '';
|
||||
const targetTest = typeof input?.targetTest === 'string' ? input.targetTest.trim() : '';
|
||||
const exitCode =
|
||||
typeof input?.exitCode === 'number' && Number.isFinite(input.exitCode) ? input.exitCode : null;
|
||||
if (!command || !targetTest || exitCode === null) return null;
|
||||
return { command, exitCode, output, targetTest };
|
||||
}
|
||||
|
||||
/**
|
||||
* Classify a persisted RED-phase test run. Fail-closed: malformed input, an
|
||||
* unparseable/empty TAP summary, a file-named (load/crash) failure, or a
|
||||
* failure that is not the target test's are all INVALID_RED — only a nonzero
|
||||
* exit WITH the distinctly-named target test failing is RED_EVIDENCE_OK.
|
||||
* Never throws.
|
||||
*/
|
||||
export function classifyRedEvidence(input: RedEvidenceInput): RedEvidenceResult {
|
||||
const parsed = readInput(input);
|
||||
if (!parsed) {
|
||||
return {
|
||||
verdict: 'INVALID_RED',
|
||||
reason: 'invalid_record',
|
||||
evidence: {
|
||||
command: typeof input?.command === 'string' ? input.command : '',
|
||||
exit_code: null,
|
||||
target_test: '',
|
||||
tests: 0,
|
||||
pass: 0,
|
||||
fail: 0,
|
||||
failing_tests: [],
|
||||
},
|
||||
};
|
||||
}
|
||||
const { command, exitCode, output, targetTest } = parsed;
|
||||
const summary = parseNodeTestSummary(output);
|
||||
const failing = tapFailedTestNames(output);
|
||||
const evidence = {
|
||||
command,
|
||||
exit_code: exitCode,
|
||||
target_test: targetTest,
|
||||
tests: summary.tests,
|
||||
pass: summary.pass,
|
||||
fail: summary.fail,
|
||||
failing_tests: failing,
|
||||
};
|
||||
|
||||
// Existing fail-fast rule, now machine-checked: exit 0 during RED is an
|
||||
// unexpected GREEN — the feature may already exist or the test is wrong.
|
||||
if (exitCode === 0) {
|
||||
return { verdict: 'INVALID_RED', reason: 'unexpected_green', evidence };
|
||||
}
|
||||
// Zero-test discovery: the discovery pattern / fixture matched no tests.
|
||||
// A run that executed nothing cannot prove anything about the behavior.
|
||||
if (summary.tests === 0) {
|
||||
return { verdict: 'INVALID_RED', reason: 'zero_tests_discovered', evidence };
|
||||
}
|
||||
// Nonzero exit but TAP reports no failing test: harness/setup/parser crash
|
||||
// whose failure never reached a test assertion (or unparseable output).
|
||||
if (summary.fail === 0 || failing.length === 0) {
|
||||
return { verdict: 'INVALID_RED', reason: 'nonzero_exit_without_test_failure', evidence };
|
||||
}
|
||||
// Fixture/load failure: every failing entry is named like the target FILE —
|
||||
// node reports a load-time crash (throw-on-require, syntax error, ENOENT
|
||||
// fixture) as a file-named `not ok 1 - <file>`, never the target test.
|
||||
const targetBase = baseOf(input?.targetFile ?? '');
|
||||
const distinctlyNamed = failing.filter((n) => (targetBase ? baseOf(n) !== targetBase : true));
|
||||
if (distinctlyNamed.length === 0) {
|
||||
return { verdict: 'INVALID_RED', reason: 'fixture_or_load_failure', evidence };
|
||||
}
|
||||
// Unrelated failure: real tests ran and failed, but none is the target test
|
||||
// the plan named — an unrelated assertion must not authorize GREEN.
|
||||
if (!distinctlyNamed.includes(targetTest)) {
|
||||
return { verdict: 'INVALID_RED', reason: 'no_target_test_failure', evidence };
|
||||
}
|
||||
return { verdict: 'RED_EVIDENCE_OK', reason: 'target_test_failed', evidence };
|
||||
}
|
||||
|
||||
/**
|
||||
* Project a classification into the persisted record shape — command, exit
|
||||
* code, failing test, expected, actual, verdict, reason — so the evidence
|
||||
* survives past the terminal and the gate can re-verify it deterministically.
|
||||
* Pure: JSON-serializable, no timestamps (the record's mtime/commit carries time).
|
||||
*/
|
||||
export function buildRedEvidenceRecord(input: RedEvidenceInput, result: RedEvidenceResult): RedEvidenceRecord {
|
||||
const failingTest =
|
||||
result.evidence.failing_tests.find((n) => n === result.evidence.target_test) ??
|
||||
result.evidence.failing_tests[0] ??
|
||||
null;
|
||||
return {
|
||||
command: result.evidence.command,
|
||||
exit_code: result.evidence.exit_code,
|
||||
failing_test: failingTest,
|
||||
target_test: result.evidence.target_test,
|
||||
expected: typeof input?.expected === 'string' ? input.expected : null,
|
||||
actual: typeof input?.actual === 'string' ? input.actual : null,
|
||||
verdict: result.verdict,
|
||||
reason: result.reason,
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user