fix(#3770): require intentional RED evidence before GREEN (#4279)

* test(3770): add failing tests for intentional RED evidence gate

RED: classifyRedEvidence / buildRedEvidenceRecord / check tdd-red-evidence do
not exist yet; every row fails on require. Per #3770 only an intentional
target-test failure may authorize GREEN; zero-test discovery, fixture crashes,
unrelated failures, and unexpected green are INVALID_RED.

* fix(3770): require intentional RED evidence before GREEN

Only an intentional failure of the TARGET test (distinctly named, TAP-reported
assertion failure) classifies as RED_EVIDENCE_OK and authorizes GREEN. Zero-test
discovery, fixture/load crashes (file-named failures), nonzero exits without a
failing test, unrelated failures, unexpected greens, and malformed/missing
records are INVALID_RED and block GREEN.

- src/tdd-red-evidence.cts: pure classifier + persisted record builder (reuses
  the prohibition-enforcement TAP primitives; fail-closed, never throws)
- check tdd-red-evidence <record.json>: validates the persisted record
  (command, exit code, failing test, expected, actual)
- gsd-executor.md / references/tdd.md / references/execute-mvp-tdd.md: RED now
  requires the evidence record + gate verdict, not a nonzero exit or a RED: tag

* chore(3770): regenerate inventory manifest for tdd-red-evidence.cjs

* fix(3770): fit executor fail-fast under size cap, fix unrelated-failure fixture, ignore generated lib

- gsd-executor.md: compress the #3770 fail-fast rule to one line (49149 B <
  49152 cap; line-count parity keeps the #2751 PROSE_ALLOWLIST line 816 valid)
- tests: the row-6 fixture used String.replace (first-occurrence), so the
  `not ok` line still named the target test and the classifier was right to
  accept it; replaceAll makes the failure genuinely unrelated
- eslint.config.mjs: ignore tsc-generated bin/lib/tdd-red-evidence.cjs
  (lint the src/*.cts source, per ADR-457 migration rule)

Emitted-Drift-Ack-Growth: gsd-executor.md — the #3770 fail-fast rule now requires intentional RED evidence (check tdd-red-evidence) before GREEN; +172 bytes, kept under the LARGE cap and on one line

* chore(3770): add changeset

* chore(3770): backfill PR number in changeset

---------

Co-authored-by: sim <sim@local>
This commit is contained in:
Tom Boucher
2026-09-04 16:44:38 -04:00
committed by GitHub
parent 2f4f7538e9
commit 8249ebcf6e
10 changed files with 560 additions and 14 deletions

View File

@@ -37,6 +37,7 @@ const { getRoadmapPhaseWithFallback } = roadmapModule;
import gapCheckerModule = require('./gap-checker.cjs');
const { runGapAnalysis } = gapCheckerModule;
import { routeProhibitionEnforcement } from './prohibition-enforcement.cjs';
import { classifyRedEvidence, buildRedEvidenceRecord } from './tdd-red-evidence.cjs';
// eslint-disable-next-line @typescript-eslint/no-require-imports
import gatePredicateEval = require('./gate-predicate-evaluator.cjs');
const { evaluatePredicate } = gatePredicateEval;
@@ -935,6 +936,83 @@ function cmdTddReviewCheckpoint(projectDir: string, args: string[], raw: boolean
output(result, raw, undefined);
}
// ─── tdd-red-evidence (#3770) ──────────────────────────────────────────────────
/**
* tdd-red-evidence: validates a persisted RED-phase test-run record for a
* `type: tdd` plan (#3770). Only an INTENTIONAL failure of the target test
* (verdict RED_EVIDENCE_OK) may authorize GREEN; zero-test discovery, fixture/
* load crashes, nonzero exits without a failing test, unrelated failures, and
* unexpected greens are INVALID_RED and block GREEN.
*
* The record is the JSON the executor persists after running the RED command:
* { command, exitCode, output, targetTest, targetFile?, expected?, actual? }
* Fail-closed: a missing/unreadable/unparseable record is INVALID_RED
* (reason unreadable_record), never a pass.
*
* Args: check tdd-red-evidence <record.json>
*/
function cmdTddRedEvidence(_projectDir: string, args: string[], raw: boolean): void {
const recordPath = typeof args[2] === 'string' ? args[2] : '';
if (!recordPath) {
error('tdd-red-evidence requires a record path: check tdd-red-evidence <record.json>', ERROR_REASON.SDK_MISSING_ARG);
return;
}
const resolved = path.resolve(recordPath);
const text = readIfExists(resolved);
const input = ((): Record<string, unknown> | null => {
if (!text) return null;
try {
return (JSON.parse(text) ?? {}) as Record<string, unknown>;
} catch {
return null;
}
})();
if (!input) {
output(
{
passed: false,
block: true,
verdict: 'INVALID_RED',
reason: 'unreadable_record',
record: resolved,
readError: text ? `record is not valid JSON: ${resolved}` : `record not found or unreadable: ${resolved}`,
},
raw,
undefined,
);
return;
}
const evidenceInput = {
command: input['command'],
exitCode: input['exitCode'],
output: input['output'],
targetTest: input['targetTest'],
targetFile: input['targetFile'],
expected: input['expected'],
actual: input['actual'],
};
const result = classifyRedEvidence(evidenceInput);
const record = buildRedEvidenceRecord(evidenceInput, result);
output(
{
// Uniform gate contract: block = !passed. INVALID_RED blocks GREEN.
passed: result.verdict === 'RED_EVIDENCE_OK',
block: result.verdict !== 'RED_EVIDENCE_OK',
verdict: result.verdict,
reason: result.reason,
evidence: result.evidence,
record,
message:
result.verdict === 'RED_EVIDENCE_OK'
? `RED evidence verified: target test "${result.evidence.target_test}" failed as expected (exit ${result.evidence.exit_code}). GREEN authorized.`
: `INVALID_RED (${result.reason}): GREEN blocked. Fix the RED phase — only an intentional failure of target test "${result.evidence.target_test}" authorizes production edits.`,
},
raw,
undefined,
);
}
/**
* Resolve a phase argument to an absolute phase directory, or '' when it
* cannot be resolved. Shared by every `check` arm that probes a phase's
@@ -1701,6 +1779,12 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void {
cmdTddReviewCheckpoint(cwd, args, raw);
return;
}
if (subcommand === 'tdd-red-evidence') {
// #3770: intentional-RED evidence gate — only a target-test failure may
// authorize GREEN. Validates the persisted record; never executes anything.
cmdTddRedEvidence(cwd, args, raw);
return;
}
if (subcommand === 'ui-safety-gate') {
cmdUiSafetyGate(cwd, args, raw);
return;
@@ -1744,7 +1828,7 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void {
routeProhibitionEnforcement(args, raw);
return;
}
error('Unknown check subcommand. Available: api-coverage-verify-pre, auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, predicate, prohibition-enforcement, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-command-paths, verify-failure-directions, verify-schema-drift, verify-codebase-drift, verify-context-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND);
error('Unknown check subcommand. Available: api-coverage-verify-pre, auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, predicate, prohibition-enforcement, tdd-red-evidence, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-command-paths, verify-failure-directions, verify-schema-drift, verify-codebase-drift, verify-context-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND);
}
export = {
@@ -1757,6 +1841,7 @@ export = {
cmdVerifyCommandPaths,
cmdVerifyFailureDirections,
cmdTddReviewCheckpoint,
cmdTddRedEvidence,
cmdCheckPredicate,
buildPredicateDeps,
parsePredicateFlags,

198
src/tdd-red-evidence.cts Normal file
View File

@@ -0,0 +1,198 @@
/**
* TDD RED-evidence classification (#3770).
*
* The `type: tdd` executor gate previously accepted ANY nonzero test command as
* RED: syntax errors, zero-test discovery, fixture crashes, parser errors, and
* unrelated assertions all authorized production edits (GREEN). This module
* defines the compact RED evidence the gate now requires — the TARGET test's
* identity plus a matching assertion failure — and classifies a persisted test
* run into exactly one verdict:
*
* RED_EVIDENCE_OK — nonzero exit AND the target test failed as a REAL test
* (distinctly named, TAP-reported failure). The ONLY
* verdict that may advance to GREEN.
* INVALID_RED — everything else, with a machine-readable reason:
* unexpected_green | zero_tests_discovered |
* nonzero_exit_without_test_failure |
* fixture_or_load_failure | no_target_test_failure |
* invalid_record | unreadable_record (router arm).
*
* Everything here is PURE — no fs, no spawn, no clock — so the executor can
* persist the record (command, exit code, failing test, expected, actual) and
* validate it via `gsd_run check tdd-red-evidence <record.json>`.
*
* TAP parsing reuses the proven primitives from prohibition-enforcement
* (`parseNodeTestSummary`, `tapFailedTestNames`) — the same contract the
* prohibition probe's fail-first prover already relies on (#1259).
*/
import { parseNodeTestSummary, tapFailedTestNames } from './prohibition-enforcement.cjs';
export type RedEvidenceVerdict = 'RED_EVIDENCE_OK' | 'INVALID_RED';
export type RedEvidenceReason =
| 'target_test_failed'
| 'unexpected_green'
| 'zero_tests_discovered'
| 'nonzero_exit_without_test_failure'
| 'fixture_or_load_failure'
| 'no_target_test_failure'
| 'invalid_record'
| 'unreadable_record';
/** The raw run record the executor persists after the RED-phase test command. */
export interface RedEvidenceInput {
/** The exact test command that was run (persisted verbatim). */
command: unknown;
/** The command's exit code. */
exitCode: unknown;
/** The command's combined stdout (TAP for node --test). */
output: unknown;
/** Identity of the target test named by the plan (its `test('...')` name). */
targetTest: unknown;
/** Path of the test file the target test lives in (file-named failures are crashes). */
targetFile?: unknown;
/** Expected result stated by the plan's <behavior> (persisted verbatim). */
expected?: unknown;
/** Actual result observed in the failing assertion (persisted verbatim). */
actual?: unknown;
}
/** The classification verdict plus the compact evidence it was decided on. */
export interface RedEvidenceResult {
verdict: RedEvidenceVerdict;
reason: RedEvidenceReason;
evidence: {
command: string;
exit_code: number | null;
target_test: string;
tests: number;
pass: number;
fail: number;
failing_tests: string[];
};
}
/** The persisted RED evidence record (acceptance: command, exit code, failing test, expected, actual). */
export interface RedEvidenceRecord {
command: string;
exit_code: number | null;
failing_test: string | null;
target_test: string;
expected: string | null;
actual: string | null;
verdict: RedEvidenceVerdict;
reason: RedEvidenceReason;
}
/** Basename of a path-like string ('' for non-strings) — separators `/` and `\`. */
function baseOf(p: unknown): string {
return typeof p === 'string' ? (p.split(/[\\/]/).pop() ?? p) : '';
}
/** Coerce and validate the raw record's scalar fields. Returns null exit_code only when absent/non-numeric. */
function readInput(input: RedEvidenceInput): {
command: string;
exitCode: number | null;
output: string;
targetTest: string;
} | null {
const command = typeof input?.command === 'string' ? input.command : '';
const output = typeof input?.output === 'string' ? input.output : '';
const targetTest = typeof input?.targetTest === 'string' ? input.targetTest.trim() : '';
const exitCode =
typeof input?.exitCode === 'number' && Number.isFinite(input.exitCode) ? input.exitCode : null;
if (!command || !targetTest || exitCode === null) return null;
return { command, exitCode, output, targetTest };
}
/**
* Classify a persisted RED-phase test run. Fail-closed: malformed input, an
* unparseable/empty TAP summary, a file-named (load/crash) failure, or a
* failure that is not the target test's are all INVALID_RED — only a nonzero
* exit WITH the distinctly-named target test failing is RED_EVIDENCE_OK.
* Never throws.
*/
export function classifyRedEvidence(input: RedEvidenceInput): RedEvidenceResult {
const parsed = readInput(input);
if (!parsed) {
return {
verdict: 'INVALID_RED',
reason: 'invalid_record',
evidence: {
command: typeof input?.command === 'string' ? input.command : '',
exit_code: null,
target_test: '',
tests: 0,
pass: 0,
fail: 0,
failing_tests: [],
},
};
}
const { command, exitCode, output, targetTest } = parsed;
const summary = parseNodeTestSummary(output);
const failing = tapFailedTestNames(output);
const evidence = {
command,
exit_code: exitCode,
target_test: targetTest,
tests: summary.tests,
pass: summary.pass,
fail: summary.fail,
failing_tests: failing,
};
// Existing fail-fast rule, now machine-checked: exit 0 during RED is an
// unexpected GREEN — the feature may already exist or the test is wrong.
if (exitCode === 0) {
return { verdict: 'INVALID_RED', reason: 'unexpected_green', evidence };
}
// Zero-test discovery: the discovery pattern / fixture matched no tests.
// A run that executed nothing cannot prove anything about the behavior.
if (summary.tests === 0) {
return { verdict: 'INVALID_RED', reason: 'zero_tests_discovered', evidence };
}
// Nonzero exit but TAP reports no failing test: harness/setup/parser crash
// whose failure never reached a test assertion (or unparseable output).
if (summary.fail === 0 || failing.length === 0) {
return { verdict: 'INVALID_RED', reason: 'nonzero_exit_without_test_failure', evidence };
}
// Fixture/load failure: every failing entry is named like the target FILE —
// node reports a load-time crash (throw-on-require, syntax error, ENOENT
// fixture) as a file-named `not ok 1 - <file>`, never the target test.
const targetBase = baseOf(input?.targetFile ?? '');
const distinctlyNamed = failing.filter((n) => (targetBase ? baseOf(n) !== targetBase : true));
if (distinctlyNamed.length === 0) {
return { verdict: 'INVALID_RED', reason: 'fixture_or_load_failure', evidence };
}
// Unrelated failure: real tests ran and failed, but none is the target test
// the plan named — an unrelated assertion must not authorize GREEN.
if (!distinctlyNamed.includes(targetTest)) {
return { verdict: 'INVALID_RED', reason: 'no_target_test_failure', evidence };
}
return { verdict: 'RED_EVIDENCE_OK', reason: 'target_test_failed', evidence };
}
/**
* Project a classification into the persisted record shape — command, exit
* code, failing test, expected, actual, verdict, reason — so the evidence
* survives past the terminal and the gate can re-verify it deterministically.
* Pure: JSON-serializable, no timestamps (the record's mtime/commit carries time).
*/
export function buildRedEvidenceRecord(input: RedEvidenceInput, result: RedEvidenceResult): RedEvidenceRecord {
const failingTest =
result.evidence.failing_tests.find((n) => n === result.evidence.target_test) ??
result.evidence.failing_tests[0] ??
null;
return {
command: result.evidence.command,
exit_code: result.evidence.exit_code,
failing_test: failingTest,
target_test: result.evidence.target_test,
expected: typeof input?.expected === 'string' ? input.expected : null,
actual: typeof input?.actual === 'string' ? input.actual : null,
verdict: result.verdict,
reason: result.reason,
};
}