diff --git a/.changeset/steady-birds-hum.md b/.changeset/steady-birds-hum.md new file mode 100644 index 000000000..e2817b6d1 --- /dev/null +++ b/.changeset/steady-birds-hum.md @@ -0,0 +1,5 @@ +--- +type: Fixed +pr: 4279 +--- +**TDD executor now requires intentional RED evidence before GREEN** — a RED-phase test command that exits nonzero no longer authorizes production edits unless the persisted evidence record shows the TARGET test failing a real assertion. Syntax errors, zero-test discovery, fixture crashes, parser errors, and unrelated assertions classify as INVALID_RED and block GREEN. (#3770) diff --git a/agents/gsd-executor.md b/agents/gsd-executor.md index fb27422a0..7f84f769f 100644 --- a/agents/gsd-executor.md +++ b/agents/gsd-executor.md @@ -408,7 +408,7 @@ reference is the single source; do not improvise a variant. When the plan frontmatter has `type: tdd`, the entire plan follows the RED/GREEN/REFACTOR cycle as a single feature. Gate sequence is mandatory: -**Fail-fast rule:** If a test passes unexpectedly during the RED phase (before any implementation), STOP. The feature may already exist or the test is not testing what you think. Investigate and fix the test before proceeding to GREEN. Do NOT skip RED by proceeding with a passing test. +**Fail-fast rules (#3770):** If a test passes unexpectedly during RED, STOP — do NOT skip RED. A nonzero exit alone is NOT RED either: persist the RED evidence (command, exit code, failing test, expected, actual) and verify with `gsd_run check tdd-red-evidence ` — only `RED_EVIDENCE_OK` (the TARGET test failed an assertion for the behavior) authorizes GREEN; any INVALID_RED (zero tests, fixture crash, parser error, wrong test) blocks it. **Gate sequence validation:** After completing the plan, verify in git log: 1. A `test(...)` commit exists (RED gate) diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index 5c8e69972..c0db5d56e 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -526,6 +526,7 @@ "surface.cjs", "task-command-router.cjs", "task-content-resolution.cjs", + "tdd-red-evidence.cjs", "teams-status.cjs", "template.cjs", "text-lines.cjs", diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index 976aa5995..0dc618e5e 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -487,6 +487,7 @@ Full listing: `gsd-core/bin/lib/*.cjs`. | `broken-windows.cjs` | Broken-windows ledger library (issue #1950) — typed IR + I/O for `.planning/WINDOWS.md` (cross-phase defect register); pure `parseLedger`/`renderLedger`/`appendWindow`/`markWaived`/`markFixed`/`openCount` + I/O `cmdWindowsStatus`/`cmdWindowsAppend`/`cmdWindowsWaive`/`cmdWindowsMarkFixed`; frozen `REASON` enum for typed-error assertions; CLI surface `gsd-tools windows status\|append\|waive\|fixed`. Generated from `src/broken-windows.cts` | | `capability-writer.cjs` | Capability State Writer (ADR-1213) — write-side inverse of the resolver; projects desired per-capability enabled/gates onto surface + config substrates, then re-resolves (assert-and-report); exports `setCapabilityState` and I/O handler `cmdCapabilitySet`; command surface: `gsd-tools capability set [--on\|--off] [--gate =]` | | `check-command-router.cjs` | Thin CJS subcommand router adapter for `gsd-tools check` | +| `tdd-red-evidence.cjs` | TDD RED-evidence classifier (issue #3770, compiled from `src/tdd-red-evidence.cts`, gitignored) — pure `classifyRedEvidence`/`buildRedEvidenceRecord`: only an intentional failure of the TARGET test (distinctly named, TAP-reported assertion failure with nonzero exit) is `RED_EVIDENCE_OK`; zero-test discovery, fixture/load crashes (file-named failures), nonzero exits without a failing test, unrelated failures, unexpected greens, and malformed records are `INVALID_RED` (fail-closed, never throws); CLI surface `gsd_run check tdd-red-evidence ` validates the persisted record (command, exit code, failing test, expected, actual) | | `claude-orchestration-command-router.cjs` | ADR-959 capability command router for Claude orchestration — workflow-backend detection and emission (#1143) | | `claude-orchestration.cjs` | Claude Orchestration capability (#1143) — Workflow-tool backend detection + emitter; `detectWorkflowBackend` fail-closed gate (`{available, backend: 'workflow'\|'inline', reason}`, degrades to today's inline behavior unless every gate opens) and `emitWorkflowScript` (maps GSD's wave/plan model onto Workflow primitives: wave → sequential `parallel()` barriers, plan → `agent(...)` with per-plan worktree isolation mirroring the inline path). Pure, zero external dependencies, never throws; never invokes the Workflow tool itself | | `cli-exit.cjs` | `ExitError` class and `runMain()` helper — CLI entrypoints throw `ExitError` instead of calling `process.exit()`; `runMain()` translates the outcome into `process.exitCode` so output flushes cleanly | diff --git a/eslint.config.mjs b/eslint.config.mjs index a66279160..7b7435964 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -119,6 +119,8 @@ export default tseslint.config( 'gsd-core/bin/lib/probe-core.cjs', 'gsd-core/bin/lib/spec-section.cjs', 'gsd-core/bin/lib/prohibition-enforcement.cjs', + // #3770: tsc-generated runtime artifact — lint the src/tdd-red-evidence.cts source. + 'gsd-core/bin/lib/tdd-red-evidence.cjs', 'gsd-core/bin/lib/ui-consideration-probe.cjs', 'gsd-core/bin/lib/code-review-flags.cjs', 'gsd-core/bin/lib/code-review-depth.cjs', diff --git a/gsd-core/references/execute-mvp-tdd.md b/gsd-core/references/execute-mvp-tdd.md index 25bd1118a..4ffcdf797 100644 --- a/gsd-core/references/execute-mvp-tdd.md +++ b/gsd-core/references/execute-mvp-tdd.md @@ -15,12 +15,15 @@ If any of these is false, the gate is inactive — execution proceeds normally. For each task gated by TDD, the executor MUST verify (before running the implementation step): 1. **A failing-test commit exists.** Search git log on the current branch for a commit matching `test({phase}-{plan})` whose subject mentions the same plan as the current task. The commit must touch a test file (`*.test.*`, `*.spec.*`, `tests/**`). -2. **The test was actually red.** The commit message body or the executor's recent shell history must show the test failed when first run. Acceptable evidence: - - Commit message contains `RED:` prefix or `(RED)` tag - - Recent terminal output shows `FAIL` or non-zero exit on the new test before any implementation commit +2. **The test was actually red — INTENTIONALLY (#3770).** A nonzero exit is not RED by itself: syntax errors, zero-test discovery, fixture crashes, parser errors, and unrelated assertions are INVALID_RED. The executor must persist the RED evidence record (command, exit code, failing test, expected result from ``, actual result) and verify it: + ```bash + gsd_run check tdd-red-evidence --raw + ``` + - `RED_EVIDENCE_OK` (reason `target_test_failed`): the TARGET test named by the plan failed on a real assertion — the ONLY verdict that authorizes GREEN. + - `INVALID_RED` (reasons `unexpected_green`, `zero_tests_discovered`, `nonzero_exit_without_test_failure`, `fixture_or_load_failure`, `no_target_test_failure`, `invalid_record`, `unreadable_record`): the gate trips — halt, fix the RED phase (test identity, fixture, discovery), and re-verify before any implementation step. A `RED:` prefix or `(RED)` tag in the commit message is NOT sufficient evidence on its own. 3. **No implementation commit yet.** No `feat({phase}-{plan})` commit may exist for the same plan ID before the failing-test commit. -If any check fails, the gate trips. +If any check fails, the gate trips. For check 2, an INVALID_RED verdict (`check tdd-red-evidence`) trips the gate — the executor MUST halt and block the implementation step. ## What "behavior-adding task" means @@ -41,7 +44,7 @@ The executor MUST: ``` ### TDD GATE TRIPPED — Plan {plan_id}, Task {task_id} - Reason: {missing_red_commit | red_commit_not_failing | feat_before_test} + Reason: {missing_red_commit | red_commit_not_failing | feat_before_test | invalid_red} Behavior expected to be tested: - {first behavior bullet} @@ -69,7 +72,7 @@ The `--force-mvp-gate` flag is documented but not introduced by this plan — it ## What this gate does NOT do - It does not enforce REFACTOR commits. REFACTOR remains optional (per `gsd-core/references/tdd.md`). -- It does not check test quality (the test could be trivially passing). That's the planner's job. +- It does not check test quality (the test could be trivially weak). That's the planner's job. It DOES check that the RED failure was intentional — the target test failing an assertion (#3770). - It does not run tests. The executor only inspects git log + file system. Running tests is the implementation step's job. - It does not gate config-only or doc-only tasks (see "behavior-adding task" definition). diff --git a/gsd-core/references/tdd.md b/gsd-core/references/tdd.md index 8e41e23df..0411aec4d 100644 --- a/gsd-core/references/tdd.md +++ b/gsd-core/references/tdd.md @@ -94,9 +94,10 @@ After completion, create SUMMARY.md with: **RED - Write failing test:** 1. Create test file following project conventions 2. Write test describing expected behavior (from `` element) -3. Run test - it MUST fail -4. If test passes: feature exists or test is wrong. Investigate. -5. Commit: `test({phase}-{plan}): add failing test for [feature]` +3. Run test - it MUST fail **intentionally** (#3770): the TARGET test you named must be the test that fails, on an assertion for the planned behavior. A nonzero exit alone is NOT RED — syntax errors, zero-test discovery, fixture crashes, parser errors, and unrelated assertions are INVALID_RED and must not authorize GREEN. +4. Persist the RED evidence record (command, exit code, failing test, expected result, actual result) and verify it: `gsd_run check tdd-red-evidence `. Only verdict `RED_EVIDENCE_OK` satisfies the RED gate; `INVALID_RED` blocks GREEN until the RED phase is fixed. +5. If test passes: feature exists or test is wrong. Investigate. +6. Commit: `test({phase}-{plan}): add failing test for [feature]` **GREEN - Implement to pass:** 1. Write minimal code to make test pass @@ -256,15 +257,16 @@ When `workflow.tdd_mode` is enabled in config, the RED/GREEN/REFACTOR gate seque | Gate | Required | Commit Pattern | Validation | |------|----------|---------------|------------| -| RED | Yes | `test({phase}-{plan}): ...` | Test exists AND fails before implementation | +| RED | Yes | `test({phase}-{plan}): ...` | Test exists AND fails before implementation — intentionally: `check tdd-red-evidence` returns `RED_EVIDENCE_OK` (target test failed on an assertion for the behavior; anything else is INVALID_RED) | | GREEN | Yes | `feat({phase}-{plan}): ...` | Test passes after implementation | | REFACTOR | No | `refactor({phase}-{plan}): ...` | Tests still pass after cleanup | ### Fail-Fast Rules 1. **Unexpected GREEN in RED phase:** If the test passes before any implementation code is written, STOP. The feature may already exist or the test is wrong. Investigate before proceeding. -2. **Missing RED commit:** If no `test(...)` commit precedes the `feat(...)` commit, the TDD discipline was violated. Flag in SUMMARY.md. -3. **REFACTOR breaks tests:** Undo the refactor immediately. Commit was premature — refactor in smaller steps. +2. **INVALID_RED in RED phase (#3770):** A nonzero exit is not RED by itself. Zero-test discovery, fixture/load crashes, nonzero exits with no failing test, unrelated failing tests, and unexpected greens all classify as INVALID_RED (`gsd_run check tdd-red-evidence`). STOP and fix the RED phase — do NOT proceed to GREEN. +3. **Missing RED commit:** If no `test(...)` commit precedes the `feat(...)` commit, the TDD discipline was violated. Flag in SUMMARY.md. +4. **REFACTOR breaks tests:** Undo the refactor immediately. Commit was premature — refactor in smaller steps. ### Executor Gate Validation diff --git a/src/check-command-router.cts b/src/check-command-router.cts index 13e8c9914..c7d79b337 100644 --- a/src/check-command-router.cts +++ b/src/check-command-router.cts @@ -37,6 +37,7 @@ const { getRoadmapPhaseWithFallback } = roadmapModule; import gapCheckerModule = require('./gap-checker.cjs'); const { runGapAnalysis } = gapCheckerModule; import { routeProhibitionEnforcement } from './prohibition-enforcement.cjs'; +import { classifyRedEvidence, buildRedEvidenceRecord } from './tdd-red-evidence.cjs'; // eslint-disable-next-line @typescript-eslint/no-require-imports import gatePredicateEval = require('./gate-predicate-evaluator.cjs'); const { evaluatePredicate } = gatePredicateEval; @@ -935,6 +936,83 @@ function cmdTddReviewCheckpoint(projectDir: string, args: string[], raw: boolean output(result, raw, undefined); } +// ─── tdd-red-evidence (#3770) ────────────────────────────────────────────────── + +/** + * tdd-red-evidence: validates a persisted RED-phase test-run record for a + * `type: tdd` plan (#3770). Only an INTENTIONAL failure of the target test + * (verdict RED_EVIDENCE_OK) may authorize GREEN; zero-test discovery, fixture/ + * load crashes, nonzero exits without a failing test, unrelated failures, and + * unexpected greens are INVALID_RED and block GREEN. + * + * The record is the JSON the executor persists after running the RED command: + * { command, exitCode, output, targetTest, targetFile?, expected?, actual? } + * Fail-closed: a missing/unreadable/unparseable record is INVALID_RED + * (reason unreadable_record), never a pass. + * + * Args: check tdd-red-evidence + */ +function cmdTddRedEvidence(_projectDir: string, args: string[], raw: boolean): void { + const recordPath = typeof args[2] === 'string' ? args[2] : ''; + if (!recordPath) { + error('tdd-red-evidence requires a record path: check tdd-red-evidence ', ERROR_REASON.SDK_MISSING_ARG); + return; + } + const resolved = path.resolve(recordPath); + const text = readIfExists(resolved); + const input = ((): Record | null => { + if (!text) return null; + try { + return (JSON.parse(text) ?? {}) as Record; + } catch { + return null; + } + })(); + if (!input) { + output( + { + passed: false, + block: true, + verdict: 'INVALID_RED', + reason: 'unreadable_record', + record: resolved, + readError: text ? `record is not valid JSON: ${resolved}` : `record not found or unreadable: ${resolved}`, + }, + raw, + undefined, + ); + return; + } + const evidenceInput = { + command: input['command'], + exitCode: input['exitCode'], + output: input['output'], + targetTest: input['targetTest'], + targetFile: input['targetFile'], + expected: input['expected'], + actual: input['actual'], + }; + const result = classifyRedEvidence(evidenceInput); + const record = buildRedEvidenceRecord(evidenceInput, result); + output( + { + // Uniform gate contract: block = !passed. INVALID_RED blocks GREEN. + passed: result.verdict === 'RED_EVIDENCE_OK', + block: result.verdict !== 'RED_EVIDENCE_OK', + verdict: result.verdict, + reason: result.reason, + evidence: result.evidence, + record, + message: + result.verdict === 'RED_EVIDENCE_OK' + ? `RED evidence verified: target test "${result.evidence.target_test}" failed as expected (exit ${result.evidence.exit_code}). GREEN authorized.` + : `INVALID_RED (${result.reason}): GREEN blocked. Fix the RED phase — only an intentional failure of target test "${result.evidence.target_test}" authorizes production edits.`, + }, + raw, + undefined, + ); +} + /** * Resolve a phase argument to an absolute phase directory, or '' when it * cannot be resolved. Shared by every `check` arm that probes a phase's @@ -1701,6 +1779,12 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void { cmdTddReviewCheckpoint(cwd, args, raw); return; } + if (subcommand === 'tdd-red-evidence') { + // #3770: intentional-RED evidence gate — only a target-test failure may + // authorize GREEN. Validates the persisted record; never executes anything. + cmdTddRedEvidence(cwd, args, raw); + return; + } if (subcommand === 'ui-safety-gate') { cmdUiSafetyGate(cwd, args, raw); return; @@ -1744,7 +1828,7 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void { routeProhibitionEnforcement(args, raw); return; } - error('Unknown check subcommand. Available: api-coverage-verify-pre, auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, predicate, prohibition-enforcement, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-command-paths, verify-failure-directions, verify-schema-drift, verify-codebase-drift, verify-context-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND); + error('Unknown check subcommand. Available: api-coverage-verify-pre, auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, predicate, prohibition-enforcement, tdd-red-evidence, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-command-paths, verify-failure-directions, verify-schema-drift, verify-codebase-drift, verify-context-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND); } export = { @@ -1757,6 +1841,7 @@ export = { cmdVerifyCommandPaths, cmdVerifyFailureDirections, cmdTddReviewCheckpoint, + cmdTddRedEvidence, cmdCheckPredicate, buildPredicateDeps, parsePredicateFlags, diff --git a/src/tdd-red-evidence.cts b/src/tdd-red-evidence.cts new file mode 100644 index 000000000..49372d6f0 --- /dev/null +++ b/src/tdd-red-evidence.cts @@ -0,0 +1,198 @@ +/** + * TDD RED-evidence classification (#3770). + * + * The `type: tdd` executor gate previously accepted ANY nonzero test command as + * RED: syntax errors, zero-test discovery, fixture crashes, parser errors, and + * unrelated assertions all authorized production edits (GREEN). This module + * defines the compact RED evidence the gate now requires — the TARGET test's + * identity plus a matching assertion failure — and classifies a persisted test + * run into exactly one verdict: + * + * RED_EVIDENCE_OK — nonzero exit AND the target test failed as a REAL test + * (distinctly named, TAP-reported failure). The ONLY + * verdict that may advance to GREEN. + * INVALID_RED — everything else, with a machine-readable reason: + * unexpected_green | zero_tests_discovered | + * nonzero_exit_without_test_failure | + * fixture_or_load_failure | no_target_test_failure | + * invalid_record | unreadable_record (router arm). + * + * Everything here is PURE — no fs, no spawn, no clock — so the executor can + * persist the record (command, exit code, failing test, expected, actual) and + * validate it via `gsd_run check tdd-red-evidence `. + * + * TAP parsing reuses the proven primitives from prohibition-enforcement + * (`parseNodeTestSummary`, `tapFailedTestNames`) — the same contract the + * prohibition probe's fail-first prover already relies on (#1259). + */ + +import { parseNodeTestSummary, tapFailedTestNames } from './prohibition-enforcement.cjs'; + +export type RedEvidenceVerdict = 'RED_EVIDENCE_OK' | 'INVALID_RED'; + +export type RedEvidenceReason = + | 'target_test_failed' + | 'unexpected_green' + | 'zero_tests_discovered' + | 'nonzero_exit_without_test_failure' + | 'fixture_or_load_failure' + | 'no_target_test_failure' + | 'invalid_record' + | 'unreadable_record'; + +/** The raw run record the executor persists after the RED-phase test command. */ +export interface RedEvidenceInput { + /** The exact test command that was run (persisted verbatim). */ + command: unknown; + /** The command's exit code. */ + exitCode: unknown; + /** The command's combined stdout (TAP for node --test). */ + output: unknown; + /** Identity of the target test named by the plan (its `test('...')` name). */ + targetTest: unknown; + /** Path of the test file the target test lives in (file-named failures are crashes). */ + targetFile?: unknown; + /** Expected result stated by the plan's (persisted verbatim). */ + expected?: unknown; + /** Actual result observed in the failing assertion (persisted verbatim). */ + actual?: unknown; +} + +/** The classification verdict plus the compact evidence it was decided on. */ +export interface RedEvidenceResult { + verdict: RedEvidenceVerdict; + reason: RedEvidenceReason; + evidence: { + command: string; + exit_code: number | null; + target_test: string; + tests: number; + pass: number; + fail: number; + failing_tests: string[]; + }; +} + +/** The persisted RED evidence record (acceptance: command, exit code, failing test, expected, actual). */ +export interface RedEvidenceRecord { + command: string; + exit_code: number | null; + failing_test: string | null; + target_test: string; + expected: string | null; + actual: string | null; + verdict: RedEvidenceVerdict; + reason: RedEvidenceReason; +} + +/** Basename of a path-like string ('' for non-strings) — separators `/` and `\`. */ +function baseOf(p: unknown): string { + return typeof p === 'string' ? (p.split(/[\\/]/).pop() ?? p) : ''; +} + +/** Coerce and validate the raw record's scalar fields. Returns null exit_code only when absent/non-numeric. */ +function readInput(input: RedEvidenceInput): { + command: string; + exitCode: number | null; + output: string; + targetTest: string; +} | null { + const command = typeof input?.command === 'string' ? input.command : ''; + const output = typeof input?.output === 'string' ? input.output : ''; + const targetTest = typeof input?.targetTest === 'string' ? input.targetTest.trim() : ''; + const exitCode = + typeof input?.exitCode === 'number' && Number.isFinite(input.exitCode) ? input.exitCode : null; + if (!command || !targetTest || exitCode === null) return null; + return { command, exitCode, output, targetTest }; +} + +/** + * Classify a persisted RED-phase test run. Fail-closed: malformed input, an + * unparseable/empty TAP summary, a file-named (load/crash) failure, or a + * failure that is not the target test's are all INVALID_RED — only a nonzero + * exit WITH the distinctly-named target test failing is RED_EVIDENCE_OK. + * Never throws. + */ +export function classifyRedEvidence(input: RedEvidenceInput): RedEvidenceResult { + const parsed = readInput(input); + if (!parsed) { + return { + verdict: 'INVALID_RED', + reason: 'invalid_record', + evidence: { + command: typeof input?.command === 'string' ? input.command : '', + exit_code: null, + target_test: '', + tests: 0, + pass: 0, + fail: 0, + failing_tests: [], + }, + }; + } + const { command, exitCode, output, targetTest } = parsed; + const summary = parseNodeTestSummary(output); + const failing = tapFailedTestNames(output); + const evidence = { + command, + exit_code: exitCode, + target_test: targetTest, + tests: summary.tests, + pass: summary.pass, + fail: summary.fail, + failing_tests: failing, + }; + + // Existing fail-fast rule, now machine-checked: exit 0 during RED is an + // unexpected GREEN — the feature may already exist or the test is wrong. + if (exitCode === 0) { + return { verdict: 'INVALID_RED', reason: 'unexpected_green', evidence }; + } + // Zero-test discovery: the discovery pattern / fixture matched no tests. + // A run that executed nothing cannot prove anything about the behavior. + if (summary.tests === 0) { + return { verdict: 'INVALID_RED', reason: 'zero_tests_discovered', evidence }; + } + // Nonzero exit but TAP reports no failing test: harness/setup/parser crash + // whose failure never reached a test assertion (or unparseable output). + if (summary.fail === 0 || failing.length === 0) { + return { verdict: 'INVALID_RED', reason: 'nonzero_exit_without_test_failure', evidence }; + } + // Fixture/load failure: every failing entry is named like the target FILE — + // node reports a load-time crash (throw-on-require, syntax error, ENOENT + // fixture) as a file-named `not ok 1 - `, never the target test. + const targetBase = baseOf(input?.targetFile ?? ''); + const distinctlyNamed = failing.filter((n) => (targetBase ? baseOf(n) !== targetBase : true)); + if (distinctlyNamed.length === 0) { + return { verdict: 'INVALID_RED', reason: 'fixture_or_load_failure', evidence }; + } + // Unrelated failure: real tests ran and failed, but none is the target test + // the plan named — an unrelated assertion must not authorize GREEN. + if (!distinctlyNamed.includes(targetTest)) { + return { verdict: 'INVALID_RED', reason: 'no_target_test_failure', evidence }; + } + return { verdict: 'RED_EVIDENCE_OK', reason: 'target_test_failed', evidence }; +} + +/** + * Project a classification into the persisted record shape — command, exit + * code, failing test, expected, actual, verdict, reason — so the evidence + * survives past the terminal and the gate can re-verify it deterministically. + * Pure: JSON-serializable, no timestamps (the record's mtime/commit carries time). + */ +export function buildRedEvidenceRecord(input: RedEvidenceInput, result: RedEvidenceResult): RedEvidenceRecord { + const failingTest = + result.evidence.failing_tests.find((n) => n === result.evidence.target_test) ?? + result.evidence.failing_tests[0] ?? + null; + return { + command: result.evidence.command, + exit_code: result.evidence.exit_code, + failing_test: failingTest, + target_test: result.evidence.target_test, + expected: typeof input?.expected === 'string' ? input.expected : null, + actual: typeof input?.actual === 'string' ? input.actual : null, + verdict: result.verdict, + reason: result.reason, + }; +} diff --git a/tests/tdd-red-evidence.test.cjs b/tests/tdd-red-evidence.test.cjs new file mode 100644 index 000000000..1d25fecbc --- /dev/null +++ b/tests/tdd-red-evidence.test.cjs @@ -0,0 +1,249 @@ +'use strict'; + +/** + * RED-evidence classification for `type: tdd` plans (#3770). + * + * Module: gsd-core/bin/lib/tdd-red-evidence.cjs (compiled from src/tdd-red-evidence.cts) + * Router: `check tdd-red-evidence ` in check-command-router.cjs + * + * #3770: the TDD executor accepted ANY nonzero test command as RED. Syntax + * errors, zero-test discovery, fixture crashes, parser errors, and unrelated + * assertions all authorized production edits (GREEN). Only an intentional + * failure of the TARGET test for the planned behavior may advance to GREEN; + * every other nonzero outcome is INVALID_RED. + * + * Row numbers map to .gsd/bug/fix-3770-tdd-red-evidence/50-test-matrix.md. + * TAP fixtures below are captured verbatim from `node --test --test-reporter tap` + * (Node v26) for each failure class. + */ + +const { describe, test, before, after } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const os = require('node:os'); +const path = require('node:path'); + +const { cleanup, runGsdTools } = require('./helpers.cjs'); +const { + classifyRedEvidence, + buildRedEvidenceRecord, +} = require('../gsd-core/bin/lib/tdd-red-evidence.cjs'); + +// ─── TAP fixtures (captured from node --test --test-reporter tap) ───────────── + +/** Row 1: intentional target failure — `not ok 1 - rejects empty email`, # fail 1, exit 1. */ +const TARGET_FAILURE_TAP = [ + 'TAP version 13', + '# Subtest: rejects empty email', + 'not ok 1 - rejects empty email', + ' ---', + ' duration_ms: 1.15', + " error: 'Expected values to be strictly equal. 1 !== 2'", + " code: 'ERR_ASSERTION'", + ' ...', + '1..1', + '# tests 1', + '# suites 0', + '# pass 0', + '# fail 1', + '# cancelled 0', + '# skipped 0', + '# todo 0', + '# duration_ms 46.8', + '', +].join('\n'); + +/** Rows 2: zero-test discovery — discovery matched zero tests, harness exits nonzero. */ +const ZERO_TESTS_TAP = [ + 'TAP version 13', + '1..0', + '# tests 0', + '# suites 0', + '# pass 0', + '# fail 0', + '# cancelled 0', + '# skipped 0', + '# todo 0', + '# duration_ms 3.1', + '', +].join('\n'); + +/** Row 3: fixture crash / load throw — the failure is FILE-NAMED, not the target test. */ +const FIXTURE_CRASH_TAP = [ + 'TAP version 13', + '# Subtest: crash.test.cjs', + 'not ok 1 - crash.test.cjs', + ' ---', + ' duration_ms: 28.3', + " type: 'test'", + ' location: \'crash.test.cjs:1:1\'', + " failureType: 'testCodeFailure'", + ' exitCode: 1', + " error: 'test failed'", + " code: 'ERR_TEST_FAILURE'", + ' ...', + '1..1', + '# tests 1', + '# suites 0', + '# pass 0', + '# fail 1', + '# cancelled 0', + '# skipped 0', + '# todo 0', + '# duration_ms 28.3', + '', +].join('\n'); + +/** Row 6: an unrelated test fails — a real assertion, but not the target test. */ +const UNRELATED_FAILURE_TAP = TARGET_FAILURE_TAP.replaceAll( + 'rejects empty email', + 'unrelated legacy behavior', +); + +function validRedInput(overrides = {}) { + return { + command: 'node --test tests/email.test.cjs', + exitCode: 1, + output: TARGET_FAILURE_TAP, + targetTest: 'rejects empty email', + targetFile: 'tests/email.test.cjs', + expected: 'ValidationError for empty input', + actual: '1 !== 2', + ...overrides, + }; +} + +// ─── Pure classifier (#3770 regression rows) ───────────────────────────────── + +describe('classifyRedEvidence (#3770)', () => { + test('row 1 — classifyRedEvidence accepts intentional target failure', () => { + const result = classifyRedEvidence(validRedInput()); + assert.equal(result.verdict, 'RED_EVIDENCE_OK', 'an intentional target failure is the only valid RED'); + assert.equal(result.reason, 'target_test_failed'); + assert.equal(result.evidence.failing_tests[0], 'rejects empty email'); + assert.equal(result.evidence.exit_code, 1); + }); + + test('row 2 — classifyRedEvidence rejects zero-test discovery', () => { + const result = classifyRedEvidence(validRedInput({ output: ZERO_TESTS_TAP })); + assert.equal(result.verdict, 'INVALID_RED'); + assert.equal(result.reason, 'zero_tests_discovered'); + }); + + test('row 3 — classifyRedEvidence rejects fixture crash', () => { + const result = classifyRedEvidence( + validRedInput({ output: FIXTURE_CRASH_TAP, targetFile: 'tests/crash.test.cjs' }), + ); + assert.equal(result.verdict, 'INVALID_RED'); + assert.equal(result.reason, 'fixture_or_load_failure'); + }); + + test('row 4 — classifyRedEvidence rejects nonzero exit without a failing test', () => { + const result = classifyRedEvidence(validRedInput({ output: TARGET_FAILURE_TAP.replace('# fail 1', '# fail 0') })); + assert.equal(result.verdict, 'INVALID_RED'); + assert.equal(result.reason, 'nonzero_exit_without_test_failure'); + }); + + test('row 5 — classifyRedEvidence rejects unexpected green', () => { + const result = classifyRedEvidence(validRedInput({ exitCode: 0 })); + assert.equal(result.verdict, 'INVALID_RED'); + assert.equal(result.reason, 'unexpected_green'); + }); + + test('row 6 — classifyRedEvidence rejects unrelated failing test', () => { + const result = classifyRedEvidence(validRedInput({ output: UNRELATED_FAILURE_TAP })); + assert.equal(result.verdict, 'INVALID_RED'); + assert.equal(result.reason, 'no_target_test_failure'); + }); + + test('fail-closed — malformed record fields are INVALID_RED, never a crash', () => { + const result = classifyRedEvidence({ command: null, exitCode: '1', output: 42, targetTest: '' }); + assert.equal(result.verdict, 'INVALID_RED'); + assert.equal(result.reason, 'invalid_record'); + }); +}); + +// ─── Persisted record (acceptance: command, exit code, failing test, expected, actual) ── + +describe('buildRedEvidenceRecord (#3770)', () => { + test('row 7 — buildRedEvidenceRecord persists the RED evidence fields', () => { + const input = validRedInput(); + const result = classifyRedEvidence(input); + const record = buildRedEvidenceRecord(input, result); + assert.equal(record.command, 'node --test tests/email.test.cjs'); + assert.equal(record.exit_code, 1); + assert.equal(record.failing_test, 'rejects empty email'); + assert.equal(record.expected, 'ValidationError for empty input'); + assert.equal(record.actual, '1 !== 2'); + assert.equal(record.verdict, 'RED_EVIDENCE_OK'); + assert.equal(record.reason, 'target_test_failed'); + }); +}); + +// ─── check tdd-red-evidence router arm ──────────────────────────────────────── + +describe('check tdd-red-evidence verb (#3770)', () => { + /** Root for record-file fixtures; removed in after(). */ + let ROOT = ''; + + before(() => { ROOT = fs.mkdtempSync(path.join(os.tmpdir(), 'tdd-red-evidence-')); }); + after(() => cleanup(ROOT)); + + function writeRecord(name, record) { + const file = path.join(ROOT, name); + fs.writeFileSync(file, JSON.stringify(record), 'utf8'); + return file; + } + + test('row 8 — check tdd-red-evidence accepts a valid persisted record', () => { + const file = writeRecord('valid.json', validRedInput()); + const result = runGsdTools(['check', 'tdd-red-evidence', file, '--raw'], ROOT); + assert.ok(result.success, `expected success, stderr: ${result.error}`); + const payload = JSON.parse(result.output); + assert.equal(payload.passed, true); + assert.equal(payload.verdict, 'RED_EVIDENCE_OK'); + assert.equal(payload.reason, 'target_test_failed'); + }); + + test('row 9 — check tdd-red-evidence rejects a crash record', () => { + const file = writeRecord( + 'crash.json', + validRedInput({ output: FIXTURE_CRASH_TAP, targetFile: 'tests/crash.test.cjs' }), + ); + const result = runGsdTools(['check', 'tdd-red-evidence', file, '--raw'], ROOT); + const payload = JSON.parse(result.output); + assert.equal(payload.passed, false); + assert.equal(payload.verdict, 'INVALID_RED'); + assert.equal(payload.reason, 'fixture_or_load_failure'); + }); + + test('row 10 — check tdd-red-evidence fails closed on missing record', () => { + const result = runGsdTools( + ['check', 'tdd-red-evidence', path.join(ROOT, 'does-not-exist.json'), '--raw'], + ROOT, + ); + const payload = JSON.parse(result.output); + assert.equal(payload.passed, false); + assert.equal(payload.verdict, 'INVALID_RED'); + assert.equal(payload.reason, 'unreadable_record'); + }); +}); + +// ─── Spec surfaces (#3770 acceptance: gate must require evidence before GREEN) ─ + +describe('executor spec requires intentional RED evidence before GREEN (#3770)', () => { + const read = (p) => fs.readFileSync(path.join(__dirname, '..', p), 'utf8'); + + test('row 11 — executor spec names INVALID_RED and blocks GREEN without evidence', () => { + const agent = read('agents/gsd-executor.md'); + const tddRef = read('gsd-core/references/tdd.md'); + const mvpRef = read('gsd-core/references/execute-mvp-tdd.md'); + for (const [name, content] of [['gsd-executor.md', agent], ['tdd.md', tddRef], ['execute-mvp-tdd.md', mvpRef]]) { + assert.match(content, /INVALID_RED/, `${name} must name the INVALID_RED verdict`); + assert.match(content, /tdd-red-evidence/, `${name} must wire the check tdd-red-evidence gate`); + } + // The gate must block GREEN on invalid RED, not merely warn. + assert.match(mvpRef, /INVALID_RED[^\n]{0,120}(block|halt|trip|STOP)/i, + 'execute-mvp-tdd.md must halt GREEN on INVALID_RED'); + }); +});