diff --git a/.changeset/gentle-orcas-click.md b/.changeset/gentle-orcas-click.md new file mode 100644 index 000000000..e17689f51 --- /dev/null +++ b/.changeset/gentle-orcas-click.md @@ -0,0 +1,5 @@ +--- +type: Changed +pr: 753 +--- +The `gsd-verifier` agent no longer re-runs the full workspace test suite once per must-have during Step 7b spot-checks — it enumerates tests to prove existence and runs a single named test to prove a pass, invoking the full suite at most once per verification. diff --git a/agents/gsd-verifier.md b/agents/gsd-verifier.md index 07d149d2c..b4a1610d1 100644 --- a/agents/gsd-verifier.md +++ b/agents/gsd-verifier.md @@ -466,8 +466,11 @@ ls $BUILD_OUTPUT_DIR/*.{js,css} 2>/dev/null | wc -l # Module exports expected functions node -e "const m = require('$MODULE_PATH'); console.log(typeof m.$FUNCTION_NAME)" 2>/dev/null | grep -q "function" -# Test suite passes (if tests exist for this phase's code) -npm test -- --grep "$PHASE_TEST_PATTERN" 2>&1 | grep -q "passing" +# A test EXISTS (existence proof — enumerate, do NOT run the suite) +cargo test -- --list 2>/dev/null | grep -q "$PHASE_TEST_PATTERN" # pytest --collect-only -q · npx vitest list · go test -list '.*' + +# A specific test PASSES (run ONE named test, never the whole suite) +cargo test "$TEST_NAME" -- --exact # pytest -k "$TEST_NAME" · npx vitest run -t "$TEST_NAME" ``` 2. **Run each check** and record pass/fail: @@ -487,6 +490,7 @@ npm test -- --grep "$PHASE_TEST_PATTERN" 2>&1 | grep -q "passing" - Each check must complete in under 10 seconds - Do not start servers or services — only test what's already runnable - Do not modify state (no writes, no mutations, no side effects) +- **Run the full workspace test command at most once per verification.** Never filter a full run per must-have (` 2>&1 | grep X` repeated per truth) — it re-runs everything and yields no new evidence. Prove a test exists by enumeration (`--list` / `--collect-only`); prove one passes via a single named test. If a full run is genuinely required, run it once and `grep` the saved output. - If the project has no runnable entry points yet, skip with: "Step 7b: SKIPPED (no runnable entry points)" ## Step 7c: Probe Execution diff --git a/docs/AGENTS.md b/docs/AGENTS.md index b63a9176f..ff048bda2 100644 --- a/docs/AGENTS.md +++ b/docs/AGENTS.md @@ -292,6 +292,7 @@ GSD uses a multi-agent architecture where thin orchestrators (workflow files) sp - Logs issues for `/gsd-verify-work` to address - Milestone scope filtering: gaps addressed in later phases are marked as "deferred", not reported as failures (v1.32) - **Test quality audit** (v1.32): verifies that tests prove what they claim by checking for disabled/skipped tests on requirements, circular test patterns (system generating its own expected values), assertion strength (existence vs. value vs. behavioral), and expected value provenance. Blockers from test quality audit override an otherwise passing verification +- Runs the full workspace test suite at most once per verification — proves a test *exists* by enumeration and that it *passes* via a single named test, never re-running the whole suite per must-have. --- diff --git a/tests/verifier-spotcheck-test-discipline.test.cjs b/tests/verifier-spotcheck-test-discipline.test.cjs new file mode 100644 index 000000000..df7ac5e56 --- /dev/null +++ b/tests/verifier-spotcheck-test-discipline.test.cjs @@ -0,0 +1,57 @@ +'use strict'; +const test = require('node:test'); +const assert = require('node:assert'); +const fs = require('node:fs'); +const path = require('node:path'); + +// Issue #25: gsd-verifier Step 7b must not present the runner-specific +// full-suite "| grep" anti-pattern, and must steer toward +// enumerate-for-existence / single-named-test-for-pass. +const verifierPath = path.join(__dirname, '..', 'agents', 'gsd-verifier.md'); +const content = fs.readFileSync(verifierPath, 'utf8'); + +// Scope assertions to the Step 7b section so the guard tracks the right place. +const start = content.indexOf('## Step 7b'); +const end = content.indexOf('## Step 7c', start); +assert.ok(start !== -1 && end !== -1 && end > start, 'Step 7b section not found'); +const step7b = content.slice(start, end); + +test('Step 7b drops the misleading full-suite grep example', () => { + assert.ok( + !step7b.includes('npm test -- --grep "$PHASE_TEST_PATTERN" 2>&1 | grep -q "passing"'), + 'the runner-specific `npm test --grep` example should be removed (it mis-generalizes to ` | grep`)' + ); +}); + +test('Step 7b steers existence proofs to test enumeration', () => { + assert.ok( + /cargo test -- --list/.test(step7b), + 'Step 7b should show the correct `cargo test -- --list` enumeration form' + ); + assert.ok( + /pytest --collect-only/.test(step7b) && + /vitest list/.test(step7b) && + /go test -list/.test(step7b), + 'Step 7b should list cross-ecosystem enumeration commands (pytest/vitest/go)' + ); +}); + +test('Step 7b steers pass-checks to a single named test', () => { + assert.ok( + /-- --exact/.test(step7b) && + /pytest -k/.test(step7b) && + /vitest run -t/.test(step7b), + 'Step 7b should show single-named-test commands across ecosystems' + ); +}); + +test('Step 7b forbids re-running the full suite per must-have', () => { + assert.ok( + /at most once per verification/i.test(step7b), + 'Step 7b should forbid invoking the full workspace test command more than once per verification' + ); + assert.ok( + /grep/i.test(step7b) && /per must-have|per truth/i.test(step7b), + 'Step 7b should explicitly call out the per-must-have full-suite grep anti-pattern' + ); +});