diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index a7a5b0a54..60e171b42 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -501,6 +501,14 @@ Required cases where relevant: Property-style parser tests are encouraged for high-risk parsers. They must be deterministic: pin the seed, bound the iteration count, and print replay data on failure. +##### Fixture provenance (#2371) + +**A gate's fixtures may not be derived from the gate's own writer, grammar, or docstring examples. A negative fixture must come from a source that does not know the gate exists.** + +This is stricter than the adversarial-input rule above and exists because of it: `tests/fixtures/adversarial/` covers hostile input, but a fixture written by the parser's own author — even a deliberately "realistic" one — is still drawn from the author's mental model of the format. It can only ever confirm what the author already believed, never surface what they didn't anticipate. A property-test generator has the same failure mode one level up: seeding the generator from the writer/render function that produces the same format makes the document shape a constant, so the property can never explore a document the writer wouldn't produce (see the document-shaped vs. writer-seeded property tests in `tests/api-coverage.test.cjs` for a worked example — the writer-seeded one cannot fail against a decoy table; the document-shaped one can). + +For a gate whose fixtures come from real user reports, put them under `tests/fixtures/representative//` with a `MANIFEST.json` labeling each fixture's source issue and expected gate verdict, and drive them through the gate's real CLI entrypoint (gate-verdict altitude), not the parser function in isolation — see `tests/fixtures/representative/README.md` and `tests/representative-corpus.test.cjs`. If the gate is not yet fixed, do not mark the assertion `{ todo: true }` and do not skip it: this repo's test-runner (`gsd-test` / `gsd-test-runner`) has no concept of node:test's `todo` option — its JSONL result parser only recognizes `kind: "pass" | "fail"`, so a thrown todo-marked test is still counted as a real failure and blocks the push gate. Instead record BOTH the correct target verdict (`expected*`) and the exact current observed verdict (`currentBuggyOutput`) in the manifest, and assert against `currentBuggyOutput` — an honest, non-vacuous characterization of today's known-broken behavior that passes today and breaks loudly the moment the real fix changes the observed output, forcing the assertion to be flipped to `expected*`. + #### Filesystem writes and installers Changes to install/uninstall flows, generated artifact writers, state/config writers, worktree safety, or any code that writes under `.planning`, runtime config dirs, `.claude`, `.codex`, `hooks`, or generated files must include fault-injection coverage where the seam allows it. diff --git a/gsd-core/bin/gsd-tools.cjs b/gsd-core/bin/gsd-tools.cjs index f3900b475..2f31a4779 100755 --- a/gsd-core/bin/gsd-tools.cjs +++ b/gsd-core/bin/gsd-tools.cjs @@ -285,8 +285,6 @@ const { routeInitCommand } = require('./lib/init-command-router.cjs'); // here, invoked from case 'init' below. const { warnIfStaleBake } = require('./lib/stale-bake-guard.cjs'); const loopResolver = require('./lib/loop-resolver.cjs'); -const capabilityState = require('./lib/capability-state.cjs'); -const capabilityWriter = require('./lib/capability-writer.cjs'); const { routePhaseCommand } = require('./lib/phase-command-router.cjs'); const { routePhasesCommand } = require('./lib/phases-command-router.cjs'); const { routeValidateCommand } = require('./lib/validate-command-router.cjs'); diff --git a/tests/api-coverage.test.cjs b/tests/api-coverage.test.cjs index 9bff9ab45..68dd49868 100644 --- a/tests/api-coverage.test.cjs +++ b/tests/api-coverage.test.cjs @@ -20,6 +20,25 @@ const fc = require('fast-check'); const MODULE_PATH = path.join(__dirname, '..', 'gsd-core', 'bin', 'lib', 'api-coverage.cjs'); +// Shared row-shape generators for the coverage-matrix property tests below +// (the parse/render bijection and the #2371 document-shaped property both +// build matrices from the same canonical row shape — a single declaration +// here means the two properties can't silently desync). +const capabilityGen = fc.stringMatching(/^[a-z][a-z0-9-]{0,14}$/); +const rowGen = fc.record({ + capability: capabilityGen, + decision: fc.constantFrom('INTEGRATE', 'OPT-OUT'), + // Reasons are short prose (e.g. "not needed yet"). The matrix is a + // markdown table, so cell text is format-safe: no pipes / newlines. + reason: fc.stringMatching(/^[a-z0-9 ,.\-!?]{0,20}$/), +}); +// OPT-OUT rows must carry a non-empty reason for the round-trip to validate. +const validRowGen = rowGen.map((r) => + r.decision === 'OPT-OUT' && r.reason.trim() === '' + ? { ...r, reason: 'because' } + : { ...r, reason: r.reason.trim() } +); + describe('detectApiIntegration — pure detector (#1562)', () => { let mod; try { @@ -346,20 +365,6 @@ describe('coverage matrix — parse/render bijection (fast-check)', () => { const { renderCoverageMatrix, validateCoverageMatrix } = mod; test('any valid row set renders and re-validates to the same counts', () => { - const capabilityGen = fc.stringMatching(/^[a-z][a-z0-9-]{0,14}$/); - const rowGen = fc.record({ - capability: capabilityGen, - decision: fc.constantFrom('INTEGRATE', 'OPT-OUT'), - // Reasons are short prose (e.g. "not needed yet"). The matrix is a - // markdown table, so cell text is format-safe: no pipes / newlines. - reason: fc.stringMatching(/^[a-z0-9 ,.\-!?]{0,20}$/), - }); - // OPT-OUT rows must carry a non-empty reason for the round-trip to validate. - const validRowGen = rowGen.map((r) => - r.decision === 'OPT-OUT' && r.reason.trim() === '' - ? { ...r, reason: 'because' } - : { ...r, reason: r.reason.trim() } - ); const matrixGen = fc.uniqueArray(validRowGen, { minLength: 1, maxLength: 8, @@ -379,6 +384,121 @@ describe('coverage matrix — parse/render bijection (fast-check)', () => { }); }); +// ────────────────────────────────────────────────────────────────────────────── +// Document-shaped property (#2371): the bijection test above generates ROWS and +// renders them through the writer, so the document shape is a constant — it +// cannot generate a second table, a decoy table, or surrounding prose, and so +// cannot fail against #2366's bugs. This property generates the DOCUMENT +// space instead: a canonical matrix interleaved with content a real +// COVERAGE.md may legitimately contain that is NOT the matrix. See +// tests/fixtures/representative/README.md and CONTRIBUTING.md's "Fixture +// provenance" section for the full rationale. +// ────────────────────────────────────────────────────────────────────────────── + +describe('coverage matrix — document-shaped fast-check (extract-exactly-canonical, #2371)', () => { + let mod; + try { + mod = require(MODULE_PATH); + } catch (err) { + throw new Error(`Could not require ${MODULE_PATH}. Run "npm run build:lib" first. Underlying: ${err.message}`); + } + const { parseCoverageMatrix, renderCoverageMatrix } = mod; + + // Uses the shared capabilityGen/rowGen/validRowGen declared at module scope + // above (same generators the bijection test uses), so the two properties + // exercise the same canonical-row space and can't silently desync. + const canonicalMatrixGen = fc.uniqueArray(validRowGen, { + minLength: 1, + maxLength: 5, + selector: (r) => r.capability.toLowerCase(), + }); + + // Decoy blocks: content a document may legitimately contain that is NOT the + // canonical matrix. Kept to three explicit, independently-readable shapes + // rather than a generic "random markdown" generator — a combinatorial but + // opaque generator is exactly the kind of cleverness that's unrunnable to + // debug when it fails (Kernighan's Law). + const proseDecoyGen = fc.constantFrom( + '## Notes\n\nSee the ADR for background.', + 'This phase also touches the auth helper.', + '## Risks\n\n- Rollout risk is low.', + ); + + const summaryTableDecoyGen = fc + .record({ + label: fc.stringMatching(/^[a-z][a-z0-9 ]{0,10}$/), + integrateCount: fc.nat({ max: 50 }), + optoutCount: fc.nat({ max: 50 }), + }) + .map( + ({ label, integrateCount, optoutCount }) => + `## Coverage summary\n\n| tier | INTEGRATE | OPT-OUT |\n|---|---|---|\n` + + `| ${label} | ${integrateCount} | ${optoutCount} |` + ); + + const secondSectionMatrixGen = fc + .uniqueArray(validRowGen, { minLength: 1, maxLength: 3, selector: (r) => r.capability.toLowerCase() }) + .map((rows) => `## Transferred to a later phase\n\n${renderCoverageMatrix(rows)}`); + + const decoyGen = fc.oneof(proseDecoyGen, summaryTableDecoyGen, secondSectionMatrixGen); + + const documentGen = fc.record({ + canonicalRows: canonicalMatrixGen, + decoysBefore: fc.array(decoyGen, { maxLength: 2 }), + decoysAfter: fc.array(decoyGen, { maxLength: 2 }), + }); + + // #2371's own test-runner (gsd-test / gsd-test-runner v1.6.2) has no concept + // of node:test's `todo` option: its JSONL result parser + // (internal/pipeline/parse.go's parseJSONL, gsd-test-runner repo) only + // recognizes `kind: "pass" | "fail"` and hard-errors on anything else, so a + // `{ todo: true }` test whose body throws is still counted as a failure in + // the tool's own verdict — verified directly against that source, not + // assumed. So this property uses fc's non-throwing `fc.check` (returns + // `RunDetails` instead of throwing — see fast-check's runners docs) and + // asserts on `.failed` directly: today the invariant genuinely does NOT + // hold (that is #2366), so `report.failed === true` is an honest, + // non-vacuous, currently-PASSING characterization of today's known-broken + // reality — not a fake pass. The moment #2366 makes the invariant hold for + // real, `report.failed` becomes `false` and THIS assertion fails loudly, + // forcing whoever's fix landed to notice and flip it. The fix itself stays + // owned by #2366. + test( + 'given a document containing exactly one canonical matrix plus arbitrary other content, ' + + 'the parser extracts exactly that matrix\'s rows and ignores everything else ' + + '(currently violated — #2366)', + () => { + const report = fc.check( + fc.property(documentGen, ({ canonicalRows, decoysBefore, decoysAfter }) => { + const canonicalBlock = renderCoverageMatrix(canonicalRows); + const doc = [...decoysBefore, canonicalBlock, ...decoysAfter].join('\n\n'); + + const result = parseCoverageMatrix(doc); + + const expectedByCap = new Map(canonicalRows.map((r) => [r.capability.toLowerCase(), r])); + const actualByCap = new Map(result.rows.map((r) => [r.capability.toLowerCase(), r])); + + if (actualByCap.size !== expectedByCap.size) return false; + for (const [cap, expected] of expectedByCap) { + const actual = actualByCap.get(cap); + if (!actual || actual.decision !== expected.decision) return false; + } + return result.errors.length === 0; + }), + { numRuns: 100 } + ); + assert.strictEqual( + report.failed, + true, + 'This property is expected to be VIOLATED today (#2366 — a decoy summary table or a ' + + 'second canonical-schema section corrupts the result or spuriously errors). If this ' + + 'assertion fails, the property now HOLDS — #2366 appears fixed; replace this ' + + 'characterization with a real fc.assert of the invariant.' + ); + } + ); +}); + // ────────────────────────────────────────────────────────────────────────────── // CLI entry point (STDIN → exit codes mirror grep, like assumption-delta) // ────────────────────────────────────────────────────────────────────────────── diff --git a/tests/fixtures/golden-install-parity/antigravity.json b/tests/fixtures/golden-install-parity/antigravity.json index c600950c3..ad0164839 100644 --- a/tests/fixtures/golden-install-parity/antigravity.json +++ b/tests/fixtures/golden-install-parity/antigravity.json @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "ea841e2865248e74", - "gsd-core/bin/gsd-tools.cjs": "48a2355e15c585c2", + "gsd-core/bin/gsd-tools.cjs": "0cfb0889b6fc5928", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/augment.json b/tests/fixtures/golden-install-parity/augment.json index aab27bf49..b33a155e9 100644 --- a/tests/fixtures/golden-install-parity/augment.json +++ b/tests/fixtures/golden-install-parity/augment.json @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/claude-local.json b/tests/fixtures/golden-install-parity/claude-local.json index 737959d88..2328b37e6 100644 --- a/tests/fixtures/golden-install-parity/claude-local.json +++ b/tests/fixtures/golden-install-parity/claude-local.json @@ -109,7 +109,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/claude.json b/tests/fixtures/golden-install-parity/claude.json index 7017b37d9..69f8c200d 100644 --- a/tests/fixtures/golden-install-parity/claude.json +++ b/tests/fixtures/golden-install-parity/claude.json @@ -38,7 +38,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/cline.json b/tests/fixtures/golden-install-parity/cline.json index 96f1f82d6..cf643a457 100644 --- a/tests/fixtures/golden-install-parity/cline.json +++ b/tests/fixtures/golden-install-parity/cline.json @@ -42,7 +42,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "476aa24e8c4f03cf", - "gsd-core/bin/gsd-tools.cjs": "bc864a9bf3a21f8b", + "gsd-core/bin/gsd-tools.cjs": "ecb7831b68ecb4a9", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/codebuddy.json b/tests/fixtures/golden-install-parity/codebuddy.json index 367c000d7..35df34926 100644 --- a/tests/fixtures/golden-install-parity/codebuddy.json +++ b/tests/fixtures/golden-install-parity/codebuddy.json @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/codex.json b/tests/fixtures/golden-install-parity/codex.json index b9219a4f3..66b1557b2 100644 --- a/tests/fixtures/golden-install-parity/codex.json +++ b/tests/fixtures/golden-install-parity/codex.json @@ -145,7 +145,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/copilot.json b/tests/fixtures/golden-install-parity/copilot.json index 4ee6ae734..0fbed9dea 100644 --- a/tests/fixtures/golden-install-parity/copilot.json +++ b/tests/fixtures/golden-install-parity/copilot.json @@ -40,7 +40,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "ea841e2865248e74", - "gsd-core/bin/gsd-tools.cjs": "48a2355e15c585c2", + "gsd-core/bin/gsd-tools.cjs": "0cfb0889b6fc5928", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/cursor.json b/tests/fixtures/golden-install-parity/cursor.json index 2f4d16eae..e560ebd17 100644 --- a/tests/fixtures/golden-install-parity/cursor.json +++ b/tests/fixtures/golden-install-parity/cursor.json @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "2525f1ae8b086828", - "gsd-core/bin/gsd-tools.cjs": "39106b83bdc47046", + "gsd-core/bin/gsd-tools.cjs": "6e31bf5c3515b677", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/hermes.json b/tests/fixtures/golden-install-parity/hermes.json index 51959eb3b..876701d76 100644 --- a/tests/fixtures/golden-install-parity/hermes.json +++ b/tests/fixtures/golden-install-parity/hermes.json @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "3a3409215044af9f", - "gsd-core/bin/gsd-tools.cjs": "b663dd5ecf091d51", + "gsd-core/bin/gsd-tools.cjs": "b0385bba7000b281", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/kilo.json b/tests/fixtures/golden-install-parity/kilo.json index fe70fd2ca..ebed16f01 100644 --- a/tests/fixtures/golden-install-parity/kilo.json +++ b/tests/fixtures/golden-install-parity/kilo.json @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/kimi.json b/tests/fixtures/golden-install-parity/kimi.json index 77683a54c..296881790 100644 --- a/tests/fixtures/golden-install-parity/kimi.json +++ b/tests/fixtures/golden-install-parity/kimi.json @@ -103,7 +103,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/opencode.json b/tests/fixtures/golden-install-parity/opencode.json index 13f95264e..38625d76a 100644 --- a/tests/fixtures/golden-install-parity/opencode.json +++ b/tests/fixtures/golden-install-parity/opencode.json @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/pi.json b/tests/fixtures/golden-install-parity/pi.json index b48b64ac5..1c79fa95b 100644 --- a/tests/fixtures/golden-install-parity/pi.json +++ b/tests/fixtures/golden-install-parity/pi.json @@ -6,7 +6,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/qwen.json b/tests/fixtures/golden-install-parity/qwen.json index 2237eb0af..4bb29b5d0 100644 --- a/tests/fixtures/golden-install-parity/qwen.json +++ b/tests/fixtures/golden-install-parity/qwen.json @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "6e98d76e955e35a2", - "gsd-core/bin/gsd-tools.cjs": "b0e0c49e82e33a9c", + "gsd-core/bin/gsd-tools.cjs": "8bdd0b02837b7a6e", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/trae.json b/tests/fixtures/golden-install-parity/trae.json index 76c09634d..d90a49f7a 100644 --- a/tests/fixtures/golden-install-parity/trae.json +++ b/tests/fixtures/golden-install-parity/trae.json @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "de4627dff103d527", - "gsd-core/bin/gsd-tools.cjs": "4e19039b6a346562", + "gsd-core/bin/gsd-tools.cjs": "d7b5484a84a5bd15", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/windsurf.json b/tests/fixtures/golden-install-parity/windsurf.json index 2c3e5ec65..3c5302f73 100644 --- a/tests/fixtures/golden-install-parity/windsurf.json +++ b/tests/fixtures/golden-install-parity/windsurf.json @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "5636ca0b726871b2", - "gsd-core/bin/gsd-tools.cjs": "196f1439ef0da939", + "gsd-core/bin/gsd-tools.cjs": "09a4dc673d2fd74e", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/golden-install-parity/zcode.json b/tests/fixtures/golden-install-parity/zcode.json index 69249b4cd..111485476 100644 --- a/tests/fixtures/golden-install-parity/zcode.json +++ b/tests/fixtures/golden-install-parity/zcode.json @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "2041725176ce6b66", + "gsd-core/bin/gsd-tools.cjs": "48b736dd16863746", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", "gsd-core/bin/shared/config-schema.manifest.json": "0109a5a9866a24e0", diff --git a/tests/fixtures/representative/README.md b/tests/fixtures/representative/README.md new file mode 100644 index 000000000..2242de632 --- /dev/null +++ b/tests/fixtures/representative/README.md @@ -0,0 +1,82 @@ +# Representative Gate Fixtures (#2371) + +Every fixture in this tree is **verbatim** (or, where noted, a minimal +faithful subset) of an artifact that a real user actually produced and +reported against a real GSD gate. None of it was written by a gate's own +author to exercise that gate. + +## Why this directory exists, and why `tests/fixtures/adversarial/` isn't enough + +`tests/fixtures/adversarial/` covers hostile input — unicode, CRLF, nested +fences, heredoc breakout. Nobody attacked the gates these fixtures target. +A developer wrote an ordinary, well-formed artifact — a plan, a coverage +matrix, a CONTEXT.md — that a gate misjudged anyway. That input is neither +synthetic-happy nor adversarial; it's simply *real*, and until #2371 no gate +had coverage for it. + +The pattern this corpus exists to break: a gate's test fixtures were +authored by the same person (or model) who wrote the gate, from the same +mental model, so they can only confirm what the author already believed — +never surface what the author didn't anticipate. See #2371 for the full +diagnosis (four incidents across three gates in ten days, including a +fast-check property test whose generator was seeded from the parser's own +writer function and therefore could not fail). + +## Rule + +**A gate's fixtures may not be derived from the gate's own writer, grammar, +or docstring examples. A negative fixture must come from a source that +does not know the gate exists.** Recorded in `CONTRIBUTING.md` under +"Fixture provenance." + +## Layout + +Each subdirectory is one gate: + +- `api-coverage-detector/` — `detectApiIntegration` (#2365) +- `api-coverage-matrix/` — `parseCoverageMatrix` (#2366) +- `audit-uat/` — `parseUatItems` / `parseVerificationItems` (#2286, fixed by #2317) +- `decision-coverage-guard/` — `extractDecisions`'s could-not-parse guard (#1365 gap, #2347) + +Each carries its own `README.md` (what each fixture is and where it came +from) and `MANIFEST.json` (machine-readable: file → source issue → gate → +expected verdict). `tests/representative-corpus.test.cjs` loads every +manifest and drives each fixture through the gate's real CLI entrypoint — +never the parser function directly — so the assertion is at gate-verdict +altitude (the boolean/JSON a user actually sees), not parse-tree altitude. + +## Why the still-broken fixtures assert `currentBuggyOutput`, not a red `todo` + +Three of the four gates here are still open bugs (#2365, #2366, #2347) at +the time this corpus was added. Their `MANIFEST.json` entries carry BOTH +the correct target verdict (`expected*` — what the eventual fix must +produce) and the exact CURRENT observed verdict (`currentBuggyOutput` — +what today's code actually returns). The test asserts against +`currentBuggyOutput`: an honest, non-vacuous characterization of today's +known-broken reality, not a fake pass. + +This is deliberately NOT node:test's `todo` option. `todo` looked like the +right tool — a todo test executes and reports its failure without +affecting Node's own process exit code +(https://nodejs.org/api/test.html#test-options) — but this repo's actual +test-runner (`gsd-test` / `gsd-test-runner` v1.6.2) has no concept of it: +its JSONL result parser (`internal/pipeline/parse.go`'s `parseJSONL`, in +the separate `gsd-test-runner` repo) only recognizes `kind: "pass" | "fail"` +and hard-errors on anything else — verified directly against that source, +not assumed. A `{ todo: true }` test whose body throws is still counted as +a real failure in `gsd-test`'s own verdict, which would block the push +gate exactly as if it weren't marked todo at all. + +Asserting `currentBuggyOutput` sidesteps this because the test genuinely +passes today — no runner-level "expected failure" feature required. The +fixes belong to #2365 / #2366 / #2347, not to this corpus. When one of +those lands, the corresponding assertion will fail (the gate now returns +something other than the pinned buggy value) — at that point, flip the +test to assert `expected*` instead and delete the stale +`currentBuggyOutput`. + +The `audit-uat/` corpus has no `todo`: #2286 was fixed by #2317 before this +corpus was written, so its assertions are ordinary, currently-passing +tests — the proof that a representative fixture, driven through the real +gate, is not automatically doomed to fail. It demonstrates the methodology +working, not just the gaps it finds. diff --git a/tests/fixtures/representative/api-coverage-detector/MANIFEST.json b/tests/fixtures/representative/api-coverage-detector/MANIFEST.json new file mode 100644 index 000000000..80dc98a30 --- /dev/null +++ b/tests/fixtures/representative/api-coverage-detector/MANIFEST.json @@ -0,0 +1,29 @@ +{ + "gate": "api-coverage.verify-pre", + "sourceIssue": "#2365", + "fixtures": [ + { + "file": "nextjs-route-path.txt", + "expectedDetected": false, + "currentBuggyOutput": { "detected": true, "signal": { "verb": "integration", "noun": "api" } }, + "note": "First-party Next.js route path. The noun-boundary class [^a-zA-Z0-9] treats '/' as a word boundary, so 'api' inside the path matches as if it were prose. Highest blast radius: any Next.js project names a route file." + }, + { + "file": "unrelated-verb-noun.txt", + "expectedDetected": false, + "currentBuggyOutput": { "detected": true, "signal": { "verb": "wiring", "noun": "endpoint" } }, + "note": "'wiring' and 'endpoint' co-occur on one line in unrelated clauses, reverse semantic order, no integration described. The verb/noun regexes are independently exec'd over the whole line with no proximity or grammatical relation." + }, + { + "file": "threat-model-prose.txt", + "expectedDetected": false, + "currentBuggyOutput": { "detected": true, "signal": { "verb": "(surface)", "noun": "api" } }, + "note": "Threat-model table cell describing a LOCAL interface. SERVICE_SURFACE_API_RE matches any capitalized word before API/SDK/REST/GraphQL; the stopword denylist cannot enumerate every ordinary English word that precedes 'API' in a sentence." + }, + { + "file": "non-integration-assertion.txt", + "expectedDetected": false, + "note": "This line explicitly ASSERTS non-integration ('no new command/dependency') and is still read as an integration signal in the real $gsd-verify-work occurrence this was drawn from — but in isolation it already returns detected:false today (the multi-signal real occurrence needed the OTHER lines' signals to trip the gate). No currentBuggyOutput: this fixture already passes for the right reason." + } + ] +} diff --git a/tests/fixtures/representative/api-coverage-detector/README.md b/tests/fixtures/representative/api-coverage-detector/README.md new file mode 100644 index 000000000..8c61867f1 --- /dev/null +++ b/tests/fixtures/representative/api-coverage-detector/README.md @@ -0,0 +1,23 @@ +# API-coverage detector fixtures (#2365) + +Verbatim reproduction lines from #2365, each used as a `PLAN.md` body and +driven through `check api-coverage.verify-pre ` — the real +blocking gate, not `detectApiIntegration()` called in isolation. + +- `nextjs-route-path.txt` — a first-party Next.js route path read as an + external API signal because `/` is a word boundary. +- `unrelated-verb-noun.txt` — a verb and a noun on the same line, unrelated + clauses, no compound relation. +- `threat-model-prose.txt` — a threat-model table cell describing a LOCAL + interface, misread as a third-party service name. +- `non-integration-assertion.txt` — a line that explicitly states no new + integration was added, misread as evidence of one. + +The first three currently `detected: true`; the gate expects `false` for +all four (see `MANIFEST.json`'s `expectedDetected`/`currentBuggyOutput` +fields). `non-integration-assertion.txt` already returns `detected: false` +in isolation, so it's asserted directly. The other three are asserted +against their `currentBuggyOutput` in `tests/representative-corpus.test.cjs` +— a characterization of today's known-broken behavior, not a `todo` (see +`tests/fixtures/representative/README.md` for why `todo` doesn't work with +this repo's test-runner) — until #2365 lands. diff --git a/tests/fixtures/representative/api-coverage-detector/nextjs-route-path.txt b/tests/fixtures/representative/api-coverage-detector/nextjs-route-path.txt new file mode 100644 index 000000000..c711b9fa3 --- /dev/null +++ b/tests/fixtures/representative/api-coverage-detector/nextjs-route-path.txt @@ -0,0 +1 @@ +Run integration tests for src/app/api/profile/route.test.ts \ No newline at end of file diff --git a/tests/fixtures/representative/api-coverage-detector/non-integration-assertion.txt b/tests/fixtures/representative/api-coverage-detector/non-integration-assertion.txt new file mode 100644 index 000000000..091523f04 --- /dev/null +++ b/tests/fixtures/representative/api-coverage-detector/non-integration-assertion.txt @@ -0,0 +1 @@ +Reuse existing validator APIs; no new command/dependency. \ No newline at end of file diff --git a/tests/fixtures/representative/api-coverage-detector/threat-model-prose.txt b/tests/fixtures/representative/api-coverage-detector/threat-model-prose.txt new file mode 100644 index 000000000..8818574e5 --- /dev/null +++ b/tests/fixtures/representative/api-coverage-detector/threat-model-prose.txt @@ -0,0 +1 @@ +| Tampering | Resolver-only API rejects arbitrary caller URLs. | \ No newline at end of file diff --git a/tests/fixtures/representative/api-coverage-detector/unrelated-verb-noun.txt b/tests/fixtures/representative/api-coverage-detector/unrelated-verb-noun.txt new file mode 100644 index 000000000..b95aaf924 --- /dev/null +++ b/tests/fixtures/representative/api-coverage-detector/unrelated-verb-noun.txt @@ -0,0 +1 @@ +Render the page and prove label endpoint, filename, and CSV/XLSX wiring. \ No newline at end of file diff --git a/tests/fixtures/representative/api-coverage-matrix/MANIFEST.json b/tests/fixtures/representative/api-coverage-matrix/MANIFEST.json new file mode 100644 index 000000000..ebb110536 --- /dev/null +++ b/tests/fixtures/representative/api-coverage-matrix/MANIFEST.json @@ -0,0 +1,23 @@ +{ + "gate": "api-coverage.verify-pre", + "sourceIssue": "#2366", + "fixtures": [ + { + "file": "multi-table-with-summary.md", + "pairedPlan": "Integrate the Stripe API for payment processing.", + "expectedCounts": { "surface": 3, "integrate": 1, "optout": 2 }, + "expectedErrorCount": 0, + "expectedBlock": false, + "currentBuggyOutput": { + "block": true, + "error_count": 3, + "errors": [ + "row: decision \"**OPT-OUT**\" not in {INTEGRATE, OPT-OUT}", + "row: decision \"DECISION\" not in {INTEGRATE, OPT-OUT}", + "row: decision \"12\" not in {INTEGRATE, OPT-OUT}" + ] + }, + "note": "Self-contained 15-line COVERAGE.md from #2366's own reproduction: one canonical matrix (search/skip), a second section-split canonical table (widget) for a transferred capability, and a decoy 3-column summary table whose 2nd cell reads 'INTEGRATE'. Today's parser invents a 'tier' capability from the summary table's header row (silent corruption, zero errors) and drops 'skip' (bolded decision rejected), while also spuriously erroring on the second header — the errors array above shows the rejected-DECISION-header and the rejected-bolded-OPT-OUT and rejected-numeric-'12'-cell paths, none of which mention the silently-invented 'tier' row (that corruption produces no error at all, which is the headline finding). Expected once fixed: exactly 3 rows (search, skip, widget), 0 errors — see expectedCounts/expectedErrorCount/expectedBlock." + } + ] +} diff --git a/tests/fixtures/representative/api-coverage-matrix/README.md b/tests/fixtures/representative/api-coverage-matrix/README.md new file mode 100644 index 000000000..38850cd2a --- /dev/null +++ b/tests/fixtures/representative/api-coverage-matrix/README.md @@ -0,0 +1,23 @@ +# API-coverage matrix fixture (#2366) + +`multi-table-with-summary.md` is the verbatim 15-line `repro-coverage.md` +from #2366's own reproduction, used as a `COVERAGE.md` body and driven +through `check api-coverage.verify-pre ` (paired with a PLAN.md +that integrates an API, so a matrix is required) — the real blocking gate, +not `parseCoverageMatrix()` called in isolation. + +Contains, in one file: the canonical `| capability | decision | reason |` +matrix, a second canonical-schema table under a "Transferred to a later +phase" heading (the section-split use case #2366 names as legitimate), and +a decoy 3-column "Coverage summary" table whose header row happens to read +`| tier | INTEGRATE | OPT-OUT |`. + +Expected once fixed: 3 rows (`search`, `skip`, `widget`), 0 errors (see +`MANIFEST.json`'s `expectedCounts`/`expectedErrorCount`/`expectedBlock`). +Today's parser instead invents a `tier` capability from the summary table +(silent corruption — zero errors reported for that path) while rejecting +the bolded `skip` decision and two other cells, producing 3 errors and +`block: true` — pinned in `MANIFEST.json`'s `currentBuggyOutput` and +asserted directly in `tests/representative-corpus.test.cjs` (a +characterization of today's known-broken behavior, not a `todo` — see +`tests/fixtures/representative/README.md` for why) until #2366 lands. diff --git a/tests/fixtures/representative/api-coverage-matrix/multi-table-with-summary.md b/tests/fixtures/representative/api-coverage-matrix/multi-table-with-summary.md new file mode 100644 index 000000000..200c7d92c --- /dev/null +++ b/tests/fixtures/representative/api-coverage-matrix/multi-table-with-summary.md @@ -0,0 +1,18 @@ +# API Coverage — demo + +| capability | decision | reason | +|---|---|---| +| search | INTEGRATE | | +| skip | **OPT-OUT** | not needed yet | + +## Transferred to a later phase + +| capability | decision | reason | +|---|---|---| +| widget | OPT-OUT | deferred to 9 | + +## Coverage summary + +| tier | INTEGRATE | OPT-OUT | +|---|---|---| +| phase 8 | 12 | 6 | diff --git a/tests/fixtures/representative/audit-uat/MANIFEST.json b/tests/fixtures/representative/audit-uat/MANIFEST.json new file mode 100644 index 000000000..3f0b7c6b4 --- /dev/null +++ b/tests/fixtures/representative/audit-uat/MANIFEST.json @@ -0,0 +1,20 @@ +{ + "gate": "audit-uat", + "sourceIssue": "#2286", + "fixedBy": "#2317", + "fixtures": [ + { + "file": "gaps-section-uat.md", + "filenameSuffix": "-UAT.md", + "expectedMinItems": 1, + "note": "A '## Gaps' bullet entry with status: open. Before #2317, parseUatItems never scanned this section at all, so this file's real outstanding item was silently invisible." + }, + { + "file": "human-verification-frontmatter.md", + "filenameSuffix": "-VERIFICATION.md", + "expectedMinItems": 1, + "note": "Frontmatter declares status: human_needed with a populated human_verification: array. Before #2317, parseVerificationItems never read the frontmatter's structured array and only recognized specific body shapes, so this file read as zero items." + } + ], + "expectedTotalItems": 2 +} diff --git a/tests/fixtures/representative/audit-uat/README.md b/tests/fixtures/representative/audit-uat/README.md new file mode 100644 index 000000000..1f1dd8fe9 --- /dev/null +++ b/tests/fixtures/representative/audit-uat/README.md @@ -0,0 +1,17 @@ +# Audit-UAT fixtures (#2286, fixed by #2317) + +Verbatim reproduction files from #2286, driven through `gsd-tools +audit-uat --raw` (the real CLI gate) rather than `parseUatItems` / +`parseVerificationItems` called in isolation. + +- `gaps-section-uat.md` — a UAT file whose only outstanding finding lives + in a `## Gaps` bullet entry. +- `human-verification-frontmatter.md` — a VERIFICATION file whose + frontmatter declares a structured `human_verification:` array. + +This is the one corpus in `tests/fixtures/representative/` with **no** +`todo` marker. #2286 was fixed by #2317 (merged) before this corpus was +written, so `total_items >= 2` is a normal, currently-passing assertion — +proof that a representative fixture, driven through the real gate, is not +automatically doomed to fail. It demonstrates the methodology working end +to end, not just the gaps it finds in the other three gates. diff --git a/tests/fixtures/representative/audit-uat/gaps-section-uat.md b/tests/fixtures/representative/audit-uat/gaps-section-uat.md new file mode 100644 index 000000000..39d0fee5f --- /dev/null +++ b/tests/fixtures/representative/audit-uat/gaps-section-uat.md @@ -0,0 +1,4 @@ +## Gaps + +- truth: "SC1: some success criterion" + status: open diff --git a/tests/fixtures/representative/audit-uat/human-verification-frontmatter.md b/tests/fixtures/representative/audit-uat/human-verification-frontmatter.md new file mode 100644 index 000000000..2f6b1bea7 --- /dev/null +++ b/tests/fixtures/representative/audit-uat/human-verification-frontmatter.md @@ -0,0 +1,11 @@ +--- +status: human_needed +human_verification: + - test: "Confirm the widget renders correctly" +--- + +## Human Verification Required + +### 1. Widget render check + +**Confirm the widget appears as expected on the dashboard.** diff --git a/tests/fixtures/representative/decision-coverage-guard/MANIFEST.json b/tests/fixtures/representative/decision-coverage-guard/MANIFEST.json new file mode 100644 index 000000000..f51db1228 --- /dev/null +++ b/tests/fixtures/representative/decision-coverage-guard/MANIFEST.json @@ -0,0 +1,18 @@ +{ + "gate": "check.decision-coverage-plan", + "sourceIssue": "#2347", + "fixtures": [ + { + "file": "d5-prefix-context.md", + "expectedReason": "could-not-parse", + "expectedPassed": false, + "currentBuggyOutput": { + "passed": true, + "skipped": true, + "reason": "no trackable decisions", + "total": 0 + }, + "note": "A block using the D5-01 ID-prefix shape from #2347's own reproduction ('- **D5-01:** some decision'), repeated twice so 'populated but 0 extracted' is unambiguous. The original report used 23 real decisions under a project-specific D5- prefix convention; this fixture preserves the exact grammar mismatch, not the count. The #1365 guard's evidence test (/\\bD-[A-Za-z0-9]/) shares the parser's own D- grammar, so it is blind to exactly the input class it exists to catch: both see nothing, and the gate reports passed:true, skipped:true, reason:'no trackable decisions' instead of failing loud." + } + ] +} diff --git a/tests/fixtures/representative/decision-coverage-guard/README.md b/tests/fixtures/representative/decision-coverage-guard/README.md new file mode 100644 index 000000000..bdba0a51d --- /dev/null +++ b/tests/fixtures/representative/decision-coverage-guard/README.md @@ -0,0 +1,24 @@ +# Decision-coverage guard fixture (#2347) + +`d5-prefix-context.md` is the verbatim reproduction shape from #2347 — the +`- **D5-01:** some decision` bullet given in the issue's own "Steps to +reproduce" — used as a CONTEXT.md `` block and driven through +`query check.decision-coverage-plan ` (the real +CLI gate; see `tests/decisions.test.cjs` for the established pattern this +follows), not `extractDecisions()` called in isolation. + +The #1365 fail-loud guard's "is this decision-shaped?" evidence test +(`/\bD-[A-Za-z0-9]/`) reuses the same `D-` grammar as the parser it guards. +For any ID prefix the parser cannot read — `D5-01` here — the guard sees +no evidence either, so the two failure modes the guard exists to +distinguish (`none-present` vs `could-not-parse`) collapse into +`none-present`, and a populated, genuinely decision-shaped CONTEXT.md +passes silently. + +Expected once fixed: `reason: 'could-not-parse'`, `passed: false` (see +`MANIFEST.json`'s `expectedReason`/`expectedPassed`). Today's gate instead +reports `passed: true, skipped: true, reason: 'no trackable decisions'` — +pinned in `MANIFEST.json`'s `currentBuggyOutput` and asserted directly in +`tests/representative-corpus.test.cjs` (a characterization of today's +known-broken behavior, not a `todo` — see +`tests/fixtures/representative/README.md` for why) until #2347 lands. diff --git a/tests/fixtures/representative/decision-coverage-guard/d5-prefix-context.md b/tests/fixtures/representative/decision-coverage-guard/d5-prefix-context.md new file mode 100644 index 000000000..0d9215ae6 --- /dev/null +++ b/tests/fixtures/representative/decision-coverage-guard/d5-prefix-context.md @@ -0,0 +1,4 @@ + +- **D5-01:** some decision +- **D5-02:** some other decision + diff --git a/tests/representative-corpus.test.cjs b/tests/representative-corpus.test.cjs new file mode 100644 index 000000000..cc86c1073 --- /dev/null +++ b/tests/representative-corpus.test.cjs @@ -0,0 +1,261 @@ +'use strict'; + +/** + * Representative-corpus gate tests (#2371). + * + * Every fixture under tests/fixtures/representative/ is verbatim (or a + * minimal faithful subset) of a real reported artifact — never invented to + * match a gate's own grammar. See tests/fixtures/representative/README.md + * for the full rationale and CONTRIBUTING.md's "Fixture provenance" rule. + * + * Each gate is driven through its real CLI entrypoint (gate-verdict + * altitude), matching the established pattern in + * tests/api-coverage-gate-e2e.test.cjs and tests/decisions.test.cjs — never + * the parser function called in isolation. + * + * Three fixtures (across api-coverage-detector, api-coverage-matrix, + * decision-coverage-guard) encode gates that are still open bugs (#2365, + * #2366, #2347). For those, MANIFEST.json carries BOTH the correct target + * verdict (`expected*` — what the fix must produce) and the exact CURRENT + * observed verdict (`currentBuggyOutput` — what today's code actually + * returns). The test asserts against `currentBuggyOutput`: an honest, + * non-vacuous characterization of today's known-broken reality, not a fake + * pass. This assertion WILL fail, loudly, the moment the underlying bug is + * fixed and the gate starts returning something other than the pinned + * buggy value — at which point whoever's fix landed must update the + * assertion to check `expected*` instead (and can delete `currentBuggyOutput`). + * + * Why not node:test's `todo` option: this repo's own test-runner + * (gsd-test / gsd-test-runner v1.6.2) has no concept of it. Its JSONL + * result parser (internal/pipeline/parse.go's parseJSONL, gsd-test-runner + * repo) only recognizes `kind: "pass" | "fail"` — verified directly against + * that source — so a `{ todo: true }` test whose body throws is still + * counted as a real failure in the tool's own verdict. Characterization + * (assert the known-current value) sidesteps this because the test + * genuinely passes today; it needs no runner-level "expected failure" + * feature at all. + * + * The audit-uat corpus (#2286, fixed by #2317) has no currentBuggyOutput: + * it already asserts the correct behavior directly, because the bug is + * already fixed — proof the methodology works end to end, not just a + * record of gaps. + */ + +const { describe, test, afterEach } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const os = require('node:os'); +const path = require('node:path'); +const { execFileSync } = require('node:child_process'); + +const { cleanup } = require('./helpers.cjs'); + +const TOOLS_PATH = path.join(__dirname, '..', 'gsd-core', 'bin', 'gsd-tools.cjs'); +const FIXTURES_ROOT = path.join(__dirname, 'fixtures', 'representative'); + +const TEST_ENV_BASE = { + GSD_SESSION_KEY: '', + CODEX_THREAD_ID: '', + CLAUDE_SESSION_ID: '', + CLAUDE_CODE_SSE_PORT: '', + OPENCODE_SESSION_ID: '', + GEMINI_SESSION_ID: '', + CURSOR_SESSION_ID: '', + WINDSURF_SESSION_ID: '', + TERM_SESSION: '', + WT_SESSION: '', + TMUX_PANE: '', + ZELLIJ_SESSION_NAME: '', + TTY: '', + SSH_TTY: '', +}; + +function runTools(args, cwd) { + try { + const stdout = execFileSync(process.execPath, [TOOLS_PATH, ...args], { + cwd, + encoding: 'utf-8', + env: { ...process.env, ...TEST_ENV_BASE }, + timeout: 60000, + }); + return { success: true, output: stdout.trim(), error: '' }; + } catch (err) { + return { + success: false, + output: err.stdout?.toString().trim() || '', + error: err.stderr?.toString().trim() || err.message, + }; + } +} + +function readManifest(gateDir) { + const raw = fs.readFileSync(path.join(FIXTURES_ROOT, gateDir, 'MANIFEST.json'), 'utf8'); + return JSON.parse(raw); +} + +function readFixture(gateDir, file) { + return fs.readFileSync(path.join(FIXTURES_ROOT, gateDir, file), 'utf8'); +} + +function makeProject() { + const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gsd-repcorpus-')); + fs.mkdirSync(path.join(tmpDir, '.planning', 'phases'), { recursive: true }); + fs.writeFileSync(path.join(tmpDir, '.planning', 'config.json'), '{}', 'utf8'); + return tmpDir; +} + +function makePhaseDir(projectDir, slug) { + const dir = path.join(projectDir, '.planning', 'phases', slug); + fs.mkdirSync(dir, { recursive: true }); + return dir; +} + +// ─── api-coverage-detector (#2365) ──────────────────────────────────────────── + +describe('representative corpus — api-coverage detector (#2365)', () => { + let tmpDir; + afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } }); + + const manifest = readManifest('api-coverage-detector'); + + for (const fx of manifest.fixtures) { + const label = fx.currentBuggyOutput ? `${fx.file} → currently detected:true (#2365)` : `${fx.file} → detected:false`; + test(label, () => { + tmpDir = makeProject(); + const phaseDir = makePhaseDir(tmpDir, '01-repcorpus'); + const body = readFixture('api-coverage-detector', fx.file); + fs.writeFileSync(path.join(phaseDir, '01-PLAN.md'), `# Plan\n${body}\n`, 'utf8'); + + const r = runTools(['check', 'api-coverage.verify-pre', phaseDir, '--raw'], tmpDir); + assert.ok(r.success, `gate should succeed (JSON). stderr: ${r.error}`); + const j = JSON.parse(r.output); + + if (fx.currentBuggyOutput) { + assert.strictEqual(j.detected, fx.currentBuggyOutput.detected, + `${fx.file}: expected today's known-buggy detected:${fx.currentBuggyOutput.detected}, got ${JSON.stringify(j)}. ` + + `If this now differs, #2365 may be fixed — check against expectedDetected:${fx.expectedDetected} instead.`); + assert.strictEqual(j.signals?.[0]?.verb, fx.currentBuggyOutput.signal.verb, `${fx.file}: signal.verb`); + assert.strictEqual(j.signals?.[0]?.noun, fx.currentBuggyOutput.signal.noun, `${fx.file}: signal.noun`); + } else { + assert.strictEqual(j.detected, fx.expectedDetected, + `${fx.file}: expected detected:${fx.expectedDetected}, got ${JSON.stringify(j)}`); + } + }); + } +}); + +// ─── api-coverage-matrix (#2366) ────────────────────────────────────────────── + +describe('representative corpus — api-coverage matrix (#2366)', () => { + let tmpDir; + afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } }); + + const manifest = readManifest('api-coverage-matrix'); + + for (const fx of manifest.fixtures) { + const label = fx.currentBuggyOutput + ? `${fx.file} → currently silently-corrupted + spurious errors (#2366)` + : `${fx.file} → exactly the canonical rows, 0 errors`; + test(label, () => { + tmpDir = makeProject(); + const phaseDir = makePhaseDir(tmpDir, '01-repcorpus'); + fs.writeFileSync(path.join(phaseDir, '01-PLAN.md'), `# Plan\n${fx.pairedPlan}\n`, 'utf8'); + fs.writeFileSync(path.join(phaseDir, 'COVERAGE.md'), readFixture('api-coverage-matrix', fx.file), 'utf8'); + + const r = runTools(['check', 'api-coverage.verify-pre', phaseDir, '--raw'], tmpDir); + assert.ok(r.success, `gate should succeed (JSON). stderr: ${r.error}`); + const j = JSON.parse(r.output); + + if (fx.currentBuggyOutput) { + assert.strictEqual(j.block, fx.currentBuggyOutput.block, + `${fx.file}: expected today's known-buggy block:${fx.currentBuggyOutput.block}, got ${JSON.stringify(j)}. ` + + `If this now differs, #2366 may be fixed — check against expectedBlock:${fx.expectedBlock} instead.`); + assert.strictEqual(j.error_count, fx.currentBuggyOutput.error_count, `${fx.file}: error_count`); + assert.deepStrictEqual(j.errors, fx.currentBuggyOutput.errors, `${fx.file}: errors`); + } else { + assert.strictEqual(j.block, fx.expectedBlock, `${fx.file}: block. Got ${JSON.stringify(j)}`); + assert.deepStrictEqual(j.counts, fx.expectedCounts, `${fx.file}: counts. Got ${JSON.stringify(j)}`); + assert.strictEqual((j.errors || []).length, fx.expectedErrorCount, + `${fx.file}: errors. Got ${JSON.stringify(j.errors)}`); + } + }); + } +}); + +// ─── audit-uat (#2286, fixed by #2317 — asserts correct behavior directly) ──── + +describe('representative corpus — audit-uat (#2286, fixed by #2317)', () => { + let tmpDir; + afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } }); + + const manifest = readManifest('audit-uat'); + + test('Gaps-section + human-verification-frontmatter fixtures both surface as real items', () => { + tmpDir = makeProject(); + const phaseDir = makePhaseDir(tmpDir, '01-repcorpus'); + for (const fx of manifest.fixtures) { + fs.writeFileSync( + path.join(phaseDir, `01${fx.filenameSuffix}`), + readFixture('audit-uat', fx.file), + 'utf8', + ); + } + + const r = runTools(['audit-uat', '--raw'], tmpDir); + assert.ok(r.success, `audit-uat should succeed. stderr: ${r.error}`); + const j = JSON.parse(r.output); + assert.ok( + j.summary.total_items >= manifest.expectedTotalItems, + `expected total_items >= ${manifest.expectedTotalItems}, got ${JSON.stringify(j.summary)}`, + ); + // Per-fixture check (not just the aggregate): a regression that moves + // items between files while preserving the total would slip past the + // total_items check above but not this one. + for (const fx of manifest.fixtures) { + const fileName = `01${fx.filenameSuffix}`; + const fileResult = j.results.find((r2) => r2.file === fileName); + assert.ok(fileResult, `expected a result entry for ${fileName}, got ${JSON.stringify(j.results)}`); + assert.ok( + fileResult.items.length >= fx.expectedMinItems, + `${fileName}: expected items.length >= ${fx.expectedMinItems}, got ${fileResult.items.length}`, + ); + } + }); +}); + +// ─── decision-coverage-guard (#2347) ────────────────────────────────────────── + +describe('representative corpus — decision-coverage guard (#2347)', () => { + let tmpDir; + afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } }); + + const manifest = readManifest('decision-coverage-guard'); + + for (const fx of manifest.fixtures) { + const label = fx.currentBuggyOutput + ? `${fx.file} → currently passed:true, skipped:true (#2347)` + : `${fx.file} → outcome could-not-parse, passed:false`; + test(label, () => { + tmpDir = makeProject(); + const phaseDir = makePhaseDir(tmpDir, '01-repcorpus'); + const contextPath = path.join(phaseDir, 'CONTEXT.md'); + fs.writeFileSync(contextPath, readFixture('decision-coverage-guard', fx.file), 'utf8'); + + const r = runTools(['query', 'check.decision-coverage-plan', phaseDir, contextPath], tmpDir); + assert.ok(r.success, `gate should succeed (JSON). stderr: ${r.error}`); + const j = JSON.parse(r.output); + + if (fx.currentBuggyOutput) { + assert.strictEqual(j.passed, fx.currentBuggyOutput.passed, + `${fx.file}: expected today's known-buggy passed:${fx.currentBuggyOutput.passed}, got ${JSON.stringify(j)}. ` + + `If this now differs, #2347 may be fixed — check against expectedPassed:${fx.expectedPassed} instead.`); + assert.strictEqual(j.skipped, fx.currentBuggyOutput.skipped, `${fx.file}: skipped`); + assert.strictEqual(j.reason, fx.currentBuggyOutput.reason, `${fx.file}: reason`); + assert.strictEqual(j.total, fx.currentBuggyOutput.total, `${fx.file}: total`); + } else { + assert.strictEqual(j.passed, fx.expectedPassed, `${fx.file}: passed. Got ${JSON.stringify(j)}`); + assert.strictEqual(j.reason, fx.expectedReason, `${fx.file}: reason. Got ${JSON.stringify(j)}`); + } + }); + } +});