From d30c99bc9252a5ca739bc720ffbe12c9cc3f2f31 Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Thu, 13 Aug 2026 21:22:03 -0400 Subject: [PATCH] chore(#3421): delete orphan verify-phase workflow, migrate live gates to verifier (#3422) * chore(#1892): delete orphan verify-phase workflow, migrate live gates to verifier reference * test(#1892): retarget structural suites from verify-phase.md to verifier-phase-gates.md * chore(#1892): reword retired-workflow mentions for removed-but-needed lint * test(#1892): correct stale surface labels in retargeted suites * docs(#1892): add verifier-phase-gates row to locale inventories * chore(#3421): backfill changeset pr number --------- Co-authored-by: sim --- .changeset/graceful-wolves-wake.md | 5 + agents/gsd-verifier.md | 1 + docs/AGENTS.md | 2 +- docs/INVENTORY-MANIFEST.json | 2 +- docs/INVENTORY.md | 9 +- ...606-prohibition-enforcement-verify-seam.md | 2 +- docs/adr/1820-spec-optional-predicate-rail.md | 2 +- docs/adr/550-spec-phase-probe-contract.md | 8 +- .../1192-adr-test-audit-2026-06-13.md | 2 +- docs/ja-JP/INVENTORY.md | 4 +- docs/ko-KR/INVENTORY.md | 4 +- docs/pt-BR/INVENTORY.md | 4 +- docs/zh-CN/INVENTORY.md | 4 +- gsd-core/references/planning-config.md | 4 +- gsd-core/references/verifier-phase-gates.md | 175 ++++++ gsd-core/templates/phase-prompt.md | 2 - gsd-core/workflows/code-review.md | 2 +- gsd-core/workflows/verify-phase.md | 574 ------------------ src/normalize-test-command.cts | 2 +- src/probe-core.cts | 2 +- .../1955-verifier-coincidental-reliance.json | 2 +- tests/execute-phase-active-flags.test.cjs | 15 +- tests/fixtures/install-tree/antigravity.json | 2 +- tests/fixtures/install-tree/augment.json | 2 +- tests/fixtures/install-tree/claude-local.json | 2 +- tests/fixtures/install-tree/claude.json | 2 +- tests/fixtures/install-tree/cline.json | 2 +- tests/fixtures/install-tree/codebuddy.json | 2 +- tests/fixtures/install-tree/codex.json | 2 +- tests/fixtures/install-tree/copilot.json | 2 +- tests/fixtures/install-tree/cursor.json | 2 +- tests/fixtures/install-tree/hermes.json | 2 +- tests/fixtures/install-tree/kilo.json | 2 +- tests/fixtures/install-tree/kimi-code.json | 2 +- tests/fixtures/install-tree/kimi.json | 2 +- tests/fixtures/install-tree/opencode.json | 2 +- tests/fixtures/install-tree/pi.json | 2 +- tests/fixtures/install-tree/qwen.json | 2 +- tests/fixtures/install-tree/trae.json | 2 +- tests/fixtures/install-tree/windsurf.json | 2 +- tests/fixtures/install-tree/zcode.json | 2 +- tests/plan-phase-drift-guard.test.cjs | 31 +- tests/planner-language-regression.test.cjs | 2 - tests/prohibition-probe.validators.test.cjs | 2 +- tests/test-gate-watch-mode.test.cjs | 27 +- tests/verifier-behavior-unverified.test.cjs | 4 +- tests/verifier-coincidental-reliance.test.cjs | 37 +- tests/verifier-deferred-items.test.cjs | 41 +- tests/verify-mvp-uat.test.cjs | 16 +- tests/verify-test-quality.test.cjs | 25 +- tests/windows-robustness.test.cjs | 1 - tests/workflow-size-budget.test.cjs | 14 +- 52 files changed, 313 insertions(+), 752 deletions(-) create mode 100644 .changeset/graceful-wolves-wake.md create mode 100644 gsd-core/references/verifier-phase-gates.md delete mode 100644 gsd-core/workflows/verify-phase.md diff --git a/.changeset/graceful-wolves-wake.md b/.changeset/graceful-wolves-wake.md new file mode 100644 index 000000000..bea0ceb03 --- /dev/null +++ b/.changeset/graceful-wolves-wake.md @@ -0,0 +1,5 @@ +--- +type: Removed +pr: 3422 +--- +**Removed the orphaned `verify-phase` workflow (~40 KB shipped to every runtime, never loaded)** — its still-live verification gates (decision-coverage validation, test-quality audit, infrastructure-phase human-verification scoping) moved to a reference the verifier agent actually loads, so they run again instead of shipping as dead prose; installs are ~40 KB lighter and PRs to the verifier no longer mirror a dead twin to keep lockstep tests green. (#1891) diff --git a/agents/gsd-verifier.md b/agents/gsd-verifier.md index 4020dda21..909227a61 100644 --- a/agents/gsd-verifier.md +++ b/agents/gsd-verifier.md @@ -41,6 +41,7 @@ Every truth must resolve to VERIFIED, FAILED (BLOCKER), or UNCERTAIN (WARNING wi @~/.claude/gsd-core/references/verification-overrides.md @~/.claude/gsd-core/references/gates.md +@~/.claude/gsd-core/references/verifier-phase-gates.md This agent implements the **Escalation Gate** pattern (surfaces unresolvable gaps to the developer for decision). diff --git a/docs/AGENTS.md b/docs/AGENTS.md index ff51f0855..613b58a48 100644 --- a/docs/AGENTS.md +++ b/docs/AGENTS.md @@ -321,7 +321,7 @@ Two further dimensions carry no number: **Verify Command Format Sanity** and - **Behavior-dependent calibration (#966):** a must-have that asserts a state transition or a cancellation/cleanup/ordering invariant is marked `⚠️ PRESENT_BEHAVIOR_UNVERIFIED` (not `VERIFIED`) when no test exercises it — excluded from the `verified_truths` score, counted in the `behavior_unverified` frontmatter field, and routed to human verification, so a clean `N/N` certifies behavioral evidence rather than mere symbol presence. - **Coincidental-reliance advisory (#1955):** a truth that reaches `✓ VERIFIED` is additionally asked *why* it holds. When the recorded evidence shows the truth holding for an incidental reason — `undeclared-precondition`, `incidental-ordering`, or `fixture-only` — the verdict is qualified as `✓ VERIFIED (coincidental-reliance)` and the truth is listed in the `coincidental_reliance_items` frontmatter field with what to harden. This is **advisory**: the base `✓ VERIFIED` token is unchanged, the truth still counts toward `verified_truths`, the overall `status` is unaffected, and no human-verification item is emitted — a passing phase still passes. It classifies evidence the verifier already gathered rather than asking it to rate its own confidence — but it is honestly an **endogenous** check, and `gsd-core/references/honest-verifier.md` records that endogenous gates are measurably weaker than the exogenous `backstop` tag it routes on. Advisory status is the consequence, not a coincidence: a miss costs exactly today's behaviour (a plain `✓ VERIFIED`) and a false positive costs one line of prose, never a failed phase, so a weaker mechanism is affordable here in a way it would not be on a pass/fail axis. Its precision is unmeasured. It complements the two existing axes: `PRESENT_BEHAVIOR_UNVERIFIED` is *no* behavioral evidence, `insufficient_spec` is an under-specified truth, and this is evidence that exists and passes for the wrong reason. - The advisory is carried by two surfaces. `agents/gsd-verifier.md` (Step 3, sub-step 5c) holds the detection rule for the subagent path. The non-subagent path, `gsd-core/workflows/verify-phase.md`, receives it through the eager `@`-import of `gsd-core/templates/verification-report.md`, whose `## Guidelines` carry the same instruction — `verify-phase.md` itself is deliberately **not** edited, because it sits 29 bytes under the DEFAULT tier hard cap in `tests/workflow-size-budget.test.cjs` and absorbing rubric prose there requires a lazy extraction first. + The advisory is carried by two surfaces. `agents/gsd-verifier.md` (Step 3, sub-step 5c) holds the detection rule, and the verifier's eagerly-imported `gsd-core/references/verifier-phase-gates.md` points at the canonical report template `@~/.claude/gsd-core/templates/verification-report.md`, whose `## Guidelines` carry the same instruction. (The former third surface, the retired `verify-phase` workflow, was deleted as an orphan in #1892 — every verification path is subagent-shaped today.) --- diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index b7c504f61..2e9697977 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -199,7 +199,6 @@ "undo.md", "update.md", "validate-phase.md", - "verify-phase.md", "verify-work.md" ], "references": [ @@ -299,6 +298,7 @@ "user-story-template.md", "verification-overrides.md", "verification-patterns.md", + "verifier-phase-gates.md", "verifier-wiring-patterns.md", "verify-mvp-mode.md", "workstream-flag.md", diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index f91f85a78..ff6b96318 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -264,10 +264,9 @@ Full roster at `gsd-core/workflows/*.md`. Workflows are thin orchestrators that | `thread.md` | Create, list, close, or resume persistent context threads for cross-session work. | `/gsd-thread` | | `update.md` | Update GSD to latest version with changelog display. | `/gsd-update` | | `validate-phase.md` | Retroactively audit and fill Nyquist validation gaps for a completed phase. | `/gsd-validate-phase` | -| `verify-phase.md` | Verify phase goal achievement through goal-backward analysis. | `execute-phase.md` (post-execution) | | `verify-work.md` | Conversational UAT with auto-diagnosis — produces UAT.md and fix plans. | `/gsd-verify-work` | -> **Note:** Some workflows have no direct user-facing command (e.g. `execute-plan.md`, `verify-phase.md`, `transition.md`, `node-repair.md`, `diagnose-issues.md`) — they are invoked internally by orchestrator workflows. `discovery-phase.md` is an alternate entry for `/gsd-new-project`. +> **Note:** Some workflows have no direct user-facing command (e.g. `execute-plan.md`, `transition.md`, `node-repair.md`, `diagnose-issues.md`) — they are invoked internally by orchestrator workflows. `discovery-phase.md` is an alternate entry for `/gsd-new-project`. (The former `verify-phase` workflow — goal-backward verification with no loader of its own — was deleted in #1892; its still-live gates moved to `references/verifier-phase-gates.md` behind `gsd-verifier`.) ### Workflow Sub-Files @@ -305,6 +304,7 @@ Full roster at `gsd-core/references/*.md`. References are shared knowledge docum | `model-profile-resolution.md` | Model resolution algorithm documentation. | | `verification-patterns.md` | How to verify different artifact types. | | `verification-overrides.md` | Per-artifact verification override rules. | +| `verifier-phase-gates.md` | Verifier-time gates eagerly imported by `gsd-verifier` (migrated from the retired `verify-phase` workflow, #1892): decision-coverage validation (#2492), test-quality audit, and infrastructure-phase human-verification scoping (#2504). | | `planning-config.md` | Full config schema and behavior. | | `security-asvs-levels.md` | OWASP ASVS level definitions for GSD threat modeling — per-level planner disposition rigor and auditor verification depth (L1 opportunistic, L2 standard, L3 comprehensive). | | `git-integration.md` | Git commit, branching, and history patterns. | @@ -598,8 +598,9 @@ Full listing: `gsd-core/bin/lib/*.cjs`. | `surface.cjs` | Runtime surface module — manages the runtime enable/disable surface state independently of the install-time profile marker (ADR-0011 Phase 2) | | `task-command-router.cjs` | Thin CJS subcommand router adapter for `gsd-tools task` | | `template.cjs` | Template selection and filling with variable substitution | -| `text-lines.cjs` | Line-terminator handling seam — `splitLines`/`normalizeEol`/`detectEol`/`joinLines`, the sole owner of `\r?\n` splitting and CRLF normalization; closes #3360's split-then-match fix in `frontmatter.cjs` (ADR-3212 §3, epic #3212 Phase 2, #3413) | -| `normalize-test-command.cjs` | Normalizes a resolved test command to a one-shot form so a watch-mode runner (vitest/jest) cannot hang a verification gate (#1857); shared by all four test-command gates (regression, post-merge, audit-fix, verify-phase) | +| `text-lines.cjs` | Line-terminator handling seam — `splitLines`/`normalizeEol`/`detectEol`/`joinLines`, the sole owner of ` ? +` splitting and CRLF normalization; closes #3360's split-then-match fix in `frontmatter.cjs` (ADR-3212 §3, epic #3212 Phase 2, #3413) | +| `normalize-test-command.cjs` | Normalizes a resolved test command to a one-shot form so a watch-mode runner (vitest/jest) cannot hang a verification gate (#1857); shared by all three live test-command gates (regression, post-merge, audit-fix) | | `uat.cjs` | UAT file parsing, verification debt tracking, audit-uat support | | `uat-predicate.cjs` | UAT-passed predicate — markdown-aware evaluation of HUMAN-UAT results; returns pass only when all required checks pass; ignores false-positive contexts (frontmatter, fenced code, blockquotes, HTML comments) | | `ui-consideration-probe.cjs` | Spec-completeness UI-consideration probe (compiled from `src/ui-consideration-probe.cts`, gitignored) — the third adapter of the `probe-core` resolution model (ADR-550 Decision 7): element-kind classification, applicable-category relevance filter, consideration proposal, `proposeElements`/`autoResolve` (propose-then-confirm + the `--auto` never-dismiss floor), and the `{explicit, backstop}` validators; delegates merge/rollup/CLI to `probe-core`; exports `classifyElement`, `applicableCategories`, `proposeConsiderations`, `proposeElements`, `autoResolve`, `analyzeCoverage`, `UI_TAXONOMY` (#1867) | diff --git a/docs/adr/1606-prohibition-enforcement-verify-seam.md b/docs/adr/1606-prohibition-enforcement-verify-seam.md index dca4a2eee..d30773845 100644 --- a/docs/adr/1606-prohibition-enforcement-verify-seam.md +++ b/docs/adr/1606-prohibition-enforcement-verify-seam.md @@ -262,4 +262,4 @@ the control fails and the proof does not pass. - **`gsd-core/references/prohibition-probe.md`** — the portable runtime reference. - **`docs/how-to/resolve-prohibition-findings.md`** — user-facing resolution guide. - Code: `src/prohibition-enforcement.cts`, `src/probe-core.cts` (`projectProhibitions`), - `gsd-core/workflows/verify-phase.md`. Issues: #644, #1259, #1278, #1279, #1346, #1906. + `gsd-core/workflows/verify-phase (retired in #1892)`. Issues: #644, #1259, #1278, #1279, #1346, #1906. diff --git a/docs/adr/1820-spec-optional-predicate-rail.md b/docs/adr/1820-spec-optional-predicate-rail.md index 632d02c8c..9adb1166d 100644 --- a/docs/adr/1820-spec-optional-predicate-rail.md +++ b/docs/adr/1820-spec-optional-predicate-rail.md @@ -7,7 +7,7 @@ ## Context -The edge-probe and prohibition-probe families surface a phase's unwritten edges and must-NOTs. Before #1820 they ran **only** in `spec-phase` (Step 5.5 / 5.6), so their value reached the verifier **only when a phase authored a SPEC.md**. But the verifier reads `must_haves` from **PLAN.md** frontmatter (`verify-phase.md`), not from SPEC.md — SPEC.md is only an optional source `plan-phase` lifts from. A phase planned without a SPEC therefore shipped with an empty predicate set and the probes' reach was lost. Per the load-bearing premise *verifier reach = spec reach*, that is a silent coverage hole precisely where a spec is thinnest. +The edge-probe and prohibition-probe families surface a phase's unwritten edges and must-NOTs. Before #1820 they ran **only** in `spec-phase` (Step 5.5 / 5.6), so their value reached the verifier **only when a phase authored a SPEC.md**. But the verifier reads `must_haves` from **PLAN.md** frontmatter (`verify-phase` (workflow, retired in #1892)), not from SPEC.md — SPEC.md is only an optional source `plan-phase` lifts from. A phase planned without a SPEC therefore shipped with an empty predicate set and the probes' reach was lost. Per the load-bearing premise *verifier reach = spec reach*, that is a silent coverage hole precisely where a spec is thinnest. The fix wires the probe generators into `plan-phase`'s `must_haves` authoring as a **spec-optional fallback**: when a section is absent (no SPEC, or a SPEC that omits/empties that section), the fallback runs the probe and authors the predicates directly into PLAN.md. Deciding *whether a section was supplied* requires detecting the SPEC's `## Edge Coverage` / `## Prohibitions (must-NOT)` sections and counting their resolved rows. That detection first lived as ad-hoc `awk` inline in the workflow body, which hard-coded the header strings at the call site and hand-rolled markdown-table row counting — brittleness that produced two real bugs (an exact `^## Prohibitions$` anchor that missed the canonical `## Prohibitions (must-NOT)` heading, and a single-table row-counting assumption). Centralizing the detection made the knowledge testable and shared — and created a new Module seam, which is what makes this an ADR-worthy decision (contributor-standards: *"an ADR is required when a decision introduces a Module seam that other code will depend on"*). diff --git a/docs/adr/550-spec-phase-probe-contract.md b/docs/adr/550-spec-phase-probe-contract.md index 54a79b66c..6fe1861ff 100644 --- a/docs/adr/550-spec-phase-probe-contract.md +++ b/docs/adr/550-spec-phase-probe-contract.md @@ -93,7 +93,7 @@ Decision 4 describes the `test`-tier as a "**Hard gate in both interactive and a - The **negative-test enforcement mechanism** — locating the wired mechanical check, running it for a genuine **non-vacuous** pass, and building the `enforcementEvidence` that flips a passing test-tier item green — **landed in #1259** as the deterministic `check prohibition-enforcement` sub-command (authored as `src/prohibition-enforcement.cts`, compiled by `build:lib` to the gitignored `gsd-core/bin/lib/prohibition-enforcement.cjs`). It accepts BOTH wired-check kinds — a `node --test` negative test (requiring a real reported test, not the empty file `node --test` would count as one passing "test") OR a lint/AST rule run through the project flat config as `eslint --format json` filtered by `ruleId` (so plugin rules like `local/*` load — bare `--rule` cannot) — and is anchored on the in-tree `local/no-source-grep` rule (dogfooding the existing must-NOT proof, ADR-550 D4; the #644 corpus had zero authored test-tier prohibitions, so no contrived consumer was minted). A passing wired check disposes green; a missing, non-attested, or genuinely-non-passing check hard-gates (flagged, non-green) in both interactive and autonomous modes. - **Honest scope — `failFirst` is caller-ATTESTED, not machine-proven (tracked follow-up).** What #1259 lands is the *execution + non-vacuous-pass* half: the producer requires the caller to attest `failFirst: true` and requires the check to genuinely run and pass. It does **not** yet independently prove the check *fails-on-violation* (the literal `regression-must-fail-first` property), because cheap proof of that at verify time needs running the check against a known **violation fixture** — deferred as a follow-up (#1279; the descriptor auto-locate half is #1278). Until then the red-first property rests on caller attestation, surfaced transparently in the evidence record. This closes the permanent-`gaps_found` dead-end with a genuinely-executed gate without overclaiming machine-proven fail-first. -Net effect on D4: the *guarantee* ("a `test`-tier prohibition is never a silent pass") was preserved through the fail-closed-now half and is now joined by the genuine-execution half — a test-tier prohibition with a passing, non-vacuous wired check can reach `green`/`passed`, and a missing/failing one hard-gates. The previously-unreachable green branch in `dispositionForProhibition()` is reachable from the live pipeline, and the fail-closed default backs every miss/fail. The one remaining gap to D4's literal intent — *machine-proven* fail-first — is documented above as a tracked follow-up. The decision also lives in `src/probe-core.cts` comments, `src/prohibition-enforcement.cts`, `verify-phase.md`, and the #644 / #1259 changesets. +Net effect on D4: the *guarantee* ("a `test`-tier prohibition is never a silent pass") was preserved through the fail-closed-now half and is now joined by the genuine-execution half — a test-tier prohibition with a passing, non-vacuous wired check can reach `green`/`passed`, and a missing/failing one hard-gates. The previously-unreachable green branch in `dispositionForProhibition()` is reachable from the live pipeline, and the fail-closed default backs every miss/fail. The one remaining gap to D4's literal intent — *machine-proven* fail-first — is documented above as a tracked follow-up. The decision also lives in `src/probe-core.cts` comments, `src/prohibition-enforcement.cts`, `verify-phase` (workflow, retired in #1892), and the #644 / #1259 changesets. This enforcement seam is the concrete instance of **ADR-857 open-question §147** — the deferred "deterministic CI conformance test for the verifier↔predicate contract." Per D6 it lands on the **core verify rail** (non-toggleable substrate), never in `capabilities/`: the verifier consuming a contract-shaped, deterministic predicate is core, not an opt-in capability. @@ -111,7 +111,7 @@ This addendum ratifies three contract points: > **PR-review flag — PROPOSED, renamable conventions (zero live consumers).** Both **`GSD_PROHIB_SUBJECT`** and **`CheckDescriptor.violationFixture`** are net-new surface introduced by this PR with **ZERO live in-tree consumers** — there is no in-tree `node --test` prohibition yet (the #1259 dogfood anchor and the #1279 lint-rule dogfood are both the LINT-rule `local/no-source-grep`; node-test fail-first is exercised only by SYNTHETIC temp fixtures in tests). They are therefore forward-looking scaffolding, and a later rename (or replacing the env var with an argv) is a **mechanical, zero-migration find/replace**. They are surfaced here explicitly so the maintainer can **rename or replace them at PR review** — the natural ratification point, exactly as #1278's ADR addendum was reviewed at PR time — without any migration cost. The `failFirst` DEMOTION is likewise open to the reviewer weighing outright removal; the rationale for keeping it as a hint is recorded above. -Net effect on D4: the *guarantee* ("a `test`-tier prohibition is never a silent pass") was preserved at every step — fail-closed-now (#644), genuine-execution (#1259), and now **machine-proven fail-first (#1279)**. A `test`-tier prohibition reaches `green`/`passed` ONLY when the wired check both genuinely, non-vacuously passes AND is independently proven to fail on a violation; every miss/fail/un-provable hard-gates. The decision also lives in `src/prohibition-enforcement.cts` comments, `gsd-core/references/prohibition-probe.md`, `gsd-core/workflows/verify-phase.md`, and the #1279 changeset. +Net effect on D4: the *guarantee* ("a `test`-tier prohibition is never a silent pass") was preserved at every step — fail-closed-now (#644), genuine-execution (#1259), and now **machine-proven fail-first (#1279)**. A `test`-tier prohibition reaches `green`/`passed` ONLY when the wired check both genuinely, non-vacuously passes AND is independently proven to fail on a violation; every miss/fail/un-provable hard-gates. The decision also lives in `src/prohibition-enforcement.cts` comments, `gsd-core/references/prohibition-probe.md`, `gsd-core/workflows/verify-phase (retired in #1892)`, and the #1279 changeset. **Review corrections (#1314 maintainer review) — two soundness items:** - **node-test fixture-existence guard (was fail-OPEN) — FIXED.** The node-test prover originally guarded only `if (!fixture)`. A missing/typo'd/stale `violationFixture` path made `GSD_PROHIB_SUBJECT` point at a non-existent file; an honest negative test then threw ENOENT *inside its callback* — a failing test named distinctly from the file — which `isNonVacuousNodeTestRed` accepted as proof, **forging a green from a setup crash** (asymmetric with the lint-rule path, which fail-CLOSES on `< 1` file result). Fixed by requiring `fs.existsSync(path.resolve(cwd, fixture))` before spawning (symmetric fail-closed; resolved against the producer's `cwd` to match the child's resolution). **Residual (#1346) — now MITIGATED by an optional control; see the 2026-06-21 addendum below:** existence is necessary but not sufficient — a deceptive test that reds merely *because* `GSD_PROHIB_SUBJECT` is set (not because the subject's CONTENT violates) was still accepted; a generic always-on proof is impossible, so #1346 adds an **opt-in clean-subject control** that proves content-dependence when the author supplies one (and the residual remains, documented, only for checks with no control fixture). @@ -143,7 +143,7 @@ This ratifies the **deterministic SOURCE** for the test-tier `CheckDescriptor` t 5. **Out of scope (unchanged boundaries).** Machine-proven fail-first (a violation-fixture / RuleTester-invalid proof replacing the `failFirst` caller attestation) stays tracked as **#1279**. The `dispositionForProhibition` green/fail-closed **policy** is untouched. No new check kinds are added. -Net effect on D3: the prohibition-item shape is extended with three optional, backward-compatible flat-scalar keys that give the test-tier locate a deterministic spec-phase source; the contract's CI-testable surface (D5) gains the projection round-trip parity (CHK-03), the fail-closed guard (CHK-06), and the byte-stable backward-compat fixture (CHK-07). The decision also lives in `src/probe-core.cts` / `src/prohibition-enforcement.cts` comments, the `verify-phase.md` / `spec-phase.md` prose, and the #1278 changeset. +Net effect on D3: the prohibition-item shape is extended with three optional, backward-compatible flat-scalar keys that give the test-tier locate a deterministic spec-phase source; the contract's CI-testable surface (D5) gains the projection round-trip parity (CHK-03), the fail-closed guard (CHK-06), and the byte-stable backward-compat fixture (CHK-07). The decision also lives in `src/probe-core.cts` / `src/prohibition-enforcement.cts` comments, the `verify-phase` (workflow, retired in #1892) / `spec-phase.md` prose, and the #1278 changeset. ## Addendum (2026-06-25, #1154) — honest verifier: the truth-axis disposition mirror of D4 @@ -159,7 +159,7 @@ This records the **truth-axis half of Decision 4** that the original ADR deliber 5. **Two measured properties define the design (maintainer Decision 2; caveats to record).** *Exogenous, not endogenous:* abstention is triggered by the external `backstop` tag, never a self-judged "abstain if unsure" — endogenous abstention was measured near-useless on true blind spots (100% → 67% vs exogenous 100% → 17%; N17). *Routing, not diagnosis:* the verdict does not name the omitted rule (the held-out test carries it). **Evidence honesty:** N17 is n=27, 1 rep — **direction-finding, not powered**; the effect is large and monotone but real-world precision depends on the edge-probe's *true* `backstop` recall/precision (the experiment modeled a perfect tagger), which is why the over-abstention guard and the capable-tier requirement are load-bearing acceptance criteria. **Model-tier coupling:** abstention is reliable on the default `gsd-verifier` tier (`sonnet`+); the budget tier (`haiku`) heeds the tag only inconsistently and degrades toward current behavior — captured as a documented cost (and a test) so a tier regression is caught, not discovered in production. -Net effect: the truth-axis `backstop` tier gains the verify-time disposition D4 gave the prohibition judgment tier; the contract's CI-testable surface (D5) gains the truth-axis projection round-trip parity and the abstain-on-unconfirmed-backstop regression. The decision also lives in `src/probe-core.cts` comments, `gsd-core/references/honest-verifier.md`, the `plan-phase.md` / `verify-phase.md` / `agents/gsd-verifier.md` prose, and the #1154 changeset. +Net effect: the truth-axis `backstop` tier gains the verify-time disposition D4 gave the prohibition judgment tier; the contract's CI-testable surface (D5) gains the truth-axis projection round-trip parity and the abstain-on-unconfirmed-backstop regression. The decision also lives in `src/probe-core.cts` comments, `gsd-core/references/honest-verifier.md`, the `plan-phase.md` / `verify-phase` (workflow, retired in #1892) / `agents/gsd-verifier.md` prose, and the #1154 changeset. ## Addendum (2026-06-22) — Alternatives considered (recall / representation / packaging side) diff --git a/docs/issueevidence/1192-adr-test-audit-2026-06-13.md b/docs/issueevidence/1192-adr-test-audit-2026-06-13.md index 229e8b1be..517e3902f 100644 --- a/docs/issueevidence/1192-adr-test-audit-2026-06-13.md +++ b/docs/issueevidence/1192-adr-test-audit-2026-06-13.md @@ -159,7 +159,7 @@ All `redesign` + `keep-but-improve` verdicts, **plus the 6 retire-candidates tha | `tests/review-reviewer-selection.test.cjs` | Lines 17–27 KNOWN_REVIEWER_SLUGS | Tests a compiled-CJS constant's shape. Verification found the slug membership DOES drive `resolveReviewerSelection`'s warn/drop logic at runtime — it's the only guard against silent slug-list drift. | Collapse to one behavioral test: `resolveReviewerSelection` with a config'd known slug yields no warning + slug in `selected`; an unknown slug yields a warning. | | `tests/enh-191-retire-sdk-package.test.cjs` | Lines 52–58 AGENTS.md absence | Off-topic for SDK retirement, but NOT valueless: `bin/install.js:~10963` writes a root `AGENTS.md` on local Copilot install; this is the only guard against that artifact committing. | Extract to a repo-layout governance test (e.g. `tests/repo-layout.test.cjs`); rename to state the real invariant (no ad-hoc root instruction file vs CONTEXT.md/ADRs); link the copilot install path. | | `tests/capability-registry.test.cjs` | Lines 558–584 drift test | Steps 2–3 are a file-I/O tautology and skip the real `--check` path (`stripGeneratedComment` + `normalizeLineEndings`). But it is the only unit-level coverage of the drift-comparison pipeline. | Call the actual comparison expression: assert a stale version survives both `stripGeneratedComment` and `normalizeLineEndings` and is detected; assert a comment-only-timestamp change is NOT flagged after stripping. | -| `tests/verify-test-quality.test.cjs` | 3 describes (disabled / circular / assertion-strength detection) | Self-referential: define a regex inline, write a matching fixture, assert the regex matches — cannot go red for any production change. The detector lives as prose in `verify-phase.md`. | Replace with a structural guard reading `gsd-core/workflows/verify-phase.md`: assert the `audit_test_quality` step tag exists and contains the skip-pattern, circular-detection, and assertion-strength-table markers. | +| `tests/verify-test-quality.test.cjs` | 3 describes (disabled / circular / assertion-strength detection) | Self-referential: define a regex inline, write a matching fixture, assert the regex matches — cannot go red for any production change. The detector lives as prose in `verify-phase` (workflow, retired in #1892). | Replace with a structural guard reading `gsd-core/workflows/verify-phase (retired in #1892)`: assert the `audit_test_quality` step tag exists and contains the skip-pattern, circular-detection, and assertion-strength-table markers. | #### Other redesign / keep-but-improve (high-signal selection) diff --git a/docs/ja-JP/INVENTORY.md b/docs/ja-JP/INVENTORY.md index fb6e9d737..d5ab21208 100644 --- a/docs/ja-JP/INVENTORY.md +++ b/docs/ja-JP/INVENTORY.md @@ -259,10 +259,9 @@ | `thread.md` | クロスセッション作業のための永続的なコンテキストスレッドを作成、一覧表示、クローズ、または再開。 | `/gsd-thread` | | `update.md` | 変更履歴の表示付きで GSD を最新バージョンに更新。 | `/gsd-update` | | `validate-phase.md` | 完了したフェーズの Nyquist バリデーションのギャップを遡及監査して埋める。 | `/gsd-validate-phase` | -| `verify-phase.md` | ゴール後退型分析によってフェーズ目標の達成を検証。 | `execute-phase.md` (post-execution) | | `verify-work.md` | 自動診断付きの会話型 UAT — UAT.md と修正プランを作成。 | `/gsd-verify-work` | -> **注記:** 一部のワークフローには直接ユーザー向けのコマンドがありません(例: `execute-plan.md`、`verify-phase.md`、`transition.md`、`node-repair.md`、`diagnose-issues.md`)— これらはオーケストレーターワークフローによって内部的に呼び出されます。`discovery-phase.md` は `/gsd-new-project` の代替エントリーポイントです。 +> **注記:** 一部のワークフローには直接ユーザー向けのコマンドがありません(例: `execute-plan.md`、`transition.md`、`node-repair.md`、`diagnose-issues.md`)— これらはオーケストレーターワークフローによって内部的に呼び出されます。`discovery-phase.md` は `/gsd-new-project` の代替エントリーポイントです。 --- @@ -280,6 +279,7 @@ | `model-profile-resolution.md` | モデル解決アルゴリズムのドキュメント。 | | `verification-patterns.md` | 異なるアーティファクトタイプの検証方法。 | | `verification-overrides.md` | アーティファクトごとの検証オーバーライドルール。 | +| `verifier-phase-gates.md` | gsd-verifier が eager import する検証時ゲート(廃止された verify-phase ワークフローから移行、#1892):デシジョンカバレッジ検証(#2492)、テスト品質監査、インフラストラクチャフェーズの human-verification スコープ(#2504)。 | | `planning-config.md` | 完全な設定スキーマと動作。 | | `git-integration.md` | git コミット、ブランチ、履歴パターン。 | | `git-planning-commit.md` | 計画ディレクトリのコミット規約。 | diff --git a/docs/ko-KR/INVENTORY.md b/docs/ko-KR/INVENTORY.md index 5848e5d05..48330de87 100644 --- a/docs/ko-KR/INVENTORY.md +++ b/docs/ko-KR/INVENTORY.md @@ -259,10 +259,9 @@ | `thread.md` | 세션 간 작업을 위한 영속 컨텍스트 스레드 생성, 목록, 닫기, 재개. | `/gsd-thread` | | `update.md` | 체인지로그 표시와 함께 GSD를 최신 버전으로 업데이트. | `/gsd-update` | | `validate-phase.md` | 완료된 단계의 나이퀴스트 검증 공백을 소급 감사 및 채움. | `/gsd-validate-phase` | -| `verify-phase.md` | 목표 역방향 분석을 통한 단계 목표 달성 검증. | `execute-phase.md` (실행 후) | | `verify-work.md` | 자동 진단이 포함된 대화형 UAT — UAT.md 및 수정 계획 생성. | `/gsd-verify-work` | -> **참고:** 일부 워크플로우는 직접적인 사용자 대면 명령어가 없습니다(예: `execute-plan.md`, `verify-phase.md`, `transition.md`, `node-repair.md`, `diagnose-issues.md`) — 이들은 오케스트레이터 워크플로우에 의해 내부적으로 호출됩니다. `discovery-phase.md`는 `/gsd-new-project`의 대체 진입점입니다. +> **참고:** 일부 워크플로우는 직접적인 사용자 대면 명령어가 없습니다(예: `execute-plan.md`, `transition.md`, `node-repair.md`, `diagnose-issues.md`) — 이들은 오케스트레이터 워크플로우에 의해 내부적으로 호출됩니다. `discovery-phase.md`는 `/gsd-new-project`의 대체 진입점입니다. --- @@ -280,6 +279,7 @@ | `model-profile-resolution.md` | 모델 해석 알고리즘 문서. | | `verification-patterns.md` | 다양한 아티팩트 유형 검증 방법. | | `verification-overrides.md` | 아티팩트별 검증 재정의 규칙. | +| `verifier-phase-gates.md` | gsd-verifier가 즉시 로드하는 검증 시점 게이트(폐기된 verify-phase 워크플로우에서 이전, #1892): 의사결정 커버리지 검증(#2492), 테스트 품질 감사, 인프라 페이즈 human-verification 스코핑(#2504). | | | `planning-config.md` | 전체 설정 스키마 및 동작. | | `git-integration.md` | Git 커밋, 브랜칭, 히스토리 패턴. | | `git-planning-commit.md` | 계획 디렉터리 커밋 관례. | diff --git a/docs/pt-BR/INVENTORY.md b/docs/pt-BR/INVENTORY.md index 949b1f3f3..0b61c5cf8 100644 --- a/docs/pt-BR/INVENTORY.md +++ b/docs/pt-BR/INVENTORY.md @@ -259,10 +259,9 @@ Registro completo em `gsd-core/workflows/*.md`. Workflows são orquestradores en | `thread.md` | Cria, lista, fecha ou retoma threads de contexto persistentes para trabalho entre sessões. | `/gsd-thread` | | `update.md` | Atualiza o GSD para a versão mais recente com exibição do changelog. | `/gsd-update` | | `validate-phase.md` | Audita retroativamente e preenche lacunas de validação Nyquist para uma fase concluída. | `/gsd-validate-phase` | -| `verify-phase.md` | Verifica o alcance dos objetivos da fase por meio de análise retroativa a partir dos objetivos. | `execute-phase.md` (pós-execução) | | `verify-work.md` | UAT conversacional com autodiagnóstico — produz UAT.md e planos de correção. | `/gsd-verify-work` | -> **Nota:** Alguns workflows não têm comando direto voltado ao usuário (p. ex. `execute-plan.md`, `verify-phase.md`, `transition.md`, `node-repair.md`, `diagnose-issues.md`) — eles são invocados internamente por workflows orquestradores. `discovery-phase.md` é uma entrada alternativa para `/gsd-new-project`. +> **Nota:** Alguns workflows não têm comando direto voltado ao usuário (p. ex. `execute-plan.md`, `transition.md`, `node-repair.md`, `diagnose-issues.md`) — eles são invocados internamente por workflows orquestradores. `discovery-phase.md` é uma entrada alternativa para `/gsd-new-project`. --- @@ -280,6 +279,7 @@ Registro completo em `gsd-core/references/*.md`. Referências são documentos de | `model-profile-resolution.md` | Documentação do algoritmo de resolução de modelo. | | `verification-patterns.md` | Como verificar diferentes tipos de artefato. | | `verification-overrides.md` | Regras de substituição de verificação por artefato. | +| `verifier-phase-gates.md` | Gates de verificação carregados eager pelo gsd-verifier (migrados do workflow verify-phase aposentado, #1892): validação de cobertura de decisões (#2492), auditoria de qualidade de testes, escopo de human-verification para fases de infraestrutura (#2504). | | | `planning-config.md` | Esquema completo de configuração e comportamento. | | `git-integration.md` | Padrões de commit git, ramificação e histórico. | | `git-planning-commit.md` | Convenções de commit do diretório de planejamento. | diff --git a/docs/zh-CN/INVENTORY.md b/docs/zh-CN/INVENTORY.md index 48cb2d98b..4cd5c913e 100644 --- a/docs/zh-CN/INVENTORY.md +++ b/docs/zh-CN/INVENTORY.md @@ -259,10 +259,9 @@ | `thread.md` | 为跨会话工作创建、列出、关闭或恢复持久上下文线程。 | `/gsd-thread` | | `update.md` | 将 GSD 更新到最新版本并显示变更日志。 | `/gsd-update` | | `validate-phase.md` | 回溯审计并填补已完成阶段的奈奎斯特验证空缺。 | `/gsd-validate-phase` | -| `verify-phase.md` | 通过目标反向分析验证阶段目标的达成情况。 | `execute-phase.md`(执行后) | | `verify-work.md` | 带自动诊断的对话式 UAT — 生成 UAT.md 和修复计划。 | `/gsd-verify-work` | -> **注意:** 某些工作流没有直接面向用户的命令(例如 `execute-plan.md`、`verify-phase.md`、`transition.md`、`node-repair.md`、`diagnose-issues.md`)— 它们由编排器工作流在内部调用。`discovery-phase.md` 是 `/gsd-new-project` 的备用入口。 +> **注意:** 某些工作流没有直接面向用户的命令(例如 `execute-plan.md`、`transition.md`、`node-repair.md`、`diagnose-issues.md`)— 它们由编排器工作流在内部调用。`discovery-phase.md` 是 `/gsd-new-project` 的备用入口。 --- @@ -280,6 +279,7 @@ | `model-profile-resolution.md` | 模型解析算法文档。 | | `verification-patterns.md` | 如何验证不同的产物类型。 | | `verification-overrides.md` | 每种产物的验证覆盖规则。 | +| `verifier-phase-gates.md` | 由 gsd-verifier 急切加载的验证期门禁(自已退役的 verify-phase 工作流迁移,#1892):决策覆盖校验(#2492)、测试质量审计、基础设施阶段的 human-verification 划定(#2504)。 | | | `planning-config.md` | 完整的配置模式和行为。 | | `git-integration.md` | Git 提交、分支和历史模式。 | | `git-planning-commit.md` | 规划目录提交约定。 | diff --git a/gsd-core/references/planning-config.md b/gsd-core/references/planning-config.md index 6ffef613a..9beedde7f 100644 --- a/gsd-core/references/planning-config.md +++ b/gsd-core/references/planning-config.md @@ -36,7 +36,7 @@ Configuration options for `.planning/` directory behavior. | `git.quick_branch_template` | `null` | Optional branch template for quick-task runs | | `workflow.use_worktrees` | `true` | Whether executor agents run in isolated git worktrees. Set to `false` to disable worktrees — agents execute sequentially on the main working tree instead. Recommended for solo developers or when worktree merges cause issues. Note: if your branch is ahead of `origin/HEAD` (a diverged milestone or feature branch), GSD auto-degrades to sequential and prints a warning; set `worktree.baseRef:"head"` in `.claude/settings.local.json` to restore parallel execution. See the branch-divergence note below. | | `workflow.subagent_timeout` | `300000` | Timeout in milliseconds for parallel subagent tasks (e.g. codebase mapping). Increase for large codebases or slower models. Default: 300000 (5 minutes). | -| `workflow.test_command` | `null` | Custom shell command run as the regression/test gate by verify-phase, execute-phase, audit-fix, and post-merge-gate. When unset, GSD auto-detects (Makefile / package.json / Cargo.toml / go.mod / pyproject.toml). Example: `npm test`. | +| `workflow.test_command` | `null` | Custom shell command run as the regression/test gate by execute-phase, audit-fix, and post-merge-gate. When unset, GSD auto-detects (Makefile / package.json / Cargo.toml / go.mod / pyproject.toml). Example: `npm test`. | | `workflow.build_command` | `null` | Custom shell command run as the build gate by the post-merge gate. When unset, the build step is skipped/auto-detected. Example: `npm run build`. | | `workflow.inline_plan_threshold` | `2` | Plans with this many tasks or fewer execute inline (Pattern C) instead of spawning a subagent. Avoids ~14K token spawn overhead for small plans. Set to `0` to always spawn subagents. | | `manager.flags.discuss` | `""` | Flags passed to `/gsd:discuss-phase` when dispatched from manager (e.g. `"--auto --analyze"`) | @@ -266,7 +266,7 @@ Set via `workflow.*` namespace in config.json (e.g., `"workflow": { "research": | `workflow.skip_discuss` | boolean | `false` | `true`, `false` | Skip discuss phase entirely | | `workflow.use_worktrees` | boolean | `true` | `true`, `false` | Run executor agents in isolated git worktrees | | `workflow.subagent_timeout` | number | `300000` | Any positive integer (ms) | Timeout for parallel subagent tasks (default: 5 minutes) | -| `workflow.test_command` | string\|null | `null` | Any shell command | Regression/test gate command run by verify-phase, execute-phase, audit-fix, and post-merge-gate. Unset → GSD auto-detects (Makefile / package.json / Cargo.toml / go.mod / pyproject.toml). | +| `workflow.test_command` | string\|null | `null` | Any shell command | Regression/test gate command run by execute-phase, audit-fix, and post-merge-gate. Unset → GSD auto-detects (Makefile / package.json / Cargo.toml / go.mod / pyproject.toml). | | `workflow.build_command` | string\|null | `null` | Any shell command | Build gate command run by the post-merge gate. Unset → build step auto-detected/skipped. | | `workflow.mvp_mode` | boolean | `false` | `true`, `false` | Persist the MVP-mode flag in config so every phase defaults to MVP framing without requiring `--mvp` on the CLI. Resolved via the chain: `--mvp` CLI flag → ROADMAP.md `**Mode:** mvp` field → this config value → `false`. When `true`, the planner, executor, verifier, and discovery surfaces (progress, stats, graphify) all treat the phase as an MVP vertical slice (UI → API → DB) of one user-visible capability. | | `workflow.context_guard_mode` | string | `"warn"` | `"auto"`, `"warn"`, `"off"` | Context exhaustion guard mode for `execute-phase`. Before each wave, the orchestrator self-assesses context pressure using degradation signals from `context-budget.md`. `"warn"` (default): emit a warning and recommend `/gsd:pause-work` when POOR tier is detected. `"auto"`: automatically invoke `/gsd:pause-work` before the next wave when POOR tier is detected. `"off"`: disable the guard. The guard is heuristic — no programmatic context-% API exists. | diff --git a/gsd-core/references/verifier-phase-gates.md b/gsd-core/references/verifier-phase-gates.md new file mode 100644 index 000000000..ef36459e2 --- /dev/null +++ b/gsd-core/references/verifier-phase-gates.md @@ -0,0 +1,175 @@ +# Verifier Phase Gates + +> Loaded eagerly by `agents/gsd-verifier.md` (``). Carries the three +> verification-time gates that lived in the retired `verify-phase` workflow +> (#1892 / epic #1891 F7): decision-coverage validation (#2492), the test-quality audit, +> and infrastructure-phase human-verification scoping (#2504). Run each at its named +> agent step; `gsd_run` is the launcher shim defined in the agent's own Step 1 block. + +## verify_decisions — Decision Coverage Gate (run after Step 6, requirements coverage) + + +**Decision coverage validation gate (issue #2492).** + +After requirements coverage, also check that each trackable CONTEXT.md +`` entry shows up somewhere in the shipped artifacts (plans, +SUMMARY.md, files modified by the phase, or recent commit subjects on the +phase branch). + +This gate is **non-blocking / warning only** by deliberate asymmetry with +the plan-phase translation gate. The plan-phase gate already blocked at +translation time, so by the time verification runs every decision has +either been translated or explicitly deferred. This gate's job is to +surface decisions that *were* translated but vanished during execution — +that's a soft signal because "honors a decision" is a fuzzy substring +heuristic, and we don't want a paraphrase miss to fail an otherwise good +phase. + +**Skip if** `workflow.context_coverage_gate` is explicitly set to `false` +(absent key = enabled). Also skip cleanly when CONTEXT.md is missing or has +no `` block. + +```bash +GATE_CFG=$(gsd_run query config-get workflow.context_coverage_gate 2>/dev/null || echo "true") +if [ "$GATE_CFG" != "false" ]; then + CONTEXT_PATH=$(ls "${PHASE_DIR}"/*-CONTEXT.md 2>/dev/null | head -1) # #2962: not a for-glob (zsh aborts) + DECISION_RESULT=$(gsd_run query check.decision-coverage-verify "${PHASE_DIR}" "${CONTEXT_PATH}") +fi +``` + +The handler returns JSON `{ skipped, blocking: false, total, honored, +not_honored: [...], message }`. + +**Reporting:** Append the handler's `message` (a `### Decision Coverage` +section) to VERIFICATION.md regardless of outcome — even when all +decisions are honored, recording the count helps reviewers spot drift over +time. Set `decision_coverage` in the verification result to +`{honored, total, not_honored: [...]}` so downstream tooling can read it. + +**Status impact:** none. The decision gate does NOT influence the +`gaps_found` / `human_needed` / `passed` decision tree in Step 9. Its +findings are warnings the user reviews and may act on by re-opening the +phase or by acknowledging the decision was abandoned intentionally. + + +## audit_test_quality (run after Step 7b, alongside anti-patterns) + + +**Verify that tests PROVE what they claim to prove.** + +This step catches test-level deceptions that pass all prior checks: files exist, are substantive, are wired, and tests pass — but the tests don't actually validate the requirement. + +**1. Identify requirement-linked test files** + +From PLAN and SUMMARY files, map each requirement to the test files that are supposed to prove it. + +**2. Disabled test scan** + +For ALL test files linked to requirements, search for disabled/skipped patterns: + +```bash +grep -rn -E "it\.skip|describe\.skip|test\.skip|xit\(|xdescribe\(|xtest\(|@pytest\.mark\.skip|@unittest\.skip|#\[ignore\]|\.pending|it\.todo|test\.todo" "$TEST_FILE" +``` + +**Rule:** A disabled test linked to a requirement = requirement NOT tested. +- 🛑 BLOCKER if the disabled test is the only test proving that requirement +- ⚠️ WARNING if other active tests also cover the requirement + +**3. Circular test detection** + +Search for scripts/utilities that generate expected values by running the system under test: + +```bash +grep -rn -E "writeFileSync|writeFile|fs\.write|open\(.*w\)" "$TEST_DIRS" +``` + +For each match, check if it also imports the system/service/module being tested. If a script both imports the system-under-test AND writes expected output values → CIRCULAR. + +**Circular test indicators:** +- Script imports a service AND writes to fixture files +- Expected values have comments like "computed from engine", "captured from baseline" +- Script filename contains "capture", "baseline", "generate", "snapshot" in test context +- Expected values were added in the same commit as the test assertions + +**Rule:** A test comparing system output against values generated by the same system is circular. It proves consistency, not correctness. + +**4. Expected value provenance** (for comparison/parity/migration requirements) + +When a requirement demands comparison with an external source ("identical to X", "matches Y", "same output as Z"): + +- Is the external source actually invoked or referenced in the test pipeline? +- Do fixture files contain data sourced from the external system? +- Or do all expected values come from the new system itself or from mathematical formulas? + +**Provenance classification:** +- VALID: Expected value from external/legacy system output, manual capture, or independent oracle +- PARTIAL: Expected value from mathematical derivation (proves formula, not system match) +- CIRCULAR: Expected value from the system being tested +- UNKNOWN: No provenance information — treat as SUSPECT + +**5. Assertion strength** + +For each test linked to a requirement, classify the strongest assertion: + +| Level | Examples | Proves | +|-------|---------|--------| +| Existence | `toBeDefined()`, `!= null` | Something returned | +| Type | `typeof x === 'number'` | Correct shape | +| Status | `code === 200` | No error | +| Value | `toEqual(expected)`, `toBeCloseTo(x)` | Specific value | +| Behavioral | Multi-step workflow assertions | End-to-end correctness | + +If a requirement demands value-level or behavioral-level proof and the test only has existence/type/status assertions → INSUFFICIENT. + +**6. Coverage quantity** + +If a requirement specifies a quantity of test cases (e.g., "30 calculations"), check if the actual number of active (non-skipped) test cases meets the requirement. + +**Reporting — add to VERIFICATION.md:** + +```markdown +### Test Quality Audit + +| Test File | Linked Req | Active | Skipped | Circular | Assertion Level | Verdict | +|-----------|-----------|--------|---------|----------|-----------------|---------| + +**Disabled tests on requirements:** {N} → {BLOCKER if any req has ONLY disabled tests} +**Circular patterns detected:** {N} → {BLOCKER if any} +**Insufficient assertions:** {N} → {WARNING} +``` + +**Impact on status:** Any BLOCKER from test quality audit → overall status = `gaps_found` (Step 9 rule 1), regardless of other checks passing. + + +## identify_human_verification — infrastructure/foundation scoping (apply at Step 8) + +**First: determine if this is an infrastructure/foundation phase.** + +Infrastructure and foundation phases — code foundations, database schema, internal APIs, data models, build tooling, CI/CD, internal service integrations — have no user-facing elements by definition. For these phases: + +- Do NOT invent artificial manual steps (e.g., "manually run git commits", "manually invoke methods", "manually check database state"). +- Mark human verification as **N/A** with rationale: "Infrastructure/foundation phase — no user-facing elements to test manually." +- Set `human_verification: []` and do **not** produce a `human_needed` status solely due to lack of user-facing features. +- Only add human verification items if the phase goal or success criteria explicitly describe something a user would interact with (UI, CLI command output visible to end users, external service UX). +- **Exception — behavior-unverified truths still count.** A truth marked ⚠️ PRESENT_BEHAVIOR_UNVERIFIED (a state transition or a cancellation/cleanup/ordering invariant with no test exercising it) is a behavioral-evidence gap, not an artificial user-facing step. Record it in `behavior_unverified_items` and emit a human-verification item for it **even on an infrastructure/foundation phase** — these invariants are exactly where infra phases hide runtime state leaks. Such a truth drives `human_needed`; the auto-pass-UAT shortcut applies only to the absence of user-facing UX, never to a behavior-unverified invariant. + +**How to determine if a phase is infrastructure/foundation:** +- Phase goal or name contains: "foundation", "infrastructure", "schema", "database", "internal API", "data model", "scaffolding", "pipeline", "tooling", "CI", "migrations", "service layer", "backend", "core library" +- Phase success criteria describe only technical artifacts (files exist, tests pass, schema is valid) with no user interaction required +- There is no UI, CLI output visible to end users, or real-time behavior to observe + +**If the phase IS infrastructure/foundation:** auto-pass UAT — skip the human verification items list entirely, **except any ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth (see exception above), which still emits a human-verification item and drives `human_needed`.** Log: + +```markdown +## Human Verification + +N/A — Infrastructure/foundation phase with no user-facing elements. +All acceptance criteria are verifiable programmatically. +``` + +**If the phase IS user-facing:** only flag items that genuinely require a human — per the Step 8 always/uncertain lists already in the agent. Do not invent steps. + +## Lazy references + +- **Per-stack verification patterns:** before Step 4 (artifact verification) on an unfamiliar stack, Read `~/.claude/gsd-core/references/verification-patterns.md` — the grep catalog for React/Next.js components, API routes, database schema, and the universal stub patterns. Read it lazily (only the sections for the stack under verification); it is too large to load wholesale on every run. +- **Canonical report shape:** the emitted VERIFICATION.md follows `@~/.claude/gsd-core/templates/verification-report.md` — the template whose Guidelines and row shapes `src/uat.cts` treats as canonical when consuming verification output. diff --git a/gsd-core/templates/phase-prompt.md b/gsd-core/templates/phase-prompt.md index bdf9efd6b..d4427e1dd 100644 --- a/gsd-core/templates/phase-prompt.md +++ b/gsd-core/templates/phase-prompt.md @@ -606,5 +606,3 @@ Task completion ≠ Goal achievement. A task "create chat component" can complet 4. Verification subagent checks must_haves against codebase 5. Gaps found → fix plans created → execute → re-verify 6. All must_haves pass → phase complete - -See `~/.claude/gsd-core/workflows/verify-phase.md` for verification logic. diff --git a/gsd-core/workflows/code-review.md b/gsd-core/workflows/code-review.md index 9adc0f639..91a263eec 100644 --- a/gsd-core/workflows/code-review.md +++ b/gsd-core/workflows/code-review.md @@ -306,7 +306,7 @@ fi **Post-processing (all tiers):** -1. **Expand tilde paths:** SUMMARY.md `key-files` entries may record a `~/...`-prefixed path (e.g. `~/.claude/gsd-core/workflows/verify-phase.md`). Bash only tilde-expands a literal `~` written in source text, never one arriving as the value of an already-expanded variable, so every later `[ -f "$file" ]` check must see a real, expanded path or it misclassifies the file as deleted. +1. **Expand tilde paths:** SUMMARY.md `key-files` entries may record a `~/...`-prefixed path (e.g. `~/.claude/gsd-core/workflows/verify-work.md`). Bash only tilde-expands a literal `~` written in source text, never one arriving as the value of an already-expanded variable, so every later `[ -f "$file" ]` check must see a real, expanded path or it misclassifies the file as deleted. ```bash EXPANDED_FILES=() for file in "${REVIEW_FILES[@]}"; do diff --git a/gsd-core/workflows/verify-phase.md b/gsd-core/workflows/verify-phase.md deleted file mode 100644 index a4ac515e2..000000000 --- a/gsd-core/workflows/verify-phase.md +++ /dev/null @@ -1,574 +0,0 @@ - -Verify phase goal achievement through goal-backward analysis. Check that the codebase delivers what the phase promised, not just that tasks completed. - -Executed by a verification subagent spawned from execute-phase.md. - - - -**Task completion ≠ Goal achievement** - -A task "create chat component" can be marked complete when the component is a placeholder. The task was done — but the goal "working chat interface" was not achieved. - -Goal-backward verification: -1. What must be TRUE for the goal to be achieved? -2. What must EXIST for those truths to hold? -3. What must be WIRED for those artifacts to function? -4. What must TESTS PROVE for those truths to be evidenced? - -Then verify each level against the actual codebase. - - - -@~/.claude/gsd-core/references/verification-patterns.md -@~/.claude/gsd-core/templates/verification-report.md - - - - - -Load phase operation context: - -```bash -_GSD_SHIM_NAME="gsd-tools.cjs"; _GSD_RUNTIME_ROOT="${RUNTIME_DIR:-$(git rev-parse --show-toplevel 2>/dev/null || pwd)}"; GSD_TOOLS="${_GSD_RUNTIME_ROOT}/gsd-core/bin/${_GSD_SHIM_NAME}"; if [ -f "$GSD_TOOLS" ]; then gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${_GSD_RUNTIME_ROOT}/.claude/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${_GSD_RUNTIME_ROOT}/.claude/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${_GSD_RUNTIME_ROOT}/.codex/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${_GSD_RUNTIME_ROOT}/.codex/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif command -v gsd-tools >/dev/null 2>&1; then GSD_TOOLS="$(command -v gsd-tools)"; gsd_run() { "$GSD_TOOLS" "$@"; }; elif [ -f "${CLAUDE_CONFIG_DIR:-$HOME/.claude}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${CLAUDE_CONFIG_DIR:-$HOME/.claude}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${HERMES_HOME:-$HOME/.hermes}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${HERMES_HOME:-$HOME/.hermes}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${CURSOR_CONFIG_DIR:-$HOME/.cursor}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${CURSOR_CONFIG_DIR:-$HOME/.cursor}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${CODEX_HOME:-$HOME/.codex}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${CODEX_HOME:-$HOME/.codex}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${GEMINI_CONFIG_DIR:-$HOME/.gemini}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${GEMINI_CONFIG_DIR:-$HOME/.gemini}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${COPILOT_CONFIG_DIR:-$HOME/.copilot}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${COPILOT_CONFIG_DIR:-$HOME/.copilot}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${WINDSURF_CONFIG_DIR:-$HOME/.codeium/windsurf}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${WINDSURF_CONFIG_DIR:-$HOME/.codeium/windsurf}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${AUGMENT_CONFIG_DIR:-$HOME/.augment}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${AUGMENT_CONFIG_DIR:-$HOME/.augment}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${TRAE_CONFIG_DIR:-$HOME/.trae}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${TRAE_CONFIG_DIR:-$HOME/.trae}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${QWEN_CONFIG_DIR:-$HOME/.qwen}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${QWEN_CONFIG_DIR:-$HOME/.qwen}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${CODEBUDDY_CONFIG_DIR:-$HOME/.codebuddy}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${CODEBUDDY_CONFIG_DIR:-$HOME/.codebuddy}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${CLINE_CONFIG_DIR:-$HOME/.cline}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${CLINE_CONFIG_DIR:-$HOME/.cline}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${GROK_AGENTS_HOME:-$HOME/.agents}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${GROK_AGENTS_HOME:-$HOME/.agents}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${ANTIGRAVITY_CONFIG_DIR:-$HOME/.gemini/antigravity}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${ANTIGRAVITY_CONFIG_DIR:-$HOME/.gemini/antigravity}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${OPENCODE_CONFIG_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/opencode}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${OPENCODE_CONFIG_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/opencode}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; elif [ -f "${KILO_CONFIG_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/kilo}/gsd-core/bin/${_GSD_SHIM_NAME}" ]; then GSD_TOOLS="${KILO_CONFIG_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/kilo}/gsd-core/bin/${_GSD_SHIM_NAME}"; gsd_run() { node "$GSD_TOOLS" "$@"; }; else echo "ERROR: gsd-tools.cjs not found at $GSD_TOOLS and gsd-tools is not on PATH. Run: npx -y @opengsd/gsd-core@latest --claude --local" >&2; exit 1; fi; if [ -n "${CLAUDE_ENV_FILE:-}" ] && [ -n "${GSD_TOOLS:-}" ]; then printf "export PATH='%s':\"\$PATH\"\n" "${GSD_TOOLS%/*}" >> "$CLAUDE_ENV_FILE" 2>/dev/null || true; fi -INIT=$(gsd_run query init.phase-op "${PHASE_ARG}") -if [[ "$INIT" == @file:* ]]; then INIT=$(cat "${INIT#@file:}"); fi -``` - -Extract from init JSON: `phase_dir`, `phase_number`, `phase_name`, `has_plans`, `plan_count`. - -Then load phase details and list plans/summaries: -```bash -gsd_run query roadmap.get-phase "${phase_number}" -grep -E "^| ${phase_number}" .planning/REQUIREMENTS.md 2>/dev/null || true -ls "$phase_dir"/*-SUMMARY.md "$phase_dir"/*-PLAN.md 2>/dev/null || true -``` - -Load full milestone phases for deferred-item filtering (Step 9b): -```bash -gsd_run query roadmap.analyze -``` - -Extract **phase goal** from ROADMAP.md (the outcome to verify, not tasks), **requirements** from REQUIREMENTS.md if it exists, and **all milestone phases** from roadmap analyze (for cross-referencing gaps against later phases). - - - -**Option A: Must-haves in PLAN frontmatter** - -Use `gsd-tools.cjs query` verify handlers (or legacy gsd-tools) to extract must_haves from each PLAN: - -```bash -shopt -s nullglob 2>/dev/null; setopt NULL_GLOB 2>/dev/null -for plan in "$PHASE_DIR"/*-PLAN.md; do - MUST_HAVES=$(gsd_run query frontmatter.get "$plan" --field must_haves) - echo "=== $plan ===" && echo "$MUST_HAVES" -done -``` - -Returns JSON: `{ truths: [...], artifacts: [...], key_links: [...], prohibitions: [...] }` - -Aggregate all must_haves across plans for phase-level verification. - -**Prohibitions (`must_haves.prohibitions`, ADR-550 D3 — the must-NOT sibling block):** When a plan carries `must_haves.prohibitions`, extract each `{ statement, status, verification }` item and route it by `verification` tier in verdict assembly (ADR-550 D4, "B-with-guard", 2026-06-12 maintainer decision). These are NEGATIVE checks (the must-NOT must NOT have happened), distinct from positive `truths`: - -- **judgment-tier → mode-dependent soft-gate.** Interactive verify defers each item to the end-of-phase human checkpoint (`human_verify_mode: end-of-phase`). Autonomous verify records a NON-AUTHORITATIVE LLM-judge verdict + a prominent `unverified-prohibition — human review recommended` flag (autonomous completion reads "complete with N flagged prohibitions"). NEVER a silent pass; NEVER a hard halt of an AFK run. -- **test-tier → ENFORCED via `check prohibition-enforcement` (green on pass, hard-gate on miss/fail).** Accept the `verification: test` value (the SPEC↔must_haves.prohibitions projection contract holds — no forced schema change later). For each test-tier item, the verifier builds `request.check` **DETERMINISTICALLY from the projected descriptor** — it does NOT invent `{ kind, target, rule }`. Read the flat scalar keys `check_kind` / `check_target` / `check_rule` / `check_violation_fixture` off the `must_haves.prohibitions` item and reconstruct the `CheckDescriptor` via the `descriptorFromProjection` adapter in `prohibition-enforcement` (`descriptorFromProjection(projectedItem)` → `{ kind: check_kind, target: check_target, rule?: check_rule, violationFixture?: check_violation_fixture }`). The `violationFixture` (a path to a KNOWN-BAD subject) is the field that gates **green** and it is **now projected** (`check_violation_fixture`, #1346) — so a prohibition authored with all four scalars greens through the projection alone, **zero hand-authoring at verify time**. Do NOT rely on `failFirst`: it is DEMOTED (#1279) and greens nothing on its own; an item with no projected fixture hard-gates fail-closed. Invoke the producer (CLI surface unchanged): - - ```bash - gsd_run check prohibition-enforcement - ``` - - where `` carries `{ prohibition, check, mode }` — `check` being the wired mechanical-check descriptor `{ kind: 'node-test' | 'lint-rule', target, rule?, violationFixture, cleanFixture?, failFirst? }`, with `kind`/`target`/`rule`/`violationFixture`/`cleanFixture` now sourced from the projected `check_*` scalars (not author/verifier invention — #1278 + #1279 + #1346). For `node-test`, `target` (from `check_target`) is the negative-test file path; for `lint-rule`, `target` is the PATH to lint and `rule` (from `check_rule`) is the eslint rule id (e.g. `local/no-source-grep`) — both required (a lint-rule without `rule` is not a valid wired check). `violationFixture` (from `check_violation_fixture`) is the path to a KNOWN-BAD subject the producer runs the check against to **machine-prove fail-first** (for `node-test`, injected via the `GSD_PROHIB_SUBJECT` env convention — #1279); the optional `cleanFixture` (from `check_clean_fixture`) is a KNOWN-CLEAN control subject the `node-test` prover ALSO requires to stay GREEN, proving the RED is content-caused (#1346); `failFirst` is a DEMOTED, non-authoritative hint kept only for backward route-JSON shape (no path greens on it alone — FF-08). The producer LOCATES the wired check from the projection, **machine-proves it is fail-first** by running it against the violation and confirming it goes RED, RUNS it for a genuine non-vacuous pass, builds `enforcementEvidence`, and emits the `dispositionForProhibition()` verdict (#1259 + #1278 + #1279, ADR-550 D5d). Fail-first is **machine-proven, not caller-attested** — absent a provable violation the producer fails closed, never falling back to attestation. Route the result by its typed fields: - - **`status: 'green'`, `flagged: false`** (a genuinely-passing wired negative test / lint rule, `located: true`, non-empty `evidence`) → the item is satisfiable → it can reach **passed**. - - **missing, non-attested, or genuinely-non-passing check** (`located: false` OR `status: 'unverified'`, `flagged: true`) → **hard-gate**: disposes flagged-unverified, NEVER green, routing to `gaps_found` in BOTH interactive and autonomous modes (a failing mechanical check blocks even AFK; ADR-550 D4 / D3). The deterministic fail-closed default backing every miss/fail is `dispositionForProhibition()` in probe-core (`status: 'unverified'`, `flagged: true` on empty `enforcementEvidence`). - - > **Descriptor source — deterministic locate + machine-proof compose (#1278 + #1346, DELIVERED).** The `check` descriptor's `{ kind, target, rule, violationFixture }` is now sourced **deterministically from the projected `check_kind` / `check_target` / `check_rule` / `check_violation_fixture` scalars** on the `must_haves.prohibitions` item (authored at `/gsd:spec-phase`, projected by `projectProhibitions`, read back via the `descriptorFromProjection` adapter). So both halves close with **zero manual descriptor authoring** — the verifier neither invents the locate (#1278) nor hand-supplies the violation fixture (#1346): a prohibition authored with all four scalars machine-proves fail-first and greens end-to-end through the projection alone (removing the spoofable invent-at-verify-time surface; ADR-857 §147 exogenous grading). **Fail-closed is preserved:** an item with NO projected descriptor, a PARTIAL one (e.g. a `lint-rule` missing `check_rule`), OR a descriptor with **no `check_violation_fixture`** makes `descriptorFromProjection` return `null` / an under-specified or fixture-less descriptor, which falls through to the producer's fail-closed paths (`located: false`, or located-but-unprovable) → flagged-unverified, NEVER green, in BOTH modes. `failFirst` is demoted and greens nothing on its own (#1279, FF-08). Causation (**#1346**): supplying `check_clean_fixture` adds an opt-in control — the `node-test` prover also requires GREEN on a known-clean subject, proving the RED is content-caused; with no clean fixture that one residual case (a deceptive test reding merely because the env var is set) stays a documented constraint, an author opting into the stronger proof by wiring a clean control. - -**Option B: Use Success Criteria from ROADMAP.md** - -If no must_haves in frontmatter (MUST_HAVES returns error or empty), check for Success Criteria: - -```bash -PHASE_DATA=$(gsd_run query roadmap.get-phase "${phase_number}" --raw) -``` - -Parse the `success_criteria` array from the JSON output. If non-empty: -1. Use each Success Criterion directly as a **truth** (they are already written as observable, testable behaviors) -2. Derive **artifacts** (concrete file paths for each truth) -3. Derive **key links** (critical wiring where stubs hide) -4. Document the must-haves before proceeding - -Success Criteria from ROADMAP.md are the contract — they override PLAN-level must_haves when both exist. - -**Option C: Derive from phase goal (fallback)** - -If no must_haves in frontmatter AND no Success Criteria in ROADMAP: -1. State the goal from ROADMAP.md -2. Derive **truths** (3-7 observable behaviors, each testable) -3. Derive **artifacts** (concrete file paths for each truth) -4. Derive **key links** (critical wiring where stubs hide) -5. Document derived must-haves before proceeding - - - -For each observable truth, determine if the codebase enables it. - -**Status:** ✓ VERIFIED (all supporting artifacts pass — and, for a behavior-dependent truth, a behavioral test exercises the asserted behavior) | ⚠️ PRESENT_BEHAVIOR_UNVERIFIED (present + wired, but a state transition or cancellation/cleanup/ordering invariant is exercised by no test — routes to human verification, excluded from the score) | ✗ FAILED (artifact missing/stub/unwired) | ? UNCERTAIN (needs human) - -For each truth: identify supporting artifacts → check artifact status → check wiring → determine truth status. - -**Behavior-dependent truths:** when a truth asserts a state transition or a cancellation/cleanup/ordering invariant, symbol presence + wiring is necessary but not sufficient — the code can be present and wired yet still leak state on the path the invariant covers. Mark such a truth ✓ VERIFIED only when a pre-existing test exercises the transition/invariant and passes (one named test, never the full suite); otherwise mark it ⚠️ PRESENT_BEHAVIOR_UNVERIFIED, emit a human-verification item, and exclude it from the verified score. - -**Non-inferable (`backstop`) truths (#1154):** a `must_haves.truths` item in object form `{ statement, verification: backstop }` is non-inferable — the correct behavior is not derivable from the spec alone, so the verifier cannot self-detect the gap and would false-pass it confidently. Branch on the `verification: backstop` field (read via `truthVerification()`, never prose): if confirmable with **explicit evidence** (a passing wired held-out/property test, or a directly-observed behavior) → ✓ VERIFIED; otherwise **abstain** — mark ⚠️ `insufficient_spec`, emit an `unverified — held-out test recommended` human-verification item, exclude from the verified score (routes to `human_needed`). Exogenous only (never a self-judged "abstain if unsure"); an inferable truth is never abstained. See `references/honest-verifier.md`. - -**Example:** Truth "User can see existing messages" depends on Chat.tsx (renders), /api/chat GET (provides), Message model (schema). If Chat.tsx is a stub or API returns hardcoded [] → FAILED. If all exist, are substantive, and connected → VERIFIED. - - - -Use `gsd-tools.cjs query verify.artifacts` (or legacy gsd-tools) for artifact verification against must_haves in each PLAN: - -```bash -shopt -s nullglob 2>/dev/null; setopt NULL_GLOB 2>/dev/null -for plan in "$PHASE_DIR"/*-PLAN.md; do - ARTIFACT_RESULT=$(gsd_run query verify.artifacts "$plan") - echo "=== $plan ===" && echo "$ARTIFACT_RESULT" -done -``` - -Parse JSON result: `{ all_passed, passed, total, artifacts: [{path, exists, issues, passed}] }` - -**Artifact status from result:** -- `exists=false` → MISSING -- `issues` not empty → STUB (check issues for "Only N lines" or "Missing pattern") -- `passed=true` → VERIFIED (Levels 1-2 pass) - -**Level 3 — Wired (manual check for artifacts that pass Levels 1-2):** -```bash -grep -r "import.*$artifact_name" src/ --include="*.ts" --include="*.tsx" # IMPORTED -grep -r "$artifact_name" src/ --include="*.ts" --include="*.tsx" | grep -v "import" # USED -``` -WIRED = imported AND used. ORPHANED = exists but not imported/used. - -| Exists | Substantive | Wired | Status | -|--------|-------------|-------|--------| -| ✓ | ✓ | ✓ | ✓ VERIFIED | -| ✓ | ✓ | ✗ | ⚠️ ORPHANED | -| ✓ | ✗ | - | ✗ STUB | -| ✗ | - | - | ✗ MISSING | - -**Export-level spot check (WARNING severity):** - -For artifacts that pass Level 3, spot-check individual exports: -- Extract key exported symbols (functions, constants, classes — skip types/interfaces) -- For each, grep for usage outside the defining file -- Flag exports with zero external call sites as "exported but unused" - -This catches dead stores like `setPlan()` that exist in a wired file but are -never actually called. Report as WARNING — may indicate incomplete cross-plan -wiring or leftover code from plan revisions. - - - -Use `gsd-tools.cjs query verify.key-links` (or legacy gsd-tools) for key link verification against must_haves in each PLAN: - -```bash -shopt -s nullglob 2>/dev/null; setopt NULL_GLOB 2>/dev/null -for plan in "$PHASE_DIR"/*-PLAN.md; do - LINKS_RESULT=$(gsd_run query verify.key-links "$plan") - echo "=== $plan ===" && echo "$LINKS_RESULT" -done -``` - -Parse JSON result: `{ all_verified, verified, total, links: [{from, to, via, verified, detail}] }` - -**Link status from result:** -- `verified=true` → WIRED -- `verified=false` with "not found" → NOT_WIRED -- `verified=false` with "Pattern not found" → PARTIAL - -**Fallback patterns (if key_links not in must_haves):** - -| Pattern | Check | Status | -|---------|-------|--------| -| Component → API | fetch/axios call to API path, response used (await/.then/setState) | WIRED / PARTIAL (call but unused response) / NOT_WIRED | -| API → Database | Prisma/DB query on model, result returned via res.json() | WIRED / PARTIAL (query but not returned) / NOT_WIRED | -| Form → Handler | onSubmit with real implementation (fetch/axios/mutate/dispatch), not console.log/empty | WIRED / STUB (log-only/empty) / NOT_WIRED | -| State → Render | useState variable appears in JSX (`{stateVar}` or `{stateVar.property}`) | WIRED / NOT_WIRED | - -Record status and evidence for each key link. - - - -If REQUIREMENTS.md exists: -```bash -grep -E "Phase ${PHASE_NUM}" .planning/REQUIREMENTS.md 2>/dev/null || true -``` - -For each requirement: parse description → identify supporting truths/artifacts → status: ✓ SATISFIED / ✗ BLOCKED / ? NEEDS HUMAN. - - - -**Decision coverage validation gate (issue #2492).** - -After requirements coverage, also check that each trackable CONTEXT.md -`` entry shows up somewhere in the shipped artifacts (plans, -SUMMARY.md, files modified by the phase, or recent commit subjects on the -phase branch). - -This gate is **non-blocking / warning only** by deliberate asymmetry with -the plan-phase translation gate. The plan-phase gate already blocked at -translation time, so by the time verification runs every decision has -either been translated or explicitly deferred. This gate's job is to -surface decisions that *were* translated but vanished during execution — -that's a soft signal because "honors a decision" is a fuzzy substring -heuristic, and we don't want a paraphrase miss to fail an otherwise good -phase. - -**Skip if** `workflow.context_coverage_gate` is explicitly set to `false` -(absent key = enabled). Also skip cleanly when CONTEXT.md is missing or has -no `` block. - -```bash -GATE_CFG=$(gsd_run query config-get workflow.context_coverage_gate 2>/dev/null || echo "true") -if [ "$GATE_CFG" != "false" ]; then - CONTEXT_PATH=$(ls "${PHASE_DIR}"/*-CONTEXT.md 2>/dev/null | head -1) # #2962: not a for-glob (zsh aborts) - DECISION_RESULT=$(gsd_run query check.decision-coverage-verify "${PHASE_DIR}" "${CONTEXT_PATH}") -fi -``` - -The handler returns JSON `{ skipped, blocking: false, total, honored, -not_honored: [...], message }`. - -**Reporting:** Append the handler's `message` (a `### Decision Coverage` -section) to VERIFICATION.md regardless of outcome — even when all -decisions are honored, recording the count helps reviewers spot drift over -time. Set `decision_coverage` in the verification result to -`{honored, total, not_honored: [...]}` so downstream tooling can read it. - -**Status impact:** none. The decision gate does NOT influence the -`gaps_found` / `human_needed` / `passed` decision tree in -`determine_status`. Its findings are warnings the user reviews and may act -on by re-opening the phase or by acknowledging the decision was abandoned -intentionally. - - - -**Run the project's test suite and CLI commands to verify behavior, not just structure.** - -Static checks (grep, file existence, wiring) catch structural gaps but miss runtime -failures. This step runs actual tests and project commands to verify the phase goal -is behaviorally achieved. - -This follows Anthropic's harness engineering principle: separating generation from -evaluation, with the evaluator interacting with the running system rather than -inspecting static artifacts. - -**Step 1: Run test suite** - -```bash -# Resolve test command: project config > Makefile > language sniff -TEST_CMD=$(gsd_run query config-get workflow.test_command --default "" --raw 2>/dev/null || true) -if [ -z "$TEST_CMD" ]; then - if [ -f "Makefile" ] && grep -q "^test:" Makefile; then - TEST_CMD="make test" - elif [ -f "Justfile" ] || [ -f "justfile" ]; then - TEST_CMD="just test" - elif [ -f "package.json" ]; then - TEST_CMD="npm test" - elif [ -f "Cargo.toml" ]; then - TEST_CMD="cargo test" - elif [ -f "go.mod" ]; then - TEST_CMD="go test ./..." - elif [ -f "pyproject.toml" ] || [ -f "requirements.txt" ]; then - TEST_CMD="python -m pytest -q --tb=short 2>&1 || uv run python -m pytest -q --tb=short" - else - TEST_CMD="false" - echo "⚠ No test runner detected — skipping test suite" - fi -fi -# Run all tests (timeout: 5 min). #1857: normalize to one-shot so watch mode exits. -TEST_CMD=$(gsd_run query normalize-test-command "$TEST_CMD" --cwd . 2>/dev/null || echo "$TEST_CMD") -TEST_EXIT=0 -gsd_run run-with-timeout 300 -- bash -c "$TEST_CMD" 2>&1 -TEST_EXIT=$? -if [ "${TEST_EXIT}" -eq 0 ]; then - echo "✓ Test suite passed" -elif [ "${TEST_EXIT}" -eq 124 ]; then - echo "⚠ Test suite timed out after 5 minutes — likely watch/dev mode" -else - echo "✗ Test suite failed (exit code ${TEST_EXIT})" -fi -``` - -Record: total tests, passed, failed, coverage (if available). - -**If any tests fail:** Mark as `behavioral_failures` — these are BLOCKER severity -regardless of whether static checks passed. A phase cannot be verified if tests fail. - -**Step 2: Run project CLI/commands from success criteria (if testable)** - -For each success criterion that describes a user command (e.g., "User can run -`mixtiq validate`", "User can run `npm start`"): - -1. Check if the command exists and required inputs are available: - - Look for example files in `templates/`, `fixtures/`, `test/`, `examples/`, or `testdata/` - - Check if the CLI binary/script exists on PATH or in the project -2. **If no suitable inputs or fixtures exist:** Mark as `? NEEDS HUMAN` with reason - "No test fixtures available — requires manual verification" and move on. - Do NOT invent example inputs. -3. If inputs are available: run the command and verify it exits successfully. - -```bash -# Only run if both command and input exist -if command -v {project_cli} &>/dev/null && [ -f "{example_input}" ]; then - {project_cli} {example_input} 2>&1 -fi -``` - -Record: command, exit code, output summary, pass/fail (or SKIPPED if no fixtures). - -**Step 3: Report** - -``` -## Behavioral Verification - -| Check | Result | Detail | -|-------|--------|--------| -| Test suite | {N} passed, {M} failed | {first failure if any} | -| {CLI command 1} | ✓ / ✗ | {output summary} | -| {CLI command 2} | ✓ / ✗ | {output summary} | -``` - -**If all behavioral checks pass:** Continue to scan_antipatterns. -**If any fail:** Add to verification gaps with BLOCKER severity. - - - -Extract files modified in this phase from SUMMARY.md, scan each: - -| Pattern | Search | Severity | -|---------|--------|----------| -| TBD/FIXME/XXX without same-line `issue #123`, `PR #123`, `#123`, or `DEF-*` reference | `grep -n -e TBD -e FIXME -e XXX` | 🛑 Blocker | -| TODO/HACK | `grep -n -e TODO -e HACK` | ⚠️ Warning | -| Placeholder content | `grep -n -iE "placeholder\|coming soon\|will be here"` | 🛑 Blocker | -| Empty returns | `grep -n -E "return null\|return \{\}\|return \[\]\|=> \{\}"` | ⚠️ Warning | -| Log-only functions | Functions containing only console.log | ⚠️ Warning | - -Categorize: 🛑 Blocker (prevents goal) | ⚠️ Warning (incomplete) | ℹ️ Info (notable). - - - -**Verify that tests PROVE what they claim to prove.** - -This step catches test-level deceptions that pass all prior checks: files exist, are substantive, are wired, and tests pass — but the tests don't actually validate the requirement. - -**1. Identify requirement-linked test files** - -From PLAN and SUMMARY files, map each requirement to the test files that are supposed to prove it. - -**2. Disabled test scan** - -For ALL test files linked to requirements, search for disabled/skipped patterns: - -```bash -grep -rn -E "it\.skip|describe\.skip|test\.skip|xit\(|xdescribe\(|xtest\(|@pytest\.mark\.skip|@unittest\.skip|#\[ignore\]|\.pending|it\.todo|test\.todo" "$TEST_FILE" -``` - -**Rule:** A disabled test linked to a requirement = requirement NOT tested. -- 🛑 BLOCKER if the disabled test is the only test proving that requirement -- ⚠️ WARNING if other active tests also cover the requirement - -**3. Circular test detection** - -Search for scripts/utilities that generate expected values by running the system under test: - -```bash -grep -rn -E "writeFileSync|writeFile|fs\.write|open\(.*w\)" "$TEST_DIRS" -``` - -For each match, check if it also imports the system/service/module being tested. If a script both imports the system-under-test AND writes expected output values → CIRCULAR. - -**Circular test indicators:** -- Script imports a service AND writes to fixture files -- Expected values have comments like "computed from engine", "captured from baseline" -- Script filename contains "capture", "baseline", "generate", "snapshot" in test context -- Expected values were added in the same commit as the test assertions - -**Rule:** A test comparing system output against values generated by the same system is circular. It proves consistency, not correctness. - -**4. Expected value provenance** (for comparison/parity/migration requirements) - -When a requirement demands comparison with an external source ("identical to X", "matches Y", "same output as Z"): - -- Is the external source actually invoked or referenced in the test pipeline? -- Do fixture files contain data sourced from the external system? -- Or do all expected values come from the new system itself or from mathematical formulas? - -**Provenance classification:** -- VALID: Expected value from external/legacy system output, manual capture, or independent oracle -- PARTIAL: Expected value from mathematical derivation (proves formula, not system match) -- CIRCULAR: Expected value from the system being tested -- UNKNOWN: No provenance information — treat as SUSPECT - -**5. Assertion strength** - -For each test linked to a requirement, classify the strongest assertion: - -| Level | Examples | Proves | -|-------|---------|--------| -| Existence | `toBeDefined()`, `!= null` | Something returned | -| Type | `typeof x === 'number'` | Correct shape | -| Status | `code === 200` | No error | -| Value | `toEqual(expected)`, `toBeCloseTo(x)` | Specific value | -| Behavioral | Multi-step workflow assertions | End-to-end correctness | - -If a requirement demands value-level or behavioral-level proof and the test only has existence/type/status assertions → INSUFFICIENT. - -**6. Coverage quantity** - -If a requirement specifies a quantity of test cases (e.g., "30 calculations"), check if the actual number of active (non-skipped) test cases meets the requirement. - -**Reporting — add to VERIFICATION.md:** - -```markdown -### Test Quality Audit - -| Test File | Linked Req | Active | Skipped | Circular | Assertion Level | Verdict | -|-----------|-----------|--------|---------|----------|----------------|---------| - -**Disabled tests on requirements:** {N} → {BLOCKER if any req has ONLY disabled tests} -**Circular patterns detected:** {N} → {BLOCKER if any} -**Insufficient assertions:** {N} → {WARNING} -``` - -**Impact on status:** Any BLOCKER from test quality audit ��� overall status = `gaps_found`, regardless of other checks passing. - - - -**First: determine if this is an infrastructure/foundation phase.** - -Infrastructure and foundation phases — code foundations, database schema, internal APIs, data models, build tooling, CI/CD, internal service integrations — have no user-facing elements by definition. For these phases: - -- Do NOT invent artificial manual steps (e.g., "manually run git commits", "manually invoke methods", "manually check database state"). -- Mark human verification as **N/A** with rationale: "Infrastructure/foundation phase — no user-facing elements to test manually." -- Set `human_verification: []` and do **not** produce a `human_needed` status solely due to lack of user-facing features. -- Only add human verification items if the phase goal or success criteria explicitly describe something a user would interact with (UI, CLI command output visible to end users, external service UX). -- **Exception — behavior-unverified truths still count.** A truth marked ⚠️ PRESENT_BEHAVIOR_UNVERIFIED (a state transition or a cancellation/cleanup/ordering invariant with no test exercising it) is a behavioral-evidence gap, not an artificial user-facing step. Record it in `behavior_unverified_items` and emit a human-verification item for it **even on an infrastructure/foundation phase** — these invariants are exactly where infra phases hide runtime state leaks. Such a truth drives `human_needed`; the auto-pass-UAT shortcut applies only to the absence of user-facing UX, never to a behavior-unverified invariant. - -**How to determine if a phase is infrastructure/foundation:** -- Phase goal or name contains: "foundation", "infrastructure", "schema", "database", "internal API", "data model", "scaffolding", "pipeline", "tooling", "CI", "migrations", "service layer", "backend", "core library" -- Phase success criteria describe only technical artifacts (files exist, tests pass, schema is valid) with no user interaction required -- There is no UI, CLI output visible to end users, or real-time behavior to observe - -**If the phase IS infrastructure/foundation:** auto-pass UAT — skip the human verification items list entirely, **except any ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth (see exception above), which still emits a human-verification item and drives `human_needed`.** Log: - -```markdown -## Human Verification - -N/A — Infrastructure/foundation phase with no user-facing elements. -All acceptance criteria are verifiable programmatically. -``` - -**If the phase IS user-facing:** Only flag items that genuinely require a human. Do not invent steps. - -**Always needs human (user-facing phases only):** Visual appearance, user flow completion, real-time behavior (WebSocket/SSE), external service integration, performance feel, error message clarity. - -**Needs human if uncertain (user-facing phases only):** Complex wiring grep can't trace, dynamic state-dependent behavior, edge cases. - -Format each as: Test Name → What to do → Expected result → Why can't verify programmatically. - - - -Classify status using this decision tree IN ORDER (most restrictive first): - -1. IF any truth FAILED, artifact MISSING/STUB, key link NOT_WIRED, blocker found, **or test quality audit found blockers (disabled requirement tests, circular tests)**: - → **gaps_found** - -2. IF any `must_haves.prohibitions` item disposes as flagged-unverified (ADR-550 D4): - - **test-tier, fail-closed when the wired check is MISSING OR FAILS** (now run via `check prohibition-enforcement` — `located: false`, or `dispositionForProhibition()` returns `status: 'unverified'`, `flagged: true`): → **gaps_found** in both interactive and autonomous modes (never green; a missing/failing mechanical check is an unverified gap). A test-tier item whose wired check PASSES disposes `status: 'green'`, `flagged: false` and is NOT a gap — it can reach **passed**. - - **judgment-tier, autonomous run** (non-authoritative LLM-judge verdict): emit the `unverified-prohibition — human review recommended` flag and classify → **human_needed** (autonomous completion reads "complete with N flagged prohibitions"; never a silent pass, never a hard halt). - - **judgment-tier, interactive run**: route to the end-of-phase human checkpoint → **human_needed**. - -2b. IF any `must_haves.truths` item carries the `verification: backstop` marker (#1154 — the verify-time truth-axis mirror of ADR-550 D4) AND the verifier cannot confirm it with **explicit evidence** (a wired held-out/property-based test that PASSES, or a directly-observed behavior — i.e. `dispositionForUnverifiableTruth()` returns `status: 'unverified'`, `flagged: true`, `reason: 'insufficient_spec'`): - - **abstain → human_needed**, NEVER `passed` and never silently graded green. Emit a prominent `unverified — held-out test recommended` flag carrying the distinguishable `reason: insufficient_spec` (so it is not conflated with ordinary manual-UAT `human_needed`). - - *Autonomous run:* record it and continue — completion reads "complete with N unverified non-inferable checks"; never a hard halt of an AFK run. *Interactive run:* route to the end-of-phase human checkpoint. - - **Exogenous only:** abstention fires SOLELY on the `backstop` tag, never a self-judged "abstain if unsure" (N17). An **inferable** truth is NEVER abstained (over-abstention guard); a `backstop` truth WITH a passing wired held-out test reaches **passed**. Reliable on capable tiers (`sonnet`+); the budget `haiku` tier degrades — see `references/honest-verifier.md`. - -3. IF the previous step produced ANY human verification items — this includes every ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth and every abstained `insufficient_spec` backstop truth: - → **human_needed** (even if all other truths VERIFIED) - -4. IF all checks pass AND no human verification items AND no flagged prohibitions AND no abstained (`insufficient_spec`) truths: - → **passed** - -**passed is ONLY valid when no human verification items, no flagged prohibitions, AND no abstained `insufficient_spec` truths exist.** Neither a prohibition (must-NOT) nor an unconfirmable non-inferable truth can ever be silently absorbed into a `passed` verdict — that is the core failure mode ADR-550 D4 forbids (now closed on both the prohibition and truth axes). - -A ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truth is never FAILED and never VERIFIED: it does not trigger gaps_found (the code is present and wired) and is not counted as verified (its runtime behavior was not exercised). It routes through the existing human_needed sink — no new overall status. - -**Score:** `verified_truths / total_truths` — `verified_truths` counts ✓ VERIFIED truths plus PASSED (override) truths; excluded are ⚠️ PRESENT_BEHAVIOR_UNVERIFIED truths (the `behavior_unverified` count) and abstained ⚠️ `insufficient_spec` backstop truths (#1154) — both are not ✓ VERIFIED and both route to `human_needed`. A headline N/N therefore certifies behavioral evidence for every behavior-dependent truth and explicit evidence for every non-inferable one, not merely symbol presence. - - - -Before reporting gaps, cross-reference each gap against later phases in the milestone using the full roadmap data loaded in load_context (from `roadmap analyze`). - -For each potential gap identified in determine_status: -1. Check if the gap's failed truth or missing item is covered by a later phase's goal or success criteria -2. **Match criteria:** The gap's concern appears in a later phase's goal text, success criteria text, or the later phase's name clearly suggests it covers this area -3. If a clear match is found → move the gap to a `deferred` list with the matching phase reference and evidence text -4. If no match in any later phase → keep as a real `gap` - -**Important:** Be conservative. Only defer a gap when there is clear, specific evidence in a later phase. Vague or tangential matches should NOT cause deferral — when in doubt, keep it as a real gap. - -**Deferred items do NOT affect the status determination.** Recalculate after filtering: -- If gaps list is now empty and no human items exist → `passed` -- If gaps list is now empty but human items exist → `human_needed` -- If gaps list still has items → `gaps_found` - -Include deferred items in VERIFICATION.md frontmatter (`deferred:` section) and body (Deferred Items table) for transparency. If no deferred items exist, omit these sections. - - - -If gaps_found: - -1. **Cluster related gaps:** API stub + component unwired → "Wire frontend to backend". Multiple missing → "Complete core implementation". Wiring only → "Connect existing components". - -2. **Generate plan per cluster:** Objective, 2-3 tasks (files/action/verify each), re-verify step. Keep focused: single concern per plan. - -3. **Order by dependency:** Fix missing → fix stubs → fix wiring → **fix test evidence** → verify. - - - -```bash -REPORT_PATH="$PHASE_DIR/${PHASE_NUM}-VERIFICATION.md" -``` - -Fill template sections: frontmatter (phase/timestamp/status/score), goal achievement, artifact table, wiring table, requirements coverage, anti-patterns, human verification, gaps summary, fix plans (if gaps_found), metadata. - -See ~/.claude/gsd-core/templates/verification-report.md for complete template. - - - -Return status (`passed` | `gaps_found` | `human_needed`), score (N/M must-haves), report path. - -If gaps_found: list gaps + recommended fix plan names. -If human_needed: list items requiring human testing. - -Orchestrator routes: `passed` → update_roadmap | `gaps_found` → create/execute fixes, re-verify | `human_needed` → present to user. - - - - - -- [ ] Must-haves established (from frontmatter or derived) -- [ ] All truths verified with status and evidence -- [ ] All artifacts checked at all three levels -- [ ] All key links verified -- [ ] Requirements coverage assessed (if applicable) -- [ ] CONTEXT.md decisions checked against shipped artifacts (#2492 — non-blocking) -- [ ] Anti-patterns scanned and categorized -- [ ] Test quality audited (disabled tests, circular patterns, assertion strength, provenance) -- [ ] Human verification items identified -- [ ] Overall status determined -- [ ] Deferred items filtered against later milestone phases (if gaps found) -- [ ] Fix plans generated (if gaps_found after filtering) -- [ ] VERIFICATION.md created with complete report -- [ ] Results returned to orchestrator - diff --git a/src/normalize-test-command.cts b/src/normalize-test-command.cts index fd5a3bcd6..3bea56019 100644 --- a/src/normalize-test-command.cts +++ b/src/normalize-test-command.cts @@ -22,7 +22,7 @@ * hang or take super-linear time on an adversarial `workflow.test_command`. * * Single source of truth: all four test-command gates (regression, post-merge, - * audit-fix, verify-phase) route their resolved command through this helper so + * audit-fix) route their resolved command through this helper so * the paths cannot drift. * * Leaf module — depends only on node:fs / node:path. diff --git a/src/probe-core.cts b/src/probe-core.cts index aca8b211f..fb62c8c64 100644 --- a/src/probe-core.cts +++ b/src/probe-core.cts @@ -494,7 +494,7 @@ export function dispositionForProhibition( } // D4 GUARD: a judgment-tier (or unknown-tier) prohibition is NEVER a silent green from this - // deterministic helper — it always routes to human/LLM judgment review (ADR-550 D4; verify-phase.md). + // deterministic helper — it always routes to human/LLM judgment review (ADR-550 D4; gsd-verifier.md + references/verifier-phase-gates.md). // Only a test-tier item with wired enforcement evidence may go green; the producer that supplies // that evidence (`prohibition-enforcement`, #1259) runs the wired check and requires a genuine pass. if (tier === 'test') { diff --git a/tests/emitted-drift-acks/1955-verifier-coincidental-reliance.json b/tests/emitted-drift-acks/1955-verifier-coincidental-reliance.json index d0ce65e29..1cdde5309 100644 --- a/tests/emitted-drift-acks/1955-verifier-coincidental-reliance.json +++ b/tests/emitted-drift-acks/1955-verifier-coincidental-reliance.json @@ -1,6 +1,6 @@ { "version": 1, "paths": { - "gsd-verifier.md": "#1955: Step 3 gains sub-step 5c (the coincidental-reliance advisory), one Step 9 score bullet, one Observable Truths example row, and the coincidental_reliance_items frontmatter block. Growth is those four additions only — no existing text was rewritten. Deliberately dense rather than extracted: the issue's approved scope says 'No new files', and ADR-1610 Decision 4 / tests/workflow-size-budget.test.cjs:32-40 name eager @-import relocation as gaming the size proxy (it shrinks the measured file while total loaded context is unchanged or larger); this agent has no lazy-read seam. After the change the file sits at 48994 bytes against the LARGE tier hard cap of 49152 (tests/agent-size-budget.test.cjs), i.e. 158 bytes of headroom. That is deliberate and disclosed: the cap is not crossed and is not raised, but the next contributor who needs room in this agent must do a lazy extraction rather than add prose. Content justification: the verifier's goal-backward pass grades THAT a truth holds and never WHY, so a truth can read VERIFIED while resting on a coincidence — a precondition nothing guarantees, an ordering nothing enforces, or a fixture-only truth. 5c classifies the evidence already recorded rather than asking for a confidence rating, but it is honestly an endogenous check and so weaker than the exogenous backstop tag gsd-core/references/honest-verifier.md routes on — which is exactly why it is advisory only: it changes no score, no status, and emits no human-verification item, so a passing phase still passes. gsd-core/workflows/verify-phase.md is deliberately NOT edited (it has 29 bytes under its DEFAULT tier cap); it receives the rule through its existing eager @-import of gsd-core/templates/verification-report.md, whose Guidelines now carry the imperative check rather than only the row shape. tests/verifier-coincidental-reliance.test.cjs asserts both halves of that route and locks out a silent second copy appearing in the workflow." + "gsd-verifier.md": "#1955: Step 3 gains sub-step 5c (the coincidental-reliance advisory), one Step 9 score bullet, one Observable Truths example row, and the coincidental_reliance_items frontmatter block. Growth is those four additions only — no existing text was rewritten. Deliberately dense rather than extracted: the issue's approved scope says 'No new files', and ADR-1610 Decision 4 / tests/workflow-size-budget.test.cjs:32-40 name eager @-import relocation as gaming the size proxy (it shrinks the measured file while total loaded context is unchanged or larger); this agent has no lazy-read seam. After the change the file sits at 48994 bytes against the LARGE tier hard cap of 49152 (tests/agent-size-budget.test.cjs), i.e. 158 bytes of headroom. That is deliberate and disclosed: the cap is not crossed and is not raised, but the next contributor who needs room in this agent must do a lazy extraction rather than add prose. Content justification: the verifier's goal-backward pass grades THAT a truth holds and never WHY, so a truth can read VERIFIED while resting on a coincidence — a precondition nothing guarantees, an ordering nothing enforces, or a fixture-only truth. 5c classifies the evidence already recorded rather than asking it to rate its own confidence — but it is honestly an endogenous check and so weaker than the exogenous backstop tag gsd-core/references/honest-verifier.md routes on — which is exactly why it is advisory only: it changes no score, no status, and emits no human-verification item, so a passing phase still passes. gsd-core/workflows/verify-phase.md is deliberately NOT edited (it has 29 bytes under its DEFAULT tier cap); it receives the rule through its existing eager @-import of gsd-core/templates/verification-report.md, whose Guidelines now carry the imperative check rather than only the row shape. tests/verifier-coincidental-reliance.test.cjs asserts both halves of that route and locks out a silent second copy appearing in the workflow. — #1892 append (epic #1891 F7, merged into this fragment because two ack sources may never name the same path): +55 bytes, one required_reading line (@~/.claude/gsd-core/references/verifier-phase-gates.md) so the verifier eagerly loads the three verification-time gates migrated out of the now-DELETED orphan workflow gsd-core/workflows/verify-phase.md (decision-coverage validation #2492, test-quality audit, infrastructure-phase human-verification scoping #2504) — the #1955-era routing through that workflow's template import retired with it. Net loaded context shrinks ~30 KB (40,931-byte orphan stops shipping and loading, replaced by a 9,951-byte reference plus this line); not proxy-gaming per ADR-1610 D4 — no existing prose was relocated out of the measured file, live behavior that previously reached no runtime is restored. Agent sits at 49,049 of 49,152 bytes (103 bytes of headroom), cap unchanged." } } diff --git a/tests/execute-phase-active-flags.test.cjs b/tests/execute-phase-active-flags.test.cjs index 7109726a4..4caa72688 100644 --- a/tests/execute-phase-active-flags.test.cjs +++ b/tests/execute-phase-active-flags.test.cjs @@ -82,7 +82,7 @@ describe('execute-phase command: active flags are explicit', () => { * Regression test for #2396: hardcoded host-level test commands bypass * container-only project Makefiles. * - * Fix: execute-phase.md, verify-phase.md, and audit-fix.md must check for + * Fix: execute-phase.md and audit-fix.md must check for * Makefile with a test target (and other wrappers) before falling through * to hardcoded language-sniffed commands. */ @@ -95,7 +95,6 @@ const fs = require('fs'); const path = require('path'); const EXECUTE_PHASE_PATH = path.join(__dirname, '..', 'gsd-core', 'workflows', 'execute-phase.md'); -const VERIFY_PHASE_PATH = path.join(__dirname, '..', 'gsd-core', 'workflows', 'verify-phase.md'); const AUDIT_FIX_PATH = path.join(__dirname, '..', 'gsd-core', 'workflows', 'audit-fix.md'); // #1857: execute-phase's regression-gate test-command resolution was extracted // to this step file (execute-phase.md is size-frozen — phase-6 capstone). @@ -164,10 +163,6 @@ describe('bug-2396: Makefile test target must take priority over hardcoded comma assert.ok(fs.existsSync(EXECUTE_PHASE_PATH), 'execute-phase.md should exist'); }); - test('verify-phase.md exists', () => { - assert.ok(fs.existsSync(VERIFY_PHASE_PATH), 'verify-phase.md should exist'); - }); - test('audit-fix.md exists', () => { assert.ok(fs.existsSync(AUDIT_FIX_PATH), 'audit-fix.md should exist'); }); @@ -176,10 +171,6 @@ describe('bug-2396: Makefile test target must take priority over hardcoded comma assertMakefileCheckBeforeNpmTest(REGRESSION_GATE_PATH, 'regression-gate.md'); }); - test('verify-phase.md: Makefile check precedes npm test', () => { - assertMakefileCheckBeforeNpmTest(VERIFY_PHASE_PATH, 'verify-phase.md'); - }); - test('audit-fix.md: Makefile check precedes npm test', () => { assertMakefileCheckBeforeNpmTest(AUDIT_FIX_PATH, 'audit-fix.md'); }); @@ -188,10 +179,6 @@ describe('bug-2396: Makefile test target must take priority over hardcoded comma assertConfigGetBeforeMakefile(REGRESSION_GATE_PATH, 'regression-gate.md'); }); - test('verify-phase.md: workflow.test_command config checked first (within bash block)', () => { - assertConfigGetBeforeMakefile(VERIFY_PHASE_PATH, 'verify-phase.md'); - }); - test('audit-fix.md: workflow.test_command config checked first (within bash block)', () => { assertConfigGetBeforeMakefile(AUDIT_FIX_PATH, 'audit-fix.md'); }); diff --git a/tests/fixtures/install-tree/antigravity.json b/tests/fixtures/install-tree/antigravity.json index 66964e1d8..2c053b4bc 100644 --- a/tests/fixtures/install-tree/antigravity.json +++ b/tests/fixtures/install-tree/antigravity.json @@ -163,6 +163,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -366,7 +367,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/augment.json b/tests/fixtures/install-tree/augment.json index 88faa6cbc..64ed0cbc3 100644 --- a/tests/fixtures/install-tree/augment.json +++ b/tests/fixtures/install-tree/augment.json @@ -234,6 +234,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -437,7 +438,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/claude-local.json b/tests/fixtures/install-tree/claude-local.json index 1edfce4d8..ccf6b2b44 100644 --- a/tests/fixtures/install-tree/claude-local.json +++ b/tests/fixtures/install-tree/claude-local.json @@ -233,6 +233,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -436,7 +437,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/claude.json b/tests/fixtures/install-tree/claude.json index b64a3012d..34e45ab05 100644 --- a/tests/fixtures/install-tree/claude.json +++ b/tests/fixtures/install-tree/claude.json @@ -162,6 +162,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -365,7 +366,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/cline.json b/tests/fixtures/install-tree/cline.json index 4ccf9af3e..aaff7c14e 100644 --- a/tests/fixtures/install-tree/cline.json +++ b/tests/fixtures/install-tree/cline.json @@ -166,6 +166,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -369,7 +370,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/codebuddy.json b/tests/fixtures/install-tree/codebuddy.json index 616aeaca5..17e5885cb 100644 --- a/tests/fixtures/install-tree/codebuddy.json +++ b/tests/fixtures/install-tree/codebuddy.json @@ -234,6 +234,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -437,7 +438,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/codex.json b/tests/fixtures/install-tree/codex.json index d62d45316..0ba864867 100644 --- a/tests/fixtures/install-tree/codex.json +++ b/tests/fixtures/install-tree/codex.json @@ -269,6 +269,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -472,7 +473,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/copilot.json b/tests/fixtures/install-tree/copilot.json index 93dcda8be..a351152a8 100644 --- a/tests/fixtures/install-tree/copilot.json +++ b/tests/fixtures/install-tree/copilot.json @@ -164,6 +164,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -367,7 +368,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/cursor.json b/tests/fixtures/install-tree/cursor.json index 29c979a08..9fa7d46bb 100644 --- a/tests/fixtures/install-tree/cursor.json +++ b/tests/fixtures/install-tree/cursor.json @@ -163,6 +163,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -366,7 +367,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/hermes.json b/tests/fixtures/install-tree/hermes.json index 55738333e..597f4f942 100644 --- a/tests/fixtures/install-tree/hermes.json +++ b/tests/fixtures/install-tree/hermes.json @@ -163,6 +163,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -366,7 +367,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/kilo.json b/tests/fixtures/install-tree/kilo.json index ac782c5f9..4ac74fc46 100644 --- a/tests/fixtures/install-tree/kilo.json +++ b/tests/fixtures/install-tree/kilo.json @@ -234,6 +234,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -437,7 +438,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/kimi-code.json b/tests/fixtures/install-tree/kimi-code.json index d6daead57..39075ff08 100644 --- a/tests/fixtures/install-tree/kimi-code.json +++ b/tests/fixtures/install-tree/kimi-code.json @@ -195,6 +195,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -398,7 +399,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/kimi.json b/tests/fixtures/install-tree/kimi.json index 575b49e01..8dbb37198 100644 --- a/tests/fixtures/install-tree/kimi.json +++ b/tests/fixtures/install-tree/kimi.json @@ -231,6 +231,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -434,7 +435,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/opencode.json b/tests/fixtures/install-tree/opencode.json index b78f2e2a1..1c0b298a4 100644 --- a/tests/fixtures/install-tree/opencode.json +++ b/tests/fixtures/install-tree/opencode.json @@ -234,6 +234,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -437,7 +438,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/pi.json b/tests/fixtures/install-tree/pi.json index defc30f5c..be8e8103b 100644 --- a/tests/fixtures/install-tree/pi.json +++ b/tests/fixtures/install-tree/pi.json @@ -131,6 +131,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -334,7 +335,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/qwen.json b/tests/fixtures/install-tree/qwen.json index aaa750cc6..31bb02088 100644 --- a/tests/fixtures/install-tree/qwen.json +++ b/tests/fixtures/install-tree/qwen.json @@ -163,6 +163,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -366,7 +367,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/trae.json b/tests/fixtures/install-tree/trae.json index b3a739d28..20667051a 100644 --- a/tests/fixtures/install-tree/trae.json +++ b/tests/fixtures/install-tree/trae.json @@ -163,6 +163,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -366,7 +367,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/windsurf.json b/tests/fixtures/install-tree/windsurf.json index 9666f1e9d..e023b072f 100644 --- a/tests/fixtures/install-tree/windsurf.json +++ b/tests/fixtures/install-tree/windsurf.json @@ -163,6 +163,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -366,7 +367,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/fixtures/install-tree/zcode.json b/tests/fixtures/install-tree/zcode.json index 8c936cece..b8db25fd3 100644 --- a/tests/fixtures/install-tree/zcode.json +++ b/tests/fixtures/install-tree/zcode.json @@ -234,6 +234,7 @@ "gsd-core/references/user-story-template.md", "gsd-core/references/verification-overrides.md", "gsd-core/references/verification-patterns.md", + "gsd-core/references/verifier-phase-gates.md", "gsd-core/references/verifier-wiring-patterns.md", "gsd-core/references/verify-mvp-mode.md", "gsd-core/references/workstream-flag.md", @@ -437,7 +438,6 @@ "gsd-core/workflows/update.md", "gsd-core/workflows/update/steps/channel-banner.md", "gsd-core/workflows/validate-phase.md", - "gsd-core/workflows/verify-phase.md", "gsd-core/workflows/verify-work.md", "gsd-core/workflows/verify-work/steps/automated-ui-verification.md", "gsd-core/workflows/verify-work/steps/mvp-uat-framing.md", diff --git a/tests/plan-phase-drift-guard.test.cjs b/tests/plan-phase-drift-guard.test.cjs index 43bdf66df..86d617d47 100644 --- a/tests/plan-phase-drift-guard.test.cjs +++ b/tests/plan-phase-drift-guard.test.cjs @@ -1589,7 +1589,7 @@ describe('enh-2430 — INVENTORY sync', () => { /** * Bug #2492: Add gates to ensure discuss-phase decisions are translated to * plans (plan-phase, BLOCKING) and verified against shipped artifacts - * (verify-phase, NON-BLOCKING). + * (verifier-phase-gates reference, NON-BLOCKING). * * These workflow files are loaded as prompts by the corresponding subagents. * The tests below verify that the prompt text contains the gate steps and @@ -1603,7 +1603,7 @@ const fs = require('fs'); const path = require('path'); const PLAN_PHASE = path.join(__dirname, '..', 'gsd-core', 'workflows', 'plan-phase.md'); -const VERIFY_PHASE = path.join(__dirname, '..', 'gsd-core', 'workflows', 'verify-phase.md'); +const VERIFY_GATES = path.join(__dirname, '..', 'gsd-core', 'references', 'verifier-phase-gates.md'); const SCHEMA_MANIFEST_JSON = path.join(__dirname, '..', 'gsd-core', 'bin', 'shared', 'config-schema.manifest.json'); describe('plan-phase decision-coverage gate (#2492)', () => { @@ -1700,20 +1700,20 @@ describe('plan-phase decision-coverage gate (#2492)', () => { }); }); -describe('verify-phase decision-coverage gate (#2492)', () => { - const md = fs.readFileSync(VERIFY_PHASE, 'utf-8'); +describe('verifier-phase-gates decision-coverage gate (#2492)', () => { + const md = fs.readFileSync(VERIFY_GATES, 'utf-8'); test('contains a verify_decisions step', () => { assert.ok( /verify_decisions/.test(md), - 'verify-phase.md must define a verify_decisions step', + 'verifier-phase-gates.md must define a verify_decisions step', ); }); test('invokes the check.decision-coverage-verify handler', () => { assert.ok( md.includes('check.decision-coverage-verify'), - 'verify-phase.md must call gsd-sdk query check.decision-coverage-verify', + 'verifier-phase-gates.md must call gsd-sdk query check.decision-coverage-verify', ); }); @@ -1721,14 +1721,14 @@ describe('verify-phase decision-coverage gate (#2492)', () => { const lower = md.toLowerCase(); assert.ok( lower.includes('non-blocking') || lower.includes('warning only') || lower.includes('not block'), - 'verify-phase.md must declare the decision gate is non-blocking', + 'verifier-phase-gates.md must declare the decision gate is non-blocking', ); }); test('mentions workflow.context_coverage_gate skip clause', () => { assert.ok( md.includes('workflow.context_coverage_gate'), - 'verify-phase.md must reference workflow.context_coverage_gate to allow skipping', + 'verifier-phase-gates.md must reference workflow.context_coverage_gate to allow skipping', ); }); }); @@ -1741,6 +1741,21 @@ describe('runtime wiring for #2492 gates', () => { 'workflow.context_coverage_gate must be present in config-schema manifest', ); }); + + test('gsd-verifier eagerly imports verifier-phase-gates.md (#1892)', () => { + // The decision-coverage gate (and the other migrated verify-time gates) only + // reach the runtime if the verifier agent actually loads the reference that + // now carries them — a reference nothing imports is the exact orphan class + // epic #1891 exists to remove. + const agent = fs.readFileSync( + path.join(__dirname, '..', 'agents', 'gsd-verifier.md'), + 'utf-8', + ); + assert.ok( + agent.includes('@~/.claude/gsd-core/references/verifier-phase-gates.md'), + 'agents/gsd-verifier.md must @-import references/verifier-phase-gates.md in required_reading', + ); + }); }); }); } diff --git a/tests/planner-language-regression.test.cjs b/tests/planner-language-regression.test.cjs index e512f80da..f13eeac10 100644 --- a/tests/planner-language-regression.test.cjs +++ b/tests/planner-language-regression.test.cjs @@ -124,8 +124,6 @@ const ALLOWLIST = { 'fast.md': ['time_sizing'], // Execute-phase uses a configurable test-gate timeout (workflow.test_gate_timeout, #1857) 'execute-phase.md': ['time_sizing'], - // Verify-phase uses a configurable test-gate timeout (workflow.test_gate_timeout, #1857) - 'verify-phase.md': ['time_sizing'], // Map-codebase documents subagent_timeout 'map-codebase.md': ['time_sizing'], // Help documents CodeRabbit timing diff --git a/tests/prohibition-probe.validators.test.cjs b/tests/prohibition-probe.validators.test.cjs index 69d0f5f05..18455257f 100644 --- a/tests/prohibition-probe.validators.test.cjs +++ b/tests/prohibition-probe.validators.test.cjs @@ -8,7 +8,7 @@ // (resolution: null; the checkable content is `statement`). The prior config required a // non-empty `resolution` and threw on 100% of the fixtures. // WR-01 — dispositionForProhibition must NEVER return a silent green for a judgment-tier item, -// regardless of enforcement evidence (ADR-550 D4 / verify-phase.md). +// regardless of enforcement evidence (ADR-550 D4 / gsd-verifier.md). // // Fixture-driven on purpose: the validators are exercised against the SAME expected.json files the // docs-fixtures parity test pins, so the validator and the corpus can never silently diverge. diff --git a/tests/test-gate-watch-mode.test.cjs b/tests/test-gate-watch-mode.test.cjs index 6d1b19185..086409853 100644 --- a/tests/test-gate-watch-mode.test.cjs +++ b/tests/test-gate-watch-mode.test.cjs @@ -12,9 +12,8 @@ * - surface a timeout (exit 124) with a watch-mode hint — the regression gate * ABORTS, the others surface clearly (never silently ignored). * - * verify-phase's gate was ALREADY bounded (a fixed `run-with-timeout 300`, not a hang), so - * it only needs the normalizer (so a watch runner exits fast) and keeps its own - * fixed 5-minute bound — asserted separately below. + * verify-phase's gate (a fourth, already-bounded surface) was deleted with its orphan + * workflow in #1892; the gates above are the complete live set. */ 'use strict'; @@ -33,7 +32,6 @@ const REGRESSION_GATE = path.join(ROOT, 'gsd-core', 'workflows', 'execute-phase' const REGRESSION_GATE_RUN = path.join(ROOT, 'gsd-core', 'workflows', 'execute-phase', 'steps', 'regression-gate-run.md'); const POST_MERGE_GATE = path.join(ROOT, 'gsd-core', 'workflows', 'execute-phase', 'steps', 'post-merge-gate.md'); const AUDIT_FIX = path.join(ROOT, 'gsd-core', 'workflows', 'audit-fix.md'); -const VERIFY_PHASE = path.join(ROOT, 'gsd-core', 'workflows', 'verify-phase.md'); const EXECUTE_PHASE = path.join(ROOT, 'gsd-core', 'workflows', 'execute-phase.md'); function read(p) { return fs.readFileSync(p, 'utf-8'); } @@ -66,18 +64,7 @@ describe('#1857: test gates normalize to one-shot and bound with a timeout', () }); } - // verify-phase is already bounded (fixed `run-with-timeout 300`, not a hang); it only - // needs the normalizer so a watch runner exits fast, and names watch mode on 124. - describe('verify-phase gate (already bounded — normalize-only)', () => { - test('routes the resolved command through the shared normalize-test-command helper', () => { - assert.match(read(VERIFY_PHASE), /normalize-test-command/, 'verify-phase must call the shared normalize-test-command helper'); - }); - test('surfaces its fixed timeout (exit 124) naming watch/dev mode', () => { - const c = read(VERIFY_PHASE); - assert.match(c, /-eq 124/, 'verify-phase must handle the timeout exit code (124)'); - assert.match(c, /watch\/dev mode/, 'verify-phase must name watch/dev mode as the likely cause on timeout'); - }); - }); + // verify-phase's normalize-only block was removed with the orphan workflow (#1892). test('the regression gate ABORTS (halts) on a watch-mode timeout', () => { const c = read(REGRESSION_GATE); @@ -92,7 +79,7 @@ describe('#1857: test gates normalize to one-shot and bound with a timeout', () test('the gates share ONE normalizer — the helper is a single source of truth', () => { // The behaviour lives in src/normalize-test-command.cts; every gate invokes it // by the same verb name, so a change to watch-defeat logic touches one place. - for (const file of [REGRESSION_GATE_RUN, POST_MERGE_GATE, AUDIT_FIX, VERIFY_PHASE]) { + for (const file of [REGRESSION_GATE_RUN, POST_MERGE_GATE, AUDIT_FIX]) { assert.match(read(file), /gsd_run query normalize-test-command/); } }); @@ -107,8 +94,9 @@ describe('#1857: test gates normalize to one-shot and bound with a timeout', () // first build file). Every gate that resolves a build/test command this way MUST // pass `--raw` so an unset key is a genuinely empty bash string the `-z` guard // catches. This is a defect CLASS — post-merge-gate.md was the reported instance, -// but regression-gate.md, verify-phase.md, and audit-fix.md shared it, so the -// guard sweeps all of them (a single-file check gave false confidence). config-get's +// but regression-gate.md and audit-fix.md shared it, so the guard sweeps them (a +// single-file check gave false confidence; the fourth original member, +// verify-phase.md, was deleted as an orphan in #1892). config-get's // own `--raw` behaviour is covered in config-get-default.test.cjs. describe('#2350: every gate resolves build/test commands with --raw', () => { // Each gate file that reads workflow.build_command / workflow.test_command to @@ -116,7 +104,6 @@ describe('#2350: every gate resolves build/test commands with --raw', () => { const GATE_FILES = [ ['post-merge gate', POST_MERGE_GATE], ['regression gate', REGRESSION_GATE_RUN], - ['verify-phase gate', VERIFY_PHASE], ['audit-fix gate', AUDIT_FIX], ]; diff --git a/tests/verifier-behavior-unverified.test.cjs b/tests/verifier-behavior-unverified.test.cjs index 76b1c9046..3901a7e17 100644 --- a/tests/verifier-behavior-unverified.test.cjs +++ b/tests/verifier-behavior-unverified.test.cjs @@ -98,10 +98,10 @@ test('VERIFICATION.md templates carry behavior_unverified + the new truth-state' assert.match(standalone, /behavior_unverified_items/); }); -const verifyPhase = fs.readFileSync(path.join(ROOT, 'gsd-core', 'workflows', 'verify-phase.md'), 'utf-8'); +const verifyPhase = fs.readFileSync(path.join(ROOT, 'gsd-core', 'references', 'verifier-phase-gates.md'), 'utf-8'); const planningArtifacts = fs.readFileSync(path.join(ROOT, 'docs', 'reference', 'planning-artifacts.md'), 'utf-8'); -test('shipped verify-phase workflow mirrors the behavior-unverified calibration', () => { +test('shipped verifier-phase-gates reference mirrors the behavior-unverified calibration', () => { assert.match(verifyPhase, /PRESENT_BEHAVIOR_UNVERIFIED/); assert.match(verifyPhase, /behavior_unverified/); assert.match(verifyPhase, /state transition/i); diff --git a/tests/verifier-coincidental-reliance.test.cjs b/tests/verifier-coincidental-reliance.test.cjs index 821ddbe1d..43c156a44 100644 --- a/tests/verifier-coincidental-reliance.test.cjs +++ b/tests/verifier-coincidental-reliance.test.cjs @@ -51,7 +51,7 @@ const read = (...p) => fs.readFileSync(path.join(ROOT, ...p), 'utf-8'); const verifier = read('agents', 'gsd-verifier.md'); const template = read('gsd-core', 'templates', 'verification-report.md'); const agentsDoc = read('docs', 'AGENTS.md'); -const verifyPhase = read('gsd-core', 'workflows', 'verify-phase.md'); +const verifyPhase = read('gsd-core', 'references', 'verifier-phase-gates.md'); const QUALIFIER = '✓ VERIFIED (coincidental-reliance)'; const REASONS = ['undeclared-precondition', 'incidental-ordering', 'fixture-only']; @@ -225,7 +225,7 @@ describe('#1955: coincidental-reliance advisory — the report surface', () => { }); }); -describe('#1955: cross-surface parity (agent, template, verify-phase workflow)', () => { +describe('#1955: cross-surface parity (agent, template, verifier gate reference)', () => { test('PARITY: agent and standalone template agree on the advisory vocabulary', () => { // Generative-fix-divergence gate: two surfaces render the same report, so a // token added to one and not the other is the defect this test exists for. @@ -245,20 +245,19 @@ describe('#1955: cross-surface parity (agent, template, verify-phase workflow)', assert.match(guidelines, /coincidental-reliance/); }); - test('verify-phase workflow reaches the rule through its eager template import', () => { - // The third surface. `gsd-core/workflows/verify-phase.md` is the - // non-subagent verification path and reimplements the truth rubric inline, - // but it sits 29 bytes under the DEFAULT tier hard cap in - // tests/workflow-size-budget.test.cjs, so the rule is NOT duplicated into - // it. It reaches the rule instead through the eager `@`-import of the - // template, whose Guidelines carry the instruction — not merely the output - // shape. Both halves of that claim are asserted here, because either one - // silently failing turns the workflow surface into an undetected - // divergence. + test('verifier gate reference reaches the rule through the canonical template', () => { + // The third surface. `gsd-core/references/verifier-phase-gates.md` is the + // verifier agent's eagerly-imported gate reference (migrated from the + // retired workflows/verify-phase.md in #1892). It does NOT reimplement the + // truth rubric — the rule is NOT duplicated into it. It reaches the rule + // instead through its pointer to the canonical report template, whose + // Guidelines carry the instruction — not merely the output shape. Both + // halves of that claim are asserted here, because either one silently + // failing turns the reference surface into an undetected divergence. assert.match( verifyPhase, /@[^\n]*gsd-core\/templates\/verification-report\.md/, - 'verify-phase.md must eagerly import the verification-report template', + 'verifier-phase-gates.md must point at the verification-report template', ); const guidelines = template.slice(template.indexOf('**Per-truth states')); assert.match( @@ -269,16 +268,16 @@ describe('#1955: cross-surface parity (agent, template, verify-phase workflow)', }); test('the workflow surface carries no divergent copy of the rule', () => { - // Characterization, not aspiration: verify-phase.md deliberately holds NO - // copy of the detection prose today. If a future change adds one, this - // assertion fails and forces a decision — duplicate it deliberately and - // update this test, or keep the single template-carried source. Silent - // partial duplication across the two surfaces is the failure mode + // Characterization, not aspiration: verifier-phase-gates.md deliberately + // holds NO copy of the detection prose today. If a future change adds one, + // this assertion fails and forces a decision — duplicate it deliberately + // and update this test, or keep the single template-carried source. Silent + // partial duplication across the surfaces is the failure mode // (generative fix divergence) this locks out. assert.doesNotMatch( verifyPhase, /coincidental-reliance/, - 'verify-phase.md must not grow a second copy of the rule without a deliberate decision', + 'verifier-phase-gates.md must not grow a second copy of the rule without a deliberate decision', ); }); diff --git a/tests/verifier-deferred-items.test.cjs b/tests/verifier-deferred-items.test.cjs index 6bbadd4ef..26ee0f163 100644 --- a/tests/verifier-deferred-items.test.cjs +++ b/tests/verifier-deferred-items.test.cjs @@ -112,43 +112,10 @@ describe('verifier deferred-items filtering (#1624)', () => { }); }); - // ── verify-phase.md (workflow) ───────────────────────────────────────────── - - describe('gsd-core/workflows/verify-phase.md', () => { - const workflowPath = path.join(ROOT, 'gsd-core', 'workflows', 'verify-phase.md'); - let workflowContent; - - test('file exists', () => { - assert.ok(fs.existsSync(workflowPath), 'verify-phase.md should exist'); - workflowContent = fs.readFileSync(workflowPath, 'utf-8'); - }); - - test('loads roadmap analyze in context step', () => { - workflowContent = workflowContent || fs.readFileSync(workflowPath, 'utf-8'); - assert.ok( - workflowContent.includes('roadmap analyze'), - 'verify-phase.md should load roadmap analyze in its context step' - ); - }); - - test('contains filter_deferred_items step', () => { - workflowContent = workflowContent || fs.readFileSync(workflowPath, 'utf-8'); - assert.ok( - workflowContent.includes('filter_deferred_items') || - workflowContent.includes('Filter Deferred'), - 'verify-phase.md should contain a deferred-item filtering step' - ); - }); - - test('success criteria mentions deferred filtering', () => { - workflowContent = workflowContent || fs.readFileSync(workflowPath, 'utf-8'); - assert.ok( - workflowContent.includes('Deferred items filtered') || - workflowContent.includes('deferred items filtered'), - 'success criteria should mention deferred item filtering' - ); - }); - }); + // ── verify-phase.md (workflow) — DELETED #1892 ───────────────────────────── + // The orphan workflow gsd-core/workflows/verify-phase.md was removed (0 loaders; + // live deferred-item filtering is carried by gsd-verifier.md Step 9b, asserted + // in the describe block above). // sdk/prompts/workflows/verify-phase.md removed in 377a6d2 — SDK loads installed workflow directly. diff --git a/tests/verify-mvp-uat.test.cjs b/tests/verify-mvp-uat.test.cjs index 48cc39719..d5620c819 100644 --- a/tests/verify-mvp-uat.test.cjs +++ b/tests/verify-mvp-uat.test.cjs @@ -90,7 +90,7 @@ describe('verify-work — MVP mode UAT framing', () => { * "manually invoke methods", "manually check database state" — and left * work half-finished specifically to create things for a human to do. * - * Fix: The verify-phase workflow's identify_human_verification step must + * Fix: The verifier reference's identify_human_verification step must * explicitly handle phases with no user-facing elements by auto-passing UAT * with a logged rationale instead of inventing manual steps. */ @@ -103,7 +103,7 @@ const fs = require('fs'); const path = require('path'); const VERIFY_PHASE_PATH = path.join( - __dirname, '..', 'gsd-core', 'workflows', 'verify-phase.md' + __dirname, '..', 'gsd-core', 'references', 'verifier-phase-gates.md' ); /** @@ -119,10 +119,10 @@ function extractSection(content, heading) { } describe('bug #2504: UAT auto-pass for foundation/infrastructure phases', () => { - test('verify-phase workflow file exists', () => { + test('verifier-phase-gates reference file exists', () => { assert.ok( fs.existsSync(VERIFY_PHASE_PATH), - 'gsd-core/workflows/verify-phase.md should exist' + 'gsd-core/references/verifier-phase-gates.md should exist' ); }); @@ -142,7 +142,7 @@ describe('bug #2504: UAT auto-pass for foundation/infrastructure phases', () => assert.ok( hasInfrastructureHandling, - 'verify-phase.md identify_human_verification step must explicitly handle ' + + 'verifier-phase-gates.md identify_human_verification step must explicitly handle ' + 'infrastructure/foundation phases that have no user-facing elements. Without ' + 'this, agents invent artificial manual steps to satisfy UAT requirements ' + '(root cause of #2504).' @@ -164,7 +164,7 @@ describe('bug #2504: UAT auto-pass for foundation/infrastructure phases', () => assert.ok( hasAutoPass, - 'verify-phase.md identify_human_verification step must contain language about ' + + 'verifier-phase-gates.md identify_human_verification step must contain language about ' + 'auto-passing or skipping UAT for phases without user-facing elements. Agents ' + 'must not invent manual steps when there is nothing user-facing to test ' + '(root cause of #2504).' @@ -196,7 +196,7 @@ describe('bug #2504: UAT auto-pass for foundation/infrastructure phases', () => assert.ok( hasProhibition, - 'verify-phase.md identify_human_verification step must explicitly prohibit ' + + 'verifier-phase-gates.md identify_human_verification step must explicitly prohibit ' + 'inventing artificial manual UAT steps for infrastructure phases. The current ' + 'wording causes agents to create fake "manually run git commits" steps to ' + 'satisfy UAT mandates (root cause of #2504).' @@ -217,7 +217,7 @@ describe('bug #2504: UAT auto-pass for foundation/infrastructure phases', () => assert.ok( hasNaState, - 'verify-phase.md identify_human_verification step must include some concept of ' + + 'verifier-phase-gates.md identify_human_verification step must include some concept of ' + 'a "not applicable" or N/A UAT state for phases with no user-facing elements. ' + 'This prevents agents from blocking phase completion on invented manual steps ' + '(root cause of #2504).' diff --git a/tests/verify-test-quality.test.cjs b/tests/verify-test-quality.test.cjs index 45d4aa2da..8939a7cb1 100644 --- a/tests/verify-test-quality.test.cjs +++ b/tests/verify-test-quality.test.cjs @@ -1,8 +1,11 @@ // allow-test-rule: source-text-is-the-product -// Structural guard: reads gsd-core/workflows/verify-phase.md and asserts that -// the audit_test_quality step contains the skip-pattern marker, circular-detection -// marker, provenance-classification contract, and assertion-strength table markers. -// Goes red if that workflow guidance is removed or the step is renamed/deleted. +// Structural guard: reads gsd-core/references/verifier-phase-gates.md (the +// gsd-verifier agent's eagerly-imported gate reference; its content was +// migrated from the retired workflows/verify-phase.md orphan in #1892) and +// asserts that the audit_test_quality step contains the skip-pattern marker, +// circular-detection marker, provenance-classification contract, and +// assertion-strength table markers. +// Goes red if that guidance is removed or the step is renamed/deleted. 'use strict'; @@ -15,8 +18,8 @@ const WORKFLOW_PATH = path.join( __dirname, '..', 'gsd-core', - 'workflows', - 'verify-phase.md' + 'references', + 'verifier-phase-gates.md' ); // Locate the audit_test_quality step boundaries so sub-assertions are scoped @@ -33,7 +36,7 @@ function extractAuditStep(src) { } // workflowSrc and auditStepSrc are populated in the before() hook so that a -// missing or renamed verify-phase.md produces a descriptive test FAILURE rather +// missing or renamed verifier-phase-gates.md produces a descriptive test FAILURE rather // than a module-load crash that prevents any test from registering. let workflowSrc = null; let auditStepSrc = null; @@ -41,22 +44,22 @@ let auditStepSrc = null; before(() => { assert.ok( fs.existsSync(WORKFLOW_PATH), - `verify-phase.md not found at expected path: ${WORKFLOW_PATH} — ` + + `verifier-phase-gates.md not found at expected path: ${WORKFLOW_PATH} — ` + 'the file may have been renamed or moved' ); workflowSrc = fs.readFileSync(WORKFLOW_PATH, 'utf8'); auditStepSrc = extractAuditStep(workflowSrc); }); -describe('verify-phase.md audit_test_quality structural guard', () => { - test('verify-phase.md exists at gsd-core/workflows/verify-phase.md', () => { +describe('verifier-phase-gates.md audit_test_quality structural guard', () => { + test('verifier-phase-gates.md exists at gsd-core/references/verifier-phase-gates.md', () => { assert.ok( fs.existsSync(WORKFLOW_PATH), `missing workflow file: ${WORKFLOW_PATH}` ); }); - test('audit_test_quality step is present in verify-phase.md', () => { + test('audit_test_quality step is present in verifier-phase-gates.md', () => { assert.ok( auditStepSrc !== null, ` not found in ${WORKFLOW_PATH} — the step ` + diff --git a/tests/windows-robustness.test.cjs b/tests/windows-robustness.test.cjs index 1e830c270..ca9f58830 100644 --- a/tests/windows-robustness.test.cjs +++ b/tests/windows-robustness.test.cjs @@ -90,7 +90,6 @@ describe('workflow shell robustness', () => { 'resume-project.md', 'progress.md', 'transition.md', - 'verify-phase.md', 'verify-work.md', 'discuss-phase.md', 'plan-phase.md', diff --git a/tests/workflow-size-budget.test.cjs b/tests/workflow-size-budget.test.cjs index 589ea6e94..f74c4a859 100644 --- a/tests/workflow-size-budget.test.cjs +++ b/tests/workflow-size-budget.test.cjs @@ -91,14 +91,14 @@ const WORKFLOWS_DIR = path.join(__dirname, '..', 'gsd-core', 'workflows'); // headroom (vs the old GRACE=3000 hug): // XL 96 KiB — high-water execute-phase.md 93,400 → ~4.8 KB headroom // LARGE 60 KiB — high-water docs-update.md 55,468 → ~5.8 KB headroom -// DEFAULT 40 KiB — high-water verify-phase.md 40,931 → 29 BYTES headroom +// DEFAULT 40 KiB — high-water settings.md 40,352 → ~608 B headroom // (DEFAULT is deliberately the tightest: a single-purpose workflow approaching -// 40 KiB is the strongest extraction signal of the three. verify-phase.md is -// effectively AT the red line — the next edit to it must be preceded by a lazy -// extraction, not absorbed. Measured 2026-08-09 via measureWorkflows(); the -// previous note here named settings-advanced.md at 39,160 with ~1.8 KB of -// headroom, which was stale on both the file and the number and invited an -// edit that would have crossed the cap.) +// 40 KiB is the strongest extraction signal of the three. The previous DEFAULT +// high-water, verify-phase.md at 40,931 (29 bytes of headroom), was deleted as +// an orphan in #1892 — 0 loaders, with its still-live gates migrated to +// gsd-core/references/verifier-phase-gates.md behind the gsd-verifier agent. +// Measured 2026-08-13 via measureWorkflows() after that deletion; the note +// before that named settings-advanced.md at 39,160, stale on both counts.) const XL_CAP = 98304; // 96 KiB const LARGE_CAP = 61440; // 60 KiB const DEFAULT_CAP = 40960; // 40 KiB