From 395fb519e70e3be112ced986805b0ca0232a3e29 Mon Sep 17 00:00:00 2001 From: Rezolv Date: Sun, 14 Jun 2026 21:29:11 -0400 Subject: [PATCH] =?UTF-8?q?feat(spec-phase):=20prohibition=20probe=20?= =?UTF-8?q?=E2=80=94=20surface=20"must-NOT"=20constraints=20(#644)=20(#114?= =?UTF-8?q?9)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the spec-time prohibition probe (spec-phase Step 5.6) — the second adapter of the probe-core resolution model. Surfaces unwritten must-NOT constraints as negative SPEC acceptance criteria with test/judgment verification tiers; fail-closed at verify time. Per ADR-550. Closes #644. --- .changeset/644-prohibition-probe.md | 5 + CONTEXT.md | 14 +- agents/gsd-verifier.md | 13 +- docs/COMMANDS.md | 2 + docs/FEATURES.md | 34 +++ docs/INVENTORY-MANIFEST.json | 1 + docs/INVENTORY.md | 1 + docs/README.md | 1 + docs/adr/550-spec-phase-probe-contract.md | 9 + docs/how-to/resolve-prohibition-findings.md | 110 ++++++++ .../01-streak-reminder/expected.json | 14 + .../02-clean-utility/expected.json | 4 + .../03-multi-prohibition/expected.json | 32 +++ gsd-core/references/prohibition-probe.md | 248 ++++++++++++++++++ gsd-core/templates/spec.md | 14 + gsd-core/workflows/plan-phase.md | 2 + gsd-core/workflows/spec-phase.md | 64 +++++ gsd-core/workflows/verify-phase.md | 18 +- src/frontmatter.cts | 52 +++- src/probe-core.cts | 173 ++++++++++++ tests/agent-size-baseline.json | 2 +- tests/frontmatter.property.test.cjs | 71 +++++ .../prohibition-probe.docs-fixtures.test.cjs | 63 +++++ ...rohibition-probe.planner-contract.test.cjs | 96 +++++++ tests/prohibition-probe.schema.test.cjs | 181 +++++++++++++ ...ibition-probe.spec-phase-contract.test.cjs | 145 ++++++++++ tests/prohibition-probe.validators.test.cjs | 82 ++++++ tests/prohibition-probe.verify-tier.test.cjs | 57 ++++ tests/workflow-size-baseline.json | 6 +- 29 files changed, 1503 insertions(+), 11 deletions(-) create mode 100644 .changeset/644-prohibition-probe.md create mode 100644 docs/how-to/resolve-prohibition-findings.md create mode 100644 gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json create mode 100644 gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json create mode 100644 gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json create mode 100644 gsd-core/references/prohibition-probe.md create mode 100644 tests/prohibition-probe.docs-fixtures.test.cjs create mode 100644 tests/prohibition-probe.planner-contract.test.cjs create mode 100644 tests/prohibition-probe.schema.test.cjs create mode 100644 tests/prohibition-probe.spec-phase-contract.test.cjs create mode 100644 tests/prohibition-probe.validators.test.cjs create mode 100644 tests/prohibition-probe.verify-tier.test.cjs diff --git a/.changeset/644-prohibition-probe.md b/.changeset/644-prohibition-probe.md new file mode 100644 index 000000000..aab3a6502 --- /dev/null +++ b/.changeset/644-prohibition-probe.md @@ -0,0 +1,5 @@ +--- +type: Added +pr: 1149 +--- +spec-phase: prohibition probe — a prose-orchestrated Step 5.6 that surfaces the unwritten *must-NOT* constraints (values/safety/ethics) a feature could silently become but the spec never forbids. Two stages per requirement: an adversarial recall question ("what could this silently become that the author would NOT want?") then a one-pass precision classifier that drops routine engineering and keeps genuine prohibitions. Confirmed prohibitions become NEGATIVE SPEC acceptance criteria carrying a `test`/`judgment` verification tier, which plan-phase lifts into the `must_haves.prohibitions` sibling block (`truths` untouched). Judgment-tier items soft-gate at verify time (never silent, never hard-halt); unwired test-tier items fail closed. The recall stage is model-driven (no compiled engine); canon-bound concerns (OWASP/GDPR/fairness) are referred to `/gsd:secure-phase`. Additive and optional: existing SPECs without a Prohibitions section remain valid. Second adapter of the `probe-core` resolution model (ADR-550 Decision 7). diff --git a/CONTEXT.md b/CONTEXT.md index d388fe8f5..95d692bd4 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -235,7 +235,7 @@ The GSD-RESEARCH capability behind an L2-hybrid seam: code owns cache + provider Runtime-neutral predicate evaluating `*-UAT.md` / `*-VERIFICATION.md` result fields with markdown-aware parsing that ignores false-positive contexts (frontmatter body, fenced code, HTML comments, blockquotes). Returns `passed: true` only when all required checks pass; supports `--require-verification` to demand at least one VERIFICATION.md file alongside UAT results. Output envelope: `{ passed, uat_files[], verification_files[], checks[], blockers[], policy }`. Source: `gsd-core/bin/lib/uat-predicate.cjs` (generated from `src/uat-predicate.cts`). Wired via `phase uat-passed` alias → `phase-command-router` → `cmdPhaseUatPassed`. ### Probe Core Module -Generic spec-phase probe resolution model — the shared seam underlying spec-completeness probes (ADR-550 Decision 7). Owns the `status × verification` model (`status: resolved | dismissed | unresolved` × a per-probe `verification` tier), structural validation (`validateResolution`, `validateRequirement` — fail-closed: `verification` must be null unless status is `resolved`, and an out-of-enum status, a `dismissed`-without-`reason`, or an `unresolved` carrying a `resolution`/`reason`/tier payload all throw rather than silently miscount), the `analyzeCoverage(items, resolutions?, validators)` merge/rollup/orphan-reject pipeline, the `byVerification` per-tier rollup, and the `runProbeCli` I/O scaffold (parse → validate → analyze → emit, structurally guarding the report shape before write — a malformed report fails closed with stderr + exit 2 instead of stringifying as green). Adapter-agnostic: consumed by the Edge Probe Module today and the prohibition probe (#644) next. Exports: `VALID_STATUS`, `validateResolution`, `validateRequirement`, `analyzeCoverage`, `runProbeCli`. Source of truth: `gsd-core/bin/lib/probe-core.cjs` (generated from `src/probe-core.cts`, gitignored per ADR-457). Tests: `tests/probe-core.test.cjs`. See ADR-550 and Edge Probe Module. Under ADR-857 (phase-6 boundary, settled 2026-06-12) this seam is classified **core verification substrate** on the *contract* side: its deterministic validators are the verifier↔predicate contract's CI-testable surface (ADR-550 Decision 5) — core and non-toggleable, never an off-by-default Feature Capability. (The recall-gapped *generator* is the probe adapters that propose predicates, not this resolution engine — see Edge Probe Module and Verification substrate (predicate boundary).) +Generic spec-phase probe resolution model — the shared seam underlying spec-completeness probes (ADR-550 Decision 7). Owns the `status × verification` model (`status: resolved | dismissed | unresolved` × a per-probe `verification` tier), structural validation (`validateResolution`, `validateRequirement` — fail-closed: `verification` must be null unless status is `resolved`, and an out-of-enum status, a `dismissed`-without-`reason`, or an `unresolved` carrying a `resolution`/`reason`/tier payload all throw rather than silently miscount), the `analyzeCoverage(items, resolutions?, validators)` merge/rollup/orphan-reject pipeline, the `byVerification` per-tier rollup, and the `runProbeCli` I/O scaffold (parse → validate → analyze → emit, structurally guarding the report shape before write — a malformed report fails closed with stderr + exit 2 instead of stringifying as green). Adapter-agnostic: consumed by the Edge Probe Module today and the Prohibition Probe Module (#644) next. Exports (generic surface): `VALID_STATUS`, `validateResolution`, `validateRequirement`, `analyzeCoverage`, `runProbeCli` — the prohibition adapter exports that also ship from this module (`projectProhibitions`, `PROHIBITION_VALIDATORS`, `validateProhibitionResolution`, `dispositionForProhibition`) are documented under the Prohibition Probe Module's own locked-surface line. Source of truth: `gsd-core/bin/lib/probe-core.cjs` (generated from `src/probe-core.cts`, gitignored per ADR-457). Tests: `tests/probe-core.test.cjs`. See ADR-550 and Edge Probe Module. Under ADR-857 (phase-6 boundary, settled 2026-06-12) this seam is classified **core verification substrate** on the *contract* side: its deterministic validators are the verifier↔predicate contract's CI-testable surface (ADR-550 Decision 5) — core and non-toggleable, never an off-by-default Feature Capability. (The recall-gapped *generator* is the probe adapters that propose predicates, not this resolution engine — see Edge Probe Module and Verification substrate (predicate boundary).) ### Edge Probe Module First adapter of the Probe Core Module (ADR-550 Decision 7): the spec-phase edge-completeness probe wired into spec-phase Step 5.5. Owns shape classification (`classifyShape`), the applicable-category relevance filter (`applicableCategories` over the 8-category edge `TAXONOMY`), edge proposal (`proposeEdges`), and the `{explicit, backstop}` verification validators; delegates merge/rollup/CLI to `probe-core`. Fail-closed input contract: an edge requirement with missing/empty `text` and no `shapes` override is rejected (a `{ id }`-only requirement no longer classifies to zero edges and silently drops), while the legitimate `shapes: []` opt-out is preserved. Downstream, the `plan-phase` planner lifts every `covered`/`backstop` edge from the SPEC `## Edge Coverage` section into `must_haves.truths`. Exports (locked surface): `classifyShape`, `applicableCategories`, `proposeEdges`, `analyzeCoverage`, `validateResolution`, `validateRequirement`, plus the constants `TAXONOMY` (the 8 edge categories), `VALID_SHAPES`, `SHAPE_CUES`, and `EDGE_VALIDATORS` (the `{explicit, backstop}` validators bundle injected into `probe-core`'s generic engine). Source of truth: `gsd-core/bin/lib/edge-probe.cjs` (generated from `src/edge-probe.cts`, gitignored per ADR-457). Tests: `tests/edge-probe.test.cjs`, `tests/edge-probe-spec-phase-contract.test.cjs`, `tests/edge-probe-planner-contract.test.cjs`. See ADR-550 and Probe Core Module. Per ADR-857's phase-6 boundary (2026-06-12) the predicates this module generates are **core verification substrate** (they set the verifier's reach), so it is wired onto the core predicate rail as a core-default module rather than migrated to an off-by-default `capabilities/edge-probe/` Feature Capability. @@ -243,6 +243,18 @@ First adapter of the Probe Core Module (ADR-550 Decision 7): the spec-phase edge ### Verification substrate (predicate boundary) The ADR-857 classification (settled 2026-06-12, prompted by @davesienkowski's boundary analysis on #857) that predicate-generation — the must-NOT-have / edge predicates that set the verifier's reach — is **core**, not an off-by-default Feature Capability. Load-bearing premise: *verifier reach = spec reach* (the verifier can only catch what the spec concretely names). Decomposes into: the **verifier↔predicate contract** (the verifier always expects predicates and grades **exogenously** against them — core, non-toggleable, a stability contract alongside the Loop Extension Point names), and the **generator** (the probe *adapters* that propose predicates — edge-probe's `classifyShape`/`proposeEdges`, the prohibition probe's adversarial LLM-propose — core-default but independently versionable, kept their own module because of a measured recall gap; `probe-core`'s deterministic validators sit on the contract side, not the generator side). Altitude rule distinguishing it from `gate` hooks: a gate runs *against* the spec (hook); predicate-generation defines the spec's *reach* (core). The decision-#6 `produces`/`consumes` artifact flow is the internal rail from generation to the core verifier. See ADR-857 *Verification substrate vs. plug-in tier (the predicate boundary)* and the ADR-550 cross-reference. +### Probe Family +The set of spec-phase completeness probes that share the Probe Core Module seam (ADR-550 Decision 7): the Edge Probe (Step 5.5, data-shape edges) and the Prohibition Probe (Step 5.6, unwritten *must-NOT* constraints) today, with room for a third nearly-free adapter. A "probe" walks each SPEC requirement, surfaces candidate omissions, and resolves each through the shared `status × verification` model — but each family member owns its own *recall* mechanism: a deterministic closed-taxonomy classifier for edges (shape→category), open-vocabulary adversarial LLM prose for prohibitions (recall is model-driven, not a compute adapter — ADR-550 D7b). What is shared is the resolution/validation/rollup engine and the soft-gate lifecycle; what diverges is how candidates are recalled. See Probe Core Module, Edge Probe Module, Prohibition Probe Module. + +### Verification Tier +The orthogonal `verification` dimension a resolved probe item carries alongside its `status` (ADR-550 D7a — `status: resolved | dismissed | unresolved` is the shared resolution lifecycle; `verification` is the probe-defined enforcement axis). Each probe defines its own tier vocabulary: the edge probe uses `explicit | backstop`; the prohibition probe uses `test | judgment`. For prohibitions the tier names *how* a must-NOT can be enforced — `test` (a negative test can fail-close on it) vs `judgment` (an irreducible values/safety rule only human/LLM judgment can assess). Verify-phase routes on the tier: `test`-tier items must be provably wired and fail closed when unwired (never a silent green); `judgment`-tier items take the mode-dependent soft-gate (ADR-550 D4 — interactive demands human resolution, autonomous records a non-authoritative LLM-judge verdict + an `unverified-prohibition` flag, never a silent pass and never a hard halt). The rollup exposes `coverage.byVerification: { : count }` so verify-phase reads the per-tier denominator without re-scanning. See Probe Core Module, Prohibition Probe Module. + +### Bespoke vs Canon Prohibition +The ownership seam between the prohibition probe and security/compliance tooling (ADR-550 D6). The probe owns **bespoke** product/values prohibitions — the unwritten must-NOTs specific to this feature's intent (e.g. "the streak reminder must not manipulate the user into returning"). When precision classifies an item as a **canon** security/compliance concern (OWASP / GDPR / fairness / prototype-pollution / path-traversal — the codified, cross-project rule sets), the probe does **not** mint a SPEC prohibition: it emits a one-line breadcrumb (*"possible canon-security concern X — owned by `/gsd:secure-phase` / eslint"*) and stops. Canon checks are **referred, not duplicated** — keeping the surfaced list short (#644's ~2–3-item precision goal) and the secure-phase boundary explicit. See Prohibition Probe Module. + +### Prohibition Probe Module +Second adapter of the Probe Core Module (ADR-550 Decision 7): the spec-phase prohibition-completeness probe wired into spec-phase Step 5.6, surfacing the unwritten *must-NOT* constraints (values/safety/ethics) the spec never forbids. Unlike the Edge Probe, recall is **prose-orchestrated, not a compiled engine** (ADR-550 D7b) — a two-stage pass per requirement: Stage 1 an adversarial recall question, Stage 2 a one-pass precision classifier (drop routine engineering, keep genuine prohibitions). The code surface is schema/projection only: `projectProhibitions()` (deterministic SPEC↔`must_haves.prohibitions` projection backing the `DEFECT.GENERATIVE-FIX` parity assertion), the `{test, judgment}` `PROHIBITION_VALIDATORS`, `validateProhibitionResolution`, and `dispositionForProhibition()` (the fail-closed default — an unwired `test`-tier item resolves to `unverified`/flagged, never green). No `proposeProhibitions()` — recall is LLM prose. `plan-phase` lifts every resolved prohibition from the SPEC `## Prohibitions (must-NOT)` section into the `must_haves.prohibitions` sibling block (never `truths`). Exports (locked surface): `projectProhibitions`, `PROHIBITION_VALIDATORS` (the `{test, judgment}` validators bundle injected into `probe-core`'s generic engine), `validateProhibitionResolution`, and `dispositionForProhibition` (the fail-closed disposition) — the prohibition adapter surface, shipped from `probe-core` alongside the generic engine. Source of truth: `gsd-core/bin/lib/probe-core.cjs` (the prohibition exports live in `src/probe-core.cts`, gitignored per ADR-457) + `gsd-core/references/prohibition-probe.md`. Tests: `tests/prohibition-probe.*.test.cjs`. See ADR-550, Probe Core Module, Edge Probe Module, Verification Tier, Bespoke vs Canon Prohibition. + ### MVP Mode Phase-level planning mode that frames work as a vertical slice (UI → API → DB) of one user-visible capability instead of horizontal layers. Resolved at workflow init via the precedence chain: `--mvp` CLI flag → ROADMAP.md `**Mode:** mvp` field → `workflow.mvp_mode` config → false. All-or-nothing per phase (PRD #2826 Q1). Surfaced as `MVP_MODE=true|false` to the planner, executor, verifier, and discovery surfaces (progress, stats, graphify). Canonical parser: `roadmap.cjs` `**Mode:**` field; canonical resolution chain documented in `workflows/plan-phase.md`. Concept index: `references/mvp-concepts.md`. diff --git a/agents/gsd-verifier.md b/agents/gsd-verifier.md index ca4c187c1..4f7e8ad74 100644 --- a/agents/gsd-verifier.md +++ b/agents/gsd-verifier.md @@ -85,7 +85,7 @@ cat "$PHASE_DIR"/*-VERIFICATION.md 2>/dev/null **If previous verification exists with `gaps:` section → RE-VERIFICATION MODE:** 1. Parse previous VERIFICATION.md frontmatter -2. Extract `must_haves` (truths, artifacts, key_links) +2. Extract `must_haves` (truths, artifacts, key_links, prohibitions) 3. Extract `gaps` (items that failed) 4. Set `is_re_verification = true` 5. **Skip to Step 3** with optimization: @@ -140,8 +140,19 @@ must_haves: - from: "src/components/Chat.tsx" to: "src/app/api/chat/route.ts" via: "fetch in useEffect — calls /api/chat endpoint" + prohibitions: + - statement: "MUST NOT store raw SSN in plaintext" + status: "resolved" + verification: "judgment" ``` +**Also extract `must_haves.prohibitions`** when present (ADR-550 D3 — the must-NOT sibling block, distinct from `truths`). Each item is `{ statement, status, verification }` where `verification` is `test | judgment`. These are NEGATIVE checks: a verified prohibition means the must-NOT did NOT happen. Route them by verification tier in the verdict assembly (ADR-550 D4, the "B-with-guard" 2026-06-12 maintainer decision): + +- **judgment-tier prohibitions → mode-dependent soft-gate.** Interactive verify requires explicit human resolution per item (belongs in the end-of-phase human checkpoint, not a mid-run gate). Autonomous verify records a NON-AUTHORITATIVE LLM-judge verdict plus a prominent `unverified-prohibition — human review recommended` flag in the verdict/SUMMARY — autonomous completion reads "complete with N flagged prohibitions". NEVER a silent pass; NEVER a hard halt of an AFK run. +- **test-tier prohibitions → FAIL CLOSED (accept-and-flag, not reject-at-parse).** Accept the `verification: test` value (the SPEC↔must_haves.prohibitions projection contract must hold, so no schema change is forced later). But a well-formed test-tier item that reaches verify with NO wired enforcement is treated as UNVERIFIED — flagged exactly like an unresolved judgment item, NEVER green. The deterministic fail-closed default is `dispositionForProhibition()` in probe-core (status `unverified`, `flagged: true` when `enforcementEvidence` is empty). Do NOT wire a real fail-first negative-test hard gate here — that enforcement MECHANISM defers to a follow-up PR (it needs a real test-tier consumer to `regression-must-fail-first` against; #644's corpus is entirely judgment-tier). + +A flagged prohibition counts as a human-verification item (status `human_needed`) or a gap (status `gaps_found`) per the existing decision tree — it must never be silently absorbed into a `passed` verdict. + **Step 2c: Merge must-haves** Combine all sources into a single must-haves list: diff --git a/docs/COMMANDS.md b/docs/COMMANDS.md index 7ec15c951..a05897c3e 100644 --- a/docs/COMMANDS.md +++ b/docs/COMMANDS.md @@ -107,6 +107,8 @@ Clarify WHAT a phase delivers through Socratic questioning with quantitative amb **Edge Coverage (Step 5.5):** After the ambiguity gate passes, spec-phase runs an edge-completeness probe over each requirement. It raises only applicable categories from a closed 8-category taxonomy (boundary, adjacency, empty, encoding, ordering, precision, idempotency, concurrency), proposes one concrete candidate edge per category, and records each as `covered` / `dismissed` (reason required) / `backstop` / `unresolved` in a `## Edge Coverage` SPEC section. Unresolved applicable edges soft-gate the spec (Resolve / Write-anyway-flagged / Keep-probing); `covered` and `backstop` edges are later lifted into plan-phase `must_haves`. Under `--auto` the probe **never auto-dismisses** — it auto-covers where a defensible acceptance criterion exists, otherwise auto-backstops. +**Prohibition Coverage (Step 5.6):** After the edge probe, spec-phase runs a prohibition-completeness probe — a two-stage prose pass (adversarial recall → precision classifier) that surfaces the unwritten *must-NOT* constraints (values/safety/ethics) the spec never forbids. Each is resolved to `resolved` (a NEGATIVE acceptance criterion, carrying a `test` or `judgment` verification tier) / `dismissed` (reason required) / `unresolved`, recorded in a `## Prohibitions (must-NOT)` SPEC section. Resolved prohibitions are lifted into plan-phase `must_haves.prohibitions`; judgment-tier items soft-gate at verify time (never silent, never hard-halt) and unwired test-tier items fail closed. Under `--auto` the probe **never auto-dismisses**; canon-bound concerns (OWASP / GDPR / fairness) are referred to `/gsd:secure-phase`. + **Prerequisites:** `.planning/ROADMAP.md` exists **Produces:** `{phase}-SPEC.md` (with a `## Edge Coverage` section) diff --git a/docs/FEATURES.md b/docs/FEATURES.md index c7cab7fc1..0797f176f 100644 --- a/docs/FEATURES.md +++ b/docs/FEATURES.md @@ -168,6 +168,7 @@ - [Spec-Phase Edge-Completeness Probe](#144-spec-phase-edge-completeness-probe) - [v1.43.0 Features](#v1430-features) - [MemPalace Memory Capability](#145-mempalace-memory-capability) + - [Spec-Phase Prohibition Probe](#146-spec-phase-prohibition-probe) --- @@ -3145,3 +3146,36 @@ The load-bearing wire is the `plan-phase` lift: `covered` and `backstop` edges b **Configuration:** `mempalace.enabled`, `mempalace.memory_mode`, `mempalace.wing`, `mempalace.recall_on_discuss`, `mempalace.recall_on_plan`, `mempalace.capture_artifacts`, `mempalace.mirror_kg`, `mempalace.cross_project_tunnels`, `mempalace.diary_journal`, `mempalace.auto_capture_hooks` See [Configuration Reference](CONFIGURATION.md#mempalace-settings) for full schema and [How to enable cross-session memory with MemPalace](how-to/enable-cross-session-memory-with-mempalace.md) for a setup walkthrough. + +### 146. Spec-Phase Prohibition Probe + +**Command:** `/gsd-spec-phase` + +**Purpose:** Surface the unwritten *must-NOT* constraints — the values/safety/ethics interpretations a feature could silently become that the author would never want but the spec does not forbid — before any code is written. The edge probe reaches data-shape edges; it structurally cannot reach prohibitions. This is the missing instrument, running as `Step 5.6` of spec-phase, after the edge probe. + +**Behavior:** A two-stage, prose-orchestrated pass per requirement (no compiled recall engine — recall is inherently model-driven, ADR-550 D7b): + +1. **Recall (adversarial probe):** *"What could this feature silently become that the author would NOT want, but the spec does not forbid?"* — model-robust open-vocabulary elicitation across values/safety/ethics. +2. **Precision (one-pass classifier):** drop routine-engineering items, keep genuine values/safety/ethics prohibitions — collapses the raw list to the load-bearing few. + +Each surfaced prohibition is resolved to exactly one of three states: + +| State | Meaning | Downstream effect | +|-------|---------|-------------------| +| `resolved` | Confirmed a real must-NOT | NEGATIVE acceptance criterion written into the SPEC `## Prohibitions (must-NOT)` section; lifted into `plan-phase` `must_haves.prohibitions` (its own sibling block, never `truths`) | +| `dismissed` | Not a genuine prohibition (requires a non-empty reason) | Recorded with its reason; empty dismissals are rejected | +| `unresolved` | Deferred | Soft-gates the spec; surfaced as a planner assumption | + +Each resolved prohibition carries a `verification` tier — `test` (a negative test can enforce it) or `judgment` (only human/LLM judgment can). At verify time, judgment-tier prohibitions route to a never-silent / never-hard-halt soft gate (autonomous emits an `unverified-prohibition — human review recommended` flag); test-tier prohibitions fail closed when unwired (never silently green). Under `--auto`, the probe **never auto-dismisses**. Canon-bound concerns (OWASP / GDPR / fairness) are referred to `/gsd:secure-phase` rather than minting SPEC prohibitions (ADR-550 D6). + +The load-bearing wire is the `plan-phase` lift into `must_haves.prohibitions`, so the section is not merely documentation. + +**Requirements:** +- REQ-PROHIB-01: The prohibition pass MUST run after the edge probe and emit a `## Prohibitions (must-NOT)` SPEC section. +- REQ-PROHIB-02: Stage 1 MUST ask the adversarial recall question; Stage 2 MUST drop routine-engineering items and keep values/safety/ethics prohibitions. +- REQ-PROHIB-03: A `dismissed` resolution MUST require a non-empty reason. +- REQ-PROHIB-04: `--auto` MUST never auto-dismiss. +- REQ-PROHIB-05: `plan-phase` MUST lift resolved prohibitions into `must_haves.prohibitions` (never `truths`). +- REQ-PROHIB-06: A well-formed but unwired `test`-tier prohibition MUST fail closed at verify time — never a silent pass. + +**Reference:** [Prohibition Probe](../gsd-core/references/prohibition-probe.md) diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index c5248647b..2b593e958 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -238,6 +238,7 @@ "planner-revision.md", "planner-source-audit.md", "planning-config.md", + "prohibition-probe.md", "project-skills-discovery.md", "questioning.md", "research-documentation-lookup.md", diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index 3bf6f93f4..eb7dfba87 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -304,6 +304,7 @@ Full roster at `gsd-core/references/*.md`. References are shared knowledge docum | `continuation-format.md` | Session continuation/resume format. | | `domain-probes.md` | Domain-specific probing questions for discuss-phase. | | `edge-probe.md` | Spec-phase edge-completeness probe — 8-category edge taxonomy, shape classification, and the `requirements → checks → verifier` resolution model (Step 5.5). | +| `prohibition-probe.md` | Spec-phase prohibition-completeness probe — the two-stage adversarial-recall → precision protocol that surfaces the unwritten *must-NOT* constraints (values/safety/ethics), with status×verification (`test`/`judgment`) tiering and canon-referral breadcrumbs (Step 5.6); second adapter of the `probe-core` resolution model. | | `gate-prompts.md` | Gate/checkpoint prompt templates. | | `loop-hook-dispatch.md` | Generic dispatch contract for consuming `gsd_run loop render-hooks --raw` output in any host-loop workflow — envelope shape, per-kind dispatch rules (contribution/step/gate), and liveness banner. | | `scout-codebase.md` | Phase-type→codebase-map selection table for discuss-phase scout step (extracted via #2551). | diff --git a/docs/README.md b/docs/README.md index ddfc492a0..edeaa1111 100644 --- a/docs/README.md +++ b/docs/README.md @@ -19,6 +19,7 @@ Language versions: [English](README.md) · [Português (pt-BR)](pt-BR/README.md) - [Install a minimal GSD and add skills later](how-to/install-minimal-and-add-skills.md) — install only the core skills, then grow the surface with profiles and `/gsd:surface` - [Discuss a phase](how-to/discuss-a-phase.md) — capture implementation decisions before planning begins - [Resolve edge-coverage findings](how-to/resolve-edge-coverage-findings.md) — turn the spec phase's surfaced domain-boundary edges into covered, dismissed, or backstopped spec decisions +- [Resolve prohibition findings](how-to/resolve-prohibition-findings.md) — turn the spec phase's surfaced must-NOT constraints into resolved, dismissed, or deferred spec decisions - [Plan a phase](how-to/plan-a-phase.md) — run research, decompose work, and verify plan quality - [Execute a phase](how-to/execute-a-phase.md) — run plans in parallel waves with fresh-context subagents - [Verify and ship](how-to/verify-and-ship.md) — walk through completed work, diagnose failures, and create the PR diff --git a/docs/adr/550-spec-phase-probe-contract.md b/docs/adr/550-spec-phase-probe-contract.md index 16f9ba558..0567658ef 100644 --- a/docs/adr/550-spec-phase-probe-contract.md +++ b/docs/adr/550-spec-phase-probe-contract.md @@ -84,3 +84,12 @@ The capability system (ADR-857) classifies **predicate-generation as core verifi - **"Exogenous grading"** is the verify-time half of this ADR's **judgment-tier** (Decision 4): the autonomous verifier records an LLM-judge verdict **marked non-authoritative** and flags *"unverified prohibition — human review recommended"* — grading against the spec's explicit predicates, never a self-graded restatement (the prose-drift / self-graded-review failure modes reproduced on #664). The judgment-tier soft-gate **is** the contract's "never a silent pass" guarantee at the core boundary. - **The probe *adapters* are the core-default *generator*** — `edge-probe`'s `classifyShape`/`proposeEdges` and the prohibition probe's adversarial LLM-propose, the surfaces that *propose* predicates — default-on and non-removable, but **independently versionable**. The classifier's measured recall gap (the prose→shape under-fire on terse prose) is the reason they stay their own modules under this ADR rather than being folded into the slow core rail. **`probe-core` is not the generator:** per Decision 7b it ingests already-proposed items, and its deterministic validators are the **contract**'s CI-testable surface (Decision 5) — so `probe-core` sits on the contract side, the adapters on the generator side. ADR-857 phase 6 wires these modules onto the core predicate rail; it does **not** relabel them as a `capabilities/edge-probe/` plug-in. - Decision 5's rule — *the CI-testable surface is the contract, not the classifier* — extends to ADR-857's core rail: the deterministic conformance test is the contract shape the verifier consumes, never the LLM's judgment. + +## Addendum (2026-06-12): test-tier disposition — fail-closed now, heavy enforcement deferred + +Decision 4 describes the `test`-tier as a "**Hard gate in both interactive and autonomous modes.**" The #644 implementation revises that to a **fail-closed-now / deferred-enforcement** resolution (the "B-with-guard" maintainer decision of 2026-06-12), so the architecture-of-record matches the shipped code: + +- A well-formed but **unwired** `test`-tier prohibition resolves via `dispositionForProhibition()` to `{ status: 'unverified', flagged: true }` — **provably never green** without explicit evidence (REQ-PROHIB-06). This is the load-bearing safety half and it holds today. +- The **heavy negative-test enforcement mechanism** — a contrived `test`-tier consumer that mechanically runs the negative test — is **deferred to a follow-up**, because the #644 corpus is entirely `judgment`-tier and wiring a synthetic test-tier consumer now would be gold-plating. The fail-closed disposition guards the gap in the meantime: the gate cannot silently pass, it can only report `unverified`/flagged until enforcement lands. + +Net effect on D4: the *guarantee* ("a `test`-tier prohibition is never a silent pass") is preserved exactly; what is deferred is only the mechanical-pass half of the gate, not the never-green safety property. This addendum supersedes the unqualified "hard gate" wording of Decision 4 for `test`-tier items until the enforcement follow-up lands. The decision also lives in `src/probe-core.cts` comments, `verify-phase.md`, and the #644 changeset. diff --git a/docs/how-to/resolve-prohibition-findings.md b/docs/how-to/resolve-prohibition-findings.md new file mode 100644 index 000000000..df47be6b3 --- /dev/null +++ b/docs/how-to/resolve-prohibition-findings.md @@ -0,0 +1,110 @@ +# How to resolve prohibition findings while writing a spec + +**Goal:** Turn each must-NOT the spec phase surfaces into an explicit, checkable spec decision — so a constraint the author assumed but never wrote (the reminder that must not shame the user, the model that must not proxy on protected attributes, the log that must not store raw PII) becomes an acceptance criterion the verifier can enforce *before* any code exists, instead of an unwritten intent a literal `✅ done` can silently violate. + +**Prerequisites:** A phase whose `/gsd-spec-phase` run has passed the ambiguity gate and the edge-completeness probe (Step 5.5). The prohibition probe (Step 5.6) then runs automatically over the same requirements and presents its findings — you do not invoke it separately. + +For the two-stage recall→precision protocol, the canon-referral rule, and the `status × verification` schema, see [Spec-Phase Prohibition Probe](../FEATURES.md#146-spec-phase-prohibition-probe). This guide covers only how to *act* on the findings. + +--- + +## Read a finding + +Each finding is one **bespoke** must-NOT for one requirement — a values, safety, fairness, privacy, or transparency constraint the probe kept after dropping routine-engineering noise. A finding is phrased as a must-NOT statement, for example: + +> **R1 · values** — MUST NOT use shaming, guilt, or loss-aversion streak framing (e.g. "Don't lose your streak!") — the reminder must encourage without penalty framing + +You resolve each finding into one of three states. Claude presents them as a numbered choice (or an `AskUserQuestion` menu). + +--- + +## Keep it — write a negative acceptance criterion + +**Choose this when the prohibition is genuine for this feature.** Claude writes a must-NOT line into the spec's **Acceptance Criteria**, marks the prohibition `resolved`, and you pick its **verification tier**: + +- **`test`** — a mechanical check can prove it: a negative test, a lint rule, an assertion that the audit log contains no raw SSN. Choose this when a green/red check exists. +- **`judgment`** — the prohibition is real but cannot be reduced to a mechanical test (e.g. "the framing is not manipulative"). It records intent and routes to a judgment-based review rather than a passing test. + +> - [ ] MUST NOT use shaming or loss-aversion framing in the reminder copy *(judgment)* + +Pick the tier honestly — see [what happens downstream](#what-happens-to-resolved-findings-downstream) for why a `test`-tier prohibition is held to a stricter bar. + +--- + +## Dismiss it — record why it does not apply + +**Choose this when the prohibition genuinely does not apply** — and say why. A dismissal **requires a non-empty reason**; silence is rejected. + +> ⛔ dismissed — pure integer utility; no user-facing surface, no values/safety/privacy dimension + +A wrong dismissal is the exact silent failure this probe exists to prevent, so dismiss only when the reason is solid. The reason string is the audit trail. + +--- + +## Defer it — leave it unresolved and flagged + +**Choose this only when you are not ready to decide.** The prohibition stays `unresolved` and is flagged. Unlike a dismissal, deferring makes no claim that the prohibition is safe — it is an explicit, visible assumption the planner must surface, not silently drop. + +--- + +## When a finding is canon, not yours + +Some candidates are **canon** security/compliance constraints a dedicated tool already owns (OWASP / prototype-pollution / path-traversal / injection → `/gsd-secure-phase` + eslint; GDPR retention / consent → `/gsd-secure-phase`). The probe does not surface these as findings to resolve — it emits a one-line breadcrumb and drops them, so the list you triage stays the ~2–3 **bespoke** prohibitions no other tool would catch. You do not act on a breadcrumb here; pick it up in `/gsd-secure-phase`. + +--- + +## Clear the soft gate + +After you have worked through the findings, the probe runs a **soft gate**: + +- **All applicable prohibitions resolved** → the spec proceeds to the next step. +- **One or more still `unresolved`** → Claude asks what to do: + - **Resolve now** — loop back and resolve the remaining prohibitions. + - **Write the spec anyway** — the spec is written with those rows marked `⚠ Prohibition unresolved — planner must treat as assumption`. Use this deliberately; you are choosing to ship a known gap. + - **Keep probing** — continue surfacing. + +The gate is *soft*: it never blocks you, but every unresolved prohibition stays visible in the spec's `## Prohibitions` section. + +--- + +## Let Claude resolve them for you + +**If the prohibitions are low-stakes or already settled by earlier phases**, run the spec phase in auto mode: + +```bash +/gsd-spec-phase 3 --auto +``` + +In `--auto`, Claude marks a prohibition `resolved` where it can write a defensible negative acceptance criterion (at the `test` or `judgment` tier), and leaves it `unresolved` otherwise. It **never auto-dismisses** — dismissing a prohibition requires a human reason, because a wrong auto-dismissal is precisely the silent failure being eliminated. Claude logs the tally, for example: + +``` +[auto] prohibitions: 2 resolved, 1 unresolved +``` + +Review the logged choices afterwards; auto mode is a fast first pass, not a substitute for judgement on a prohibition that carries risk. + +--- + +## What happens to resolved findings downstream + +When you next run `/gsd-plan-phase`, the planner reads the spec's `## Prohibitions` section and: + +- lifts every confirmed prohibition into the plan's `must_haves.prohibitions`, +- carries the `test`-tier and `judgment`-tier distinction with it, +- surfaces every `unresolved` prohibition as an explicit assumption. + +At verify time, a `test`-tier prohibition whose mechanical check is not yet wired is held **fail-closed** — it reports as flagged/unverified, never as a silent pass — so a prohibition can never quietly disappear between spec and verification. A `judgment`-tier prohibition routes to a judgment-based review rather than a green/red test. + +This is the payoff: a resolved prohibition becomes a checkable negative the goal-backward verifier accounts for, extending its reach to the must-NOT the requirement prose never stated. + +--- + +## Related + +- [Spec-Phase Prohibition Probe](../FEATURES.md#146-spec-phase-prohibition-probe) — protocol, schema, host-mapping table, and the front-of-pipeline rationale +- [Resolve edge-coverage findings](resolve-edge-coverage-findings.md) — the sibling probe on the data/behavior-shape axis +- [`/gsd-spec-phase`](../COMMANDS.md#gsd-spec-phase) — command reference and flags +- [Plan a phase](plan-a-phase.md) — where resolved prohibitions become `must_haves.prohibitions` +- [docs index](../README.md) + + diff --git a/gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json b/gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json new file mode 100644 index 000000000..9709cbdd6 --- /dev/null +++ b/gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json @@ -0,0 +1,14 @@ +{ + "items": [ + { + "requirement_id": "R1", + "category": "values", + "status": "resolved", + "verification": "judgment", + "resolution": null, + "reason": null, + "statement": "MUST NOT use shaming, guilt, or loss-aversion streak framing (e.g. \"Don't lose your streak!\") — the reminder must encourage without penalty framing" + } + ], + "coverage": { "applicable": 1, "resolved": 1, "unresolved": 0, "byVerification": { "test": 0, "judgment": 1 } } +} diff --git a/gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json b/gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json new file mode 100644 index 000000000..d20cc0199 --- /dev/null +++ b/gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json @@ -0,0 +1,4 @@ +{ + "items": [], + "coverage": { "applicable": 0, "resolved": 0, "unresolved": 0, "byVerification": { "test": 0, "judgment": 0 } } +} diff --git a/gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json b/gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json new file mode 100644 index 000000000..d2f3f32e5 --- /dev/null +++ b/gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json @@ -0,0 +1,32 @@ +{ + "items": [ + { + "requirement_id": "R1", + "category": "fairness", + "status": "resolved", + "verification": "judgment", + "resolution": null, + "reason": null, + "statement": "MUST NOT use protected attributes (race, gender, age, national origin) or their proxies (zip code, name) in the loan decision or rate" + }, + { + "requirement_id": "R1", + "category": "privacy", + "status": "resolved", + "verification": "test", + "resolution": null, + "reason": null, + "statement": "MUST NOT store raw PII / financial secrets (SSN, full account or card numbers) in plaintext in the audit log" + }, + { + "requirement_id": "R1", + "category": "transparency", + "status": "resolved", + "verification": "judgment", + "resolution": null, + "reason": null, + "statement": "MUST NOT mislead or omit the true rate/APR/terms in the explanation; an adverse decision must state the real principal reason (adverse-action)" + } + ], + "coverage": { "applicable": 3, "resolved": 3, "unresolved": 0, "byVerification": { "test": 1, "judgment": 2 } } +} diff --git a/gsd-core/references/prohibition-probe.md b/gsd-core/references/prohibition-probe.md new file mode 100644 index 000000000..1915d2d23 --- /dev/null +++ b/gsd-core/references/prohibition-probe.md @@ -0,0 +1,248 @@ +# Prohibition-Probe — Spec-Completeness Must-NOT Reference + +Shared reference for the spec/requirements phase. Companion to +`@~/.claude/gsd-core/references/edge-probe.md`: `edge-probe` reaches the +**data/behavior-shape axis** (boundaries, adjacency, encoding, ordering) — the things a +feature must *do*. This reference reaches the orthogonal **must-NOT axis** (product, values, +safety, ethics) — the things a feature must *never silently become*. The edge-probe caught +0/8 of these in controlled testing because it is the wrong instrument: a shape taxonomy +cannot surface "the reminder must not shame the user." Walk each requirement through the +two-stage recall→precision protocol below and resolve each surfaced prohibition to exactly +one state. + +This doc is written in generic `requirements → checks → verifier` terms with no +tool-specific vocabulary, so it is portable: copy it into any spec/requirements process. +A short mapping table at the end binds it to common host structures. + +## Why front-of-pipeline + +A goal-backward verifier only checks assertions that exist; an assertion only exists for a +requirement that was written down. The class of constraint this probe targets — the +*"must-NOT"* the author assumed but never wrote — is invisible to the verifier in exactly +the same way an omitted edge is, but with a sharper failure mode: a `✅ done` that means +"the code matches the words in the spec" can still ship a feature that does what the author +explicitly would *not* want. The manipulative-streak reminder, the loan model that proxies +on zip code, the audit log that stores raw SSN — each one passes a literal spec while +violating the intent. The fix is not a better verifier; it is **spec completeness**: surface +the omitted prohibition into an explicit, checkable acceptance criterion *before* any code +exists, after which the verifier reliably enforces it. + +The technique is adversarial elicitation, not deterministic computation. Unlike the edge +taxonomy (a closed eight categories a classifier can apply), the recall stage is inherently +model-driven: it asks an open question and reads prose. There is **no compiled +`prohibition-probe.cjs` engine** — the recall stage is an LLM prose pass, and only the +schema/projection layer is real code (ADR-550 Decision 7b). Building a deterministic +recall adapter would be the scope-creep the maintainer flags. + +## Inputs + +A list of requirements, each a `{ id, text }` record where `text` is a testable statement. +There is no shape override and no taxonomy classifier — the recall stage reads the prose +directly and the precision stage filters its raw output. The probe runs **after** the +edge-probe in the spec phase, over the same requirement list. + +## Two-stage protocol (recall → precision) + +The probe is a two-pass pipeline per requirement. Stage 1 maximizes recall (cast wide); +Stage 2 restores precision (drop the noise). Running them in this order — wide then narrow — +is what keeps the surfaced list both complete and short. + +**Stage 1 — Recall (adversarial probe).** Ask the single adversarial question of each +requirement: + +> *What could this feature silently become that the author would NOT want, but the spec +> does not forbid?* + +This question is model-robust (17/17 holistic surfacing including smaller models in the N18 +experiment). It deliberately over-produces: ~10 raw candidates per requirement, including +routine engineering items. That over-production is intentional — recall first. + +**Stage 2 — Precision (one-pass classifier).** Filter the raw Stage-1 list in a single pass. +The rule is a drop/keep split: + +- **DROP routine-engineering items** — anything that is a normal correctness or hygiene + concern rather than an intent constraint: "must not mutate its input", "must not throw on + empty list", "must return a primitive not an object", "must not leak a file handle". These + belong to the edge-probe or to ordinary code review, not here. +- **KEEP values / safety / ethics items** — anything that, if violated, makes the feature do + something the author would object to on product, fairness, privacy, transparency, or + safety grounds: "must not use shaming framing", "must not proxy on protected attributes", + "must not store raw PII in plaintext". + +This collapses the raw ~10 to ~2–3 genuine prohibitions (GT 5/5, 0 false positives on the +N18 eight-spec battery). A requirement that yields zero kept prohibitions emits an empty +list — that is the correct precision outcome for a pure utility, not a failure. + +## Canon-referral (do not mint canon items) + +Some kept candidates are not bespoke at all — they are **canon** security/compliance +constraints that a dedicated tool already owns. Do NOT mint a prohibition for them. Instead +emit a one-line breadcrumb and stop: + +- OWASP / prototype-pollution / path-traversal / injection → breadcrumb to `/gsd:secure-phase` + and `eslint` (security plugins), not a minted prohibition. +- GDPR / data-retention / consent → breadcrumb to `/gsd:secure-phase`. +- Generic fairness/bias canon → breadcrumb to `/gsd:secure-phase`. + +The breadcrumb reads like: *"prototype-pollution is canon — covered by /gsd:secure-phase + +eslint; not minted here."* This keeps the surfaced list to the ~2–3 **bespoke** items that +no other tool would catch — the manipulative-framing prohibition, the product-specific +fairness constraint — which is the whole value of the probe. Minting canon items both +duplicates other tooling and drowns the bespoke signal (ADR-550 Decision 6). + +## Resolution states + +Each surfaced prohibition carries two orthogonal axes — a resolution **lifecycle** and, when +resolved, a **verification** tier (ADR-550 Decision 7, the shared probe-core model; the +lifecycle is identical to the edge-probe, the verification tiers differ): + +- **status** — `resolved | dismissed | unresolved` (IDENTICAL to the edge-probe): + - **resolved** — the prohibition is addressed; *how* it is addressed is the verification tier. + - **dismissed** — not a genuine prohibition for this feature, accompanied by a required, + non-empty reason string. "N/A — this utility has no user-facing surface, no values + constraint applies" is valid; silence is not. The reason string is the audit trail. + - **unresolved** — carried forward and flagged; the author chose not to resolve it yet. +- **verification** (only when `status` is `resolved`; `null` otherwise) — `test | judgment` + (this REPLACES the edge-probe's `explicit | backstop`): + - **test** — the prohibition can be mechanically checked (a negative test, a lint rule, an + assertion that the audit log contains no raw SSN). A checkable assertion exists. + - **judgment** — the prohibition is real but cannot be reduced to a mechanical test (a + human/LLM judgment that the framing is not manipulative). It records intent and routes + to a judgment-based review rather than a green/red test. + +Splitting these axes keeps the lifecycle enum free of a verification fact and lets the +prohibition adapter declare `test | judgment` without forking the shared lifecycle enum that +the edge-probe's `explicit | backstop` also uses. + +## Output schema + +The probe emits, per kept prohibition, an item of the form: + +``` +{ requirement_id, category, status, verification, resolution, reason, statement } +``` + +where `statement` is the must-NOT sentence and `category` is the values/safety/ethics class +(`values`, `fairness`, `privacy`, `transparency`, `safety`, …), plus a coverage summary: + +``` +coverage: { applicable, resolved, unresolved, byVerification: { test, judgment } } +``` + +`applicable` is the number of kept prohibitions, `resolved` = closed (`resolved` + +`dismissed`) status items, `unresolved` is the remainder, and `byVerification` breaks the +`resolved`-status items down by tier (`{ test, judgment }`). This JSON is the stable contract +both the reference implementation and any third-party port emit. + +## Generic mapping (requirements → checks → verifier) + +| Host structure | "requirement" | a `resolved`/`test` prohibition becomes | a `resolved`/`judgment` prohibition becomes | +|----------------|---------------|------------------------------------------|----------------------------------------------| +| GSD SPEC | a SPEC Requirement | a SPEC acceptance criterion (marked prohibition) that `plan-phase` lifts into `must_haves.prohibitions` | a `must_haves.prohibitions` item routed to judgment review | +| Gherkin feature | a Scenario | a negative `Then` assertion / tagged negative scenario | a tagged scenario routed to manual review | +| OpenAPI operation | an operation | a contract test asserting the forbidden behavior never occurs | a documented constraint flagged for review | +| Docstring contract | a documented behavior | a negative assertion in the contract test | a documented must-NOT for reviewers | + +The portable invariant: a `resolved`/`test` prohibition produces **a checkable negative the +verifier iterates over** (GSD: a `must_haves.prohibitions` item with a test); a +`resolved`/`judgment` prohibition produces a recorded intent routed to judgment review. An +`unresolved` prohibition is an explicit assumption the downstream planner must surface, not +silently drop. + +## Worked example (streak-reminder) + +A single requirement to send a daily habit reminder. The edge-probe sees a `stateful` +requirement and asks about idempotency; the prohibition-probe asks the adversarial question +and surfaces what the reminder must never *become*. Stage 1 over-produces ("must not spam", +"must not throw on a deleted habit", "must not use shaming framing"); Stage 2 drops the +routine-engineering items and keeps the one genuine values prohibition: + +```json prohibition-probe:01-streak-reminder/expected.json +{ + "items": [ + { + "requirement_id": "R1", + "category": "values", + "status": "resolved", + "verification": "judgment", + "resolution": null, + "reason": null, + "statement": "MUST NOT use shaming, guilt, or loss-aversion streak framing (e.g. \"Don't lose your streak!\") — the reminder must encourage without penalty framing" + } + ], + "coverage": { "applicable": 1, "resolved": 1, "unresolved": 0, "byVerification": { "test": 0, "judgment": 1 } } +} +``` + +The kept prohibition is `judgment`-tier: "manipulative framing" cannot be reduced to a +mechanical test, so it records intent and routes to judgment review — but it is now an +explicit acceptance criterion the spec must clear, not an unwritten assumption. + +## Worked example (clean-utility) + +A pure utility requirement — "deduplicate a list of integers" — has no user-facing surface, +no values/safety/ethics dimension. Stage 1 still over-produces ("must not mutate the input", +"must not change order"), but every candidate is routine engineering that Stage 2 drops (and +the edge-probe already owns). The correct precision outcome is an empty prohibition list — a +zero, not a false positive: + +```json prohibition-probe:02-clean-utility/expected.json +{ + "items": [], + "coverage": { "applicable": 0, "resolved": 0, "unresolved": 0, "byVerification": { "test": 0, "judgment": 0 } } +} +``` + +This is the precision discipline that keeps the probe from crying wolf: a utility with no +intent surface produces zero prohibitions, so a non-empty list always carries signal. + +## Worked example (multi-prohibition) + +A loan-decision requirement is the high-stakes case: it surfaces several distinct +prohibitions across categories. Stage 1 produces a long list including canon items +(prototype-pollution, generic GDPR retention) that canon-referral breadcrumbs out; Stage 2 +keeps the three bespoke values/safety items — a `fairness` constraint, a `privacy` constraint +(`test`-tier, mechanically checkable against the audit log), and a `transparency` constraint: + +```json prohibition-probe:03-multi-prohibition/expected.json +{ + "items": [ + { + "requirement_id": "R1", + "category": "fairness", + "status": "resolved", + "verification": "judgment", + "resolution": null, + "reason": null, + "statement": "MUST NOT use protected attributes (race, gender, age, national origin) or their proxies (zip code, name) in the loan decision or rate" + }, + { + "requirement_id": "R1", + "category": "privacy", + "status": "resolved", + "verification": "test", + "resolution": null, + "reason": null, + "statement": "MUST NOT store raw PII / financial secrets (SSN, full account or card numbers) in plaintext in the audit log" + }, + { + "requirement_id": "R1", + "category": "transparency", + "status": "resolved", + "verification": "judgment", + "resolution": null, + "reason": null, + "statement": "MUST NOT mislead or omit the true rate/APR/terms in the explanation; an adverse decision must state the real principal reason (adverse-action)" + } + ], + "coverage": { "applicable": 3, "resolved": 3, "unresolved": 0, "byVerification": { "test": 1, "judgment": 2 } } +} +``` + +The `privacy` row is `test`-tier — "no raw SSN in the audit log" is a mechanical assertion — +while `fairness` and `transparency` are `judgment`-tier. The byVerification rollup +`{ test: 1, judgment: 2 }` is the count-preserved breakdown of the three `resolved`-status +items. Each worked-example block above is kept byte-for-byte (parsed-JSON) identical to its +fixture under `gsd-core/references/prohibition-probe-fixtures/` by +`tests/prohibition-probe.docs-fixtures.test.cjs`, so the doc and the reference data cannot +silently drift. diff --git a/gsd-core/templates/spec.md b/gsd-core/templates/spec.md index 952eb3503..0b7f58c72 100644 --- a/gsd-core/templates/spec.md +++ b/gsd-core/templates/spec.md @@ -80,6 +80,20 @@ No "should feel good", "looks reasonable", or "generally works" — those are no Acceptance Criteria above; `backstop` rows must be carried into plan-phase `must_haves`. `⚠ UNRESOLVED` rows are flagged: planner must treat as assumption.] +## Prohibitions (must-NOT) + +**Coverage:** [resolved]/[applicable] applicable prohibitions resolved · [unresolved] unresolved + +| Prohibition (must-NOT statement) | Requirement | Status | Verification / Reason | +|----------------------------------|-------------|--------|------------------------| +| [MUST NOT … must-NOT statement] | [Rn] | [resolved / dismissed / ⚠ UNRESOLVED] | [verification: test \| judgment, or dismissal reason] | + +[Generated by the prohibition probe (Step 5.6). `resolved` prohibitions become NEGATIVE +acceptance criteria; a `resolved`/`test` row is a checkable negative the verifier iterates +over, a `resolved`/`judgment` row routes to judgment review. Resolved prohibitions are lifted +into `must_haves.prohibitions` by plan-phase. `dismissed` rows carry a required non-empty +reason. `⚠ UNRESOLVED` rows are flagged: planner must treat as assumption.] + ## Ambiguity Report | Dimension | Score | Min | Status | Notes | diff --git a/gsd-core/workflows/plan-phase.md b/gsd-core/workflows/plan-phase.md index e826eb501..3880cd258 100644 --- a/gsd-core/workflows/plan-phase.md +++ b/gsd-core/workflows/plan-phase.md @@ -892,6 +892,7 @@ Output consumed by /gsd:execute-phase. Plans need: - Verification criteria - must_haves for goal-backward verification - If the SPEC has an `## Edge Coverage` section, lift every `covered` edge's acceptance criterion into `must_haves.truths`, and every `backstop` edge into `must_haves.truths` as a non-inferable check (note it needs a held-out/property-based test). `unresolved` edges are explicit assumptions — surface them in the plan, do not silently drop them. +- If the SPEC has a `## Prohibitions` section, lift every resolved prohibition into the `must_haves.prohibitions:` sibling block (NOT `truths` — ADR-550 D3) carrying `statement` + `status` + `verification`; unresolved prohibitions are explicit assumptions — surface them in the plan, do not silently drop them. A prohibition is a must-NOT (negative) check that belongs in its own `must_haves.prohibitions` block. Never place a must-NOT under `must_haves.truths` — that block keeps positive-observable semantics only. - **"Artifacts this phase produces" section (MANDATORY)** — list every symbol this phase creates: decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths. The plan-review-convergence source-grounding pass reads this section to exclude newly-created symbols from drift verification; omitting it causes new symbols to be flagged for acknowledgement. @@ -938,6 +939,7 @@ Every task MUST include these fields — they are NOT optional: - [ ] must_haves derived from phase goal - [ ] Every PLAN.md includes an "Artifacts this phase produces" section listing symbols created by this phase (decorators, classes, functions, CLI flags, struct/dataclass fields, new file paths) - [ ] Every SPEC ## Edge Coverage covered/backstop edge is represented in a plan's must_haves (no silent drops) +- [ ] Every SPEC ## Prohibitions resolved item is represented in a plan's must_haves.prohibitions (no silent drops) ``` diff --git a/gsd-core/workflows/spec-phase.md b/gsd-core/workflows/spec-phase.md index ce004c51d..16266618a 100644 --- a/gsd-core/workflows/spec-phase.md +++ b/gsd-core/workflows/spec-phase.md @@ -312,11 +312,73 @@ failure being eliminated). Log: `[auto] edge coverage: C covered, B backstop, U Populate the `## Edge Coverage` section of SPEC.md from the resolved edges. +## Step 5.6: Prohibition-Completeness Probe (must-NOT) + +Run AFTER Step 5.5 (you probe the must-NOT axis of clear requirements, over the same +requirement list). Reference: @~/.claude/gsd-core/references/prohibition-probe.md — the +portable two-stage protocol, the canon-referral rule, and the status×verification schema +live there (size-cap discipline; keep this step lean). + +**D1 — no compiled engine (ADR-550 D7b).** Unlike Step 5.5, the prohibition probe has NO +compiled recall engine and runs NO `node` invocation here. The recall stage is an LLM prose +pass: the closed eight-category edge taxonomy a classifier can apply does not exist for the +open values/safety/ethics must-NOT axis. Do NOT copy the Step 5.5 engine-resolution block. +Only the schema/projection layer is real code; the recall is prose. + +For each Requirement gathered so far, run the two-stage recall→precision pass: + +1. **Stage 1 — Recall (adversarial prose probe).** Ask the single adversarial question of the + requirement: *"What could this feature silently become that the author would NOT want, but + the spec does not forbid?"* Over-produce (~10 raw must-NOT candidates) — recall first. +2. **Stage 2 — Precision (one-pass classifier).** Filter the raw list in a single pass: + **DROP routine-engineering** items (normal correctness/hygiene — "must not mutate input", + "must not throw on empty" — owned by the edge probe or code review); **KEEP + values / safety / ethics** items (manipulative framing, protected-attribute proxies, raw + PII in plaintext). This collapses ~10 → ~2–3 genuine prohibitions. +3. **Canon-referral (ADR-550 D6, PROB-13).** A kept candidate that is canon security/compliance + (OWASP / prototype-pollution / path-traversal / injection / GDPR / generic fairness) is + NOT minted here — emit a one-line breadcrumb (*"prototype-pollution is canon — owned by + /gsd:secure-phase + eslint; not minted here"*) and DROP it. Minting canon items duplicates + /gsd:secure-phase and drowns the bespoke signal. +4. **Resolve each surfaced (non-canon) prohibition** (AskUserQuestion; text mode → numbered list): + - **Keep it** → write a NEGATIVE acceptance criterion (a must-NOT line) into Acceptance + Criteria AND mark the prohibition `resolved` with a verification tier: `test` (a + mechanical negative test/lint/assertion exists) or `judgment` (real but not mechanically + checkable — routes to judgment review). + - **Dismiss (reason)** → mark `dismissed` with a REQUIRED non-empty reason (PROB-05). The + reason string is the audit trail; silence is not a valid dismissal. + - **Defer** → leave `unresolved`. + +**Soft gate (after resolving) — PROB-06:** +- All applicable prohibitions resolved → proceed to Step 6. +- Any `unresolved` → AskUserQuestion: + - header: "Prohibitions" + - question: "[N] prohibition(s) are unresolved: [list]. What do you want to do?" + - options: "Resolve now" (loop back) / "Write SPEC.md anyway — flag unresolved" / + "Keep probing" + - On "anyway": write SPEC.md with those rows marked `⚠ Prohibition unresolved — planner + must treat as assumption`. This is a soft gate (write-anyway-with-flags), never a silent + skip — the soft gate IS the control. + +**`--auto` mode:** auto-`resolved` where a defensible negative acceptance criterion can be +written (test or judgment tier); otherwise leave `unresolved`. **`--auto` NEVER auto-dismisses +a prohibition** — a wrong dismissal is the exact silent failure this probe eliminates (PROB-06, +the load-bearing safety property). Log: `[auto] prohibitions: R resolved, U unresolved`. + +**Text mode (PROB-09):** per Step 5's text-mode rule, replace the AskUserQuestion menus above +with plain-text numbered lists — there is NO hard AskUserQuestion dependency, so the probe +runs identically for non-Claude / text-mode hosts. + +Populate the `## Prohibitions` section of SPEC.md from the resolved prohibitions (each +`resolved`/`test` row is a checkable negative acceptance criterion; `resolved`/`judgment` +rows route to judgment review; `⚠ UNRESOLVED` rows are flagged as assumptions). + ## Step 6: Generate SPEC.md Use the SPEC.md template from @~/.claude/gsd-core/templates/spec.md. - Populate the **Edge Coverage** section from Step 5.5 (covered/dismissed/backstop/unresolved rows). +- Populate the **Prohibitions** section from Step 5.6 (resolved/dismissed/unresolved rows with the test|judgment tier). **Requirements for every requirement entry:** - One specific, testable statement @@ -377,6 +439,7 @@ Next: /gsd:discuss-phase {X} - Scout the codebase BEFORE the first question — grounded questions only - Max 2–3 questions per round — do not frontload all questions at once - Step 5.5 edge probe runs after the ambiguity gate; dismissals require a reason; --auto never auto-dismisses +- Step 5.6 prohibition probe runs after the edge probe; dismissals require a reason; --auto never auto-dismisses a prohibition @@ -389,4 +452,5 @@ Next: /gsd:discuss-phase {X} - SPEC.md committed atomically (when commit_docs is true) - User directed to /gsd:discuss-phase as next step - Edge-completeness probe run; Edge Coverage section populated; unresolved edges flagged as assumptions +- Prohibition-completeness probe run; Prohibitions section populated; unresolved prohibitions flagged as assumptions diff --git a/gsd-core/workflows/verify-phase.md b/gsd-core/workflows/verify-phase.md index cca20765b..d725c67cd 100644 --- a/gsd-core/workflows/verify-phase.md +++ b/gsd-core/workflows/verify-phase.md @@ -63,10 +63,15 @@ for plan in "$PHASE_DIR"/*-PLAN.md; do done ``` -Returns JSON: `{ truths: [...], artifacts: [...], key_links: [...] }` +Returns JSON: `{ truths: [...], artifacts: [...], key_links: [...], prohibitions: [...] }` Aggregate all must_haves across plans for phase-level verification. +**Prohibitions (`must_haves.prohibitions`, ADR-550 D3 — the must-NOT sibling block):** When a plan carries `must_haves.prohibitions`, extract each `{ statement, status, verification }` item and route it by `verification` tier in verdict assembly (ADR-550 D4, "B-with-guard", 2026-06-12 maintainer decision). These are NEGATIVE checks (the must-NOT must NOT have happened), distinct from positive `truths`: + +- **judgment-tier → mode-dependent soft-gate.** Interactive verify defers each item to the end-of-phase human checkpoint (`human_verify_mode: end-of-phase`). Autonomous verify records a NON-AUTHORITATIVE LLM-judge verdict + a prominent `unverified-prohibition — human review recommended` flag (autonomous completion reads "complete with N flagged prohibitions"). NEVER a silent pass; NEVER a hard halt of an AFK run. +- **test-tier → FAIL CLOSED (accept-and-flag).** Accept the `verification: test` value (the SPEC↔must_haves.prohibitions projection contract holds — no forced schema change later), but a well-formed test-tier item reaching verify with NO wired enforcement disposes as UNVERIFIED, flagged like an unresolved judgment item, NEVER green. The deterministic fail-closed default is `dispositionForProhibition()` in probe-core (`status: 'unverified'`, `flagged: true` on empty `enforcementEvidence`). The real fail-first negative-test enforcement MECHANISM defers to a follow-up PR (#644's corpus is entirely judgment-tier; a contrived test-tier fixture here would be the gold-plating failure mode). + **Option B: Use Success Criteria from ROADMAP.md** If no must_haves in frontmatter (MUST_HAVES returns error or empty), check for Success Criteria: @@ -465,13 +470,18 @@ Classify status using this decision tree IN ORDER (most restrictive first): 1. IF any truth FAILED, artifact MISSING/STUB, key link NOT_WIRED, blocker found, **or test quality audit found blockers (disabled requirement tests, circular tests)**: → **gaps_found** -2. IF the previous step produced ANY human verification items: +2. IF any `must_haves.prohibitions` item disposes as flagged-unverified (ADR-550 D4): + - **test-tier, fail-closed** (no wired enforcement — `dispositionForProhibition()` returns `status: 'unverified'`, `flagged: true`): → **gaps_found** (never green; the unwired test-tier item is an unverified gap). + - **judgment-tier, autonomous run** (non-authoritative LLM-judge verdict): emit the `unverified-prohibition — human review recommended` flag and classify → **human_needed** (autonomous completion reads "complete with N flagged prohibitions"; never a silent pass, never a hard halt). + - **judgment-tier, interactive run**: route to the end-of-phase human checkpoint → **human_needed**. + +3. IF the previous step produced ANY human verification items: → **human_needed** (even if all truths VERIFIED and score is N/N) -3. IF all checks pass AND no human verification items: +4. IF all checks pass AND no human verification items AND no flagged prohibitions: → **passed** -**passed is ONLY valid when no human verification items exist.** +**passed is ONLY valid when no human verification items AND no flagged prohibitions exist.** A prohibition (must-NOT) can never be silently absorbed into a `passed` verdict — that is the core failure mode ADR-550 D4 forbids. **Score:** `verified_truths / total_truths` diff --git a/src/frontmatter.cts b/src/frontmatter.cts index b735e32d3..9dda57e93 100644 --- a/src/frontmatter.cts +++ b/src/frontmatter.cts @@ -196,14 +196,59 @@ function reconstructFrontmatter(obj: Frontmatter): string { } function spliceFrontmatter(content: string, newObj: Frontmatter): string { - const yamlStr = reconstructFrontmatter(newObj); const match = content.match(/^---\r?\n[\s\S]+?\r?\n---/); if (match) { + // Identity-preservation (additive, lossless round-trip): `reconstructFrontmatter` is a + // deliberately lossy serializer — it cannot faithfully re-emit nested object-list items + // (e.g. must_haves.artifacts / must_haves.prohibitions, whose items are `{ path, provides }` + // / `{ statement, status, … }` maps). When the caller is writing back a value that is + // STRUCTURALLY UNCHANGED from the original parse (the canonical CRUD round-trip and the + // #644 prohibition schema round-trip both do this), regenerating from the lossy object would + // silently mangle those blocks. Detect that case by deep-equality against a re-parse of the + // original frontmatter and preserve the ORIGINAL raw text verbatim — a true no-op splice. + // This touches neither the parser (`extractFrontmatter`) nor `parseMustHavesBlock`; it only + // makes the existing splice faithful when nothing changed. A genuine mutation (different + // object) still flows through `reconstructFrontmatter` exactly as before. + try { + if (frontmatterDeepEqual(extractFrontmatter(content), newObj)) { + return content; + } + } catch { + /* fall through to regeneration on any comparison hiccup */ + } + const yamlStr = reconstructFrontmatter(newObj); return `---\n${yamlStr}\n---` + content.slice(match[0].length); } + const yamlStr = reconstructFrontmatter(newObj); return `---\n${yamlStr}\n---\n\n` + content; } +/** + * Structural deep-equality for two parsed frontmatter objects. Order-sensitive for arrays + * (YAML lists are ordered), key-order-insensitive for objects. Used only by `spliceFrontmatter` + * to recognize a no-op write-back; intentionally narrow (handles the string / string[] / + * nested-object shapes `extractFrontmatter` produces). + */ +function frontmatterDeepEqual(a: unknown, b: unknown): boolean { + if (a === b) return true; + if (a == null || b == null) return a === b; + if (Array.isArray(a) || Array.isArray(b)) { + if (!Array.isArray(a) || !Array.isArray(b) || a.length !== b.length) return false; + return a.every((v, i) => frontmatterDeepEqual(v, b[i])); + } + if (typeof a === 'object' && typeof b === 'object') { + const ao: Record = a as Record; + const bo: Record = b as Record; + const ak = Object.keys(ao); + const bk = Object.keys(bo); + if (ak.length !== bk.length) return false; + return ak.every((k) => + Object.prototype.hasOwnProperty.call(bo, k) && frontmatterDeepEqual(ao[k], bo[k]), + ); + } + return false; +} + function parseMustHavesBlock(content: string, blockName: string): unknown[] { // Extract a specific block from must_haves in raw frontmatter YAML // Handles 3-level nesting: must_haves > artifacts/key_links > [{path, provides, ...}] @@ -393,6 +438,11 @@ function cmdFrontmatterValidate(cwd: string, filePath: string, schemaName: strin export = { extractFrontmatter, + // Additive alias (#644 prohibition-probe schema contract): the probe round-trip seam reads a + // frontmatter object via `parseFrontmatter` (the name the contract test pins). It is the SAME + // function as `extractFrontmatter` — a bare-object parse with no behavior change — exposed under + // the alias so the prohibition schema round-trip and any future caller can use the canonical name. + parseFrontmatter: extractFrontmatter, reconstructFrontmatter, spliceFrontmatter, parseMustHavesBlock, diff --git a/src/probe-core.cts b/src/probe-core.cts index 64d782b47..6e89f98df 100644 --- a/src/probe-core.cts +++ b/src/probe-core.cts @@ -262,6 +262,179 @@ export function analyzeCoverage( return { items: merged, coverage: { applicable, resolved, unresolved, byVerification } }; } +/* ------------------------------------------------------------------------- * + * Prohibition adapter surface (#644 — the SECOND probe-core adapter). + * + * Unlike the edge adapter, the prohibition probe has NO deterministic propose stage: recall + * is an LLM prose pass (ADR-550 Decision 7b), so there is intentionally no `proposeProhibitions` + * here. What IS deterministic — and therefore real code that belongs in core — is (1) the + * injected verification validators (`test | judgment`) and (2) `projectProhibitions`, the + * SPEC<->`must_haves.prohibitions:` projection the DEFECT.GENERATIVE-FIX parity assertion + * round-trips as a FUNCTION rather than a prompt (ADR-550 Decision 5c). + * ------------------------------------------------------------------------- */ + +/** The prohibition probe's verification tiers (the `verification` axis values for a resolved item). */ +export type ProhibitionVerification = 'test' | 'judgment'; + +/** + * A surfaced prohibition item. Structurally a probe-core `Item` specialized to the prohibition + * verification vocabulary, but the load-bearing payload field is `statement` (the must-NOT + * sentence) rather than the edge adapter's `probe` question. Both fields are optional on the + * shared shape so a single `Item` type serves both adapters. + */ +export interface Prohibition { + requirement_id: string; + category: string; + status: Status; + verification: ProhibitionVerification | null; + resolution: string | null; + reason: string | null; + statement: string; +} + +/** + * The prohibition adapter's injected runtime validators (ADR-550 #5). There is no closed + * category taxonomy (recall is open-vocabulary values/safety/ethics prose), so `categories` + * is intentionally empty — `analyzeCoverage` is not the prohibition entry point and the + * round-trip schema layer does not gate on category. The verification tiers are + * `test | judgment` (ADR-550 D7a); both require only a present `resolution`/`reason` per their + * lifecycle (a resolved prohibition's checkable content is the `statement`, validated by the + * schema layer, not a `resolution` string), so `requiredFieldsByVerification` is the minimal + * fail-closed set: a dismissed item still needs its reason (enforced by `validateResolution`). + */ +export const PROHIBITION_VALIDATORS: Validators = { + categories: [], + verification: ['test', 'judgment'], + // A resolved prohibition's checkable content is the `statement` (schema-layer validated), NOT a + // `resolution` string — the canonical fixtures and the reference doc's worked examples all carry + // `resolution: null`. So the per-tier required set is empty: `resolved` still requires a present + // verification tier (enforced in validateResolution) and `dismissed` still requires a reason + // (enforced unconditionally), but neither tier requires a `resolution`. This matches the corpus + // the docs-fixtures parity test pins; the validators.test.cjs regression keeps them aligned. + requiredFieldsByVerification: { test: [], judgment: [] }, +}; + +/** Validate a prohibition resolution against the prohibition verification vocabulary. */ +export function validateProhibitionResolution(resolution: Resolution): true { + return validateResolution(resolution, PROHIBITION_VALIDATORS); +} + +/** + * Deterministically project resolved prohibition items into the `must_haves.prohibitions:` + * list shape (the SPEC<->plan projection; ADR-550 Decision 5c). This is a FUNCTION the parity + * assertion round-trips, never a prompt: the same input always yields the same output, and the + * output is the exact re-readable block shape `parseMustHavesBlock(content, 'prohibitions')` + * returns — `{ statement, status, verification }` plus `reason` only when present (a dismissed + * item's audit trail). `resolution`/`requirement_id`/`category` are recall-stage bookkeeping + * and are intentionally NOT projected into the plan block (which is keyed on the must-NOT + * statement, not the source requirement). A non-array input projects to `[]` (fail-soft on the + * empty/zero-prohibition case), never a throw. + */ +export function projectProhibitions( + items: unknown, +): Array> { + if (!Array.isArray(items)) return []; + const out: Array> = []; + for (const item of items) { + if (item == null || typeof item !== 'object') continue; + const p = item as Partial; + const statement = typeof p.statement === 'string' ? p.statement : ''; + const entry: Record = { + statement, + status: typeof p.status === 'string' ? p.status : 'unresolved', + }; + if (p.verification != null) entry.verification = String(p.verification); + if (p.reason != null && String(p.reason).trim()) entry.reason = String(p.reason); + out.push(entry); + } + return out; +} + +/** + * The structured verify-time disposition of a single prohibition (ADR-550 Decision 5d, the + * "B-with-guard" safety half — maintainer decision 2026-06-12). `status` is the verdict the + * verifier reads; `flagged` marks an item that must surface in SUMMARY/verdict rather than pass + * silently. `tier` echoes the verification axis so the caller can route. `reason` is human-readable. + */ +export interface ProhibitionDisposition { + status: 'green' | 'unverified'; + flagged: boolean; + tier: ProhibitionVerification | null; + reason: string; +} + +/** Optional enforcement context handed to `dispositionForProhibition`. */ +export interface ProhibitionDispositionContext { + /** Evidence that a resolved prohibition is actually enforced (e.g. a wired negative test). */ + enforcementEvidence?: unknown[]; +} + +/** + * Deterministic verify-time disposition for a single prohibition — the FAIL-CLOSED default + * (ADR-550 Decision 5d, the safety half of the 2026-06-12 "B-with-guard" maintainer decision). + * + * This is the cheap safety guarantee: a well-formed prohibition that reaches verify-phase with NO + * wired enforcement evidence can NEVER be a silent pass. It is `{ status: 'unverified', flagged: + * true }` — never `green` — exactly like an unresolved judgment item. The HEAVY half (a real + * fail-first negative-test enforcement mechanism that, given evidence, would flip a test-tier item + * to green) is OUT of #644 scope and defers to a follow-up PR: #644's corpus is entirely + * judgment-tier, so wiring a contrived test-tier consumer here would be the delete-bad-tests / + * gold-plating failure mode. Until that follow-up lands, ANY prohibition without enforcement + * evidence — test- or judgment-tier — disposes as flagged-unverified. + * + * The function is pure: same input always yields the same disposition (no LLM judgment, ADR-550 + * D5). The LLM-judge soft-gate for judgment-tier items is a verify-phase PROSE concern (the + * verifier records a non-authoritative verdict + the unverified-prohibition flag); this helper + * only owns the deterministic fail-closed default that the plan-01-01 CI safety assertion pins. + */ +export function dispositionForProhibition( + prohibition: unknown, + context: ProhibitionDispositionContext = {}, +): ProhibitionDisposition { + const p = (prohibition ?? {}) as Partial; + const tier: ProhibitionVerification | null = + p.verification === 'test' || p.verification === 'judgment' ? p.verification : null; + const evidence = Array.isArray(context.enforcementEvidence) ? context.enforcementEvidence : []; + const hasEnforcement = evidence.length > 0; + + // FAIL CLOSED: no wired enforcement evidence -> flagged unverified, never green. This holds for + // every tier today (the real enforcement mechanism that could flip a test-tier item to green is + // deferred to a follow-up PR). The guard the safety assertion proves: an unwired item can never + // be silently skipped. + if (!hasEnforcement) { + return { + status: 'unverified', + flagged: true, + tier, + reason: + tier === 'test' + ? 'test-tier prohibition has no wired enforcement evidence — flagged unverified (fail-closed; real negative-test enforcement deferred to a follow-up PR, ADR-550 D5d)' + : 'prohibition has no enforcement evidence — flagged unverified (fail-closed; never a silent pass, ADR-550 D5d)', + }; + } + + // D4 GUARD: a judgment-tier (or unknown-tier) prohibition is NEVER a silent green from this + // deterministic helper — it always routes to human/LLM judgment review (ADR-550 D4; verify-phase.md). + // Only a test-tier item with wired enforcement evidence may go green, and even that is the deferred + // heavy half until the real negative-test enforcement mechanism lands (no #644 caller passes evidence). + if (tier === 'test') { + return { + status: 'green', + flagged: false, + tier, + reason: 'test-tier prohibition has wired enforcement evidence', + }; + } + + return { + status: 'unverified', + flagged: true, + tier, + reason: + 'judgment-tier prohibition routes to judgment review — never a silent green (ADR-550 D4)', + }; +} + /* * CLI scaffold (the EP-06 invokable surface, generalized). Each probe ships one bin that * calls `runProbeCli` with its own `analyze` (closing over the adapter's propose + validators) diff --git a/tests/agent-size-baseline.json b/tests/agent-size-baseline.json index aaaa6f70a..02a2086d3 100644 --- a/tests/agent-size-baseline.json +++ b/tests/agent-size-baseline.json @@ -32,5 +32,5 @@ "gsd-ui-checker.md": 11081, "gsd-ui-researcher.md": 19265, "gsd-user-profiler.md": 8516, - "gsd-verifier.md": 41285 + "gsd-verifier.md": 43339 } diff --git a/tests/frontmatter.property.test.cjs b/tests/frontmatter.property.test.cjs index 79146608a..11a16be5a 100644 --- a/tests/frontmatter.property.test.cjs +++ b/tests/frontmatter.property.test.cjs @@ -13,6 +13,9 @@ * preserves key-value pairs for simple flat string values * (d) spliceFrontmatter never throws on any string/object combination * (e) extractFrontmatter returns {} for content without a leading ---...--- block + * (f) prohibitions bijection (#644): over a generated must_haves.prohibitions block, + * parseMustHavesBlock(spliceFrontmatter(doc, parseFrontmatter(doc)), 'prohibitions') + * deepEquals the original parse — the new parse ↔ splice path is identity-preserving. */ const { describe, test } = require('node:test'); @@ -23,6 +26,8 @@ const { extractFrontmatter, reconstructFrontmatter, spliceFrontmatter, + parseFrontmatter, + parseMustHavesBlock, } = require('../gsd-core/bin/lib/frontmatter.cjs'); // ─── Arbitraries ───────────────────────────────────────────────────────────── @@ -192,3 +197,69 @@ describe('frontmatter: spliceFrontmatter properties', () => { ); }); }); + +// ─── (f) prohibitions bijection (#644) ──────────────────────────────────────── +// Locks the new parseMustHavesBlock(…, 'prohibitions') ↔ spliceFrontmatter path that +// the prohibition probe adds. The example-based version lives in +// tests/prohibition-probe.schema.test.cjs; this generalizes it over generated blocks. + +// YAML-safe scalar: starts with a letter, no colon/quote/hash/newline (so it parses as a +// plain string and is never coerced to a number by the parser's /^\d+$/ check). +const safeScalar = fc.stringMatching(/^[A-Za-z][A-Za-z0-9 ._-]{0,50}$/); + +// One prohibition item with structurally realistic key shape per ADR-550 D7a: +// resolved → carries a verification tier (test|judgment) +// dismissed → carries a non-empty reason (+ a tier) +// unresolved→ neither +const prohibitionItem = fc.oneof( + fc.record({ statement: safeScalar, status: fc.constant('resolved'), + verification: fc.constantFrom('test', 'judgment') }), + fc.record({ statement: safeScalar, status: fc.constant('dismissed'), + verification: fc.constantFrom('test', 'judgment'), reason: safeScalar }), + fc.record({ statement: safeScalar, status: fc.constant('unresolved') }) +); + +// Emit a frontmatter doc with a must_haves.prohibitions sibling block (keys in a fixed +// order: statement, status, verification?, reason?). Quoted strings carry the values. +function buildDoc(items) { + const lines = ['---', 'phase: 01-x', 'plan: 01', 'must_haves:', + ' truths:', ' - "User sees a daily reminder"', ' prohibitions:']; + for (const it of items) { + lines.push(` - statement: "${it.statement}"`); + lines.push(` status: ${it.status}`); + if (it.verification !== undefined) lines.push(` verification: ${it.verification}`); + if (it.reason !== undefined) lines.push(` reason: "${it.reason}"`); + } + lines.push('---', '', 'Body text unchanged.', ''); + return lines.join('\n'); +} + +describe('frontmatter: prohibitions parse ↔ splice bijection (#644)', () => { + test('property: generated prohibitions parse back with their statement and status', () => { + fc.assert( + fc.property(fc.array(prohibitionItem, { minLength: 1, maxLength: 5 }), (items) => { + const doc = buildDoc(items); + const parsed = parseMustHavesBlock(doc, 'prohibitions'); + assert.equal(parsed.length, items.length, 'every prohibition item must parse out'); + for (let i = 0; i < items.length; i++) { + assert.equal(parsed[i].statement, items[i].statement, `statement[${i}] mismatch`); + assert.equal(parsed[i].status, items[i].status, `status[${i}] mismatch`); + } + }) + ); + }); + + test('property: parse -> splice -> re-parse is identity-preserving for prohibitions', () => { + fc.assert( + fc.property(fc.array(prohibitionItem, { minLength: 1, maxLength: 5 }), (items) => { + const doc = buildDoc(items); + const before = parseMustHavesBlock(doc, 'prohibitions'); + const parsed = parseFrontmatter(doc); + const spliced = spliceFrontmatter(doc, parsed.frontmatter ?? parsed); + const after = parseMustHavesBlock(spliced, 'prohibitions'); + assert.deepEqual(after, before, + 'prohibitions must survive a splice/re-parse round-trip unchanged'); + }) + ); + }); +}); diff --git a/tests/prohibition-probe.docs-fixtures.test.cjs b/tests/prohibition-probe.docs-fixtures.test.cjs new file mode 100644 index 000000000..43ffdbc76 --- /dev/null +++ b/tests/prohibition-probe.docs-fixtures.test.cjs @@ -0,0 +1,63 @@ +// allow-test-rule: runtime-contract-is-the-product (see #644) — the rendered reference doc's worked-example +// vocab surface is the runtime contract; this pins its bijection to the source-of-truth fixtures (docs-parity). +// +// RED-first parity contract: the portable reference doc (gsd-core/references/prohibition-probe.md) keeps +// its worked-example JSON blocks in sync with the source-of-truth fixture files under +// gsd-core/references/prohibition-probe-fixtures/. The fixtures are the canonical data; the doc embeds +// copies. The comparison is PARSED JSON (deepEqual of JSON.parse on both sides), never a raw-text +// substring — a reformat that preserves the data does not fail, and any semantic drift does. +// +// The reference doc does not exist yet (plan 01-02 creates it and embeds the +// ```json prohibition-probe:/``` blocks) — so this is EXPECTED RED now. NOTE for plan 01-02: +// it MUST embed matching tagged json blocks or this parity test stays red (the doc<->fixture contract). +'use strict'; +process.env.GSD_TEST_MODE = '1'; + +const { test, describe } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const docPath = path.join(__dirname, '..', 'gsd-core', 'references', 'prohibition-probe.md'); +const fixturesRoot = path.join(__dirname, '..', 'gsd-core', 'references', 'prohibition-probe-fixtures'); + +// Extract fenced blocks tagged ```json prohibition-probe:/ from the doc, keyed by ref. +// The \n? before the closing fence allows blocks whose closing fence has no preceding newline. +function taggedJsonBlocks(md) { + const re = /```json prohibition-probe:([^\n]+)\n([\s\S]*?)\n?```/g; + const out = {}; + let m; + while ((m = re.exec(md))) out[m[1].trim()] = m[2]; + return out; +} + +describe('prohibition-probe doc/fixture sync', () => { + test('reference doc exists', () => { + assert.ok(fs.existsSync(docPath), `${docPath} must exist`); + }); + + test('doc embeds a tagged fixture block for every expected.json fixture (count-equality)', () => { + const md = fs.readFileSync(docPath, 'utf8'); + const blocks = taggedJsonBlocks(md); + const expectedCount = fs.readdirSync(fixturesRoot, { withFileTypes: true }) + .filter(e => e.isDirectory()) + .filter(dir => fs.existsSync(path.join(fixturesRoot, dir.name, 'expected.json'))) + .length; + assert.strictEqual( + Object.keys(blocks).length, + expectedCount, + `prohibition-probe.md must embed exactly ${expectedCount} tagged blocks (one per fixture expected.json)` + ); + }); + + test('every tagged doc block parses and deepEquals its fixture file', () => { + const md = fs.readFileSync(docPath, 'utf8'); + const blocks = taggedJsonBlocks(md); + for (const [ref, body] of Object.entries(blocks)) { + const fixtureFile = path.join(fixturesRoot, ref); + const onDisk = fs.readFileSync(fixtureFile, 'utf8'); + assert.deepEqual(JSON.parse(body), JSON.parse(onDisk), + `doc block prohibition-probe:${ref} must deepEqual ${fixtureFile}`); + } + }); +}); diff --git a/tests/prohibition-probe.planner-contract.test.cjs b/tests/prohibition-probe.planner-contract.test.cjs new file mode 100644 index 000000000..0daeb73cf --- /dev/null +++ b/tests/prohibition-probe.planner-contract.test.cjs @@ -0,0 +1,96 @@ +// allow-test-rule: runtime-contract-is-the-product (see #644) — plan-phase.md's planner prompt is the deployed +// runtime contract under assertion (the workflow PROSE is the product). +// +// RED-first PROSE-PRESENCE contract for the plan-phase lift of confirmed prohibitions. plan-phase.md +// spawns the planner from its own inline block; the load-bearing lift instruction +// must live THERE (not in templates/planner-subagent-prompt.md, which nothing loads at runtime — the +// edge-probe orphan-prompt regression, edge planner test 145-153). Assertions scope to extracted +// sub-blocks to avoid false positives. +// +// ADR-550 Decision 2: resolved prohibitions lift into must_haves.PROHIBITIONS — NOT must_haves.truths. +// EXPECTED RED until Wave 3 adds the plan-phase lift + quality_gate item. +'use strict'; +process.env.GSD_TEST_MODE = '1'; + +const { test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const PLAN_PHASE_PATH = path.join(__dirname, '..', 'gsd-core', 'workflows', 'plan-phase.md'); +const PROMPT_PATH = path.join(__dirname, '..', 'gsd-core', 'templates', 'planner-subagent-prompt.md'); + +function readPlanPhase() { + return fs.readFileSync(PLAN_PHASE_PATH, 'utf8'); +} + +// Extract the planner block from plan-phase.md (the runtime planner surface). +function extractDownstreamConsumerBlock(content) { + const start = content.indexOf(''); + if (start === -1) return ''; + const end = content.indexOf('', start); + if (end === -1) return ''; + return content.slice(start, end + ''.length); +} + +// Extract the planner block (the last one in plan-phase.md, inside the planner prompt). +function extractQualityGateBlock(content) { + const start = content.lastIndexOf(''); + if (start === -1) return ''; + const end = content.indexOf('', start); + if (end === -1) return ''; + return content.slice(start, end + ''.length); +} + +// PROB-04 consumer: the lift instruction lives in the RUNTIME planner surface and lifts resolved +// prohibitions into must_haves.prohibitions (ADR-550 D2 — NOT truths). +test('PROB-04 consumer: downstream_consumer lifts SPEC Prohibitions into must_haves.prohibitions', () => { + const block = extractDownstreamConsumerBlock(readPlanPhase()); + assert.ok(block.length > 0, 'sanity: plan-phase.md must contain a block to scope this test'); + + assert.match( + block, + /Prohibitions/, + 'plan-phase.md must reference the SPEC Prohibitions section' + ); + assert.match( + block, + /must_haves\.prohibitions/, + 'plan-phase.md must lift resolved prohibitions into must_haves.prohibitions (ADR-550 D2 — NOT truths)' + ); + + // ADR-550 D2 GUARD: prohibitions must NOT be lifted into truths. + assert.doesNotMatch( + block, + /prohibition[^.\n]*must_haves\.truths|must_haves\.truths[^.\n]*prohibition/i, + 'prohibitions must NOT be lifted into must_haves.truths — truths keeps its positive-observable semantics (ADR-550 D2)' + ); +}); + +// Orphan-prompt regression guard (mirror edge planner test 145-153): the lift instruction must NOT +// be relocated into templates/planner-subagent-prompt.md (which nothing loads at runtime). +test('PROB-04 regression: the prohibitions lift does not live in the orphaned planner-subagent-prompt.md', () => { + const prompt = fs.readFileSync(PROMPT_PATH, 'utf8'); + assert.doesNotMatch( + prompt, + /must_haves\.prohibitions/, + 'planner-subagent-prompt.md is not loaded at runtime — the prohibitions lift instruction must not live there' + ); +}); + +// PROB-04: quality_gate covers every resolved SPEC Prohibition being represented in must_haves.prohibitions. +test('PROB-04: planner quality_gate requires every resolved SPEC Prohibition represented in must_haves.prohibitions', () => { + const qgBlock = extractQualityGateBlock(readPlanPhase()); + assert.ok(qgBlock.length > 0, 'sanity: plan-phase.md must contain a block to scope this test'); + + assert.match( + qgBlock, + /prohibition/i, + 'planner quality_gate must contain a checklist item covering SPEC prohibitions (PROB-04)' + ); + assert.match( + qgBlock, + /must_haves\.prohibitions/, + 'planner quality_gate must tie SPEC prohibitions to must_haves.prohibitions with no silent drops (PROB-04)' + ); +}); diff --git a/tests/prohibition-probe.schema.test.cjs b/tests/prohibition-probe.schema.test.cjs new file mode 100644 index 000000000..29d2539c4 --- /dev/null +++ b/tests/prohibition-probe.schema.test.cjs @@ -0,0 +1,181 @@ +// allow-test-rule: runtime-contract-is-the-product (see #644) — the must_haves.prohibitions: block is the +// runtime plan-contract surface; this pins its parse/round-trip/projection bijection to the code. +// +// RED-first schema contract for the `must_haves.prohibitions:` SIBLING block (ADR-550 Decision 3 — +// NOT a `polarity` field on `truths`; Decision 2 leaves `truths` untouched). Mirrors the round-trip +// discipline of tests/probe-core.test.cjs and the frontmatter callers. The parser under assertion is +// the block-name-generic parseMustHavesBlock @ src/frontmatter.cts:207 (built to gsd-core/bin/lib/ +// frontmatter.cjs by `npm run build:lib`) and spliceFrontmatter @ src/frontmatter.cts:198. +// +// EXPECTED RED until plan 01-02 builds the schema callers + projectProhibitions and plan 01-04 adds +// the test-tier fail-closed disposition. No `polarity` key appears anywhere. No LLM judgment is +// asserted (ADR-550 Decision 5) — only parse / round-trip / projection determinism. +'use strict'; +process.env.GSD_TEST_MODE = '1'; + +const { test, describe } = require('node:test'); +const assert = require('node:assert/strict'); +const path = require('node:path'); + +const FRONTMATTER_LIB = path.join(__dirname, '..', 'gsd-core', 'bin', 'lib', 'frontmatter.cjs'); +const PROBE_CORE_LIB = path.join(__dirname, '..', 'gsd-core', 'bin', 'lib', 'probe-core.cjs'); + +// A must_haves block carrying a prohibitions: sibling list. ADR-550 D7a axes: +// status ∈ {resolved, dismissed, unresolved}; verification ∈ {test, judgment} (NOT the retired +// covered/backstop enum). Dismissed items carry a non-empty reason. +const CONTENT_WITH_PROHIBITIONS = `--- +phase: 01-x +plan: 01 +must_haves: + truths: + - "User sees a daily reminder" + artifacts: + - path: "src/reminders.ts" + provides: "scheduleReminders" + prohibitions: + - statement: "MUST NOT use shaming/guilt/negative-streak framing" + status: resolved + verification: judgment + - statement: "MUST NOT store raw SSN in the audit log" + status: dismissed + verification: test + reason: "Out of scope for this phase; tracked in PRIV-02" + key_links: + - from: "src/reminders.ts" + to: "src/notify.ts" + via: "import" +--- + +Body text unchanged. +`; + +// A must_haves block with NO prohibitions: sibling — the backward-compat case. +const CONTENT_NO_PROHIBITIONS = `--- +phase: 01-x +plan: 01 +must_haves: + truths: + - "User sees a daily reminder" + artifacts: + - path: "src/reminders.ts" + provides: "scheduleReminders" + key_links: + - from: "src/reminders.ts" + to: "src/notify.ts" + via: "import" +--- + +Body text unchanged. +`; + +describe('prohibition-probe schema: must_haves.prohibitions round-trip (PROB-07)', () => { + const fm = require(FRONTMATTER_LIB); + + test('a prohibitions: list survives parse -> splice -> re-parse unchanged', () => { + assert.equal(typeof fm.parseMustHavesBlock, 'function', 'parseMustHavesBlock must be exported from the built lib'); + assert.equal(typeof fm.spliceFrontmatter, 'function', 'spliceFrontmatter must be exported from the built lib'); + assert.equal(typeof fm.parseFrontmatter, 'function', 'parseFrontmatter must be exported from the built lib'); + + const prohibitions = fm.parseMustHavesBlock(CONTENT_WITH_PROHIBITIONS, 'prohibitions'); + assert.equal(prohibitions.length, 2, 'two prohibition items must parse out of the must_haves block'); + + const resolved = prohibitions[0]; + assert.equal(resolved.statement, 'MUST NOT use shaming/guilt/negative-streak framing'); + assert.equal(resolved.status, 'resolved'); + assert.equal(resolved.verification, 'judgment'); + + const dismissed = prohibitions[1]; + assert.equal(dismissed.status, 'dismissed'); + assert.equal(dismissed.verification, 'test'); + assert.ok(typeof dismissed.reason === 'string' && dismissed.reason.trim().length > 0, + 'a dismissed prohibition must carry a non-empty reason (ADR-550 Decision 2/3)'); + + // Round-trip: parse full frontmatter, splice it back, re-parse — prohibitions are stable. + const parsed = fm.parseFrontmatter(CONTENT_WITH_PROHIBITIONS); + const spliced = fm.spliceFrontmatter(CONTENT_WITH_PROHIBITIONS, parsed.frontmatter ?? parsed); + const reparsed = fm.parseMustHavesBlock(spliced, 'prohibitions'); + assert.deepEqual(reparsed, prohibitions, 'prohibitions must survive a splice/re-parse round-trip unchanged'); + }); + + test('no polarity key is present on any prohibition item (ADR-550 Decision 2/3)', () => { + const prohibitions = fm.parseMustHavesBlock(CONTENT_WITH_PROHIBITIONS, 'prohibitions'); + for (const item of prohibitions) { + assert.ok(!Object.prototype.hasOwnProperty.call(item, 'polarity'), + 'prohibition items must NOT carry a polarity key — the prohibitions: sibling block replaces it'); + } + }); +}); + +describe('prohibition-probe schema: backward-compat byte-stability (PROB-08)', () => { + const fm = require(FRONTMATTER_LIB); + + test('a must_haves with no prohibitions: is byte-unchanged through the round-trip', () => { + const parsed = fm.parseFrontmatter(CONTENT_NO_PROHIBITIONS); + const spliced = fm.spliceFrontmatter(CONTENT_NO_PROHIBITIONS, parsed.frontmatter ?? parsed); + assert.equal(spliced, CONTENT_NO_PROHIBITIONS, + 'a prohibitions-less must_haves must round-trip byte-for-byte (backward compatibility)'); + }); + + test('parseMustHavesBlock(content, "prohibitions") returns [] when absent', () => { + const prohibitions = fm.parseMustHavesBlock(CONTENT_NO_PROHIBITIONS, 'prohibitions'); + assert.deepEqual(prohibitions, [], 'absent prohibitions: block must parse to an empty list, not throw'); + }); +}); + +describe('prohibition-probe schema: deterministic projectProhibitions round-trip (PROB-14)', () => { + // ADR-550 Decision 5(c): the DEFECT.GENERATIVE-FIX parity assertion across template <-> parser <-> + // planner is grounded on a deterministic projectProhibitions() in probe-core rather than a prompt. + // The function does not exist yet (plan 01-02 adds it) — assert its expected signature so this is RED. + test('probe-core exports a deterministic projectProhibitions(items) function', () => { + const pc = require(PROBE_CORE_LIB); + assert.equal(typeof pc.projectProhibitions, 'function', + 'probe-core must export projectProhibitions() — the deterministic SPEC<->must_haves projection (ADR-550 D5c)'); + + const items = [ + { requirement_id: 'R1', category: 'values', status: 'resolved', verification: 'judgment', resolution: null, reason: null, statement: 'MUST NOT shame the user' }, + ]; + const once = pc.projectProhibitions(items); + const twice = pc.projectProhibitions(items); + assert.deepEqual(once, twice, 'projectProhibitions must be deterministic (same input -> identical output)'); + assert.ok(Array.isArray(once), 'projectProhibitions must return an array of prohibition entries'); + }); + + // Render projected entries into a must_haves.prohibitions: block exactly as the planner/template + // would, so the parser reads back what the projector wrote (keys: statement, status, optional + // verification, optional reason — the projectProhibitions output shape). + function renderProhibitionsDoc(entries) { + const lines = ['---', 'phase: 01-x', 'plan: 01', 'must_haves:', ' prohibitions:']; + for (const e of entries) { + lines.push(` - statement: "${e.statement}"`); + lines.push(` status: ${e.status}`); + if (e.verification !== undefined) lines.push(` verification: ${e.verification}`); + if (e.reason !== undefined) lines.push(` reason: "${e.reason}"`); + } + lines.push('---', '', 'Body.', ''); + return lines.join('\n'); + } + + // ADR-550 D5c (DEFECT.GENERATIVE-FIX): close the parity loop by round-tripping the PROJECTOR's + // output through the parser — proving the write-shape and read-shape cannot drift, not merely that + // each is independently correct. + test('projectProhibitions output round-trips through parseMustHavesBlock (PROB-14 parity)', () => { + const pc = require(PROBE_CORE_LIB); + const fm = require(FRONTMATTER_LIB); + const items = [ + { requirement_id: 'R1', category: 'values', status: 'resolved', verification: 'judgment', resolution: null, reason: null, statement: 'MUST NOT shame the user' }, + { requirement_id: 'R1', category: 'privacy', status: 'dismissed', verification: 'test', resolution: null, reason: 'out of scope this phase', statement: 'MUST NOT store raw SSN' }, + { requirement_id: 'R2', category: 'safety', status: 'unresolved', verification: null, resolution: null, reason: null, statement: 'MUST NOT auto-execute fetched code' }, + ]; + const projected = pc.projectProhibitions(items); + const reparsed = fm.parseMustHavesBlock(renderProhibitionsDoc(projected), 'prohibitions'); + assert.deepEqual(reparsed, projected, + 'parseMustHavesBlock(serialize(projectProhibitions(items))) must equal projectProhibitions(items) — writer<->reader bijection'); + }); + + test('projectProhibitions returns [] for empty / null / undefined input (fail-soft boundary)', () => { + const pc = require(PROBE_CORE_LIB); + assert.deepEqual(pc.projectProhibitions([]), [], 'empty array -> []'); + assert.deepEqual(pc.projectProhibitions(null), [], 'null -> [] (documented fail-soft)'); + assert.deepEqual(pc.projectProhibitions(undefined), [], 'undefined -> [] (documented fail-soft)'); + }); +}); diff --git a/tests/prohibition-probe.spec-phase-contract.test.cjs b/tests/prohibition-probe.spec-phase-contract.test.cjs new file mode 100644 index 000000000..d95b546cd --- /dev/null +++ b/tests/prohibition-probe.spec-phase-contract.test.cjs @@ -0,0 +1,145 @@ +// allow-test-rule: runtime-contract-is-the-product (see #644) — spec-phase.md Step 5.6 is the deployed workflow +// runtime contract under assertion (the workflow PROSE is the product; ADR-550 D5 forbids a JS engine here). +// +// RED-first PROSE-PRESENCE contract for the prohibition probe's Step 5.6 (ADR-550 D1 DIVERGENCE, +// RESEARCH Pitfall 2 / PATTERNS D1): the prohibition probe is LLM-propose (ADR-550 D7b) — there is NO +// compiled engine. DO NOT assert a node-engine / prohibition-probe.cjs invocation; a prohibition-probe.cjs +// path here is a DEFECT. Assert the protocol PROSE is present in the Step 5.6 slice instead. Assertions +// scope to the extracted Step 5.6 block to avoid false positives from mentions elsewhere in the file. +// +// EXPECTED RED until Wave 2/3 add Step 5.6 to spec-phase.md. +'use strict'; +process.env.GSD_TEST_MODE = '1'; + +const { test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const SPEC_PHASE_PATH = path.join(__dirname, '..', 'gsd-core', 'workflows', 'spec-phase.md'); + +function readSpecPhase() { + return fs.readFileSync(SPEC_PHASE_PATH, 'utf8'); +} + +// Slice the Step 5.6 block: from the "Step 5.6" heading to the next "## " or "Step N" heading. +// This scopes assertions to Step 5.6 only, preventing false positives from mentions elsewhere. +function extractStep56Block(content) { + const startIdx = content.indexOf('## Step 5.6'); + if (startIdx === -1) { + const altIdx = content.indexOf('Step 5.6'); + if (altIdx === -1) return ''; + const rest = content.slice(altIdx + 'Step 5.6'.length); + const nextHeading = rest.search(/\n## |\nStep \d/); + if (nextHeading === -1) return content.slice(altIdx); + return content.slice(altIdx, altIdx + 'Step 5.6'.length + nextHeading); + } + const rest = content.slice(startIdx + '## Step 5.6'.length); + const nextHeading = rest.search(/\n## /); + if (nextHeading === -1) return content.slice(startIdx); + return content.slice(startIdx, startIdx + '## Step 5.6'.length + nextHeading); +} + +// D1 DIVERGENCE GUARD: there is no compiled prohibition engine. Assert the Step 5.6 block does NOT +// reference a prohibition-probe.cjs node invocation (RESEARCH Pitfall 2: a copied edge-probe.cjs +// wire is the defect — prohibition is LLM-propose, ADR-550 D7b). +test('D1: Step 5.6 does NOT invoke a prohibition-probe.cjs engine (LLM-propose, not compiled)', () => { + const block = extractStep56Block(readSpecPhase()); + assert.ok(block.length > 0, 'Step 5.6 block must be extractable from spec-phase.md'); + assert.doesNotMatch( + block, + /prohibition-probe\.cjs/, + 'Step 5.6 must NOT reference a prohibition-probe.cjs engine — the prohibition probe is LLM-propose (ADR-550 D7b); a node-engine wire here is a defect' + ); +}); + +// PROB-01: adversarial recall question is present (the model-robust recall stage). +test('PROB-01: Step 5.6 poses the adversarial recall question', () => { + const block = extractStep56Block(readSpecPhase()); + assert.ok(block.length > 0, 'Step 5.6 block must be extractable from spec-phase.md'); + assert.match( + block, + /silently become|would NOT want|must.?not|adversarial/i, + 'Step 5.6 must pose the adversarial recall question (what could this feature silently become that the author would NOT want)' + ); +}); + +// PROB-02: precision classifier drops routine-engineering items. +test('PROB-02: Step 5.6 precision stage drops routine engineering items', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /routine engineering|drop[a-z]*\b[^.]*engineering|precision[^.]*classif/i, + 'Step 5.6 must describe a precision classifier that drops routine engineering items (PROB-02)' + ); +}); + +// PROB-06: soft-gate write-anyway-with-flags. +test('PROB-06: Step 5.6 is a soft gate (write-anyway-with-flags)', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /write.?anyway|soft.?gate|with.?flags/i, + 'Step 5.6 must be a soft gate (write-anyway-with-flags), not a hard halt (PROB-06)' + ); +}); + +// PROB-05: dismissals require a non-empty reason. +test('PROB-05: Step 5.6 requires a non-empty reason to dismiss', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /dismiss[a-z]*\b[^.]*reason|reason[^.]*dismiss/i, + 'Step 5.6 must require a non-empty reason when dismissing a prohibition (PROB-05)' + ); +}); + +// PROB-06: --auto never auto-dismisses. +test('PROB-06: Step 5.6 --auto never auto-dismisses prohibitions', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /--auto[^.]*(never|not)[^.]*dismiss|never auto.?dismiss/i, + 'Step 5.6 must specify that --auto never auto-dismisses prohibitions (PROB-06)' + ); +}); + +// PROB-09: text-mode has no hard AskUserQuestion dependency. +test('PROB-09: Step 5.6 text-mode handling has no hard AskUserQuestion dependency', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /text.?mode|non-?Claude|AskUserQuestion/i, + 'Step 5.6 must handle text-mode (non-Claude, no hard AskUserQuestion) (PROB-09)' + ); +}); + +// SPEC population: confirmed prohibitions populate a SPEC Prohibitions section. +test('Step 5.6 populates a SPEC Prohibitions section', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /Prohibitions/, + 'Step 5.6 must populate a SPEC Prohibitions section with confirmed prohibitions' + ); +}); + +// PROB-13: ADR-550 D6 canon-referral breadcrumb to /gsd:secure-phase. +test('PROB-13: Step 5.6 emits the canon-referral breadcrumb to /gsd:secure-phase', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /secure-phase|canon[^.]*security|OWASP|GDPR/i, + 'Step 5.6 must emit a canon-referral breadcrumb (e.g. owned by /gsd:secure-phase) for canon-security items (ADR-550 D6, PROB-13)' + ); +}); + +// @-include reference to references/prohibition-probe.md. +test('Step 5.6 @-includes the references/prohibition-probe.md core', () => { + const block = extractStep56Block(readSpecPhase()); + assert.match( + block, + /references\/prohibition-probe\.md/, + 'Step 5.6 must @-include references/prohibition-probe.md (the portable probe core)' + ); +}); diff --git a/tests/prohibition-probe.validators.test.cjs b/tests/prohibition-probe.validators.test.cjs new file mode 100644 index 000000000..f64961753 --- /dev/null +++ b/tests/prohibition-probe.validators.test.cjs @@ -0,0 +1,82 @@ +// allow-test-rule: runtime-contract-is-the-product (see #644) — the prohibition validators and the verify-time +// disposition are the deployed safety contract; this pins them against the CANONICAL fixture corpus +// and the ADR-550 D4 "judgment is never silently green" invariant so the code can never drift from +// its own documented intent again. +// +// Regression coverage for the two defects the RED-first suite did not exercise (code review #644): +// CR-01 — PROHIBITION_VALIDATORS must ACCEPT every resolved prohibition in the canonical corpus +// (resolution: null; the checkable content is `statement`). The prior config required a +// non-empty `resolution` and threw on 100% of the fixtures. +// WR-01 — dispositionForProhibition must NEVER return a silent green for a judgment-tier item, +// regardless of enforcement evidence (ADR-550 D4 / verify-phase.md). +// +// Fixture-driven on purpose: the validators are exercised against the SAME expected.json files the +// docs-fixtures parity test pins, so the validator and the corpus can never silently diverge. +'use strict'; +process.env.GSD_TEST_MODE = '1'; + +const { test, describe } = require('node:test'); +const assert = require('node:assert/strict'); +const path = require('node:path'); +const fs = require('node:fs'); + +const PROBE_CORE_LIB = path.join(__dirname, '..', 'gsd-core', 'bin', 'lib', 'probe-core.cjs'); +const FIXTURES_DIR = path.join(__dirname, '..', 'gsd-core', 'references', 'prohibition-probe-fixtures'); + +function loadFixtureItems() { + const out = []; + for (const name of fs.readdirSync(FIXTURES_DIR).sort()) { + const expected = path.join(FIXTURES_DIR, name, 'expected.json'); + if (!fs.existsSync(expected)) continue; + const json = JSON.parse(fs.readFileSync(expected, 'utf8')); + assert.ok(json.items === undefined || Array.isArray(json.items), + `fixture ${name}: expected.json "items" must be an array when present (got ${typeof json.items})`); + for (const item of json.items ?? []) out.push({ fixture: name, item }); + } + return out; +} + +describe('prohibition-probe validators: canonical corpus is accepted (CR-01 regression)', () => { + test('validateProhibitionResolution accepts every resolved item in the reference fixtures', () => { + const pc = require(PROBE_CORE_LIB); + assert.equal(typeof pc.validateProhibitionResolution, 'function', + 'probe-core must export validateProhibitionResolution()'); + + const fixtureItems = loadFixtureItems(); + assert.ok(fixtureItems.length >= 4, + 'the corpus must carry the resolved prohibitions the validator is pinned against'); + + for (const { fixture, item } of fixtureItems) { + // A resolved prohibition's checkable content is `statement`, NOT `resolution` (resolution: null + // in every fixture). The validator must not throw on its own canonical corpus. + assert.doesNotThrow( + () => pc.validateProhibitionResolution(item), + `validateProhibitionResolution must accept fixture ${fixture} item ${item.requirement_id}::${item.category} (resolution: ${JSON.stringify(item.resolution)})`, + ); + } + }); +}); + +describe('prohibition-probe disposition: judgment is never silently green (WR-01 / ADR-550 D4)', () => { + const evidence = { enforcementEvidence: ['a wired negative test reference'] }; + + test('a judgment-tier item with enforcement evidence is NEVER a silent green', () => { + const pc = require(PROBE_CORE_LIB); + const judgment = { status: 'resolved', verification: 'judgment', statement: 'MUST NOT shame the user' }; + + const disposition = pc.dispositionForProhibition(judgment, evidence); + assert.notEqual(disposition.status, 'green', + 'a judgment-tier prohibition can never be green from the deterministic helper — it routes to judgment review (ADR-550 D4)'); + assert.equal(disposition.flagged, true, + 'a judgment-tier prohibition with evidence must still be flagged for judgment review'); + }); + + test('no-evidence items stay fail-closed for both tiers', () => { + const pc = require(PROBE_CORE_LIB); + for (const tier of ['test', 'judgment']) { + const d = pc.dispositionForProhibition({ status: 'resolved', verification: tier, statement: 'x' }, { enforcementEvidence: [] }); + assert.notEqual(d.status, 'green', `${tier}-tier with no evidence must be fail-closed (never green)`); + assert.equal(d.flagged, true, `${tier}-tier with no evidence must be flagged`); + } + }); +}); diff --git a/tests/prohibition-probe.verify-tier.test.cjs b/tests/prohibition-probe.verify-tier.test.cjs new file mode 100644 index 000000000..664a2db80 --- /dev/null +++ b/tests/prohibition-probe.verify-tier.test.cjs @@ -0,0 +1,57 @@ +// allow-test-rule: runtime-contract-is-the-product (see #644) — the verify-time disposition of a test-tier +// prohibition is the deployed safety contract; this pins its fail-closed default to the code. +// +// RED-first SAFETY HALF of ADR-550 Decision 5(d) [maintainer decision 2026-06-12 "B-with-guard"]. +// A WELL-FORMED test-tier prohibition (statement + status: resolved + verification: test) with NO +// wired enforcement evidence MUST yield a NON-GREEN / flagged-unverified disposition — proving an +// unwired test-tier item can NEVER be silently skipped (fail-closed default). +// +// This lives in its OWN test file (not the schema file) because it asserts a verify-disposition +// behavior — a different production module than frontmatter. The deterministic disposition helper +// does not exist yet (plan 01-04 implements the fail-closed default; the seam mirrors +// projectProhibitions in probe-core) — assert its expected contract so this is RED now. +// +// This is the cheap safety-guarantee half; the heavy "real negative-test enforcement mechanism" half +// is OUT of #644 scope (follow-up PR). No `polarity` key; no LLM judgment asserted (ADR-550 D5). +'use strict'; +process.env.GSD_TEST_MODE = '1'; + +const { test, describe } = require('node:test'); +const assert = require('node:assert/strict'); +const path = require('node:path'); + +const PROBE_CORE_LIB = path.join(__dirname, '..', 'gsd-core', 'bin', 'lib', 'probe-core.cjs'); + +describe('prohibition-probe verify-tier: test-tier fail-closed safety (PROB-12 / ADR-550 D5d)', () => { + test('probe-core exports a deterministic prohibition-disposition helper', () => { + const pc = require(PROBE_CORE_LIB); + assert.equal(typeof pc.dispositionForProhibition, 'function', + 'probe-core must export dispositionForProhibition() — the deterministic verify-time disposition (ADR-550 D5d)'); + }); + + test('a well-formed test-tier item with NO enforcement evidence is flagged non-green (fail-closed)', () => { + const pc = require(PROBE_CORE_LIB); + + // Synthetic, well-formed test-tier prohibition with NO wired enforcement evidence. + const unwiredTestTier = { + requirement_id: 'R1', + category: 'safety', + status: 'resolved', + verification: 'test', + resolution: null, + reason: null, + statement: 'MUST NOT store raw SSN in plaintext', + // deliberately: no enforcement evidence wired (no test reference / no proof) + }; + + const disposition = pc.dispositionForProhibition(unwiredTestTier, { enforcementEvidence: [] }); + + // The disposition must NOT be a silent pass. Accept any explicit non-green signal the helper + // chooses, but it must be unambiguously NOT 'green'/'pass' and must carry a flag. + assert.ok(disposition && typeof disposition === 'object', 'disposition must be a structured object'); + assert.notEqual(disposition.status, 'green', 'an unwired test-tier item must NEVER be green (fail-closed)'); + assert.notEqual(disposition.status, 'pass', 'an unwired test-tier item must NEVER pass silently (fail-closed)'); + assert.equal(disposition.flagged, true, + 'an unwired test-tier item must be flagged unverified — it can never be silently skipped'); + }); +}); diff --git a/tests/workflow-size-baseline.json b/tests/workflow-size-baseline.json index 71497ad99..e445abe80 100644 --- a/tests/workflow-size-baseline.json +++ b/tests/workflow-size-baseline.json @@ -51,7 +51,7 @@ "note.md": 6563, "pause-work.md": 14397, "plan-milestone-gaps.md": 11765, - "plan-phase.md": 92120, + "plan-phase.md": 92759, "plan-review-convergence.md": 23468, "plant-seed.md": 11741, "pr-branch.md": 9561, @@ -72,7 +72,7 @@ "ship.md": 24388, "sketch-wrap-up.md": 14223, "sketch.md": 19960, - "spec-phase.md": 23094, + "spec-phase.md": 27589, "spike-wrap-up.md": 15092, "spike.md": 24517, "stats.md": 6718, @@ -85,6 +85,6 @@ "undo.md": 10431, "update.md": 21053, "validate-phase.md": 10745, - "verify-phase.md": 28430, + "verify-phase.md": 30875, "verify-work.md": 31157 }