From 8de2ff9121c35a9d8729aedba1e4cbef8d7a96b6 Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Sun, 5 Jul 2026 14:04:29 -0400 Subject: [PATCH] feat(#2008): generic command-exit-zero gate-predicate evaluator (#2011) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(#2008): add generic command-exit-zero gate-predicate evaluator Third-party capability gates declared via check.predicate were rendered for display but never evaluated (only built-in check.query gates fired; the security capability's gate worked solely via a hard-coded ship.md branch). Add a generic, deps-injected gate-predicate evaluator (src/gate-predicate-evaluator.cts) that dispatches by predicate.kind. Built-in kind: command-exit-zero — runs a bounded sh -c command at the project root (via shell-command-projection.execTool), inherits env, exit 0 => pass, non-zero => block, timeout => block, fail-closed. Wire a 'check predicate' subcommand into check-command-router.cts and extend the three generic workflow gate-dispatch sites (execute:wave:post, execute:post, plan:post) to route check.predicate gates to the new evaluator. The two-step gate contract (command-failure => onError; block => halt) is unchanged. - src/gate-predicate-evaluator.cts: pure leaf, KIND_TABLE extensible - src/check-command-router.cts: cmdCheckPredicate + buildPredicateDeps + parsePredicateFlags - docs/adr/2008-*, docs/reference/gate-predicates.md, docs/how-to/command-exit-zero-gate.md - tests: 38 unit + integration tests (exit mapping, timeout, interpolation, property-based bijection, malformed-predicate fail-closed, real subprocess e2e) Closes #2008 * docs(#2008): backfill changeset pr number 2011 --- .changeset/2008-command-exit-zero-gate.md | 5 + .gitignore | 1 + CONTEXT.md | 3 + docs/INVENTORY-MANIFEST.json | 1 + docs/adr/2008-command-exit-zero-gate.md | 158 +++++++++ docs/adr/README.md | 1 + docs/how-to/command-exit-zero-gate.md | 146 ++++++++ docs/reference/gate-predicates.md | 117 ++++++ eslint.config.mjs | 1 + gsd-core/workflows/execute-phase.md | 7 +- gsd-core/workflows/plan-phase.md | 9 +- src/check-command-router.cts | 118 +++++- src/gate-predicate-evaluator.cts | 204 +++++++++++ tests/check-predicate.test.cjs | 106 ++++++ .../golden-install-parity/antigravity.json | 4 +- .../golden-install-parity/augment.json | 4 +- .../golden-install-parity/claude.json | 4 +- .../fixtures/golden-install-parity/cline.json | 4 +- .../golden-install-parity/codebuddy.json | 4 +- .../fixtures/golden-install-parity/codex.json | 4 +- .../golden-install-parity/copilot.json | 4 +- .../golden-install-parity/cursor.json | 4 +- .../golden-install-parity/hermes.json | 4 +- .../fixtures/golden-install-parity/kilo.json | 4 +- .../fixtures/golden-install-parity/kimi.json | 4 +- .../golden-install-parity/opencode.json | 4 +- .../fixtures/golden-install-parity/qwen.json | 4 +- .../fixtures/golden-install-parity/trae.json | 4 +- .../golden-install-parity/windsurf.json | 4 +- tests/gate-predicate-evaluator.test.cjs | 335 ++++++++++++++++++ tests/workflow-size-baseline.json | 4 +- 31 files changed, 1238 insertions(+), 38 deletions(-) create mode 100644 .changeset/2008-command-exit-zero-gate.md create mode 100644 docs/adr/2008-command-exit-zero-gate.md create mode 100644 docs/how-to/command-exit-zero-gate.md create mode 100644 docs/reference/gate-predicates.md create mode 100644 src/gate-predicate-evaluator.cts create mode 100644 tests/check-predicate.test.cjs create mode 100644 tests/gate-predicate-evaluator.test.cjs diff --git a/.changeset/2008-command-exit-zero-gate.md b/.changeset/2008-command-exit-zero-gate.md new file mode 100644 index 000000000..49151b5c2 --- /dev/null +++ b/.changeset/2008-command-exit-zero-gate.md @@ -0,0 +1,5 @@ +--- +type: Added +pr: 2011 +--- +**Third-party capability gates now actually fire via a generic `command-exit-zero` predicate.** — a capability's declared `check.predicate` gate was rendered for display but never evaluated (only built-in `check.query` gates were enforced, and the `security` capability's gate worked solely via a hard-coded `ship.md` branch). A new generic evaluator (`gsd_run check predicate`) now evaluates `check.predicate` blocks by `kind`; the first built-in kind `command-exit-zero` runs a bounded `sh -c` command at the project root and blocks the loop on non-zero exit (timeout → block, fail-closed). The `execute:wave:post`, `execute:post`, and `plan:post` gate-dispatch sites route `predicate` gates to the new evaluator automatically. (#2008) diff --git a/.gitignore b/.gitignore index a222f8625..9562ecd17 100644 --- a/.gitignore +++ b/.gitignore @@ -179,6 +179,7 @@ build/ /gsd-core/bin/lib/phase-command-router.cjs /gsd-core/bin/lib/surface.cjs /gsd-core/bin/lib/gap-checker.cjs +/gsd-core/bin/lib/gate-predicate-evaluator.cjs /gsd-core/bin/lib/docs.cjs /gsd-core/bin/lib/check-command-router.cjs /gsd-core/bin/lib/frontmatter.cjs diff --git a/CONTEXT.md b/CONTEXT.md index 04797d8da..31c2d72b1 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -184,6 +184,9 @@ A bundle delivering one optional GSD feature, toggled as a unit at install or af ### Loop Host Contract Generated description of what the five-step loop (Discuss → Plan → Execute → Verify → Ship) exposes as extension points: per-step loop points, agent roles, and core artifacts. Sourced from structured `` HTML-comment markers embedded near the top of each of the five step workflow files (`discuss-phase.md`, `plan-phase.md`, `execute-phase.md`, `verify-work.md`, `ship.md`). Generated by `scripts/gen-loop-host-contract.cjs` → `gsd-core/bin/lib/loop-host-contract.cjs` (ADR-894 §3 phase 3a-impl-2). Covers exactly the 12 canonical points (discuss:pre/post, plan:pre/post, execute:pre/wave:pre/wave:post/post, verify:pre/post, ship:pre/post). The generator enforces a drift guard: every declared non-orchestrator agent role must correspond to an actual agent reference in the workflow file. Consumed by `gen-capability-registry.cjs` (replaces the former inline `LOOP_HOST_CONTRACT` constant). Run `node scripts/gen-loop-host-contract.cjs --write` after editing a workflow step marker. +### Gate Predicate Evaluator Module +Pure, deps-injected evaluator for capability gate `check.predicate` blocks (#2008, ADR-2008). Prior to #2008 the registry validator accepted `check.predicate` (one of exactly-one-of `query`/`predicate`/`agentVerdict`) and the loop-resolver rendered it, but nothing EVALUATED a declared predicate — only `check.query` was enforced (dispatched via `gsd_run check `), and built-in gates like `security` worked only via hard-coded `capId` prose branches in `ship.md`/`execute-phase.md`/`verify-work.md`. This module is the generic evaluation path: `evaluatePredicate(predicate, context, deps) → { block, message, details? }` dispatches by `predicate.kind` through a `KIND_TABLE`. Built-in kind (v1): `command-exit-zero` — runs a declared command in a bounded `sh -c` subprocess (production binding: `shell-command-projection.execTool`) at the project root, inheriting env; exit 0 ⇒ pass, non-zero ⇒ block, timeout (SIGTERM) ⇒ block; interpolates `${PHASE_NUMBER}`/`${PHASE_DIR}`/`${PHASE_REQ_IDS}` from gate context. A THROWN error (malformed predicate, non-positive/non-finite timeout, command >4096 chars, unknown kind) maps at the CLI seam to a non-zero check-command exit, which the workflow's two-step gate contract treats as a step-1 command failure routed per `onError` — so an evaluator bug is never conflated with a legitimate block decision. Leaf pure module (no fs/child_process/config — subprocess seam injected). CLI entry: `gsd_run check predicate --predicate '' [--phase-dir …] [--phase-number …] [--phase-req-ids …] --raw`, wired into `check-command-router.cts:cmdCheckPredicate`; the three generic workflow gate-dispatch sites (`execute:wave:post`, `execute:post`, `plan:post`) branch on `check` shape (`query` vs `predicate`). Source of truth: `src/gate-predicate-evaluator.cts`. Docs: `docs/reference/gate-predicates.md`, `docs/how-to/command-exit-zero-gate.md`. + ### Capability Registry Generated central manifest projecting all co-located Capability declarations into one validated artifact for runtime resolution and for the install, surface, config, and loop-extension adapters. Mirrors the research-profiles / package-identity generation pattern (co-located source → generated central file). Generated by `scripts/gen-capability-registry.cjs` → `gsd-core/bin/lib/capability-registry.cjs` (ADR-894 §5 phase 3a-impl). Role-partitioned indexes: `bySkill`, `byAgent`, `byLoopPoint` (hook ordering materialized), `configKeys` (ownership map: key→capId), `configSchema` (full per-key schema: key→{ owner, type, default, description }), `runtimes`, `requiresClosure(id)`. Each feature capability's entry in `capabilities` now includes the optional `activationKey` field (the dotted config key that gates the whole capability, e.g. `"graphify.enabled"`; absent means no config gate). ADR-857 phase 3b adds `configSchema` with validated type/default/description per key, sourced from each capability's `.config` slice. ADR-857 phase 4a adds two derived views: `capabilityClusters` (`{ : [] }` — each cap's skills array, sorted, derived from the capability's `skills` declaration; consistency-gated against the hand-authored `CLUSTERS`) and `profileMembership` (`{ : { tier, profiles: [...] } }` — the tier-derived index: suffix of `PROFILE_RANK` starting at the capability's tier). Both views cover the same capability set: only capabilities that own skills (non-empty `skills` array). The generator enforces a HARD gate (throws) if a capId matching a `CLUSTERS` key has a mismatched skill set, and emits SOFT `⚠ pending-reconciliation` warnings to stderr (never to the file) for skills not yet in the hand-authored profile at the capability's tier. `install` and `surface` are UNTOUCHED (still read hand-authored constants; derived views are emitted and tested but unconsumed until cutover). Validated against the Loop Host Contract (12 points; generated by `gen-loop-host-contract.cjs` from workflow markers, phase 3a-impl-2). Run `node scripts/gen-capability-registry.cjs --write` after editing any `capabilities//capability.json`. diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index 1f84431ea..2c94b3bc9 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -337,6 +337,7 @@ "federated-config.cjs", "frontmatter.cjs", "gap-checker.cjs", + "gate-predicate-evaluator.cjs", "git-base-branch.cjs", "graphify-command-router.cjs", "graphify.cjs", diff --git a/docs/adr/2008-command-exit-zero-gate.md b/docs/adr/2008-command-exit-zero-gate.md new file mode 100644 index 000000000..fbdaccf53 --- /dev/null +++ b/docs/adr/2008-command-exit-zero-gate.md @@ -0,0 +1,158 @@ +# ADR-2008: Generic gate-predicate evaluator (`command-exit-zero`) + +| | | +|---|---| +| **Status** | Accepted | +| **Date** | 2026-07-04 | +| **Issue** | [#2008 — No generic evaluator for third-party capability gates](https://github.com/open-gsd/gsd-core/issues/2008) | +| **Supersedes** | — | +| **Amends** | ADR-0894 (capability declaration format — `check.predicate` evaluation path) | + +## Context + +Capability gates are declared in `capability.json` under `gates[].check`. The +registry validator (`capability-validator.cjs:validateGate`) recognises three +mutually-exclusive `check` shapes — `query`, `predicate`, and `agentVerdict` — +and the loop-resolver (`loop-resolver.cts:renderLoopHooks`) renders all active +gates to the workflow gate-dispatch sites. + +Prior to this ADR, only `check.query` was **enforced**: every workflow +gate-dispatch site ran `gsd_run check ${hook.check.query}`, which routes through +the fixed if-chain in `check-command-router.cts:routeCheckCommand`. A +`check.predicate` (e.g. the `security` capability's +`artifact-frontmatter-equals` declaration) was **declaration-only** — rendered +for display, never evaluated. The security capability's `threats_open == 0` +enforcement worked only because `ship.md` hard-codes a `capId == "security"` +prose branch that reads `SECURITY.md` frontmatter directly. There was no +extension path for a third-party capability's gate to actually fire. + +Issue #2008 asked for a generic, data-driven gate-evaluation path. Two candidate +directions were proposed: + +1. A generic `check.predicate` evaluator covering `artifact-frontmatter-equals` + and future kinds. +2. A documented, sandboxed **`command-exit-zero`** gate kind — "run ``, + block on non-zero exit" — matching the git-hook / CI-runner enforcement + shape. + +The maintainer chose **Option 2** (issue comment, 2026-07-04). + +## Decision + +Add a **generic gate-predicate evaluation path** with one built-in kind, +`command-exit-zero`, scoped as follows. + +### Declaration shape + +A capability declares a `command-exit-zero` gate under the existing +`check.predicate` block (already a first-class, validator-accepted shape): + +```json +"gates": [ + { + "point": "ship:pre", + "check": { + "predicate": { + "kind": "command-exit-zero", + "command": "node scripts/check.sh \"${PHASE_DIR}\"", + "timeout": 30 + } + }, + "when": "my_cap.enabled", + "blocking": true, + "onError": "halt" + } +] +``` + +`predicate.command` is required (non-empty string, ≤ 4096 chars). +`predicate.timeout` is optional (positive finite number, seconds; default 30). + +### Sandbox contract (the security core of this ADR) + +| Axis | Value | Rationale | +|---|---|---| +| Interpreter | `sh -c` (via `shell-command-projection.execTool`) | Cross-platform with the runtime's existing bash dependency; one string, no argv array to author | +| cwd | Project root (the runtime `cwd`) | Matches the user's working context; the same root existing `check.query` gates operate from | +| Environment | Inherit process env | The command runs as the user, on the user's machine, in the project they are working on — no sandbox boundary is crossed vs. the user's own shell. Override via the command itself (`env VAR=x ...`) | +| Timeout | Default 30s; overridable per-gate | Bounded execution is non-negotiable; an unbounded gate could hang the loop forever | +| Interpolation | `${PHASE_NUMBER}`, `${PHASE_DIR}`, `${PHASE_REQ_IDS}` substituted from gate context; undefined → `''`; all other `${X}` left untouched for `sh` to interpret | Parity with the context existing `check.query` gates already receive | +| Result mapping | exit 0 → `block:false`; non-zero → `block:true`; timeout (SIGTERM) → `block:true` (`timed_out`); `sh` missing (ENOENT, exit 127) → `block:true` | Fail-closed: every non-zero outcome blocks. A blocking gate with `block:true` halts per the existing two-step gate contract | +| Output cap | stderr/stdout tail embedded in `message` trimmed to 2000 chars | Keeps the `GATE_RESULT` payload context-bounded | + +### Evaluation path + +- A new pure leaf module `src/gate-predicate-evaluator.cts` owns + `evaluatePredicate(predicate, context, deps)`. It is fully deps-injected (the + subprocess seam is `runBoundedShell`) — no fs, no child_process, no config — + so it is trivially unit-testable without spawning. A `KIND_TABLE` dispatches + by `predicate.kind`; adding a future kind is a one-line registration. +- `check-command-router.cts` gains a `predicate` subcommand + (`gsd_run check predicate --predicate '' [--phase-dir …] [--phase-number …] + [--phase-req-ids …] --raw`). It parses flags, builds the production deps + (wrapping `execTool`), calls `evaluatePredicate`, and emits the standard + `{ block, message, details? }` envelope via `output()`. +- The three generic workflow gate-dispatch sites — `execute:wave:post`, + `execute:post`, `plan:post` — now branch on the gate's `check` shape: + `check.query` → existing `gsd_run check ` path; `check.predicate` → + `gsd_run check predicate`. The two-step Step-1 (command-failure → `onError`) + / Step-2 (`block` → halt) contract is **unchanged** — the predicate path + emits the same envelope and the same check-command-failure semantics. + +### Fail-closed mapping for malformed predicates + +`evaluatePredicate` **throws** for a malformed predicate (missing/non-string +command, non-positive timeout, oversized command, unknown kind). The CLI +wrapper maps a throw to `error()` (non-zero exit), which the workflow treats as +a **Step-1 command failure** — routed per the gate's `onError` (`halt` or +`skip`). This deliberately does **not** conflate an evaluator bug with a +legitimate gate-block decision: a recognised predicate returns +`{ block, … }`; an unrecognised one fails the check command. + +## Trust model + +Capabilities are opt-in installs (like npm packages): installing one already +trusts it to ship skills, agents, and hooks that run arbitrary code. +`command-exit-zero` is therefore **not a new trust boundary** — it is another +code-execution path for already-trusted capabilities. Security does not rely on +secrecy (Kerckhoffs): the command is declared in plain JSON, and safety comes +from bounded timeout + fail-closed mapping + the opt-in install, not from +hiding the mechanism. + +## Consequences + +- **Positive:** Third-party capability gates now actually fire. The path is + generic; future predicate kinds (e.g. a revival of `artifact-frontmatter-equals` + to retire the hard-coded `ship.md` security branch) register in `KIND_TABLE` + without further workflow changes. +- **Positive:** The evaluator is a pure leaf with injected I/O, matching the + ADR-857 module decomposition and the repo's test conventions. +- **Negative / documented limitation:** A command that backgrounds a child + (`sleep 100 &`) can outlive the timeout kill of its direct `sh` parent — + `execTool` kills the direct child on SIGTERM, not the whole process group. + This is the same property the existing `check.query` gates have (they too can + spawn long-running subprocesses). Full process-group kill is a future + hardening, out of scope here; the threat is a malicious capability, which is + already trusted. +- **Negative:** `sh` must be present. On Windows without a POSIX shell + (git-bash / WSL), `command-exit-zero` gates fail-closed with exit 127. This + matches the runtime's existing bash dependency for workflows. + +## Out of scope + +- Implementing the `artifact-frontmatter-equals` kind (the maintainer chose + Option 2 over Option 1; the kind table is structured for it but it is not + registered). +- Retiring the hard-coded `security` branch in `ship.md`. +- `check.agentVerdict` evaluation (advisory, `blocking:false`-forced, separate + concern). +- Process-group kill on timeout. + +## References + +- Issue: [#2008](https://github.com/open-gsd/gsd-core/issues/2008) +- Parent bundle: #2004 +- ADR-0894 (capability declaration format) +- ADR-0857 (capability system / module decomposition) +- `src/gate-predicate-evaluator.cts`, `src/check-command-router.cts` +- Diátaxis docs: `docs/reference/gate-predicates.md`, `docs/how-to/command-exit-zero-gate.md` diff --git a/docs/adr/README.md b/docs/adr/README.md index afe84e846..5922a7422 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -63,6 +63,7 @@ See **[CONTRIBUTING.md — "Proposing an ADR or PRD"](../../CONTRIBUTING.md#prop | [1593-skill-mapping-converter-methodology.md](1593-skill-mapping-converter-methodology.md) | Skill mapping & converter methodology across runtimes | Accepted | | [1769-state-md-transition-module.md](1769-state-md-transition-module.md) | STATE.md Transition Module — intent-based transitions over scattered RMW callbacks | Proposed | | [1817-state-md-rebuild-derivability-contract.md](1817-state-md-rebuild-derivability-contract.md) | STATE.md rebuild — derivability contract (capstone 11th transition) | Accepted | +| [2008-command-exit-zero-gate.md](2008-command-exit-zero-gate.md) | Generic gate-predicate evaluator with a `command-exit-zero` kind (#2008) | Accepted | ## Seam map diff --git a/docs/how-to/command-exit-zero-gate.md b/docs/how-to/command-exit-zero-gate.md new file mode 100644 index 000000000..473402d08 --- /dev/null +++ b/docs/how-to/command-exit-zero-gate.md @@ -0,0 +1,146 @@ +# How-to: add a `command-exit-zero` gate to a capability + +> **Diátaxis quadrant:** How-To. A step-by-step recipe for a capability author +> who wants a gate that runs a shell command and blocks the loop on non-zero +> exit. For the full specification, see +> [Reference: gate predicates](../reference/gate-predicates.md). + +## When to use this + +Use a `command-exit-zero` gate when your capability can be verified by an +existing command-line check — a test runner, a linter, a custom validator, a +git-hook-style probe. The gate runs your command at a loop point you choose and +blocks the loop (advisory or hard) when it exits non-zero. + +This is the generic extension path introduced in #2008 / ADR-2008. Before it, +only built-in `check.query` gates could fire; a third-party capability's +declared gate was display-only. + +## Prerequisites + +- A capability with a `capability.json` (see ADR-0894). +- A command that exits `0` on success and non-zero on failure, runnable via + `sh`. On Windows, `sh` must be present (git-bash / WSL) — otherwise the gate + fails closed with exit 127. + +## Step 1 — author the command + +Write a command that follows the **exit-0-on-success** contract. It receives +the project root as its cwd and the inherited process environment. + +```bash +# Example: a bundled check script shipped with the capability +node "${GSD_CAP_DIR}/checks/pre-ship.js" +``` + +You may interpolate three loop-context placeholders directly into the command: + +| Placeholder | Meaning | +|---|---| +| `${PHASE_NUMBER}` | the active phase number (e.g. `03`) | +| `${PHASE_DIR}` | the active phase directory | +| `${PHASE_REQ_IDS}` | the phase's requirement ids | + +Anything else (`${HOME}`, `$VAR`, etc.) is left for `sh` to interpret against +the inherited env. + +## Step 2 — declare the gate + +Add a `gates` entry to your `capability.json`. Pick the loop `point` (one of +the 12 canonical points), set `blocking` and `onError`, and gate it on a +config key with `when`: + +```json +{ + "gates": [ + { + "point": "ship:pre", + "check": { + "predicate": { + "kind": "command-exit-zero", + "command": "node \"${GSD_CAP_DIR}/checks/pre-ship.js\" \"${PHASE_DIR}\"", + "timeout": 30 + } + }, + "when": "my_cap.enabled", + "blocking": true, + "onError": "halt" + } + ] +} +``` + +Field rules (enforced by the evaluator): + +- `predicate.command` — required, non-empty string, ≤ 4096 chars. +- `predicate.timeout` — optional, positive finite number of seconds (default 30). +- `predicate.kind` — must be exactly `"command-exit-zero"`. + +## Step 3 — choose `blocking` and `onError` + +| Field | Effect | +|---|---| +| `blocking: true` | A `block: true` result **halts** the loop at this point. | +| `blocking: false` | Advisory — the gate prints its `message` and the loop continues. | +| `onError: "halt"` | If the check command itself fails (malformed predicate, unknown kind, or a command that cannot be evaluated), halt. | +| `onError: "skip"` | On a check-command failure, warn and continue (do not read `block`). | + +> A non-zero exit of **your** command is a gate *result* (`block: true`), not a +> check-command failure — `onError` does not apply to it. `onError` covers only +> the case where the gate could not be evaluated at all. + +## Step 4 — test the gate directly + +You can run the evaluator standalone to verify your predicate before wiring it +into a loop point: + +```bash +gsd_run check predicate \ + --predicate '{"kind":"command-exit-zero","command":"node checks/pre-ship.js \"${PHASE_DIR}\"","timeout":30}' \ + --phase-dir ".planning/phases/03-my-phase" \ + --phase-number "03" \ + --raw +``` + +Expected output on success: + +```json +{ "block": false, "message": "command exited 0", "details": { "kind": "command-exit-zero", "exitCode": 0 } } +``` + +Expected output when your command fails (e.g. exit 1): + +```json +{ "block": true, "message": "command exited 1: ", "details": { "kind": "command-exit-zero", "exitCode": 1, "signal": null } } +``` + +## Step 5 — verify it fires in the loop + +Once declared and the capability is active, the loop-resolver renders the gate +at your chosen point. The workflow gate-dispatch (`execute:wave:post`, +`execute:post`, `plan:post`, `ship:pre`, …) detects the `predicate` shape and +dispatches the evaluator automatically — no further wiring needed. + +```bash +gsd_run loop render-hooks ship:pre --raw +``` + +## Gotchas + +- **Backgrounded children escape the timeout.** `command: "sleep 100 &"` returns + before the sleep finishes; the timeout kills the `sh` parent, not the + backgrounded child. Keep your command foreground, or have it self-manage its + children. (ADR-2008, documented limitation.) +- **`sh` must be on PATH.** On Windows without git-bash/WSL, the gate fails + closed (exit 127). This matches the runtime's existing bash dependency. +- **Output is trimmed.** Only the last 2000 chars of stderr/stdout surface in + the gate `message`. Emit a concise diagnostic; don't rely on grepping long + output downstream. +- **Env is inherited.** The command sees the full GSD process environment. Do + not put secrets in the command string; read them from env or a file like any + shell script. + +## Related + +- [Reference: gate predicates](../reference/gate-predicates.md) +- [ADR-2008](../adr/2008-command-exit-zero-gate.md) diff --git a/docs/reference/gate-predicates.md b/docs/reference/gate-predicates.md new file mode 100644 index 000000000..145d6bbdc --- /dev/null +++ b/docs/reference/gate-predicates.md @@ -0,0 +1,117 @@ +# Gate predicates (reference) + +> **Diátaxis quadrant:** Reference. This is the canonical specification of the +> capability gate `check.predicate` evaluation path. For a step-by-step +> authoring guide, see [How-to: add a command-exit-zero gate](../how-to/command-exit-zero-gate.md). + +A capability gate's `check` block carries exactly one of three shapes +(`query`, `predicate`, `agentVerdict`), enforced by the registry validator +(`capability-validator.cjs:validateGate`). This page documents the +**`predicate`** shape and the kinds the built-in evaluator recognises. + +## Declaration + +```json +"gates": [ + { + "point": "", + "check": { + "predicate": { + "kind": "", + "" + } + }, + "when": "", + "blocking": true, + "onError": "halt" + } +] +``` + +The gate envelope (`point`, `when`, `blocking`, `onError`) follows the standard +contract documented in ADR-0894 (capability declaration format) and the +`Loop Host Contract` glossary entry in `CONTEXT.md`. This page covers only +`check.predicate`. + +## Evaluation path + +1. The loop-resolver (`gsd-tools loop render-hooks `) renders the active + gate hook (including its `check.predicate` declaration) to the workflow. +2. The workflow gate-dispatch reads the hook in-context and, when the `check` + shape is `predicate`, runs: + ```bash + gsd_run check predicate --predicate '' [--phase-dir …] [--phase-number …] [--phase-req-ids …] --raw + ``` +3. `check-command-router.cts:cmdCheckPredicate` parses the predicate, builds the + production subprocess binding, and calls + `gate-predicate-evaluator.cjs:evaluatePredicate`, which dispatches by + `predicate.kind`. +4. The evaluator returns the standard gate envelope: + ```json + { "block": , "message": "", "details": { … } } + ``` +5. The workflow applies the **two-step gate contract** unchanged: + - **Step 1** — if the check command itself failed (non-zero exit, e.g. a + malformed predicate / unknown kind), route per `onError` (`halt` or `skip`). + - **Step 2** — if the command succeeded, a `blocking: true` gate halts on + `block: true`; an advisory gate shows `message` and continues. + +## Built-in kinds + +### `command-exit-zero` + +Runs a declared command in a bounded `sh -c` subprocess; **exit 0 → pass, +non-zero → block, timeout → block.** See ADR-2008 for the full sandbox +contract. + +| Field | Type | Required | Default | Notes | +|---|---|---|---|---| +| `kind` | string | yes | — | Must be `"command-exit-zero"` | +| `command` | string | yes | — | The shell command. Non-empty, ≤ 4096 chars | +| `timeout` | number | no | `30` | Positive finite number, seconds | + +**Interpolation.** Before execution, three placeholders are substituted from +the gate context; all others are left untouched for `sh` to interpret: + +| Placeholder | Source | Workflow flag | +|---|---|---| +| `${PHASE_NUMBER}` | the active phase number | `--phase-number` | +| `${PHASE_DIR}` | the active phase directory | `--phase-dir` | +| `${PHASE_REQ_IDS}` | the phase's requirement ids | `--phase-req-ids` | + +An undefined placeholder interpolates to the empty string. + +**Sandbox.** cwd = project root; env = inherited from the GSD process; killed +(SIGTERM) on timeout. The command runs as the user, on the user's machine — +there is no sandbox boundary vs. the user's own shell. See ADR-2008 "Trust +model". + +**Result mapping.** + +| Command outcome | `block` | `message` | +|---|---|---| +| exit 0 | `false` | `command exited 0` | +| exit N (non-zero) | `true` | `command exited N: ` | +| timeout (SIGTERM) | `true` | `command timed out after s: ` | +| `sh` missing (ENOENT, exit 127) | `true` | `command exited 127: sh: not found` | + +**Validation errors (throw → check-command failure → Step-1 / `onError`).** + +- Missing, non-string, empty, or whitespace-only `command`. +- `command` longer than 4096 chars. +- `timeout` present but not a positive finite number. +- Unknown `kind`. + +## Extensibility + +The evaluator dispatches through a `KIND_TABLE`. Adding a new built-in kind is +a one-line registration in `gate-predicate-evaluator.cts` — no workflow changes +required, since the workflow dispatches any `check.predicate` to the same +`gsd_run check predicate` subcommand. + +## Related + +- [ADR-2008](../adr/2008-command-exit-zero-gate.md) — full decision record. +- [How-to: add a command-exit-zero gate](../how-to/command-exit-zero-gate.md). +- ADR-0894 — capability declaration format. +- `src/gate-predicate-evaluator.cts`, `src/check-command-router.cts`. diff --git a/eslint.config.mjs b/eslint.config.mjs index 57bce37c2..580c90822 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -170,6 +170,7 @@ export default tseslint.config( 'gsd-core/bin/lib/roadmap-command-router.cjs', 'gsd-core/bin/lib/state-command-router.cjs', 'gsd-core/bin/lib/gap-checker.cjs', + 'gsd-core/bin/lib/gate-predicate-evaluator.cjs', 'gsd-core/bin/lib/config.cjs', 'gsd-core/bin/lib/profile-output.cjs', 'gsd-core/bin/lib/commands.cjs', diff --git a/gsd-core/workflows/execute-phase.md b/gsd-core/workflows/execute-phase.md index c2dc43acb..18a96b1fa 100644 --- a/gsd-core/workflows/execute-phase.md +++ b/gsd-core/workflows/execute-phase.md @@ -921,7 +921,7 @@ increases monotonically across waves. `{status}` is `complete` (success), **If `activeHooks` is empty or absent:** Skip silently to step 5.8. - **For each active entry where `kind == "gate"`** (process in array order), run the gate check: + **For each active entry where `kind == "gate"`** (process in array order), run the gate check — for a `predicate` gate (ADR-2008 / #2008) substitute `gsd_run check predicate --predicate '' --phase-number "${PHASE_NUMBER}" --raw` for the `check.query` form: ```bash GATE_RESULT=$(gsd_run check ${hook.check.query} "${PHASE_NUMBER}" --raw) @@ -1198,15 +1198,14 @@ Code review found issues. Consider running: **Error handling:** If the Skill invocation fails or throws, catch the error, display "Code review encountered an error (non-blocking): {error}" and proceed to gate dispatch. Review failures must never block execution. -**Execute:post gate hook dispatch.** After code review, dispatch all active gate hooks from `EXECUTE_POST_HOOKS_JSON` where `kind == "gate"`: +**Execute:post gate hook dispatch.** After code review, dispatch all active gate hooks from `EXECUTE_POST_HOOKS_JSON` where `kind == "gate"`. For each, run `gsd_run check ${hook.check.query} "${PHASE_NUMBER}" --raw`, or — for a `predicate` gate (ADR-2008 / #2008) — `gsd_run check predicate --predicate '' --phase-number "${PHASE_NUMBER}" --raw`: -For each active gate hook: ```bash GATE_RESULT=$(gsd_run check ${hook.check.query} "${PHASE_NUMBER}" --raw) CHECK_EXIT=$? ``` -**Gate evaluation** uses the same two-step contract as `execute:wave:post` above: **Step 1** — if the check command failed (non-zero `CHECK_EXIT`, empty/unparseable output), `onError == "halt"` stops and surfaces the error, `onError == "skip"` warns and continues to the next hook (do not read `block`). **Step 2** (command succeeded) — a blocking gate (`hook.blocking == true`) halts on `GATE_RESULT.block == true` with its message/table (never bypassed by `onError`); an advisory gate (`hook.blocking == false`) shows its `table`/summary when `block == true` or `message` is non-empty, then continues; a blocking gate with `block == false` continues silently. +**Gate evaluation** uses the same two-step contract as `execute:wave:post` above (Step 1: command-failure → `onError`; Step 2: `block == true` halts a blocking gate; an advisory gate shows its `message`/`table` and continues). **TDD review escalation (overrides the advisory default for the `tdd.review-checkpoint` gate only).** The tdd `execute:post` gate is declared `blocking: false`, so by the generic contract above it displays its `message`/table and continues. There is ONE documented exception (see `~/.claude/gsd-core/references/execute-mvp-tdd.md`): when `MVP_MODE=true` AND `TDD_MODE=true` AND `GATE_RESULT.block == true` (one or more TDD plans miss a RED or GREEN gate commit), the end-of-phase TDD review escalates from advisory to **blocking under MVP+TDD** — refuse to mark the phase complete and present: diff --git a/gsd-core/workflows/plan-phase.md b/gsd-core/workflows/plan-phase.md index b457c4350..c5f245a34 100644 --- a/gsd-core/workflows/plan-phase.md +++ b/gsd-core/workflows/plan-phase.md @@ -1468,12 +1468,19 @@ PHASE_REQ_IDS=$(gsd_run query init.plan-phase "$PHASE" --pick phase_req_ids 2>/d Read the `activeHooks` array from `PLAN_POST_HOOKS_JSON` in-context. If the `gap-analysis` gate hook is absent (capability inactive), skip this step. -**For each active entry where `kind == "gate"`** (process in array order): +**For each active entry where `kind == "gate"`** (process in array order). **Dispatch by check shape** (the registry validates exactly one of `query`/`predicate`/`agentVerdict`): ```bash +# named-query gate: GATE_RESULT=$(gsd_run check ${hook.check.query} "${PHASE_DIR}" "${PHASE_REQ_IDS}" --raw) CHECK_EXIT=$? ``` +OR, for a generic `predicate` gate (ADR-2008 / #2008), inline the predicate as compact JSON (note the `--phase-dir`/`--phase-req-ids` flags feed `${PHASE_DIR}`/`${PHASE_REQ_IDS}` interpolation): +```bash +GATE_RESULT=$(gsd_run check predicate --predicate '' --phase-dir "${PHASE_DIR}" --phase-req-ids "${PHASE_REQ_IDS}" --raw) +CHECK_EXIT=$? +``` +(Read the hook's `check` object in-context to pick the branch; a gate with neither is a malformed registry entry — skip with a warning.) **Step 1 — did the CHECK COMMAND itself succeed?** If the check command failed (non-zero `CHECK_EXIT`, empty output, or unparseable JSON): diff --git a/src/check-command-router.cts b/src/check-command-router.cts index 7aec8d1d3..c3201b502 100644 --- a/src/check-command-router.cts +++ b/src/check-command-router.cts @@ -32,6 +32,10 @@ const { getRoadmapPhaseWithFallback } = roadmapModule; import gapCheckerModule = require('./gap-checker.cjs'); const { runGapAnalysis } = gapCheckerModule; import { routeProhibitionEnforcement } from './prohibition-enforcement.cjs'; +// eslint-disable-next-line @typescript-eslint/no-require-imports +import gatePredicateEval = require('./gate-predicate-evaluator.cjs'); +const { evaluatePredicate } = gatePredicateEval; +import { execTool } from './shell-command-projection.cjs'; // ─── Helpers ────────────────────────────────────────────────────────────────── @@ -882,6 +886,106 @@ interface RouteCheckCommandOptions { raw: boolean; } +// ─── predicate (generic gate-predicate evaluator, #2008) ────────────────────── + +/** + * Production subprocess binding for the gate-predicate evaluator. Wraps the + * bounded `execTool` seam (shell-command-projection) as a `runBoundedShell` + * the pure evaluator consumes. `sh -c` runs the interpolated command; the + * subprocess inherits the process env and is killed (SIGTERM) on timeout. + * + * `timedOut` is derived from the kill signal: spawnSync sets `signal: 'SIGTERM'` + * when the `timeout` fires, distinct from a normal non-zero exit code. A command + * that self-terminates with SIGTERM is indistinguishable at this seam and is + * reported as a timeout — either way the gate blocks (non-zero), so the outcome + * is fail-closed and correct. See ADR-2008. + */ +function buildPredicateDeps() { + return { + runBoundedShell(opts: { command: string; cwd: string; timeoutMs: number }): { + exitCode: number | null; + stdout: string; + stderr: string; + signal: NodeJS.Signals | null; + timedOut: boolean; + } { + const r = execTool('sh', ['-c', opts.command], { cwd: opts.cwd, timeout: opts.timeoutMs }); + return { + exitCode: r.exitCode, + stdout: r.stdout, + stderr: r.stderr, + signal: r.signal, + timedOut: r.signal === 'SIGTERM', + }; + }, + }; +} + +/** Parse `--flag value` pairs from an args array into a map (last write wins). */ +function parsePredicateFlags(args: string[]): Record { + const out: Record = {}; + for (let i = 0; i < args.length; i++) { + const a = args[i]; + if (typeof a !== 'string') continue; + if (!a.startsWith('--')) continue; + const key = a.slice(2); + const next = args[i + 1]; + if (key.length > 0 && typeof next === 'string' && !next.startsWith('--')) { + out[key] = next; + i++; + } + } + return out; +} + +/** + * `check predicate` — generic evaluator for capability gate `check.predicate` + * blocks (#2008). The workflow gate-dispatch invokes this for any gate whose + * `check` carries a `predicate` (instead of a `query`); the predicate object is + * passed as `--predicate ''`. Emits the standard `{ block, message, + * details? }` gate contract on success. A malformed predicate / unknown kind + * THROWS inside the evaluator and is mapped here to `error()` (non-zero exit), + * which the workflow's two-step gate contract treats as a step-1 command failure + * routed per the gate's `onError`. + * + * Invocation: + * gsd_run check predicate --predicate '' \ + * [--phase-dir ] [--phase-number ] [--phase-req-ids ] --raw + * + * The subprocess runs at the runtime project root (the `cwd` passed to this + * router), inheriting the process env. Interpolation placeholders + * ${PHASE_NUMBER}/${PHASE_DIR}/${PHASE_REQ_IDS} are substituted from the flags. + */ +function cmdCheckPredicate(projectDir: string, args: string[], raw: boolean): void { + const flags = parsePredicateFlags(args); + const predicateJson = flags['predicate']; + if (!predicateJson) { + error('predicate requires --predicate (the gate hook check.predicate object)', ERROR_REASON.SDK_MISSING_ARG); + return; + } + let predicate: unknown; + try { + predicate = JSON.parse(predicateJson); + } catch { + error('predicate --predicate value must be valid JSON', ERROR_REASON.USAGE); + return; + } + const ctx = { + cwd: projectDir, + phaseNumber: flags['phase-number'], + phaseDir: flags['phase-dir'], + phaseReqIds: flags['phase-req-ids'], + }; + let result; + try { + result = evaluatePredicate(predicate, ctx, buildPredicateDeps()); + } catch (e) { + error(`gate predicate evaluation failed: ${(e as Error).message}`, ERROR_REASON.USAGE); + return; + } + output(result, raw, undefined); +} + function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void { // Normalize dots to hyphens in the subcommand so both forms are accepted. // This makes `check.query = "ui.plan-gate"` (dotted form in capability.json gates) @@ -934,6 +1038,15 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void { cmdVerifyCodebaseDrift(cwd, raw); return; } + if (subcommand === 'predicate') { + // Generic gate-predicate evaluator (#2008). The workflow gate-dispatch calls + // this for any gate whose `check` carries a `predicate` (instead of a `query`), + // passing the predicate object as --predicate ''. NOTE: unlike the + // `check.query` subcommands above (which take positional phase args), this + // subcommand parses --flag value pairs. + cmdCheckPredicate(cwd, args, raw); + return; + } if (subcommand === 'prohibition-enforcement') { // The deterministic test-tier prohibition PRODUCER/gate (#1259, ADR-550 D5d). Locates the // wired mechanical check (node-test or lint-rule), confirms fail-first, runs it, builds @@ -942,7 +1055,7 @@ function routeCheckCommand({ args, cwd, raw }: RouteCheckCommandOptions): void { routeProhibitionEnforcement(args, raw); return; } - error('Unknown check subcommand. Available: auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, prohibition-enforcement, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-schema-drift, verify-codebase-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND); + error('Unknown check subcommand. Available: auto-mode, decision-coverage-plan, decision-coverage-verify, gap-analysis-plan-post, predicate, prohibition-enforcement, tdd-review-checkpoint, ui-plan-gate, ui-safety-gate, verify-schema-drift, verify-codebase-drift', ERROR_REASON.SDK_UNKNOWN_COMMAND); } export = { @@ -953,4 +1066,7 @@ export = { computeUiSafetyGate, cmdGapAnalysisPlanPost, cmdTddReviewCheckpoint, + cmdCheckPredicate, + buildPredicateDeps, + parsePredicateFlags, }; diff --git a/src/gate-predicate-evaluator.cts b/src/gate-predicate-evaluator.cts new file mode 100644 index 000000000..a1cabe04e --- /dev/null +++ b/src/gate-predicate-evaluator.cts @@ -0,0 +1,204 @@ +/** + * Gate Predicate Evaluator — issue #2008 / ADR-2008 + * + * Pure, deps-injected evaluator for capability gate `check.predicate` blocks. + * + * Background: the loop-resolver renders active gate hooks (carrying their + * `check.predicate` declarations) but, prior to #2008, nothing EVALUATED a + * declared predicate for non-built-in capabilities — `check.query` was the only + * enforced shape (dispatched via `gsd_run check `), and `check.predicate` + * was declaration-only. This module is the generic evaluation path. + * + * The workflow gate-dispatch calls this evaluator (via the `gsd_run check predicate` + * subcommand in check-command-router) for any gate whose `check` carries a + * `predicate` instead of a `query`. The evaluator dispatches by `predicate.kind` + * and returns the existing `{ block, message }` gate contract. A THROWN error + * (malformed predicate / unknown kind) is mapped by the CLI wrapper to a + * non-zero check-command exit, which the workflow's two-step gate contract treats + * as a step-1 command failure (routed per the gate's `onError`). + * + * Built-in kind (v1): `command-exit-zero` — run a declared command in a bounded + * `sh -c` subprocess at the project root, inheriting the process env; exit 0 => + * pass, non-zero => block, timeout => block. The production runBoundedShell + * binding is shell-command-projection.execTool (bounded spawnSync). See ADR-2008 + * for the full sandbox contract. + * + * This is a leaf pure module: no fs, no child_process, no config — the subprocess + * seam is injected so the evaluator is trivially testable without spawning. + */ + +// ─── Public constants ───────────────────────────────────────────────────────── + +/** Default command timeout for `command-exit-zero` (30s). Matches execTool default. */ +const COMMAND_EXIT_ZERO_DEFAULT_TIMEOUT_MS = 30_000; + +/** Hard cap on the stderr/stdout tail embedded in the gate `message`. */ +const COMMAND_MAX_OUTPUT_CHARS = 2000; + +/** Hard cap on the declared command length (defense-in-depth against ARGV overflow / abuse). */ +const COMMAND_MAX_LENGTH = 4096; + +/** Predicate kinds this evaluator recognises (extensible — add to KIND_TABLE). */ +const EVALUATOR_KINDS = Object.freeze(['command-exit-zero']); + +/** Placeholders interpolated into a declared command, in addition to sh's own vars. */ +const INTERPOLATION_VAR_NAMES = Object.freeze(['PHASE_NUMBER', 'PHASE_DIR', 'PHASE_REQ_IDS']); + +// ─── Types (internal; runtime API is the `export =` block) ──────────────────── + +interface PredicateContext { + /** Project root — also the cwd of the bounded subprocess. */ + cwd: string; + phaseNumber?: string; + phaseDir?: string; + phaseReqIds?: string; +} + +interface BoundedShellResult { + exitCode: number | null; + stdout: string; + stderr: string; + signal: NodeJS.Signals | null; + timedOut: boolean; +} + +interface PredicateDeps { + runBoundedShell(opts: { command: string; cwd: string; timeoutMs: number }): BoundedShellResult; +} + +interface PredicateResult { + block: boolean; + message: string; + details?: Record; +} + +// ─── Helpers ────────────────────────────────────────────────────────────────── + +const INTERPOLATION_RE = /\$\{(PHASE_NUMBER|PHASE_DIR|PHASE_REQ_IDS)\}/g; + +/** Replace the three known ${PHASE_*} placeholders with context values (undefined => ''). */ +function interpolate(command: string, ctx: PredicateContext): string { + return command.replace(INTERPOLATION_RE, (_whole, name: string): string => { + if (name === 'PHASE_NUMBER') return ctx.phaseNumber ?? ''; + if (name === 'PHASE_DIR') return ctx.phaseDir ?? ''; + if (name === 'PHASE_REQ_IDS') return ctx.phaseReqIds ?? ''; + return ''; + }); +} + +/** Cap a string at COMMAND_MAX_OUTPUT_CHARS so gate messages stay context-bounded. */ +function trimToMax(s: string): string { + return s.length > COMMAND_MAX_OUTPUT_CHARS ? s.slice(0, COMMAND_MAX_OUTPUT_CHARS) : s; +} + +function isNonEmptyString(v: unknown): v is string { + return typeof v === 'string' && v.trim().length > 0; +} + +// ─── Kind: command-exit-zero ────────────────────────────────────────────────── + +function evaluateCommandExitZero( + predicate: Record, + ctx: PredicateContext, + deps: PredicateDeps, +): PredicateResult { + const command = predicate['command']; + if (!isNonEmptyString(command)) { + throw new Error('command-exit-zero predicate requires a non-empty string "command"'); + } + if (command.length > COMMAND_MAX_LENGTH) { + throw new Error(`command-exit-zero predicate "command" exceeds max length ${COMMAND_MAX_LENGTH}`); + } + + let timeoutMs = COMMAND_EXIT_ZERO_DEFAULT_TIMEOUT_MS; + const rawTimeout = predicate['timeout']; + if (rawTimeout !== undefined) { + if (typeof rawTimeout !== 'number' || !Number.isFinite(rawTimeout) || rawTimeout <= 0) { + throw new Error('command-exit-zero predicate "timeout" must be a positive finite number (seconds)'); + } + timeoutMs = Math.floor(rawTimeout * 1000); + } + + const interpolated = interpolate(command, ctx); + const res = deps.runBoundedShell({ command: interpolated, cwd: ctx.cwd, timeoutMs }); + + if (res.timedOut) { + return { + block: true, + message: trimToMax(`command timed out after ${Math.round(timeoutMs / 1000)}s: ${res.stderr || interpolated}`), + details: { kind: 'command-exit-zero', timedOut: true, signal: res.signal }, + }; + } + + if (res.exitCode === 0) { + return { + block: false, + message: 'command exited 0', + details: { kind: 'command-exit-zero', exitCode: 0 }, + }; + } + + // Non-zero (incl. null exit from a signal kill) => block. Surface code + stderr/stdout tail. + const code = res.exitCode === null ? '' : String(res.exitCode); + const tail = trimToMax(res.stderr || res.stdout || ''); + const message = tail ? `command exited ${code}: ${tail}` : `command exited ${code}`; + return { + block: true, + message: trimToMax(message), + details: { kind: 'command-exit-zero', exitCode: res.exitCode, signal: res.signal }, + }; +} + +// ─── Kind dispatch table ────────────────────────────────────────────────────── + +const KIND_TABLE: Record, ctx: PredicateContext, deps: PredicateDeps) => PredicateResult> = { + 'command-exit-zero': evaluateCommandExitZero, +}; + +// ─── Public entry point ─────────────────────────────────────────────────────── + +/** + * Evaluate a capability gate `check.predicate`. + * + * Returns `{ block, message }` for any successfully-recognised predicate. + * THROWS for a malformed predicate, missing context/deps, or unknown `kind` — + * the CLI wrapper converts a throw into a non-zero check-command exit so the + * workflow's `onError` (step-1) contract applies. This keeps the gate fail-closed + * without conflating an evaluator bug with a legitimate gate-block decision. + */ +function evaluatePredicate(predicate: unknown, context: unknown, deps: unknown): PredicateResult { + if (!predicate || typeof predicate !== 'object' || Array.isArray(predicate)) { + throw new Error('predicate must be an object'); + } + const ctx = context as PredicateContext; + if (!ctx || typeof ctx.cwd !== 'string' || ctx.cwd.length === 0) { + throw new Error('predicate context requires a non-empty "cwd"'); + } + const d = deps as PredicateDeps; + if (!d || typeof d.runBoundedShell !== 'function') { + throw new Error('predicate deps require a "runBoundedShell" function'); + } + + const pred = predicate as Record; + const kind = pred['kind']; + if (typeof kind !== 'string' || kind.length === 0) { + throw new Error('predicate.kind must be a non-empty string'); + } + + const handler = KIND_TABLE[kind]; + if (!handler) { + throw new Error(`Unknown predicate kind: "${kind}". Known kinds: ${EVALUATOR_KINDS.join(', ')}`); + } + return handler(pred, ctx, d); +} + +export = { + evaluatePredicate, + evaluateCommandExitZero, + interpolate, + COMMAND_EXIT_ZERO_DEFAULT_TIMEOUT_MS, + COMMAND_MAX_OUTPUT_CHARS, + COMMAND_MAX_LENGTH, + EVALUATOR_KINDS, + INTERPOLATION_VAR_NAMES, +}; diff --git a/tests/check-predicate.test.cjs b/tests/check-predicate.test.cjs new file mode 100644 index 000000000..ea9658546 --- /dev/null +++ b/tests/check-predicate.test.cjs @@ -0,0 +1,106 @@ +'use strict'; + +/** + * Integration tests for the `check predicate` subcommand wiring (#2008). + * + * These exercise the PRODUCTION stack: the real `buildPredicateDeps()` binding + * (which wraps shell-command-projection.execTool → bounded `sh -c` spawnSync) and + * the `parsePredicateFlags` arg parser. The pure evaluator logic is covered by + * gate-predicate-evaluator.test.cjs; this file proves the wiring holds against + * real subprocess exit codes and a real timeout kill. + * + * Commands run are instant (`true` / `false` / `exit 3`) or tightly bounded + * (a 100ms timeout killing `sleep 1`), so there is no orphan/leak risk. + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); + +const { evaluatePredicate } = require('../gsd-core/bin/lib/gate-predicate-evaluator.cjs'); +const { buildPredicateDeps, parsePredicateFlags } = require('../gsd-core/bin/lib/check-command-router.cjs'); + +// ─── buildPredicateDeps: real subprocess exit mapping ───────────────────────── + +describe('buildPredicateDeps — real bounded sh -c subprocess', () => { + const deps = buildPredicateDeps(); + const cwd = process.cwd(); + + test('`true` => exitCode 0, not timed out', () => { + const r = deps.runBoundedShell({ command: 'true', cwd, timeoutMs: 5000 }); + assert.equal(r.exitCode, 0); + assert.equal(r.timedOut, false); + }); + + test('`false` => exitCode 1, not timed out', () => { + const r = deps.runBoundedShell({ command: 'false', cwd, timeoutMs: 5000 }); + assert.equal(r.exitCode, 1); + assert.equal(r.timedOut, false); + }); + + test('`exit 3` => exitCode 3', () => { + const r = deps.runBoundedShell({ command: 'exit 3', cwd, timeoutMs: 5000 }); + assert.equal(r.exitCode, 3); + }); + + test('stderr is captured from the subprocess', () => { + const r = deps.runBoundedShell({ command: 'echo oops >&2; exit 4', cwd, timeoutMs: 5000 }); + assert.equal(r.exitCode, 4); + assert.match(r.stderr, /oops/); + }); + + test('timeout kills the subprocess (SIGTERM => timedOut:true)', () => { + const r = deps.runBoundedShell({ command: 'sleep 1', cwd, timeoutMs: 100 }); + assert.equal(r.timedOut, true); + assert.equal(r.signal, 'SIGTERM'); + }); +}); + +// ─── evaluatePredicate + production deps: end-to-end exit mapping ───────────── + +describe('evaluatePredicate + production deps — command-exit-zero e2e', () => { + const deps = buildPredicateDeps(); + const ctx = { cwd: process.cwd() }; + + test('command `true` => block:false', () => { + const res = evaluatePredicate({ kind: 'command-exit-zero', command: 'true' }, ctx, deps); + assert.equal(res.block, false); + }); + + test('command `false` => block:true', () => { + const res = evaluatePredicate({ kind: 'command-exit-zero', command: 'false' }, ctx, deps); + assert.equal(res.block, true); + assert.match(res.message, /1/); + }); + + test('interpolation reaches the real shell ($PHASE_NUMBER via flag context)', () => { + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'test "${PHASE_NUMBER}" = "07" && true || false' }, + { cwd: process.cwd(), phaseNumber: '07' }, + deps, + ); + assert.equal(res.block, false); + }); +}); + +// ─── parsePredicateFlags ─────────────────────────────────────────────────────── + +describe('parsePredicateFlags', () => { + test('extracts --flag value pairs, skips positional + bare --flags', () => { + const out = parsePredicateFlags(['check', 'predicate', '--predicate', '{"kind":"x"}', '--phase-number', '03', '--raw']); + assert.deepEqual(out, { predicate: '{"kind":"x"}', 'phase-number': '03' }); + }); + + test('last write wins for repeated flags', () => { + const out = parsePredicateFlags(['--phase-number', '01', '--phase-number', '02']); + assert.equal(out['phase-number'], '02'); + }); + + test('value that starts with -- is not consumed (treated as a flag)', () => { + const out = parsePredicateFlags(['--predicate', '--phase-number']); + assert.equal('predicate' in out, false); + }); + + test('empty args => empty map', () => { + assert.deepEqual(parsePredicateFlags([]), {}); + }); +}); diff --git a/tests/fixtures/golden-install-parity/antigravity.json b/tests/fixtures/golden-install-parity/antigravity.json index 8ad9da951..478d209d6 100644 --- a/tests/fixtures/golden-install-parity/antigravity.json +++ b/tests/fixtures/golden-install-parity/antigravity.json @@ -228,7 +228,7 @@ "gsd-core/workflows/docs-update.md": "1c3a0d23ecc9605c", "gsd-core/workflows/edit-phase.md": "f685fb7063616826", "gsd-core/workflows/eval-review.md": "2da6215ce79b2f7c", - "gsd-core/workflows/execute-phase.md": "e7713d986ebf45e0", + "gsd-core/workflows/execute-phase.md": "660fb92b8ed43129", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "1da22ba32f524c6e", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "bb9113e7bf953797", @@ -264,7 +264,7 @@ "gsd-core/workflows/note.md": "3ce09c0aa0a20599", "gsd-core/workflows/pause-work.md": "3196d681d4dd8c71", "gsd-core/workflows/plan-milestone-gaps.md": "94b193dfc9ca3681", - "gsd-core/workflows/plan-phase.md": "659b52bd338720fe", + "gsd-core/workflows/plan-phase.md": "1fecfb77b67d4833", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "4099ef6d0868de60", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "473d4d6e5b1f6340", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "e9de7a96bbfff261", diff --git a/tests/fixtures/golden-install-parity/augment.json b/tests/fixtures/golden-install-parity/augment.json index 0ecfccd2f..c4ab8c706 100644 --- a/tests/fixtures/golden-install-parity/augment.json +++ b/tests/fixtures/golden-install-parity/augment.json @@ -298,7 +298,7 @@ "gsd-core/workflows/docs-update.md": "806ada831b961116", "gsd-core/workflows/edit-phase.md": "e5624ac6e3f8bef5", "gsd-core/workflows/eval-review.md": "dfcfb4f8ce031fae", - "gsd-core/workflows/execute-phase.md": "5daeb8fac7cab295", + "gsd-core/workflows/execute-phase.md": "5383dabd9fd40206", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "8898e0ea533cc643", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "abd2aca069c04a80", @@ -334,7 +334,7 @@ "gsd-core/workflows/note.md": "5a99eb396c744619", "gsd-core/workflows/pause-work.md": "7bcbdf27ba957c8b", "gsd-core/workflows/plan-milestone-gaps.md": "02fee851c82e3b25", - "gsd-core/workflows/plan-phase.md": "dc0beefd1316a860", + "gsd-core/workflows/plan-phase.md": "f80ff87cec37c25c", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "b36f77ac7344a072", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "d64097345ffac40d", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "49f58c3f75be3eb5", diff --git a/tests/fixtures/golden-install-parity/claude.json b/tests/fixtures/golden-install-parity/claude.json index f173ebd0d..440cb4fbf 100644 --- a/tests/fixtures/golden-install-parity/claude.json +++ b/tests/fixtures/golden-install-parity/claude.json @@ -227,7 +227,7 @@ "gsd-core/workflows/docs-update.md": "98c30bec9350542f", "gsd-core/workflows/edit-phase.md": "a3ab51739c5c021c", "gsd-core/workflows/eval-review.md": "f47fd1a7a1ca3308", - "gsd-core/workflows/execute-phase.md": "fe39dd64ff9db8f6", + "gsd-core/workflows/execute-phase.md": "d8ffe62c26c3f28e", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "c133828cf177a772", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "abd2aca069c04a80", @@ -263,7 +263,7 @@ "gsd-core/workflows/note.md": "42b66686b2c102cb", "gsd-core/workflows/pause-work.md": "0be71264eafd16dc", "gsd-core/workflows/plan-milestone-gaps.md": "1976bf2001969719", - "gsd-core/workflows/plan-phase.md": "f0401a3221b74b58", + "gsd-core/workflows/plan-phase.md": "21a81a2c468c53c2", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "4099ef6d0868de60", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "d64097345ffac40d", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "e9de7a96bbfff261", diff --git a/tests/fixtures/golden-install-parity/cline.json b/tests/fixtures/golden-install-parity/cline.json index 67dc509ab..7a8644def 100644 --- a/tests/fixtures/golden-install-parity/cline.json +++ b/tests/fixtures/golden-install-parity/cline.json @@ -231,7 +231,7 @@ "gsd-core/workflows/docs-update.md": "4aee7d852ab26710", "gsd-core/workflows/edit-phase.md": "db501e21c762d2b2", "gsd-core/workflows/eval-review.md": "e211cca4eea94930", - "gsd-core/workflows/execute-phase.md": "61b798388f3c28d7", + "gsd-core/workflows/execute-phase.md": "20299af3e665f539", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "84b49fd65d290399", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "bcb4ab75df626846", @@ -267,7 +267,7 @@ "gsd-core/workflows/note.md": "5a99eb396c744619", "gsd-core/workflows/pause-work.md": "9c1acf8c30a244fd", "gsd-core/workflows/plan-milestone-gaps.md": "95ca791b0867fe2d", - "gsd-core/workflows/plan-phase.md": "e4256d987d376112", + "gsd-core/workflows/plan-phase.md": "b906c040d22d0424", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "b36f77ac7344a072", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "ae59c247ad4a57d2", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "090c31e22b1508fe", diff --git a/tests/fixtures/golden-install-parity/codebuddy.json b/tests/fixtures/golden-install-parity/codebuddy.json index 99b10d015..84fd4f82a 100644 --- a/tests/fixtures/golden-install-parity/codebuddy.json +++ b/tests/fixtures/golden-install-parity/codebuddy.json @@ -298,7 +298,7 @@ "gsd-core/workflows/docs-update.md": "806ada831b961116", "gsd-core/workflows/edit-phase.md": "e5624ac6e3f8bef5", "gsd-core/workflows/eval-review.md": "dfcfb4f8ce031fae", - "gsd-core/workflows/execute-phase.md": "d3c1583906de7408", + "gsd-core/workflows/execute-phase.md": "05280ea81a40a29f", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "8898e0ea533cc643", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "abd2aca069c04a80", @@ -334,7 +334,7 @@ "gsd-core/workflows/note.md": "5a99eb396c744619", "gsd-core/workflows/pause-work.md": "7bcbdf27ba957c8b", "gsd-core/workflows/plan-milestone-gaps.md": "02fee851c82e3b25", - "gsd-core/workflows/plan-phase.md": "9fc35fa37928c14e", + "gsd-core/workflows/plan-phase.md": "9ad0d58d4abbfcb3", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "b36f77ac7344a072", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "d64097345ffac40d", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "49f58c3f75be3eb5", diff --git a/tests/fixtures/golden-install-parity/codex.json b/tests/fixtures/golden-install-parity/codex.json index ea74f7ab3..268d8d358 100644 --- a/tests/fixtures/golden-install-parity/codex.json +++ b/tests/fixtures/golden-install-parity/codex.json @@ -263,7 +263,7 @@ "gsd-core/workflows/docs-update.md": "c5e1ea372f4a9ee8", "gsd-core/workflows/edit-phase.md": "c83ef0701c19c455", "gsd-core/workflows/eval-review.md": "68d7d96bd415629b", - "gsd-core/workflows/execute-phase.md": "5db223634afa2b88", + "gsd-core/workflows/execute-phase.md": "5ac844394468bef0", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "7ae407e4435c9a2a", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "abd2aca069c04a80", @@ -299,7 +299,7 @@ "gsd-core/workflows/note.md": "664da466ab989f9d", "gsd-core/workflows/pause-work.md": "3c2ee96295959527", "gsd-core/workflows/plan-milestone-gaps.md": "8ee841bcc7adc836", - "gsd-core/workflows/plan-phase.md": "5d3ed552933f81da", + "gsd-core/workflows/plan-phase.md": "58ee6f71fa269c46", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "d838b87563feedf6", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "44a12264f8b61d03", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "f5edc589cab52a7b", diff --git a/tests/fixtures/golden-install-parity/copilot.json b/tests/fixtures/golden-install-parity/copilot.json index 929527ce6..14fa8a7c1 100644 --- a/tests/fixtures/golden-install-parity/copilot.json +++ b/tests/fixtures/golden-install-parity/copilot.json @@ -229,7 +229,7 @@ "gsd-core/workflows/docs-update.md": "15436f1f87ce9bd2", "gsd-core/workflows/edit-phase.md": "a619911fd32d1f8e", "gsd-core/workflows/eval-review.md": "b29095c5e7149975", - "gsd-core/workflows/execute-phase.md": "dcb483d5fceffac1", + "gsd-core/workflows/execute-phase.md": "63945642ac949204", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "8d835dbbc9bc72cf", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "a01e6aae9f3e416e", @@ -265,7 +265,7 @@ "gsd-core/workflows/note.md": "4a5ee74cf2fc1f54", "gsd-core/workflows/pause-work.md": "324e04e675dc7f7f", "gsd-core/workflows/plan-milestone-gaps.md": "c0eb896eb42e22d4", - "gsd-core/workflows/plan-phase.md": "371be4167168d461", + "gsd-core/workflows/plan-phase.md": "cc53a19bfc3831e8", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "4099ef6d0868de60", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "d911a080dc1cece6", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "e9de7a96bbfff261", diff --git a/tests/fixtures/golden-install-parity/cursor.json b/tests/fixtures/golden-install-parity/cursor.json index e887e96a3..093f7ae1e 100644 --- a/tests/fixtures/golden-install-parity/cursor.json +++ b/tests/fixtures/golden-install-parity/cursor.json @@ -298,7 +298,7 @@ "gsd-core/workflows/docs-update.md": "b05a59f3079f6fa7", "gsd-core/workflows/edit-phase.md": "a3ab51739c5c021c", "gsd-core/workflows/eval-review.md": "8cc5788e32c15783", - "gsd-core/workflows/execute-phase.md": "c90b217abff9bbb3", + "gsd-core/workflows/execute-phase.md": "9ef2d83a416994dd", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "c133828cf177a772", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "abd2aca069c04a80", @@ -334,7 +334,7 @@ "gsd-core/workflows/note.md": "1c1e466c764e3deb", "gsd-core/workflows/pause-work.md": "0be71264eafd16dc", "gsd-core/workflows/plan-milestone-gaps.md": "1976bf2001969719", - "gsd-core/workflows/plan-phase.md": "ba95020c3495b191", + "gsd-core/workflows/plan-phase.md": "f1f89fbd2bb723fb", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "e06ccd4d4c0703fb", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "d64097345ffac40d", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "3bed01c3c906ac52", diff --git a/tests/fixtures/golden-install-parity/hermes.json b/tests/fixtures/golden-install-parity/hermes.json index e01911fb4..9a3096f10 100644 --- a/tests/fixtures/golden-install-parity/hermes.json +++ b/tests/fixtures/golden-install-parity/hermes.json @@ -228,7 +228,7 @@ "gsd-core/workflows/docs-update.md": "c9df0f7df1ebec79", "gsd-core/workflows/edit-phase.md": "57d807e21fe70355", "gsd-core/workflows/eval-review.md": "d16eebac88386c05", - "gsd-core/workflows/execute-phase.md": "0167a319ed9a5bc5", + "gsd-core/workflows/execute-phase.md": "f05a000ab97b5b25", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "b890bafa4a0c8bd3", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "e34415dec69b8432", @@ -264,7 +264,7 @@ "gsd-core/workflows/note.md": "42b66686b2c102cb", "gsd-core/workflows/pause-work.md": "2c3abcaa1fa6d2e2", "gsd-core/workflows/plan-milestone-gaps.md": "419744c1191354af", - "gsd-core/workflows/plan-phase.md": "e1dcabd0499a03f9", + "gsd-core/workflows/plan-phase.md": "bdd3aade4649708a", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "4099ef6d0868de60", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "99949122a3bd455e", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "62f8e4f3b475fe5f", diff --git a/tests/fixtures/golden-install-parity/kilo.json b/tests/fixtures/golden-install-parity/kilo.json index bb833dd29..6df66038c 100644 --- a/tests/fixtures/golden-install-parity/kilo.json +++ b/tests/fixtures/golden-install-parity/kilo.json @@ -298,7 +298,7 @@ "gsd-core/workflows/docs-update.md": "87bf2a6b7b6ec9db", "gsd-core/workflows/edit-phase.md": "a3ab51739c5c021c", "gsd-core/workflows/eval-review.md": "f6183650f7fcabf9", - "gsd-core/workflows/execute-phase.md": "5c4b59ee5e04a022", + "gsd-core/workflows/execute-phase.md": "73e6a5c54af797bd", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "c133828cf177a772", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "abd2aca069c04a80", @@ -334,7 +334,7 @@ "gsd-core/workflows/note.md": "f8c2842a2217f776", "gsd-core/workflows/pause-work.md": "8b81699a46ca8e9b", "gsd-core/workflows/plan-milestone-gaps.md": "1976bf2001969719", - "gsd-core/workflows/plan-phase.md": "a7642fb6484cf12e", + "gsd-core/workflows/plan-phase.md": "b58ec889ce1d1707", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "4099ef6d0868de60", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "44a12264f8b61d03", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "e9de7a96bbfff261", diff --git a/tests/fixtures/golden-install-parity/kimi.json b/tests/fixtures/golden-install-parity/kimi.json index 9ec4454ab..43ea7c136 100644 --- a/tests/fixtures/golden-install-parity/kimi.json +++ b/tests/fixtures/golden-install-parity/kimi.json @@ -264,7 +264,7 @@ "gsd-core/workflows/docs-update.md": "806ada831b961116", "gsd-core/workflows/edit-phase.md": "e5624ac6e3f8bef5", "gsd-core/workflows/eval-review.md": "dfcfb4f8ce031fae", - "gsd-core/workflows/execute-phase.md": "444d5f17dfc1437a", + "gsd-core/workflows/execute-phase.md": "fea95b7022cad03f", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "8898e0ea533cc643", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "abd2aca069c04a80", @@ -300,7 +300,7 @@ "gsd-core/workflows/note.md": "5a99eb396c744619", "gsd-core/workflows/pause-work.md": "7bcbdf27ba957c8b", "gsd-core/workflows/plan-milestone-gaps.md": "02fee851c82e3b25", - "gsd-core/workflows/plan-phase.md": "e69e435b47d77070", + "gsd-core/workflows/plan-phase.md": "5ab304e905e4e234", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "b36f77ac7344a072", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "d64097345ffac40d", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "49f58c3f75be3eb5", diff --git a/tests/fixtures/golden-install-parity/opencode.json b/tests/fixtures/golden-install-parity/opencode.json index 0c28a1ef0..ce361ce90 100644 --- a/tests/fixtures/golden-install-parity/opencode.json +++ b/tests/fixtures/golden-install-parity/opencode.json @@ -298,7 +298,7 @@ "gsd-core/workflows/docs-update.md": "93e969c3afb446a0", "gsd-core/workflows/edit-phase.md": "ed948a400a0146c7", "gsd-core/workflows/eval-review.md": "1ba1af74a43c07db", - "gsd-core/workflows/execute-phase.md": "f0ca3839e0d0cf8b", + "gsd-core/workflows/execute-phase.md": "24561c1fbc7b710e", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "a640b093a5a7b028", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "8b42f5df1a2c231f", @@ -334,7 +334,7 @@ "gsd-core/workflows/note.md": "0d1374f2a2257858", "gsd-core/workflows/pause-work.md": "c9f0b8826845dda7", "gsd-core/workflows/plan-milestone-gaps.md": "5ec459734bf7570c", - "gsd-core/workflows/plan-phase.md": "03745f51d63d99b8", + "gsd-core/workflows/plan-phase.md": "f449b7d8f334ebb9", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "4099ef6d0868de60", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "fa7f29c323f4cb57", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "e9de7a96bbfff261", diff --git a/tests/fixtures/golden-install-parity/qwen.json b/tests/fixtures/golden-install-parity/qwen.json index 6a70ce791..b1b07ced4 100644 --- a/tests/fixtures/golden-install-parity/qwen.json +++ b/tests/fixtures/golden-install-parity/qwen.json @@ -228,7 +228,7 @@ "gsd-core/workflows/docs-update.md": "12433c1e3cf9e40e", "gsd-core/workflows/edit-phase.md": "4ba06a05cb29f3b7", "gsd-core/workflows/eval-review.md": "0ab8368c8edf91fb", - "gsd-core/workflows/execute-phase.md": "c739c018da3d2c0d", + "gsd-core/workflows/execute-phase.md": "e79f8e56b5d9ccdb", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "9320986ccf0e8a60", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "4116471249699255", @@ -264,7 +264,7 @@ "gsd-core/workflows/note.md": "42b66686b2c102cb", "gsd-core/workflows/pause-work.md": "54f1c0a9e79e2c59", "gsd-core/workflows/plan-milestone-gaps.md": "1b820f4e70acce86", - "gsd-core/workflows/plan-phase.md": "574fc8008416e6a7", + "gsd-core/workflows/plan-phase.md": "fa7053b515270b1d", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "4099ef6d0868de60", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "451a287826b36227", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "d050d8d551ed1756", diff --git a/tests/fixtures/golden-install-parity/trae.json b/tests/fixtures/golden-install-parity/trae.json index 89594dce5..8b4c7054a 100644 --- a/tests/fixtures/golden-install-parity/trae.json +++ b/tests/fixtures/golden-install-parity/trae.json @@ -228,7 +228,7 @@ "gsd-core/workflows/docs-update.md": "dd1050b32c2dc170", "gsd-core/workflows/edit-phase.md": "38f9c9945073ac8c", "gsd-core/workflows/eval-review.md": "48fdef1cf0e67525", - "gsd-core/workflows/execute-phase.md": "5e36e2287a165064", + "gsd-core/workflows/execute-phase.md": "706a2890f7556367", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "44d46d3e98942efc", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "a2fc97089560ec0a", @@ -264,7 +264,7 @@ "gsd-core/workflows/note.md": "acc9130fb1e94f0b", "gsd-core/workflows/pause-work.md": "09a6b8980f7b771a", "gsd-core/workflows/plan-milestone-gaps.md": "022822b3b6971b75", - "gsd-core/workflows/plan-phase.md": "8eddecda8fcd8765", + "gsd-core/workflows/plan-phase.md": "857f539a33efa3a5", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "e06ccd4d4c0703fb", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "382dac17d70ebfac", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "619946c879f33b9d", diff --git a/tests/fixtures/golden-install-parity/windsurf.json b/tests/fixtures/golden-install-parity/windsurf.json index 1e40b993c..f750157b7 100644 --- a/tests/fixtures/golden-install-parity/windsurf.json +++ b/tests/fixtures/golden-install-parity/windsurf.json @@ -228,7 +228,7 @@ "gsd-core/workflows/docs-update.md": "a3c2ec2c856aaabc", "gsd-core/workflows/edit-phase.md": "1666ca537fbef030", "gsd-core/workflows/eval-review.md": "bb01b3300db963bc", - "gsd-core/workflows/execute-phase.md": "20063847abe7ce75", + "gsd-core/workflows/execute-phase.md": "1bd9bf34f86de403", "gsd-core/workflows/execute-phase/steps/codebase-drift-gate.md": "579a3dcf8d69de31", "gsd-core/workflows/execute-phase/steps/per-plan-worktree-gate.md": "7ebb7d1af6082028", "gsd-core/workflows/execute-phase/steps/post-merge-gate.md": "bbd704d6b6f0787b", @@ -264,7 +264,7 @@ "gsd-core/workflows/note.md": "1c1e466c764e3deb", "gsd-core/workflows/pause-work.md": "5ce6a137bd9fc0f8", "gsd-core/workflows/plan-milestone-gaps.md": "19911e87fd4185f2", - "gsd-core/workflows/plan-phase.md": "f6e8c5e38d5c5a52", + "gsd-core/workflows/plan-phase.md": "20062e8d37f19e3d", "gsd-core/workflows/plan-phase/steps/closed-phase-gate.md": "e06ccd4d4c0703fb", "gsd-core/workflows/plan-phase/steps/prd-express-path.md": "7c2be59e3e19299a", "gsd-core/workflows/plan-phase/steps/windows-troubleshooting.md": "3fed4740a91d0443", diff --git a/tests/gate-predicate-evaluator.test.cjs b/tests/gate-predicate-evaluator.test.cjs new file mode 100644 index 000000000..d3e2579a1 --- /dev/null +++ b/tests/gate-predicate-evaluator.test.cjs @@ -0,0 +1,335 @@ +'use strict'; + +/** + * Unit tests for the gate-predicate evaluator (pure core) — issue #2008. + * + * The evaluator is a pure, deps-injected function: + * evaluatePredicate(predicate, context, deps) -> { block, message, details? } + * Malformed predicates / unknown kinds THROW (the CLI wrapper maps a throw to a + * check-command failure so the workflow's `onError` step-1 contract applies). + * + * Built-in kind: `command-exit-zero` — runs a declared command via an injected + * bounded-shell seam; exit 0 => pass, non-zero => block, timeout => block. + * + * Per RULESET.TESTS.boundary-coverage: exit code boundary (0/1/2), timedOut + * true/false, message-trim boundary (MAX / MAX+1), timeout pass-through. + * Per RULESET.TESTS.property-based: interpolation bijection property (fast-check). + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fc = require('fast-check'); + +const { + evaluatePredicate, + evaluateCommandExitZero, + interpolate, + COMMAND_EXIT_ZERO_DEFAULT_TIMEOUT_MS, + COMMAND_MAX_OUTPUT_CHARS, + COMMAND_MAX_LENGTH, + EVALUATOR_KINDS, +} = require('../gsd-core/bin/lib/gate-predicate-evaluator.cjs'); + +// ─── Fake bounded-shell seam ────────────────────────────────────────────────── + +/** Build a fake runBoundedShell that records the invocation and returns a preset result. */ +function fakeShell(preset) { + const calls = []; + const run = (opts) => { + calls.push(opts); + return { + exitCode: preset.exitCode ?? 0, + stdout: preset.stdout ?? '', + stderr: preset.stderr ?? '', + signal: preset.signal ?? null, + timedOut: preset.timedOut ?? false, + }; + }; + return { run, calls }; +} + +const baseCtx = { cwd: '/proj', phaseNumber: '03', phaseDir: '/proj/.planning/phases/03-x', phaseReqIds: 'R-1' }; + +// ─── command-exit-zero: exit-code boundary (0 / 1 / 2) ───────────────────────── + +describe('evaluatePredicate — command-exit-zero exit mapping', () => { + test('exit 0 => block:false (gate passes)', () => { + const shell = fakeShell({ exitCode: 0 }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'true' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.block, false); + assert.match(res.message, /exit/i); + }); + + test('exit 1 => block:true', () => { + const shell = fakeShell({ exitCode: 1, stderr: 'boom' }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'false' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.block, true); + assert.match(res.message, /1/); + assert.match(res.message, /boom/); + }); + + test('exit 2 => block:true (boundary above 1)', () => { + const shell = fakeShell({ exitCode: 2 }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'bad' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.block, true); + }); + + test('exit 127 (ENOENT) => block:true, surfaces not-found', () => { + const shell = fakeShell({ exitCode: 127, stderr: 'sh: not found' }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'nope' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.block, true); + assert.match(res.message, /not found|127/); + }); + + test('non-zero exit with empty stderr still yields a block message', () => { + const shell = fakeShell({ exitCode: 3, stderr: '', stdout: '' }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'x' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.block, true); + assert.match(res.message, /3/); + }); +}); + +// ─── timeout ────────────────────────────────────────────────────────────────── + +describe('evaluatePredicate — command-exit-zero timeout', () => { + test('timedOut => block:true with timed-out message', () => { + const shell = fakeShell({ timedOut: true, exitCode: null, signal: 'SIGTERM' }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'sleep 100', timeout: 5 }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.block, true); + assert.match(res.message, /timed out|timeout/i); + }); + + test('default timeout applied when predicate.timeout absent', () => { + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: 'true' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(shell.calls[0].timeoutMs, COMMAND_EXIT_ZERO_DEFAULT_TIMEOUT_MS); + }); + + test('custom timeout (seconds) honored and converted to ms', () => { + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: 'true', timeout: 90 }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(shell.calls[0].timeoutMs, 90_000); + }); +}); + +// ─── interpolation ──────────────────────────────────────────────────────────── + +describe('evaluatePredicate — interpolation', () => { + test('${PHASE_DIR}, ${PHASE_NUMBER}, ${PHASE_REQ_IDS} are substituted', () => { + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: 'check ${PHASE_DIR} ${PHASE_NUMBER} ${PHASE_REQ_IDS}' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal( + shell.calls[0].command, + 'check /proj/.planning/phases/03-x 03 R-1', + ); + }); + + test('undefined context var => empty string (no leftover placeholder)', () => { + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: 'check ${PHASE_REQ_IDS}' }, + { cwd: '/proj' }, + { runBoundedShell: shell.run }, + ); + assert.equal(shell.calls[0].command, 'check '); + }); + + test('foreign ${HOME} placeholder left untouched (shell interprets)', () => { + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: 'echo ${HOME}' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(shell.calls[0].command, 'echo ${HOME}'); + }); +}); + +// ─── output trimming ────────────────────────────────────────────────────────── + +describe('evaluatePredicate — message trimming', () => { + test('stderr longer than COMMAND_MAX_OUTPUT_CHARS is trimmed', () => { + const long = 'E'.repeat(COMMAND_MAX_OUTPUT_CHARS + 50); + const shell = fakeShell({ exitCode: 1, stderr: long }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'x' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.block, true); + assert.equal(res.message.length, COMMAND_MAX_OUTPUT_CHARS); + }); + + test('stderr exactly at COMMAND_MAX_OUTPUT_CHARS is not trimmed (boundary)', () => { + const exact = 'E'.repeat(COMMAND_MAX_OUTPUT_CHARS); + const shell = fakeShell({ exitCode: 1, stderr: exact }); + const res = evaluatePredicate( + { kind: 'command-exit-zero', command: 'x' }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.equal(res.message.length, COMMAND_MAX_OUTPUT_CHARS); + }); +}); + +// ─── malformed predicate / fail-closed ──────────────────────────────────────── + +describe('evaluatePredicate — malformed predicate throws (maps to check-cmd failure)', () => { + test('missing command throws', () => { + assert.throws( + () => evaluatePredicate({ kind: 'command-exit-zero' }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /command/i, + ); + }); + + test('non-string command throws', () => { + assert.throws( + () => evaluatePredicate({ kind: 'command-exit-zero', command: 42 }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /command/i, + ); + }); + + test('empty-string command throws', () => { + assert.throws( + () => evaluatePredicate({ kind: 'command-exit-zero', command: ' ' }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /command/i, + ); + }); + + test('oversized command throws (ARGV-overflow guard)', () => { + const huge = 'a'.repeat(COMMAND_MAX_LENGTH + 1); + assert.throws( + () => evaluatePredicate({ kind: 'command-exit-zero', command: huge }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /max length/i, + ); + }); + + test('non-positive timeout throws', () => { + assert.throws( + () => evaluatePredicate({ kind: 'command-exit-zero', command: 'x', timeout: 0 }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /timeout/i, + ); + assert.throws( + () => evaluatePredicate({ kind: 'command-exit-zero', command: 'x', timeout: -5 }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /timeout/i, + ); + }); + + test('non-finite timeout throws', () => { + assert.throws( + () => evaluatePredicate({ kind: 'command-exit-zero', command: 'x', timeout: 'forever' }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /timeout/i, + ); + }); + + test('unknown kind throws', () => { + assert.throws( + () => evaluatePredicate({ kind: 'no-such-kind', command: 'x' }, baseCtx, { runBoundedShell: fakeShell({}).run }), + /kind|unknown|no-such-kind/i, + ); + }); + + test('predicate not an object throws', () => { + assert.throws( + () => evaluatePredicate('nope', baseCtx, { runBoundedShell: fakeShell({}).run }), + /predicate/i, + ); + }); +}); + +// ─── contract surface ───────────────────────────────────────────────────────── + +describe('evaluatePredicate — exported contract surface', () => { + test('EVALUATOR_KINDS advertises command-exit-zero', () => { + assert.ok(Array.isArray(EVALUATOR_KINDS)); + assert.ok(EVALUATOR_KINDS.includes('command-exit-zero')); + }); + + test('default timeout is 30s', () => { + assert.equal(COMMAND_EXIT_ZERO_DEFAULT_TIMEOUT_MS, 30_000); + }); + + test('cwd is passed through to the shell seam', () => { + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: 'true' }, + { cwd: '/custom/proj' }, + { runBoundedShell: shell.run }, + ); + assert.equal(shell.calls[0].cwd, '/custom/proj'); + }); +}); + +// ─── property-based: interpolation bijection on non-placeholder strings ─────── + +describe('evaluatePredicate — interpolation property (fast-check)', () => { + test('strings without the 3 placeholders pass through unchanged', () => { + fc.assert( + fc.property( + fc.string({ maxLength: 40 }).filter((s) => s.trim().length > 0 && !/\$\{(PHASE_NUMBER|PHASE_DIR|PHASE_REQ_IDS)\}/.test(s)), + (cmd) => { + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: cmd }, + baseCtx, + { runBoundedShell: shell.run }, + ); + // sh -c receives exactly the input; only the 3 known placeholders would have been rewritten. + assert.equal(shell.calls[0].command, cmd); + }), + { numRuns: 100 }, + ); + }); + + test('every placeholder is fully replaced (no leftover ${PHASE_*})', () => { + fc.assert( + fc.property(fc.string({ maxLength: 20 }), (noise) => { + const cmd = `${noise} ${noise}`; + const shell = fakeShell({ exitCode: 0 }); + evaluatePredicate( + { kind: 'command-exit-zero', command: `\${PHASE_DIR}${cmd}\${PHASE_NUMBER}` }, + baseCtx, + { runBoundedShell: shell.run }, + ); + assert.doesNotMatch(shell.calls[0].command, /\$\{PHASE_(DIR|NUMBER|REQ_IDS)\}/); + }), + { numRuns: 100 }, + ); + }); +}); diff --git a/tests/workflow-size-baseline.json b/tests/workflow-size-baseline.json index 6a0fab970..424db6752 100644 --- a/tests/workflow-size-baseline.json +++ b/tests/workflow-size-baseline.json @@ -24,7 +24,7 @@ "docs-update.md": 55662, "edit-phase.md": 12883, "eval-review.md": 9923, - "execute-phase.md": 93515, + "execute-phase.md": 93484, "execute-plan.md": 32611, "explore.md": 10497, "extract-learnings.md": 12849, @@ -52,7 +52,7 @@ "note.md": 6563, "pause-work.md": 14397, "plan-milestone-gaps.md": 11765, - "plan-phase.md": 89775, + "plan-phase.md": 90411, "plan-review-convergence.md": 23468, "plant-seed.md": 11741, "pr-branch.md": 15919,