From 46ba02acde4c36fd025b7f507ead725f892717e9 Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Sun, 26 Jul 2026 01:42:47 -0400 Subject: [PATCH] feat(#2630): phase-estimation module, smart-zone config key, and cli verbs (#2661) * feat(#2630): add phase-estimation module, smart-zone config key, and cli verbs * fix(#2630): document smart_zone_tokens, refresh golden fixtures, fix null-proto property assertions * fix(#2630): align smart_zone_tokens write/read validation and harden estimation tests * chore(#2630): backfill changeset pr to 2661 --- .changeset/fierce-rams-wave.md | 5 + .gitignore | 2 + CONTEXT.md | 7 +- docs/CONFIGURATION.md | 21 +- docs/INVENTORY-MANIFEST.json | 2 + docs/INVENTORY.md | 2 + eslint.config.mjs | 2 + gsd-core/bin/gsd-tools.cjs | 5 + .../bin/shared/config-defaults.manifest.json | 1 + .../bin/shared/config-schema.manifest.json | 1 + gsd-core/references/planning-config.md | 1 + src/config-loader.cts | 1 + src/config.cts | 16 + src/estimate-cli.cts | 158 ++++ src/phase-estimation.cts | 359 +++++++++ .../golden-install-parity/antigravity.json | 8 +- .../golden-install-parity/augment.json | 8 +- .../golden-install-parity/claude-local.json | 8 +- .../golden-install-parity/claude.json | 8 +- .../fixtures/golden-install-parity/cline.json | 8 +- .../golden-install-parity/codebuddy.json | 8 +- .../fixtures/golden-install-parity/codex.json | 8 +- .../golden-install-parity/copilot.json | 8 +- .../golden-install-parity/cursor.json | 8 +- .../golden-install-parity/hermes.json | 8 +- .../fixtures/golden-install-parity/kilo.json | 8 +- .../golden-install-parity/kimi-code.json | 8 +- .../fixtures/golden-install-parity/kimi.json | 8 +- .../golden-install-parity/opencode.json | 8 +- tests/fixtures/golden-install-parity/pi.json | 8 +- .../fixtures/golden-install-parity/qwen.json | 8 +- .../fixtures/golden-install-parity/trae.json | 8 +- .../golden-install-parity/windsurf.json | 8 +- .../fixtures/golden-install-parity/zcode.json | 8 +- tests/phase-estimation.test.cjs | 681 ++++++++++++++++++ 35 files changed, 1319 insertions(+), 97 deletions(-) create mode 100644 .changeset/fierce-rams-wave.md create mode 100644 src/estimate-cli.cts create mode 100644 src/phase-estimation.cts create mode 100644 tests/phase-estimation.test.cjs diff --git a/.changeset/fierce-rams-wave.md b/.changeset/fierce-rams-wave.md new file mode 100644 index 000000000..a8b26602d --- /dev/null +++ b/.changeset/fierce-rams-wave.md @@ -0,0 +1,5 @@ +--- +type: Added +pr: 2661 +--- +**Phase effort estimation against a calibrated smart-zone budget** — plans can now be sized against a configurable token budget (`workflow.smart_zone_tokens`, default 100000) instead of a static heuristic, and the estimate self-corrects against measured reality. Adds the `estimate-check` and `estimate-calibration` query verbs. (#2630) diff --git a/.gitignore b/.gitignore index 53c018a18..c1044d29a 100644 --- a/.gitignore +++ b/.gitignore @@ -184,6 +184,8 @@ build/ /gsd-core/bin/lib/capability-activation.cjs /gsd-core/bin/lib/federated-config.cjs /gsd-core/bin/lib/phase-locator.cjs +/gsd-core/bin/lib/phase-estimation.cjs +/gsd-core/bin/lib/estimate-cli.cjs /gsd-core/bin/lib/roadmap-parser.cjs /gsd-core/bin/lib/drift.cjs /gsd-core/bin/lib/cjs-command-router-adapter.cjs diff --git a/CONTEXT.md b/CONTEXT.md index 24cdbed87..e2adc048d 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -20,6 +20,9 @@ Module owning the pure phase-id parsing and matching helpers: phase-name normali ### Phase Lifecycle Module Module owning phase create, rename, complete, remove, list, and plan-index operations, plus phase-dir prefix validation, STATE.md staleness detection, and auto-prune behaviour. Entry point: `gsd-core/bin/lib/phase.cjs` (CJS surface). Typed phase events: `GSDPhaseStartEvent`, `GSDPhaseStepStartEvent`, `GSDPhaseStepCompleteEvent`, `GSDPhaseCompleteEvent`. (The SDK native-query surface, the `types.ts` event definitions, `phase-runner.ts`, and `phase-prompt.ts` were retired with the SDK package per ADR-0174.) +### Phase Estimation Module +Module owning phase-effort estimation and its calibration against measured reality (ADR-2629, epic #1952). Pure — no I/O, no config reads; the CLI seam (`src/estimate-cli.cts`, verbs `estimate-check` / `estimate-calibration`) owns reading `.planning/config.json` and `.planning/estimation-calibration.json`. Interface: `parseEstimate`/`renderEstimate` (the PLAN.md `estimate: {tokens, tasks, confidence}` block), `parseActuals`/`renderActuals` (the SUMMARY.md `actuals: {tokens, tasks, commits}` block), `deriveConfidence(sampleCount) → low|med|high`, `classifyAgainstBudget(estimate, budget) → {overBudget, ratio, recommendation, budgetValid}`, `computeCalibration(samples) → {factor, sampleCount, applied, confidence, clamped}`, `applyCalibration`, `parseCalibrationDocument`/`renderCalibrationDocument`, and `measureTokens` (a re-export of `prompt-budget`'s `estimateTokens`). **Domain terms: _smart zone_** — the usable prefix of a model's context window before output quality degrades, expressed as the configurable `workflow.smart_zone_tokens` budget (default 100000, a *policy default* rather than a benchmark constant since the effective ceiling is model/task-dependent); **_estimate/actuals_** — a projected phase cost recorded at plan time and the measured cost recorded at completion, both on the **same `estimateTokens` scale** so their ratio measures the miss rather than a difference between two measurement methods. Two invariants: (1) every signal is **exogenous** — the correction routes on a measured actual/estimate ratio and `confidence` routes on a calibration sample count, never on a model's self-assessment (this project measured self-rated confidence and found it weak — `gsd-core/references/honest-verifier.md:25-29`; see `.out-of-scope/general-purpose-agent-prompt-skills.md`); (2) the over-budget flag is **advisory** — a warning plus a split recommendation, never a block. Calibration is median-of-ratios, clamped to `[0.5, 3.0]`, and inert below 3 samples. Source of truth: `gsd-core/bin/lib/phase-estimation.cjs` (generated from `src/phase-estimation.cts`). Test anchor: `tests/phase-estimation.test.cjs`. + ### Verification Module Module owning the canonical phase-verification status projection shared by phase transition, progress, manager, autonomous, and closeout readiness paths. `readVerificationStatus(phaseDir, opts?)` reads the first `*-VERIFICATION.md` frontmatter `status`, maps it through `VERIFICATION_ROUTING_TABLE`, and fail-closes — only `{passed}` satisfies the canonical gate; `missing`/`unknown`/`gaps_found`/`human_needed`/`stale` all route away from "complete" (#1522). `findStaleVerificationSummary` flags a SUMMARY newer than the VERIFICATION file (status `stale`). Both honor a no-throw, degrade-to-safe contract (any FS error → `missing` / not-stale) and an injectable `opts.fs` seam. Source of truth: `gsd-core/bin/lib/verification.cjs` (generated from `src/verification.cts`). @@ -113,7 +116,7 @@ Cross-seam principle (ADR-1411, epic #1411): context resolution — config loadi Diagnostic-output convention for the Resolution Provenance principle (ADR-1411 P3, #1416). Config-interpreting read verbs expose `Resolution { value, configured, reason, warnings }` (`src/resolution.cts`); agent-skills is the first adopter, where `value = { block, skills_count }` and `source`/`degraded` remain config-provenance extras outside the envelope. Other read verbs expose at least `warnings[]` (e.g. capability-state `{ runtimeConfigDir, capabilities, warnings? }`) without `configured`/`reason`, which are meaningful only for config-interpreting verbs. Mutation verbs expose `warnings[]` (advisory) PLUS `errors[]` (operation-not-applied), e.g. capability-writer `{ capabilities, warnings, errors }`. The shared seam across all shapes is `warnings: string[]`; a single generic `Resolution` across read+write verbs was rejected by the deletion test (`configured`/`reason` are meaningless for capability verbs; `errors[]` cannot fold into `warnings[]`) — ADR-1411 P3 amendment. Recurrence prevention is delivered by P4's CI guard (a configured input resolving empty must carry a `reason`), not by a shared envelope. A CI guard (`scripts/lint-resolution-provenance.cjs`, wired into `lint:ci`) enforces that every registered config-interpreting read verb keeps a `configured_empty`/`not_configured` contract test; the registry in that script is the registration point for future verbs (ADR-1411 P4 / #1417). ### Worktree Safety Policy Module -CJS Module owning worktree lifecycle safety policy for the GSD orchestration layer. Interface: `resolveWorktreeContext(cwd, deps) → WorktreeContext` (linked-worktree root mapping), `parseWorktreePorcelain(output) → WorktreeEntry[]` (porcelain parser, skips detached HEAD), `planWorktreePrune(repoRoot, opts, deps) → PrunePlan` (metadata-prune plan, never destructive by default), `executeWorktreePrunePlan(plan, deps) → PruneResult` (executes prune; degrades gracefully on git timeout), `listLinkedWorktreePaths(repoRoot, deps) → LinkedPathsResult`, `inspectWorktreeHealth(repoRoot, opts, deps) → HealthResult` (orphan + stale detection), `snapshotWorktreeInventory(repoRoot, opts, deps) → InventoryResult`, `planWorktreeWaveCleanup(repoRoot, manifest) → CleanupPlan` (manifest-scoped, fail-closed), `executeWorktreeWaveCleanupPlan(plan, deps) → CleanupResult`, `planWorktreeRecordAgent(manifestRaw, fields) → RecordAgentPlan` (write-strict per-agent manifest append; validates each field at write time via the same `normalizeCleanupManifestEntry` rules the reader enforces; fail-closed on a missing/garbled field or a duplicate `(worktree_path, branch)` the reader would dedup away), `cmdWorktreeRecordAgent(cwd, args, deps) → RecordAgentCmdResult` (thin deps-injectable IO wrapper for the `worktree record-agent` verb), `planWorktreeCreate(fields) → WorktreeCreatePlan` (write-strict `worktree create` planner — same missing-field-hint and `normalizeCleanupManifestEntry` validation as `planWorktreeRecordAgent`, pure/no-git), `executeWorktreeCreatePlan(plan, repoRoot, deps) → WorktreeCreateResult` (bounded `git rev-parse --verify` base check THEN `git worktree add -b `; fail-closed `base_unresolved`/`git_timeout`/`worktree_add_failed`; returns `cwd` — the working directory an executor spawn would use), `cmdWorktreeCreate(cwd, args, deps) → WorktreeCreateCmdResult` (CLI verb: plans, creates the worktree, then appends the manifest entry so it is immediately manageable by cleanup-wave/reap-orphans; dedupes by `(worktree_path, branch)`). #2584 ADR-1239 Codex-binding amendment: `worktree create` is the git-worktree-creation primitive for `dispatch.isolation: orchestrator-worktree` hosts, CONSUMED since Phase 3 (#2627) by `execute-phase`'s wave scheduler. Accepts an optional `--root `: when supplied, `--path` must resolve inside it (`path_outside_root`), confining a worktree the orchestrator will spawn a process into; absent the flag, behavior is unchanged from Phase 2. Source of truth: `gsd-core/bin/lib/worktree-safety.cjs`. Timeout path: all git subprocess calls are bounded; callers receive `ok:false, reason:'git_timed_out'` rather than a thrown exception. Test anchor: `tests/worktree-safety.test.cjs`. The `core.cjs` re-export spine was retired in epic #1267: this module absorbed the two thin compositional wrappers that squatted in Core — `resolveWorktreeRoot(cwd, deps)` (a projection over `resolveWorktreeContext`) and `pruneOrphanedWorktrees(...)` (sequences `planWorktreePrune` + `executeWorktreePrunePlan` with a timeout warning) — so callers reach this single worktree-lifecycle seam directly. `gitWorktreeInfoInternal` did NOT move here — worktree-info detection belongs to the Git Query Module. +CJS Module owning worktree lifecycle safety policy for the GSD orchestration layer. Interface: `resolveWorktreeContext(cwd, deps) → WorktreeContext` (linked-worktree root mapping), `parseWorktreePorcelain(output) → WorktreeEntry[]` (porcelain parser, skips detached HEAD), `planWorktreePrune(repoRoot, opts, deps) → PrunePlan` (metadata-prune plan, never destructive by default), `executeWorktreePrunePlan(plan, deps) → PruneResult` (executes prune; degrades gracefully on git timeout), `listLinkedWorktreePaths(repoRoot, deps) → LinkedPathsResult`, `inspectWorktreeHealth(repoRoot, opts, deps) → HealthResult` (orphan + stale detection), `snapshotWorktreeInventory(repoRoot, opts, deps) → InventoryResult`, `planWorktreeWaveCleanup(repoRoot, manifest) → CleanupPlan` (manifest-scoped, fail-closed), `executeWorktreeWaveCleanupPlan(plan, deps) → CleanupResult`, `planWorktreeRecordAgent(manifestRaw, fields) → RecordAgentPlan` (write-strict per-agent manifest append; validates each field at write time via the same `normalizeCleanupManifestEntry` rules the reader enforces; fail-closed on a missing/garbled field or a duplicate `(worktree_path, branch)` the reader would dedup away), `cmdWorktreeRecordAgent(cwd, args, deps) → RecordAgentCmdResult` (thin deps-injectable IO wrapper for the `worktree record-agent` verb), `planWorktreeCreate(fields) → WorktreeCreatePlan` (write-strict `worktree create` planner — same missing-field-hint and `normalizeCleanupManifestEntry` validation as `planWorktreeRecordAgent`, pure/no-git), `executeWorktreeCreatePlan(plan, repoRoot, deps) → WorktreeCreateResult` (bounded `git rev-parse --verify` base check THEN `git worktree add -b `; fail-closed `base_unresolved`/`git_timeout`/`worktree_add_failed`; returns `cwd` — the working directory an executor spawn would use), `cmdWorktreeCreate(cwd, args, deps) → WorktreeCreateCmdResult` (CLI verb: plans, creates the worktree, then appends the manifest entry so it is immediately manageable by cleanup-wave/reap-orphans; dedupes by `(worktree_path, branch)`). #2584 ADR-1239 Codex-binding amendment, Phase 2: `worktree create` is the git-worktree-creation primitive for `dispatch.isolation: orchestrator-worktree` hosts — declared and testable but UNCONSUMED (no scheduler calls it yet; Phase 3 wires it). Source of truth: `gsd-core/bin/lib/worktree-safety.cjs`. Timeout path: all git subprocess calls are bounded; callers receive `ok:false, reason:'git_timed_out'` rather than a thrown exception. Test anchor: `tests/worktree-safety.test.cjs`. The `core.cjs` re-export spine was retired in epic #1267: this module absorbed the two thin compositional wrappers that squatted in Core — `resolveWorktreeRoot(cwd, deps)` (a projection over `resolveWorktreeContext`) and `pruneOrphanedWorktrees(...)` (sequences `planWorktreePrune` + `executeWorktreePrunePlan` with a timeout warning) — so callers reach this single worktree-lifecycle seam directly. `gitWorktreeInfoInternal` did NOT move here — worktree-info detection belongs to the Git Query Module. ### Worktree Lifecycle Module Workflow contract seam covering agent worktree lifecycle orchestration rules. The `worktree_branch_check` block lives in one canonical fragment (`gsd-core/references/worktree-branch-check.md`) that `execute-phase.md`, `quick.md`, `diagnose-issues.md`, and `execute-plan.md` embed at dispatch. Key invariants: `worktree_branch_check` is **verify-only and fail-closed** — the orchestrator owns worktree lifecycle and base recovery, so the sub-agent holds no state-correction primitives; HEAD attachment verified via `git symbolic-ref`; positive allow-list `^worktree-agent-*` enforced; `git update-ref` on protected refs is prohibited; on base mismatch the sub-agent halts with `exit 42` and surfaces to the orchestrator (#48); the orchestrator runs a cwd-drift guard at `execute_waves` entry that resolves the worktree root and refuses drift into an agent worktree (#48); cleanup is manifest-scoped (`WAVE_WORKTREE_MANIFEST`) not global-discovery-based; worktree spawning is sequential (one `run_in_background` at a time to avoid `config.lock` contention). Test anchor: `tests/worktree.test.cjs`. @@ -128,7 +131,7 @@ Module owning bounded, never-throw git repository introspection — the single s Module owning runtime identity normalization at runtime-selection seams. Canonicalizes alias signals from env/config (`GSD_RUNTIME`, `.planning/config.json:runtime`) to supported runtime IDs so output emitters and query runtime gates stay consistent across naming variants (for example `codex-app`/`codex-cli` -> `codex`). Sources: `gsd-core/bin/lib/runtime-name-policy.cjs`, alias manifest `gsd-core/bin/shared/runtime-aliases.manifest.json`. ### Host-Integration Interface -Pure, additive, no-I/O Module owning the versioned, negotiated contract over the six host-integration interface points (command, dispatch, model, hooks, state, artifact) — ADR-1239 Phase A. Extends the ADR-1016 runtime descriptor with nine closed-vocabulary axes carried under `capability.json` `runtime.hostIntegration`: `embeddingMode` (`imperative|declarative`), `commandSurface` (`slash-file|slash-programmatic|slash-toml|palette|prose-only`), `dispatch` (`{namedDispatch,nested,maxDepth,background,backgroundDispatch,subagentToolkit,isolation}`), `modelMode` (`active|passive`), `hookBus` (`host|engine|none`), `stateIO` (`filesystem|sandboxed-storage|session-log-append`), `transport` (`mcp|native-extension`), `runtime` (`node|bun|sandboxed-web|python|go|rust|electron|other`), `effortSurface` (`argv|none` — how reasoning effort reaches the host; ADR-1239 amendment #2481, the first axis whose consumer is an invocation-time argument rather than an install-time artifact). `dispatch.isolation` (`harness-worktree|orchestrator-worktree|none` — how a host isolates concurrent same-wave executors; ADR-1239 Codex-binding amendment #2584; negotiated per-sub-field and CONSUMED since Phase 3 (#2627) by `execute-phase`, which branches on the negotiated value rather than on a runtime id, via the `gsd_run query dispatch-isolation` CLI — the sibling of `dispatch-should-flatten`). `resolveOrchestratorExec(orchestratorExec, cwd, prompt?) → { ok:true, command, args, cwd } | { ok:false, reason }` (#2584 Phase 2, extended in Phase 3 (#2627); pure, no I/O — resolves the `runtime.orchestratorExec` descriptor field, a sibling of `runtime.hostBehaviors` in `capability.json` carrying `{command, args?, cwdFlag?, promptFlag?}`, into the concrete argv/cwd a process-spawn primitive would use for a `dispatch.isolation: orchestrator-worktree` host; appends `[cwdFlag, cwd]` to `args` when `cwdFlag` is a non-empty string, e.g. codex `exec --cd `, opencode `run --dir `, kimi `--work-dir `; when `cwdFlag` is `null`/absent — kimi-code's process-cwd case — no flag is appended and `cwd` alone is returned for the caller to bind via the subprocess's own working-directory option; when `prompt` is supplied it is appended last — behind `promptFlag` when that is a non-empty string (kimi/kimi-code `--prompt

`), otherwise positionally (codex `exec

`, opencode `run

`) — so prompt passing is descriptor data rather than a per-host scheduler branch; omitting `prompt` yields a resolution byte-identical to Phase 2's; fail-closed `missing_command`/`invalid_cwd`/`invalid_args`/`invalid_cwd_flag`/`invalid_prompt_flag`/`invalid_prompt`, plus `unsafe_leading_dash_prompt`/`unsafe_leading_dash_cwd` mirroring the git-argument leading-dash guard in the Worktree Safety Policy Module — a dash-leading positional is parsed by the spawned CLI as a flag. CONSUMED since Phase 3 (#2627)). The harness-side counterpart is the `runtime.harnessIsolationFlag` descriptor string (claude `isolation="worktree"`, cursor `--worktree`), which `dispatch.isolation: harness-worktree` hosts must declare so the scheduler passes a declared token instead of branching on a runtime id; a host declaring `harness-worktree` without it degrades to `none`. Interface: `negotiateHostCapabilities(host, engine?) → { protocolVersion, effective, points, warnings }` enforcing the trust-boundary invariant `effective ⊆ host-declared ∩ engine-known` (never augment with an undeclared or unknown/future-`protocolVersion` value — fail-closed via the most-restrictive-known `SAFE_DEFAULTS`); `degradationFor(point, axes) → { level, fallback }` (a pure Full/Degraded/Absent ladder table, never throws); `profileOf(axes) → 'programmatic-cli'|'declarative-cli'|'ide'|null`; plus `PROTOCOL_VERSION` (integer, starts at 1 — distinct from the package `version`/`engines.gsd` semver), `HOST_INTEGRATION_AXES` (the frozen closed vocabulary, single source of truth), `PROFILE_BASELINES`, and `shouldFlattenDispatch(dispatch) → boolean` (ADR-1239 Phase B / #1708 — graduates the #853 rule: returns `true` = run the orchestrator inline UNLESS the host is documented to background a nesting-capable orchestrator (`background === true && backgroundDispatch === true`); fail-closed to inline; exposed to the plan/execute workflows via the `gsd_run query dispatch-should-flatten --raw` CLI, which replaced the former scattered `RUNTIME === 'codex'` prose check). The runtime-descriptor validator (`gsd-core/bin/lib/capability-validator.cjs` `validateRuntimeBody`) mirrors the closed vocabulary inline (exported as `_HOST_INTEGRATION_VOCAB`) and is kept in lock-step by the parity guard `tests/host-integration-validator-parity.test.cjs`. Orthogonal axes (resolved explicitly per ADR-1239 Phase A): `commandStyle` (GSD emission style, retained) vs `commandSurface` (host surface type); `hookEvents` dialect vs `hookBus` ownership (a host with `hooksSurface:none` may still be `hookBus:host` — e.g. opencode); `runtimeCompat` (feature→host) vs these negotiated runtime→engine axes. Phase A defined the interface; Phase B (#1679) wires it incrementally — `destSubpath` write-confinement (#1704) and the typed documentation-sourced #853 dispatch-flatten (#1708, the first consumer of a negotiated `dispatch` axis); adapters/MCP/host-bindings remain Phases C–E. Source of truth: `gsd-core/bin/lib/host-integration.cjs` (generated from `src/host-integration.cts`). See ADR-1239 and ADR-1016. +Pure, additive, no-I/O Module owning the versioned, negotiated contract over the six host-integration interface points (command, dispatch, model, hooks, state, artifact) — ADR-1239 Phase A. Extends the ADR-1016 runtime descriptor with nine closed-vocabulary axes carried under `capability.json` `runtime.hostIntegration`: `embeddingMode` (`imperative|declarative`), `commandSurface` (`slash-file|slash-programmatic|slash-toml|palette|prose-only`), `dispatch` (`{namedDispatch,nested,maxDepth,background,backgroundDispatch,subagentToolkit,isolation}`), `modelMode` (`active|passive`), `hookBus` (`host|engine|none`), `stateIO` (`filesystem|sandboxed-storage|session-log-append`), `transport` (`mcp|native-extension`), `runtime` (`node|bun|sandboxed-web|python|go|rust|electron|other`), `effortSurface` (`argv|none` — how reasoning effort reaches the host; ADR-1239 amendment #2481, the first axis whose consumer is an invocation-time argument rather than an install-time artifact). `dispatch.isolation` (`harness-worktree|orchestrator-worktree|none` — how a host isolates concurrent same-wave executors; ADR-1239 Codex-binding amendment #2584; declared and negotiated but not yet consumed by any scheduler — Phase 1 of #2584). `resolveOrchestratorExec(orchestratorExec, cwd) → { ok:true, command, args, cwd } | { ok:false, reason }` (#2584 Phase 2, pure, no I/O — resolves the `runtime.orchestratorExec` descriptor field, a sibling of `runtime.hostBehaviors` in `capability.json` carrying `{command, args?, cwdFlag?}`, into the concrete argv/cwd a process-spawn primitive would use for a `dispatch.isolation: orchestrator-worktree` host; appends `[cwdFlag, cwd]` to `args` when `cwdFlag` is a non-empty string, e.g. codex `exec --cd `, opencode `run --dir `, kimi `--work-dir `; when `cwdFlag` is `null`/absent — kimi-code's process-cwd case — no flag is appended and `cwd` alone is returned for the caller to bind via the subprocess's own working-directory option; fail-closed `missing_command`/`invalid_cwd`/`invalid_args`/`invalid_cwd_flag`; declared and testable but UNCONSUMED — no scheduler spawns anything with it yet, Phase 3 wires it). Interface: `negotiateHostCapabilities(host, engine?) → { protocolVersion, effective, points, warnings }` enforcing the trust-boundary invariant `effective ⊆ host-declared ∩ engine-known` (never augment with an undeclared or unknown/future-`protocolVersion` value — fail-closed via the most-restrictive-known `SAFE_DEFAULTS`); `degradationFor(point, axes) → { level, fallback }` (a pure Full/Degraded/Absent ladder table, never throws); `profileOf(axes) → 'programmatic-cli'|'declarative-cli'|'ide'|null`; plus `PROTOCOL_VERSION` (integer, starts at 1 — distinct from the package `version`/`engines.gsd` semver), `HOST_INTEGRATION_AXES` (the frozen closed vocabulary, single source of truth), `PROFILE_BASELINES`, and `shouldFlattenDispatch(dispatch) → boolean` (ADR-1239 Phase B / #1708 — graduates the #853 rule: returns `true` = run the orchestrator inline UNLESS the host is documented to background a nesting-capable orchestrator (`background === true && backgroundDispatch === true`); fail-closed to inline; exposed to the plan/execute workflows via the `gsd_run query dispatch-should-flatten --raw` CLI, which replaced the former scattered `RUNTIME === 'codex'` prose check). The runtime-descriptor validator (`gsd-core/bin/lib/capability-validator.cjs` `validateRuntimeBody`) mirrors the closed vocabulary inline (exported as `_HOST_INTEGRATION_VOCAB`) and is kept in lock-step by the parity guard `tests/host-integration-validator-parity.test.cjs`. Orthogonal axes (resolved explicitly per ADR-1239 Phase A): `commandStyle` (GSD emission style, retained) vs `commandSurface` (host surface type); `hookEvents` dialect vs `hookBus` ownership (a host with `hooksSurface:none` may still be `hookBus:host` — e.g. opencode); `runtimeCompat` (feature→host) vs these negotiated runtime→engine axes. Phase A defined the interface; Phase B (#1679) wires it incrementally — `destSubpath` write-confinement (#1704) and the typed documentation-sourced #853 dispatch-flatten (#1708, the first consumer of a negotiated `dispatch` axis); adapters/MCP/host-bindings remain Phases C–E. Source of truth: `gsd-core/bin/lib/host-integration.cjs` (generated from `src/host-integration.cts`). See ADR-1239 and ADR-1016. ### Statusline Host-integration hook (`hooks/gsd-statusline.js`) that renders the session status line: model name, context-window meter, workspace directory, and the GSD-state segment (`formatGsdState()` projecting `.planning/` STATE.md). Opt-in segments are gated by `.planning/config.json` keys (`statusline.show_last_command`, `statusline.context_position`, plus the approved `statusline.show_context_tokens` and `statusline.state_format`), each registered across `gsd-core/bin/shared/config-schema.manifest.json` + `src/config.cts` + the `loadConfig` whitelist + `docs/CONFIGURATION.md`. The compact GSD-state format consumes the canonical status vocabulary from `normalizeStateStatus()` (STATE.md Document Module) rather than a parallel keyword list. **Data-source boundary (ADR-2164):** the statusline sources only local, read-only data — it refines the stdin payload Claude Code already sends and may add a new *local* source (e.g. `git`), but does not read credentials or call external/network APIs for data; account/usage/platform-level state is out of scope. diff --git a/docs/CONFIGURATION.md b/docs/CONFIGURATION.md index 4d77d2056..bf9e3944e 100644 --- a/docs/CONFIGURATION.md +++ b/docs/CONFIGURATION.md @@ -301,12 +301,13 @@ All workflow toggles follow the **absent = enabled** pattern. If a key is missin | `workflow.ui_review` | boolean | `true` | Run visual quality audit (`/gsd-ui-review`) after phase execution in autonomous mode. When `false`, the UI audit step is skipped. | | `workflow.node_repair` | boolean | `true` | Autonomous task repair on verification failure | | `workflow.node_repair_budget` | number | `2` | Max repair attempts per failed task | +| `workflow.smart_zone_tokens` | number | `100000` | Smart-zone token budget for phase-effort estimation (#2630, [ADR-2629](adr/2629-phase-effort-estimation-calibration.md)). A phase whose estimate exceeds this is flagged with a split recommendation — **advisory only, never a block**. This is a *policy default, not a benchmark constant*: LLM output quality degrades before the advertised context window is full, but the effective ceiling is model-, task-, and distractor-dependent, so no universal number exists. Lower it for models that degrade early; the estimate-vs-actual calibration loop corrects the figure per project over time. Must be a positive integer. | | `workflow.research_before_questions` | boolean | `false` | Run research before discussion questions instead of after | | `workflow.discuss_mode` | string | `'discuss'` | Controls how `/gsd-discuss-phase` gathers context. `'discuss'` (default) asks questions one-by-one. `'assumptions'` reads the codebase first, generates structured assumptions with confidence levels, and only asks you to correct what's wrong. Added in v1.28 | | `workflow.max_discuss_passes` | number | `3` | Maximum number of question rounds in discuss-phase before the workflow stops asking. Useful in headless/auto mode to prevent infinite discussion loops. | | `workflow.skip_discuss` | boolean | `false` | When `true`, `/gsd-autonomous` bypasses the discuss-phase entirely, writing minimal CONTEXT.md from the ROADMAP phase goal. Useful for projects where developer preferences are fully captured in PROJECT.md/REQUIREMENTS.md. Added in v1.28 | | `workflow.text_mode` | boolean | `false` | Replaces AskUserQuestion TUI menus with plain-text numbered lists. Required for Claude Code remote sessions (`/rc` mode) where TUI menus don't render. Can also be set per-session with `--text` flag on discuss-phase. Added in v1.28 | -| `workflow.use_worktrees` | boolean | `true` | When `false`, disables git worktree isolation for parallel execution. Users who prefer sequential execution or whose environment does not support worktrees can disable this. Added in v1.31. **Branch-divergence note:** when your branch has diverged from `origin/HEAD`, GSD auto-degrades to sequential and prints a warning. See [`worktree.baseRef`](#worktree-settings) to restore parallel execution on a diverged branch. **Per-runtime note:** whether this key can be honored depends on the runtime's declared `dispatch.isolation` capability, not on its name (#2584). Three cases: runtimes whose own harness isolates each executor (**Claude Code**, **Cursor**) run parallel worktrees natively; runtimes that expose a headless exec with an explicit working directory (**Codex**, **OpenCode**, **Kimi**, **Kimi Code**) get parallel worktrees that GSD itself creates, validates and merges; every other runtime declares no isolation primitive, and forcing `use_worktrees: true` there still fails closed before any executor dispatch (#1515, #1521). See [Executor isolation per runtime](#executor-isolation-per-runtime). | +| `workflow.use_worktrees` | boolean | `true` | When `false`, disables git worktree isolation for parallel execution. Users who prefer sequential execution or whose environment does not support worktrees can disable this. Added in v1.31. **Branch-divergence note:** when your branch has diverged from `origin/HEAD`, GSD auto-degrades to sequential and prints a warning. See [`worktree.baseRef`](#worktree-settings) to restore parallel execution on a diverged branch. **Non-Claude note:** git worktree isolation uses Claude Code's `isolation="worktree"` agent primitive, which no other runtime honors. On any non-Claude install (Codex, Cursor, Antigravity, Qwen, etc.) a runtime-neutral `.planning/config.json` resolves the runtime to that install's own id and defaults this key to `false`; forcing `use_worktrees: true` on a non-Claude install fails closed before any executor dispatch (#1515, #1521). | | `workflow.worktree_skip_hooks` | boolean | `false` | When `true`, executor agents in worktree mode pass `--no-verify` (skipping pre-commit hooks) and post-wave hook validation runs against the merged result instead. Opt-in escape hatch for projects whose hooks cannot run in agent worktrees. Default `false` runs hooks on every commit (#2924). | | `workflow.code_review` | boolean | `true` | Enable `/gsd-code-review` and `/gsd-code-review --fix` commands. When `false`, the commands exit with a configuration gate message. Added in v1.34 | | `workflow.code_review_depth` | string | `standard` | Default review depth for `/gsd-code-review`: `quick` (pattern-matching only), `standard` (per-file analysis), or `deep` (cross-file with import graphs). Can be overridden per-run with `--depth=`. Added in v1.34 | @@ -347,24 +348,6 @@ All workflow toggles follow the **absent = enabled** pattern. If a key is missin |---------|------|---------|-------------| | `worktree.baseRef` | string | (unset) | Controls which ref the worktree-based parallel executor uses as the base when creating new phase/wave worktrees. When unset, the executor bases new worktrees on the repository default branch (`origin/HEAD`); if the current branch has diverged, execute-phase auto-degrades to sequential execution rather than halting (as of v1.4.0). Set to `"head"` to base new worktrees on the local `HEAD` instead — the appropriate choice when working on a branch that has diverged from the default branch, as it prevents the exit-42 base-mismatch halt and allows wave-based parallel execution to proceed normally. See [Fix the worktree base-mismatch (exit 42) error](how-to/fix-worktree-base-mismatch.md). | -### Executor isolation per runtime - -When `/gsd-execute-phase` runs a wave containing several independent plans, it can execute them concurrently — but only if the runtime can keep each executor isolated. Two executors sharing one checkout race on files, git state, hooks, and `.planning/`. Which runtimes can do this is a **declared capability** (`dispatch.isolation`), not a hardcoded list, so the scheduler behaves the same way for every host that declares the same value. - -| Isolation | Runtimes | What happens | -|---|---|---| -| `harness-worktree` | `claude`, `cursor` | The runtime's own harness creates and binds a git worktree per executor. GSD passes the host's isolation flag and runs no git itself. | -| `orchestrator-worktree` | `codex`, `opencode`, `kimi`, `kimi-code` | The runtime has no harness-native isolation, but exposes a headless exec that accepts a working directory. **GSD** creates the worktree, spawns each executor into it, then validates and merges the result. All git operations are performed by GSD, never by the sandboxed executor. | -| `none` | every other runtime | No isolation primitive — plans in a wave run sequentially. Setting `workflow.use_worktrees: true` here fails closed before any executor is dispatched. | - -You do not configure this directly: set `workflow.use_worktrees` and GSD negotiates the rest. `use_worktrees: false` forces sequential execution on **every** runtime, including the ones that support isolation. An unknown or undeclared isolation value always degrades to sequential — GSD never guesses its way into an unisolated parallel run. - -To see what your current runtime negotiated: - -```bash -gsd-tools query dispatch-isolation --json -``` - ## Code Quality Settings The `code_quality.*` namespace gates optional structural-analysis tooling that augments `/gsd-code-review`. Settings are additive: each tool is independently opt-in and off by default. diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index c22cf2011..af42580b2 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -354,6 +354,7 @@ "drift.cjs", "edge-probe.cjs", "embedding-adapter.cjs", + "estimate-cli.cjs", "eval-command-router.cjs", "eval.cjs", "external-descriptor-trust.cjs", @@ -399,6 +400,7 @@ "package-identity.cjs", "package-legitimacy.cjs", "phase-command-router.cjs", + "phase-estimation.cjs", "phase-id.cjs", "phase-lifecycle.cjs", "phase-locator.cjs", diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index 42c1357e1..e379a1227 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -456,6 +456,7 @@ Full listing: `gsd-core/bin/lib/*.cjs`. | `edge-probe.cjs` | Spec-completeness edge probe (compiled from `src/edge-probe.cts`, gitignored) — the first adapter of the `probe-core` resolution model (ADR-550 Decision 7): shape classification, applicable-category relevance filter, edge proposal, and the `{explicit, backstop}` verification validators; delegates merge/rollup/CLI to `probe-core`; exports `classifyShape`, `applicableCategories`, `proposeEdges`, `analyzeCoverage`, `validateResolution`, `TAXONOMY` (#550) | | `eval-command-router.cjs` | Routes the `eval.score` verb (compiled from `src/eval-command-router.cts`, gitignored) — thin dispatcher into the eval scoring module (#1579) | | `eval.cjs` | Deterministic eval scoring (compiled from `src/eval.cts`, gitignored) — `computeEvalScore` (coverage*0.6 + infra*0.4, bands 80/60/40) + `cmdEvalScore` CLI domain guard; moves the gsd-eval-auditor's weighted arithmetic out of the prompt into code (#10 / #1579) | +| `estimate-cli.cjs` | I/O seam over `phase-estimation.cjs` — the `estimate-check` and `estimate-calibration` query verbs; reads the `workflow.smart_zone_tokens` budget and `.planning/estimation-calibration.json`, both degrading to defaults rather than failing planning (#2630) | | `fallow-runner.cjs` | Fallow audit adapter for `/gsd-code-review`: binary resolution (`PATH` then `node_modules/.bin`), actionable missing-binary errors, and structural findings normalization | | `federated-config.cjs` | Defensive merge of capability-declared config slices into the loadConfig return value — ADR-857 phase 3b; exports `mergeFederatedConfig({ configSchema, isCentralKey, userConfig })` → `{ values, validKeys, warnings }`; live for migrated Capability keys that are atomically removed from the central config schema | | `frontmatter.cjs` | YAML frontmatter CRUD operations | @@ -489,6 +490,7 @@ Full listing: `gsd-core/bin/lib/*.cjs`. | `package-identity.cjs` | Generated single source for GSD's published-package coordinates (npm name, bin name, repo slug, changelog URL, manual-install command), derived from package.json; read by the update worker, `check-latest-version`, and installer (#498) | | `package-legitimacy.cjs` | Registry-API package legitimacy verdicts (OK/SUS/SLOP) from npm/PyPI/crates, slopcheck optional | | `phase-command-router.cjs` | Thin CJS subcommand router adapter for `gsd-tools phase` | +| `phase-estimation.cjs` | Pure phase-effort estimation — `estimate`/`actuals` schema parse+render, smart-zone budget classification, and estimate-vs-actual calibration (median ratio, clamped, sample-gated). Confidence is derived from calibration sample count, never self-rated (ADR-2629) | | `phase-id.cjs` | Pure phase-id parsing/matching helpers — normalize, token match, milestone/phase-dir id parsing, phase-markdown regex builders (extracted from `core.cjs`, ADR-857) | | `phase-lifecycle.cjs` | Pure-computation phase lifecycle helpers extracted from the phase-lifecycle SDK handler | | `phase-locator.cjs` | Phase-directory search/location — active + archived phase-dir discovery, phase-id matching against the filesystem (extracted from `core.cjs`, ADR-857) | diff --git a/eslint.config.mjs b/eslint.config.mjs index 6ad716168..032db8121 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -154,6 +154,8 @@ export default tseslint.config( 'gsd-core/bin/lib/core-utils.cjs', 'gsd-core/bin/lib/io.cjs', 'gsd-core/bin/lib/phase-id.cjs', + 'gsd-core/bin/lib/phase-estimation.cjs', + 'gsd-core/bin/lib/estimate-cli.cjs', 'gsd-core/bin/lib/normalize-test-command.cjs', 'gsd-core/bin/lib/config-loader.cjs', 'gsd-core/bin/lib/phase-locator.cjs', diff --git a/gsd-core/bin/gsd-tools.cjs b/gsd-core/bin/gsd-tools.cjs index fffb2630e..de0d7e197 100755 --- a/gsd-core/bin/gsd-tools.cjs +++ b/gsd-core/bin/gsd-tools.cjs @@ -268,6 +268,7 @@ const roadmap = require('./lib/roadmap.cjs'); const { detectAssumptionDelta } = require('./lib/assumption-delta.cjs'); const verify = require('./lib/verify.cjs'); const config = require('./lib/config.cjs'); +const estimateCli = require('./lib/estimate-cli.cjs'); const template = require('./lib/template.cjs'); const milestone = require('./lib/milestone.cjs'); const commands = require('./lib/commands.cjs'); @@ -2182,6 +2183,10 @@ const HOST_COMMAND_ROUTERS = { 'config-set': routeConfigSet, 'config-set-model-profile': routeConfigSetModelProfile, 'config-get': routeConfigGet, + // Phase-effort estimation (#2630, ADR-2629). A PAIR of verbs, so leaves + // rather than a family — ADR-2346 promotes to a family only at >=3. + 'estimate-check': ({ args, cwd, raw }) => estimateCli.cmdEstimateCheck(cwd, args.slice(1), raw), + 'estimate-calibration': ({ args, cwd, raw }) => estimateCli.cmdEstimateCalibration(cwd, args.slice(1), raw), 'config-new-project': routeConfigNewProject, 'config-path': routeConfigPath, 'migrate-config': routeMigrateConfig, diff --git a/gsd-core/bin/shared/config-defaults.manifest.json b/gsd-core/bin/shared/config-defaults.manifest.json index a373ea3ed..a7bc443fd 100644 --- a/gsd-core/bin/shared/config-defaults.manifest.json +++ b/gsd-core/bin/shared/config-defaults.manifest.json @@ -33,6 +33,7 @@ "_auto_chain_active": false, "node_repair": true, "node_repair_budget": 2, + "smart_zone_tokens": 100000, "ui_phase": true, "ui_safety_gate": true, "text_mode": false, diff --git a/gsd-core/bin/shared/config-schema.manifest.json b/gsd-core/bin/shared/config-schema.manifest.json index 1676e2afd..585389370 100644 --- a/gsd-core/bin/shared/config-schema.manifest.json +++ b/gsd-core/bin/shared/config-schema.manifest.json @@ -19,6 +19,7 @@ "workflow.auto_advance", "workflow.node_repair", "workflow.node_repair_budget", + "workflow.smart_zone_tokens", "workflow.human_verify_mode", "workflow.text_mode", "workflow.research_before_questions", diff --git a/gsd-core/references/planning-config.md b/gsd-core/references/planning-config.md index ac509c862..6e4b89e81 100644 --- a/gsd-core/references/planning-config.md +++ b/gsd-core/references/planning-config.md @@ -255,6 +255,7 @@ Set via `workflow.*` namespace in config.json (e.g., `"workflow": { "research": | `workflow.auto_advance` | boolean | `false` | `true`, `false` | Auto-advance to next phase after completion | | `workflow.node_repair` | boolean | `true` | `true`, `false` | Attempt automatic repair of failed plan nodes | | `workflow.node_repair_budget` | number | `2` | Any positive integer | Max repair retries per failed node | +| `workflow.smart_zone_tokens` | number | `100000` | Any positive integer | Smart-zone token budget for phase-effort estimation (#2630, ADR-2629). A phase whose estimate exceeds this is flagged with a split recommendation — advisory only, never a block. A *policy default*, not a benchmark constant: degradation begins before the advertised context window is full, but the effective ceiling is model- and task-dependent, so the calibration loop corrects it per project. _Alias:_ `smart_zone_tokens` is the flat-key form used in `CONFIG_DEFAULTS`; `workflow.smart_zone_tokens` is the canonical namespaced form. | | `workflow.ai_integration_phase` | boolean | `true` | `true`, `false` | Run /gsd:ai-integration-phase before planning AI system phases | | `workflow.api_coverage_gate` | boolean | `true` | `true`, `false` | Require an explicit API-coverage decision (full-by-default, opt-out-not-opt-in) before a phase that integrates an external API/SDK/service can seal. At plan:pre prompts a COVERAGE.md matrix; at verify:pre a blocking gate fails the seal unless the matrix exists with every non-integrated capability an explicit, reasoned opt-out (#1562) | | `workflow.ui_phase` | boolean | `true` | `true`, `false` | Generate UI-SPEC.md for frontend phases | diff --git a/src/config-loader.cts b/src/config-loader.cts index ebf3181d0..0017e725a 100644 --- a/src/config-loader.cts +++ b/src/config-loader.cts @@ -124,6 +124,7 @@ const CONFIG_DEFAULTS = { security_asvs_level: _getNestedConfigDefault('workflow', 'security_asvs_level'), security_block_on: _getNestedConfigDefault('workflow', 'security_block_on'), post_planning_gaps: _getNestedConfigDefault('workflow', 'post_planning_gaps'), + smart_zone_tokens: _getNestedConfigDefault('workflow', 'smart_zone_tokens'), }; /** diff --git a/src/config.cts b/src/config.cts index 359b643ed..c83d6101d 100644 --- a/src/config.cts +++ b/src/config.cts @@ -89,6 +89,9 @@ const SCHEMA_DEFAULTS: Record = { 'executor.stall_detect_interval_minutes': 5, 'executor.stall_threshold_minutes': 10, 'git.create_tag': true, + // Derived from the defaults manifest rather than restated, so the manifest + // stays the single source of truth for the smart-zone budget (#2630). + 'workflow.smart_zone_tokens': CONFIG_DEFAULTS.smart_zone_tokens, }; /** @@ -773,6 +776,19 @@ function cmdConfigSet(cwd: string, keyPath: string | undefined, value: string | } } + // Smart-zone token budget (#2630, ADR-2629). Same shape as context_window: + // a positive integer token count. A POLICY default, not a benchmark constant. + // Number.isSafeInteger, NOT Number.isInteger: the read side + // (estimate-cli readSmartZoneBudget) accepts only safe integers, so an + // isInteger-only gate would let config-set 'succeed' on a value past 2^53 + // that estimate-check then silently ignores in favour of the default. + // Accept and honour must agree. + if (kp === 'workflow.smart_zone_tokens') { + if (typeof parsedValue !== 'number' || !Number.isSafeInteger(parsedValue) || parsedValue < 1) { + error(`Invalid workflow.smart_zone_tokens '${val}'. Must be a positive integer (token count).`, ERROR_REASON.USAGE); + } + } + // Post-planning gap checker (#2493) if (kp === 'workflow.post_planning_gaps') { if (typeof parsedValue !== 'boolean') { diff --git a/src/estimate-cli.cts b/src/estimate-cli.cts new file mode 100644 index 000000000..a1c5e78f1 --- /dev/null +++ b/src/estimate-cli.cts @@ -0,0 +1,158 @@ +/** + * Estimate CLI — the I/O seam over the pure phase-estimation module. + * + * Epic #1952 Phase 1 (#2630). Design lock: docs/adr/2629-phase-effort-estimation-calibration.md. + * + * `phase-estimation.cts` is pure policy; everything that touches disk or config + * lives here. Two leaf verbs (a pair, so leaves rather than a family per + * ADR-2346's ">=3 subcommands" rule): + * + * gsd-tools query estimate-check --tokens + * gsd-tools query estimate-calibration + * + * Both degrade rather than fail. A missing or corrupt + * `.planning/estimation-calibration.json` yields an inert calibration + * (factor 1, applied false) instead of breaking planning — the file is a disk + * trust boundary that steers planning output, so it is parsed defensively and + * never trusted structurally. + */ + +import fs from 'node:fs'; +import path from 'node:path'; + +// eslint-disable-next-line @typescript-eslint/no-require-imports -- io.cjs is an export= CommonJS module +import io = require('./io.cjs'); +// eslint-disable-next-line @typescript-eslint/no-require-imports -- phase-estimation.cjs is an export= CommonJS module +import estimation = require('./phase-estimation.cjs'); +// eslint-disable-next-line @typescript-eslint/no-require-imports -- planning-workspace.cjs is an export= CommonJS module +import planningWorkspace = require('./planning-workspace.cjs'); +// eslint-disable-next-line @typescript-eslint/no-require-imports -- config-loader.cjs is an export= CommonJS module +import configLoader = require('./config-loader.cjs'); + +const { output, error, ERROR_REASON } = io; +const { planningDir } = planningWorkspace; +const { CONFIG_DEFAULTS } = configLoader; + +/** Filename of the persisted calibration document, written by extract-learnings (Phase 3). */ +export const CALIBRATION_FILENAME = 'estimation-calibration.json'; + +function defaultBudget(): number { + const fromManifest = Number(CONFIG_DEFAULTS.smart_zone_tokens); + return Number.isSafeInteger(fromManifest) && fromManifest > 0 ? fromManifest : 100000; +} + +/** + * Read the configured smart-zone budget, degrading to the manifest default. + * + * Reads config.json directly rather than through the flat loadConfig + * projection: a hand-edited config can hold any value, and this seam must + * validate rather than assume. An out-of-shape value falls back to the default + * instead of propagating NaN into the comparison. + */ +export function readSmartZoneBudget(cwd: string): number { + try { + const configPath = path.join(planningDir(cwd), 'config.json'); + const parsed: unknown = JSON.parse(fs.readFileSync(configPath, 'utf-8')); + if (parsed !== null && typeof parsed === 'object') { + const workflow = (parsed as Record)['workflow']; + if (workflow !== null && typeof workflow === 'object') { + const value = (workflow as Record)['smart_zone_tokens']; + if (typeof value === 'number' && Number.isSafeInteger(value) && value > 0) return value; + } + } + } catch { + // Absent, unreadable, or malformed config — the default is the answer. + } + return defaultBudget(); +} + +/** Read and defensively parse the calibration history. Never throws. */ +export function readCalibrationSamples(cwd: string): ReturnType { + let raw: string; + try { + raw = fs.readFileSync(path.join(planningDir(cwd), CALIBRATION_FILENAME), 'utf-8'); + } catch { + return []; + } + return estimation.parseCalibrationDocument(raw); +} + +/** + * Parse `--tokens `. + * + * Rejects a missing value, an empty/whitespace value, a value that is really + * the next flag, and anything that is not a positive integer. Uses an exact + * digit match rather than Number()/parseInt so that "1; touch x", "1e5", + * "0x10", and " 1 " are all refused — the value reaches us as argv, is never + * shell-interpolated, and must not be coerced into looking valid. + */ +export function parseTokensFlag(args: string[]): number { + const idx = args.indexOf('--tokens'); + if (idx === -1) { + error('Usage: estimate-check --tokens ', ERROR_REASON.USAGE); + } + + const value = args[idx + 1]; + if (value === undefined || value.startsWith('--') || !/^[0-9]+$/.test(value)) { + error( + `Invalid --tokens ${JSON.stringify(value ?? '')}. Must be a positive integer (token count).`, + ERROR_REASON.USAGE, + ); + } + + const parsed = Number(value); + if (!Number.isSafeInteger(parsed) || parsed < 1) { + error( + `Invalid --tokens ${JSON.stringify(value)}. Must be a positive integer (token count).`, + ERROR_REASON.USAGE, + ); + } + return parsed; +} + +/** + * `estimate-check --tokens ` — classify an estimate against the configured + * smart-zone budget, with the current calibration applied. + * + * The advisory contract (ADR-2629 Decision 5): this reports, it never blocks. + * Exit status is 0 whether or not the estimate is over budget; `over_budget` + * in the payload is the signal. + */ +export function cmdEstimateCheck(cwd: string, args: string[], raw: boolean): void { + const rawTokens = parseTokensFlag(args); + const budget = readSmartZoneBudget(cwd); + const calibration = estimation.computeCalibration(readCalibrationSamples(cwd)); + const calibratedTokens = estimation.applyCalibration(rawTokens, calibration.factor); + const classification = estimation.classifyAgainstBudget(calibratedTokens, budget); + + output({ + raw_tokens: rawTokens, + calibrated_tokens: calibratedTokens, + budget, + over_budget: classification.overBudget, + budget_valid: classification.budgetValid, + ratio: Number(classification.ratio.toFixed(4)), + recommendation: classification.recommendation, + confidence: calibration.confidence, + calibration_applied: calibration.applied, + calibration_factor: calibration.factor, + sample_count: calibration.sampleCount, + }, raw); +} + +/** + * `estimate-calibration` — report the current correction factor and the + * history behind it. + */ +export function cmdEstimateCalibration(cwd: string, _args: string[], raw: boolean): void { + const calibration = estimation.computeCalibration(readCalibrationSamples(cwd)); + + output({ + factor: calibration.factor, + applied: calibration.applied, + sample_count: calibration.sampleCount, + confidence: calibration.confidence, + clamped: calibration.clamped, + min_samples: estimation.MIN_CALIBRATION_SAMPLES, + }, raw); +} diff --git a/src/phase-estimation.cts b/src/phase-estimation.cts new file mode 100644 index 000000000..faeadd630 --- /dev/null +++ b/src/phase-estimation.cts @@ -0,0 +1,359 @@ +/** + * Phase Estimation — estimate/actuals schema, smart-zone threshold policy, and + * estimate-vs-actual calibration. + * + * Epic #1952, Phase 1 (#2630). Design lock: docs/adr/2629-phase-effort-estimation-calibration.md. + * + * Pure functions only — no I/O, no config reads. Callers supply the budget and + * the raw calibration document; this module decides policy over them. The CLI + * seam (gsd-tools) owns reading `.planning/config.json` and + * `.planning/estimation-calibration.json`. + * + * Two properties this module exists to preserve, both from ADR-2629: + * + * 1. Every signal is EXOGENOUS. The correction routes on a measured + * actual/estimate ratio; `confidence` routes on a calibration sample + * count. Nothing routes on a model's self-assessment. This project + * measured self-rated confidence and found it weak + * (gsd-core/references/honest-verifier.md:25-29 — "on a true blind spot it + * stays confidently wrong"), which is why deriveConfidence() takes a + * sample count and there is no "how sure are you?" input anywhere here. + * + * 2. Estimate and actual share ONE measurement scale — estimateTokens() from + * prompt-budget. A ratio between two different measurement methods would + * measure the methods, not the miss. measureTokens() below is the single + * re-export so no consumer reaches for a second estimator. + * + * ADR-457 build-at-publish: source here, compiled to + * gsd-core/bin/lib/phase-estimation.cjs (gitignored). + */ + +// eslint-disable-next-line @typescript-eslint/no-require-imports -- prompt-budget.cjs is an export= CommonJS module +import promptBudget = require('./prompt-budget.cjs'); + +const { estimateTokens } = promptBudget; + +/** Confidence in an estimate. DERIVED from calibration sample count — never self-rated. */ +export type Confidence = 'low' | 'med' | 'high'; + +export const CONFIDENCE_VALUES: readonly Confidence[] = Object.freeze(['low', 'med', 'high'] as const); + +/** Below this many calibration samples, no correction is applied (ADR-2629 Decision 4). */ +export const MIN_CALIBRATION_SAMPLES = 3; + +/** Sample-count thresholds for derived confidence (ADR-2629 Decision 1). */ +export const CONFIDENCE_MED_MIN_SAMPLES = 3; +export const CONFIDENCE_HIGH_MIN_SAMPLES = 6; + +/** Correction-factor clamp. Outside this range the estimator is wrong in kind, not degree. */ +export const CALIBRATION_FACTOR_MIN = 0.5; +export const CALIBRATION_FACTOR_MAX = 3.0; + +/** Schema version for the persisted calibration document. */ +export const CALIBRATION_SCHEMA_VERSION = 1; + +export interface PhaseEstimate { + tokens: number; + tasks: number; + confidence: Confidence; +} + +export interface PhaseActuals { + tokens: number; + tasks: number; + commits: number; +} + +export interface BudgetClassification { + /** True only when the estimate strictly exceeds the budget. At the budget exactly, false. */ + overBudget: boolean; + /** estimate / budget. 0 when the budget is unusable. */ + ratio: number; + /** Human-facing split advice. Null unless overBudget. Advisory — never a block. */ + recommendation: string | null; + /** False when the supplied budget was not a positive finite number. */ + budgetValid: boolean; +} + +export interface CalibrationSample { + estimateTokens: number; + actualTokens: number; +} + +export interface CalibrationResult { + /** Multiply a raw estimate by this. Exactly 1 when not applied. */ + factor: number; + /** Count of USABLE samples (both sides positive and finite). */ + sampleCount: number; + /** True once sampleCount >= MIN_CALIBRATION_SAMPLES. */ + applied: boolean; + /** Derived from sampleCount — the same signal, surfaced for the estimate block. */ + confidence: Confidence; + /** True when the median ratio fell outside the clamp and was pinned to a bound. */ + clamped: boolean; +} + +/** + * A positive, finite, safe integer. Rejects NaN, Infinity, negatives, zero, + * non-integers, and anything past MAX_SAFE_INTEGER (where integer arithmetic + * silently stops being exact). + */ +function isPositiveInt(value: unknown): value is number { + return typeof value === 'number' + && Number.isSafeInteger(value) + && value > 0; +} + +function isPositiveFinite(value: unknown): value is number { + return typeof value === 'number' && Number.isFinite(value) && value > 0; +} + +function isConfidence(value: unknown): value is Confidence { + return typeof value === 'string' && (CONFIDENCE_VALUES as readonly string[]).includes(value); +} + +/** + * A usable calibration sample: both sides present, positive, and finite. + * A zero or negative estimate would divide to Infinity or flip the ratio's + * sign, so those are dropped rather than coerced. + */ +function isCalibrationSample(value: unknown): value is CalibrationSample { + if (value === null || typeof value !== 'object' || Array.isArray(value)) return false; + const record = value as Record; + return isPositiveFinite(record['estimateTokens']) && isPositiveFinite(record['actualTokens']); +} + +/** + * Measure text on the canonical scale. The ONE estimator both the estimate and + * the actuals must use — see property 2 in the module header. + */ +export function measureTokens(text: string | null | undefined): number { + return estimateTokens(text); +} + +/** + * Derive confidence from how much measured history backs the estimate. + * + * Exogenous by construction: the input is a count, not a judgment. A non-integer + * or negative count degrades to 'low' rather than throwing — an unusable history + * is exactly the low-confidence case. + */ +export function deriveConfidence(sampleCount: unknown): Confidence { + if (typeof sampleCount !== 'number' || !Number.isFinite(sampleCount) || sampleCount < 0) return 'low'; + if (sampleCount >= CONFIDENCE_HIGH_MIN_SAMPLES) return 'high'; + if (sampleCount >= CONFIDENCE_MED_MIN_SAMPLES) return 'med'; + return 'low'; +} + +/** + * Classify an estimate against the smart-zone budget. + * + * Boundary contract (ADR-2629 Decision 3 + RULESET.TESTS.boundary-coverage.fixtures): + * budget-1 → under, budget → under, budget+1 → over. The comparison is strictly + * greater-than, so landing exactly on the budget is not a violation. + * + * An unusable budget (hand-edited config, missing key) never fabricates a + * violation: it reports budgetValid=false and overBudget=false, so a broken + * config cannot spam split recommendations. + */ +export function classifyAgainstBudget(estimate: unknown, budget: unknown): BudgetClassification { + if (!isPositiveFinite(budget) || !isPositiveFinite(estimate)) { + return { overBudget: false, ratio: 0, recommendation: null, budgetValid: isPositiveFinite(budget) }; + } + + const ratio = estimate / budget; + if (estimate <= budget) { + return { overBudget: false, ratio, recommendation: null, budgetValid: true }; + } + + const slices = Math.ceil(ratio); + return { + overBudget: true, + ratio, + recommendation: + `Estimated ${estimate} tokens exceeds the ${budget}-token smart-zone budget ` + + `(${ratio.toFixed(2)}x). Consider splitting this phase into about ${slices} ` + + `slices — a tracer plus ${slices - 1} expansion slice(s) — so each runs inside the budget.`, + budgetValid: true, + }; +} + +/** Median of a non-empty numeric array. Caller guarantees non-empty. */ +function median(sorted: number[]): number { + const mid = Math.floor(sorted.length / 2); + if (sorted.length % 2 === 1) return sorted[mid]; + return (sorted[mid - 1] + sorted[mid]) / 2; +} + +/** + * Compute the correction factor from estimate/actual history. + * + * Median, not mean — one pathological phase (an aborted run, a mass rename) + * must not swing every later projection. Clamped, because a ratio outside + * [0.5, 3.0] means the estimator is wrong in kind and amplifying it would make + * the next estimate worse, not better. + * + * Samples missing either side, or carrying a non-positive/non-finite value, are + * dropped rather than coerced — a zero estimate would divide to Infinity. + */ +export function computeCalibration(samples: unknown): CalibrationResult { + const candidates: unknown[] = Array.isArray(samples) ? samples : []; + const usable = candidates.filter(isCalibrationSample); + + const sampleCount = usable.length; + const confidence = deriveConfidence(sampleCount); + + if (sampleCount < MIN_CALIBRATION_SAMPLES) { + return { factor: 1, sampleCount, applied: false, confidence, clamped: false }; + } + + const ratios = usable.map((s) => s.actualTokens / s.estimateTokens).sort((a, b) => a - b); + const raw = median(ratios); + const factor = Math.min(CALIBRATION_FACTOR_MAX, Math.max(CALIBRATION_FACTOR_MIN, raw)); + + return { factor, sampleCount, applied: true, confidence, clamped: factor !== raw }; +} + +/** + * Apply a correction factor to a raw estimate. Rounds to an integer because + * `estimate.tokens` is an integer field; floors at 1 so a heavy shrink factor + * can never produce a zero-token estimate. + */ +export function applyCalibration(rawTokens: unknown, factor: unknown): number { + if (!isPositiveFinite(rawTokens)) return 0; + if (!isPositiveFinite(factor)) return Math.max(1, Math.round(rawTokens)); + // Bound the product: an inexact float past MAX_SAFE_INTEGER would masquerade + // as an integer token count. Unreachable through today's CLI (which is + // safe-integer bounded) but the function is exported and must not depend on + // its caller for that guarantee. + const scaled = Math.round(rawTokens * factor); + return Math.min(Number.MAX_SAFE_INTEGER, Math.max(1, scaled)); +} + +/** Pull the `estimate:` mapping out of an already-parsed frontmatter object. */ +function estimateBlockOf(input: unknown): unknown { + if (input === null || typeof input !== 'object') return null; + const record = input as Record; + return Object.prototype.hasOwnProperty.call(record, 'estimate') ? record['estimate'] : record; +} + +/** + * Parse an estimate block. Returns null for anything that is not a complete, + * well-typed estimate — a partial block is not a usable estimate, and silently + * defaulting a missing field would fabricate data the planner never produced. + * + * Accepts either the whole frontmatter object (`{estimate: {...}}`) or the + * estimate mapping itself, so callers need not unwrap. + */ +export function parseEstimate(input: unknown): PhaseEstimate | null { + const block = estimateBlockOf(input); + if (block === null || typeof block !== 'object' || Array.isArray(block)) return null; + + const record = block as Record; + const tokens = record['tokens']; + const tasks = record['tasks']; + const confidence = record['confidence']; + + if (!isPositiveInt(tokens) || !isPositiveInt(tasks) || !isConfidence(confidence)) return null; + + return { tokens, tasks, confidence }; +} + +/** Pull the `actuals:` mapping out of an already-parsed frontmatter object. */ +function actualsBlockOf(input: unknown): unknown { + if (input === null || typeof input !== 'object') return null; + const record = input as Record; + return Object.prototype.hasOwnProperty.call(record, 'actuals') ? record['actuals'] : record; +} + +/** + * Parse an actuals block. `commits` may be 0 — a phase can legitimately record + * zero commits — so it is validated as a non-negative integer while tokens and + * tasks stay strictly positive. + */ +export function parseActuals(input: unknown): PhaseActuals | null { + const block = actualsBlockOf(input); + if (block === null || typeof block !== 'object' || Array.isArray(block)) return null; + + const record = block as Record; + const tokens = record['tokens']; + const tasks = record['tasks']; + const commits = record['commits']; + + if (!isPositiveInt(tokens) || !isPositiveInt(tasks)) return null; + if (typeof commits !== 'number' || !Number.isSafeInteger(commits) || commits < 0) return null; + + return { tokens, tasks, commits }; +} + +/** + * Render an estimate as the YAML block that lands in PLAN.md frontmatter. + * Inverse of parseEstimate over the same value domain — the bijection the + * property test pins. + */ +export function renderEstimate(estimate: PhaseEstimate): string { + return [ + 'estimate:', + ` tokens: ${estimate.tokens}`, + ` tasks: ${estimate.tasks}`, + ` confidence: ${estimate.confidence}`, + ].join('\n'); +} + +/** Render an actuals block for SUMMARY.md frontmatter. Inverse of parseActuals. */ +export function renderActuals(actuals: PhaseActuals): string { + return [ + 'actuals:', + ` tokens: ${actuals.tokens}`, + ` tasks: ${actuals.tasks}`, + ` commits: ${actuals.commits}`, + ].join('\n'); +} + +export interface CalibrationDocument { + schema_version: number; + samples: CalibrationSample[]; +} + +/** + * Parse the persisted calibration document. + * + * This is a trust boundary: the file is on disk, may be hand-edited, and its + * contents steer planning output. Every failure mode degrades to an empty + * sample set rather than throwing or partially trusting — malformed JSON, a + * non-object root, a missing/!== current schema_version, a non-array samples + * field, or individual malformed samples. + * + * A schema_version we do not recognize is refused outright rather than + * best-effort read: a future writer may change the ratio's meaning, and + * misreading it would silently corrupt every subsequent estimate. + */ +export function parseCalibrationDocument(raw: unknown): CalibrationSample[] { + if (typeof raw !== 'string' || raw.trim() === '') return []; + + let parsed: unknown; + try { + parsed = JSON.parse(raw); + } catch { + return []; + } + + if (parsed === null || typeof parsed !== 'object' || Array.isArray(parsed)) return []; + + const doc = parsed as Record; + if (doc['schema_version'] !== CALIBRATION_SCHEMA_VERSION) return []; + if (!Array.isArray(doc['samples'])) return []; + + // Rebuild each sample from its two known fields rather than passing the + // parsed object through — a hostile document cannot smuggle extra keys + // (or a __proto__ payload) into anything downstream. + return (doc['samples'] as unknown[]) + .filter(isCalibrationSample) + .map((s) => ({ estimateTokens: s.estimateTokens, actualTokens: s.actualTokens })); +} + +/** Serialize a calibration document. Inverse of parseCalibrationDocument. */ +export function renderCalibrationDocument(samples: CalibrationSample[]): string { + const doc: CalibrationDocument = { schema_version: CALIBRATION_SCHEMA_VERSION, samples }; + return `${JSON.stringify(doc, null, 2)}\n`; +} diff --git a/tests/fixtures/golden-install-parity/antigravity.json b/tests/fixtures/golden-install-parity/antigravity.json index 7fb37cf67..4253c8bcc 100644 --- a/tests/fixtures/golden-install-parity/antigravity.json +++ b/tests/fixtures/golden-install-parity/antigravity.json @@ -39,10 +39,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "ea841e2865248e74", - "gsd-core/bin/gsd-tools.cjs": "3bc9c80de40245e9", + "gsd-core/bin/gsd-tools.cjs": "b28392bf9a4dc9bb", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "8bc541aabc2e143c", @@ -123,7 +123,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "3294f01933ac111e", + "gsd-core/references/planning-config.md": "a5fa9b0aec6f321d", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/augment.json b/tests/fixtures/golden-install-parity/augment.json index 18e87133a..cc0f460c2 100644 --- a/tests/fixtures/golden-install-parity/augment.json +++ b/tests/fixtures/golden-install-parity/augment.json @@ -110,10 +110,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -194,7 +194,7 @@ "gsd-core/references/planner-reviews.md": "dda0193a0fbd4947", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "b867e826fab06880", + "gsd-core/references/planning-config.md": "ab3b698fe23bab9b", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/claude-local.json b/tests/fixtures/golden-install-parity/claude-local.json index 93070f1ee..5021f5d2d 100644 --- a/tests/fixtures/golden-install-parity/claude-local.json +++ b/tests/fixtures/golden-install-parity/claude-local.json @@ -109,10 +109,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -193,7 +193,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "9bf731bae8d712a6", + "gsd-core/references/planning-config.md": "2dc9cc2ac26ccdcd", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/claude.json b/tests/fixtures/golden-install-parity/claude.json index 02be55ce7..ac584464f 100644 --- a/tests/fixtures/golden-install-parity/claude.json +++ b/tests/fixtures/golden-install-parity/claude.json @@ -38,10 +38,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -122,7 +122,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "9bf731bae8d712a6", + "gsd-core/references/planning-config.md": "2dc9cc2ac26ccdcd", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/cline.json b/tests/fixtures/golden-install-parity/cline.json index fa6653ff0..e1f9c0a7a 100644 --- a/tests/fixtures/golden-install-parity/cline.json +++ b/tests/fixtures/golden-install-parity/cline.json @@ -42,10 +42,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "476aa24e8c4f03cf", - "gsd-core/bin/gsd-tools.cjs": "ab7a26552bae55dc", + "gsd-core/bin/gsd-tools.cjs": "695c409ab2407247", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -126,7 +126,7 @@ "gsd-core/references/planner-reviews.md": "dda0193a0fbd4947", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "35550d812bdcd6ec", + "gsd-core/references/planning-config.md": "fe89ba623a93d84b", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/codebuddy.json b/tests/fixtures/golden-install-parity/codebuddy.json index 03813638b..d0476ea99 100644 --- a/tests/fixtures/golden-install-parity/codebuddy.json +++ b/tests/fixtures/golden-install-parity/codebuddy.json @@ -110,10 +110,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -194,7 +194,7 @@ "gsd-core/references/planner-reviews.md": "dda0193a0fbd4947", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "b867e826fab06880", + "gsd-core/references/planning-config.md": "ab3b698fe23bab9b", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/codex.json b/tests/fixtures/golden-install-parity/codex.json index 05e96afb0..f35e7fc39 100644 --- a/tests/fixtures/golden-install-parity/codex.json +++ b/tests/fixtures/golden-install-parity/codex.json @@ -145,10 +145,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -229,7 +229,7 @@ "gsd-core/references/planner-reviews.md": "7889bfa28e82156b", "gsd-core/references/planner-revision.md": "2ebf1a714d1ec4bf", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "eedc6daa26054994", + "gsd-core/references/planning-config.md": "bf430b3ea2bf44ff", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/copilot.json b/tests/fixtures/golden-install-parity/copilot.json index 991790e8e..b1be9f6d9 100644 --- a/tests/fixtures/golden-install-parity/copilot.json +++ b/tests/fixtures/golden-install-parity/copilot.json @@ -40,10 +40,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "ea841e2865248e74", - "gsd-core/bin/gsd-tools.cjs": "3bc9c80de40245e9", + "gsd-core/bin/gsd-tools.cjs": "b28392bf9a4dc9bb", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "10226e9512dd44bf", @@ -124,7 +124,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "788802092a535772", + "gsd-core/references/planning-config.md": "6ab82d1d7a354521", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/cursor.json b/tests/fixtures/golden-install-parity/cursor.json index 144836b08..d715726f7 100644 --- a/tests/fixtures/golden-install-parity/cursor.json +++ b/tests/fixtures/golden-install-parity/cursor.json @@ -110,10 +110,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "2525f1ae8b086828", - "gsd-core/bin/gsd-tools.cjs": "2ad128ce1ccf606e", + "gsd-core/bin/gsd-tools.cjs": "1b15863444a36838", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -194,7 +194,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "0cda9fb6e0dea14f", + "gsd-core/references/planning-config.md": "1f52183344323a34", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/hermes.json b/tests/fixtures/golden-install-parity/hermes.json index d36fea163..3b35d6288 100644 --- a/tests/fixtures/golden-install-parity/hermes.json +++ b/tests/fixtures/golden-install-parity/hermes.json @@ -39,10 +39,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "3a3409215044af9f", - "gsd-core/bin/gsd-tools.cjs": "c5ee45330e4b4e59", + "gsd-core/bin/gsd-tools.cjs": "fb40ca85a5fc3f9b", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -123,7 +123,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "b2c11e48f913d3e5", + "gsd-core/references/planning-config.md": "207597aadbb12108", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/kilo.json b/tests/fixtures/golden-install-parity/kilo.json index 64dbe7162..36660b43b 100644 --- a/tests/fixtures/golden-install-parity/kilo.json +++ b/tests/fixtures/golden-install-parity/kilo.json @@ -110,10 +110,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -194,7 +194,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "f51934f2fcfa9943", + "gsd-core/references/planning-config.md": "b7a4fb3d8215ac6a", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/kimi-code.json b/tests/fixtures/golden-install-parity/kimi-code.json index 1e0d27bf5..b790b1f49 100644 --- a/tests/fixtures/golden-install-parity/kimi-code.json +++ b/tests/fixtures/golden-install-parity/kimi-code.json @@ -67,10 +67,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -151,7 +151,7 @@ "gsd-core/references/planner-reviews.md": "dda0193a0fbd4947", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "b867e826fab06880", + "gsd-core/references/planning-config.md": "ab3b698fe23bab9b", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/kimi.json b/tests/fixtures/golden-install-parity/kimi.json index fff3470bb..595012cfe 100644 --- a/tests/fixtures/golden-install-parity/kimi.json +++ b/tests/fixtures/golden-install-parity/kimi.json @@ -103,10 +103,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -187,7 +187,7 @@ "gsd-core/references/planner-reviews.md": "dda0193a0fbd4947", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "b867e826fab06880", + "gsd-core/references/planning-config.md": "ab3b698fe23bab9b", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/opencode.json b/tests/fixtures/golden-install-parity/opencode.json index fc36c2881..c9101c645 100644 --- a/tests/fixtures/golden-install-parity/opencode.json +++ b/tests/fixtures/golden-install-parity/opencode.json @@ -110,10 +110,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -194,7 +194,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "f51934f2fcfa9943", + "gsd-core/references/planning-config.md": "b7a4fb3d8215ac6a", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/pi.json b/tests/fixtures/golden-install-parity/pi.json index 234596a62..312e69a47 100644 --- a/tests/fixtures/golden-install-parity/pi.json +++ b/tests/fixtures/golden-install-parity/pi.json @@ -6,10 +6,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -90,7 +90,7 @@ "gsd-core/references/planner-reviews.md": "dda0193a0fbd4947", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "b867e826fab06880", + "gsd-core/references/planning-config.md": "ab3b698fe23bab9b", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/qwen.json b/tests/fixtures/golden-install-parity/qwen.json index f04d01908..d16e80d8e 100644 --- a/tests/fixtures/golden-install-parity/qwen.json +++ b/tests/fixtures/golden-install-parity/qwen.json @@ -39,10 +39,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "6e98d76e955e35a2", - "gsd-core/bin/gsd-tools.cjs": "658a035b209dd82d", + "gsd-core/bin/gsd-tools.cjs": "4cc7b3241d1422b6", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -123,7 +123,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "8491550a179906fd", + "gsd-core/references/planning-config.md": "7eaeb9dc43f0f9f1", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/trae.json b/tests/fixtures/golden-install-parity/trae.json index 62442596c..cb8998036 100644 --- a/tests/fixtures/golden-install-parity/trae.json +++ b/tests/fixtures/golden-install-parity/trae.json @@ -39,10 +39,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "de4627dff103d527", - "gsd-core/bin/gsd-tools.cjs": "4ad9a090f406f549", + "gsd-core/bin/gsd-tools.cjs": "83769f08d1cc0216", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -123,7 +123,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "c14f59545fbfb981", + "gsd-core/references/planning-config.md": "e333d958ae0179dd", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/windsurf.json b/tests/fixtures/golden-install-parity/windsurf.json index 615387375..7915d0bbc 100644 --- a/tests/fixtures/golden-install-parity/windsurf.json +++ b/tests/fixtures/golden-install-parity/windsurf.json @@ -39,10 +39,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "5636ca0b726871b2", - "gsd-core/bin/gsd-tools.cjs": "2afce7fb45247a8b", + "gsd-core/bin/gsd-tools.cjs": "1ef071e09d8edcb6", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -123,7 +123,7 @@ "gsd-core/references/planner-reviews.md": "da39eace09a10743", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "3c961d736c3d9f4f", + "gsd-core/references/planning-config.md": "6afdf982d6b1663e", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/fixtures/golden-install-parity/zcode.json b/tests/fixtures/golden-install-parity/zcode.json index 9673733e8..0793eb61b 100644 --- a/tests/fixtures/golden-install-parity/zcode.json +++ b/tests/fixtures/golden-install-parity/zcode.json @@ -110,10 +110,10 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "77acbef416b9e5f9", + "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", "gsd-core/bin/gsd_run": "62d9b647ede212e6", - "gsd-core/bin/shared/config-defaults.manifest.json": "517e6a7c1e9f4f16", - "gsd-core/bin/shared/config-schema.manifest.json": "2f60b9eaa6cf6bc1", + "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", + "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", "gsd-core/bin/shared/model-catalog.json": "59f3233132969e27", "gsd-core/bin/shared/runtime-aliases.manifest.json": "e14346bdbd0ebd64", "gsd-core/bin/verify-reapply-patches.cjs": "caec5dbce11e3904", @@ -194,7 +194,7 @@ "gsd-core/references/planner-reviews.md": "dda0193a0fbd4947", "gsd-core/references/planner-revision.md": "86ba8a511f081f05", "gsd-core/references/planner-source-audit.md": "7de5bdb07232ce0b", - "gsd-core/references/planning-config.md": "b867e826fab06880", + "gsd-core/references/planning-config.md": "ab3b698fe23bab9b", "gsd-core/references/prohibition-probe-fixtures/01-streak-reminder/expected.json": "f10df472f2846cc6", "gsd-core/references/prohibition-probe-fixtures/02-clean-utility/expected.json": "31e8a781eeffe020", "gsd-core/references/prohibition-probe-fixtures/03-multi-prohibition/expected.json": "70a532a7cc1b6ae8", diff --git a/tests/phase-estimation.test.cjs b/tests/phase-estimation.test.cjs new file mode 100644 index 000000000..dfd632871 --- /dev/null +++ b/tests/phase-estimation.test.cjs @@ -0,0 +1,681 @@ +/** + * Phase estimation — schema, smart-zone threshold policy, and calibration. + * + * Epic #1952 Phase 1 (#2630). Design lock: docs/adr/2629-phase-effort-estimation-calibration.md. + * + * The two invariants worth stating up front, because most of these tests exist + * to defend them: + * + * 1. Confidence is DERIVED from calibration sample count, never self-rated. + * This project measured self-rated confidence and found it weak + * (gsd-core/references/honest-verifier.md:25-29). A future edit that adds + * a "how sure are you?" input should fail here. + * 2. The budget comparison is strictly greater-than, so an estimate landing + * exactly on the budget is NOT a violation. Boundary fixtures at + * limit-1 / limit / limit+1 per RULESET.TESTS.boundary-coverage.fixtures. + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); +const fc = require('fast-check'); + +const { createTempProject, cleanup, runGsdTools } = require('./helpers.cjs'); + +const est = require('../gsd-core/bin/lib/phase-estimation.cjs'); + +// ─── deriveConfidence — exogenous, sample-count driven ────────────────────── + +describe('deriveConfidence', () => { + // Boundary fixtures at both thresholds: limit-1 / limit / limit+1. + test('routes on sample count at the med threshold (2/3/4)', () => { + assert.equal(est.deriveConfidence(2), 'low'); + assert.equal(est.deriveConfidence(3), 'med'); + assert.equal(est.deriveConfidence(4), 'med'); + }); + + test('routes on sample count at the high threshold (5/6/7)', () => { + assert.equal(est.deriveConfidence(5), 'med'); + assert.equal(est.deriveConfidence(6), 'high'); + assert.equal(est.deriveConfidence(7), 'high'); + }); + + test('zero history is low confidence', () => { + assert.equal(est.deriveConfidence(0), 'low'); + }); + + test('unusable counts degrade to low rather than throwing', () => { + for (const bad of [-1, NaN, Infinity, -Infinity, null, undefined, '6', {}, []]) { + assert.equal(est.deriveConfidence(bad), 'low', `${String(bad)} must degrade to low`); + } + }); + + test('only ever returns a declared confidence value', () => { + // f(n) === f(n) would hold for ANY deterministic function, including one + // that always returned 'high'. Constrain the codomain instead. + fc.assert(fc.property(fc.integer({ min: 0, max: 500 }), (n) => { + assert.ok( + est.CONFIDENCE_VALUES.includes(est.deriveConfidence(n)), + `deriveConfidence(${n}) returned a value outside CONFIDENCE_VALUES`, + ); + }), { numRuns: 100, seed: 19520, verbose: true }); + // ...and that every declared value is actually reachable, so the enum and + // the thresholds cannot drift apart. + assert.deepEqual( + [...new Set([0, 3, 6].map((n) => est.deriveConfidence(n)))].sort(), + [...est.CONFIDENCE_VALUES].sort(), + ); + }); + + test('is monotonic — more history never lowers confidence', () => { + const rank = { low: 0, med: 1, high: 2 }; + fc.assert(fc.property( + fc.integer({ min: 0, max: 200 }), + fc.integer({ min: 0, max: 200 }), + (a, b) => { + const [lo, hi] = a <= b ? [a, b] : [b, a]; + assert.ok( + rank[est.deriveConfidence(lo)] <= rank[est.deriveConfidence(hi)], + `confidence dropped going from ${lo} to ${hi} samples`, + ); + }, + ), { numRuns: 200, seed: 19521, verbose: true }); + }); +}); + +// ─── classifyAgainstBudget — the smart-zone threshold ─────────────────────── + +describe('classifyAgainstBudget', () => { + const BUDGET = 100000; + + test('boundary: limit-1 is under, limit is under, limit+1 is over', () => { + assert.equal(est.classifyAgainstBudget(BUDGET - 1, BUDGET).overBudget, false, 'budget-1 must be under'); + assert.equal(est.classifyAgainstBudget(BUDGET, BUDGET).overBudget, false, 'exactly at budget must NOT be a violation'); + assert.equal(est.classifyAgainstBudget(BUDGET + 1, BUDGET).overBudget, true, 'budget+1 must be over'); + }); + + test('under budget carries no recommendation', () => { + const r = est.classifyAgainstBudget(50000, BUDGET); + assert.equal(r.recommendation, null); + assert.equal(r.budgetValid, true); + assert.ok(Math.abs(r.ratio - 0.5) < 1e-9); + }); + + test('over budget recommends a slice count derived from the ratio', () => { + const r = est.classifyAgainstBudget(250000, BUDGET); + assert.equal(r.overBudget, true); + assert.ok(Math.abs(r.ratio - 2.5) < 1e-9); + assert.ok(typeof r.recommendation === 'string' && r.recommendation.length > 0); + // ceil(2.5) === 3 slices. + assert.match(r.recommendation, /\b3\b/); + }); + + test('an unusable budget never fabricates a violation', () => { + for (const bad of [0, -1, NaN, Infinity, null, undefined, '100000', {}]) { + const r = est.classifyAgainstBudget(999999999, bad); + assert.equal(r.overBudget, false, `budget ${String(bad)} must not report a violation`); + assert.equal(r.budgetValid, false, `budget ${String(bad)} must report itself invalid`); + assert.equal(r.recommendation, null); + } + }); + + test('an unusable estimate reports no violation but keeps a valid budget flagged valid', () => { + const r = est.classifyAgainstBudget(NaN, BUDGET); + assert.equal(r.overBudget, false); + assert.equal(r.budgetValid, true, 'the budget was fine; the estimate was not'); + }); + + test('property: overBudget is exactly estimate > budget', () => { + fc.assert(fc.property( + fc.integer({ min: 1, max: 1000000 }), + fc.integer({ min: 1, max: 1000000 }), + (estimate, budget) => { + assert.equal(est.classifyAgainstBudget(estimate, budget).overBudget, estimate > budget); + }, + ), { numRuns: 300, seed: 19522, verbose: true }); + }); +}); + +// ─── computeCalibration — median, clamped, sample-gated ───────────────────── + +const sample = (estimateTokens, actualTokens) => ({ estimateTokens, actualTokens }); + +describe('computeCalibration', () => { + test('boundary: no correction below the minimum sample count (2/3)', () => { + const two = [sample(100, 200), sample(100, 200)]; + const r2 = est.computeCalibration(two); + assert.equal(r2.applied, false, '2 samples must not apply a correction'); + assert.equal(r2.factor, 1, 'unapplied factor must be exactly 1'); + assert.equal(r2.sampleCount, 2); + + const three = [sample(100, 200), sample(100, 200), sample(100, 200)]; + const r3 = est.computeCalibration(three); + assert.equal(r3.applied, true, '3 samples must apply a correction'); + assert.equal(r3.factor, 2, 'median of [2,2,2] is 2'); + }); + + test('a consistently-underestimated project gets a larger future estimate (AC4)', () => { + // Every phase cost roughly twice its estimate. + const history = [sample(50000, 98000), sample(60000, 121000), sample(40000, 82000)]; + const r = est.computeCalibration(history); + + assert.equal(r.applied, true); + assert.ok(r.factor > 1, `expected an upward correction, got ${r.factor}`); + + const raw = 60000; + const corrected = est.applyCalibration(raw, r.factor); + assert.ok(corrected > raw, `calibrated estimate ${corrected} must exceed the raw ${raw}`); + }); + + test('a consistently-overestimated project gets a smaller future estimate', () => { + const history = [sample(100000, 60000), sample(80000, 48000), sample(90000, 54000)]; + const r = est.computeCalibration(history); + + assert.equal(r.applied, true); + assert.ok(r.factor < 1, `expected a downward correction, got ${r.factor}`); + assert.ok(est.applyCalibration(50000, r.factor) < 50000); + }); + + test('median averages the two middle ratios on an even-length history', () => { + // Every explicit-value case elsewhere uses 3 samples (odd), so the + // even-length averaging branch had no fixed-value assertion. + const history = [sample(100, 100), sample(100, 120), sample(100, 140), sample(100, 160)]; + const r = est.computeCalibration(history); + assert.equal(r.sampleCount, 4); + // ratios [1.0, 1.2, 1.4, 1.6] -> median = (1.2 + 1.4) / 2 = 1.3 + assert.ok(Math.abs(r.factor - 1.3) < 1e-9, `expected ~1.3, got ${r.factor}`); + }); + + test('median resists a single pathological outlier', () => { + // Two honest 1.0 phases plus one aborted run that burned 50x. + const history = [sample(100, 100), sample(100, 100), sample(100, 5000)]; + const r = est.computeCalibration(history); + assert.equal(r.factor, 1, 'median must ignore the outlier a mean would chase'); + }); + + test('factor is clamped at both bounds and reports the clamp', () => { + const huge = [sample(100, 100000), sample(100, 100000), sample(100, 100000)]; + const rHigh = est.computeCalibration(huge); + assert.equal(rHigh.factor, est.CALIBRATION_FACTOR_MAX); + assert.equal(rHigh.clamped, true); + + const tiny = [sample(100000, 100), sample(100000, 100), sample(100000, 100)]; + const rLow = est.computeCalibration(tiny); + assert.equal(rLow.factor, est.CALIBRATION_FACTOR_MIN); + assert.equal(rLow.clamped, true); + + const inRange = [sample(100, 150), sample(100, 150), sample(100, 150)]; + assert.equal(est.computeCalibration(inRange).clamped, false); + }); + + test('drops unusable samples instead of coercing them', () => { + const mixed = [ + sample(100, 200), + sample(0, 200), // zero estimate would divide to Infinity + sample(100, 0), + sample(-100, 200), + sample(100, NaN), + sample(100, 200), + null, + 'nope', + { estimateTokens: '100', actualTokens: '200' }, + sample(100, 200), + ]; + const r = est.computeCalibration(mixed); + assert.equal(r.sampleCount, 3, 'only the three well-formed samples count'); + assert.equal(r.applied, true); + assert.equal(r.factor, 2); + }); + + test('non-array input degrades to an empty history', () => { + for (const bad of [null, undefined, 'samples', 42, {}]) { + const r = est.computeCalibration(bad); + assert.equal(r.sampleCount, 0); + assert.equal(r.applied, false); + assert.equal(r.factor, 1); + assert.equal(r.confidence, 'low'); + } + }); + + test('confidence always agrees with deriveConfidence on the usable count', () => { + fc.assert(fc.property( + fc.array(fc.tuple(fc.integer({ min: 1, max: 10000 }), fc.integer({ min: 1, max: 10000 })), { maxLength: 12 }), + (pairs) => { + const samples = pairs.map(([e, a]) => sample(e, a)); + const r = est.computeCalibration(samples); + assert.equal(r.sampleCount, samples.length); + assert.equal(r.confidence, est.deriveConfidence(samples.length)); + }, + ), { numRuns: 200, seed: 19523, verbose: true }); + }); + + test('property: factor always lands inside the clamp', () => { + fc.assert(fc.property( + fc.array(fc.tuple(fc.integer({ min: 1, max: 100000 }), fc.integer({ min: 1, max: 100000 })), { minLength: 3, maxLength: 20 }), + (pairs) => { + const r = est.computeCalibration(pairs.map(([e, a]) => sample(e, a))); + assert.ok(r.factor >= est.CALIBRATION_FACTOR_MIN, `factor ${r.factor} below clamp`); + assert.ok(r.factor <= est.CALIBRATION_FACTOR_MAX, `factor ${r.factor} above clamp`); + }, + ), { numRuns: 300, seed: 19524, verbose: true }); + }); +}); + +describe('applyCalibration', () => { + test('never returns a zero-token estimate', () => { + assert.equal(est.applyCalibration(1, 0.5), 1); + assert.ok(est.applyCalibration(2, est.CALIBRATION_FACTOR_MIN) >= 1); + }); + + test('an unusable factor leaves the estimate intact', () => { + assert.equal(est.applyCalibration(1234, NaN), 1234); + assert.equal(est.applyCalibration(1234, null), 1234); + assert.equal(est.applyCalibration(1234, 0), 1234); + }); + + test('an unusable raw estimate yields 0', () => { + assert.equal(est.applyCalibration(0, 2), 0); + assert.equal(est.applyCalibration(NaN, 2), 0); + }); +}); + +// ─── schema parse/render ─────────────────────────────────────────────────── + +describe('parseEstimate', () => { + test('accepts a whole frontmatter object or the estimate mapping itself', () => { + const expected = { tokens: 60000, tasks: 5, confidence: 'med' }; + assert.deepEqual(est.parseEstimate({ estimate: expected }), expected); + assert.deepEqual(est.parseEstimate(expected), expected); + }); + + test('rejects an incomplete block rather than defaulting a missing field', () => { + assert.equal(est.parseEstimate({ tokens: 100, tasks: 2 }), null, 'missing confidence'); + assert.equal(est.parseEstimate({ tokens: 100, confidence: 'low' }), null, 'missing tasks'); + assert.equal(est.parseEstimate({ tasks: 2, confidence: 'low' }), null, 'missing tokens'); + }); + + test('rejects hostile and malformed values', () => { + const bad = [ + null, undefined, 'estimate', 42, [], + { tokens: 0, tasks: 1, confidence: 'low' }, + { tokens: -5, tasks: 1, confidence: 'low' }, + { tokens: 1.5, tasks: 1, confidence: 'low' }, + { tokens: NaN, tasks: 1, confidence: 'low' }, + { tokens: Infinity, tasks: 1, confidence: 'low' }, + { tokens: Number.MAX_SAFE_INTEGER + 2, tasks: 1, confidence: 'low' }, + { tokens: '60000', tasks: 1, confidence: 'low' }, + { tokens: 100, tasks: 0, confidence: 'low' }, + { tokens: 100, tasks: 1, confidence: 'certain' }, + { tokens: 100, tasks: 1, confidence: '' }, + { tokens: 100, tasks: 1, confidence: 1 }, + { estimate: null }, + { estimate: [] }, + ]; + for (const value of bad) { + assert.equal(est.parseEstimate(value), null, `must reject ${JSON.stringify(value) ?? String(value)}`); + } + }); +}); + +describe('parseActuals', () => { + test('accepts zero commits but not zero tokens or tasks', () => { + assert.deepEqual( + est.parseActuals({ actuals: { tokens: 74000, tasks: 5, commits: 0 } }), + { tokens: 74000, tasks: 5, commits: 0 }, + ); + assert.equal(est.parseActuals({ tokens: 0, tasks: 5, commits: 1 }), null); + assert.equal(est.parseActuals({ tokens: 100, tasks: 0, commits: 1 }), null); + }); + + test('rejects negative or malformed commits', () => { + assert.equal(est.parseActuals({ tokens: 100, tasks: 1, commits: -1 }), null); + assert.equal(est.parseActuals({ tokens: 100, tasks: 1, commits: 1.5 }), null); + assert.equal(est.parseActuals({ tokens: 100, tasks: 1, commits: '3' }), null); + assert.equal(est.parseActuals({ tokens: 100, tasks: 1 }), null); + }); +}); + +describe('estimate/actuals schema disjointness', () => { + // estimateBlockOf/actualsBlockOf fall back to treating the whole record as + // the block when the wrapper key is absent. That is only safe while the two + // schemas require disjoint fields. Pin it: if either schema ever gains the + // other's disambiguator, this fails loudly instead of silently cross-parsing. + test('an actuals block never parses as an estimate, and vice versa', () => { + const actualsBlock = { tokens: 74000, tasks: 5, commits: 7 }; + const estimateBlock = { tokens: 60000, tasks: 5, confidence: 'med' }; + + assert.equal(est.parseEstimate(actualsBlock), null, 'actuals must not parse as an estimate'); + assert.equal(est.parseActuals(estimateBlock), null, 'an estimate must not parse as actuals'); + }); +}); + +describe('estimate/actuals round-trip', () => { + test('property: parseEstimate(renderEstimate(e)) === e', () => { + fc.assert(fc.property( + fc.record({ + tokens: fc.integer({ min: 1, max: 5000000 }), + tasks: fc.integer({ min: 1, max: 200 }), + confidence: fc.constantFrom('low', 'med', 'high'), + }), + (estimate) => { + const rendered = est.renderEstimate(estimate); + // Render emits YAML; parse the scalar lines back into an object the + // parser accepts, proving the rendered text carries every field. + const parsedBack = {}; + for (const line of rendered.split('\n').slice(1)) { + const m = /^ {2}(\w+): (.+)$/.exec(line); + assert.ok(m, `unparseable rendered line: ${line}`); + parsedBack[m[1]] = m[1] === 'confidence' ? m[2] : Number(m[2]); + } + // fast-check's fc.record yields a null-prototype object; deepEqual + // compares prototypes, so rebuild a plain object to compare values. + assert.deepEqual(est.parseEstimate(parsedBack), { + tokens: estimate.tokens, tasks: estimate.tasks, confidence: estimate.confidence, + }); + }, + ), { numRuns: 300, seed: 19525, verbose: true }); + }); + + test('property: parseActuals(renderActuals(a)) === a', () => { + fc.assert(fc.property( + fc.record({ + tokens: fc.integer({ min: 1, max: 5000000 }), + tasks: fc.integer({ min: 1, max: 200 }), + commits: fc.integer({ min: 0, max: 500 }), + }), + (actuals) => { + const rendered = est.renderActuals(actuals); + const parsedBack = {}; + for (const line of rendered.split('\n').slice(1)) { + const m = /^ {2}(\w+): (.+)$/.exec(line); + assert.ok(m, `unparseable rendered line: ${line}`); + parsedBack[m[1]] = Number(m[2]); + } + // Same null-prototype caveat as the estimate round-trip above. + assert.deepEqual(est.parseActuals(parsedBack), { + tokens: actuals.tokens, tasks: actuals.tasks, commits: actuals.commits, + }); + }, + ), { numRuns: 300, seed: 19526, verbose: true }); + }); +}); + +// ─── calibration document — a disk trust boundary ────────────────────────── + +describe('parseCalibrationDocument', () => { + test('reads a well-formed document', () => { + const raw = est.renderCalibrationDocument([sample(100, 200), sample(300, 400)]); + assert.deepEqual(est.parseCalibrationDocument(raw), [sample(100, 200), sample(300, 400)]); + }); + + test('round-trips through render', () => { + fc.assert(fc.property( + fc.array(fc.tuple(fc.integer({ min: 1, max: 100000 }), fc.integer({ min: 1, max: 100000 })), { maxLength: 15 }), + (pairs) => { + const samples = pairs.map(([e, a]) => sample(e, a)); + assert.deepEqual(est.parseCalibrationDocument(est.renderCalibrationDocument(samples)), samples); + }, + ), { numRuns: 200, seed: 19527, verbose: true }); + }); + + test('refuses an unrecognized schema_version outright', () => { + const future = JSON.stringify({ schema_version: 99, samples: [sample(100, 200)] }); + assert.deepEqual(est.parseCalibrationDocument(future), [], + 'a future schema may redefine the ratio — best-effort reading it would corrupt every later estimate'); + + const missing = JSON.stringify({ samples: [sample(100, 200)] }); + assert.deepEqual(est.parseCalibrationDocument(missing), []); + }); + + test('degrades to empty on malformed and hostile input', () => { + const bad = [ + '', ' ', 'not json', '{', '[]', 'null', '"string"', '42', + JSON.stringify({ schema_version: 1 }), + JSON.stringify({ schema_version: 1, samples: 'nope' }), + JSON.stringify({ schema_version: 1, samples: {} }), + JSON.stringify({ schema_version: '1', samples: [] }), + null, undefined, 42, {}, + ]; + for (const value of bad) { + assert.deepEqual(est.parseCalibrationDocument(value), [], + `must degrade to [] for ${String(value).slice(0, 40)}`); + } + }); + + test('drops individually malformed samples but keeps the good ones', () => { + const raw = JSON.stringify({ + schema_version: 1, + samples: [ + sample(100, 200), + { estimateTokens: 0, actualTokens: 5 }, + { estimateTokens: 'x', actualTokens: 5 }, + null, + [1, 2], + sample(300, 400), + ], + }); + assert.deepEqual(est.parseCalibrationDocument(raw), [sample(100, 200), sample(300, 400)]); + }); + + test('a prototype-pollution payload cannot reach Object.prototype', () => { + const raw = JSON.stringify({ + schema_version: 1, + samples: [{ estimateTokens: 100, actualTokens: 200, __proto__: { polluted: true } }], + }); + const parsed = est.parseCalibrationDocument(raw); + assert.equal(parsed.length, 1); + assert.equal({}.polluted, undefined, 'Object.prototype must not be polluted'); + assert.deepEqual(parsed[0], sample(100, 200), 'only the two known fields are carried forward'); + }); +}); + +// ─── measureTokens — one scale for estimate and actual ───────────────────── + +describe('measureTokens', () => { + test('is the same scale prompt-budget uses', () => { + const { estimateTokens } = require('../gsd-core/bin/lib/prompt-budget.cjs'); + fc.assert(fc.property(fc.string({ maxLength: 400 }), (s) => { + assert.equal(est.measureTokens(s), estimateTokens(s), + 'estimate and actuals must share one estimator or the ratio is meaningless'); + }), { numRuns: 200, seed: 19528, verbose: true }); + }); + + test('empty and nullish inputs measure zero', () => { + assert.equal(est.measureTokens(''), 0); + assert.equal(est.measureTokens(null), 0); + assert.equal(est.measureTokens(undefined), 0); + }); +}); + +// ─── config key: workflow.smart_zone_tokens ──────────────────────────────── + +describe('workflow.smart_zone_tokens config key', () => { + test('defaults to 100000 with no config file written', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const r = runGsdTools('query config-get workflow.smart_zone_tokens --raw', tmpDir); + assert.ok(r.success, `config-get should resolve the schema default: ${r.error}`); + assert.equal(String(r.output).trim(), '100000'); + }); + + test('config-set persists an override', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const set = runGsdTools('config-set workflow.smart_zone_tokens 60000', tmpDir); + assert.ok(set.success, `config-set should accept the key: ${set.error}`); + + const configPath = path.join(tmpDir, '.planning', 'config.json'); + const config = JSON.parse(fs.readFileSync(configPath, 'utf-8')); + assert.equal(config.workflow?.smart_zone_tokens, 60000, 'value must be persisted under workflow.'); + + const get = runGsdTools('query config-get workflow.smart_zone_tokens --raw', tmpDir); + assert.equal(String(get.output).trim(), '60000'); + }); + + test('rejects non-positive-integer values', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + for (const bad of ['0', '-1', '1.5', 'abc', 'Infinity', '']) { + const r = runGsdTools(`config-set workflow.smart_zone_tokens ${bad === '' ? '""' : bad}`, tmpDir); + assert.ok(!r.success, `config-set must reject ${JSON.stringify(bad)}`); + } + }); +}); + +// ─── CLI verbs ───────────────────────────────────────────────────────────── + +describe('query estimate-check', () => { + test('reports under-budget against the configured budget', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const r = runGsdTools('query estimate-check --tokens 50000', tmpDir); + assert.ok(r.success, `estimate-check should succeed: ${r.error}`); + const out = JSON.parse(r.output); + assert.equal(out.over_budget, false); + assert.equal(out.budget, 100000); + }); + + test('reports over-budget with a recommendation and honors a configured budget', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + runGsdTools('config-set workflow.smart_zone_tokens 40000', tmpDir); + const r = runGsdTools('query estimate-check --tokens 90000', tmpDir); + assert.ok(r.success, `estimate-check should succeed: ${r.error}`); + + const out = JSON.parse(r.output); + assert.equal(out.budget, 40000); + assert.equal(out.over_budget, true); + assert.ok(typeof out.recommendation === 'string' && out.recommendation.length > 0); + }); + + test('boundary: exactly at the configured budget is not over', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + runGsdTools('config-set workflow.smart_zone_tokens 40000', tmpDir); + const at = JSON.parse(runGsdTools('query estimate-check --tokens 40000', tmpDir).output); + assert.equal(at.over_budget, false); + + const over = JSON.parse(runGsdTools('query estimate-check --tokens 40001', tmpDir).output); + assert.equal(over.over_budget, true); + }); + + test('rejects a missing, empty, or malformed --tokens value', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + for (const args of [ + 'query estimate-check', + 'query estimate-check --tokens', + 'query estimate-check --tokens ""', + 'query estimate-check --tokens abc', + 'query estimate-check --tokens -5', + 'query estimate-check --tokens 0', + ]) { + const r = runGsdTools(args, tmpDir); + assert.ok(!r.success, `must reject: ${args}`); + assert.ok(!/\bat Object\.|\bat Module\./.test(String(r.error ?? '')), 'no stack trace in failure output'); + } + }); + + test('does not shell-interpolate a hostile --tokens value', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const marker = path.join(tmpDir, 'pwned.txt'); + const r = runGsdTools(`query estimate-check --tokens "1; touch ${marker}"`, tmpDir); + assert.ok(!r.success, 'a command-substitution payload must be rejected as a bad number'); + assert.equal(fs.existsSync(marker), false, 'no shell interpolation of attacker-controlled values'); + }); +}); + +describe('query estimate-calibration', () => { + test('reports an inert calibration with no history', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const r = runGsdTools('query estimate-calibration', tmpDir); + assert.ok(r.success, `estimate-calibration should succeed with no history: ${r.error}`); + + const out = JSON.parse(r.output); + assert.equal(out.applied, false); + assert.equal(out.factor, 1); + assert.equal(out.sample_count, 0); + assert.equal(out.confidence, 'low'); + }); + + test('applies a correction once enough history exists', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + fs.writeFileSync( + path.join(tmpDir, '.planning', 'estimation-calibration.json'), + est.renderCalibrationDocument([sample(100, 200), sample(100, 200), sample(100, 200)]), + ); + + const out = JSON.parse(runGsdTools('query estimate-calibration', tmpDir).output); + assert.equal(out.applied, true); + assert.equal(out.factor, 2); + assert.equal(out.sample_count, 3); + assert.equal(out.confidence, 'med'); + }); + + test('degrades to inert on a corrupt calibration file rather than failing planning', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + fs.writeFileSync(path.join(tmpDir, '.planning', 'estimation-calibration.json'), '{ not json'); + + const r = runGsdTools('query estimate-calibration', tmpDir); + assert.ok(r.success, 'a corrupt calibration file must not fail the command'); + const out = JSON.parse(r.output); + assert.equal(out.applied, false); + assert.equal(out.factor, 1); + }); + + test('survives an unreadable calibration file', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const target = path.join(tmpDir, '.planning', 'estimation-calibration.json'); + fs.writeFileSync(target, est.renderCalibrationDocument([sample(100, 200)])); + + // Drives the REAL readCalibrationSamples — an earlier version of this test + // re-implemented the try/catch inline and would have kept passing if the + // production guard were deleted. + const cli = require('../gsd-core/bin/lib/estimate-cli.cjs'); + + // Deterministic IO fault injection: monkeypatch the fs method and restore in + // finally. Never chmod 0o000 — root bypasses mode bits, so the test would + // silently pass with zero coverage in root Docker/CI. + const originalReadFileSync = fs.readFileSync; + let sawInjectedRead = false; + fs.readFileSync = function patched(p, ...rest) { + if (String(p).endsWith('estimation-calibration.json')) { + sawInjectedRead = true; + throw Object.assign(new Error('injected EACCES'), { code: 'EACCES' }); + } + return originalReadFileSync.call(this, p, ...rest); + }; + let samples; + try { + samples = cli.readCalibrationSamples(tmpDir); + } finally { + fs.readFileSync = originalReadFileSync; + } + + assert.equal(sawInjectedRead, true, 'the injected fault must actually have fired'); + assert.deepEqual(samples, [], 'an unreadable file degrades to no history'); + + // And the degraded history must still yield an inert calibration. + const calibration = est.computeCalibration(samples); + assert.equal(calibration.applied, false); + assert.equal(calibration.factor, 1); + }); +});