From bd570618d4b1b80c2450cf625936526578b1d17d Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Sun, 26 Jul 2026 16:28:56 -0400 Subject: [PATCH] feat(#2632): executor actuals and the closed estimate-calibration loop (#2672) * feat(#2632): record executor actuals and close the estimate calibration loop * fix(#2632): calibrate against the raw projection so the loop converges * test(#2632): add closed-loop convergence guard and codify the feedback-loop rule * fix(#2632): pair calibration samples per plan; atomic write; amend adr * chore(#2632): backfill changeset pr to 2672 * fix(#2632): retry renameSync on transient windows errnos and clean up the temp --- .changeset/vivid-cranes-click.md | 5 + CONTEXT.md | 3 +- agents/gsd-executor.md | 11 +- agents/gsd-planner.md | 5 +- ...629-phase-effort-estimation-calibration.md | 4 +- docs/reference/plan-md.md | 2 +- docs/reference/planning-artifacts.md | 11 + gsd-core/bin/gsd-tools.cjs | 1 + gsd-core/templates/summary-minimal.md | 4 + gsd-core/templates/summary-standard.md | 4 + gsd-core/templates/summary.md | 7 + gsd-core/workflows/extract-learnings.md | 21 + src/estimate-cli.cts | 153 ++++++++ src/phase-estimation.cts | 76 +++- tests/agent-size-baseline.json | 4 +- tests/estimate-calibrate.test.cjs | 365 ++++++++++++++++++ tests/estimate-loop-convergence.test.cjs | 146 +++++++ .../golden-install-parity/antigravity.json | 14 +- .../golden-install-parity/augment.json | 14 +- .../golden-install-parity/claude-local.json | 14 +- .../golden-install-parity/claude.json | 14 +- .../fixtures/golden-install-parity/cline.json | 14 +- .../golden-install-parity/codebuddy.json | 14 +- .../fixtures/golden-install-parity/codex.json | 18 +- .../golden-install-parity/copilot.json | 14 +- .../golden-install-parity/cursor.json | 14 +- .../golden-install-parity/hermes.json | 14 +- .../fixtures/golden-install-parity/kilo.json | 14 +- .../golden-install-parity/kimi-code.json | 14 +- .../fixtures/golden-install-parity/kimi.json | 14 +- .../golden-install-parity/opencode.json | 14 +- tests/fixtures/golden-install-parity/pi.json | 10 +- .../fixtures/golden-install-parity/qwen.json | 14 +- .../fixtures/golden-install-parity/trae.json | 14 +- .../golden-install-parity/windsurf.json | 14 +- .../fixtures/golden-install-parity/zcode.json | 14 +- tests/workflow-size-baseline.json | 2 +- 37 files changed, 943 insertions(+), 147 deletions(-) create mode 100644 .changeset/vivid-cranes-click.md create mode 100644 tests/estimate-calibrate.test.cjs create mode 100644 tests/estimate-loop-convergence.test.cjs diff --git a/.changeset/vivid-cranes-click.md b/.changeset/vivid-cranes-click.md new file mode 100644 index 000000000..bd795bfe1 --- /dev/null +++ b/.changeset/vivid-cranes-click.md @@ -0,0 +1,5 @@ +--- +type: Added +pr: 2672 +--- +**Estimates now calibrate against reality** — the executor records what a phase actually cost into SUMMARY.md, and `/gsd:extract-learnings` computes the estimate-vs-actual correction so future plan estimates improve for your project. (#2632) diff --git a/CONTEXT.md b/CONTEXT.md index e2adc048d..5c431a91a 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -21,7 +21,7 @@ Module owning the pure phase-id parsing and matching helpers: phase-name normali Module owning phase create, rename, complete, remove, list, and plan-index operations, plus phase-dir prefix validation, STATE.md staleness detection, and auto-prune behaviour. Entry point: `gsd-core/bin/lib/phase.cjs` (CJS surface). Typed phase events: `GSDPhaseStartEvent`, `GSDPhaseStepStartEvent`, `GSDPhaseStepCompleteEvent`, `GSDPhaseCompleteEvent`. (The SDK native-query surface, the `types.ts` event definitions, `phase-runner.ts`, and `phase-prompt.ts` were retired with the SDK package per ADR-0174.) ### Phase Estimation Module -Module owning phase-effort estimation and its calibration against measured reality (ADR-2629, epic #1952). Pure — no I/O, no config reads; the CLI seam (`src/estimate-cli.cts`, verbs `estimate-check` / `estimate-calibration`) owns reading `.planning/config.json` and `.planning/estimation-calibration.json`. Interface: `parseEstimate`/`renderEstimate` (the PLAN.md `estimate: {tokens, tasks, confidence}` block), `parseActuals`/`renderActuals` (the SUMMARY.md `actuals: {tokens, tasks, commits}` block), `deriveConfidence(sampleCount) → low|med|high`, `classifyAgainstBudget(estimate, budget) → {overBudget, ratio, recommendation, budgetValid}`, `computeCalibration(samples) → {factor, sampleCount, applied, confidence, clamped}`, `applyCalibration`, `parseCalibrationDocument`/`renderCalibrationDocument`, and `measureTokens` (a re-export of `prompt-budget`'s `estimateTokens`). **Domain terms: _smart zone_** — the usable prefix of a model's context window before output quality degrades, expressed as the configurable `workflow.smart_zone_tokens` budget (default 100000, a *policy default* rather than a benchmark constant since the effective ceiling is model/task-dependent); **_estimate/actuals_** — a projected phase cost recorded at plan time and the measured cost recorded at completion, both on the **same `estimateTokens` scale** so their ratio measures the miss rather than a difference between two measurement methods. Two invariants: (1) every signal is **exogenous** — the correction routes on a measured actual/estimate ratio and `confidence` routes on a calibration sample count, never on a model's self-assessment (this project measured self-rated confidence and found it weak — `gsd-core/references/honest-verifier.md:25-29`; see `.out-of-scope/general-purpose-agent-prompt-skills.md`); (2) the over-budget flag is **advisory** — a warning plus a split recommendation, never a block. Calibration is median-of-ratios, clamped to `[0.5, 3.0]`, and inert below 3 samples. Source of truth: `gsd-core/bin/lib/phase-estimation.cjs` (generated from `src/phase-estimation.cts`). Test anchor: `tests/phase-estimation.test.cjs`. +Module owning phase-effort estimation and its calibration against measured reality (ADR-2629, epic #1952). Pure — no I/O, no config reads; the CLI seam (`src/estimate-cli.cts`, verbs `estimate-check` / `estimate-calibration`) owns reading `.planning/config.json` and `.planning/estimation-calibration.json`. Interface: `parseEstimate`/`renderEstimate` (the PLAN.md `estimate: {tokens, tasks, confidence}` block), `parseActuals`/`renderActuals` (the SUMMARY.md `actuals: {tokens, tasks, commits}` block), `deriveConfidence(sampleCount) → low|med|high`, `classifyAgainstBudget(estimate, budget) → {overBudget, ratio, recommendation, budgetValid}`, `computeCalibration(samples) → {factor, sampleCount, applied, confidence, clamped}`, `applyCalibration`, `parseCalibrationDocument`/`renderCalibrationDocument`, `extractFrontmatterBlock` (leading-`---`-anchored scalar-block reader; hand-rolled because core ships no external deps), `calibrationBasis` (returns `estimate.raw_tokens` when present, else `tokens` — calibration must measure actual/raw or the loop un-corrects itself), and `measureTokens` (a re-export of `prompt-budget`'s `estimateTokens`). **Domain terms: _smart zone_** — the usable prefix of a model's context window before output quality degrades, expressed as the configurable `workflow.smart_zone_tokens` budget (default 100000, a *policy default* rather than a benchmark constant since the effective ceiling is model/task-dependent); **_estimate/actuals_** — a projected phase cost recorded at plan time and the measured cost recorded at completion, both on the **same `estimateTokens` scale** so their ratio measures the miss rather than a difference between two measurement methods. Two invariants: (1) every signal is **exogenous** — the correction routes on a measured actual/estimate ratio and `confidence` routes on a calibration sample count, never on a model's self-assessment (this project measured self-rated confidence and found it weak — `gsd-core/references/honest-verifier.md:25-29`; see `.out-of-scope/general-purpose-agent-prompt-skills.md`); (2) the over-budget flag is **advisory** — a warning plus a split recommendation, never a block. Calibration is median-of-ratios, clamped to `[0.5, 3.0]`, and inert below 3 samples. CLI seam verbs: `estimate-check` (classify one figure; `--calibrated` when the input already has the factor applied — omitting it squares the correction), `estimate-calibration` (report the current factor), `estimate-calibrate` (#2632 — pair every completed phase's PLAN `estimate` with its SUMMARY `actuals`, rebuild `.planning/estimation-calibration.json` idempotently, and report the result; this is what closes the loop). Source of truth: `gsd-core/bin/lib/phase-estimation.cjs` (generated from `src/phase-estimation.cts`) and `src/estimate-cli.cts`. Test anchors: `tests/phase-estimation.test.cjs`, `tests/estimate-calibrate.test.cjs`. ### Verification Module Module owning the canonical phase-verification status projection shared by phase transition, progress, manager, autonomous, and closeout readiness paths. `readVerificationStatus(phaseDir, opts?)` reads the first `*-VERIFICATION.md` frontmatter `status`, maps it through `VERIFICATION_ROUTING_TABLE`, and fail-closes — only `{passed}` satisfies the canonical gate; `missing`/`unknown`/`gaps_found`/`human_needed`/`stale` all route away from "complete" (#1522). `findStaleVerificationSummary` flags a SUMMARY newer than the VERIFICATION file (status `stale`). Both honor a no-throw, degrade-to-safe contract (any FS error → `missing` / not-stale) and an injectable `opts.fs` seam. Source of truth: `gsd-core/bin/lib/verification.cjs` (generated from `src/verification.cts`). @@ -451,6 +451,7 @@ The prompt-level data/instruction isolation seam for untrusted web/document ingr `RULESET.TESTS.coderabbit-fix-prefer=behavioral tests (call exported fn, capture JSON, assert typed fields) over source-grep` `RULESET.TESTS.diagnostics=after JSON.parse, assert output shape (Array.isArray(output.phases)) with raw-output-prefix diagnostics before .map() — prevents opaque TypeErrors when CLI output shape changes` `RULESET.TESTS.boundary-coverage=tests MUST exercise inputs at and near the threshold/limit, not only trivial-fit and trivial-overflow; pick inputs where N ∈ {limit-1, limit, limit+1} and where pre-trim/pre-check accumulators ≈ effective limit; "very small" and "very large" inputs alone do not constitute edge-case coverage and routinely miss off-by-one + reservation-accounting bugs` +`RULESET.TESTS.feedback-loop-convergence=when a feature's OUTPUT feeds back into its own INPUT (calibration, retry backoff, adaptive budgets, ratchets, any self-correcting signal), step-wise tests are NOT sufficient evidence of correctness: they assert `given X return Y` while the defect lives in the TRAJECTORY across iterations. Required: a closed-loop test that (a) drives the REAL end-to-end surface — not the pure core alone, since composition bugs live between surfaces — for N >= 2x the loop's window, (b) asserts convergence on the known-true value, (c) asserts the fixed point (an already-correct history must produce NO correction), and (d) asserts boundedness under an adversarial/oscillating history. Two defects shipped past a green ~26,800-test suite in epic #1952 for want of exactly this: calibration applied twice across two surfaces (factor^2, #2631) and calibration measured against its own corrected output so it oscillated to ~1.41 instead of converging on 2.0 (#2632). Every unit, boundary, property and round-trip test passed for both. HOW TO SPOT ONE (the detection tell, not a judgment call): the feature's own acceptance criterion carries a TEMPORAL QUANTIFIER — "after N phases", "subsequent", "over time", "improves", "learns", "adapts". That phrasing means the claim is about a TRAJECTORY, so a step-wise `given X return Y` test does not test the claim that was made. #1952's AC4 read "After N phases, the error is computed and applied as a correction to SUBSEQUENT estimates" — the tell was in plain sight and was still tested as a point. Survey of this repo (2026-07): estimation calibration is the ONLY true instance; size/mutation ratchets are exempt because they fail on both growth AND shrinkage (cannot self-satisfy), and retry ladders (node_repair_budget, plan_bounce_passes, provider_escalation) terminate rather than feed back. Test anchor: tests/estimate-loop-convergence.test.cjs` `RULESET.TESTS.boundary-coverage.fixtures=for any code with budget/limit/quota/threshold parameter, test suite MUST include: (a) input where SUT estimate == limit exactly, (b) input where estimate == limit - 1, (c) input where estimate == limit + 1, (d) input where any internal reserve/safety constant pushes baseline within reserve-distance of limit (catches early-pressure firing)` `RULESET.TESTS.boundary-coverage.anti-pattern=test suites that pair budget:1_000_000 (trivially fits) with budget:1 (trivially overflows) and skip the boundary region; failure mode that shipped PR #3708 UNNEEDED_TRIM + FALSE_HARDFAIL regressions (commit 2df566ed, fixed bde1ae8f)` `LEARNING.prompt-budget.boundary-gap=PR #3708 commit 2df566ed reserved NOTE_RESERVE_TOKENS in pressure-threshold AND in minSet pre-check; both buggy paths only fire when baseTokens ∈ (effectiveBudget - NOTE_RESERVE_TOKENS, effectiveBudget]; original test suite used budgets far from that band so neither path was exercised; fix bde1ae8f confines NOTE_RESERVE accounting to post-trim assembly path only; future budget/limit code MUST add boundary fixtures per RULESET.TESTS.boundary-coverage.fixtures` diff --git a/agents/gsd-executor.md b/agents/gsd-executor.md index 1ab8d43d7..ce827d567 100644 --- a/agents/gsd-executor.md +++ b/agents/gsd-executor.md @@ -638,7 +638,16 @@ This file is the canonical output of this step. The orchestrator reads `.plannin **Use template:** @~/.claude/gsd-core/templates/summary.md -**Frontmatter:** phase, plan, subsystem, tags, dependency graph (requires/provides/affects), tech-stack (added/patterns), key-files (created/modified), decisions, metrics (duration, completed date), status (`status: complete` — required so the audit-open scanner recognises the summary as done). +**Frontmatter:** phase, plan, subsystem, tags, dependency graph (requires/provides/affects), tech-stack (added/patterns), key-files (created/modified), decisions, metrics (duration, completed date), status (`status: complete` — required so the audit-open scanner recognises the summary as done), and `actuals` (#2632). + +**`actuals` (required when the plan carried an `estimate`):** record what the phase ACTUALLY cost, on the SAME scale the estimate used — `estimateTokens` (chars/4) over the realized diff, NOT a harness token count. Mixing scales measures the measurement methods, not the miss. +```yaml +actuals: + tokens: 74000 # chars/4 over the files you actually changed + tasks: 5 # tasks completed + commits: 7 # commits made +``` +These pair with the plan's `estimate` to calibrate future estimates (ADR-2629). Do not round to look closer to the estimate — a flattering number corrupts every later projection. **Title:** `# Phase [X] Plan [Y]: [Name] Summary` diff --git a/agents/gsd-planner.md b/agents/gsd-planner.md index a6a6b320b..97890e5e5 100644 --- a/agents/gsd-planner.md +++ b/agents/gsd-planner.md @@ -293,8 +293,8 @@ See @~/.claude/gsd-core/references/planner-guidance.md for dependency graph buil Full rules: @~/.claude/gsd-core/references/context-budget.md (Phase Sizing). Read before sizing. - **2-3 tasks per plan.** **ALWAYS split if:** >3 tasks, multiple subsystems, or any task touching >5 files. -- **Emit `estimate`** in PLAN.md frontmatter: run the `estimate-calibration` query, multiply your raw - token projection by its calibration factor, and copy its `confidence` verbatim — derived from the +- **Emit `estimate`**: run `estimate-calibration`; `tokens` = raw projection x factor, `raw_tokens` = that + projection before the factor (calibration measures actual/raw), `confidence` verbatim — derived from sample count, never self-rated. - **Over the smart-zone budget?** Re-slice: tracer + expansion slices. Advisory, never a block. @@ -318,6 +318,7 @@ user_setup: [] # Human-required setup (omit if empty) estimate: # Projected execution cost (see Estimate Emission) tokens: 60000 # calibrated projection + raw_tokens: 30000 # pre-factor projection tasks: 3 # task count the projection assumes confidence: low # low | med | high — DERIVED from sample count, never self-rated diff --git a/docs/adr/2629-phase-effort-estimation-calibration.md b/docs/adr/2629-phase-effort-estimation-calibration.md index 6708d8e81..578832db0 100644 --- a/docs/adr/2629-phase-effort-estimation-calibration.md +++ b/docs/adr/2629-phase-effort-estimation-calibration.md @@ -66,7 +66,7 @@ The effective ceiling is model-, task-, and distractor-dependent. **100k is a co ### 4. Calibration: median ratio, clamped, with a minimum sample count ``` -ratio_i = actuals_i.tokens / estimate_i.tokens (phases with BOTH fields) +ratio_i = actuals_i.tokens / estimate_i.raw_tokens (per PLAN, with BOTH fields) factor = clamp(median(ratio_i), 0.5, 3.0) when n >= 3 factor = 1.0 when n < 3 ``` @@ -74,6 +74,8 @@ factor = 1.0 when n < 3 - **Median, not mean** — one pathological phase (an aborted run, a mass rename) must not swing the projection for every later phase. - **Clamped to [0.5, 3.0]** — bounds the blast radius of a degenerate history; a factor outside that range indicates the estimator is wrong in kind, not in degree, and should be fixed rather than amplified. - **`n >= 3` before any correction applies** — below that, the sample says more about variance than about bias. +- **The denominator is the RAW projection, not the emitted (already-corrected) figure** — amended #2632. Measuring `actual / calibrated` is self-defeating: once the correction works the observed ratio approaches 1, which drags the median back toward 1 and un-corrects the next estimate. Simulated over 10 phases against a true 2x miss it oscillates and settles near 1.41 instead of converging on 2.0. Plans therefore record `estimate.raw_tokens` alongside the calibrated `estimate.tokens`, and `calibrationBasis()` prefers it (falling back to `tokens` for plans written before #2632, where no factor had yet been applied). +- **Samples are per PLAN, not per phase** — amended #2632. A phase holds several `--PLAN.md` files; pairing at phase granularity cross-pairs one plan's projection with another's cost and discards the rest. Persisted to `.planning/estimation-calibration.json` with a `schema_version` field, written by `extract-learnings`, read at plan time. Versioned from the first write so the schema can migrate without a silent misread. diff --git a/docs/reference/plan-md.md b/docs/reference/plan-md.md index 3a3da716c..73a628d0a 100644 --- a/docs/reference/plan-md.md +++ b/docs/reference/plan-md.md @@ -73,7 +73,7 @@ must_haves: | `requirements` | Yes | array of IDs | Requirement IDs from ROADMAP.md that this plan addresses. Every phase requirement ID must appear in at least one plan's `requirements` field. Empty arrays are a BLOCKER. | | `user_setup` | No | array of objects | External-service setup steps that Claude cannot automate (account creation, secret retrieval, dashboard configuration). When present, execute-phase generates a `USER-SETUP.md` checklist for the developer. | | `status` | No | `superseded` | Marks a plan that was deliberately reassigned or abandoned mid-phase and will never be executed. A `status: superseded` plan is excluded from the phase's plan and summary counts, so it never holds the phase below 100%. See [Superseded plans](#superseded-plans). Any other value (or the field's absence) has no effect on counting. | -| `estimate` | No | object | Projected execution cost: `{tokens, tasks, confidence}` (#2631, [ADR-2629](../adr/2629-phase-effort-estimation-calibration.md)). `tokens` is an `estimateTokens`-scale projection with the project's calibration factor **already applied** (which is why the plan-checker passes `--calibrated` to `estimate-check` — re-applying it would square the correction); `confidence` (`low`/`med`/`high`) is **derived from the calibration sample count, never self-rated**. Additive and optional — a plan without it behaves exactly as before. A plan estimated above `workflow.smart_zone_tokens` is flagged with a split recommendation at plan time; the flag is advisory and never blocks. | +| `estimate` | No | object | Projected execution cost: `{tokens, raw_tokens, tasks, confidence}` (#2631, [ADR-2629](../adr/2629-phase-effort-estimation-calibration.md)). `tokens` is an `estimateTokens`-scale projection with the project's calibration factor **already applied** (which is why the plan-checker passes `--calibrated` to `estimate-check` — re-applying it would square the correction); `confidence` (`low`/`med`/`high`) is **derived from the calibration sample count, never self-rated**. Additive and optional — a plan without it behaves exactly as before. A plan estimated above `workflow.smart_zone_tokens` is flagged with a split recommendation at plan time; the flag is advisory and never blocks. | | `must_haves` | Yes | object | Goal-backward verification criteria. See below. | ### Superseded plans diff --git a/docs/reference/planning-artifacts.md b/docs/reference/planning-artifacts.md index d62c32005..969db01ff 100644 --- a/docs/reference/planning-artifacts.md +++ b/docs/reference/planning-artifacts.md @@ -189,6 +189,17 @@ See [PLAN.md schema](plan-md.md) for the full field reference. | **Produced by** | `execute-phase` executor agent (written at the end of each plan's execution). | | **Consumed by** | `/gsd-progress` (phase status); `gsd-planner` (when a subsequent plan has a genuine dependency on prior plan output); `milestone-summary`. | +**`actuals` frontmatter (#2632, [ADR-2629](../adr/2629-phase-effort-estimation-calibration.md)).** When the phase's PLAN carried an `estimate`, the executor records what it actually cost: + +```yaml +actuals: + tokens: 74000 # estimateTokens scale (chars/4) over the realized diff + tasks: 5 + commits: 7 +``` + +`tokens` uses the **same scale as the estimate**, not a harness-reported token count — an executor subagent cannot read its own consumption, and a ratio between two different measurement methods would measure the methods rather than the miss. `/gsd:extract-learnings` pairs each phase's estimate with its actuals via `gsd_run query estimate-calibrate`, writes `.planning/estimation-calibration.json`, and the planner applies the resulting factor to subsequent estimates. Additive and optional: a summary without `actuals` simply contributes no calibration sample. + ### `-VERIFICATION.md` | | | diff --git a/gsd-core/bin/gsd-tools.cjs b/gsd-core/bin/gsd-tools.cjs index de0d7e197..f20e12d91 100755 --- a/gsd-core/bin/gsd-tools.cjs +++ b/gsd-core/bin/gsd-tools.cjs @@ -2187,6 +2187,7 @@ const HOST_COMMAND_ROUTERS = { // rather than a family — ADR-2346 promotes to a family only at >=3. 'estimate-check': ({ args, cwd, raw }) => estimateCli.cmdEstimateCheck(cwd, args.slice(1), raw), 'estimate-calibration': ({ args, cwd, raw }) => estimateCli.cmdEstimateCalibration(cwd, args.slice(1), raw), + 'estimate-calibrate': ({ args, cwd, raw }) => estimateCli.cmdEstimateCalibrate(cwd, args.slice(1), raw), 'config-new-project': routeConfigNewProject, 'config-path': routeConfigPath, 'migrate-config': routeMigrateConfig, diff --git a/gsd-core/templates/summary-minimal.md b/gsd-core/templates/summary-minimal.md index 8278c5007..4cd8ae6f3 100644 --- a/gsd-core/templates/summary-minimal.md +++ b/gsd-core/templates/summary-minimal.md @@ -6,6 +6,10 @@ tags: [searchable tech] provides: - [bullet list of what was built/delivered] affects: [list of phase names or keywords] +actuals: + tokens: [chars/4 over files actually changed] + tasks: [tasks completed] + commits: [commits made] tech-stack: added: [libraries/tools] patterns: [architectural/code patterns] diff --git a/gsd-core/templates/summary-standard.md b/gsd-core/templates/summary-standard.md index c1b851eec..5ef26eb5b 100644 --- a/gsd-core/templates/summary-standard.md +++ b/gsd-core/templates/summary-standard.md @@ -6,6 +6,10 @@ tags: [searchable tech] provides: - [bullet list of what was built/delivered] affects: [list of phase names or keywords] +actuals: + tokens: [chars/4 over files actually changed] + tasks: [tasks completed] + commits: [commits made] tech-stack: added: [libraries/tools] patterns: [architectural/code patterns] diff --git a/gsd-core/templates/summary.md b/gsd-core/templates/summary.md index c22327c31..11a235904 100644 --- a/gsd-core/templates/summary.md +++ b/gsd-core/templates/summary.md @@ -21,6 +21,13 @@ provides: - [bullet list of what this phase built/delivered] affects: [list of phase names or keywords that will need this context] +# Actuals (#2632) — pairs with the plan's `estimate` to calibrate future estimates. +# Same estimateTokens scale (chars/4 over the realized diff), never a harness token count. +actuals: + tokens: [chars/4 over files actually changed] + tasks: [tasks completed] + commits: [commits made] + # Tech tracking tech-stack: added: [libraries/tools added in this phase] diff --git a/gsd-core/workflows/extract-learnings.md b/gsd-core/workflows/extract-learnings.md index f721996be..35ee0b5b7 100644 --- a/gsd-core/workflows/extract-learnings.md +++ b/gsd-core/workflows/extract-learnings.md @@ -186,6 +186,27 @@ The body follows this structure: ``` + +Rebuild the estimate-vs-actual calibration from every completed phase (#2632, ADR-2629). + +```bash +gsd_run query estimate-calibrate +``` + +This pairs each phase's PLAN `estimate` with its SUMMARY `actuals`, writes +`.planning/estimation-calibration.json`, and reports the resulting correction factor. +The planner reads it on the next `/gsd:plan-phase`, so estimates improve for THIS project +over time. + +Report the returned `factor`, `sample_count`, and `confidence` in the summary output. +`applied: false` means fewer than 3 phases carry both an estimate and actuals — that is +expected early and is not an error. The verb rebuilds from scratch each run, so it is safe +to re-run and never accumulates duplicates. + +Phases missing either side are skipped rather than guessed: a fabricated sample would +steer every future estimate. + + Update STATE.md to reflect the learning extraction: diff --git a/src/estimate-cli.cts b/src/estimate-cli.cts index 73406c17a..89b7a09cc 100644 --- a/src/estimate-cli.cts +++ b/src/estimate-cli.cts @@ -33,6 +33,35 @@ const { output, error, ERROR_REASON } = io; const { planningDir } = planningWorkspace; const { CONFIG_DEFAULTS } = configLoader; +// WIN-1 parity (DEFECT.WINDOWS-FS-OPS): on Windows a concurrent reader, indexer, +// or AV scanner can transiently hold the rename target open. Retry the transient +// errnos with backoff, matching the writeLedger / writeConsentStore idiom. +const RENAME_RETRY_ERRNOS = new Set(['EPERM', 'EBUSY', 'EACCES']); +const RENAME_MAX_ATTEMPTS = 3; +const RENAME_RETRY_BACKOFF_MS = 50; +let _renameSleepBuf: Int32Array | null = null; +function renameBackoff(): void { + if (_renameSleepBuf === null) _renameSleepBuf = new Int32Array(new SharedArrayBuffer(4)); + Atomics.wait(_renameSleepBuf, 0, 0, RENAME_RETRY_BACKOFF_MS); +} + +/** Rename with a bounded retry on the transient Windows errnos. Rethrows anything else. */ +function renameWithRetry(from: string, to: string): void { + for (let attempt = 1; ; attempt += 1) { + try { + fs.renameSync(from, to); + return; + } catch (err) { + const code = (err as NodeJS.ErrnoException).code ?? ''; + if (attempt < RENAME_MAX_ATTEMPTS && RENAME_RETRY_ERRNOS.has(code)) { + renameBackoff(); + continue; + } + throw err; + } + } +} + /** Filename of the persisted calibration document, written by extract-learnings (Phase 3). */ export const CALIBRATION_FILENAME = 'estimation-calibration.json'; @@ -151,6 +180,130 @@ export function cmdEstimateCheck(cwd: string, args: string[], raw: boolean): voi }, raw); } +/** + * Pair each completed phase's PLAN estimate with its SUMMARY actuals. + * + * A phase contributes a sample only when BOTH sides are present and well-formed. + * A plan with no `estimate` block, a summary with no `actuals`, or a malformed + * value is skipped rather than guessed — a fabricated sample would silently + * steer every future estimate. + */ +export function collectCalibrationSamples(cwd: string): estimation.CalibrationSample[] { + const phasesRoot = path.join(planningDir(cwd), 'phases'); + let phases: string[]; + try { + phases = fs.readdirSync(phasesRoot, { withFileTypes: true }) + .filter((d) => d.isDirectory()) + .map((d) => d.name) + .sort(); + } catch { + return []; + } + + const readBlock = (file: string, key: string): Record | null => { + let text: string; + try { + text = fs.readFileSync(file, 'utf-8'); + } catch { + return null; + } + return estimation.extractFrontmatterBlock(text, key); + }; + + const samples: estimation.CalibrationSample[] = []; + for (const phase of phases) { + const dir = path.join(phasesRoot, phase); + let files: string[]; + try { + files = fs.readdirSync(dir).sort(); + } catch { + continue; + } + + // Pair PER PLAN, keyed on the `-` stem, NOT per phase directory. + // A phase routinely holds several plans (docs/reference/planning-artifacts.md: + // "one file per plan"). Taking the first plan with an estimate and the first + // summary with actuals independently cross-pairs one plan's projection with + // another's cost — a fabricated sample — and discards every later plan. + const stems = new Map(); + for (const f of files) { + const m = /^(.*?)-(PLAN|SUMMARY)\.md$/.exec(f); + if (m === null) continue; + const stem = m[1]; + const entry = stems.get(stem) ?? {}; + if (m[2] === 'PLAN') entry.plan = path.join(dir, f); + else entry.summary = path.join(dir, f); + stems.set(stem, entry); + } + + for (const stem of [...stems.keys()].sort()) { + const { plan, summary } = stems.get(stem) as { plan?: string; summary?: string }; + if (plan === undefined || summary === undefined) continue; + + const estimate = estimation.parseEstimate(readBlock(plan, 'estimate')); + const actuals = estimation.parseActuals(readBlock(summary, 'actuals')); + if (estimate === null || actuals === null) continue; + + // Measure against the RAW projection — see PhaseEstimate.rawTokens for why + // measuring against the calibrated figure makes the loop self-defeating. + samples.push({ + estimateTokens: estimation.calibrationBasis(estimate), + actualTokens: actuals.tokens, + }); + } + } + return samples; +} + +/** + * `estimate-calibrate` — rebuild the calibration document from completed phases. + * + * Rebuilds from scratch every run rather than appending, so it is idempotent and + * a corrupt prior document is replaced rather than merged. This is the verb that + * closes the loop (#1952 AC4): extract-learnings invokes it, and the planner's + * next estimate reads the result. + */ +export function cmdEstimateCalibrate(cwd: string, _args: string[], raw: boolean): void { + const samples = collectCalibrationSamples(cwd); + const calibration = estimation.computeCalibration(samples); + + const target = path.join(planningDir(cwd), CALIBRATION_FILENAME); + let written = true; + let writeError: string | null = null; + try { + // Write-then-rename: a direct writeFileSync can leave a truncated file if + // interrupted, and parseCalibrationDocument treats malformed JSON exactly + // like "no history yet" — so a torn write would silently erase the + // calibration instead of surfacing. + const tmp = `${target}.tmp-${String(process.pid)}`; + try { + fs.writeFileSync(tmp, estimation.renderCalibrationDocument(samples), 'utf-8'); + renameWithRetry(tmp, target); + } catch (err) { + // Never leave the temp behind for a later run to trip over. + try { fs.rmSync(tmp, { force: true }); } catch { /* best effort */ } + throw err; + } + } catch (err) { + // Persisting is best-effort: a read-only .planning must not fail the phase. + // But report WHY — a genuine bug and a benign permission issue are otherwise + // indistinguishable to both the caller and the workflow. + written = false; + writeError = err instanceof Error ? (err.message || String(err)) : String(err); + } + + output({ + factor: calibration.factor, + applied: calibration.applied, + sample_count: calibration.sampleCount, + confidence: calibration.confidence, + clamped: calibration.clamped, + min_samples: estimation.MIN_CALIBRATION_SAMPLES, + written, + write_error: writeError, + }, raw); +} + /** * `estimate-calibration` — report the current correction factor and the * history behind it. diff --git a/src/phase-estimation.cts b/src/phase-estimation.cts index faeadd630..58fd7b89b 100644 --- a/src/phase-estimation.cts +++ b/src/phase-estimation.cts @@ -56,6 +56,19 @@ export interface PhaseEstimate { tokens: number; tasks: number; confidence: Confidence; + /** + * The planner's UNCALIBRATED projection, before the correction factor was + * applied. Optional for backward compatibility with plans written before + * #2632. + * + * Calibration MUST measure actual/raw, not actual/calibrated. Measuring + * against the already-corrected figure makes the loop self-defeating: once + * the correction works, the observed ratio approaches 1, which drags the + * median back toward 1, which un-corrects the next estimate. Simulated over + * 10 phases with a true 2x underestimate, that oscillates and settles at + * ~1.41 instead of converging on 2.0. + */ + rawTokens?: number; } export interface PhaseActuals { @@ -230,6 +243,46 @@ export function applyCalibration(rawTokens: unknown, factor: unknown): number { return Math.min(Number.MAX_SAFE_INTEGER, Math.max(1, scaled)); } +/** + * Extract a two-space-indented scalar block (`estimate:` / `actuals:`) out of a + * document's leading YAML frontmatter. + * + * Hand-rolled because gsd-core ships no external dependencies (CONTRIBUTING.md + * "No external dependencies in core") — js-yaml is a devDependency and is not + * available at runtime. Scope is deliberately narrow: the leading `---` block + * only, so a `estimate:` line inside a fenced code block in the body cannot be + * mistaken for frontmatter (the DEFECT.FRONTMATTER-SCALAR-BROAD-GREP class). + * + * Numeric-looking values are returned as numbers so parseEstimate/parseActuals + * see the types they validate; everything else stays a string. + */ +export function extractFrontmatterBlock(text: unknown, key: string): Record | null { + if (typeof text !== 'string') return null; + + // Anchor at byte 0 — CRLF-tolerant. + const fm = /^---\r?\n([\s\S]*?)\r?\n---\r?(?:\n|$)/.exec(text); + if (fm === null) return null; + + const lines = fm[1].split(/\r?\n/); + const startIdx = lines.findIndex((l) => l === `${key}:` || l.startsWith(`${key}:`)); + if (startIdx === -1) return null; + + const out: Record = Object.create(null) as Record; + for (let i = startIdx + 1; i < lines.length; i += 1) { + const line = lines[i]; + if (!/^\s/.test(line)) break; // dedent ends the block + const m = /^\s+([A-Za-z_][\w-]*):\s*(.*)$/.exec(line); + if (m === null) continue; + const rawValue = m[2].replace(/\s+#.*$/, '').trim(); + if (rawValue === '') continue; + const asNumber = Number(rawValue); + out[m[1]] = /^-?\d+(?:\.\d+)?$/.test(rawValue) && Number.isFinite(asNumber) + ? asNumber + : rawValue.replace(/^['"]|['"]$/g, ''); + } + return Object.keys(out).length > 0 ? { ...out } : null; +} + /** Pull the `estimate:` mapping out of an already-parsed frontmatter object. */ function estimateBlockOf(input: unknown): unknown { if (input === null || typeof input !== 'object') return null; @@ -256,7 +309,10 @@ export function parseEstimate(input: unknown): PhaseEstimate | null { if (!isPositiveInt(tokens) || !isPositiveInt(tasks) || !isConfidence(confidence)) return null; - return { tokens, tasks, confidence }; + const rawTokens = record['raw_tokens']; + return isPositiveInt(rawTokens) + ? { tokens, tasks, confidence, rawTokens } + : { tokens, tasks, confidence }; } /** Pull the `actuals:` mapping out of an already-parsed frontmatter object. */ @@ -292,12 +348,22 @@ export function parseActuals(input: unknown): PhaseActuals | null { * property test pins. */ export function renderEstimate(estimate: PhaseEstimate): string { - return [ + const lines = [ 'estimate:', ` tokens: ${estimate.tokens}`, - ` tasks: ${estimate.tasks}`, - ` confidence: ${estimate.confidence}`, - ].join('\n'); + ]; + if (isPositiveInt(estimate.rawTokens)) lines.push(` raw_tokens: ${estimate.rawTokens}`); + lines.push(` tasks: ${estimate.tasks}`, ` confidence: ${estimate.confidence}`); + return lines.join('\n'); +} + +/** + * The figure calibration must measure against: the uncalibrated projection when + * the plan recorded one, else the stored value (pre-#2632 plans, where the two + * were the same because no factor had yet been applied). + */ +export function calibrationBasis(estimate: PhaseEstimate): number { + return isPositiveInt(estimate.rawTokens) ? estimate.rawTokens : estimate.tokens; } /** Render an actuals block for SUMMARY.md frontmatter. Inverse of parseActuals. */ diff --git a/tests/agent-size-baseline.json b/tests/agent-size-baseline.json index 78d09ea4c..3f5aa8d7b 100644 --- a/tests/agent-size-baseline.json +++ b/tests/agent-size-baseline.json @@ -14,7 +14,7 @@ "gsd-domain-researcher.md": 7032, "gsd-eval-auditor.md": 12496, "gsd-eval-planner.md": 7008, - "gsd-executor.md": 47955, + "gsd-executor.md": 48596, "gsd-framework-selector.md": 6778, "gsd-integration-checker.md": 15238, "gsd-intel-updater.md": 18166, @@ -23,7 +23,7 @@ "gsd-pattern-mapper.md": 12487, "gsd-phase-researcher.md": 40866, "gsd-plan-checker.md": 46363, - "gsd-planner.md": 49251, + "gsd-planner.md": 49311, "gsd-project-researcher.md": 22242, "gsd-research-synthesizer.md": 13847, "gsd-roadmapper.md": 22273, diff --git a/tests/estimate-calibrate.test.cjs b/tests/estimate-calibrate.test.cjs new file mode 100644 index 000000000..83ad85d2b --- /dev/null +++ b/tests/estimate-calibrate.test.cjs @@ -0,0 +1,365 @@ +/** + * estimate-calibrate — build the calibration document from completed phases. + * + * Epic #1952 Phase 3 (#2632). Design lock: docs/adr/2629-phase-effort-estimation-calibration.md. + * + * This is the verb that makes AC4 real. Phase 1 shipped the calibration MATH; + * Phase 2 made the planner emit an estimate. Neither closes the loop, because + * nothing pairs a plan's `estimate` with its summary's `actuals` and writes the + * result. Leaving that to agent prose would make "estimates improve over time" + * unverifiable — so the pairing and the write are deterministic here, and + * extract-learnings just invokes them. + * + * The headline test is `a consistently-underestimated project produces an + * upward correction`: that is epic acceptance criterion AC4 stated as an + * executable claim. + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const { createTempProject, cleanup, runGsdTools } = require('./helpers.cjs'); +const est = require('../gsd-core/bin/lib/phase-estimation.cjs'); + +/** Write a phase dir containing a PLAN with an estimate and a SUMMARY with actuals. */ +function writePhase(tmpDir, phaseDir, { estTokens, actTokens, tasks = 3, commits = 4 }) { + const dir = path.join(tmpDir, '.planning', 'phases', phaseDir); + fs.mkdirSync(dir, { recursive: true }); + if (estTokens !== null) { + fs.writeFileSync(path.join(dir, '01-PLAN.md'), [ + '---', + 'phase: ' + phaseDir, + 'plan: 01', + 'estimate:', + ` tokens: ${estTokens}`, + ` tasks: ${tasks}`, + ' confidence: low', + 'must_haves:', + ' truths: []', + '---', + 'x', + '', + ].join('\n')); + } + if (actTokens !== null) { + fs.writeFileSync(path.join(dir, '01-SUMMARY.md'), [ + '---', + 'phase: ' + phaseDir, + 'plan: 01', + 'actuals:', + ` tokens: ${actTokens}`, + ` tasks: ${tasks}`, + ` commits: ${commits}`, + '---', + '## What shipped', + '', + ].join('\n')); + } + return dir; +} + +describe('estimate-calibrate', () => { + test('AC4: a consistently-underestimated project produces an upward correction', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + // Three phases that each cost ~2x their estimate. + writePhase(tmpDir, '01-alpha', { estTokens: 50000, actTokens: 98000 }); + writePhase(tmpDir, '02-beta', { estTokens: 60000, actTokens: 121000 }); + writePhase(tmpDir, '03-gamma', { estTokens: 40000, actTokens: 82000 }); + + const r = runGsdTools('query estimate-calibrate', tmpDir); + assert.ok(r.success, `estimate-calibrate should succeed: ${r.error}`); + + const out = JSON.parse(r.output); + assert.equal(out.sample_count, 3, 'all three phases pair up'); + assert.equal(out.applied, true); + assert.ok(out.factor > 1, `expected an upward correction, got ${out.factor}`); + + // The document must be persisted where estimate-calibration reads it. + const docPath = path.join(tmpDir, '.planning', 'estimation-calibration.json'); + assert.ok(fs.existsSync(docPath), 'calibration document must be written'); + assert.deepEqual( + est.parseCalibrationDocument(fs.readFileSync(docPath, 'utf8')).length, 3, + 'persisted document must carry all three samples', + ); + + // And the read verb must now agree — this is the loop actually closing. + const readBack = JSON.parse(runGsdTools('query estimate-calibration', tmpDir).output); + assert.equal(readBack.factor, out.factor, 'estimate-calibration must see what estimate-calibrate wrote'); + assert.equal(readBack.applied, true); + + // A subsequent estimate is therefore larger than the raw projection. + const check = JSON.parse(runGsdTools('query estimate-check --tokens 50000', tmpDir).output); + assert.ok(check.calibrated_tokens > 50000, + `a later estimate must be corrected upward, got ${check.calibrated_tokens}`); + }); + + test('a consistently-overestimated project produces a downward correction', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + writePhase(tmpDir, '01-a', { estTokens: 100000, actTokens: 60000 }); + writePhase(tmpDir, '02-b', { estTokens: 80000, actTokens: 48000 }); + writePhase(tmpDir, '03-c', { estTokens: 90000, actTokens: 54000 }); + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.ok(out.factor < 1, `expected a downward correction, got ${out.factor}`); + }); + + test('boundary: inert below the minimum sample count, applied at it', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + writePhase(tmpDir, '01-a', { estTokens: 100, actTokens: 200 }); + writePhase(tmpDir, '02-b', { estTokens: 100, actTokens: 200 }); + let out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 2); + assert.equal(out.applied, false, '2 samples must not apply a correction'); + assert.equal(out.factor, 1); + + writePhase(tmpDir, '03-c', { estTokens: 100, actTokens: 200 }); + out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 3); + assert.equal(out.applied, true, '3 samples must apply'); + assert.equal(out.factor, 2); + }); + + test('phases missing either side are skipped, not guessed', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + writePhase(tmpDir, '01-paired', { estTokens: 100, actTokens: 200 }); + writePhase(tmpDir, '02-plan-only', { estTokens: 100, actTokens: null }); + writePhase(tmpDir, '03-summary-only', { estTokens: null, actTokens: 200 }); + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 1, 'only the fully-paired phase counts'); + }); + + test('a phase whose PLAN has no estimate block contributes nothing', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const dir = path.join(tmpDir, '.planning', 'phases', '01-noest'); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, '01-PLAN.md'), '---\nphase: 01-noest\nplan: 01\n---\nbody\n'); + fs.writeFileSync(path.join(dir, '01-SUMMARY.md'), '---\nphase: 01-noest\nactuals:\n tokens: 5\n tasks: 1\n commits: 1\n---\nx\n'); + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 0); + assert.equal(out.applied, false); + }); + + test('no phases at all is a clean no-op, not an error', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const r = runGsdTools('query estimate-calibrate', tmpDir); + assert.ok(r.success, 'must not fail on an empty project'); + const out = JSON.parse(r.output); + assert.equal(out.sample_count, 0); + assert.equal(out.factor, 1); + }); + + test('re-running is idempotent — it rebuilds, never appends duplicates', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + writePhase(tmpDir, '01-a', { estTokens: 100, actTokens: 200 }); + writePhase(tmpDir, '02-b', { estTokens: 100, actTokens: 200 }); + writePhase(tmpDir, '03-c', { estTokens: 100, actTokens: 200 }); + + const first = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + const second = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.deepEqual(second, first, 'a second run must produce an identical result'); + + const doc = est.parseCalibrationDocument( + fs.readFileSync(path.join(tmpDir, '.planning', 'estimation-calibration.json'), 'utf8'), + ); + assert.equal(doc.length, 3, 'samples must not accumulate across runs'); + }); + + test('a corrupt pre-existing document is replaced, not merged', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + fs.writeFileSync(path.join(tmpDir, '.planning', 'estimation-calibration.json'), '{ not json'); + writePhase(tmpDir, '01-a', { estTokens: 100, actTokens: 200 }); + + const r = runGsdTools('query estimate-calibrate', tmpDir); + assert.ok(r.success, 'a corrupt prior document must not fail the rebuild'); + const doc = est.parseCalibrationDocument( + fs.readFileSync(path.join(tmpDir, '.planning', 'estimation-calibration.json'), 'utf8'), + ); + assert.equal(doc.length, 1); + }); + + test('the written document round-trips through the parser', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + writePhase(tmpDir, '01-a', { estTokens: 12345, actTokens: 23456 }); + + runGsdTools('query estimate-calibrate', tmpDir); + const raw = fs.readFileSync(path.join(tmpDir, '.planning', 'estimation-calibration.json'), 'utf8'); + const parsed = est.parseCalibrationDocument(raw); + assert.deepEqual(parsed, [{ estimateTokens: 12345, actualTokens: 23456 }]); + assert.equal(JSON.parse(raw).schema_version, est.CALIBRATION_SCHEMA_VERSION, + 'must stamp the current schema version so a future reader can refuse it'); + }); +}); + +// ─── convergence guard (#2632) ───────────────────────────────────────────── + +describe('calibration converges instead of oscillating', () => { + // The loop must measure actual/RAW, not actual/calibrated. Measuring against + // the already-corrected figure is self-defeating: once the correction works + // the observed ratio approaches 1, dragging the median back toward 1, which + // un-corrects the next estimate. This test pins convergence over enough + // phases for that oscillation to show up — it fails at ~1.41 if the basis + // regresses to the calibrated value. + const RAW = 50000; + const TRUE_COST = 100000; // the planner is consistently 2x low + + const simulate = (useRawBasis) => { + const samples = []; + for (let phase = 0; phase < 10; phase += 1) { + const cal = est.computeCalibration(samples); + const emitted = est.applyCalibration(RAW, cal.factor); + const estimate = { tokens: emitted, rawTokens: RAW, tasks: 3, confidence: cal.confidence }; + samples.push({ + estimateTokens: useRawBasis ? est.calibrationBasis(estimate) : estimate.tokens, + actualTokens: TRUE_COST, + }); + } + return est.computeCalibration(samples).factor; + }; + + test('measuring against the raw projection converges on the true ratio', () => { + assert.ok(Math.abs(simulate(true) - 2) < 1e-9, + `expected convergence on 2.0, got ${simulate(true)}`); + }); + + test('measuring against the calibrated figure does NOT converge', () => { + // Negative proof that the basis choice is load-bearing, not incidental. + assert.ok(simulate(false) < 1.9, + 'if this passes at ~2.0 the two bases are equivalent and this guard is vacuous'); + }); + + test('calibrationBasis prefers raw_tokens and falls back for older plans', () => { + assert.equal(est.calibrationBasis({ tokens: 100000, rawTokens: 50000, tasks: 3, confidence: 'med' }), 50000); + assert.equal(est.calibrationBasis({ tokens: 60000, tasks: 3, confidence: 'low' }), 60000, + 'a pre-#2632 plan with no raw_tokens must still contribute a sample'); + }); + + test('estimate-calibrate uses raw_tokens from the plan when present', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + // tokens=100000 (calibrated) but raw_tokens=50000; actual=100000. + // Ratio must be 100000/50000 = 2, NOT 100000/100000 = 1. + for (const phase of ['01-a', '02-b', '03-c']) { + const dir = path.join(tmpDir, '.planning', 'phases', phase); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, '01-PLAN.md'), + `---\nphase: ${phase}\nestimate:\n tokens: 100000\n raw_tokens: 50000\n tasks: 3\n confidence: med\nmust_haves:\n---\nx\n`); + fs.writeFileSync(path.join(dir, '01-SUMMARY.md'), + `---\nphase: ${phase}\nactuals:\n tokens: 100000\n tasks: 3\n commits: 5\n---\nx\n`); + } + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 3); + assert.equal(out.factor, 2, + 'ratio must be actual/raw (2.0), not actual/calibrated (1.0)'); + }); +}); + +// ─── multi-plan pairing (#2632 review BLOCKER) ───────────────────────────── + +describe('multi-plan phases pair per plan, not per phase', () => { + // A phase routinely holds several plans (`--PLAN.md`, one per plan — + // docs/reference/planning-artifacts.md). An earlier implementation took the + // first PLAN carrying an estimate and the first SUMMARY carrying actuals + // INDEPENDENTLY, which cross-paired one plan's projection with another plan's + // cost and discarded every later plan. The whole suite passed because its + // helper only ever wrote `01-PLAN.md`. + + /** Write one plan/summary pair inside a phase, using the real `-` naming. */ + const writePlan = (tmpDir, phase, pp, { estTokens, actTokens }) => { + const dir = path.join(tmpDir, '.planning', 'phases', phase); + fs.mkdirSync(dir, { recursive: true }); + const nn = phase.slice(0, 2); + if (estTokens !== null) { + fs.writeFileSync(path.join(dir, `${nn}-${pp}-PLAN.md`), + `---\nphase: ${phase}\nplan: ${pp}\nestimate:\n tokens: ${estTokens}\n` + + ` raw_tokens: ${estTokens}\n tasks: 3\n confidence: low\nmust_haves:\n---\nx\n`); + } else { + fs.writeFileSync(path.join(dir, `${nn}-${pp}-PLAN.md`), `---\nphase: ${phase}\nplan: ${pp}\n---\nx\n`); + } + fs.writeFileSync(path.join(dir, `${nn}-${pp}-SUMMARY.md`), + `---\nphase: ${phase}\nplan: ${pp}\nactuals:\n tokens: ${actTokens}\n tasks: 3\n commits: 4\n---\nx\n`); + }; + + test('never cross-pairs one plan\'s estimate with another plan\'s actuals', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + // Plan 01 has NO estimate but cheap actuals; plan 02 has both (true 2.5x). + writePlan(tmpDir, '04-multi', '01', { estTokens: null, actTokens: 30000 }); + writePlan(tmpDir, '04-multi', '02', { estTokens: 80000, actTokens: 200000 }); + + runGsdTools('query estimate-calibrate', tmpDir); + const doc = est.parseCalibrationDocument( + fs.readFileSync(path.join(tmpDir, '.planning', 'estimation-calibration.json'), 'utf8'), + ); + + assert.deepEqual(doc, [{ estimateTokens: 80000, actualTokens: 200000 }], + 'plan 02\'s estimate must pair with plan 02\'s actuals — cross-pairing fabricates a sample ' + + 'and throws away the real signal'); + }); + + test('counts every correctly-paired plan in a multi-plan phase', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + // Three plans in ONE phase, each cleanly 2x. + writePlan(tmpDir, '05-wave', '01', { estTokens: 40000, actTokens: 80000 }); + writePlan(tmpDir, '05-wave', '02', { estTokens: 50000, actTokens: 100000 }); + writePlan(tmpDir, '05-wave', '03', { estTokens: 60000, actTokens: 120000 }); + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 3, 'all three plans must contribute — not just the first'); + assert.equal(out.factor, 2); + assert.equal(out.applied, true, 'three samples in one phase must reach the minimum'); + }); + + test('a plan with no matching summary contributes nothing', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const dir = path.join(tmpDir, '.planning', 'phases', '06-partial'); + fs.mkdirSync(dir, { recursive: true }); + // 06-01 pairs; 06-02 is a plan with no summary (mid-execution). + fs.writeFileSync(path.join(dir, '06-01-PLAN.md'), + '---\nphase: 06-partial\nestimate:\n tokens: 100\n raw_tokens: 100\n tasks: 1\n confidence: low\nmust_haves:\n---\nx\n'); + fs.writeFileSync(path.join(dir, '06-01-SUMMARY.md'), + '---\nphase: 06-partial\nactuals:\n tokens: 200\n tasks: 1\n commits: 1\n---\nx\n'); + fs.writeFileSync(path.join(dir, '06-02-PLAN.md'), + '---\nphase: 06-partial\nestimate:\n tokens: 999999\n raw_tokens: 999999\n tasks: 1\n confidence: low\nmust_haves:\n---\nx\n'); + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 1, 'an in-flight plan must not contribute a half-sample'); + }); + + test('samples accumulate across BOTH plans and phases', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + writePlan(tmpDir, '01-a', '01', { estTokens: 100, actTokens: 200 }); + writePlan(tmpDir, '01-a', '02', { estTokens: 100, actTokens: 200 }); + writePlan(tmpDir, '02-b', '01', { estTokens: 100, actTokens: 200 }); + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.sample_count, 3, 'two plans in phase 1 plus one in phase 2'); + assert.equal(out.applied, true); + }); +}); diff --git a/tests/estimate-loop-convergence.test.cjs b/tests/estimate-loop-convergence.test.cjs new file mode 100644 index 000000000..4fed28ec9 --- /dev/null +++ b/tests/estimate-loop-convergence.test.cjs @@ -0,0 +1,146 @@ +/** + * Closed-loop convergence for phase-effort estimation. + * + * Epic #1952. Design lock: docs/adr/2629-phase-effort-estimation-calibration.md. + * + * WHY THIS FILE EXISTS + * ──────────────────── + * Estimation is a feedback control loop: the correction derived from past + * phases feeds back into the next estimate. Two real defects shipped past a + * green suite of ~26,800 step-wise tests during this epic, because BOTH are + * properties of the loop over time rather than of any single call: + * + * 1. Calibration applied twice (#2631 review). The planner emitted an + * already-calibrated figure and `estimate-check` re-applied the factor, so + * the effective correction was factor². Every unit test still passed — + * each function was individually correct; the COMPOSITION was not. + * + * 2. Non-convergence (#2632). Calibration measured actual/calibrated instead + * of actual/raw, so once the correction worked the observed ratio + * approached 1, dragging the median back down. Simulated over 10 phases + * the factor oscillated and settled near 1.41 instead of 2.0. Every + * boundary fixture, property test and round-trip still passed. + * + * Neither is visible to a test that asserts "given X, return Y". Both are + * visible here, because this drives the REAL verbs through many iterations and + * asserts on the TRAJECTORY of the user-visible number. + * + * The rule this encodes (CONTEXT.md RULESET.TESTS.feedback-loop-convergence): + * when a feature's output feeds back into its own input, a step-wise test is + * not sufficient evidence of correctness. Simulate the loop and assert it + * converges, stays bounded, and lands on the truth. + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const { createTempProject, cleanup, runGsdTools } = require('./helpers.cjs'); + +const RAW_PROJECTION = 50000; // what the planner would say with no history +const TRUE_COST = 100000; // reality: consistently 2x the raw projection +const TRUE_RATIO = TRUE_COST / RAW_PROJECTION; + +/** Drive one full plan→execute→calibrate cycle and return what the user sees. */ +function runPhase(tmpDir, phaseIndex) { + const factorRaw = runGsdTools('query estimate-calibration --pick factor --raw', tmpDir).output; + const factor = Number(String(factorRaw).trim()) || 1; + + // The planner emits tokens = raw x factor, and records the raw projection. + const emitted = Math.max(1, Math.round(RAW_PROJECTION * factor)); + + const dir = path.join(tmpDir, '.planning', 'phases', `${String(phaseIndex).padStart(2, '0')}-p`); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, '01-PLAN.md'), + `---\nphase: p\nestimate:\n tokens: ${emitted}\n raw_tokens: ${RAW_PROJECTION}\n` + + ` tasks: 3\n confidence: low\nmust_haves:\n---\nx\n`); + fs.writeFileSync(path.join(dir, '01-SUMMARY.md'), + `---\nphase: p\nactuals:\n tokens: ${TRUE_COST}\n tasks: 3\n commits: 5\n---\nx\n`); + + runGsdTools('query estimate-calibrate', tmpDir); + + // What the checker actually compares against the budget for this plan. + const check = JSON.parse( + runGsdTools(`query estimate-check --tokens ${emitted} --calibrated`, tmpDir).output, + ); + return { factor, emitted, userSees: check.calibrated_tokens }; +} + +describe('estimation feedback loop', () => { + test('converges on the true cost and stays there', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + const history = []; + for (let i = 1; i <= 8; i += 1) history.push(runPhase(tmpDir, i)); + + // Early phases have no history, so no correction is applied yet. + assert.equal(history[0].factor, 1, 'phase 1 has no history'); + assert.equal(history[0].emitted, RAW_PROJECTION); + + // Once enough samples exist the correction must reach the TRUE ratio… + const settled = history.slice(4); + for (const step of settled) { + assert.ok(Math.abs(step.factor - TRUE_RATIO) < 1e-9, + `factor drifted to ${step.factor}; expected ${TRUE_RATIO}. ` + + 'A factor that wanders means calibration is measuring against its own output.'); + assert.equal(step.emitted, TRUE_COST, + `emitted estimate ${step.emitted} should equal the true cost ${TRUE_COST}`); + } + + // …and the number the USER is shown must equal the emitted estimate. + // If anything re-applies the factor downstream this reads 2x and fails. + for (const step of settled) { + assert.equal(step.userSees, step.emitted, + `the checker compared ${step.userSees} against the budget but the plan says ${step.emitted} — ` + + 'a downstream surface is applying the correction a second time (factor²).'); + } + }); + + test('stays bounded under a wildly inconsistent history', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + // Alternating 10x over and 10x under: the clamp must hold and the loop + // must not run away in either direction. + const costs = [500000, 5000, 500000, 5000, 500000, 5000]; + costs.forEach((cost, i) => { + const dir = path.join(tmpDir, '.planning', 'phases', `${String(i + 1).padStart(2, '0')}-p`); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, '01-PLAN.md'), + `---\nphase: p\nestimate:\n tokens: 50000\n raw_tokens: 50000\n tasks: 3\n confidence: low\nmust_haves:\n---\nx\n`); + fs.writeFileSync(path.join(dir, '01-SUMMARY.md'), + `---\nphase: p\nactuals:\n tokens: ${cost}\n tasks: 3\n commits: 5\n---\nx\n`); + }); + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.ok(out.factor >= 0.5 && out.factor <= 3.0, + `factor ${out.factor} escaped the clamp under an adversarial history`); + + const check = JSON.parse(runGsdTools('query estimate-check --tokens 50000', tmpDir).output); + assert.ok(check.calibrated_tokens >= 25000 && check.calibrated_tokens <= 150000, + `a corrected estimate of ${check.calibrated_tokens} is outside the clamp's reachable range`); + }); + + test('an all-accurate history leaves estimates unchanged (fixed point)', (t) => { + const tmpDir = createTempProject(); + t.after(() => cleanup(tmpDir)); + + // A project whose estimates are already right must not be "corrected". + for (let i = 1; i <= 5; i += 1) { + const dir = path.join(tmpDir, '.planning', 'phases', `${String(i).padStart(2, '0')}-p`); + fs.mkdirSync(dir, { recursive: true }); + fs.writeFileSync(path.join(dir, '01-PLAN.md'), + `---\nphase: p\nestimate:\n tokens: 60000\n raw_tokens: 60000\n tasks: 3\n confidence: low\nmust_haves:\n---\nx\n`); + fs.writeFileSync(path.join(dir, '01-SUMMARY.md'), + `---\nphase: p\nactuals:\n tokens: 60000\n tasks: 3\n commits: 5\n---\nx\n`); + } + + const out = JSON.parse(runGsdTools('query estimate-calibrate', tmpDir).output); + assert.equal(out.factor, 1, 'an accurate project must be a fixed point — no correction applied'); + + const check = JSON.parse(runGsdTools('query estimate-check --tokens 60000', tmpDir).output); + assert.equal(check.calibrated_tokens, 60000, 'an accurate estimate must pass through unchanged'); + }); +}); diff --git a/tests/fixtures/golden-install-parity/antigravity.json b/tests/fixtures/golden-install-parity/antigravity.json index b2df66e25..df433c4c4 100644 --- a/tests/fixtures/golden-install-parity/antigravity.json +++ b/tests/fixtures/golden-install-parity/antigravity.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "1db46cac3f4d9889", "agents/gsd-eval-auditor.md": "1b8391f1aafb067f", "agents/gsd-eval-planner.md": "3d10fd11147f6857", - "agents/gsd-executor.md": "041c6d658e8e1002", + "agents/gsd-executor.md": "f9bd85ba91a2a042", "agents/gsd-framework-selector.md": "daa62c79619c76bf", "agents/gsd-integration-checker.md": "0643cd2d779b131c", "agents/gsd-intel-updater.md": "26c1f1e028c6346a", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "e62ee90d39084802", "agents/gsd-phase-researcher.md": "cff1196c8e8bb4fa", "agents/gsd-plan-checker.md": "893e88036b4ad419", - "agents/gsd-planner.md": "8af42f4321c34bc9", + "agents/gsd-planner.md": "e916e99dab32a95b", "agents/gsd-project-researcher.md": "85de7f562872ee9b", "agents/gsd-research-synthesizer.md": "18a2e1b30ff7ae3a", "agents/gsd-roadmapper.md": "7a8465ac6d4dd29e", @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "ea841e2865248e74", - "gsd-core/bin/gsd-tools.cjs": "b28392bf9a4dc9bb", + "gsd-core/bin/gsd-tools.cjs": "974871b2efcd2b7b", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -204,9 +204,9 @@ "gsd-core/templates/spec.md": "8734f0df4df3a34b", "gsd-core/templates/state.md": "a45a134631efe3f9", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "52abe2af968e8533", "gsd-core/templates/user-setup.md": "1da2382725db080f", "gsd-core/templates/verification-report.md": "78ab9264ce63ed7e", @@ -257,7 +257,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "5c6fcb36e16f5bf7", "gsd-core/workflows/explore.md": "965e1c0ede05f40c", - "gsd-core/workflows/extract-learnings.md": "167ea7f0e23bf496", + "gsd-core/workflows/extract-learnings.md": "82f8b60ab36e4818", "gsd-core/workflows/fast.md": "568e6c3ec00b6e00", "gsd-core/workflows/forensics.md": "b64f0309b8c3fde1", "gsd-core/workflows/graduation.md": "c892887f212c9cff", diff --git a/tests/fixtures/golden-install-parity/augment.json b/tests/fixtures/golden-install-parity/augment.json index 28fe28801..011d6f9e5 100644 --- a/tests/fixtures/golden-install-parity/augment.json +++ b/tests/fixtures/golden-install-parity/augment.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "671c9ea949889c4a", "agents/gsd-eval-auditor.md": "fcaec7b00f94c435", "agents/gsd-eval-planner.md": "a4a5b4b3f7828ba3", - "agents/gsd-executor.md": "00d887a16c249765", + "agents/gsd-executor.md": "3f03ad10b7251f68", "agents/gsd-framework-selector.md": "4b77eebbe9288d80", "agents/gsd-integration-checker.md": "fa53e2d78be1de74", "agents/gsd-intel-updater.md": "fa40e685d7441ace", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "43c6021cf7caabfa", "agents/gsd-phase-researcher.md": "f1f6fd6a3e67c7a8", "agents/gsd-plan-checker.md": "d9aca5f75d649fc0", - "agents/gsd-planner.md": "5b018b0a2df93e37", + "agents/gsd-planner.md": "5143e54864af6a63", "agents/gsd-project-researcher.md": "4531b7cc8f5e5f7d", "agents/gsd-research-synthesizer.md": "4a4f68e6c75b133a", "agents/gsd-roadmapper.md": "bb2f57695dbab32c", @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -275,9 +275,9 @@ "gsd-core/templates/spec.md": "26d55bce940f0288", "gsd-core/templates/state.md": "4d123aa6cea167fe", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -328,7 +328,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "334bb2053908a049", "gsd-core/workflows/explore.md": "3f3b4f83e23349bc", - "gsd-core/workflows/extract-learnings.md": "b6f01ca3d8f58de4", + "gsd-core/workflows/extract-learnings.md": "ab26f2cbb14efef2", "gsd-core/workflows/fast.md": "11f5cd10ae5cc7d3", "gsd-core/workflows/forensics.md": "0d500a3f5ab26913", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/fixtures/golden-install-parity/claude-local.json b/tests/fixtures/golden-install-parity/claude-local.json index 436eee65b..5ff7cb21c 100644 --- a/tests/fixtures/golden-install-parity/claude-local.json +++ b/tests/fixtures/golden-install-parity/claude-local.json @@ -15,7 +15,7 @@ "agents/gsd-domain-researcher.md": "f1e03df842ddfb95", "agents/gsd-eval-auditor.md": "d0f45fff7370bb0b", "agents/gsd-eval-planner.md": "9cc049b82897daa4", - "agents/gsd-executor.md": "406dfda62f0f8428", + "agents/gsd-executor.md": "c6a3f08a31afc6bf", "agents/gsd-framework-selector.md": "85005d716f9d98f7", "agents/gsd-integration-checker.md": "17a8ee731986564d", "agents/gsd-intel-updater.md": "4953a465db9dadc1", @@ -24,7 +24,7 @@ "agents/gsd-pattern-mapper.md": "b45b5e106775bec1", "agents/gsd-phase-researcher.md": "4772d9eada32e8bd", "agents/gsd-plan-checker.md": "b3f510f1257ff383", - "agents/gsd-planner.md": "831977f45c361561", + "agents/gsd-planner.md": "a22751cc49456236", "agents/gsd-project-researcher.md": "d7f355894519f9fe", "agents/gsd-research-synthesizer.md": "1c738df9932d325a", "agents/gsd-roadmapper.md": "453e9471ad27c7ea", @@ -109,7 +109,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -274,9 +274,9 @@ "gsd-core/templates/spec.md": "7dc900c355098d8b", "gsd-core/templates/state.md": "73e424b8c70b765c", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -327,7 +327,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "724d726833e222ed", "gsd-core/workflows/explore.md": "fabe8e5553de16c0", - "gsd-core/workflows/extract-learnings.md": "fd75072c339b58bd", + "gsd-core/workflows/extract-learnings.md": "3778273aaba5d477", "gsd-core/workflows/fast.md": "7f7687b920d79b29", "gsd-core/workflows/forensics.md": "857d7b064f4cca21", "gsd-core/workflows/graduation.md": "7a1d9e1327dce4e7", diff --git a/tests/fixtures/golden-install-parity/claude.json b/tests/fixtures/golden-install-parity/claude.json index 737e39f10..ae26f1afe 100644 --- a/tests/fixtures/golden-install-parity/claude.json +++ b/tests/fixtures/golden-install-parity/claude.json @@ -15,7 +15,7 @@ "agents/gsd-domain-researcher.md": "5f7d366251b957fe", "agents/gsd-eval-auditor.md": "fea2759beff0a642", "agents/gsd-eval-planner.md": "112f6730f23854e3", - "agents/gsd-executor.md": "16391af53f8922bb", + "agents/gsd-executor.md": "dc6785301ba81da4", "agents/gsd-framework-selector.md": "c350ee693cb1aa4e", "agents/gsd-integration-checker.md": "c8b4e65dee89c8ea", "agents/gsd-intel-updater.md": "5b41e05f90ce89d9", @@ -24,7 +24,7 @@ "agents/gsd-pattern-mapper.md": "b45b5e106775bec1", "agents/gsd-phase-researcher.md": "85217c69c1ed2ac6", "agents/gsd-plan-checker.md": "670e4ad7132a66ca", - "agents/gsd-planner.md": "0f5d0199fbce83f8", + "agents/gsd-planner.md": "cef4c830e0fa2bd8", "agents/gsd-project-researcher.md": "f468e96f8339d1e0", "agents/gsd-research-synthesizer.md": "7be02e47f4fd901b", "agents/gsd-roadmapper.md": "8a7f1f1256a6aed5", @@ -38,7 +38,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -203,9 +203,9 @@ "gsd-core/templates/spec.md": "7dc900c355098d8b", "gsd-core/templates/state.md": "73e424b8c70b765c", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -256,7 +256,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "da0c2286b1aed11d", "gsd-core/workflows/explore.md": "9348b54b526b8938", - "gsd-core/workflows/extract-learnings.md": "d8177b0c13b7e5ee", + "gsd-core/workflows/extract-learnings.md": "13929ce8a04013e0", "gsd-core/workflows/fast.md": "41a6568b873aef99", "gsd-core/workflows/forensics.md": "c01da0178fb97b21", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/fixtures/golden-install-parity/cline.json b/tests/fixtures/golden-install-parity/cline.json index 72f476b38..b46bf20f7 100644 --- a/tests/fixtures/golden-install-parity/cline.json +++ b/tests/fixtures/golden-install-parity/cline.json @@ -19,7 +19,7 @@ "agents/gsd-domain-researcher.md": "0fecdaea86466a56", "agents/gsd-eval-auditor.md": "36c44303085df2f8", "agents/gsd-eval-planner.md": "3ddea88a69b4da3f", - "agents/gsd-executor.md": "a20e280c9e648f5b", + "agents/gsd-executor.md": "eb82990b9b81e39d", "agents/gsd-framework-selector.md": "564669d479433f15", "agents/gsd-integration-checker.md": "1bbbdd3d420b994e", "agents/gsd-intel-updater.md": "42c40fffbc720d0b", @@ -28,7 +28,7 @@ "agents/gsd-pattern-mapper.md": "b526065fd2efa19c", "agents/gsd-phase-researcher.md": "c507db2ba66038f4", "agents/gsd-plan-checker.md": "70ccc072a52cc4ce", - "agents/gsd-planner.md": "aab8d4d2db4c3d93", + "agents/gsd-planner.md": "4aaa7d9fab92a01e", "agents/gsd-project-researcher.md": "049f816c6caa4316", "agents/gsd-research-synthesizer.md": "2f7dcbff50371d4c", "agents/gsd-roadmapper.md": "bbb23d3097911516", @@ -42,7 +42,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "476aa24e8c4f03cf", - "gsd-core/bin/gsd-tools.cjs": "695c409ab2407247", + "gsd-core/bin/gsd-tools.cjs": "50e8f67219cfb620", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -207,9 +207,9 @@ "gsd-core/templates/spec.md": "26d55bce940f0288", "gsd-core/templates/state.md": "4d123aa6cea167fe", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -260,7 +260,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "204a0ac94a4bb545", "gsd-core/workflows/explore.md": "db9e1a79dbe88090", - "gsd-core/workflows/extract-learnings.md": "6f39375b7dc775f9", + "gsd-core/workflows/extract-learnings.md": "9a39a7b2b7592141", "gsd-core/workflows/fast.md": "e4f74a454b6ca5e8", "gsd-core/workflows/forensics.md": "9fc65a8eed5d8bfc", "gsd-core/workflows/graduation.md": "68d83a47e9cfaa7a", diff --git a/tests/fixtures/golden-install-parity/codebuddy.json b/tests/fixtures/golden-install-parity/codebuddy.json index a84e188e8..74fb2d1d4 100644 --- a/tests/fixtures/golden-install-parity/codebuddy.json +++ b/tests/fixtures/golden-install-parity/codebuddy.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "1c1a800108a2b225", "agents/gsd-eval-auditor.md": "99012004b14ea602", "agents/gsd-eval-planner.md": "4ebdd7fe9cbb0cfe", - "agents/gsd-executor.md": "d9dca78de04bdc1c", + "agents/gsd-executor.md": "412ddd97399607a2", "agents/gsd-framework-selector.md": "7726fccc86bfeb50", "agents/gsd-integration-checker.md": "2d8339790bbb2dc3", "agents/gsd-intel-updater.md": "c51339956197cbd3", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "92cfa2e6c2a06bf3", "agents/gsd-phase-researcher.md": "6338474da1a5d65e", "agents/gsd-plan-checker.md": "3fdccb7353c69340", - "agents/gsd-planner.md": "55445462761df7b3", + "agents/gsd-planner.md": "c482d13228f2efd8", "agents/gsd-project-researcher.md": "e43c59f7f1f2f37a", "agents/gsd-research-synthesizer.md": "87955470c3c129b2", "agents/gsd-roadmapper.md": "20b69eff61a7a9fa", @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -275,9 +275,9 @@ "gsd-core/templates/spec.md": "26d55bce940f0288", "gsd-core/templates/state.md": "4d123aa6cea167fe", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -328,7 +328,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "874689d8678e6186", "gsd-core/workflows/explore.md": "3f3b4f83e23349bc", - "gsd-core/workflows/extract-learnings.md": "b6f01ca3d8f58de4", + "gsd-core/workflows/extract-learnings.md": "ab26f2cbb14efef2", "gsd-core/workflows/fast.md": "11f5cd10ae5cc7d3", "gsd-core/workflows/forensics.md": "0d500a3f5ab26913", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/fixtures/golden-install-parity/codex.json b/tests/fixtures/golden-install-parity/codex.json index 37455cefa..ae6901d54 100644 --- a/tests/fixtures/golden-install-parity/codex.json +++ b/tests/fixtures/golden-install-parity/codex.json @@ -102,8 +102,8 @@ "agents/gsd-eval-auditor.toml": "9b81d61b3c5f722d", "agents/gsd-eval-planner.md": "73f2ad2ff2797a51", "agents/gsd-eval-planner.toml": "09468ad1a34ac468", - "agents/gsd-executor.md": "697b7ed099fff17d", - "agents/gsd-executor.toml": "9a8aa31bbde42448", + "agents/gsd-executor.md": "cd4dac78e03debd1", + "agents/gsd-executor.toml": "427bb38fc924d3e8", "agents/gsd-framework-selector.md": "ebae32430887d2e0", "agents/gsd-framework-selector.toml": "637e4e021b7ec380", "agents/gsd-integration-checker.md": "9cc875676cf7d741", @@ -120,8 +120,8 @@ "agents/gsd-phase-researcher.toml": "44a3d510cd0ce3bd", "agents/gsd-plan-checker.md": "9ba8ea7643f0a7f5", "agents/gsd-plan-checker.toml": "03488397d892ea09", - "agents/gsd-planner.md": "88283bc1d8da011e", - "agents/gsd-planner.toml": "1b24b3f19b8afecc", + "agents/gsd-planner.md": "932266ed67ea0947", + "agents/gsd-planner.toml": "8fd81d2367979851", "agents/gsd-project-researcher.md": "959f2e57c3d69ed8", "agents/gsd-project-researcher.toml": "f395e8e8c4baf1ed", "agents/gsd-research-synthesizer.md": "497f85adf53259ef", @@ -145,7 +145,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -310,9 +310,9 @@ "gsd-core/templates/spec.md": "34cb8124f37c9b14", "gsd-core/templates/state.md": "3bac0c4a26c094f9", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "f612660bfd0a325a", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "c95dfa4653791314", "gsd-core/templates/user-profile.md": "52abe2af968e8533", "gsd-core/templates/user-setup.md": "1da2382725db080f", "gsd-core/templates/verification-report.md": "78ab9264ce63ed7e", @@ -363,7 +363,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "e4723e1669f7c508", "gsd-core/workflows/explore.md": "d12f68770c65093a", - "gsd-core/workflows/extract-learnings.md": "f716aa03fcb5f8da", + "gsd-core/workflows/extract-learnings.md": "44efe9756da774f9", "gsd-core/workflows/fast.md": "e4ed60f96a7b3ac8", "gsd-core/workflows/forensics.md": "2e8a01b5b44e65f3", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/fixtures/golden-install-parity/copilot.json b/tests/fixtures/golden-install-parity/copilot.json index d63f047a6..77aec7dae 100644 --- a/tests/fixtures/golden-install-parity/copilot.json +++ b/tests/fixtures/golden-install-parity/copilot.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.agent.md": "d603239b3e9fe428", "agents/gsd-eval-auditor.agent.md": "3c03009564de55c8", "agents/gsd-eval-planner.agent.md": "14751876fc2b5f16", - "agents/gsd-executor.agent.md": "6bf98b78fd75c347", + "agents/gsd-executor.agent.md": "d8439a4f44e7e6eb", "agents/gsd-framework-selector.agent.md": "cafeec0b3489be45", "agents/gsd-integration-checker.agent.md": "30439b804927acc7", "agents/gsd-intel-updater.agent.md": "238c1a886f35a25c", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.agent.md": "b1f488b0fa6a2395", "agents/gsd-phase-researcher.agent.md": "03cfb510a766fe93", "agents/gsd-plan-checker.agent.md": "0d68eb258b85e24e", - "agents/gsd-planner.agent.md": "0c222d93778eb371", + "agents/gsd-planner.agent.md": "aaef800d164c3475", "agents/gsd-project-researcher.agent.md": "d73bdbe986ffa8a6", "agents/gsd-research-synthesizer.agent.md": "f03eed4aa89e47c5", "agents/gsd-roadmapper.agent.md": "322048cf8ddcb4e5", @@ -40,7 +40,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "ea841e2865248e74", - "gsd-core/bin/gsd-tools.cjs": "b28392bf9a4dc9bb", + "gsd-core/bin/gsd-tools.cjs": "974871b2efcd2b7b", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -205,9 +205,9 @@ "gsd-core/templates/spec.md": "8734f0df4df3a34b", "gsd-core/templates/state.md": "a45a134631efe3f9", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "52abe2af968e8533", "gsd-core/templates/user-setup.md": "1da2382725db080f", "gsd-core/templates/verification-report.md": "78ab9264ce63ed7e", @@ -258,7 +258,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "c82c959f78a1917d", "gsd-core/workflows/explore.md": "35431c40bb1b245a", - "gsd-core/workflows/extract-learnings.md": "f34d0b1927545b18", + "gsd-core/workflows/extract-learnings.md": "6ed17a7b6c9f68a6", "gsd-core/workflows/fast.md": "66821090b6b8ed3b", "gsd-core/workflows/forensics.md": "459644dce26ee2ef", "gsd-core/workflows/graduation.md": "343e287b8bfac6b2", diff --git a/tests/fixtures/golden-install-parity/cursor.json b/tests/fixtures/golden-install-parity/cursor.json index e002fbc20..e245555b1 100644 --- a/tests/fixtures/golden-install-parity/cursor.json +++ b/tests/fixtures/golden-install-parity/cursor.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "56395dbdabf076f6", "agents/gsd-eval-auditor.md": "ad2840fd5cd76172", "agents/gsd-eval-planner.md": "2049dac060d00eda", - "agents/gsd-executor.md": "026606d319f6c338", + "agents/gsd-executor.md": "4246bd0e27197e6e", "agents/gsd-framework-selector.md": "4b77eebbe9288d80", "agents/gsd-integration-checker.md": "5da30584d06b878c", "agents/gsd-intel-updater.md": "b8971c5d96e63b38", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "1229c215677f740d", "agents/gsd-phase-researcher.md": "982d59921bed463d", "agents/gsd-plan-checker.md": "30bc89279d6586b7", - "agents/gsd-planner.md": "61d89d16ac357320", + "agents/gsd-planner.md": "1c0d5aaa124f56ab", "agents/gsd-project-researcher.md": "beeac940d3a10e76", "agents/gsd-research-synthesizer.md": "6315f016d55176f4", "agents/gsd-roadmapper.md": "d28e7d4bac46dde2", @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "2525f1ae8b086828", - "gsd-core/bin/gsd-tools.cjs": "1b15863444a36838", + "gsd-core/bin/gsd-tools.cjs": "c9dfb100e7f40b7f", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -275,9 +275,9 @@ "gsd-core/templates/spec.md": "7dc900c355098d8b", "gsd-core/templates/state.md": "73e424b8c70b765c", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -328,7 +328,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "09ea0688882eb30e", "gsd-core/workflows/explore.md": "9348b54b526b8938", - "gsd-core/workflows/extract-learnings.md": "d8177b0c13b7e5ee", + "gsd-core/workflows/extract-learnings.md": "13929ce8a04013e0", "gsd-core/workflows/fast.md": "77b49793e26b3323", "gsd-core/workflows/forensics.md": "a65f817d4a515291", "gsd-core/workflows/graduation.md": "0c0f4e6258ae482f", diff --git a/tests/fixtures/golden-install-parity/hermes.json b/tests/fixtures/golden-install-parity/hermes.json index 74292e9d9..6f5f4a436 100644 --- a/tests/fixtures/golden-install-parity/hermes.json +++ b/tests/fixtures/golden-install-parity/hermes.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "412cdbb05ba252ea", "agents/gsd-eval-auditor.md": "4ffb265063c318e5", "agents/gsd-eval-planner.md": "03448fc9c5774b56", - "agents/gsd-executor.md": "3ec54684a756e830", + "agents/gsd-executor.md": "53379f898b45b192", "agents/gsd-framework-selector.md": "ea9981d65d6b3429", "agents/gsd-integration-checker.md": "35b4f2969d279871", "agents/gsd-intel-updater.md": "5fe5edfae2719cb8", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "cea092600aeb3978", "agents/gsd-phase-researcher.md": "2bd0402f33d757ca", "agents/gsd-plan-checker.md": "c31dc063b5506aaf", - "agents/gsd-planner.md": "d1fe1e2653cb12de", + "agents/gsd-planner.md": "3be8aff696060e6c", "agents/gsd-project-researcher.md": "425a7df7f37a5c06", "agents/gsd-research-synthesizer.md": "9d31c87fc2c87ffa", "agents/gsd-roadmapper.md": "64dce5d5f9fa5654", @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "3a3409215044af9f", - "gsd-core/bin/gsd-tools.cjs": "fb40ca85a5fc3f9b", + "gsd-core/bin/gsd-tools.cjs": "06878bab8a74e8c1", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -204,9 +204,9 @@ "gsd-core/templates/spec.md": "7dc900c355098d8b", "gsd-core/templates/state.md": "73e424b8c70b765c", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -257,7 +257,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "0c5e0455252b5a46", "gsd-core/workflows/explore.md": "843e66f101a56dfa", - "gsd-core/workflows/extract-learnings.md": "e9e167c718949c0b", + "gsd-core/workflows/extract-learnings.md": "7f44f65fbf32a427", "gsd-core/workflows/fast.md": "c801145115755524", "gsd-core/workflows/forensics.md": "91961b811917c5c4", "gsd-core/workflows/graduation.md": "938c71f1e7650da0", diff --git a/tests/fixtures/golden-install-parity/kilo.json b/tests/fixtures/golden-install-parity/kilo.json index e399017e5..4d0ec0fe1 100644 --- a/tests/fixtures/golden-install-parity/kilo.json +++ b/tests/fixtures/golden-install-parity/kilo.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "a3874d80bcbc7380", "agents/gsd-eval-auditor.md": "630d4cd3bd6ea195", "agents/gsd-eval-planner.md": "3db12cde12aeb2c1", - "agents/gsd-executor.md": "bea8102bf46ddb7a", + "agents/gsd-executor.md": "2dfa6f871d65bab1", "agents/gsd-framework-selector.md": "ad5f2c6b9bec6270", "agents/gsd-integration-checker.md": "c503e2f4a3d8ec05", "agents/gsd-intel-updater.md": "231393da62a45b2e", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "6a5408fd11d70391", "agents/gsd-phase-researcher.md": "94818f28c498bb26", "agents/gsd-plan-checker.md": "2546e311f7e8e0e8", - "agents/gsd-planner.md": "72996d1f61ad7b0f", + "agents/gsd-planner.md": "e6bc5cf0ac9e9c06", "agents/gsd-project-researcher.md": "60573a38d3dfd9fe", "agents/gsd-research-synthesizer.md": "1f7cd286c5783c86", "agents/gsd-roadmapper.md": "277e0a3252553ab7", @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -275,9 +275,9 @@ "gsd-core/templates/spec.md": "8734f0df4df3a34b", "gsd-core/templates/state.md": "a45a134631efe3f9", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "52abe2af968e8533", "gsd-core/templates/user-setup.md": "1da2382725db080f", "gsd-core/templates/verification-report.md": "78ab9264ce63ed7e", @@ -328,7 +328,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "731d2947b623be4f", "gsd-core/workflows/explore.md": "e6c94f8b6f083109", - "gsd-core/workflows/extract-learnings.md": "d8177b0c13b7e5ee", + "gsd-core/workflows/extract-learnings.md": "13929ce8a04013e0", "gsd-core/workflows/fast.md": "41a6568b873aef99", "gsd-core/workflows/forensics.md": "c01da0178fb97b21", "gsd-core/workflows/graduation.md": "636df3c43e9b3bca", diff --git a/tests/fixtures/golden-install-parity/kimi-code.json b/tests/fixtures/golden-install-parity/kimi-code.json index 44147e7f3..ae2b3ee06 100644 --- a/tests/fixtures/golden-install-parity/kimi-code.json +++ b/tests/fixtures/golden-install-parity/kimi-code.json @@ -44,7 +44,7 @@ "agents/gsd-domain-researcher.md": "049f588663814fa2", "agents/gsd-eval-auditor.md": "54870d3b07433525", "agents/gsd-eval-planner.md": "552e9fa164c51ce8", - "agents/gsd-executor.md": "98424ccf4346c9d1", + "agents/gsd-executor.md": "258fcf3cc19fb411", "agents/gsd-framework-selector.md": "8a795f230436ad2e", "agents/gsd-integration-checker.md": "c1760a0bbd4f7bf5", "agents/gsd-intel-updater.md": "944f1d903e2e9e09", @@ -53,7 +53,7 @@ "agents/gsd-pattern-mapper.md": "68ecefd60811a669", "agents/gsd-phase-researcher.md": "2235f61764d8e969", "agents/gsd-plan-checker.md": "f3caf89525709445", - "agents/gsd-planner.md": "bf4304ea220f3874", + "agents/gsd-planner.md": "88acabd0285bb25b", "agents/gsd-project-researcher.md": "f572892f138734ff", "agents/gsd-research-synthesizer.md": "29949bf3f049a8f1", "agents/gsd-roadmapper.md": "840ac933e3b094f9", @@ -67,7 +67,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -232,9 +232,9 @@ "gsd-core/templates/spec.md": "26d55bce940f0288", "gsd-core/templates/state.md": "4d123aa6cea167fe", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -285,7 +285,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "74de71f01739f7ac", "gsd-core/workflows/explore.md": "3f3b4f83e23349bc", - "gsd-core/workflows/extract-learnings.md": "b6f01ca3d8f58de4", + "gsd-core/workflows/extract-learnings.md": "ab26f2cbb14efef2", "gsd-core/workflows/fast.md": "11f5cd10ae5cc7d3", "gsd-core/workflows/forensics.md": "0d500a3f5ab26913", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/fixtures/golden-install-parity/kimi.json b/tests/fixtures/golden-install-parity/kimi.json index 4b8489575..d29798290 100644 --- a/tests/fixtures/golden-install-parity/kimi.json +++ b/tests/fixtures/golden-install-parity/kimi.json @@ -61,7 +61,7 @@ "agents/subagents/gsd-eval-auditor.yaml": "e3d868bd5fefe938", "agents/subagents/gsd-eval-planner.md": "70f8c5727bfb9876", "agents/subagents/gsd-eval-planner.yaml": "df8499f7af297ec2", - "agents/subagents/gsd-executor.md": "8fdf9332233a66bd", + "agents/subagents/gsd-executor.md": "77f6d8e414227fcd", "agents/subagents/gsd-executor.yaml": "e29422986636fd64", "agents/subagents/gsd-framework-selector.md": "a15b7aa1e0576e16", "agents/subagents/gsd-framework-selector.yaml": "fb52c31cde27b0e3", @@ -79,7 +79,7 @@ "agents/subagents/gsd-phase-researcher.yaml": "7633c8e82617e7cc", "agents/subagents/gsd-plan-checker.md": "410639f4c4f16e7a", "agents/subagents/gsd-plan-checker.yaml": "8295181071121db8", - "agents/subagents/gsd-planner.md": "0384e53e5b5b50e5", + "agents/subagents/gsd-planner.md": "ddf881be44151dbd", "agents/subagents/gsd-planner.yaml": "2e83ee194bcd7fbd", "agents/subagents/gsd-project-researcher.md": "39bc2ec5a8b18283", "agents/subagents/gsd-project-researcher.yaml": "ce12586b0347e2dc", @@ -103,7 +103,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -268,9 +268,9 @@ "gsd-core/templates/spec.md": "26d55bce940f0288", "gsd-core/templates/state.md": "4d123aa6cea167fe", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -321,7 +321,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "74de71f01739f7ac", "gsd-core/workflows/explore.md": "3f3b4f83e23349bc", - "gsd-core/workflows/extract-learnings.md": "b6f01ca3d8f58de4", + "gsd-core/workflows/extract-learnings.md": "ab26f2cbb14efef2", "gsd-core/workflows/fast.md": "11f5cd10ae5cc7d3", "gsd-core/workflows/forensics.md": "0d500a3f5ab26913", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/fixtures/golden-install-parity/opencode.json b/tests/fixtures/golden-install-parity/opencode.json index fef9dbd01..c0f9ffb99 100644 --- a/tests/fixtures/golden-install-parity/opencode.json +++ b/tests/fixtures/golden-install-parity/opencode.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "71250e759ca9e723", "agents/gsd-eval-auditor.md": "c88890105f32ace6", "agents/gsd-eval-planner.md": "60bddb70a937f796", - "agents/gsd-executor.md": "14423bfb4dee14b8", + "agents/gsd-executor.md": "3fdbe96e721a8099", "agents/gsd-framework-selector.md": "1c0a10355e787675", "agents/gsd-integration-checker.md": "a9de5928e5a5c649", "agents/gsd-intel-updater.md": "493e07482fa6198a", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "7c6d1d9817a9c1e7", "agents/gsd-phase-researcher.md": "9874110700b41f48", "agents/gsd-plan-checker.md": "f39aca1ea261d720", - "agents/gsd-planner.md": "15c78c3192e53c9b", + "agents/gsd-planner.md": "50f660afd4ddd599", "agents/gsd-project-researcher.md": "dae210ae0b3c6e2b", "agents/gsd-research-synthesizer.md": "e02c6ad5d1b74171", "agents/gsd-roadmapper.md": "1658a40b20d8b575", @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -275,9 +275,9 @@ "gsd-core/templates/spec.md": "8734f0df4df3a34b", "gsd-core/templates/state.md": "a45a134631efe3f9", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "52abe2af968e8533", "gsd-core/templates/user-setup.md": "1da2382725db080f", "gsd-core/templates/verification-report.md": "78ab9264ce63ed7e", @@ -328,7 +328,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "bb0ce725dcf12aa2", "gsd-core/workflows/explore.md": "a514b88e98609441", - "gsd-core/workflows/extract-learnings.md": "92b3c0979604b7d0", + "gsd-core/workflows/extract-learnings.md": "38b09022eca429d8", "gsd-core/workflows/fast.md": "12187b242e6af970", "gsd-core/workflows/forensics.md": "9354cb830152fd28", "gsd-core/workflows/graduation.md": "589b9b303cff8643", diff --git a/tests/fixtures/golden-install-parity/pi.json b/tests/fixtures/golden-install-parity/pi.json index 1e393030c..9296ca40a 100644 --- a/tests/fixtures/golden-install-parity/pi.json +++ b/tests/fixtures/golden-install-parity/pi.json @@ -6,7 +6,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -171,9 +171,9 @@ "gsd-core/templates/spec.md": "26d55bce940f0288", "gsd-core/templates/state.md": "4d123aa6cea167fe", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -224,7 +224,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "3168b13c3758ed84", "gsd-core/workflows/explore.md": "3f3b4f83e23349bc", - "gsd-core/workflows/extract-learnings.md": "b6f01ca3d8f58de4", + "gsd-core/workflows/extract-learnings.md": "ab26f2cbb14efef2", "gsd-core/workflows/fast.md": "11f5cd10ae5cc7d3", "gsd-core/workflows/forensics.md": "0d500a3f5ab26913", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/fixtures/golden-install-parity/qwen.json b/tests/fixtures/golden-install-parity/qwen.json index 0bc745ed7..2c1b733dc 100644 --- a/tests/fixtures/golden-install-parity/qwen.json +++ b/tests/fixtures/golden-install-parity/qwen.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "bd054bb27beed2a7", "agents/gsd-eval-auditor.md": "57cc7458ab5de6b7", "agents/gsd-eval-planner.md": "01b665728dde4ccf", - "agents/gsd-executor.md": "b6baf92a0d0ab516", + "agents/gsd-executor.md": "ebca30765efc0b68", "agents/gsd-framework-selector.md": "82ba6abea84226b7", "agents/gsd-integration-checker.md": "90835dbc7dfa1691", "agents/gsd-intel-updater.md": "3cc4f6ddd04676ec", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "83c66c7722e8b165", "agents/gsd-phase-researcher.md": "284e55a86ae46d7f", "agents/gsd-plan-checker.md": "e80e6d51017be405", - "agents/gsd-planner.md": "f3c8b934ee07020c", + "agents/gsd-planner.md": "303c956e286d83dd", "agents/gsd-project-researcher.md": "b5baac64a15c85e2", "agents/gsd-research-synthesizer.md": "6cd9b501dc97bd50", "agents/gsd-roadmapper.md": "c357a77ab919e9e5", @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "6e98d76e955e35a2", - "gsd-core/bin/gsd-tools.cjs": "4cc7b3241d1422b6", + "gsd-core/bin/gsd-tools.cjs": "94b222441497e080", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -204,9 +204,9 @@ "gsd-core/templates/spec.md": "7dc900c355098d8b", "gsd-core/templates/state.md": "73e424b8c70b765c", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -257,7 +257,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "ea93d7a5c3d483e8", "gsd-core/workflows/explore.md": "c412905fb86f3f12", - "gsd-core/workflows/extract-learnings.md": "dd4fdb88605de49a", + "gsd-core/workflows/extract-learnings.md": "5791168e144ec997", "gsd-core/workflows/fast.md": "93ed453edddd8c0c", "gsd-core/workflows/forensics.md": "82800a3138ac1da9", "gsd-core/workflows/graduation.md": "5d5004de086db1c8", diff --git a/tests/fixtures/golden-install-parity/trae.json b/tests/fixtures/golden-install-parity/trae.json index 99ee46c34..29c168777 100644 --- a/tests/fixtures/golden-install-parity/trae.json +++ b/tests/fixtures/golden-install-parity/trae.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "b80f76874c04e515", "agents/gsd-eval-auditor.md": "470bf16303ec4d2e", "agents/gsd-eval-planner.md": "22334fde85723c9d", - "agents/gsd-executor.md": "00768988ee91ad9f", + "agents/gsd-executor.md": "191215b38db8bb1b", "agents/gsd-framework-selector.md": "7726fccc86bfeb50", "agents/gsd-integration-checker.md": "7cd2072984411c7f", "agents/gsd-intel-updater.md": "83de6ba9172891c3", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "b5d7a4abb1baecb9", "agents/gsd-phase-researcher.md": "2256f1f82212c757", "agents/gsd-plan-checker.md": "aaa9e928de1abba9", - "agents/gsd-planner.md": "ca48b09c8688e70d", + "agents/gsd-planner.md": "c99c8a75226d19b7", "agents/gsd-project-researcher.md": "ddf7794e81300032", "agents/gsd-research-synthesizer.md": "a124b00271748d07", "agents/gsd-roadmapper.md": "493ef92b42b12cf4", @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "de4627dff103d527", - "gsd-core/bin/gsd-tools.cjs": "83769f08d1cc0216", + "gsd-core/bin/gsd-tools.cjs": "11117cf7fe37c77d", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -204,9 +204,9 @@ "gsd-core/templates/spec.md": "7dc900c355098d8b", "gsd-core/templates/state.md": "73e424b8c70b765c", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -257,7 +257,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "3abba98d19b31036", "gsd-core/workflows/explore.md": "f4d0db08c11545a8", - "gsd-core/workflows/extract-learnings.md": "3fcc858b20d0d0e6", + "gsd-core/workflows/extract-learnings.md": "1ee1ee86b7d8bb43", "gsd-core/workflows/fast.md": "0f05b1e008ac2fc2", "gsd-core/workflows/forensics.md": "665546666547875d", "gsd-core/workflows/graduation.md": "defc0dfcbfeb68d2", diff --git a/tests/fixtures/golden-install-parity/windsurf.json b/tests/fixtures/golden-install-parity/windsurf.json index 5a37a1851..440bfa8e9 100644 --- a/tests/fixtures/golden-install-parity/windsurf.json +++ b/tests/fixtures/golden-install-parity/windsurf.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "56395dbdabf076f6", "agents/gsd-eval-auditor.md": "fb64fc5acf359747", "agents/gsd-eval-planner.md": "2049dac060d00eda", - "agents/gsd-executor.md": "10341df8e1f390bb", + "agents/gsd-executor.md": "09666c96d441ecfa", "agents/gsd-framework-selector.md": "4b77eebbe9288d80", "agents/gsd-integration-checker.md": "4ffb37fb230c2b90", "agents/gsd-intel-updater.md": "a81d77c143c02108", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "ada0c169daa2f0ec", "agents/gsd-phase-researcher.md": "2a45ebde829555ec", "agents/gsd-plan-checker.md": "c68b9bd6382a22a7", - "agents/gsd-planner.md": "b63b7203e0bc7beb", + "agents/gsd-planner.md": "f9e2d99c4daf51de", "agents/gsd-project-researcher.md": "f6697b316b5995ba", "agents/gsd-research-synthesizer.md": "04036f38c1d373ea", "agents/gsd-roadmapper.md": "fb62e1e3de84b5f9", @@ -39,7 +39,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "5636ca0b726871b2", - "gsd-core/bin/gsd-tools.cjs": "1ef071e09d8edcb6", + "gsd-core/bin/gsd-tools.cjs": "41a13dea56716a55", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -204,9 +204,9 @@ "gsd-core/templates/spec.md": "7dc900c355098d8b", "gsd-core/templates/state.md": "73e424b8c70b765c", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -257,7 +257,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "2ff04f06b000fe26", "gsd-core/workflows/explore.md": "e5ff3269f8ee77b6", - "gsd-core/workflows/extract-learnings.md": "af793bdf4ffd1c8a", + "gsd-core/workflows/extract-learnings.md": "894218dd99cf2131", "gsd-core/workflows/fast.md": "b03b9f595892479a", "gsd-core/workflows/forensics.md": "3d1ce16b5f605592", "gsd-core/workflows/graduation.md": "0783b074a1c7013b", diff --git a/tests/fixtures/golden-install-parity/zcode.json b/tests/fixtures/golden-install-parity/zcode.json index fb57cd52d..65c6befb4 100644 --- a/tests/fixtures/golden-install-parity/zcode.json +++ b/tests/fixtures/golden-install-parity/zcode.json @@ -16,7 +16,7 @@ "agents/gsd-domain-researcher.md": "049f588663814fa2", "agents/gsd-eval-auditor.md": "54870d3b07433525", "agents/gsd-eval-planner.md": "552e9fa164c51ce8", - "agents/gsd-executor.md": "98424ccf4346c9d1", + "agents/gsd-executor.md": "258fcf3cc19fb411", "agents/gsd-framework-selector.md": "8a795f230436ad2e", "agents/gsd-integration-checker.md": "c1760a0bbd4f7bf5", "agents/gsd-intel-updater.md": "944f1d903e2e9e09", @@ -25,7 +25,7 @@ "agents/gsd-pattern-mapper.md": "68ecefd60811a669", "agents/gsd-phase-researcher.md": "2235f61764d8e969", "agents/gsd-plan-checker.md": "f3caf89525709445", - "agents/gsd-planner.md": "bf4304ea220f3874", + "agents/gsd-planner.md": "88acabd0285bb25b", "agents/gsd-project-researcher.md": "f572892f138734ff", "agents/gsd-research-synthesizer.md": "29949bf3f049a8f1", "agents/gsd-roadmapper.md": "840ac933e3b094f9", @@ -110,7 +110,7 @@ "gsd-core/VERSION": "ef0deccd81a6723c", "gsd-core/bin/check-latest-version.cjs": "e4a224058c8f4d74", "gsd-core/bin/ensure-runtime-build.cjs": "51bc64467ab30f62", - "gsd-core/bin/gsd-tools.cjs": "ec98734a021c6157", + "gsd-core/bin/gsd-tools.cjs": "0b356a34f3654051", "gsd-core/bin/gsd_run": "62d9b647ede212e6", "gsd-core/bin/shared/config-defaults.manifest.json": "3a3581ea768cbe6a", "gsd-core/bin/shared/config-schema.manifest.json": "377dc46be7b56fe8", @@ -275,9 +275,9 @@ "gsd-core/templates/spec.md": "26d55bce940f0288", "gsd-core/templates/state.md": "4d123aa6cea167fe", "gsd-core/templates/summary-complex.md": "a5e40574fd8894dc", - "gsd-core/templates/summary-minimal.md": "7d09b5e709e2e67c", - "gsd-core/templates/summary-standard.md": "e8d9cf4a8377cdff", - "gsd-core/templates/summary.md": "23c40f6503b3ea98", + "gsd-core/templates/summary-minimal.md": "d5f40260721e307d", + "gsd-core/templates/summary-standard.md": "a4fb6df80f41b545", + "gsd-core/templates/summary.md": "85f4d37fcee6852b", "gsd-core/templates/user-profile.md": "20749f23e4c413fc", "gsd-core/templates/user-setup.md": "78b7d718b6e8d67c", "gsd-core/templates/verification-report.md": "dd5faa6254183731", @@ -328,7 +328,7 @@ "gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md": "be84efbd71e1513e", "gsd-core/workflows/execute-plan.md": "4f43b65cf80d8f2e", "gsd-core/workflows/explore.md": "3f3b4f83e23349bc", - "gsd-core/workflows/extract-learnings.md": "b6f01ca3d8f58de4", + "gsd-core/workflows/extract-learnings.md": "ab26f2cbb14efef2", "gsd-core/workflows/fast.md": "11f5cd10ae5cc7d3", "gsd-core/workflows/forensics.md": "0d500a3f5ab26913", "gsd-core/workflows/graduation.md": "973dcebcd17e1b10", diff --git a/tests/workflow-size-baseline.json b/tests/workflow-size-baseline.json index 7e7c82fad..3b35768a5 100644 --- a/tests/workflow-size-baseline.json +++ b/tests/workflow-size-baseline.json @@ -27,7 +27,7 @@ "execute-phase.md": 90143, "execute-plan.md": 35143, "explore.md": 11127, - "extract-learnings.md": 12893, + "extract-learnings.md": 13762, "fast.md": 7613, "forensics.md": 12531, "graduation.md": 11987,