chore(#191): retire sdk package seam

This commit is contained in:
Tom Boucher
2026-05-25 10:58:36 -04:00
parent 04b3be6833
commit 11918dcc37
389 changed files with 109 additions and 84177 deletions

View File

@@ -0,0 +1,5 @@
---
type: Fixed
pr: 191
---
Retired the legacy SDK package seam by deleting `sdk/`, removing the `gsd-sdk` shim/bin publishing path, and moving required shared manifests to `get-shit-done/bin/shared` for runtime/install compatibility.

View File

@@ -1,37 +0,0 @@
#!/usr/bin/env node
/**
* bin/gsd-sdk.js — back-compat shim for external callers of `gsd-sdk`.
*
* When the parent package is installed globally (`npm install -g @opengsd/get-shit-done-redux`)
* npm creates a `gsd-sdk` symlink in the global bin directory pointing at this
* file. npm correctly chmods bin entries from a tarball, so the execute-bit
* problem that afflicted the sub-install approach (issue #2453) cannot occur here.
*
* NOTE (#2775): `npx @opengsd/get-shit-done-redux` does NOT link this shim — npx only
* exposes the package's primary bin (`get-shit-done-redux`). For npx-based usage,
* the installer (`bin/install.js#installSdkIfNeeded`) self-symlinks `gsd-sdk`
* into `~/.local/bin` when needed and verifies PATH callability before
* reporting `✓ GSD SDK ready`.
*
* This shim resolves sdk/dist/cli.js relative to its own location and delegates
* to it via `node`, so `gsd-sdk <args>` behaves identically to
* `node <packageDir>/sdk/dist/cli.js <args>`.
*
* Call sites (slash commands, agent prompts, hook scripts) continue to work without
* changes because `gsd-sdk` still resolves on PATH — it just comes from this shim
* in the parent package rather than from a separately installed @opengsd/gsd-sdk.
*/
'use strict';
const path = require('path');
const { spawnSync } = require('child_process');
const cliPath = path.resolve(__dirname, '..', 'sdk', 'dist', 'cli.js');
const result = spawnSync(process.execPath, [cliPath, ...process.argv.slice(2)], {
stdio: 'inherit',
env: process.env,
});
process.exit(result.status ?? 1);

View File

@@ -215,19 +215,12 @@ const _profileArgRaw = (() => {
// configDir is resolved) and may override 'full' — see writeActiveProfile call below. // configDir is resolved) and may override 'full' — see writeActiveProfile call below.
const _profileIsCore = _profileArgRaw === 'core'; const _profileIsCore = _profileArgRaw === 'core';
const _requestedProfileName = (hasMinimal || _profileIsCore) ? 'core' : (_profileArgRaw || null); const _requestedProfileName = (hasMinimal || _profileIsCore) ? 'core' : (_profileArgRaw || null);
const hasSdk = args.includes('--sdk');
const hasNoSdk = args.includes('--no-sdk');
if (hasMinimal && _profileArgRaw) { if (hasMinimal && _profileArgRaw) {
console.error(` ${yellow}Cannot specify both --minimal/--core-only and --profile${reset}`); console.error(` ${yellow}Cannot specify both --minimal/--core-only and --profile${reset}`);
process.exit(1); process.exit(1);
} }
if (hasSdk && hasNoSdk) {
console.error(` ${yellow}Cannot specify both --sdk and --no-sdk${reset}`);
process.exit(1);
}
// Runtime selection - can be set by flags or interactive prompt // Runtime selection - can be set by flags or interactive prompt
let selectedRuntimes = []; let selectedRuntimes = [];
if (hasAll) { if (hasAll) {
@@ -8547,22 +8540,19 @@ function install(isGlobal, runtime = 'claude', options = {}) {
failures.push('get-shit-done'); failures.push('get-shit-done');
} }
// #3288 / #3571 — Copy sdk/shared manifests into the get-shit-done payload // Copy shared manifests into the get-shit-done payload
// at the co-located path that CJS modules resolve first: // at the co-located path that CJS modules resolve first:
// get-shit-done/bin/shared/*.json // get-shit-done/bin/shared/*.json
// //
// The install copies get-shit-done/ but NOT sdk/ — CJS modules' legacy // This source now lives under get-shit-done/bin/shared in-repo.
// source-repo paths (3 levels up → sdk/shared/) therefore resolve to a
// non-existent location in every post-install layout. Copying these shared
// files alongside the CJS files ensures require() succeeds without needing
// sdk/ to exist.
const sharedPayloadFiles = [ const sharedPayloadFiles = [
'model-catalog.json', 'model-catalog.json',
'config-defaults.manifest.json', 'config-defaults.manifest.json',
'config-schema.manifest.json', 'config-schema.manifest.json',
'runtime-aliases.manifest.json',
]; ];
for (const fileName of sharedPayloadFiles) { for (const fileName of sharedPayloadFiles) {
const sharedSrc = path.join(src, 'sdk', 'shared', fileName); const sharedSrc = path.join(src, 'get-shit-done', 'bin', 'shared', fileName);
const sharedDest = path.join(skillDest, 'bin', 'shared', fileName); const sharedDest = path.join(skillDest, 'bin', 'shared', fileName);
const displayPath = `get-shit-done/bin/shared/${fileName}`; const displayPath = `get-shit-done/bin/shared/${fileName}`;
if (fs.existsSync(sharedSrc)) { if (fs.existsSync(sharedSrc)) {
@@ -8574,7 +8564,7 @@ function install(isGlobal, runtime = 'claude', options = {}) {
failures.push(displayPath); failures.push(displayPath);
} }
} else { } else {
failures.push(`sdk/shared/${fileName} (source missing)`); failures.push(`get-shit-done/bin/shared/${fileName} (source missing)`);
} }
} }
@@ -11239,13 +11229,6 @@ function installAllRuntimes(runtimes, isGlobal, isInteractive) {
const finalize = (shouldInstallStatusline, shouldInstallBanner) => { const finalize = (shouldInstallStatusline, shouldInstallBanner) => {
try { try {
// Verify sdk/dist/cli.js is present and executable. The dist is shipped
// prebuilt in the tarball (fix/2441-sdk-decouple); gsd-sdk reaches users via
// the parent package's bin/gsd-sdk.js shim, so no sub-install is needed.
// Skip with --no-sdk. Skip with isLocal (#2678 — local installs don't own global npm).
// #3033: pass forceSdk so --sdk overrides the local-install skip.
installSdkIfNeeded({ isLocal: !isGlobal, forceSdk: hasSdk, throwOnFailure: true });
const printSummaries = () => { const printSummaries = () => {
for (const result of results) { for (const result of results) {
const useStatusline = statuslineRuntimes.includes(result.runtime) && shouldInstallStatusline; const useStatusline = statuslineRuntimes.includes(result.runtime) && shouldInstallStatusline;

View File

@@ -26,12 +26,17 @@ function normalizeRuntimeToken(value) {
} }
function loadAliasManifest() { function loadAliasManifest() {
try { const manifestCandidates = [
const manifestPath = path.resolve(__dirname, '../../../sdk/shared/runtime-aliases.manifest.json'); path.resolve(__dirname, '..', 'shared', 'runtime-aliases.manifest.json'),
const parsed = JSON.parse(fs.readFileSync(manifestPath, 'utf8')); path.resolve(__dirname, '../../../sdk/shared/runtime-aliases.manifest.json'),
if (parsed && typeof parsed === 'object') return parsed; ];
} catch { for (const manifestPath of manifestCandidates) {
// Fall through to fallback aliases. try {
const parsed = JSON.parse(fs.readFileSync(manifestPath, 'utf8'));
if (parsed && typeof parsed === 'object') return parsed;
} catch {
// Try next candidate.
}
} }
return FALLBACK_ALIASES; return FALLBACK_ALIASES;
} }

View File

@@ -4,8 +4,7 @@
"description": "A meta-prompting, context engineering and spec-driven development system for Claude Code, OpenCode, Gemini and Codex by TÂCHES.", "description": "A meta-prompting, context engineering and spec-driven development system for Claude Code, OpenCode, Gemini and Codex by TÂCHES.",
"bin": { "bin": {
"get-shit-done-redux": "bin/install.js", "get-shit-done-redux": "bin/install.js",
"gsd-sdk": "bin/gsd-sdk.js", "gsd-tools": "get-shit-done/bin/gsd-tools.cjs"
"gsd-tools": "bin/gsd-sdk.js"
}, },
"files": [ "files": [
"bin", "bin",
@@ -13,14 +12,7 @@
"get-shit-done", "get-shit-done",
"agents", "agents",
"hooks", "hooks",
"scripts", "scripts"
"sdk/src",
"sdk/shared",
"sdk/prompts",
"sdk/dist",
"sdk/package.json",
"sdk/package-lock.json",
"sdk/tsconfig.json"
], ],
"keywords": [ "keywords": [
"claude", "claude",
@@ -64,12 +56,11 @@
"scripts": { "scripts": {
"check:env": "bash scripts/check-env.sh", "check:env": "bash scripts/check-env.sh",
"check:integrity": "./scripts/check-npm-integrity.sh", "check:integrity": "./scripts/check-npm-integrity.sh",
"build": "npm run build:sdk", "build": "npm run build:hooks",
"build:hooks": "node scripts/build-hooks.js", "build:hooks": "node scripts/build-hooks.js",
"build:sdk": "cd sdk && npm ci && npm run build", "prepublishOnly": "npm run build:hooks",
"prepublishOnly": "npm run build:hooks && npm run build:sdk", "pretest": "npm run lint:skill-deps",
"pretest": "npm run build:sdk && npm run lint:skill-deps", "pretest:coverage": "npm run lint:skill-deps",
"pretest:coverage": "npm run build:sdk",
"lint:descriptions": "node scripts/lint-descriptions.cjs", "lint:descriptions": "node scripts/lint-descriptions.cjs",
"lint:skill-deps": "node scripts/lint-skill-deps.cjs", "lint:skill-deps": "node scripts/lint-skill-deps.cjs",
"lint:tests": "node scripts/lint-no-source-grep.cjs", "lint:tests": "node scripts/lint-no-source-grep.cjs",

View File

@@ -1,237 +0,0 @@
# Handover: Query layer + golden parity
Use this document at the start of a new session so work continues in context without re-deriving history.
**Related:** `HANDOVER-PARITY-DOCS.md` (#2302 scope); **`sdk/src/query/QUERY-HANDLERS.md`** (golden matrix, CJS↔SDK routing).
---
## Goal for the next session (primary)
**Track A (Golden/parity) is complete.** 127/128 canonicals covered — the single exception (`phases.archive`) is permanent (SDK-only, no CJS analogue). Focus shifts to the remaining #2302 acceptance criteria.
**Ongoing:** pick next gap from **`GOLDEN_PARITY_EXCEPTIONS`** / registry orphans (run `golden-policy.test.ts`) or expand **`READ_ONLY_JSON_PARITY_ROWS`** for read-only handlers still on generic exceptions. The read-only batch in **§ Next batch** below is **done**.
**Follow-up:** confirm **`GOLDEN_PARITY_EXCEPTIONS`** for any remaining read-only registry gaps (`learnings.query`, `progress.bar`, `profile-questionnaire` — still exception-only until strict rows); extend **`read-only-golden-rows.ts`** when aligned.
### Remaining work — ordered by priority
1. **Track C — Runner alignment** (not started)
- `PhaseRunner` and `InitRunner` both take `GSDTools` (subprocess bridge) as a `tools` dependency (`phase-runner.ts:55`, `init-runner.ts:70`).
- Issue #2302 says: "Align programmatic paths with the same contracts as query handlers (shared helpers or registry dispatch), **without** removing `GSDTools`."
- Concretely: where runners currently shell out via `GSDTools.run('state update …')`, they could call the typed handler (`stateUpdate()`) directly or dispatch through `createRegistry()`. This eliminates subprocess overhead on the hot path while keeping `GSDTools` exported for backward compatibility.
- Files to touch: `sdk/src/phase-runner.ts`, `sdk/src/init-runner.ts`, `sdk/src/index.ts` (re-exports). Tests: `phase-runner.integration.test.ts`, `init-e2e.integration.test.ts`, `lifecycle-e2e.integration.test.ts`.
- **Risk:** Runner integration tests are slow and sensitive to state. Approach: swap one `tools.run()` call at a time, verify the integration test still passes, then proceed to the next.
2. **Track B — CHANGELOG.md [Unreleased] entries** (not started)
- `CHANGELOG.md` has an `[Unreleased]` section but no Phase 3 entries yet.
- Add entries covering: golden parity policy gate, mutation subprocess infrastructure, handler alignment, profile-output port, CJS deprecation header.
- `docs/CLI-TOOLS.md` already references `QUERY-HANDLERS.md` and SDK query layer — may need minor polish but is substantively done.
- `QUERY-HANDLERS.md` is maintained and current.
3. **Track D — CJS deprecation headers** (done)
- `gsd-tools.cjs` already has `@deprecated` JSDoc header (lines 3-6) pointing to `gsd-sdk query` and `@opengsd/gsd-sdk`.
- No additional CJS file deletion in scope per #2302.
4. **CI verification** (should run before any PR)
- Run full integration suite: `npx vitest run --project integration` (mutation subprocess + read-only parity + golden composition).
- Verify against CI matrix expectations: Ubuntu + macOS, Node 22 + 24.
### Acceptance criteria from #2302 — status
| Criterion | Status | Notes |
| --------- | ------ | ----- |
| Policy gate | **Done** | `verifyGoldenPolicyComplete()` green; 0 orphan canonicals |
| Parity | **Done** | 127/128 covered; strict rows, mutation subprocess, composition goldens |
| Registry | **Done** | CJS-only matrix in `QUERY-HANDLERS.md`; `docs/CLI-TOOLS.md` updated |
| Runners (Track C) | **Not started** | `PhaseRunner`/`InitRunner` still use `GSDTools` subprocess bridge |
| Deprecation (Track D) | **Done** | `@deprecated` header on `gsd-tools.cjs` |
| Docs | **Partial** | `QUERY-HANDLERS.md` current; `CHANGELOG.md` [Unreleased] needs Phase 3 entries |
| CI | **Not verified** | Unit tests green (1261/1261); integration suite not run this session |
---
## Repo / branch
- **Workspace:** `D:\Repos\get-shit-done` (GSD PBR backport initiative).
- **Feature branch:** `feat/sdk-phase3-query-layer` (62 commits ahead of `main`; confirm against `origin` before merging).
- **Upstream PRs:** `open-gsd/get-shit-done-redux` issue #2302.
---
## Golden parity architecture (current)
| Piece | Role |
| ----- | ---- |
| `sdk/src/golden/registry-canonical-commands.ts` | One canonical dispatch string per unique handler (`pickCanonicalCommandName`). |
| `sdk/src/golden/golden-integration-covered.ts` | Canonicals exercised by **`golden.integration.test.ts`** (subset/full/shape tests). |
| `sdk/src/golden/read-only-golden-rows.ts` | **Strict** `JsonParityRow[]` for `read-only-parity.integration.test.ts` (`toEqual` on parsed CJS JSON vs `sdkResult.data`). |
| `sdk/src/golden/read-only-parity.integration.test.ts` | Rows from `READ_ONLY_JSON_PARITY_ROWS` + **`config-path`** (plain stdout vs `{ path }`, `path.normalize`) + **`verify.commits`**. |
| `sdk/src/golden/capture.ts` | `captureGsdToolsOutput` (JSON stdout); **`captureGsdToolsStdout`** (raw stdout, e.g. `config-path`). |
| `sdk/src/golden/golden-policy.ts` | `GOLDEN_PARITY_INTEGRATION_COVERED` = integration ∪ `readOnlyGoldenCanonicals()` ∪ **`GOLDEN_MUTATION_SUBPROCESS_COVERED`**; `GOLDEN_PARITY_EXCEPTIONS` includes `NO_CJS_SUBPROCESS_REASON`, then `MUTATION_DEFERRED_REASON` for remaining mutations, else read-only. |
| `sdk/src/golden/golden-mutation-covered.ts` | Canonicals exercised by **`mutation-subprocess.integration.test.ts`** (must match non-skipped tests). |
| `sdk/src/golden/mutation-subprocess.integration.test.ts` | Tmp fixture + `captureGsdToolsOutput` vs `registry.dispatch`; dual sandbox per comparison. |
| `sdk/src/golden/mutation-sandbox.ts` | `createMutationSandbox({ git?: boolean })` — copy fixture, optional `git init` + commit. |
| `sdk/src/golden/golden-policy.test.ts` | Calls `verifyGoldenPolicyComplete()` so every canonical is covered or excepted. |
**Invariant:** Every canonical from `getCanonicalRegistryCommands()` is either in `GOLDEN_PARITY_INTEGRATION_COVERED` or has an exception string—**never** leave orphans by removing tests.
---
## Reference pattern: porting like `scan-sessions` and `workstream.status`
These were fixed by **aligning the TypeScript handler with the CJS implementation**, then adding a row to `READ_ONLY_JSON_PARITY_ROWS`.
1. **Find the CJS source of truth**
- `scan-sessions`: `get-shit-done/bin/lib/profile-pipeline.cjs` → `cmdScanSessions`
- `workstream status`: `get-shit-done/bin/lib/workstream.cjs` → `cmdWorkstreamStatus`
- `gsd-tools.cjs` `runCommand` switch shows the top-level command and argv.
2. **Implement or adjust the SDK module**
- Example: `sdk/src/query/profile-scan-sessions.ts` mirrors the project-array build from `cmdScanSessions`; `scanSessions` in `profile.ts` parses `--path` / `--verbose`, throws when no sessions root (same error text as CJS), returns `{ data: projects }` where `projects` matches CJS JSON array.
3. **Add a parity row** in `read-only-golden-rows.ts` with `canonical`, `sdkArgs`, `cjs`, `cjsArgs` (must match what `execFile(node, [gsdToolsPath, command, ...args])` expects).
4. **Run**
`cd sdk && npm run build && npx vitest run src/golden/read-only-parity.integration.test.ts src/golden/golden-policy.test.ts --project integration --project unit`
5. **Policy**
`readOnlyGoldenCanonicals()` picks up new canonicals automatically; no manual duplicate if the canonical is already in the JSON row list.
**When not to copy line-for-line:** subprocess-only concerns (e.g. `agents_installed` / `missing_agents` differing from in-process `~` resolution). Then **normalize in the test** (see `golden.integration.test.ts` `docs-init`: sort `existing_docs`, omit install fields)—**document in QUERY-HANDLERS.md**, do not delete the assertion.
---
## Completed — Track A (golden parity)
All 127 portable canonicals have subprocess or in-process parity coverage. Summary of completed work by batch:
### Profile-output + milestone subprocess batch (latest)
**`write-profile`**, **`generate-claude-profile`**, **`generate-dev-preferences`**, **`generate-claude-md`** — implemented in **`sdk/src/query/profile-output.ts`** (templates from `get-shit-done/templates/`, same JSON as `profile-output.cjs`); re-exported from **`profile.ts`**. **`milestone.complete`** — full port of **`cmdMilestoneComplete`** in **`phase-lifecycle.ts`**; **`readModifyWriteStateMdFull`** in **`state-mutation.ts`** for STATE writes matching CJS.
### Mutation subprocess infrastructure
**`mutation-subprocess.integration.test.ts`** — tmp fixture `sdk/src/golden/fixtures/mutation-project/` + `createMutationSandbox()` (`mutation-sandbox.ts`). **`assertJsonParity`** runs CJS and SDK on **two fresh sandboxes** (factory fn) so neither run sees the other's filesystem mutations. **`GOLDEN_MUTATION_SUBPROCESS_COVERED`** lists canonicals with non-skipped subprocess assertions. Handlers covered: `config-ensure-section`, `commit`, `commitToSubrepo`, `configSetModelProfile`, `state.patch`, `frontmatter.set`/`merge`, `workstream.progress`, `workstream.set`, nine `state.*` subprocess tests, `write-profile`, `generate-claude-profile`, `generate-dev-preferences`, `generate-claude-md`, `milestone.complete`, `init.remove-workspace`.
### CJS mutation handler alignment
`commit.ts` — `--files` argv boundary, `commitToSubrepo` config check, `checkCommit` `allowed` field. `state-mutation.ts` — `readModifyWriteStateMdFull`, `statePlannedPhase`=`cmdStatePlannedPhase`, record-session/add-decision/add-blocker/resolve-blocker/record-metric/update-progress JSON shapes. `phase-lifecycle.ts` — `milestone.complete`. `workstream.ts` — `workstream.progress` (`cmdWorkstreamProgress`), `workstream.set`. `roadmap.ts` — extracted `roadmapUpdatePlanProgress` to own module. `frontmatter-mutation.ts` — `--field`/`--value`, `--data` parsing. `config-mutation.ts` — `configSetModelProfile` CJS-shaped `{ updated, profile, previousProfile, agentToModelMap }`. `config-query.ts` — `getAgentToModelMapForProfile()`.
### Read-only parity rows (earlier batches)
`progress.table` / `stats.table`, `progress.bar`, `learnings.query`, `profile-questionnaire`, `verify.references`, `init.*` composition goldens (9 handlers), `profile-sample`, `extract-messages`, `uat.render-checkpoint`, `validate.agents` + `state.get`, `skill-manifest`, `audit-open` + `audit-uat`, `intel.extract-exports`, `summary-extract` + `history-digest`, `stats.json`, `todo.match-phase`, `verify.key-links`, `verify.schema-drift`, `state-snapshot`, `state.json`/`state.load`, `scan-sessions`, `workstream.status`.
---
## Next batch — summary / audit / skill / validate / UAT / intel / profile / init
**Same workflow as above:** read `gsd-tools.cjs` `runCommand` for argv → implement/adjust `sdk/src/query/*.ts` → add `READ_ONLY_JSON_PARITY_ROWS` and/or a **named `describe` block** with documented omissions → `npm run build` → `read-only-parity.integration.test.ts` + `golden-policy.test.ts`.
| Priority | Command (CLI) | `gsd-tools.cjs` case / args | CJS implementation | SDK module | Notes |
| -------- | ------------- | -------------------------- | -------------------- | ---------- | ----- |
| ~~1~~ | ~~`summary-extract <path>`~~ `[--fields a,b]` | `summary-extract` | `commands.cjs` `cmdSummaryExtract` (~L425) | `summary.ts` `summaryExtract` | **Done:** strict `READ_ONLY_JSON_PARITY_ROWS`; `summary.ts` aligned with `commands.cjs`; `extractFrontmatterLeading` in `frontmatter.ts` for first-`---`-block parity with `frontmatter.cjs`. |
| ~~2~~ | ~~`history-digest`~~ | `history-digest` | `commands.cjs` `cmdHistoryDigest` (~L133) | `summary.ts` `historyDigest` | **Done:** same row / handler alignment as above. |
| ~~3~~ | ~~`audit-open`~~ | `audit-open` `[--json]` | `audit.cjs` `auditOpenArtifacts` + optional `formatAuditReport` | `audit-open.ts` | **Done:** `--json` parity test + `scanned_at` normalization; `sanitizeForDisplay` = `security.cjs`. |
| ~~4~~ | ~~`audit-uat`~~ | `audit-uat` | `uat.cjs` `cmdAuditUat` | `uat.ts` `auditUat` | **Done:** `auditUat` ports `cmdAuditUat` (`parseUatItems`, milestone filter, `summary.by_*`); strict `READ_ONLY_JSON_PARITY_ROWS` row. |
| ~~5~~ | ~~`skill-manifest`~~ | `skill-manifest` + args | `init.cjs` `cmdSkillManifest` (~L1829) | `skill-manifest.ts` | **Done:** strict row; `extractFrontmatterLeading` for CJS parity (see `QUERY-HANDLERS.md`). |
| ~~6~~ | ~~`validate agents`~~ | `validate` + `agents` | `verify.cjs` `cmdValidateAgents` (~L997) | `validate.ts` `validateAgents` | **Done:** strict row; `getAgentsDir` parity with `core.cjs`; `MODEL_PROFILES` includes `gsd-pattern-mapper` (sync with `model-profiles.cjs`). |
| ~~7~~ | ~~`uat render-checkpoint --file <path>`~~ | `uat` subcommand | `uat.cjs` `cmdRenderCheckpoint` | `uat.ts` `uatRenderCheckpoint` | **Done:** strict row; fixture `sdk/src/golden/fixtures/uat-render-checkpoint-sample.md`; see `QUERY-HANDLERS.md`. |
| ~~8~~ | ~~`intel extract-exports <file>`~~ | `intel` `extract-exports` | `intel.cjs` `intelExtractExports` (~L502) | `intel.ts` `intelExtractExports` | **Done:** strict row + handler parity with `intel.cjs` (fixed file e.g. `sdk/src/query/utils.ts`). |
| ~~9~~ | ~~`extract-messages`~~ | `extract-messages` + project/session flags | `profile-pipeline.cjs` | `profile.ts` `extractMessages` | **Done:** `profile-extract-messages.ts` + golden `output_file` strip + JSONL compare; fixture `extract-messages-sessions/`. |
| ~~10~~ | ~~`profile-sample`~~ | `profile-sample` | `profile-pipeline.cjs` | `profile.ts` `profileSample` | **Done:** `profile-sample.ts` + golden `output_file` strip + JSONL compare; fixture `profile-sample-sessions/`. |
| ~~11~~ | ~~**`init.*` read-only JSON**~~ | various | `init.cjs` / `init-complex` | `init.ts`, `init-complex.ts` | **Done:** `golden.integration.test.ts` + nine init composition tests; `withProjectRoot` / `subagent_timeout` / `GOLDEN_INTEGRATION_MAIN_FILE_CANONICALS`; see `QUERY-HANDLERS.md`. |
**Suggested order:** Audit/read-only batch above is complete — follow-ups via **`GOLDEN_PARITY_EXCEPTIONS`** / new strict rows as needed (`learnings.query`, `progress.bar`, `profile-questionnaire`, etc.).
**Done (this line of work):** `summary-extract` + `history-digest` — strict `READ_ONLY_JSON_PARITY_ROWS`; `summary.ts` aligned with `commands.cjs`; `extractFrontmatterLeading` in `frontmatter.ts` for first-`---`-block parity with `frontmatter.cjs`.
**Done (profile-output + milestone mutation batch):** `write-profile`, `generate-claude-profile`, `generate-dev-preferences`, `generate-claude-md` (`profile-output.ts`); `milestone.complete` (`phase-lifecycle.ts` + `readModifyWriteStateMdFull`); `GOLDEN_MUTATION_SUBPROCESS_COVERED` updated; **`MUTATION_SUBPROCESS_GAP_REASON` removed** from `golden-policy.ts`.
**Mutations** (`QUERY_MUTATION_COMMANDS`): subprocess coverage is **`mutation-subprocess.integration.test.ts`** + `GOLDEN_MUTATION_SUBPROCESS_COVERED`. Remaining mutation canonicals without a subprocess row use **`MUTATION_DEFERRED_REASON`** (see `golden-policy.ts`). For known gaps before parity, prefer **`it.skip`** with an explicit rationale in code comments or restore a dedicated gap map — do not rely on silent deferral alone.
---
## Backlog: other read-only handlers (lower priority or follow-ups)
Confirm against `GOLDEN_PARITY_EXCEPTIONS` in `golden-policy.ts` for the live list.
**Mutations:** Prefer tmp fixture + dual sandbox (see `mutation-sandbox.ts`). Do not green the suite by deleting subprocess tests; skip with **`it.skip`** and document the gap (policy entry or comment) until parity is restored.
---
## Not in the SDK registry (product decision)
- **`graphify`**, **`from-gsd2` / `gsd2-import`** — CLI-only; no registry handler.
---
## Files to know (updated)
| Path | Role |
| ---- | ---- |
| `sdk/src/query/index.ts` | `createRegistry()`, `QUERY_MUTATION_COMMANDS`. |
| `sdk/src/golden/golden-policy.ts` | Coverage set + exceptions; `verifyGoldenPolicyComplete()`. |
| `sdk/src/golden/read-only-golden-rows.ts` | Strict read-only JSON matrix. |
| `sdk/src/golden/read-only-parity.integration.test.ts` | Subprocess + dispatch parity tests. |
| `sdk/src/golden/capture.ts` | `captureGsdToolsOutput`, `captureGsdToolsStdout`. |
| `sdk/src/golden/fixtures/mutation-project/` | Ephemeral copy for mutation subprocess tests. |
| `sdk/src/golden/mutation-subprocess.integration.test.ts` | Mutation handler subprocess parity. |
| `sdk/src/golden/mutation-sandbox.ts` | `createMutationSandbox({ git?: boolean })`. |
| `sdk/src/query/profile-output.ts` | CJS-parity profile output handlers. |
| `sdk/src/phase-runner.ts` | **Track C target** — currently uses `GSDTools`. |
| `sdk/src/init-runner.ts` | **Track C target** — currently uses `GSDTools`. |
| `sdk/src/gsd-tools.ts` | Subprocess bridge; **not deleted** in Phase 3 scope. |
| `get-shit-done/bin/gsd-tools.cjs` | `runCommand` — argv routing. Has `@deprecated` header. |
| `get-shit-done/bin/lib/*.cjs` | Per-command implementations (CJS source of truth). |
---
## Commands (verification)
```bash
cd sdk
npm run build
npm run test:unit
npm run test:integration
```
Focused:
```bash
npx vitest run src/golden/read-only-parity.integration.test.ts src/golden/golden.integration.test.ts --project integration
npx vitest run src/golden/mutation-subprocess.integration.test.ts --project integration
npx vitest run src/golden/golden-policy.test.ts --project unit
```
---
## Success criteria (extend, not replace)
- **No regression:** `golden-policy.test.ts` / `verifyGoldenPolicyComplete()` stays green.
- **Track A complete:** 127/128 covered; read-only rows, mutation subprocess, composition goldens all in place.
- **Track C:** Runner alignment — `PhaseRunner` and `InitRunner` use typed handlers where possible; `GSDTools` remains exported.
- **CHANGELOG.md** [Unreleased] updated with Phase 3 entries.
- **`QUERY-HANDLERS.md`** updated when assertion style changes (full `toEqual` vs normalized subset).
**Do not "green the suite" by deleting or shrinking golden tests.** If a handler cannot match CJS byte-for-byte without product decisions, use **documented normalization** in the test or **fix the TypeScript handler** — do not silently remove assertions.
---
## Commit history (this branch)
62 commits ahead of `main` on `feat/sdk-phase3-query-layer`. Recent batch (5 commits):
```
95db59c docs(sdk): update handover for profile-output and mutation subprocess batch
05e8238 sdk(golden): mutation subprocess test infrastructure and golden policy
593d9be sdk(query): port profile output handlers from profile-output.cjs
a2d0eb6 sdk(query): CJS parity for state, phase-lifecycle, workstream, roadmap, frontmatter, config, and intel
8bd9f1d sdk(query): align commit handler with CJS --files argv and allowed field
```
**Cherry-pick notes:** Commits 1 (`8bd9f1d`) and 3 (`593d9be`) are independently cherry-pickable. Commit 2 (`a2d0eb6`) is a bulk handler alignment (13 files). Commit 4 (`05e8238`) depends on handlers from 2+3 at test-runtime but compiles independently. Commit 5 is docs-only.
---
*Update this file when registry or golden milestones change.*

View File

@@ -1,97 +0,0 @@
# Handover: Parity exceptions doc + CJS-only matrix (next session)
**Status:** The deliverables described below are implemented in `sdk/src/query/QUERY-HANDLERS.md` (sections **Golden parity: coverage and exceptions** and **CJS command surface vs SDK registry**). Use that file as the canonical registry + parity reference; this handover remains useful for issue **#2302** scope and parent **#2007** links.
Paste this document (or `@sdk/HANDOVER-PARITY-DOCS.md`) at the start of a new chat so work continues without re-auditing issue scope.
## Goal for this session
1. **Parity “exceptions” documentation** — A clear, maintainable description of where **full JSON equality** between `gsd-tools.cjs` and `createRegistry()` is **not** expected or not attempted, and why (stubs, structural-only tests, environment-dependent fields, ordering, etc.). Map this to **#2007 / #2302** expectations: no *undocumented* gap.
2. **CJS-only matrix** — A **single authoritative table**: each relevant `gsd-tools.cjs` surface (top-level command or documented cluster) → **registered in SDK** vs **permanent CLI-only** vs **alias / naming difference**, with a **one-line justification** where not registered.
## Parent tracking
- **Issue:** [open-gsd/get-shit-done-redux#2302](https://github.com/open-gsd/get-shit-done-redux/issues/2302) — Phase 3 SDK query parity, registry, docs (parent umbrella #2007).
- **Acceptance criteria touched here:** parity coverage/exceptions documented; registry audit reflected in a **matrix** (issue wording: “every required CJS surface either has a handler or appears in the CJS-only matrix with justification”).
## Repo / branch
- **Workspace:** `D:\Repos\get-shit-done` (PBR backport); adjust path if different machine.
- **Feature branch (typical):** `feat/sdk-phase3-query-layer` — confirm with `git branch` before editing.
- **Upstream:** `open-gsd/get-shit-done-redux`.
## What already exists (do not duplicate blindly)
- `sdk/src/query/QUERY-HANDLERS.md` — Registry conventions, partial “not registered” list (**graphify**, **from-gsd2**), CLI name differences (**summary-extract** vs **summary.extract**, **scaffold** vs **phase.scaffold**), **intel.update** (CJS JSON parity; refresh via agent), **skill-manifest --write** / mutation events, **docs-init** golden note (agent install fields), **stateExtractField** rule.
- `sdk/src/golden/golden.integration.test.ts` — Source of truth for **which commands** are golden-tested and **how** (full equality vs subset vs normalized `existing_docs` vs omitted fields; `init.quick` strips clock-derived keys via `init-golden-normalize.ts`).
- `sdk/src/golden/capture.ts` — `captureGsdToolsOutput()` spawns `get-shit-done/bin/gsd-tools.cjs`.
- `docs/CLI-TOOLS.md` — User-facing CLI reference; should **link** to the parity exceptions + matrix (or host a short summary with pointer to `sdk/`).
## Deliverables (suggested shape)
### A) Parity exceptions section
Add or extend a dedicated section (prefer `QUERY-HANDLERS.md` under a heading like **"Golden parity: coverage and exceptions"**, or a new `sdk/PARITY.md` if the team wants less churn in QUERY-HANDLERS — **pick one canonical location** and link from the other).
Cover at least:
| Category | Examples to document |
| ----------------------------- | ------------------------------------------------------------------------------------------------------------------------------------- |
| **Full JSON parity** | Commands where tests use `toEqual` on `sdkResult.data` vs CJS stdout JSON. |
| **Structural / field subset** | Tests that compare only selected keys (e.g. `frontmatter.get`, `find-phase` — SDK subset vs CJS). Full parity for `roadmap.analyze`, `init.*` (except `init.quick` volatile keys), etc. — see `QUERY-HANDLERS.md` matrix. |
| **Normalized comparison** | e.g. `docs-init`: `existing_docs` sorted by path; `agents_installed` / `missing_agents` omitted between subprocess vs in-process. |
| **CLI parity without in-process refresh** | `intel.update` — JSON matches CJS `intel.cjs` (spawn hint or disabled); refresh is agent-driven. |
| **Conditional behavior** | `skill-manifest`: writes only with `--write`; not in `QUERY_MUTATION_COMMANDS`. |
| **Environment / time** | `current-timestamp`: structure and format, not same instant. |
| **Not in golden suite** | Commands registered but not (yet) covered — list as **coverage gap** or **out of scope for golden** with rationale. |
### B) CJS-only matrix
Build the table by **diffing** `get-shit-done/bin/gsd-tools.cjs` `switch (command)` top-level cases against `createRegistry()` registrations in `sdk/src/query/index.ts`.
**Already documented as product-out-of-scope for registry:** **graphify**, **from-gsd2** / **gsd2-import**.
**Already documented as naming/alias differences (registered, different string):** **summary-extract** ↔ **summary.extract**; top-level **scaffold** ↔ **phase.scaffold**.
Matrix columns (suggested):
- **CJS command** (or subcommand pattern)
- **SDK dispatch name(s)** if any
- **Disposition:** Registered / CLI-only / Alias-only / Stub / N/A
- **Justification** (one line) if not a straight registered parity
Optional: footnote that `detect-custom-files` skips multi-repo root resolution in CJS (`SKIP_ROOT_RESOLUTION`) — behavior is documented in CLI; matrix can mention if relevant.
## Files likely to edit
| Path | Role |
| --------------------------------- | ----------------------------------------------------------------- |
| `sdk/src/query/QUERY-HANDLERS.md` | Primary home for exceptions + matrix, or link hub. |
| `sdk/PARITY.md` | Optional dedicated file if QUERY-HANDLERS becomes too long. |
| `docs/CLI-TOOLS.md` | Short “Parity & registry” subsection with links into `sdk/` docs. |
| `sdk/HANDOVER-GOLDEN-PARITY.md` | Optional one-line pointer to new parity doc section when done. |
## Out of scope for *this* handover session
- Implementing runner alignment (`GSDTools` → registry) — separate #2302 work.
- Adding `@deprecated` headers to `gsd-tools.cjs` — separate task.
- **CHANGELOG** — only if you batch doc work with release notes in same PR (optional).
## Verification
- No code behavior change required for pure docs; run `npm run build` in `sdk/` only if TypeScript-adjacent files were touched.
- Proofread: every **CLI-only** row has a **justification**; every **exception** in golden tests appears in the exceptions doc.
## Success criteria
- A reader can answer: **“Which commands are fully golden-parity vs partial vs stub vs untested?”** without reading the whole test file.
- A reader can answer: **“Which `gsd-tools` top-level commands are not registered and why?”** from one table.
- **#2302** acceptance bullets on parity documentation and registry matrix are satisfied for the **documentation** slice (remaining issue items may still be open for code).
---
*Created for handoff to “parity exceptions + CJS-only matrix” session. Update when the canonical doc location or golden coverage changes.*

View File

@@ -1,170 +0,0 @@
# Handover: SDK query layer (registry, CLI, parity docs)
Paste this document (or `@sdk/HANDOVER-QUERY-LAYER.md`) at the start of a new session so work continues without re-deriving scope.
## Parent tracking
- **Issue:** [open-gsd/get-shit-done-redux#2302](https://github.com/open-gsd/get-shit-done-redux/issues/2302) — Phase 3 SDK query parity, registry, docs (umbrella #2007).
- **Workspace:** `D:\Repos\get-shit-done` (PBR backport). **Upstream:** `open-gsd/get-shit-done-redux`. Confirm branch with `git branch` (typical: `feat/sdk-phase3-query-layer`).
### Scope anchors (do not confuse issues)
| Role | GitHub | Notes |
| --------------------------------------- | -------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| **Product / requirements anchor** | [#2007](https://github.com/open-gsd/get-shit-done-redux/issues/2007) | Problem statement, user stories, and target architecture for the SDK-first migration. **Do not** treat its original acceptance-checklist boxes as proof of what is merged upstream; work was split into phased PRs after maintainer review. |
| **Phase 3 execution scope** | [#2302](https://github.com/open-gsd/get-shit-done-redux/issues/2302) **+ this handover** | What this branch is actually doing now: registry/CLI parity, docs, harness gaps, runner alignment follow-ups as listed below. |
| **Patch mine (if local tree is short)** | [PR #2008](https://github.com/open-gsd/get-shit-done-redux/pull/2008) and matching branches | Large pre-phasing PR; cherry-pick or compare when something looks missing vs that line of work. |
---
## What was delivered (this line of work)
### 1. Parity documentation (`QUERY-HANDLERS.md`)
- **”Golden parity: coverage and exceptions”** — How `golden.integration.test.ts` compares SDK vs `gsd-tools.cjs` (full `toEqual`, subset, normalized `docs-init`, `intel.update` CJS parity, time-dependent fields, etc.).
- **”CJS command surface vs SDK registry”** — Naming aliases, CLI-only rows, SDK-only rows, and a **top-level `gsd-tools` command → SDK** matrix.
- `docs/CLI-TOOLS.md` — Short “Parity & registry” pointer into those sections.
- `HANDOVER-GOLDEN-PARITY.md` — One paragraph linking to the same sections.
### 2. `gsd-sdk query` tokenization (`normalizeQueryCommand`)
- **Problem:** `gsd-sdk query` used only argv[0] as the registry key, so `query state json` dispatched `state` (unregistered) instead of `state.json`.
- **Fix:** `sdk/src/query/normalize-query-command.ts` merges the same **command + subcommand** patterns as `gsd-tools` `runCommand()` (e.g. `state json` → `state.json`, `init execute-phase 9` → `init.execute-phase`, `scaffold …` → `phase.scaffold`, `progress bar` → `progress.bar`). Wired in `sdk/src/cli.ts` before `registry.dispatch()`.
- **Tests:** `sdk/src/query/normalize-query-command.test.ts`.
### 3. `phase add-batch` in the registry
- **Implementation:** `phaseAddBatch` in `sdk/src/query/phase-lifecycle.ts` — port of `cmdPhaseAddBatch` from `get-shit-done/bin/lib/phase.cjs` (batch append under one roadmap lock; sequential or `phase_naming: custom`).
- **Registration:** `phase.add-batch` and `phase add-batch` in `sdk/src/query/index.ts`; listed in `QUERY_MUTATION_COMMANDS` (dotted + space forms).
- **Tests:** `describe('phaseAddBatch')` in `sdk/src/query/phase-lifecycle.test.ts`.
- **Docs:** `QUERY-HANDLERS.md` updated — `phase add-batch` is **registered**; CLI-only table no longer lists it.
### 4. `state load` fully in the registry (split from `state json`)
Previously `state.json` and `state.load` were easy to confuse: CJS has two different commands — `cmdStateJson` (`state json`, rebuilt frontmatter) vs `cmdStateLoad` (`state load`, `loadConfig` + `state_raw` + existence flags).
- `stateJson` — `sdk/src/query/state.ts`; registry key `state.json`.
- `stateProjectLoad` — `sdk/src/query/state-project-load.ts`; registry key `state.load`. Uses `createRequire` to call `core.cjs` `loadConfig(projectDir)` from the same resolution paths as a normal install (bundled monorepo path, `projectDir/.claude/get-shit-done/...`, `~/.claude/get-shit-done/...`). `GSDTools.stateLoad()` and `formatRegistryRawStdout` for `--raw` no longer force a subprocess solely for this command.
- **Risk:** If `core.cjs` is absent (e.g. some `@opengsd/gsd-sdk`-only layouts), `state.load` throws `GSDError` — document; future option is a TS `loadConfig` port or bundling.
- **Goldens:** `read-only-parity.integration.test.ts` — one block compares `state.json` to `state json` (strip `last_updated`); another compares `state.load` to `state load` (full `toEqual`). `read-only-golden-rows.ts` `readOnlyGoldenCanonicals()` includes both `state.json` and `state.load`.
---
## Query surface completeness (snapshot)
| Status | Surface |
| ------------------------ | ------------------------------------------------------------------------------------------------ |
| **Registered** | Essentially all `gsd-tools.cjs` `runCommand` surfaces, including `phase.add-batch`. |
| **CLI-only (by design)** | `graphify`, `from-gsd2` — not in `createRegistry()`; documented in `QUERY-HANDLERS.md`. |
| **SDK-only extra** | `phases.archive` — no `gsd-tools phases archive` subcommand (CJS has `list` / `clear` only). |
**Programmatic API:** `createRegistry()` / `registry.dispatch('dotted.name', args, projectDir)`.
**CLI:** `gsd-sdk query …` — apply `normalizeQueryCommand` semantics (or pass dotted names explicitly).
**Still not unified:** `GSDTools` (`sdk/src/gsd-tools.ts`) shells out to `gsd-tools.cjs` for plan/session flows; migrating callers to the registry is separate #2302 / runner work. `state load` is **not** among the subprocess-only exceptions anymore (it uses the registry like other native query handlers when native query is active).
---
## Canonical files
| Path | Role |
| ------------------------------------------- | -------------------------------------------------------------------------------------- |
| `sdk/src/query/index.ts` | `createRegistry()`, `QUERY_MUTATION_COMMANDS`, handler wiring. |
| `sdk/src/query/state-project-load.ts` | `state.load` — CJS `cmdStateLoad` parity (`loadConfig` + `state_raw` + flags). |
| `sdk/src/query/normalize-query-command.ts` | CLI argv → registry command string. |
| `sdk/src/cli.ts` | `gsd-sdk query` path (uses `normalizeQueryCommand`). |
| `sdk/src/query/QUERY-HANDLERS.md` | Registry contracts, parity tiers, CJS matrix, mutation notes. |
| `sdk/src/golden/golden.integration.test.ts` | Golden parity vs `captureGsdToolsOutput()`. |
| `docs/CLI-TOOLS.md` | User-facing CLI; links to parity sections. |
Related handovers: `HANDOVER-GOLDEN-PARITY.md`, `HANDOVER-PARITY-DOCS.md` (older parity-doc brief; content largely folded into `QUERY-HANDLERS.md`).
---
## Roadmap: parity vs decision offloading
Work that moves **deterministic** orchestration out of AI/bash and into **SDK queries** (historically `gsd-tools.cjs`) has **two layers**. Do not confuse them:
| Layer | Goal | What “done” looks like |
| ------------------------ | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- |
| **Parity / migration** | Existing CLI behavior is **stable and testable** in the registry so callers can use `gsd-sdk query` instead of `node …/gsd-tools.cjs` without silent drift. | Goldens + `QUERY-HANDLERS.md`; same JSON/`--raw` contracts as CJS. |
| **Offloading decisions** | **New or consolidated** queries replace repeated `grep`, `ls` piped to `wc -l`, many `config-get`s, and inline `node -e` in workflows — so the model does less parsing and branching. | Fewer inline shell blocks; measurable token/step reduction on representative workflows. |
Phase 3–style registry work mainly advances **parity**. The `decision-routing-audit.md` proposals are mostly **offloading** — they assume parity exists for commands workflows already call.
### Decision-routing audit (proposed `gsd-tools` / SDK queries)
Source: `.planning/research/decision-routing-audit.md` §3. **Tier** = priority from §5 (implementation order). **Do not implement** = explicitly rejected in the audit.
| # | Proposed command | Tier | Notes |
|---|------------------|------|--------|
| 3.1 | `route next-action` | **1** | Next slash-command from `/gsd-next`-style routing. |
| 3.2 | `check gates <workflow>` | 3 | Safety gates (continue-here, error state, verification debt). |
| 3.3 | `check config-gates <workflow>` | **1** | Batch `workflow.*` config for orchestration (replaces many `config-get`s). |
| 3.4 | `check phase-ready <phase>` | **1** | Phase directory readiness + `next_step` hint. |
| 3.5 | `check auto-mode` | 2 | `auto_advance` + `_auto_chain_active` → single boolean. |
| 3.6 | `detect phase-type <phase>` | 2 | Structured UI/schema detection (replaces fragile grep). |
| 3.7 | `check completion <scope>` | 2 | Phase or milestone completion rollup. |
| 3.8 | `check verification-status <phase>` | 3 | VERIFICATION.md parsing for routing. |
| 3.9 | `check ship-ready <phase>` | 3 | Ship preflight (`ship.md`). |
| 3.10 | `route workflow-steps <workflow>` | ❌ **Do not implement** | Pre-computed step lists are unsound when mid-workflow writes change state. See `review-and-risks.md` §3.6. |
**Not in audit:** `phase-artifact-counts` was only an example in an older handover line; there is no §3.11 for it — add via a new research doc if needed.
**SDK registry (Tier 1):** **Done** — `check.config-gates`, `check.phase-ready`, `route.next-action` in `createRegistry()` (`sdk/src/query/index.ts`). Documented in `sdk/src/query/QUERY-HANDLERS.md` § Decision routing (**SDK-only** until/unless mirrored in `gsd-tools.cjs`).
**Simple roadmap (execute in order):**
1. **Harden parity** for surfaces workflows already depend on (registry dispatch, goldens, docs) so swaps from CJS to `gsd-sdk query` stay safe.
2. **Ship 1–2 high-leverage consolidation handlers** from the audit (pick based on impact and risk; examples: `check auto-mode`, `phase-artifact-counts`, `route next-action` — with **display/routing fields** required by `review-and-risks.md` if applicable). Each needs handlers, tests, and `QUERY-HANDLERS.md` notes. **Progress:** `check.auto-mode` shipped (`sdk/src/query/check-auto-mode.ts`); Tier 1 `route.next-action` already registered.
3. **Rewrite one heavy workflow** (e.g. `next.md` or a focused slice of `autonomous.md`) to consume those queries and **measure** before/after (steps, tokens, or both). **Progress:** `execute-phase.md`, `discuss-phase.md`, `discuss-phase-assumptions.md`, and `plan-phase.md` (UI gate) now use `check auto-mode` instead of paired `config-get`s where applicable.
4. **Maintain a living boundary** between SDK (**data, deterministic checks**) and workflows (**judgment, sequencing, user-facing messages**). Extend `decision-routing-audit.md` §6 (decisions that stay with the AI) and `review-and-risks.md` “Do not implement” (e.g. no pre-computed `route workflow-steps`) as you add primitives. **Progress:** audit §3.5 / Tier 2 #4 updated to reference SDK implementation.
**Gaps to keep in mind when designing new queries:** call-time vs stale data after file writes (re-query volatile fields); workflows own gates/UX; behavioral contracts (e.g. UI keyword lists) must match existing greps; `stderr`/`stdout` and JSON shapes stable for bash/`jq`; hybrid `require(core.cjs)` paths called out for minimal installs.
**Research references (repo root):** `.planning/research/decision-routing-audit.md`, `.planning/research/review-and-risks.md`, `.planning/research/inline-computation-audit.md`, `.planning/research/questions.md` (Q1 boundary). For parity mechanics, prefer `sdk/src/query/QUERY-HANDLERS.md` and `HANDOVER-GOLDEN-PARITY.md`.
---
## Suggested next session
(Strategic ordering of **parity vs decision offloading** is in **Roadmap** above.)
1. ~~**Golden test for `phase.add-batch`**~~ — Done: `sdk/src/golden/mutation-subprocess.integration.test.ts` (`phase.add-batch` JSON parity vs CJS).
2. ~~**Re-export `normalizeQueryCommand`**~~ — Done: exported from `sdk/src/query/index.ts` and `sdk/src/index.ts` (`@opengsd/gsd-sdk`).
3. **Issue #2302 follow-ups** — Runner alignment (`GSDTools` → registry where appropriate). **`configGet`** now uses `dispatchNativeJson` with canonical `config-get` (fixes subprocess argv vs real `gsd-tools.cjs`, which has no `config` + `get` top-level). Keep `graphify` / `from-gsd2` out of scope unless product reopens.
4. **Drift check** — When adding CJS commands, update `QUERY-HANDLERS.md` matrix and golden docs in the same PR.
---
## Verification commands
```bash
cd sdk
npm run build
npx vitest run src/query/normalize-query-command.test.ts src/query/phase-lifecycle.test.ts src/query/registry.test.ts --project unit
npx vitest run src/golden/golden.integration.test.ts --project integration
```
(Adjust `--project` to match `sdk/vitest.config.ts`.)
---
## Success criteria (query-layer slice)
- Parity expectations and CJS↔SDK matrix documented in one place (`QUERY-HANDLERS.md`).
- `gsd-sdk query` understands two-token command patterns like `gsd-tools`.
- `phase add-batch` implemented and registered; **only** intentional CLI-only gaps remain (**graphify**, **from-gsd2**).
---
*Created/updated for query-layer handoff. Revise when registry surface, golden coverage, or the parity/offloading roadmap changes materially.*

View File

@@ -1,53 +0,0 @@
# @opengsd/gsd-sdk
TypeScript SDK for **Get Shit Done**: deterministic query/mutation handlers, plan execution, and event-stream telemetry so agents focus on judgment, not shell plumbing.
## Install
```bash
npm install @opengsd/gsd-sdk
```
## Quickstart — programmatic
```typescript
import { GSD, createRegistry } from '@opengsd/gsd-sdk';
const gsd = new GSD({ projectDir: process.cwd(), sessionId: 'my-run' });
const tools = gsd.createTools();
const registry = createRegistry(gsd.eventStream, 'my-run');
const { data } = await registry.dispatch('state.json', [], process.cwd());
```
## Quickstart — CLI
From a project that depends on this package, **invoke the CLI with Node** (recommended in CI and local dev):
```bash
node ./node_modules/@opengsd/gsd-sdk/dist/cli.js query state.json
node ./node_modules/@opengsd/gsd-sdk/dist/cli.js query roadmap.analyze
```
If no native handler is registered for a command, the CLI can transparently shell out to `get-shit-done/bin/gsd-tools.cjs` (see stderr warning), unless `GSD_QUERY_FALLBACK=off`.
## What ships
| Area | Entry |
|------|--------|
| Query registry | `createRegistry()` in `src/query/index.ts` — same handlers as `gsd-sdk query` |
| Tools bridge | `GSDTools` — native dispatch with optional CJS subprocess fallback |
| Orchestrators | `PhaseRunner`, `InitRunner`, `GSD` |
| CLI | `gsd-sdk` — `query`, `run`, `init`, `auto` |
## Guides
- **Handler registry & contracts:** [`src/query/QUERY-HANDLERS.md`](src/query/QUERY-HANDLERS.md)
- **Repository docs** (when present): `docs/ARCHITECTURE.md`, `docs/CLI-TOOLS.md` at repo root
## Environment
| Variable | Purpose |
|----------|---------|
| `GSD_QUERY_FALLBACK` | `off` / `never` disables CLI fallback to `gsd-tools.cjs` for unknown commands |
| `GSD_AGENTS_DIR` | Override directory scanned for installed GSD agents (`~/.claude/agents` by default) |

View File

@@ -1,68 +0,0 @@
# Prompt Caching Best Practices
When building applications on the GSD SDK, system prompts that include workflow instructions (executor prompts, planner context, verification rules) are large and stable across requests. Prompt caching avoids re-processing these on every API call.
## Recommended: 1-Hour Cache TTL
Use `cache_control` with a 1-hour TTL on system prompts that include GSD workflow content:
```typescript
const response = await client.messages.create({
model: 'claude-sonnet-4-20250514',
system: [
{
type: 'text',
text: executorPrompt, // GSD workflow instructions — large, stable across requests
cache_control: { type: 'ephemeral', ttl: '1h' },
},
],
messages,
});
```
### Why 1 hour instead of the default 5 minutes
GSD workflows involve human review pauses between phases — discussing results, checking verification output, deciding next steps. The default 5-minute TTL expires during these pauses, forcing full re-processing of the system prompt on the next request.
With a 1-hour TTL:
- **Cost:** 2x write cost on cache miss (vs. 1.25x for 5-minute TTL)
- **Break-even:** Pays for itself after 3 cache hits per hour
- **GSD usage pattern:** Phase execution involves dozens of requests per hour, well above break-even
- **Cache refresh:** Every cache hit resets the TTL at no cost, so active sessions maintain warm cache throughout
### Which prompts to cache
| Prompt | Cache? | Reason |
|--------|--------|--------|
| Executor system prompt | Yes | Large (~10K tokens), identical across tasks in a phase |
| Planner system prompt | Yes | Large, stable within a planning session |
| Verifier system prompt | Yes | Large, stable within a verification session |
| User/task-specific content | No | Changes per request |
### SDK integration point
In `session-runner.ts`, the `systemPrompt.append` field carries the executor/planner prompt. When using the Claude API directly (outside the Agent SDK's `query()` helper), wrap this content with `cache_control`:
```typescript
// In runPlanSession / runPhaseStepSession, the systemPrompt is:
systemPrompt: {
type: 'preset',
preset: 'claude_code',
append: executorPrompt, // <-- this is the content to cache
}
// When calling the API directly, convert to:
system: [
{
type: 'text',
text: executorPrompt,
cache_control: { type: 'ephemeral', ttl: '1h' },
},
]
```
## References
- [Anthropic Prompt Caching documentation](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching)
- [Extended caching (1-hour TTL)](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching#extended-caching)

2502
sdk/package-lock.json generated

File diff suppressed because it is too large Load Diff

View File

@@ -1,57 +0,0 @@
{
"name": "@opengsd/gsd-sdk",
"version": "1.0.0",
"description": "GSD SDK — programmatic interface for running GSD plans via the Agent SDK",
"type": "module",
"main": "dist/index.js",
"types": "dist/index.d.ts",
"exports": {
".": {
"import": "./dist/index.js",
"types": "./dist/index.d.ts"
}
},
"bin": {
"gsd-sdk": "./dist/cli.js"
},
"files": [
"dist",
"shared",
"prompts"
],
"repository": {
"type": "git",
"url": "git+https://github.com/open-gsd/get-shit-done-redux.git",
"directory": "sdk"
},
"homepage": "https://github.com/open-gsd/get-shit-done-redux/tree/main/sdk",
"bugs": {
"url": "https://github.com/open-gsd/get-shit-done-redux/issues"
},
"author": "TÂCHES",
"license": "MIT",
"publishConfig": {
"access": "public"
},
"engines": {
"node": ">=22.0.0"
},
"scripts": {
"build": "tsc",
"prepublishOnly": "rm -rf dist && tsc && chmod +x dist/cli.js",
"test": "vitest run",
"test:unit": "vitest run --project unit",
"test:integration": "vitest run --project integration"
},
"dependencies": {
"@anthropic-ai/claude-agent-sdk": "^0.2.84",
"ws": "8.20.1"
},
"devDependencies": {
"@types/node": "^22.0.0",
"@types/ws": "^8.18.1",
"tsx": "^4.22.0",
"typescript": "^5.7.0",
"vitest": "^3.1.1"
}
}

View File

@@ -1,186 +0,0 @@
# PROJECT.md Template
Template for `.planning/PROJECT.md` — the living project context document.
<template>
```markdown
# [Project Name]
## What This Is
[Current accurate description — 2-3 sentences. What does this product do and who is it for?
Use the user's language and framing. Update whenever reality drifts from this description.]
## Core Value
[The ONE thing that matters most. If everything else fails, this must work.
One sentence that drives prioritization when tradeoffs arise.]
## Requirements
### Validated
<!-- Shipped and confirmed valuable. -->
(None yet — ship to validate)
### Active
<!-- Current scope. Building toward these. -->
- [ ] [Requirement 1]
- [ ] [Requirement 2]
- [ ] [Requirement 3]
### Out of Scope
<!-- Explicit boundaries. Includes reasoning to prevent re-adding. -->
- [Exclusion 1] — [why]
- [Exclusion 2] — [why]
## Context
[Background information that informs implementation:
- Technical environment or ecosystem
- Relevant prior work or experience
- User research or feedback themes
- Known issues to address]
## Constraints
- **[Type]**: [What] — [Why]
- **[Type]**: [What] — [Why]
Common types: Tech stack, Timeline, Budget, Dependencies, Compatibility, Performance, Security
## Key Decisions
<!-- Decisions that constrain future work. Add throughout project lifecycle. -->
| Decision | Rationale | Outcome |
|----------|-----------|---------|
| [Choice] | [Why] | [✓ Good / ⚠️ Revisit / — Pending] |
---
*Last updated: [date] after [trigger]*
```
</template>
<guidelines>
**What This Is:**
- Current accurate description of the product
- 2-3 sentences capturing what it does and who it's for
- Use the user's words and framing
- Update when the product evolves beyond this description
**Core Value:**
- The single most important thing
- Everything else can fail; this cannot
- Drives prioritization when tradeoffs arise
- Rarely changes; if it does, it's a significant pivot
**Requirements — Validated:**
- Requirements that shipped and proved valuable
- Format: `- ✓ [Requirement] — [version/phase]`
- These are locked — changing them requires explicit discussion
**Requirements — Active:**
- Current scope being built toward
- These are hypotheses until shipped and validated
- Move to Validated when shipped, Out of Scope if invalidated
**Requirements — Out of Scope:**
- Explicit boundaries on what we're not building
- Always include reasoning (prevents re-adding later)
- Includes: considered and rejected, deferred to future, explicitly excluded
**Context:**
- Background that informs implementation decisions
- Technical environment, prior work, user feedback
- Known issues or technical debt to address
- Update as new context emerges
**Constraints:**
- Hard limits on implementation choices
- Tech stack, timeline, budget, compatibility, dependencies
- Include the "why" — constraints without rationale get questioned
**Key Decisions:**
- Significant choices that affect future work
- Add decisions as they're made throughout the project
- Track outcome when known:
- ✓ Good — decision proved correct
- ⚠️ Revisit — decision may need reconsideration
- — Pending — too early to evaluate
**Last Updated:**
- Always note when and why the document was updated
- Format: `after Phase 2` or `after v1.0 milestone`
- Triggers review of whether content is still accurate
</guidelines>
<evolution>
PROJECT.md evolves throughout the project lifecycle.
These rules are embedded in the generated PROJECT.md (## Evolution section)
and implemented by transition and milestone-completion workflows.
**After each phase transition:**
1. Requirements invalidated? → Move to Out of Scope with reason
2. Requirements validated? → Move to Validated with phase reference
3. New requirements emerged? → Add to Active
4. Decisions to log? → Add to Key Decisions
5. "What This Is" still accurate? → Update if drifted
**After each milestone:**
1. Full review of all sections
2. Core Value check — still the right priority?
3. Audit Out of Scope — reasons still valid?
4. Update Context with current state (users, feedback, metrics)
</evolution>
<brownfield>
For existing codebases:
1. **Map the codebase first** — analyze the project structure and existing code before defining requirements.
2. **Infer Validated requirements** from existing code:
- What does the codebase actually do?
- What patterns are established?
- What's clearly working and relied upon?
3. **Gather Active requirements** from user:
- Present inferred current state
- Ask what they want to build next
4. **Initialize:**
- Validated = inferred from existing code
- Active = user's goals for this work
- Out of Scope = boundaries user specifies
- Context = includes current codebase state
</brownfield>
<state_reference>
STATE.md references PROJECT.md:
```markdown
## Project Reference
See: .planning/PROJECT.md (updated [date])
**Core value:** [One-liner from Core Value section]
**Current focus:** [Current phase name]
```
This ensures Claude reads current PROJECT.md context.
</state_reference>

View File

@@ -1,231 +0,0 @@
# Requirements Template
Template for `.planning/REQUIREMENTS.md` — checkable requirements that define "done."
<template>
```markdown
# Requirements: [Project Name]
**Defined:** [date]
**Core Value:** [from PROJECT.md]
## v1 Requirements
Requirements for initial release. Each maps to roadmap phases.
### Authentication
- [ ] **AUTH-01**: User can sign up with email and password
- [ ] **AUTH-02**: User receives email verification after signup
- [ ] **AUTH-03**: User can reset password via email link
- [ ] **AUTH-04**: User session persists across browser refresh
### [Category 2]
- [ ] **[CAT]-01**: [Requirement description]
- [ ] **[CAT]-02**: [Requirement description]
- [ ] **[CAT]-03**: [Requirement description]
### [Category 3]
- [ ] **[CAT]-01**: [Requirement description]
- [ ] **[CAT]-02**: [Requirement description]
## v2 Requirements
Deferred to future release. Tracked but not in current roadmap.
### [Category]
- **[CAT]-01**: [Requirement description]
- **[CAT]-02**: [Requirement description]
## Out of Scope
Explicitly excluded. Documented to prevent scope creep.
| Feature | Reason |
|---------|--------|
| [Feature] | [Why excluded] |
| [Feature] | [Why excluded] |
## Traceability
Which phases cover which requirements. Updated during roadmap creation.
| Requirement | Phase | Status |
|-------------|-------|--------|
| AUTH-01 | Phase 1 | Pending |
| AUTH-02 | Phase 1 | Pending |
| AUTH-03 | Phase 1 | Pending |
| AUTH-04 | Phase 1 | Pending |
| [REQ-ID] | Phase [N] | Pending |
**Coverage:**
- v1 requirements: [X] total
- Mapped to phases: [Y]
- Unmapped: [Z] ⚠️
---
*Requirements defined: [date]*
*Last updated: [date] after [trigger]*
```
</template>
<guidelines>
**Requirement Format:**
- ID: `[CATEGORY]-[NUMBER]` (AUTH-01, CONTENT-02, SOCIAL-03)
- Description: User-centric, testable, atomic
- Checkbox: Only for v1 requirements (v2 are not yet actionable)
**Categories:**
- Derive from research FEATURES.md categories
- Keep consistent with domain conventions
- Typical: Authentication, Content, Social, Notifications, Moderation, Payments, Admin
**v1 vs v2:**
- v1: Committed scope, will be in roadmap phases
- v2: Acknowledged but deferred, not in current roadmap
- Moving v2 → v1 requires roadmap update
**Out of Scope:**
- Explicit exclusions with reasoning
- Prevents "why didn't you include X?" later
- Anti-features from research belong here with warnings
**Traceability:**
- Empty initially, populated during roadmap creation
- Each requirement maps to exactly one phase
- Unmapped requirements = roadmap gap
**Status Values:**
- Pending: Not started
- In Progress: Phase is active
- Complete: Requirement verified
- Blocked: Waiting on external factor
</guidelines>
<evolution>
**After each phase completes:**
1. Mark covered requirements as Complete
2. Update traceability status
3. Note any requirements that changed scope
**After roadmap updates:**
1. Verify all v1 requirements still mapped
2. Add new requirements if scope expanded
3. Move requirements to v2/out of scope if descoped
**Requirement completion criteria:**
- Requirement is "Complete" when:
- Feature is implemented
- Feature is verified (tests pass, manual check done)
- Feature is committed
</evolution>
<example>
```markdown
# Requirements: CommunityApp
**Defined:** 2025-01-14
**Core Value:** Users can share and discuss content with people who share their interests
## v1 Requirements
### Authentication
- [ ] **AUTH-01**: User can sign up with email and password
- [ ] **AUTH-02**: User receives email verification after signup
- [ ] **AUTH-03**: User can reset password via email link
- [ ] **AUTH-04**: User session persists across browser refresh
### Profiles
- [ ] **PROF-01**: User can create profile with display name
- [ ] **PROF-02**: User can upload avatar image
- [ ] **PROF-03**: User can write bio (max 500 chars)
- [ ] **PROF-04**: User can view other users' profiles
### Content
- [ ] **CONT-01**: User can create text post
- [ ] **CONT-02**: User can upload image with post
- [ ] **CONT-03**: User can edit own posts
- [ ] **CONT-04**: User can delete own posts
- [ ] **CONT-05**: User can view feed of posts
### Social
- [ ] **SOCL-01**: User can follow other users
- [ ] **SOCL-02**: User can unfollow users
- [ ] **SOCL-03**: User can like posts
- [ ] **SOCL-04**: User can comment on posts
- [ ] **SOCL-05**: User can view activity feed (followed users' posts)
## v2 Requirements
### Notifications
- **NOTF-01**: User receives in-app notifications
- **NOTF-02**: User receives email for new followers
- **NOTF-03**: User receives email for comments on own posts
- **NOTF-04**: User can configure notification preferences
### Moderation
- **MODR-01**: User can report content
- **MODR-02**: User can block other users
- **MODR-03**: Admin can view reported content
- **MODR-04**: Admin can remove content
- **MODR-05**: Admin can ban users
## Out of Scope
| Feature | Reason |
|---------|--------|
| Real-time chat | High complexity, not core to community value |
| Video posts | Storage/bandwidth costs, defer to v2+ |
| OAuth login | Email/password sufficient for v1 |
| Mobile app | Web-first, mobile later |
## Traceability
| Requirement | Phase | Status |
|-------------|-------|--------|
| AUTH-01 | Phase 1 | Pending |
| AUTH-02 | Phase 1 | Pending |
| AUTH-03 | Phase 1 | Pending |
| AUTH-04 | Phase 1 | Pending |
| PROF-01 | Phase 2 | Pending |
| PROF-02 | Phase 2 | Pending |
| PROF-03 | Phase 2 | Pending |
| PROF-04 | Phase 2 | Pending |
| CONT-01 | Phase 3 | Pending |
| CONT-02 | Phase 3 | Pending |
| CONT-03 | Phase 3 | Pending |
| CONT-04 | Phase 3 | Pending |
| CONT-05 | Phase 3 | Pending |
| SOCL-01 | Phase 4 | Pending |
| SOCL-02 | Phase 4 | Pending |
| SOCL-03 | Phase 4 | Pending |
| SOCL-04 | Phase 4 | Pending |
| SOCL-05 | Phase 4 | Pending |
**Coverage:**
- v1 requirements: 18 total
- Mapped to phases: 18
- Unmapped: 0 ✓
---
*Requirements defined: 2025-01-14*
*Last updated: 2025-01-14 after initial definition*
```
</example>

View File

@@ -1,204 +0,0 @@
# Architecture Research Template
Template for `.planning/research/ARCHITECTURE.md` — system structure patterns for the project domain.
<template>
```markdown
# Architecture Research
**Domain:** [domain type]
**Researched:** [date]
**Confidence:** [HIGH/MEDIUM/LOW]
## Standard Architecture
### System Overview
```
┌─────────────────────────────────────────────────────────────┐
│ [Layer Name] │
├─────────────────────────────────────────────────────────────┤
│ ┌─────────┐ ┌─────────┐ ┌─────────┐ ┌─────────┐ │
│ │ [Comp] │ │ [Comp] │ │ [Comp] │ │ [Comp] │ │
│ └────┬────┘ └────┬────┘ └────┬────┘ └────┬────┘ │
│ │ │ │ │ │
├───────┴────────────┴────────────┴────────────┴──────────────┤
│ [Layer Name] │
├─────────────────────────────────────────────────────────────┤
│ ┌─────────────────────────────────────────────────────┐ │
│ │ [Component] │ │
│ └─────────────────────────────────────────────────────┘ │
├─────────────────────────────────────────────────────────────┤
│ [Layer Name] │
│ ┌──────────┐ ┌──────────┐ ┌──────────┐ │
│ │ [Store] │ │ [Store] │ │ [Store] │ │
│ └──────────┘ └──────────┘ └──────────┘ │
└─────────────────────────────────────────────────────────────┘
```
### Component Responsibilities
| Component | Responsibility | Typical Implementation |
|-----------|----------------|------------------------|
| [name] | [what it owns] | [how it's usually built] |
| [name] | [what it owns] | [how it's usually built] |
| [name] | [what it owns] | [how it's usually built] |
## Recommended Project Structure
```
src/
├── [folder]/ # [purpose]
│ ├── [subfolder]/ # [purpose]
│ └── [file].ts # [purpose]
├── [folder]/ # [purpose]
│ ├── [subfolder]/ # [purpose]
│ └── [file].ts # [purpose]
├── [folder]/ # [purpose]
└── [folder]/ # [purpose]
```
### Structure Rationale
- **[folder]/:** [why organized this way]
- **[folder]/:** [why organized this way]
## Architectural Patterns
### Pattern 1: [Pattern Name]
**What:** [description]
**When to use:** [conditions]
**Trade-offs:** [pros and cons]
**Example:**
```typescript
// [Brief code example showing the pattern]
```
### Pattern 2: [Pattern Name]
**What:** [description]
**When to use:** [conditions]
**Trade-offs:** [pros and cons]
**Example:**
```typescript
// [Brief code example showing the pattern]
```
### Pattern 3: [Pattern Name]
**What:** [description]
**When to use:** [conditions]
**Trade-offs:** [pros and cons]
## Data Flow
### Request Flow
```
[User Action]
↓
[Component] → [Handler] → [Service] → [Data Store]
↓ ↓ ↓ ↓
[Response] ← [Transform] ← [Query] ← [Database]
```
### State Management
```
[State Store]
↓ (subscribe)
[Components] ←→ [Actions] → [Reducers/Mutations] → [State Store]
```
### Key Data Flows
1. **[Flow name]:** [description of how data moves]
2. **[Flow name]:** [description of how data moves]
## Scaling Considerations
| Scale | Architecture Adjustments |
|-------|--------------------------|
| 0-1k users | [approach — usually monolith is fine] |
| 1k-100k users | [approach — what to optimize first] |
| 100k+ users | [approach — when to consider splitting] |
### Scaling Priorities
1. **First bottleneck:** [what breaks first, how to fix]
2. **Second bottleneck:** [what breaks next, how to fix]
## Anti-Patterns
### Anti-Pattern 1: [Name]
**What people do:** [the mistake]
**Why it's wrong:** [the problem it causes]
**Do this instead:** [the correct approach]
### Anti-Pattern 2: [Name]
**What people do:** [the mistake]
**Why it's wrong:** [the problem it causes]
**Do this instead:** [the correct approach]
## Integration Points
### External Services
| Service | Integration Pattern | Notes |
|---------|---------------------|-------|
| [service] | [how to connect] | [gotchas] |
| [service] | [how to connect] | [gotchas] |
### Internal Boundaries
| Boundary | Communication | Notes |
|----------|---------------|-------|
| [module A ↔ module B] | [API/events/direct] | [considerations] |
## Sources
- [Architecture references]
- [Official documentation]
- [Case studies]
---
*Architecture research for: [domain]*
*Researched: [date]*
```
</template>
<guidelines>
**System Overview:**
- Use ASCII box-drawing diagrams for clarity (├── └── │ ─ for structure visualization only)
- Show major components and their relationships
- Don't over-detail — this is conceptual, not implementation
**Project Structure:**
- Be specific about folder organization
- Explain the rationale for grouping
- Match conventions of the chosen stack
**Patterns:**
- Include code examples where helpful
- Explain trade-offs honestly
- Note when patterns are overkill for small projects
**Scaling Considerations:**
- Be realistic — most projects don't need to scale to millions
- Focus on "what breaks first" not theoretical limits
- Avoid premature optimization recommendations
**Anti-Patterns:**
- Specific to this domain
- Include what to do instead
- Helps prevent common mistakes during implementation
</guidelines>

View File

@@ -1,147 +0,0 @@
# Features Research Template
Template for `.planning/research/FEATURES.md` — feature landscape for the project domain.
<template>
```markdown
# Feature Research
**Domain:** [domain type]
**Researched:** [date]
**Confidence:** [HIGH/MEDIUM/LOW]
## Feature Landscape
### Table Stakes (Users Expect These)
Features users assume exist. Missing these = product feels incomplete.
| Feature | Why Expected | Complexity | Notes |
|---------|--------------|------------|-------|
| [feature] | [user expectation] | LOW/MEDIUM/HIGH | [implementation notes] |
| [feature] | [user expectation] | LOW/MEDIUM/HIGH | [implementation notes] |
| [feature] | [user expectation] | LOW/MEDIUM/HIGH | [implementation notes] |
### Differentiators (Competitive Advantage)
Features that set the product apart. Not required, but valuable.
| Feature | Value Proposition | Complexity | Notes |
|---------|-------------------|------------|-------|
| [feature] | [why it matters] | LOW/MEDIUM/HIGH | [implementation notes] |
| [feature] | [why it matters] | LOW/MEDIUM/HIGH | [implementation notes] |
| [feature] | [why it matters] | LOW/MEDIUM/HIGH | [implementation notes] |
### Anti-Features (Commonly Requested, Often Problematic)
Features that seem good but create problems.
| Feature | Why Requested | Why Problematic | Alternative |
|---------|---------------|-----------------|-------------|
| [feature] | [surface appeal] | [actual problems] | [better approach] |
| [feature] | [surface appeal] | [actual problems] | [better approach] |
## Feature Dependencies
```
[Feature A]
└──requires──> [Feature B]
└──requires──> [Feature C]
[Feature D] ──enhances──> [Feature A]
[Feature E] ──conflicts──> [Feature F]
```
### Dependency Notes
- **[Feature A] requires [Feature B]:** [why the dependency exists]
- **[Feature D] enhances [Feature A]:** [how they work together]
- **[Feature E] conflicts with [Feature F]:** [why they're incompatible]
## MVP Definition
### Launch With (v1)
Minimum viable product — what's needed to validate the concept.
- [ ] [Feature] — [why essential]
- [ ] [Feature] — [why essential]
- [ ] [Feature] — [why essential]
### Add After Validation (v1.x)
Features to add once core is working.
- [ ] [Feature] — [trigger for adding]
- [ ] [Feature] — [trigger for adding]
### Future Consideration (v2+)
Features to defer until product-market fit is established.
- [ ] [Feature] — [why defer]
- [ ] [Feature] — [why defer]
## Feature Prioritization Matrix
| Feature | User Value | Implementation Cost | Priority |
|---------|------------|---------------------|----------|
| [feature] | HIGH/MEDIUM/LOW | HIGH/MEDIUM/LOW | P1/P2/P3 |
| [feature] | HIGH/MEDIUM/LOW | HIGH/MEDIUM/LOW | P1/P2/P3 |
| [feature] | HIGH/MEDIUM/LOW | HIGH/MEDIUM/LOW | P1/P2/P3 |
**Priority key:**
- P1: Must have for launch
- P2: Should have, add when possible
- P3: Nice to have, future consideration
## Competitor Feature Analysis
| Feature | Competitor A | Competitor B | Our Approach |
|---------|--------------|--------------|--------------|
| [feature] | [how they do it] | [how they do it] | [our plan] |
| [feature] | [how they do it] | [how they do it] | [our plan] |
## Sources
- [Competitor products analyzed]
- [User research or feedback sources]
- [Industry standards referenced]
---
*Feature research for: [domain]*
*Researched: [date]*
```
</template>
<guidelines>
**Table Stakes:**
- These are non-negotiable for launch
- Users don't give credit for having them, but penalize for missing them
- Example: A community platform without user profiles is broken
**Differentiators:**
- These are where you compete
- Should align with the Core Value from PROJECT.md
- Don't try to differentiate on everything
**Anti-Features:**
- Prevent scope creep by documenting what seems good but isn't
- Include the alternative approach
- Example: "Real-time everything" often creates complexity without value
**Feature Dependencies:**
- Critical for roadmap phase ordering
- If A requires B, B must be in an earlier phase
- Conflicts inform what NOT to combine in same phase
**MVP Definition:**
- Be ruthless about what's truly minimum
- "Nice to have" is not MVP
- Launch with less, validate, then expand
</guidelines>

View File

@@ -1,200 +0,0 @@
# Pitfalls Research Template
Template for `.planning/research/PITFALLS.md` — common mistakes to avoid in the project domain.
<template>
```markdown
# Pitfalls Research
**Domain:** [domain type]
**Researched:** [date]
**Confidence:** [HIGH/MEDIUM/LOW]
## Critical Pitfalls
### Pitfall 1: [Name]
**What goes wrong:**
[Description of the failure mode]
**Why it happens:**
[Root cause — why developers make this mistake]
**How to avoid:**
[Specific prevention strategy]
**Warning signs:**
[How to detect this early before it becomes a problem]
**Phase to address:**
[Which roadmap phase should prevent this]
---
### Pitfall 2: [Name]
**What goes wrong:**
[Description of the failure mode]
**Why it happens:**
[Root cause — why developers make this mistake]
**How to avoid:**
[Specific prevention strategy]
**Warning signs:**
[How to detect this early before it becomes a problem]
**Phase to address:**
[Which roadmap phase should prevent this]
---
### Pitfall 3: [Name]
**What goes wrong:**
[Description of the failure mode]
**Why it happens:**
[Root cause — why developers make this mistake]
**How to avoid:**
[Specific prevention strategy]
**Warning signs:**
[How to detect this early before it becomes a problem]
**Phase to address:**
[Which roadmap phase should prevent this]
---
[Continue for all critical pitfalls...]
## Technical Debt Patterns
Shortcuts that seem reasonable but create long-term problems.
| Shortcut | Immediate Benefit | Long-term Cost | When Acceptable |
|----------|-------------------|----------------|-----------------|
| [shortcut] | [benefit] | [cost] | [conditions, or "never"] |
| [shortcut] | [benefit] | [cost] | [conditions, or "never"] |
| [shortcut] | [benefit] | [cost] | [conditions, or "never"] |
## Integration Gotchas
Common mistakes when connecting to external services.
| Integration | Common Mistake | Correct Approach |
|-------------|----------------|------------------|
| [service] | [what people do wrong] | [what to do instead] |
| [service] | [what people do wrong] | [what to do instead] |
| [service] | [what people do wrong] | [what to do instead] |
## Performance Traps
Patterns that work at small scale but fail as usage grows.
| Trap | Symptoms | Prevention | When It Breaks |
|------|----------|------------|----------------|
| [trap] | [how you notice] | [how to avoid] | [scale threshold] |
| [trap] | [how you notice] | [how to avoid] | [scale threshold] |
| [trap] | [how you notice] | [how to avoid] | [scale threshold] |
## Security Mistakes
Domain-specific security issues beyond general web security.
| Mistake | Risk | Prevention |
|---------|------|------------|
| [mistake] | [what could happen] | [how to avoid] |
| [mistake] | [what could happen] | [how to avoid] |
| [mistake] | [what could happen] | [how to avoid] |
## UX Pitfalls
Common user experience mistakes in this domain.
| Pitfall | User Impact | Better Approach |
|---------|-------------|-----------------|
| [pitfall] | [how users suffer] | [what to do instead] |
| [pitfall] | [how users suffer] | [what to do instead] |
| [pitfall] | [how users suffer] | [what to do instead] |
## "Looks Done But Isn't" Checklist
Things that appear complete but are missing critical pieces.
- [ ] **[Feature]:** Often missing [thing] — verify [check]
- [ ] **[Feature]:** Often missing [thing] — verify [check]
- [ ] **[Feature]:** Often missing [thing] — verify [check]
- [ ] **[Feature]:** Often missing [thing] — verify [check]
## Recovery Strategies
When pitfalls occur despite prevention, how to recover.
| Pitfall | Recovery Cost | Recovery Steps |
|---------|---------------|----------------|
| [pitfall] | LOW/MEDIUM/HIGH | [what to do] |
| [pitfall] | LOW/MEDIUM/HIGH | [what to do] |
| [pitfall] | LOW/MEDIUM/HIGH | [what to do] |
## Pitfall-to-Phase Mapping
How roadmap phases should address these pitfalls.
| Pitfall | Prevention Phase | Verification |
|---------|------------------|--------------|
| [pitfall] | Phase [X] | [how to verify prevention worked] |
| [pitfall] | Phase [X] | [how to verify prevention worked] |
| [pitfall] | Phase [X] | [how to verify prevention worked] |
## Sources
- [Post-mortems referenced]
- [Community discussions]
- [Official "gotchas" documentation]
- [Personal experience / known issues]
---
*Pitfalls research for: [domain]*
*Researched: [date]*
```
</template>
<guidelines>
**Critical Pitfalls:**
- Focus on domain-specific issues, not generic mistakes
- Include warning signs — early detection prevents disasters
- Link to specific phases — makes pitfalls actionable
**Technical Debt:**
- Be realistic — some shortcuts are acceptable
- Note when shortcuts are "never acceptable" vs. "only in MVP"
- Include the long-term cost to inform tradeoff decisions
**Performance Traps:**
- Include scale thresholds ("breaks at 10k users")
- Focus on what's relevant for this project's expected scale
- Don't over-engineer for hypothetical scale
**Security Mistakes:**
- Beyond OWASP basics — domain-specific issues
- Example: Community platforms have different security concerns than e-commerce
- Include risk level to prioritize
**"Looks Done But Isn't":**
- Checklist format for verification during execution
- Common in demos vs. production
- Prevents "it works on my machine" issues
**Pitfall-to-Phase Mapping:**
- Critical for roadmap creation
- Each pitfall should map to a phase that prevents it
- Informs phase ordering and success criteria
</guidelines>

View File

@@ -1,120 +0,0 @@
# Stack Research Template
Template for `.planning/research/STACK.md` — recommended technologies for the project domain.
<template>
```markdown
# Stack Research
**Domain:** [domain type]
**Researched:** [date]
**Confidence:** [HIGH/MEDIUM/LOW]
## Recommended Stack
### Core Technologies
| Technology | Version | Purpose | Why Recommended |
|------------|---------|---------|-----------------|
| [name] | [version] | [what it does] | [why experts use it for this domain] |
| [name] | [version] | [what it does] | [why experts use it for this domain] |
| [name] | [version] | [what it does] | [why experts use it for this domain] |
### Supporting Libraries
| Library | Version | Purpose | When to Use |
|---------|---------|---------|-------------|
| [name] | [version] | [what it does] | [specific use case] |
| [name] | [version] | [what it does] | [specific use case] |
| [name] | [version] | [what it does] | [specific use case] |
### Development Tools
| Tool | Purpose | Notes |
|------|---------|-------|
| [name] | [what it does] | [configuration tips] |
| [name] | [what it does] | [configuration tips] |
## Installation
```bash
# Core
npm install [packages]
# Supporting
npm install [packages]
# Dev dependencies
npm install -D [packages]
```
## Alternatives Considered
| Recommended | Alternative | When to Use Alternative |
|-------------|-------------|-------------------------|
| [our choice] | [other option] | [conditions where alternative is better] |
| [our choice] | [other option] | [conditions where alternative is better] |
## What NOT to Use
| Avoid | Why | Use Instead |
|-------|-----|-------------|
| [technology] | [specific problem] | [recommended alternative] |
| [technology] | [specific problem] | [recommended alternative] |
## Stack Patterns by Variant
**If [condition]:**
- Use [variation]
- Because [reason]
**If [condition]:**
- Use [variation]
- Because [reason]
## Version Compatibility
| Package A | Compatible With | Notes |
|-----------|-----------------|-------|
| [package@version] | [package@version] | [compatibility notes] |
## Sources
- [Context7 library ID] — [topics fetched]
- [Official docs URL] — [what was verified]
- [Other source] — [confidence level]
---
*Stack research for: [domain]*
*Researched: [date]*
```
</template>
<guidelines>
**Core Technologies:**
- Include specific version numbers
- Explain why this is the standard choice, not just what it does
- Focus on technologies that affect architecture decisions
**Supporting Libraries:**
- Include libraries commonly needed for this domain
- Note when each is needed (not all projects need all libraries)
**Alternatives:**
- Don't just dismiss alternatives
- Explain when alternatives make sense
- Helps user make informed decisions if they disagree
**What NOT to Use:**
- Actively warn against outdated or problematic choices
- Explain the specific problem, not just "it's old"
- Provide the recommended alternative
**Version Compatibility:**
- Note any known compatibility issues
- Critical for avoiding debugging time later
</guidelines>

View File

@@ -1,170 +0,0 @@
# Research Summary Template
Template for `.planning/research/SUMMARY.md` — executive summary of project research with roadmap implications.
<template>
```markdown
# Project Research Summary
**Project:** [name from PROJECT.md]
**Domain:** [inferred domain type]
**Researched:** [date]
**Confidence:** [HIGH/MEDIUM/LOW]
## Executive Summary
[2-3 paragraph overview of research findings]
- What type of product this is and how experts build it
- The recommended approach based on research
- Key risks and how to mitigate them
## Key Findings
### Recommended Stack
[Summary from STACK.md — 1-2 paragraphs]
**Core technologies:**
- [Technology]: [purpose] — [why recommended]
- [Technology]: [purpose] — [why recommended]
- [Technology]: [purpose] — [why recommended]
### Expected Features
[Summary from FEATURES.md]
**Must have (table stakes):**
- [Feature] — users expect this
- [Feature] — users expect this
**Should have (competitive):**
- [Feature] — differentiator
- [Feature] — differentiator
**Defer (v2+):**
- [Feature] — not essential for launch
### Architecture Approach
[Summary from ARCHITECTURE.md — 1 paragraph]
**Major components:**
1. [Component] — [responsibility]
2. [Component] — [responsibility]
3. [Component] — [responsibility]
### Critical Pitfalls
[Top 3-5 from PITFALLS.md]
1. **[Pitfall]** — [how to avoid]
2. **[Pitfall]** — [how to avoid]
3. **[Pitfall]** — [how to avoid]
## Implications for Roadmap
Based on research, suggested phase structure:
### Phase 1: [Name]
**Rationale:** [why this comes first based on research]
**Delivers:** [what this phase produces]
**Addresses:** [features from FEATURES.md]
**Avoids:** [pitfall from PITFALLS.md]
### Phase 2: [Name]
**Rationale:** [why this order]
**Delivers:** [what this phase produces]
**Uses:** [stack elements from STACK.md]
**Implements:** [architecture component]
### Phase 3: [Name]
**Rationale:** [why this order]
**Delivers:** [what this phase produces]
[Continue for suggested phases...]
### Phase Ordering Rationale
- [Why this order based on dependencies discovered]
- [Why this grouping based on architecture patterns]
- [How this avoids pitfalls from research]
### Research Flags
Phases likely needing deeper research during planning:
- **Phase [X]:** [reason — e.g., "complex integration, needs API research"]
- **Phase [Y]:** [reason — e.g., "niche domain, sparse documentation"]
Phases with standard patterns (skip research-phase):
- **Phase [X]:** [reason — e.g., "well-documented, established patterns"]
## Confidence Assessment
| Area | Confidence | Notes |
|------|------------|-------|
| Stack | [HIGH/MEDIUM/LOW] | [reason] |
| Features | [HIGH/MEDIUM/LOW] | [reason] |
| Architecture | [HIGH/MEDIUM/LOW] | [reason] |
| Pitfalls | [HIGH/MEDIUM/LOW] | [reason] |
**Overall confidence:** [HIGH/MEDIUM/LOW]
### Gaps to Address
[Any areas where research was inconclusive or needs validation during implementation]
- [Gap]: [how to handle during planning/execution]
- [Gap]: [how to handle during planning/execution]
## Sources
### Primary (HIGH confidence)
- [Context7 library ID] — [topics]
- [Official docs URL] — [what was checked]
### Secondary (MEDIUM confidence)
- [Source] — [finding]
### Tertiary (LOW confidence)
- [Source] — [finding, needs validation]
---
*Research completed: [date]*
*Ready for roadmap: yes*
```
</template>
<guidelines>
**Executive Summary:**
- Write for someone who will only read this section
- Include the key recommendation and main risk
- 2-3 paragraphs maximum
**Key Findings:**
- Summarize, don't duplicate full documents
- Link to detailed docs (STACK.md, FEATURES.md, etc.)
- Focus on what matters for roadmap decisions
**Implications for Roadmap:**
- This is the most important section
- Directly informs roadmap creation
- Be explicit about phase suggestions and rationale
- Include research flags for each suggested phase
**Confidence Assessment:**
- Be honest about uncertainty
- Note gaps that need resolution during planning
- HIGH = verified with official sources
- MEDIUM = community consensus, multiple sources agree
- LOW = single source or inference
**Integration with roadmap creation:**
- This file is loaded as context during roadmap creation
- Phase suggestions here become starting point for roadmap
- Research flags inform phase planning
</guidelines>

View File

@@ -1,202 +0,0 @@
# Roadmap Template
Template for `.planning/ROADMAP.md`.
## Initial Roadmap (v1.0 Greenfield)
```markdown
# Roadmap: [Project Name]
## Overview
[One paragraph describing the journey from start to finish]
## Phases
**Phase Numbering:**
- Integer phases (1, 2, 3): Planned milestone work
- Decimal phases (2.1, 2.2): Urgent insertions (marked with INSERTED)
Decimal phases appear between their surrounding integers in numeric order.
- [ ] **Phase 1: [Name]** - [One-line description]
- [ ] **Phase 2: [Name]** - [One-line description]
- [ ] **Phase 3: [Name]** - [One-line description]
- [ ] **Phase 4: [Name]** - [One-line description]
## Phase Details
### Phase 1: [Name]
**Goal**: [What this phase delivers]
**Depends on**: Nothing (first phase)
**Requirements**: [REQ-01, REQ-02, REQ-03] <!-- brackets optional, parser handles both formats -->
**Success Criteria** (what must be TRUE):
1. [Observable behavior from user perspective]
2. [Observable behavior from user perspective]
3. [Observable behavior from user perspective]
**Plans**: [Number of plans, e.g., "3 plans" or "TBD"]
Plans:
- [ ] 01-01: [Brief description of first plan]
- [ ] 01-02: [Brief description of second plan]
- [ ] 01-03: [Brief description of third plan]
### Phase 2: [Name]
**Goal**: [What this phase delivers]
**Depends on**: Phase 1
**Requirements**: [REQ-04, REQ-05]
**Success Criteria** (what must be TRUE):
1. [Observable behavior from user perspective]
2. [Observable behavior from user perspective]
**Plans**: [Number of plans]
Plans:
- [ ] 02-01: [Brief description]
- [ ] 02-02: [Brief description]
### Phase 2.1: Critical Fix (INSERTED)
**Goal**: [Urgent work inserted between phases]
**Depends on**: Phase 2
**Success Criteria** (what must be TRUE):
1. [What the fix achieves]
**Plans**: 1 plan
Plans:
- [ ] 02.1-01: [Description]
### Phase 3: [Name]
**Goal**: [What this phase delivers]
**Depends on**: Phase 2
**Requirements**: [REQ-06, REQ-07, REQ-08]
**Success Criteria** (what must be TRUE):
1. [Observable behavior from user perspective]
2. [Observable behavior from user perspective]
3. [Observable behavior from user perspective]
**Plans**: [Number of plans]
Plans:
- [ ] 03-01: [Brief description]
- [ ] 03-02: [Brief description]
### Phase 4: [Name]
**Goal**: [What this phase delivers]
**Depends on**: Phase 3
**Requirements**: [REQ-09, REQ-10]
**Success Criteria** (what must be TRUE):
1. [Observable behavior from user perspective]
2. [Observable behavior from user perspective]
**Plans**: [Number of plans]
Plans:
- [ ] 04-01: [Brief description]
## Progress
**Execution Order:**
Phases execute in numeric order: 2 → 2.1 → 2.2 → 3 → 3.1 → 4
| Phase | Plans Complete | Status | Completed |
|-------|----------------|--------|-----------|
| 1. [Name] | 0/3 | Not started | - |
| 2. [Name] | 0/2 | Not started | - |
| 3. [Name] | 0/2 | Not started | - |
| 4. [Name] | 0/1 | Not started | - |
```
<guidelines>
**Initial planning (v1.0):**
- Phase count depends on granularity setting (coarse: 3-5, standard: 5-8, fine: 8-12)
- Each phase delivers something coherent
- Phases can have 1+ plans (split if >3 tasks or multiple subsystems)
- Plans use naming: {phase}-{plan}-PLAN.md (e.g., 01-02-PLAN.md)
- No time estimates (this isn't enterprise PM)
- Progress table updated by execute workflow
- Plan count can be "TBD" initially, refined during planning
**Success criteria:**
- 2-5 observable behaviors per phase (from user's perspective)
- Cross-checked against requirements during roadmap creation
- Flow downstream to `must_haves` in plan-phase
- Verified by verify-phase after execution
- Format: "User can [action]" or "[Thing] works/exists"
**After milestones ship:**
- Collapse completed milestones in `<details>` tags
- Add new milestone sections for upcoming work
- Keep continuous phase numbering (never restart at 01)
</guidelines>
<status_values>
- `Not started` - Haven't begun
- `In progress` - Currently working
- `Complete` - Done (add completion date)
- `Deferred` - Pushed to later (with reason)
</status_values>
## Milestone-Grouped Roadmap (After v1.0 Ships)
After completing first milestone, reorganize with milestone groupings:
```markdown
# Roadmap: [Project Name]
## Milestones
- ✅ **v1.0 MVP** - Phases 1-4 (shipped YYYY-MM-DD)
- 🚧 **v1.1 [Name]** - Phases 5-6 (in progress)
- 📋 **v2.0 [Name]** - Phases 7-10 (planned)
## Phases
<details>
<summary>✅ v1.0 MVP (Phases 1-4) - SHIPPED YYYY-MM-DD</summary>
### Phase 1: [Name]
**Goal**: [What this phase delivers]
**Plans**: 3 plans
Plans:
- [x] 01-01: [Brief description]
- [x] 01-02: [Brief description]
- [x] 01-03: [Brief description]
[... remaining v1.0 phases ...]
</details>
### 🚧 v1.1 [Name] (In Progress)
**Milestone Goal:** [What v1.1 delivers]
#### Phase 5: [Name]
**Goal**: [What this phase delivers]
**Depends on**: Phase 4
**Plans**: 2 plans
Plans:
- [ ] 05-01: [Brief description]
- [ ] 05-02: [Brief description]
[... remaining v1.1 phases ...]
### 📋 v2.0 [Name] (Planned)
**Milestone Goal:** [What v2.0 delivers]
[... v2.0 phases ...]
## Progress
| Phase | Milestone | Plans Complete | Status | Completed |
|-------|-----------|----------------|--------|-----------|
| 1. Foundation | v1.0 | 3/3 | Complete | YYYY-MM-DD |
| 2. Features | v1.0 | 2/2 | Complete | YYYY-MM-DD |
| 5. Security | v1.1 | 0/2 | Not started | - |
```
**Notes:**
- Milestone emoji: ✅ shipped, 🚧 in progress, 📋 planned
- Completed milestones collapsed in `<details>` for readability
- Current/future milestones expanded
- Continuous phase numbering (01-99)
- Progress table includes milestone column

View File

@@ -1,194 +0,0 @@
# State Template
Template for `.planning/STATE.md` — the project's living memory.
---
## File Template
```markdown
---
gsd_state_version: '1.0' # placeholder; syncStateFrontmatter overwrites on first state.* call
status: planning
progress:
total_phases: 0
completed_phases: 0
total_plans: 0
completed_plans: 0
percent: 0
---
# Project State
## Project Reference
See: .planning/PROJECT.md (updated [date])
**Core value:** [One-liner from PROJECT.md Core Value section]
**Current focus:** [Current phase name]
## Current Position
Phase: [X] of [Y] ([Phase name])
Plan: [A] of [B] in current phase
Status: [Ready to plan / Planning / Ready to execute / In progress / Phase complete]
Last activity: [YYYY-MM-DD] — [What happened]
Progress: [░░░░░░░░░░] 0%
## Performance Metrics
**Velocity:**
- Total plans completed: [N]
- Average duration: [X] min
- Total execution time: [X.X] hours
**By Phase:**
| Phase | Plans | Total | Avg/Plan |
|-------|-------|-------|----------|
| - | - | - | - |
**Recent Trend:**
- Last 5 plans: [durations]
- Trend: [Improving / Stable / Degrading]
*Updated after each plan completion*
## Accumulated Context
### Decisions
Decisions are logged in PROJECT.md Key Decisions table.
Recent decisions affecting current work:
- [Phase X]: [Decision summary]
- [Phase Y]: [Decision summary]
### Pending Todos
[From .planning/todos/pending/ — ideas captured during sessions]
None yet.
### Blockers/Concerns
[Issues that affect future work]
None yet.
## Deferred Items
Items acknowledged and carried forward from previous milestone close:
| Category | Item | Status | Deferred At |
|----------|------|--------|-------------|
| *(none)* | | | |
## Session Continuity
Last session: [YYYY-MM-DD HH:MM]
Stopped at: [Description of last completed action]
Resume file: [Path to .continue-here*.md if exists, otherwise "None"]
```
<purpose>
STATE.md is the project's short-term memory spanning all phases and sessions.
**Problem it solves:** Information is captured in summaries, issues, and decisions but not systematically consumed. Sessions start without context.
**Solution:** A single, small file that's:
- Read first in every workflow
- Updated after every significant action
- Contains digest of accumulated context
- Enables instant session restoration
</purpose>
<lifecycle>
**Creation:** After ROADMAP.md is created (during init)
- Reference PROJECT.md (read it for current context)
- Initialize empty accumulated context sections
- Set position to "Phase 1 ready to plan"
**Reading:** First step of every workflow
- progress: Present status to user
- plan: Inform planning decisions
- execute: Know current position
- transition: Know what's complete
**Writing:** After every significant action
- execute: After SUMMARY.md created
- Update position (phase, plan, status)
- Note new decisions (detail in PROJECT.md)
- Add blockers/concerns
- transition: After phase marked complete
- Update progress bar
- Clear resolved blockers
- Refresh Project Reference date
</lifecycle>
<sections>
### Project Reference
Points to PROJECT.md for full context. Includes:
- Core value (the ONE thing that matters)
- Current focus (which phase)
- Last update date (triggers re-read if stale)
Claude reads PROJECT.md directly for requirements, constraints, and decisions.
### Current Position
Where we are right now:
- Phase X of Y — which phase
- Plan A of B — which plan within phase
- Status — current state
- Last activity — what happened most recently
- Progress bar — visual indicator of overall completion
Progress calculation: (completed plans) / (total plans across all phases) × 100%
### Performance Metrics
Track velocity to understand execution patterns:
- Total plans completed
- Average duration per plan
- Per-phase breakdown
- Recent trend (improving/stable/degrading)
Updated after each plan completion.
### Accumulated Context
**Decisions:** Reference to PROJECT.md Key Decisions table, plus recent decisions summary for quick access. Full decision log lives in PROJECT.md.
**Pending Todos:** Ideas captured during sessions.
- Count of pending todos
- Brief list if few, count if many
**Blockers/Concerns:** From "Next Phase Readiness" sections
- Issues that affect future work
- Prefix with originating phase
- Cleared when addressed
### Session Continuity
Enables instant resumption:
- When was last session
- What was last completed
- Is there a .continue-here file to resume from
</sections>
<size_constraint>
Keep STATE.md under 100 lines.
It's a DIGEST, not an archive. If accumulated context grows too large:
- Keep only 3-5 recent decisions in summary (full log in PROJECT.md)
- Keep only active blockers, remove resolved ones
The goal is "read once, know where we are" — if it's too long, that fails.
</size_constraint>

View File

@@ -1,59 +0,0 @@
/**
* One-off generator: extracts PROFILING_QUESTIONS + CLAUDE_INSTRUCTIONS from profile-output.cjs
* Run: node scripts/gen-profile-questionnaire-data.mjs
*/
import fs from 'node:fs';
import { fileURLToPath } from 'node:url';
import { dirname, join } from 'node:path';
const __dirname = dirname(fileURLToPath(import.meta.url));
const root = join(__dirname, '..', '..');
const cjs = fs.readFileSync(join(root, 'get-shit-done/bin/lib/profile-output.cjs'), 'utf-8');
const m1 = cjs.match(/const PROFILING_QUESTIONS = (\[[\s\S]*?\]);/);
const m2 = cjs.match(/const CLAUDE_INSTRUCTIONS = (\{[\s\S]*?\n\});/);
if (!m1 || !m2) {
console.error('regex extract failed');
process.exit(1);
}
const header = `/**
* Synced from get-shit-done/bin/lib/profile-output.cjs (PROFILING_QUESTIONS, CLAUDE_INSTRUCTIONS).
* Used by profileQuestionnaire for parity with cmdProfileQuestionnaire.
*/
export type ProfilingOption = { label: string; value: string; rating: string };
export type ProfilingQuestion = {
dimension: string;
header: string;
context: string;
question: string;
options: ProfilingOption[];
};
export const PROFILING_QUESTIONS: ProfilingQuestion[] = ${m1[1]};
export const CLAUDE_INSTRUCTIONS: Record<string, Record<string, string>> = ${m2[1]};
export function isAmbiguousAnswer(dimension: string, value: string): boolean {
if (dimension === 'communication_style' && value === 'd') return true;
const question = PROFILING_QUESTIONS.find((q) => q.dimension === dimension);
if (!question) return false;
const option = question.options.find((o) => o.value === value);
if (!option) return false;
return option.rating === 'mixed';
}
export function generateClaudeInstruction(dimension: string, rating: string): string {
const dimInstructions = CLAUDE_INSTRUCTIONS[dimension];
if (dimInstructions && dimInstructions[rating]) {
return dimInstructions[rating]!;
}
return \`Adapt to this developer's \${dimension.replace(/_/g, ' ')} preference: \${rating}.\`;
}
`;
const outPath = join(root, 'sdk/src/query/profile-questionnaire-data.ts');
fs.writeFileSync(outPath, header);
console.log('wrote', outPath);

View File

@@ -1,349 +0,0 @@
/**
* Contract test: assembled prompts from PromptFactory.buildPrompt() and
* InitRunner.build*Prompt() must contain zero interactive patterns.
*
* Unlike headless-prompts.test.ts (which scans raw .md files on disk),
* these tests exercise the full assembly pipeline:
* file loading → role extraction → context injection → sanitizePrompt()
*
* If any assembly step reintroduces interactive patterns that sanitizePrompt()
* doesn't catch, these tests will fail.
*/
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
import { mkdtemp, mkdir, writeFile, rm } from 'node:fs/promises';
import { join, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';
import { PromptFactory } from './phase-prompt.js';
import { InitRunner } from './init-runner.js';
import { PhaseType } from './types.js';
import type { ParsedPlan, ContextFiles, GSDEvent } from './types.js';
import type { GSDTools } from './gsd-tools.js';
import type { GSDEventStream } from './event-stream.js';
// ─── Paths ───────────────────────────────────────────────────────────────────
const __dirname = dirname(fileURLToPath(import.meta.url));
const sdkPromptsDir = join(__dirname, '..', 'prompts');
// ─── Blocked patterns (aligned with headless-prompts.test.ts) ────────────────
const BLOCKED_PATTERNS: Array<[string, RegExp]> = [
['AskUserQuestion', /AskUserQuestion\s*\(/],
['SlashCommand', /SlashCommand\s*\(/],
['/gsd: command', /\/gsd:\S+/],
['@file: reference', /@file:\S+/],
['STOP + wait directive', /\bSTOP\b\s+(?:and\s+)?(?:wait|ask)/i],
['bare STOP directive', /^\s*STOP\s*[.!]?\s*$/m],
['wait for user', /\bwait\s+for\s+(?:the\s+)?user\b/i],
['ask the user', /\bask\s+the\s+user\b/i],
];
// ─── Minimal fixtures ────────────────────────────────────────────────────────
const MINIMAL_PLAN: ParsedPlan = {
frontmatter: {
phase: '01',
plan: 'test-plan',
type: 'feature',
wave: 1,
depends_on: [],
files_modified: ['src/index.ts'],
autonomous: true,
requirements: ['R001'],
must_haves: {
truths: ['It works'],
artifacts: [{ path: 'src/index.ts', provides: 'entry point' }],
key_links: [],
},
},
objective: 'Test objective for assembled prompt contract test',
execution_context: ['This is a test context line'],
context_refs: [],
tasks: [
{
type: 'create',
name: 'Create test file',
files: ['src/test.ts'],
read_first: [],
action: 'Create a test file',
verify: 'File exists',
acceptance_criteria: ['File created'],
done: 'src/test.ts exists',
},
],
raw: '# Test Plan\n\nMinimal plan for testing.',
};
const EMPTY_CONTEXT: ContextFiles = {};
// ─── Helper ──────────────────────────────────────────────────────────────────
function assertNoBlockedPatterns(output: string, label: string): void {
for (const [patternLabel, pattern] of BLOCKED_PATTERNS) {
const matches = output.match(new RegExp(pattern.source, pattern.flags + 'g'));
expect(
matches,
`Found ${patternLabel} in ${label}: ${matches?.join(', ')}`,
).toBeNull();
}
}
// ─── PromptFactory assembled output ──────────────────────────────────────────
describe('PromptFactory assembled output', () => {
let factory: PromptFactory;
beforeAll(() => {
factory = new PromptFactory({ sdkPromptsDir });
});
const phaseTypes = Object.values(PhaseType) as PhaseType[];
for (const phaseType of phaseTypes) {
describe(`${phaseType} phase`, () => {
let output: string;
beforeAll(async () => {
output = await factory.buildPrompt(phaseType, MINIMAL_PLAN, EMPTY_CONTEXT);
});
it('produces non-empty output', () => {
expect(output.length).toBeGreaterThan(0);
});
for (const [label, pattern] of BLOCKED_PATTERNS) {
it(`contains no ${label}`, () => {
const matches = output.match(new RegExp(pattern.source, pattern.flags + 'g'));
expect(
matches,
`Found ${label} in ${phaseType} assembled prompt: ${matches?.join(', ')}`,
).toBeNull();
});
}
});
}
it('includes role section for phases with agents', async () => {
// Research, Plan, Execute, Verify all have agents; Discuss does not
const researchOutput = await factory.buildPrompt(PhaseType.Research, null, EMPTY_CONTEXT);
expect(researchOutput).toContain('## Agent Instructions');
});
it('includes purpose section from workflow files', async () => {
const planOutput = await factory.buildPrompt(PhaseType.Plan, null, EMPTY_CONTEXT);
// Plan phase should have purpose from plan-phase.md
expect(planOutput).toContain('## Purpose');
});
it('includes context section when context files provided', async () => {
const contextFiles: ContextFiles = {
state: '# State\ncurrent_phase: 01',
roadmap: '# Roadmap\n## Phase 01',
};
const output = await factory.buildPrompt(PhaseType.Research, null, contextFiles);
expect(output).toContain('## Context');
expect(output).toContain('Project State');
});
});
// ─── InitRunner assembled output ─────────────────────────────────────────────
describe('InitRunner assembled output', () => {
let tmpDir: string;
let runner: InitRunner;
// Minimal stub tools and event stream — we only call build*Prompt(), not run()
const stubTools: GSDTools = {
initNewProject: async () => ({
researcher_model: 'test',
synthesizer_model: 'test',
roadmapper_model: 'test',
commit_docs: false,
project_exists: false,
has_codebase_map: false,
has_git: true,
}),
configSet: async () => {},
commit: async () => {},
} as unknown as GSDTools;
const stubEventStream: GSDEventStream = {
emitEvent: (_event: GSDEvent) => {},
} as unknown as GSDEventStream;
beforeAll(async () => {
// Create temp directory with .planning/ structure for InitRunner file reads
tmpDir = await mkdtemp(join(tmpdir(), 'assembled-prompts-'));
const planningDir = join(tmpDir, '.planning');
const researchDir = join(planningDir, 'research');
await mkdir(researchDir, { recursive: true });
// Write minimal stubs that InitRunner reads
await writeFile(
join(planningDir, 'PROJECT.md'),
'# Test Project\n\nA minimal test project for contract testing.\n',
);
await writeFile(
join(planningDir, 'config.json'),
JSON.stringify({ mode: 'yolo', parallelization: true }, null, 2),
);
await writeFile(
join(planningDir, 'REQUIREMENTS.md'),
'# Requirements\n\n## R001 — Test Requirement\n',
);
await writeFile(
join(researchDir, 'STACK.md'),
'# Stack Research\n\nTypeScript + Node.js\n',
);
await writeFile(
join(researchDir, 'FEATURES.md'),
'# Features Research\n\nCore features identified.\n',
);
await writeFile(
join(researchDir, 'ARCHITECTURE.md'),
'# Architecture Research\n\nModular architecture.\n',
);
await writeFile(
join(researchDir, 'PITFALLS.md'),
'# Pitfalls Research\n\nCommon pitfalls noted.\n',
);
await writeFile(
join(researchDir, 'SUMMARY.md'),
'# Research Summary\n\nAll research synthesized.\n',
);
runner = new InitRunner({
projectDir: tmpDir,
tools: stubTools,
eventStream: stubEventStream,
sdkPromptsDir,
});
});
afterAll(async () => {
if (tmpDir) {
await rm(tmpDir, { recursive: true, force: true });
}
});
// Access private methods via (runner as any) — standard pattern for testing
// private methods in TypeScript without subclassing or mocking
describe('buildProjectPrompt', () => {
let output: string;
beforeAll(async () => {
output = await (runner as any).buildProjectPrompt('Build a CLI tool');
});
it('produces non-empty output', () => {
expect(output.length).toBeGreaterThan(0);
});
it('contains project template content', () => {
expect(output).toContain('PROJECT.md');
});
it('contains user input', () => {
expect(output).toContain('Build a CLI tool');
});
it('contains zero blocked patterns', () => {
assertNoBlockedPatterns(output, 'buildProjectPrompt');
});
});
describe('buildResearchPrompt', () => {
const researchTypes = ['STACK', 'FEATURES', 'ARCHITECTURE', 'PITFALLS'] as const;
for (const researchType of researchTypes) {
describe(`${researchType} research`, () => {
let output: string;
beforeAll(async () => {
output = await (runner as any).buildResearchPrompt(researchType, 'Build a CLI tool');
});
it('produces non-empty output', () => {
expect(output.length).toBeGreaterThan(0);
});
it('references the research type', () => {
expect(output).toContain(researchType);
});
it('contains zero blocked patterns', () => {
assertNoBlockedPatterns(output, `buildResearchPrompt(${researchType})`);
});
});
}
});
describe('buildSynthesisPrompt', () => {
let output: string;
beforeAll(async () => {
output = await (runner as any).buildSynthesisPrompt();
});
it('produces non-empty output', () => {
expect(output.length).toBeGreaterThan(0);
});
it('contains research content from temp files', () => {
// The synthesis prompt reads research files from disk — our stubs should appear
expect(output).toContain('Stack Research');
});
it('contains zero blocked patterns', () => {
assertNoBlockedPatterns(output, 'buildSynthesisPrompt');
});
});
describe('buildRequirementsPrompt', () => {
let output: string;
beforeAll(async () => {
output = await (runner as any).buildRequirementsPrompt();
});
it('produces non-empty output', () => {
expect(output.length).toBeGreaterThan(0);
});
it('contains project context from temp files', () => {
expect(output).toContain('Test Project');
});
it('contains zero blocked patterns', () => {
assertNoBlockedPatterns(output, 'buildRequirementsPrompt');
});
});
describe('buildRoadmapPrompt', () => {
let output: string;
beforeAll(async () => {
output = await (runner as any).buildRoadmapPrompt();
});
it('produces non-empty output', () => {
expect(output.length).toBeGreaterThan(0);
});
it('contains agent definition content', () => {
// Roadmap prompt loads gsd-roadmapper.md
expect(output).toContain('agent_definition');
});
it('contains project file content', () => {
expect(output).toContain('Test Project');
});
it('contains zero blocked patterns', () => {
assertNoBlockedPatterns(output, 'buildRoadmapPrompt');
});
});
});

View File

@@ -1,89 +0,0 @@
/**
* Bug #3589 (security): SDK `planningPaths(projectDir, workstream)` and
* `relPlanningPath(workstream)` accepted unvalidated explicit workstream
* names from direct SDK callers. Path-traversal segments (`..`, `/`, `\\`)
* would flow through `posix.join('.planning', 'workstreams', name)` and
* route planning operations outside the intended `.planning/workstreams/<name>`
* subtree.
*
* Env-sourced workstreams are pre-validated inside `planningPaths` and fall
* back to root .planning/ silently (#2791 contract). Explicit SDK arguments
* had no such gate.
*
* Fix: validate inside `relPlanningPath` so every caller — direct SDK use,
* `planningPaths`, `ContextEngine` — is protected at the same seam.
* Explicit invalid names throw; env-sourced ones still silently fall back
* because `planningPaths` filters them to `null` before calling
* `relPlanningPath`.
*/
import { describe, it, expect } from 'vitest';
import { relPlanningPath } from './workstream-utils.js';
import { planningPaths } from './query/helpers.js';
// Empty string is treated as "no workstream provided" (returns `.planning`)
// for back-compat with pre-fix behaviour; only non-empty invalid names throw.
const TRAVERSAL_CASES = [
'../../../outside',
'../escape',
'..',
'foo/bar',
'foo\\bar',
'foo bar',
'.hidden',
'/abs',
'-leading-hyphen',
];
describe('bug #3589: relPlanningPath rejects path-traversal and invalid workstream names', () => {
it('returns .planning when workstream is omitted (unchanged)', () => {
expect(relPlanningPath()).toBe('.planning');
expect(relPlanningPath(undefined)).toBe('.planning');
expect(relPlanningPath('')).toBe('.planning');
});
it('returns .planning/workstreams/<name> for valid workstream names (unchanged)', () => {
expect(relPlanningPath('frontend')).toBe('.planning/workstreams/frontend');
expect(relPlanningPath('api_v2')).toBe('.planning/workstreams/api_v2');
expect(relPlanningPath('alpha.beta-1')).toBe('.planning/workstreams/alpha.beta-1');
});
for (const bad of TRAVERSAL_CASES) {
it(`throws for invalid workstream name ${JSON.stringify(bad)}`, () => {
expect(() => relPlanningPath(bad)).toThrow(/workstream/i);
});
}
it('throws BEFORE constructing the path (no partial side effect)', () => {
let resultPath: string | null = null;
try {
resultPath = relPlanningPath('../../../outside');
} catch {
/* expected */
}
expect(resultPath).toBeNull();
});
});
describe('bug #3589: planningPaths rejects explicit invalid workstream names', () => {
it('throws for explicit ../../../outside (was silently constructing a traversal path)', () => {
expect(() => planningPaths('/tmp/projectDir', '../../../outside')).toThrow(/workstream/i);
});
it('throws for explicit slash-bearing names', () => {
expect(() => planningPaths('/tmp/projectDir', 'foo/bar')).toThrow(/workstream/i);
});
it('accepts valid explicit names and constructs the expected planning subtree', () => {
const paths = planningPaths('/tmp/projectDir', 'frontend');
expect(paths.planning.endsWith('.planning/workstreams/frontend')).toBe(true);
expect(paths.state.endsWith('.planning/workstreams/frontend/STATE.md')).toBe(true);
expect(paths.roadmap.endsWith('.planning/workstreams/frontend/ROADMAP.md')).toBe(true);
});
it('still returns root .planning when workstream is omitted', () => {
const paths = planningPaths('/tmp/projectDir');
expect(paths.planning.endsWith('.planning')).toBe(true);
expect(paths.planning).not.toContain('workstreams');
});
});

View File

@@ -1,179 +0,0 @@
/**
* Bug #3591: createGSDToolsRuntime accepts a `workstream` option, but the
* native dispatch closure passed to QueryNativeDirectAdapter dropped it
* before forwarding to registry.dispatch(). The omission silently routed
* native query handlers to root `.planning` instead of
* `.planning/workstreams/<name>` whenever a GSDTools instance was created
* with a workstream and native dispatch was used.
*
* The fix passes `opts.workstream` as the 4th argument to
* `registry.dispatch(command, args, projectDir, workstream)`. This test
* captures the dispatch closure via a constructor-seam spy on
* QueryNativeDirectAdapter, builds a mock registry whose dispatch records
* its arguments, then invokes the captured closure to verify the
* workstream is forwarded.
*/
import { describe, it, expect, vi } from 'vitest';
import { createGSDToolsRuntime } from './query-gsd-tools-runtime.js';
import * as adapterModule from './query-native-direct-adapter.js';
import * as registryModule from './query/index.js';
describe('bug #3591: createGSDToolsRuntime forwards workstream to registry.dispatch', () => {
it('native dispatch closure passes opts.workstream as 4th arg to registry.dispatch', async () => {
// Capture the `dispatch` option passed into QueryNativeDirectAdapter.
let capturedDispatch:
| ((command: string, args: string[]) => Promise<unknown>)
| null = null;
const adapterSpy = vi
.spyOn(adapterModule, 'QueryNativeDirectAdapter')
// eslint-disable-next-line @typescript-eslint/no-explicit-any
.mockImplementation((deps: any) => {
capturedDispatch = deps.dispatch;
// Return a minimal stub satisfying the runtime constructor.
return {
dispatchResult: vi.fn(),
dispatchJson: vi.fn(),
dispatchRaw: vi.fn(),
} as unknown as adapterModule.QueryNativeDirectAdapter;
});
const registry = registryModule.createRegistry();
const dispatchSpy = vi.spyOn(registry, 'dispatch');
const createRegistrySpy = vi
.spyOn(registryModule, 'createRegistry')
.mockReturnValue(registry);
try {
createGSDToolsRuntime({
projectDir: '/tmp/3591-proj',
gsdToolsPath: '/tmp/gsd-tools.cjs',
timeoutMs: 1_000,
workstream: 'frontend-ws',
shouldUseNativeQuery: () => true,
execJsonFallback: vi.fn(async () => ({})),
execRawFallback: vi.fn(async () => ''),
});
expect(adapterSpy).toHaveBeenCalled();
expect(capturedDispatch).not.toBeNull();
await capturedDispatch!('__bug-3591-unknown-cmd__', ['x']);
} catch (err) {
// unknown command is expected from the real registry
void err;
} finally {
createRegistrySpy.mockRestore();
adapterSpy.mockRestore();
}
expect(dispatchSpy).toHaveBeenCalledWith(
'__bug-3591-unknown-cmd__',
['x'],
'/tmp/3591-proj',
'frontend-ws',
);
});
it('forwards undefined workstream when the option is omitted (back-compat)', async () => {
// Same shape as above but no workstream. The closure must still pass
// projectDir; passing `undefined` for the 4th slot is the documented
// signature of registry.dispatch.
let capturedDispatch:
| ((command: string, args: string[]) => Promise<unknown>)
| null = null;
const registry = registryModule.createRegistry();
const dispatchSpy = vi.spyOn(registry, 'dispatch');
const adapterSpy = vi
.spyOn(adapterModule, 'QueryNativeDirectAdapter')
// eslint-disable-next-line @typescript-eslint/no-explicit-any
.mockImplementation((deps: any) => {
capturedDispatch = deps.dispatch;
return {
dispatchResult: vi.fn(),
dispatchJson: vi.fn(),
dispatchRaw: vi.fn(),
} as unknown as adapterModule.QueryNativeDirectAdapter;
});
const createRegistrySpy = vi
.spyOn(registryModule, 'createRegistry')
.mockReturnValue(registry);
try {
createGSDToolsRuntime({
projectDir: '/tmp/3591-proj',
gsdToolsPath: '/tmp/gsd-tools.cjs',
timeoutMs: 1_000,
// workstream intentionally omitted
shouldUseNativeQuery: () => true,
execJsonFallback: vi.fn(async () => ({})),
execRawFallback: vi.fn(async () => ''),
});
expect(capturedDispatch).not.toBeNull();
await capturedDispatch!('__bug-3591-unknown-cmd-2__', []);
} catch (err) {
// unknown command is expected from the real registry
void err;
} finally {
createRegistrySpy.mockRestore();
adapterSpy.mockRestore();
}
expect(dispatchSpy).toHaveBeenCalledWith(
'__bug-3591-unknown-cmd-2__',
[],
'/tmp/3591-proj',
undefined,
);
});
});
describe('bug #3591: end-to-end — workstream-aware probe handler receives the workstream', () => {
it('a registered probe handler receives opts.workstream as its 3rd arg', async () => {
// End-to-end path: register a probe handler on a real registry (via
// module-level export), build the runtime with a workstream, and
// assert the handler observed the workstream when invoked through the
// native dispatch closure.
const registryModule = await import('./query/index.js');
const probeRegistry = registryModule.createRegistry();
const seen: Array<{ args: string[]; projectDir: string; workstream?: string }> = [];
probeRegistry.register('__bug-3591-probe__', async (args, projectDir, workstream) => {
seen.push({ args, projectDir, workstream });
return { data: { ok: true } };
});
// The runtime builds its OWN registry internally; we can't substitute
// ours unless we mock createRegistry. Hoist a module mock for that.
const createRegistrySpy = vi
.spyOn(registryModule, 'createRegistry')
.mockReturnValue(probeRegistry);
try {
const runtime = createGSDToolsRuntime({
projectDir: '/tmp/3591-proj',
gsdToolsPath: '/tmp/gsd-tools.cjs',
timeoutMs: 1_000,
workstream: 'frontend-ws',
shouldUseNativeQuery: () => true,
execJsonFallback: vi.fn(async () => ({})),
execRawFallback: vi.fn(async () => ''),
});
await runtime.bridge.dispatchHotpath(
'__bug-3591-probe-legacy__',
[],
'__bug-3591-probe__',
['payload'],
'json',
);
} finally {
createRegistrySpy.mockRestore();
}
expect(seen).toHaveLength(1);
expect(seen[0]?.args).toEqual(['payload']);
expect(seen[0]?.projectDir).toBe('/tmp/3591-proj');
expect(seen[0]?.workstream).toBe('frontend-ws');
});
});

View File

@@ -1,388 +0,0 @@
import { describe, it, expect } from 'vitest';
import { PassThrough } from 'node:stream';
import { CLITransport } from './cli-transport.js';
import { GSDEventType, type GSDEvent, type GSDEventBase } from './types.js';
// ─── ANSI constants (mirror the source for readable assertions) ──────────────
const BOLD = '\x1b[1m';
const RESET = '\x1b[0m';
const GREEN = '\x1b[32m';
const RED = '\x1b[31m';
const YELLOW = '\x1b[33m';
const CYAN = '\x1b[36m';
const DIM = '\x1b[90m';
// ─── Helpers ─────────────────────────────────────────────────────────────────
function makeBase(overrides: Partial<GSDEventBase> = {}): Omit<GSDEventBase, 'type'> {
return {
timestamp: '2025-06-15T14:30:45.123Z',
sessionId: 'test-session',
...overrides,
};
}
function readOutput(stream: PassThrough): string {
const chunks: Buffer[] = [];
let chunk: Buffer | null;
while ((chunk = stream.read() as Buffer | null) !== null) {
chunks.push(chunk);
}
return Buffer.concat(chunks).toString('utf-8').trim();
}
// ─── Tests ───────────────────────────────────────────────────────────────────
describe('CLITransport', () => {
it('formats SessionInit event correctly', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.SessionInit,
model: 'claude-sonnet-4-20250514',
tools: ['Read', 'Write', 'Bash'],
cwd: '/home/project',
} as GSDEvent);
const output = readOutput(stream);
expect(output).toBe(
'[14:30:45] [INIT] Session started — model: claude-sonnet-4-20250514, tools: 3, cwd: /home/project',
);
});
it('formats SessionComplete in green with checkmark', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.SessionComplete,
success: true,
totalCostUsd: 1.234,
durationMs: 45600,
numTurns: 12,
result: 'done',
} as GSDEvent);
const output = readOutput(stream);
expect(output).toBe(
`[14:30:45] ${GREEN}✓ Session complete — cost: $1.23, turns: 12, duration: 45.6s${RESET}`,
);
});
it('formats SessionError in red with ✗ marker', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.SessionError,
success: false,
totalCostUsd: 0.5,
durationMs: 3000,
numTurns: 2,
errorSubtype: 'tool_error',
errors: ['file not found', 'permission denied'],
} as GSDEvent);
const output = readOutput(stream);
expect(output).toBe(
`[14:30:45] ${RED}✗ Session failed — subtype: tool_error, errors: [file not found, permission denied]${RESET}`,
);
});
it('formats PhaseStart as bold cyan banner and PhaseComplete with running cost', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.PhaseStart,
phaseNumber: '01',
phaseName: 'Authentication',
} as GSDEvent);
transport.onEvent({
...makeBase(),
type: GSDEventType.PhaseComplete,
phaseNumber: '01',
phaseName: 'Authentication',
success: true,
totalCostUsd: 2.50,
totalDurationMs: 60000,
stepsCompleted: 5,
} as GSDEvent);
const output = readOutput(stream);
const lines = output.split('\n');
expect(lines[0]).toBe(`${BOLD}${CYAN}━━━ GSD ► PHASE 01: Authentication ━━━${RESET}`);
expect(lines[1]).toBe('[14:30:45] [PHASE] Phase 01 complete — success: true, cost: $2.50, running: $0.00');
});
it('formats ToolCall with truncated input', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
const longInput = { content: 'x'.repeat(200) };
transport.onEvent({
...makeBase(),
type: GSDEventType.ToolCall,
toolName: 'Write',
toolUseId: 'tool-123',
input: longInput,
} as GSDEvent);
const output = readOutput(stream);
expect(output).toMatch(/^\[14:30:45\] \[TOOL\] Write\(.+…\)$/);
// The truncated input portion (inside parens) should be ≤80 chars
const insideParens = output.match(/Write\((.+)\)/)![1]!;
expect(insideParens.length).toBeLessThanOrEqual(80);
});
it('formats MilestoneStart as bold banner and MilestoneComplete with running cost', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.MilestoneStart,
phaseCount: 3,
prompt: 'build the app',
} as GSDEvent);
transport.onEvent({
...makeBase(),
type: GSDEventType.MilestoneComplete,
success: true,
totalCostUsd: 8.75,
totalDurationMs: 300000,
phasesCompleted: 3,
} as GSDEvent);
const output = readOutput(stream);
const lines = output.split('\n');
// MilestoneStart emits 3 lines (top bar, text, bottom bar)
expect(lines[0]).toBe(`${BOLD}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${RESET}`);
expect(lines[1]).toBe(`${BOLD} GSD Milestone — 3 phases${RESET}`);
expect(lines[2]).toBe(`${BOLD}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${RESET}`);
expect(lines[3]).toBe(`${BOLD}━━━ Milestone complete — success: true, cost: $8.75, running: $0.00 ━━━${RESET}`);
});
it('close() is callable without error', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
expect(() => transport.close()).not.toThrow();
});
it('onEvent does not throw on unknown event type variant', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
// Use a known event type that hits the default/fallback branch
transport.onEvent({
...makeBase(),
type: GSDEventType.ToolProgress,
toolName: 'Bash',
toolUseId: 'tool-456',
elapsedSeconds: 12,
} as GSDEvent);
const output = readOutput(stream);
expect(output).toBe('[14:30:45] [EVENT] tool_progress');
});
it('formats AssistantText as dim with truncation at 200 chars', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
const longText = 'A'.repeat(300);
transport.onEvent({
...makeBase(),
type: GSDEventType.AssistantText,
text: longText,
} as GSDEvent);
const output = readOutput(stream);
expect(output).toMatch(new RegExp(`^${escRe(DIM)}\\[14:30:45\\] A+…${escRe(RESET)}$`));
// Strip ANSI to check text length
const stripped = stripAnsi(output);
const agentText = stripped.split('] ')[1]!;
expect(agentText.length).toBeLessThanOrEqual(200);
});
it('formats WaveStart in yellow and WaveComplete with colored counts', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.WaveStart,
phaseNumber: '01',
waveNumber: 2,
planCount: 4,
planIds: ['plan-a', 'plan-b', 'plan-c', 'plan-d'],
} as GSDEvent);
transport.onEvent({
...makeBase(),
type: GSDEventType.WaveComplete,
phaseNumber: '01',
waveNumber: 2,
successCount: 3,
failureCount: 1,
durationMs: 25000,
} as GSDEvent);
const output = readOutput(stream);
const lines = output.split('\n');
expect(lines[0]).toBe(`${YELLOW}⟫ Wave 2 (4 plans)${RESET}`);
expect(lines[1]).toBe(
`[14:30:45] [WAVE] Wave 2 complete — ${GREEN}3 success${RESET}, ${RED}1 failed${RESET}, 25000ms`,
);
});
// ─── New tests for rich formatting ─────────────────────────────────────────
it('formats PhaseStepStart in cyan with ◆ indicator', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.PhaseStepStart,
phaseNumber: '01',
step: 'research',
} as GSDEvent);
const output = readOutput(stream);
expect(output).toBe(`${CYAN}◆ research${RESET}`);
});
it('formats PhaseStepComplete green ✓ on success, red ✗ on failure', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.PhaseStepComplete,
phaseNumber: '01',
step: 'plan',
success: true,
durationMs: 5200,
} as GSDEvent);
transport.onEvent({
...makeBase(),
type: GSDEventType.PhaseStepComplete,
phaseNumber: '01',
step: 'execute',
success: false,
durationMs: 12000,
} as GSDEvent);
const output = readOutput(stream);
const lines = output.split('\n');
expect(lines[0]).toBe(`${GREEN}✓ plan${RESET} ${DIM}5200ms${RESET}`);
expect(lines[1]).toBe(`${RED}✗ execute${RESET} ${DIM}12000ms${RESET}`);
});
it('formats InitResearchSpawn in cyan with ◆ and session count', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
transport.onEvent({
...makeBase(),
type: GSDEventType.InitResearchSpawn,
sessionCount: 4,
researchTypes: ['stack', 'features', 'architecture', 'pitfalls'],
} as GSDEvent);
const output = readOutput(stream);
expect(output).toBe(`${CYAN}◆ Spawning 4 researchers...${RESET}`);
});
it('tracks running cost across CostUpdate events', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
// First cost update
transport.onEvent({
...makeBase(),
type: GSDEventType.CostUpdate,
sessionCostUsd: 0.50,
cumulativeCostUsd: 0.50,
} as GSDEvent);
// Second cost update
transport.onEvent({
...makeBase(),
type: GSDEventType.CostUpdate,
sessionCostUsd: 0.75,
cumulativeCostUsd: 1.25,
} as GSDEvent);
const output = readOutput(stream);
const lines = output.split('\n');
expect(lines[0]).toBe(`${DIM}[14:30:45] Cost: session $0.50, running $0.50${RESET}`);
expect(lines[1]).toBe(`${DIM}[14:30:45] Cost: session $0.75, running $1.25${RESET}`);
});
it('shows running cost in PhaseComplete and MilestoneComplete after CostUpdates', () => {
const stream = new PassThrough();
const transport = new CLITransport(stream);
// Accumulate some cost
transport.onEvent({
...makeBase(),
type: GSDEventType.CostUpdate,
sessionCostUsd: 1.50,
cumulativeCostUsd: 1.50,
} as GSDEvent);
transport.onEvent({
...makeBase(),
type: GSDEventType.PhaseComplete,
phaseNumber: '02',
phaseName: 'Build',
success: true,
totalCostUsd: 1.50,
totalDurationMs: 30000,
stepsCompleted: 3,
} as GSDEvent);
transport.onEvent({
...makeBase(),
type: GSDEventType.MilestoneComplete,
success: true,
totalCostUsd: 1.50,
totalDurationMs: 30000,
phasesCompleted: 2,
} as GSDEvent);
const output = readOutput(stream);
const lines = output.split('\n');
// CostUpdate line
expect(lines[0]).toContain('running $1.50');
// PhaseComplete includes running cost
expect(lines[1]).toContain('running: $1.50');
// MilestoneComplete includes running cost
expect(lines[2]).toContain('running: $1.50');
});
});
// ─── Test utilities ──────────────────────────────────────────────────────────
/** Escape a string for use in a RegExp. */
function escRe(s: string): string {
return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
}
/** Strip ANSI escape sequences from a string. */
function stripAnsi(s: string): string {
return s.replace(/\x1b\[[0-9;]*m/g, '');
}

View File

@@ -1,130 +0,0 @@
/**
* CLI Transport — renders GSD events as rich ANSI-colored output to a Writable stream.
*
* Implements TransportHandler with colored banners, step indicators, spawn markers,
* and running cost totals. No external dependencies — ANSI codes are inline constants.
*/
import type { Writable } from 'node:stream';
import { GSDEventType, type GSDEvent, type TransportHandler } from './types.js';
// ─── ANSI escape constants (no dependency per D021) ──────────────────────────
const BOLD = '\x1b[1m';
const RESET = '\x1b[0m';
const GREEN = '\x1b[32m';
const RED = '\x1b[31m';
const YELLOW = '\x1b[33m';
const CYAN = '\x1b[36m';
const DIM = '\x1b[90m';
// ─── Helpers ─────────────────────────────────────────────────────────────────
/** Extract HH:MM:SS from an ISO-8601 timestamp. */
function formatTime(ts: string): string {
try {
const d = new Date(ts);
if (Number.isNaN(d.getTime())) return '??:??:??';
return d.toISOString().slice(11, 19);
} catch {
return '??:??:??';
}
}
/** Truncate a string to `max` characters, appending '…' if truncated. */
function truncate(s: string, max: number): string {
if (s.length <= max) return s;
return s.slice(0, max - 1) + '…';
}
/** Format a USD amount. */
function usd(n: number): string {
return `$${n.toFixed(2)}`;
}
// ─── CLITransport ────────────────────────────────────────────────────────────
export class CLITransport implements TransportHandler {
private readonly out: Writable;
private runningCostUsd = 0;
constructor(out?: Writable) {
this.out = out ?? process.stdout;
}
/** Format and write a GSD event as a rich ANSI-colored line. Never throws. */
onEvent(event: GSDEvent): void {
try {
const line = this.formatEvent(event);
this.out.write(line + '\n');
} catch {
// TransportHandler contract: onEvent must never throw
}
}
/** No-op — stdout doesn't need cleanup. */
close(): void {
// Nothing to clean up
}
// ─── Private formatting ────────────────────────────────────────────
private formatEvent(event: GSDEvent): string {
const time = formatTime(event.timestamp);
switch (event.type) {
case GSDEventType.SessionInit:
return `[${time}] [INIT] Session started — model: ${event.model}, tools: ${event.tools.length}, cwd: ${event.cwd}`;
case GSDEventType.SessionComplete:
return `[${time}] ${GREEN}✓ Session complete — cost: ${usd(event.totalCostUsd)}, turns: ${event.numTurns}, duration: ${(event.durationMs / 1000).toFixed(1)}s${RESET}`;
case GSDEventType.SessionError:
return `[${time}] ${RED}✗ Session failed — subtype: ${event.errorSubtype}, errors: [${event.errors.join(', ')}]${RESET}`;
case GSDEventType.ToolCall:
return `[${time}] [TOOL] ${event.toolName}(${truncate(JSON.stringify(event.input), 80)})`;
case GSDEventType.PhaseStart:
return `${BOLD}${CYAN}━━━ GSD ► PHASE ${event.phaseNumber}: ${event.phaseName} ━━━${RESET}`;
case GSDEventType.PhaseComplete:
return `[${time}] [PHASE] Phase ${event.phaseNumber} complete — success: ${event.success}, cost: ${usd(event.totalCostUsd)}, running: ${usd(this.runningCostUsd)}`;
case GSDEventType.PhaseStepStart:
return `${CYAN}◆ ${event.step}${RESET}`;
case GSDEventType.PhaseStepComplete:
return event.success
? `${GREEN}✓ ${event.step}${RESET} ${DIM}${event.durationMs}ms${RESET}`
: `${RED}✗ ${event.step}${RESET} ${DIM}${event.durationMs}ms${RESET}`;
case GSDEventType.WaveStart:
return `${YELLOW}⟫ Wave ${event.waveNumber} (${event.planCount} plans)${RESET}`;
case GSDEventType.WaveComplete:
return `[${time}] [WAVE] Wave ${event.waveNumber} complete — ${GREEN}${event.successCount} success${RESET}, ${RED}${event.failureCount} failed${RESET}, ${event.durationMs}ms`;
case GSDEventType.CostUpdate: {
this.runningCostUsd += event.sessionCostUsd;
return `${DIM}[${time}] Cost: session ${usd(event.sessionCostUsd)}, running ${usd(this.runningCostUsd)}${RESET}`;
}
case GSDEventType.MilestoneStart:
return `${BOLD}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${RESET}\n${BOLD} GSD Milestone — ${event.phaseCount} phases${RESET}\n${BOLD}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${RESET}`;
case GSDEventType.MilestoneComplete:
return `${BOLD}━━━ Milestone complete — success: ${event.success}, cost: ${usd(event.totalCostUsd)}, running: ${usd(this.runningCostUsd)} ━━━${RESET}`;
case GSDEventType.AssistantText:
return `${DIM}[${time}] ${truncate(event.text, 200)}${RESET}`;
case GSDEventType.InitResearchSpawn:
return `${CYAN}◆ Spawning ${event.sessionCount} researchers...${RESET}`;
// Generic fallback for event types without specific formatting
default:
return `[${time}] [EVENT] ${event.type}`;
}
}
}

View File

@@ -1,426 +0,0 @@
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
import { parseCliArgs, resolveInitInput, USAGE, type ParsedCliArgs } from './cli.js';
import { mkdir, writeFile, rm } from 'node:fs/promises';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
describe('parseCliArgs', () => {
it('parses run <prompt> with defaults', () => {
const result = parseCliArgs(['run', 'build auth']);
expect(result.command).toBe('run');
expect(result.prompt).toBe('build auth');
expect(result.help).toBe(false);
expect(result.version).toBe(false);
expect(result.wsPort).toBeUndefined();
expect(result.model).toBeUndefined();
expect(result.maxBudget).toBeUndefined();
});
it('parses --help flag', () => {
const result = parseCliArgs(['--help']);
expect(result.help).toBe(true);
expect(result.command).toBeUndefined();
});
it('parses -h short flag', () => {
const result = parseCliArgs(['-h']);
expect(result.help).toBe(true);
});
it('parses --version flag', () => {
const result = parseCliArgs(['--version']);
expect(result.version).toBe(true);
});
it('parses -v short flag', () => {
const result = parseCliArgs(['-v']);
expect(result.version).toBe(true);
});
it('parses --ws-port as number', () => {
const result = parseCliArgs(['run', 'build X', '--ws-port', '8080']);
expect(result.command).toBe('run');
expect(result.prompt).toBe('build X');
expect(result.wsPort).toBe(8080);
});
it('parses --model option', () => {
const result = parseCliArgs(['run', 'build X', '--model', 'claude-sonnet-4-6']);
expect(result.model).toBe('claude-sonnet-4-6');
});
it('parses --max-budget option', () => {
const result = parseCliArgs(['run', 'build X', '--max-budget', '10']);
expect(result.maxBudget).toBe(10);
});
it('parses --project-dir option', () => {
const result = parseCliArgs(['run', 'build X', '--project-dir', '/tmp/my-project']);
expect(result.projectDir).toBe('/tmp/my-project');
});
it('returns undefined command and prompt for empty args', () => {
const result = parseCliArgs([]);
expect(result.command).toBeUndefined();
expect(result.prompt).toBeUndefined();
expect(result.help).toBe(false);
expect(result.version).toBe(false);
});
it('parses multi-word prompts from positionals', () => {
const result = parseCliArgs(['run', 'build', 'the', 'entire', 'app']);
expect(result.prompt).toBe('build the entire app');
});
it('handles all options combined', () => {
const result = parseCliArgs([
'run', 'build auth',
'--project-dir', '/tmp/proj',
'--ws-port', '9090',
'--model', 'claude-sonnet-4-6',
'--max-budget', '15',
]);
expect(result.command).toBe('run');
expect(result.prompt).toBe('build auth');
expect(result.projectDir).toBe('/tmp/proj');
expect(result.wsPort).toBe(9090);
expect(result.model).toBe('claude-sonnet-4-6');
expect(result.maxBudget).toBe(15);
});
it('rejects unknown options (strict parser)', () => {
expect(() => parseCliArgs(['--unknown-flag'])).toThrow();
});
it('rejects unknown flags on run command', () => {
expect(() => parseCliArgs(['run', 'hello', '--not-a-real-option'])).toThrow();
});
it('parses query permissively (keeps gsd-tools flags like --pick, --json)', () => {
const result = parseCliArgs([
'query', 'state.load', '--pick', 'data', '--project-dir', 'C:\\tmp\\proj',
]);
expect(result.command).toBe('query');
expect(result.projectDir).toBe('C:\\tmp\\proj');
expect(result.queryArgv).toEqual(['state.load', '--pick', 'data']);
});
it('parses query with extra flags forwarded in queryArgv', () => {
const result = parseCliArgs([
'query', 'audit-open', '--json', '--project-dir', 'D:\\proj',
]);
expect(result.command).toBe('query');
expect(result.projectDir).toBe('D:\\proj');
expect(result.queryArgv).toEqual(['audit-open', '--json']);
});
// ─── #3019: --help inside `query <subcommand>` reaches the handler ────
it('forwards --help to queryArgv when a subcommand precedes it (#3019)', () => {
// gsd-sdk query phase add --help
// Previously: --help was harvested as global, queryArgv = ['phase', 'add'],
// help: true → main() short-circuits to top-level USAGE, never dispatching.
// Now: --help travels with the rest of queryArgv so the registry handler
// (or the gsd-tools.cjs fallback) can render contextual subcommand help.
const result = parseCliArgs(['query', 'phase', 'add', '--help']);
expect(result.command).toBe('query');
expect(result.queryArgv).toEqual(['phase', 'add', '--help']);
// The global help flag must NOT short-circuit dispatch when there is a
// subcommand to dispatch to.
expect(result.help).toBe(false);
});
it('forwards -h to queryArgv when a subcommand precedes it (#3019)', () => {
const result = parseCliArgs(['query', 'init', '-h']);
expect(result.queryArgv).toEqual(['init', '-h']);
expect(result.help).toBe(false);
});
it('treats bare `query --help` as a top-level help request (no subcommand to dispatch to)', () => {
// gsd-sdk query --help
// No subcommand follows, so the only useful response is the top-level
// USAGE. Preserve existing behavior: help: true.
const result = parseCliArgs(['query', '--help']);
expect(result.command).toBe('query');
expect(result.help).toBe(true);
// queryArgv may be empty or carry just the lone --help; either is fine
// because main() short-circuits on help when there is no subcommand.
expect((result.queryArgv ?? []).filter((x) => x !== '--help' && x !== '-h')).toEqual([]);
});
it('preserves --help position when intermixed with other query flags (#3019)', () => {
// gsd-sdk query phase --help --pick name
// The handler/fallback should see --help in argv so it can render help
// even when other flags are present.
const result = parseCliArgs(['query', 'phase', '--help', '--pick', 'name']);
expect(result.queryArgv).toEqual(['phase', '--help', '--pick', 'name']);
expect(result.help).toBe(false);
});
// ─── Init command parsing ──────────────────────────────────────────────
it('parses init with @file input', () => {
const result = parseCliArgs(['init', '@prd.md']);
expect(result.command).toBe('init');
expect(result.initInput).toBe('@prd.md');
expect(result.prompt).toBe('@prd.md');
});
it('parses init with raw text input', () => {
const result = parseCliArgs(['init', 'build a todo app']);
expect(result.command).toBe('init');
expect(result.initInput).toBe('build a todo app');
});
it('parses init with multi-word text input', () => {
const result = parseCliArgs(['init', 'build', 'a', 'todo', 'app']);
expect(result.command).toBe('init');
expect(result.initInput).toBe('build a todo app');
});
it('parses init with no input (stdin mode)', () => {
const result = parseCliArgs(['init']);
expect(result.command).toBe('init');
expect(result.initInput).toBeUndefined();
expect(result.prompt).toBeUndefined();
});
it('parses init with options', () => {
const result = parseCliArgs(['init', '@prd.md', '--project-dir', '/tmp/proj', '--model', 'claude-sonnet-4-6']);
expect(result.command).toBe('init');
expect(result.initInput).toBe('@prd.md');
expect(result.projectDir).toBe('/tmp/proj');
expect(result.model).toBe('claude-sonnet-4-6');
});
it('does not set initInput for non-init commands', () => {
const result = parseCliArgs(['run', 'build auth']);
expect(result.command).toBe('run');
expect(result.initInput).toBeUndefined();
expect(result.prompt).toBe('build auth');
});
// ─── Auto command parsing ──────────────────────────────────────────────
it('parses auto command with no prompt', () => {
const result = parseCliArgs(['auto']);
expect(result.command).toBe('auto');
expect(result.prompt).toBeUndefined();
expect(result.initInput).toBeUndefined();
});
it('parses auto with --project-dir', () => {
const result = parseCliArgs(['auto', '--project-dir', '/tmp/x']);
expect(result.command).toBe('auto');
expect(result.projectDir).toBe('/tmp/x');
});
it('parses auto with --ws-port', () => {
const result = parseCliArgs(['auto', '--ws-port', '9090']);
expect(result.command).toBe('auto');
expect(result.wsPort).toBe(9090);
});
it('parses auto with all options combined', () => {
const result = parseCliArgs([
'auto',
'--project-dir', '/tmp/proj',
'--ws-port', '8080',
'--model', 'claude-sonnet-4-6',
'--max-budget', '20',
]);
expect(result.command).toBe('auto');
expect(result.projectDir).toBe('/tmp/proj');
expect(result.wsPort).toBe(8080);
expect(result.model).toBe('claude-sonnet-4-6');
expect(result.maxBudget).toBe(20);
});
it('auto command does not set initInput', () => {
const result = parseCliArgs(['auto']);
expect(result.initInput).toBeUndefined();
});
// ─── Auto --init parsing ──────────────────────────────────────────────
it('parses auto --init with @file', () => {
const result = parseCliArgs(['auto', '--init', '@prd.md']);
expect(result.command).toBe('auto');
expect(result.init).toBe('@prd.md');
expect(result.initInput).toBeUndefined();
});
it('parses auto --init with raw text', () => {
const result = parseCliArgs(['auto', '--init', 'build a todo app']);
expect(result.command).toBe('auto');
expect(result.init).toBe('build a todo app');
});
it('parses auto --init with other options', () => {
const result = parseCliArgs([
'auto',
'--init', '@spec.md',
'--project-dir', '/tmp/proj',
'--model', 'claude-sonnet-4-6',
'--max-budget', '25',
]);
expect(result.command).toBe('auto');
expect(result.init).toBe('@spec.md');
expect(result.projectDir).toBe('/tmp/proj');
expect(result.model).toBe('claude-sonnet-4-6');
expect(result.maxBudget).toBe(25);
});
it('init is undefined when --init not provided', () => {
const result = parseCliArgs(['auto']);
expect(result.init).toBeUndefined();
});
it('init is undefined for non-auto commands', () => {
const result = parseCliArgs(['run', 'build auth']);
expect(result.init).toBeUndefined();
});
});
// ─── resolveInitInput tests ──────────────────────────────────────────────────
describe('resolveInitInput', () => {
let tmpDir: string;
beforeEach(async () => {
tmpDir = join(tmpdir(), `cli-init-test-${Date.now()}-${Math.random().toString(36).slice(2)}`);
await mkdir(tmpDir, { recursive: true });
});
afterEach(async () => {
await rm(tmpDir, { recursive: true, force: true });
});
function makeArgs(overrides: Partial<ParsedCliArgs>): ParsedCliArgs {
return {
command: 'init',
prompt: undefined,
initInput: undefined,
init: undefined,
projectDir: tmpDir,
wsPort: undefined,
model: undefined,
maxBudget: undefined,
help: false,
version: false,
...overrides,
};
}
it('reads file contents when input starts with @', async () => {
const prdPath = join(tmpDir, 'prd.md');
await writeFile(prdPath, '# My PRD\n\nBuild a todo app');
const result = await resolveInitInput(makeArgs({ initInput: '@prd.md' }));
expect(result).toBe('# My PRD\n\nBuild a todo app');
});
it('resolves @file path relative to projectDir', async () => {
const subDir = join(tmpDir, 'docs');
await mkdir(subDir, { recursive: true });
await writeFile(join(subDir, 'spec.md'), 'specification content');
const result = await resolveInitInput(makeArgs({ initInput: '@docs/spec.md' }));
expect(result).toBe('specification content');
});
it('throws descriptive error when @file does not exist', async () => {
await expect(
resolveInitInput(makeArgs({ initInput: '@nonexistent.md' }))
).rejects.toThrow('file not found');
});
it('returns raw text as-is when input does not start with @', async () => {
const result = await resolveInitInput(makeArgs({ initInput: 'build a todo app' }));
expect(result).toBe('build a todo app');
});
it('throws TTY error when no input and stdin is TTY', async () => {
// In test environment, stdin.isTTY is typically undefined (not a TTY),
// but we can verify the function throws when stdin is a TTY by
// checking the error path directly via the export.
// This test verifies the raw text path works for empty-like scenarios.
const result = await resolveInitInput(makeArgs({ initInput: 'some text' }));
expect(result).toBe('some text');
});
it('reads @file with absolute path', async () => {
const absPath = join(tmpDir, 'absolute-prd.md');
await writeFile(absPath, 'absolute path content');
// Absolute paths are resolved relative to projectDir, so we need
// to use the relative form or the absolute form via @
const result = await resolveInitInput(makeArgs({ initInput: `@${absPath}` }));
expect(result).toBe('absolute path content');
});
it('preserves whitespace in raw text input', async () => {
const input = ' build a todo app with spaces ';
const result = await resolveInitInput(makeArgs({ initInput: input }));
expect(result).toBe(input);
});
it('reads large file content from @file', async () => {
const largeContent = 'x'.repeat(10000) + '\n# PRD\nDescription here';
await writeFile(join(tmpDir, 'large.md'), largeContent);
const result = await resolveInitInput(makeArgs({ initInput: '@large.md' }));
expect(result).toBe(largeContent);
});
});
// ─── USAGE text tests ────────────────────────────────────────────────────────
describe('USAGE', () => {
it('includes auto command', () => {
expect(USAGE).toContain('auto');
});
it('describes auto as autonomous lifecycle', () => {
expect(USAGE).toMatch(/auto\s+.*autonomous/i);
});
it('documents --init option', () => {
expect(USAGE).toContain('--init');
expect(USAGE).toContain('Bootstrap from a PRD');
});
});

View File

@@ -1,589 +0,0 @@
#!/usr/bin/env node
/**
* CLI entry point for gsd-sdk.
*
* Usage: gsd-sdk run "<prompt>" [--project-dir <dir>] [--ws-port <port>]
* [--model <model>] [--max-budget <n>]
*/
import { parseArgs } from 'node:util';
import { readFile } from 'node:fs/promises';
import { resolve, join, isAbsolute } from 'node:path';
import { fileURLToPath } from 'node:url';
import { GSD } from './index.js';
import { CLITransport } from './cli-transport.js';
import { WSTransport } from './ws-transport.js';
import { InitRunner } from './init-runner.js';
import { validateWorkstreamName } from './workstream-utils.js';
import { loadConfig } from './config.js';
import { assertRuntimeSupportsAutoMode } from './runtime-gate.js';
import { runQueryCliCommand } from './query/query-cli-adapter.js';
// ─── Parsed CLI args ─────────────────────────────────────────────────────────
export interface ParsedCliArgs {
command: string | undefined;
prompt: string | undefined;
/** For 'init' command: the raw input source (@file, text, or undefined for stdin). */
initInput: string | undefined;
/** For 'auto --init': bootstrap from a PRD before running the autonomous loop. */
init: string | undefined;
projectDir: string;
wsPort: number | undefined;
model: string | undefined;
maxBudget: number | undefined;
/** Workstream name for multi-workstream projects. Routes .planning/ to .planning/workstreams/<name>/. */
ws: string | undefined;
help: boolean;
version: boolean;
/**
* When `command === 'query'`, tokens after `query` with only known SDK flags removed.
* Extra flags are kept so handlers that share gsd-tools-style argv (e.g. `--pick`) still receive them.
*/
queryArgv?: string[];
}
/**
* Parse `gsd-sdk query …` without rejecting unknown flags (query argv is forwarded to the registry).
*/
function parseCliArgsQueryPermissive(argv: string[]): ParsedCliArgs {
let projectDir = process.cwd();
let ws: string | undefined;
let wsPort: number | undefined;
let model: string | undefined;
let maxBudget: number | undefined;
let help = false;
let version = false;
const queryArgv: string[] = [];
let i = 1;
while (i < argv.length) {
const a = argv[i];
if (a === '--project-dir' && argv[i + 1]) {
projectDir = argv[i + 1];
i += 2;
continue;
}
if (a === '--ws' && argv[i + 1]) {
ws = argv[i + 1];
i += 2;
continue;
}
if (a === '--ws-port' && argv[i + 1]) {
wsPort = Number(argv[i + 1]);
i += 2;
continue;
}
if (a === '--model' && argv[i + 1]) {
model = argv[i + 1];
i += 2;
continue;
}
if (a === '--max-budget' && argv[i + 1]) {
maxBudget = Number(argv[i + 1]);
i += 2;
continue;
}
// #3019: do NOT consume -h / --help here unconditionally. Pushing the
// flag onto queryArgv lets the registered handler (or the gsd-tools.cjs
// fallback) render contextual subcommand help. We still set the global
// `help` flag when the flag appears, but only short-circuit dispatch in
// main() when there is no real subcommand to dispatch to (i.e. the only
// tokens in queryArgv are the help flags themselves). That preserves
// `gsd-sdk query --help` → top-level USAGE while letting
// `gsd-sdk query phase add --help` reach the handler.
if (a === '-h' || a === '--help') {
help = true;
queryArgv.push(a);
i += 1;
continue;
}
if (a === '-v' || a === '--version') {
version = true;
i += 1;
continue;
}
queryArgv.push(a);
i += 1;
}
// If the user typed a real subcommand (anything other than help flags
// alone in queryArgv), do NOT short-circuit to top-level USAGE on help.
// The handler/fallback will render contextual help.
const nonHelpTokens = queryArgv.filter((t) => t !== '-h' && t !== '--help');
if (help && nonHelpTokens.length > 0) {
help = false;
}
return {
command: 'query',
prompt: undefined,
initInput: undefined,
init: undefined,
projectDir,
wsPort,
model,
maxBudget,
ws,
help,
version,
queryArgv,
};
}
/**
* Parse CLI arguments into a structured object.
* Exported for testing — the main() function uses this internally.
*/
export function parseCliArgs(argv: string[]): ParsedCliArgs {
if (argv[0] === 'query') {
return parseCliArgsQueryPermissive(argv);
}
const { values, positionals } = parseArgs({
args: argv,
options: {
'project-dir': { type: 'string', default: process.cwd() },
'ws-port': { type: 'string' },
ws: { type: 'string' },
model: { type: 'string' },
'max-budget': { type: 'string' },
init: { type: 'string' },
help: { type: 'boolean', short: 'h', default: false },
version: { type: 'boolean', short: 'v', default: false },
},
allowPositionals: true,
strict: true,
});
const command = positionals[0] as string | undefined;
const prompt = positionals.slice(1).join(' ') || undefined;
// For 'init' command, the positional after 'init' is the input source.
// For 'run' command, it's the prompt. Both use positionals[1+].
const initInput = command === 'init' ? prompt : undefined;
return {
command,
prompt,
initInput,
init: values.init as string | undefined,
projectDir: values['project-dir'] as string,
wsPort: values['ws-port'] ? Number(values['ws-port']) : undefined,
model: values.model as string | undefined,
maxBudget: values['max-budget'] ? Number(values['max-budget']) : undefined,
ws: values.ws as string | undefined,
help: values.help as boolean,
version: values.version as boolean,
};
}
// ─── Usage ───────────────────────────────────────────────────────────────────
export const USAGE = `
Usage: gsd-sdk <command> [args] [options]
Commands:
run <prompt> Run a full milestone from a text prompt
auto Run the full autonomous lifecycle (discover -> execute -> advance)
init [input] Bootstrap a new project from a PRD or description
input can be:
@path/to/prd.md Read input from a file
"description" Use text directly
(empty) Read from stdin
query <argv...> Registered query handlers only (longest-prefix argv match; see QUERY-HANDLERS.md)
Use --pick <field> to extract a specific field from JSON output
Options:
--init <input> Bootstrap from a PRD before running (auto only)
Accepts @path/to/prd.md or "description text"
--project-dir <dir> Project directory (default: cwd)
--ws <name> Route .planning/ to .planning/workstreams/<name>/
--ws-port <port> Enable WebSocket transport on <port>
--model <model> Override LLM model
--max-budget <n> Max budget per step in USD
-h, --help Show this help
-v, --version Show version
`.trim();
/**
* Read the package version from package.json.
*/
async function getVersion(): Promise<string> {
try {
const pkgPath = resolve(fileURLToPath(import.meta.url), '..', '..', 'package.json');
const raw = await readFile(pkgPath, 'utf-8');
const pkg = JSON.parse(raw) as { version?: string };
return pkg.version ?? 'unknown';
} catch {
return 'unknown';
}
}
// ─── Init input resolution ───────────────────────────────────────────────────
/**
* Resolve the init command input to a string.
*
* - `@path/to/file.md` → reads the file contents
* - Raw text → returns as-is
* - No input → reads from stdin (with TTY detection)
*
* Exported for testing.
*/
export async function resolveInitInput(args: ParsedCliArgs): Promise<string> {
const input = args.initInput;
if (input && input.startsWith('@')) {
// File path: strip @ prefix, resolve relative to projectDir
const filePath = resolve(args.projectDir, input.slice(1));
try {
return await readFile(filePath, 'utf-8');
} catch (err) {
throw new Error(`Cannot read input file "${filePath}": ${(err as NodeJS.ErrnoException).code === 'ENOENT' ? 'file not found' : (err as Error).message}`);
}
}
if (input) {
// Raw text
return input;
}
// No input — read from stdin
return readStdin();
}
/**
* Read all data from stdin. Rejects if stdin is a TTY with no piped data.
*/
async function readStdin(): Promise<string> {
const { stdin } = process;
if (stdin.isTTY) {
throw new Error(
'No input provided. Usage:\n' +
' gsd-sdk init @path/to/prd.md\n' +
' gsd-sdk init "build a todo app"\n' +
' cat prd.md | gsd-sdk init'
);
}
return new Promise<string>((resolve, reject) => {
const chunks: Buffer[] = [];
stdin.on('data', (chunk: Buffer) => chunks.push(chunk));
stdin.on('end', () => resolve(Buffer.concat(chunks).toString('utf-8')));
stdin.on('error', reject);
});
}
// ─── Main ────────────────────────────────────────────────────────────────────
export async function main(argv: string[] = process.argv.slice(2)): Promise<void> {
let args: ParsedCliArgs;
try {
args = parseCliArgs(argv);
} catch (err) {
console.error(`Error: ${(err as Error).message}`);
console.error(USAGE);
process.exitCode = 1;
return;
}
if (args.help) {
console.log(USAGE);
return;
}
if (args.version) {
const ver = await getVersion();
console.log(`gsd-sdk v${ver}`);
return;
}
// Validate --ws flag if provided
if (args.ws !== undefined && !validateWorkstreamName(args.ws)) {
console.error(`Error: Invalid workstream name "${args.ws}". Use alphanumeric, hyphens, underscores, or dots only.`);
process.exitCode = 1;
return;
}
// ─── Query command ──────────────────────────────────────────────────────
if (args.command === 'query') {
const result = await runQueryCliCommand({
projectDir: args.projectDir,
ws: args.ws,
queryArgv: args.queryArgv,
});
for (const line of result.stderrLines) console.error(line);
for (const chunk of result.stdoutChunks) process.stdout.write(chunk);
process.exitCode = result.exitCode;
return;
}
// Fall back to GSD_WORKSTREAM env var when --ws is not supplied (#2791).
// gsd-tools.cjs resolves the active workstream via this env var; parity
// means gsd-sdk command paths see the same .planning/ path as gsd-tools.
if (args.ws === undefined && process.env.GSD_WORKSTREAM) {
const envWs = process.env.GSD_WORKSTREAM;
if (validateWorkstreamName(envWs)) {
args = { ...args, ws: envWs };
}
}
// Multi-repo project-root resolution (issue #2623).
{
const { findProjectRoot } = await import('./query/helpers.js');
args = { ...args, projectDir: findProjectRoot(args.projectDir) };
}
if (args.command !== 'run' && args.command !== 'init' && args.command !== 'auto') {
console.error('Error: Expected "gsd-sdk run <prompt>", "gsd-sdk auto", "gsd-sdk init [input]", or "gsd-sdk query <command>"');
console.error(USAGE);
process.exitCode = 1;
return;
}
if (args.command === 'run' && !args.prompt) {
console.error('Error: "gsd-sdk run" requires a prompt');
console.error(USAGE);
process.exitCode = 1;
return;
}
// ─── Init command ─────────────────────────────────────────────────────────
if (args.command === 'init') {
let input: string;
try {
input = await resolveInitInput(args);
} catch (err) {
console.error(`Error: ${(err as Error).message}`);
process.exitCode = 1;
return;
}
console.log(`[init] Resolved input: ${input.length} chars`);
// Build GSD instance for tools and event stream
const gsd = new GSD({
projectDir: args.projectDir,
model: args.model,
maxBudgetUsd: args.maxBudget,
workstream: args.ws,
});
// Wire CLI transport
const cliTransport = new CLITransport();
gsd.addTransport(cliTransport);
// Optional WebSocket transport
let wsTransport: WSTransport | undefined;
if (args.wsPort !== undefined) {
wsTransport = new WSTransport({ port: args.wsPort });
await wsTransport.start();
gsd.addTransport(wsTransport);
console.log(`WebSocket transport listening on port ${args.wsPort}`);
}
try {
const tools = gsd.createTools();
const runner = new InitRunner({
projectDir: args.projectDir,
tools,
eventStream: gsd.eventStream,
config: {
maxBudgetPerSession: args.maxBudget,
orchestratorModel: args.model,
},
});
const result = await runner.run(input);
// Print completion summary
const status = result.success ? 'SUCCESS' : 'FAILED';
const stepCount = result.steps.length;
const passedSteps = result.steps.filter(s => s.success).length;
const cost = result.totalCostUsd.toFixed(2);
const duration = (result.totalDurationMs / 1000).toFixed(1);
const artifactList = result.artifacts.join(', ');
console.log(`\n[${status}] ${passedSteps}/${stepCount} steps, $${cost}, ${duration}s`);
if (result.artifacts.length > 0) {
console.log(`Artifacts: ${artifactList}`);
}
if (!result.success) {
// Log failed steps
for (const step of result.steps) {
if (!step.success && step.error) {
console.error(` ✗ ${step.step}: ${step.error}`);
}
}
process.exitCode = 1;
}
} catch (err) {
console.error(`Fatal error: ${(err as Error).message}`);
process.exitCode = 1;
} finally {
cliTransport.close();
if (wsTransport) {
wsTransport.close();
}
}
return;
}
// ─── Auto command ─────────────────────────────────────────────────────────
if (args.command === 'auto') {
// #2832: refuse to silently route non-Claude runtime projects through the
// Claude Agent SDK. Load project config (best effort — falls back to
// defaults when missing) and gate before constructing GSD/InitRunner.
try {
const cfg = await loadConfig(args.projectDir, args.ws);
assertRuntimeSupportsAutoMode(cfg);
} catch (err) {
console.error(`Fatal error: ${(err as Error).message}`);
process.exitCode = 1;
return;
}
const gsd = new GSD({
projectDir: args.projectDir,
model: args.model,
maxBudgetUsd: args.maxBudget,
autoMode: true,
workstream: args.ws,
});
// Wire CLI transport (always active)
const cliTransport = new CLITransport();
gsd.addTransport(cliTransport);
// Optional WebSocket transport
let wsTransport: WSTransport | undefined;
if (args.wsPort !== undefined) {
wsTransport = new WSTransport({ port: args.wsPort });
await wsTransport.start();
gsd.addTransport(wsTransport);
console.log(`WebSocket transport listening on port ${args.wsPort}`);
}
try {
// If --init provided, bootstrap project first
if (args.init) {
const initInput = await resolveInitInput({
...args,
command: 'init',
initInput: args.init,
});
console.log(`[auto] Bootstrapping project from --init (${initInput.length} chars)`);
const tools = gsd.createTools();
const runner = new InitRunner({
projectDir: args.projectDir,
tools,
eventStream: gsd.eventStream,
config: {
maxBudgetPerSession: args.maxBudget,
orchestratorModel: args.model,
},
});
const initResult = await runner.run(initInput);
const initStatus = initResult.success ? 'SUCCESS' : 'FAILED';
const stepCount = initResult.steps.length;
const passedSteps = initResult.steps.filter(s => s.success).length;
const initCost = initResult.totalCostUsd.toFixed(2);
const initDuration = (initResult.totalDurationMs / 1000).toFixed(1);
console.log(`[init ${initStatus}] ${passedSteps}/${stepCount} steps, $${initCost}, ${initDuration}s`);
if (!initResult.success) {
for (const step of initResult.steps) {
if (!step.success && step.error) {
console.error(` ✗ ${step.step}: ${step.error}`);
}
}
process.exitCode = 1;
return;
}
}
const result = await gsd.run('');
// Final summary
const status = result.success ? 'SUCCESS' : 'FAILED';
const phases = result.phases.length;
const cost = result.totalCostUsd.toFixed(2);
const duration = (result.totalDurationMs / 1000).toFixed(1);
console.log(`\n[${status}] ${phases} phase(s), $${cost}, ${duration}s`);
if (!result.success) {
process.exitCode = 1;
}
} catch (err) {
console.error(`Fatal error: ${(err as Error).message}`);
process.exitCode = 1;
} finally {
cliTransport.close();
if (wsTransport) {
wsTransport.close();
}
}
return;
}
// ─── Run command ─────────────────────────────────────────────────────────
// Build GSD instance
const gsd = new GSD({
projectDir: args.projectDir,
model: args.model,
maxBudgetUsd: args.maxBudget,
workstream: args.ws,
});
// Wire CLI transport (always active)
const cliTransport = new CLITransport();
gsd.addTransport(cliTransport);
// Optional WebSocket transport
let wsTransport: WSTransport | undefined;
if (args.wsPort !== undefined) {
wsTransport = new WSTransport({ port: args.wsPort });
await wsTransport.start();
gsd.addTransport(wsTransport);
console.log(`WebSocket transport listening on port ${args.wsPort}`);
}
try {
const result = await gsd.run(args.prompt!);
// Final summary
const status = result.success ? 'SUCCESS' : 'FAILED';
const phases = result.phases.length;
const cost = result.totalCostUsd.toFixed(2);
const duration = (result.totalDurationMs / 1000).toFixed(1);
console.log(`\n[${status}] ${phases} phase(s), $${cost}, ${duration}s`);
if (!result.success) {
process.exitCode = 1;
}
} catch (err) {
console.error(`Fatal error: ${(err as Error).message}`);
process.exitCode = 1;
} finally {
// Clean up transports
cliTransport.close();
if (wsTransport) {
wsTransport.close();
}
}
}
// ─── Auto-run when invoked directly ──────────────────────────────────────────
main();

View File

@@ -1,277 +0,0 @@
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
import { loadConfig, CONFIG_DEFAULTS } from './config.js';
import { mkdir, writeFile, rm } from 'node:fs/promises';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
describe('loadConfig', () => {
let tmpDir: string;
let fakeHome: string;
let prevHome: string | undefined;
let prevGsdHome: string | undefined;
beforeEach(async () => {
tmpDir = join(tmpdir(), `gsd-config-test-${Date.now()}-${Math.random().toString(36).slice(2)}`);
await mkdir(join(tmpDir, '.planning'), { recursive: true });
// Isolate ~/.gsd/defaults.json by pointing HOME at an empty tmp dir.
fakeHome = join(tmpdir(), `gsd-home-test-${Date.now()}-${Math.random().toString(36).slice(2)}`);
await mkdir(fakeHome, { recursive: true });
prevHome = process.env.HOME;
process.env.HOME = fakeHome;
// Also isolate GSD_HOME (loadUserDefaults prefers it over HOME).
prevGsdHome = process.env.GSD_HOME;
delete process.env.GSD_HOME;
});
afterEach(async () => {
await rm(tmpDir, { recursive: true, force: true });
await rm(fakeHome, { recursive: true, force: true });
if (prevHome === undefined) delete process.env.HOME;
else process.env.HOME = prevHome;
if (prevGsdHome === undefined) delete process.env.GSD_HOME;
else process.env.GSD_HOME = prevGsdHome;
});
async function writeUserDefaults(defaults: unknown) {
await mkdir(join(fakeHome, '.gsd'), { recursive: true });
await writeFile(join(fakeHome, '.gsd', 'defaults.json'), JSON.stringify(defaults));
}
it('returns all defaults when config file is missing', async () => {
// No config.json created
await rm(join(tmpDir, '.planning', 'config.json'), { force: true });
const config = await loadConfig(tmpDir);
expect(config).toEqual(CONFIG_DEFAULTS);
});
it('returns all defaults when config file is empty', async () => {
await writeFile(join(tmpDir, '.planning', 'config.json'), '');
const config = await loadConfig(tmpDir);
expect(config).toEqual(CONFIG_DEFAULTS);
});
it('loads valid config and merges with defaults', async () => {
const userConfig = {
model_profile: 'fast',
workflow: { research: false },
};
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify(userConfig),
);
const config = await loadConfig(tmpDir);
expect(config.model_profile).toBe('fast');
expect(config.workflow.research).toBe(false);
// Other workflow defaults preserved
expect(config.workflow.plan_check).toBe(true);
expect(config.workflow.verifier).toBe(true);
// Top-level defaults preserved
expect(config.commit_docs).toBe(true);
expect(config.parallelization).toBe(true);
});
it('partial config merges correctly for nested objects', async () => {
const userConfig = {
git: { branching_strategy: 'milestone' },
hooks: { context_warnings: false },
};
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify(userConfig),
);
const config = await loadConfig(tmpDir);
expect(config.git.branching_strategy).toBe('milestone');
// Other git defaults preserved
expect(config.git.phase_branch_template).toBe('gsd/phase-{phase}-{slug}');
expect(config.hooks.context_warnings).toBe(false);
});
it('preserves unknown top-level keys', async () => {
const userConfig = { custom_key: 'custom_value' };
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify(userConfig),
);
const config = await loadConfig(tmpDir);
expect(config.custom_key).toBe('custom_value');
});
it('merges agent_skills', async () => {
const userConfig = {
agent_skills: { planner: 'custom-skill' },
};
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify(userConfig),
);
const config = await loadConfig(tmpDir);
expect(config.agent_skills).toEqual({ planner: 'custom-skill' });
});
// ─── Negative tests ─────────────────────────────────────────────────────
it('throws on malformed JSON', async () => {
await writeFile(
join(tmpDir, '.planning', 'config.json'),
'{bad json',
);
await expect(loadConfig(tmpDir)).rejects.toThrow(/Failed to parse config/);
});
it('throws when config is not an object (array)', async () => {
await writeFile(
join(tmpDir, '.planning', 'config.json'),
'[1, 2, 3]',
);
await expect(loadConfig(tmpDir)).rejects.toThrow(/must be a JSON object/);
});
it('throws when config is not an object (string)', async () => {
await writeFile(
join(tmpDir, '.planning', 'config.json'),
'"just a string"',
);
await expect(loadConfig(tmpDir)).rejects.toThrow(/must be a JSON object/);
});
it('ignores unknown keys without error', async () => {
const userConfig = {
totally_unknown: true,
another_unknown: { nested: 'value' },
};
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify(userConfig),
);
const config = await loadConfig(tmpDir);
// Should load fine, with unknowns passed through
expect(config.model_profile).toBe('balanced');
expect((config as Record<string, unknown>).totally_unknown).toBe(true);
});
it('handles wrong value types gracefully (user sets string instead of bool)', async () => {
const userConfig = {
commit_docs: 'yes', // should be boolean but we don't validate types
parallelization: 0,
};
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify(userConfig),
);
const config = await loadConfig(tmpDir);
// We pass through the user's values as-is — runtime code handles type mismatches
expect(config.commit_docs).toBe('yes');
expect(config.parallelization).toBe(0);
});
// ─── User-level defaults (~/.gsd/defaults.json) ─────────────────────────
// Regression: issue #2652 — SDK loadConfig ignored user-level defaults
// for pre-project Codex installs, so init.quick still emitted Claude
// model aliases from MODEL_PROFILES via resolveModel even when the user
// had `resolve_model_ids: "omit"` in ~/.gsd/defaults.json.
//
// Mirrors current CJS parity expectations for SDK loadConfig + resolveModel:
// in pre-project context, loadConfig ignores ~/.gsd/defaults.json so
// resolveModel/MODEL_PROFILES do not emit aliases when resolve_model_ids
// is "omit". Once a project is initialized, config.json is authoritative,
// because buildNewProjectConfig bakes user defaults into project config
// at /gsd-new-project time.
it('pre-project: ignores user defaults and uses built-in defaults', async () => {
await writeUserDefaults({ resolve_model_ids: 'omit' });
const config = await loadConfig(tmpDir);
// BEHAVIOR CHANGE (Cycle 3, #3536): CONFIG_DEFAULTS now sourced from
// sdk/shared/config-defaults.manifest.json which includes resolve_model_ids: false.
// The key is NOT undefined — it has the manifest default (false), not the user
// default ('omit'), confirming that user-level ~/.gsd/defaults.json is still ignored.
expect((config as Record<string, unknown>).resolve_model_ids).toBe(false);
expect(config.model_profile).toBe('balanced');
expect(config.workflow.plan_check).toBe(true);
});
it('pre-project: keeps built-in nested defaults even when user defaults exist', async () => {
await writeUserDefaults({
git: { branching_strategy: 'milestone' },
agent_skills: { planner: 'user-skill' },
});
const config = await loadConfig(tmpDir);
expect(config.git.branching_strategy).toBe('none');
expect(config.git.phase_branch_template).toBe('gsd/phase-{phase}-{slug}');
expect(config.agent_skills).toEqual({});
});
it('project config is authoritative over user defaults (CJS parity)', async () => {
// User defaults set resolve_model_ids: "omit", but project config omits it.
// Per CJS core.cjs loadConfig (#1683): once .planning/config.json exists,
// ~/.gsd/defaults.json is ignored — buildNewProjectConfig already baked
// the user defaults in at project creation time.
await writeUserDefaults({
resolve_model_ids: 'omit',
model_profile: 'fast',
});
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify({ model_profile: 'quality' }),
);
const config = await loadConfig(tmpDir);
expect(config.model_profile).toBe('quality');
// User-defaults not layered when project config present.
// BEHAVIOR CHANGE (Cycle 3, #3536): resolve_model_ids is now false (manifest default),
// not undefined — confirming user defaults are still ignored (value is NOT 'omit').
expect((config as Record<string, unknown>).resolve_model_ids).toBe(false);
});
it('ignores malformed ~/.gsd/defaults.json', async () => {
await mkdir(join(fakeHome, '.gsd'), { recursive: true });
await writeFile(join(fakeHome, '.gsd', 'defaults.json'), '{not json');
const config = await loadConfig(tmpDir);
// Falls back to built-in defaults
expect(config).toEqual(CONFIG_DEFAULTS);
});
it('maps legacy top-level branching_strategy into git.branching_strategy', async () => {
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify({ branching_strategy: 'phase' }),
);
const config = await loadConfig(tmpDir);
expect(config.git.branching_strategy).toBe('phase');
});
it('git.branching_strategy overrides legacy top-level branching_strategy when both are present', async () => {
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify({ branching_strategy: 'phase', git: { branching_strategy: 'milestone' } }),
);
const config = await loadConfig(tmpDir);
expect(config.git.branching_strategy).toBe('milestone');
});
it('does not mutate CONFIG_DEFAULTS between calls', async () => {
const before = structuredClone(CONFIG_DEFAULTS);
await writeFile(
join(tmpDir, '.planning', 'config.json'),
JSON.stringify({ model_profile: 'fast', workflow: { research: false } }),
);
await loadConfig(tmpDir);
expect(CONFIG_DEFAULTS).toEqual(before);
});
});

View File

@@ -1,210 +0,0 @@
/**
* Config reader — loads `.planning/config.json` and merges with defaults.
*
* Mirrors the default structure from `get-shit-done/bin/lib/config.cjs`
* `buildNewProjectConfig()`.
*/
import { readFile } from 'node:fs/promises';
import { join } from 'node:path';
import { relPlanningPath } from './workstream-utils.js';
import {
CONFIG_DEFAULTS as CANONICAL_CONFIG_DEFAULTS,
mergeDefaults as canonicalMergeDefaults,
normalizeLegacyKeys,
} from './config/index.js';
// ─── Types ───────────────────────────────────────────────────────────────────
export interface GitConfig {
branching_strategy: string;
phase_branch_template: string;
milestone_branch_template: string;
quick_branch_template: string | null;
}
export interface WorkflowConfig {
research: boolean;
plan_check: boolean;
verifier: boolean;
nyquist_validation: boolean;
/** Mirrors gsd-tools flat `config.tdd_mode` (from `workflow.tdd_mode`). */
tdd_mode: boolean;
/**
* Issue #3309. `end-of-phase` (default) suppresses mid-flight
* `<task type="checkpoint:human-verify">` task emission; the planner
* embeds verification details into the relevant `auto` task's
* `<verify><human-check>` block and the verifier harvests them at
* end-of-phase into the existing HUMAN-UAT.md path. `mid-flight`
* restores the pre-#3309 behavior where the executor halts at each
* `checkpoint:human-verify` task and pays a full executor cold-start
* cost (CLAUDE.md, MEMORY.md, STATE.md, plan re-read on respawn) per
* round-trip.
*/
human_verify_mode: 'mid-flight' | 'end-of-phase';
auto_advance: boolean;
/** Internal auto-chain flag used by workflow routing. */
_auto_chain_active?: boolean;
node_repair: boolean;
node_repair_budget: number;
ui_phase: boolean;
ui_safety_gate: boolean;
text_mode: boolean;
research_before_questions: boolean;
discuss_mode: string;
skip_discuss: boolean;
/** Maximum self-discuss passes in auto/headless mode before forcing proceed. Default: 3. */
max_discuss_passes: number;
/** Subagent timeout in ms (matches `get-shit-done/bin/lib/core.cjs` default 300000). */
subagent_timeout: number;
/**
* Issue #2492. When true (default), enforces that every trackable decision in
* CONTEXT.md `<decisions>` is referenced by at least one plan (translation
* gate, blocking) and reports decisions not honored by shipped artifacts at
* verify-phase (validation gate, non-blocking). Set false to disable both.
*/
context_coverage_gate: boolean;
/**
* Issue #105. When false, the primary checkout is shared or pinned (concurrent
* sessions / deliberate base-branch lock) and the commit handler must NOT
* auto-switch HEAD to the strategy branch. String value `"false"` is accepted
* for resilience against YAML/JSON parsers that leave boolean-like fields as
* strings.
*/
use_worktrees?: boolean | string;
}
export interface HooksConfig {
context_warnings: boolean;
}
export interface GSDConfig {
model_profile: string;
commit_docs: boolean;
parallelization: boolean;
search_gitignored: boolean;
brave_search: boolean;
firecrawl: boolean;
exa_search: boolean;
git: GitConfig;
workflow: WorkflowConfig;
hooks: HooksConfig;
agent_skills: Record<string, unknown>;
/** Project slug for branch templates; mirrors gsd-tools `config.project_code`. */
project_code?: string | null;
/** Interactive vs headless; mirrors gsd-tools flat `config.mode`. */
mode?: string;
[key: string]: unknown;
}
// ─── Defaults ────────────────────────────────────────────────────────────────
/**
* Canonical CONFIG_DEFAULTS delegated to the Configuration Module (ADR-3524).
* Cast to GSDConfig to preserve typed access for existing consumers.
* The canonical manifest may include additional keys beyond GSDConfig's
* declared fields (e.g. resolve_model_ids, context_window, planning.*,
* ship.*, workflow.security_*, workflow.code_review_*); these are accessible
* via the [key: string]: unknown index signature on GSDConfig.
*
* BEHAVIOR CHANGE (Cycle 3, #3536): CONFIG_DEFAULTS now includes all keys from
* sdk/shared/config-defaults.manifest.json. Keys added vs old inline literal:
* top-level: resolve_model_ids (false), context_window (200000),
* phase_naming ('sequential'), claude_md_path ('./CLAUDE.md')
* git: create_tag (true), base_branch (null)
* workflow: ai_integration_phase (true), code_review (true),
* code_review_depth ('standard'), code_review_command (null),
* pattern_mapper (true), plan_bounce (false), plan_bounce_script (null),
* plan_bounce_passes (2), auto_prune_state (false),
* post_planning_gaps (true), security_enforcement (true),
* security_asvs_level (1), security_block_on ('high'),
* context_coverage_gate: true (unchanged from old literal)
* planning: { commit_docs: true, search_gitignored: false, sub_repos: [], granularity: 'standard' }
* hooks: workflow_guard (false)
* ship: { pr_body_sections: [] }
*/
export const CONFIG_DEFAULTS: GSDConfig = CANONICAL_CONFIG_DEFAULTS as unknown as GSDConfig;
// ─── Loader ──────────────────────────────────────────────────────────────────
/**
* Load project config from `.planning/config.json`, merging with defaults.
* When project config is missing or empty, this returns `mergeDefaults({})`
* (built-in defaults only; no `~/.gsd/defaults.json` layering).
* Throws on malformed JSON with a helpful error message.
*/
export async function loadConfig(projectDir: string, workstream?: string): Promise<GSDConfig> {
const configPath = join(projectDir, relPlanningPath(workstream), 'config.json');
const rootConfigPath = join(projectDir, '.planning', 'config.json');
let raw: string;
let projectConfigFound = false;
try {
raw = await readFile(configPath, 'utf-8');
projectConfigFound = true;
} catch {
// If workstream config missing, fall back to root config
if (workstream) {
try {
raw = await readFile(rootConfigPath, 'utf-8');
projectConfigFound = true;
} catch {
raw = '';
}
} else {
raw = '';
}
}
// Pre-project context: no .planning/config.json exists.
// Use built-in defaults only so SDK query parity stays stable across machines.
if (!projectConfigFound) {
return mergeDefaults({});
}
const trimmed = raw.trim();
if (trimmed === '') {
// Empty project config — treat as no project config.
return mergeDefaults({});
}
let parsed: Record<string, unknown>;
try {
parsed = JSON.parse(trimmed);
} catch (err) {
const msg = err instanceof Error ? err.message : String(err);
throw new Error(`Failed to parse config at ${configPath}: ${msg}`);
}
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
throw new Error(`Config at ${configPath} must be a JSON object`);
}
// Project config exists — user-level defaults are ignored (CJS parity).
// `buildNewProjectConfig` already baked them into config.json at /gsd-new-project.
// Normalize legacy top-level keys (branching_strategy → git.branching_strategy, etc.)
// before merging with defaults, matching the Configuration Module's loadConfig pipeline.
const { parsed: normalized } = normalizeLegacyKeys(parsed);
return mergeDefaults(normalized);
}
/**
* Merge config with defaults using the Configuration Module's deep-merge.
* Delegates to canonicalMergeDefaults (ADR-3524, Cycle 3, #3536).
*
* BEHAVIOR CHANGE (Cycle 3, #3536): The old implementation used spread-per-section
* (shallow merge for git/workflow/hooks/agent_skills, spread for top-level).
* The new implementation uses recursive deep-merge via canonicalMergeDefaults,
* which means partial nested objects (e.g. { workflow: { research: false } })
* are now deep-merged rather than replacing the entire section's defaults.
* The practical difference: deep-merge preserves sibling default keys within
* nested sections even when the overlay only specifies one key — which was
* already the intended behavior of the old spread-per-section approach.
* Legacy branching_strategy top-level → git.branching_strategy normalization
* is now handled by normalizeLegacyKeys inside canonicalMergeDefaults's pipeline
* (via loadConfig); for the raw mergeDefaults path, legacy key handling is
* delegated to the canonical module.
*/
function mergeDefaults(parsed: Record<string, unknown>): GSDConfig {
return canonicalMergeDefaults(parsed) as unknown as GSDConfig;
}

View File

@@ -1,318 +0,0 @@
/**
* Pinning tests for the Configuration Module (ADR-3524 §6).
*
* These tests pin the public interface contract. They are RED until
* sdk/src/config/index.ts is created (Cycle 2).
*
* Test precedent: sdk/src/config.test.ts (vitest + fs fixtures).
*/
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
import { mkdirSync, mkdtempSync, writeFileSync, readFileSync, rmSync } from 'node:fs';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import {
loadConfig,
normalizeLegacyKeys,
mergeDefaults,
migrateOnDisk,
CONFIG_DEFAULTS,
} from './index.js';
// ─── Helpers ─────────────────────────────────────────────────────────────────
function makeTmpProject(): string {
const dir = mkdtempSync(join(tmpdir(), 'gsd-cfg-test-'));
mkdirSync(join(dir, '.planning'), { recursive: true });
return dir;
}
function writeConfig(dir: string, data: unknown): void {
writeFileSync(join(dir, '.planning', 'config.json'), JSON.stringify(data, null, 2));
}
function readConfigRaw(dir: string): string {
return readFileSync(join(dir, '.planning', 'config.json'), 'utf-8');
}
function cleanupDir(dir: string): void {
rmSync(dir, { recursive: true, force: true });
}
// ─── loadConfig ──────────────────────────────────────────────────────────────
describe('loadConfig', () => {
let tmpDir: string;
beforeEach(() => {
tmpDir = makeTmpProject();
});
afterEach(() => {
cleanupDir(tmpDir);
});
it('returns CONFIG_DEFAULTS when config.json is missing', async () => {
const config = await loadConfig(tmpDir);
expect(config).toEqual(CONFIG_DEFAULTS);
});
it('returns CONFIG_DEFAULTS when config.json is empty {}', async () => {
writeConfig(tmpDir, {});
const config = await loadConfig(tmpDir);
expect(config).toEqual(CONFIG_DEFAULTS);
});
it('returns nested git.branching_strategy when already nested', async () => {
writeConfig(tmpDir, { git: { branching_strategy: 'phase' } });
const config = await loadConfig(tmpDir);
expect(config.git.branching_strategy).toBe('phase');
});
it('normalizes legacy top-level branching_strategy to git.branching_strategy', async () => {
writeConfig(tmpDir, { branching_strategy: 'phase' });
const config = await loadConfig(tmpDir);
expect(config.git.branching_strategy).toBe('phase');
});
it('does NOT write disk when normalizing legacy branching_strategy', async () => {
writeConfig(tmpDir, { branching_strategy: 'phase' });
const before = readConfigRaw(tmpDir);
await loadConfig(tmpDir);
const after = readConfigRaw(tmpDir);
expect(after).toBe(before);
});
it('normalizes legacy top-level sub_repos to planning.sub_repos', async () => {
writeConfig(tmpDir, { sub_repos: ['app1', 'app2'] });
const config = await loadConfig(tmpDir);
expect((config.planning as Record<string, unknown>)?.sub_repos).toEqual(['app1', 'app2']);
});
it('handles legacy depth: comprehensive → planning.granularity: fine', async () => {
writeConfig(tmpDir, { depth: 'comprehensive' });
const config = await loadConfig(tmpDir);
// depth is a legacy key that maps to granularity
const granularity = (config as Record<string, unknown>).granularity
?? (config.planning as Record<string, unknown> | undefined)?.granularity;
expect(granularity).toBe('fine');
});
it('handles legacy depth: quick → coarse', async () => {
writeConfig(tmpDir, { depth: 'quick' });
const config = await loadConfig(tmpDir);
const granularity = (config as Record<string, unknown>).granularity
?? (config.planning as Record<string, unknown> | undefined)?.granularity;
expect(granularity).toBe('coarse');
});
it('handles legacy depth: standard → standard', async () => {
writeConfig(tmpDir, { depth: 'standard' });
const config = await loadConfig(tmpDir);
const granularity = (config as Record<string, unknown>).granularity
?? (config.planning as Record<string, unknown> | undefined)?.granularity;
expect(granularity).toBe('standard');
});
it('throws with informative error on malformed JSON', async () => {
writeFileSync(join(tmpDir, '.planning', 'config.json'), '{bad json');
await expect(loadConfig(tmpDir)).rejects.toThrow(/parse|invalid|json/i);
});
it('does not throw when .planning/config.json is missing — returns defaults', async () => {
const dir = mkdtempSync(join(tmpdir(), 'gsd-cfg-noplan-'));
// intentionally no .planning dir
try {
const config = await loadConfig(dir);
expect(config).toEqual(CONFIG_DEFAULTS);
} finally {
cleanupDir(dir);
}
});
});
// ─── normalizeLegacyKeys ─────────────────────────────────────────────────────
describe('normalizeLegacyKeys', () => {
it('migrates top-level branching_strategy to git.branching_strategy', () => {
const input = { branching_strategy: 'phase' };
const { parsed, normalizations } = normalizeLegacyKeys(input);
expect((parsed as Record<string, unknown>).branching_strategy).toBeUndefined();
expect((parsed as Record<string, Record<string, unknown>>).git?.branching_strategy).toBe('phase');
expect(normalizations).toHaveLength(1);
expect(normalizations[0]).toMatchObject({ from: 'branching_strategy', to: 'git.branching_strategy', value: 'phase' });
});
it('migrates top-level sub_repos to planning.sub_repos', () => {
const input = { sub_repos: ['app1', 'app2'] };
const { parsed, normalizations } = normalizeLegacyKeys(input);
expect((parsed as Record<string, unknown>).sub_repos).toBeUndefined();
expect((parsed as Record<string, Record<string, unknown>>).planning?.sub_repos).toEqual(['app1', 'app2']);
expect(normalizations).toHaveLength(1);
expect(normalizations[0]).toMatchObject({ from: 'sub_repos', to: 'planning.sub_repos', value: ['app1', 'app2'] });
});
it('migrates multiRepo: true to planning.sub_repos marker', () => {
const input = { multiRepo: true };
const { parsed, normalizations } = normalizeLegacyKeys(input);
expect((parsed as Record<string, unknown>).multiRepo).toBeUndefined();
expect(normalizations).toHaveLength(1);
expect(normalizations[0]).toMatchObject({ from: 'multiRepo', to: 'planning.sub_repos', requiresFilesystem: true });
});
it('migrates top-level depth to granularity (comprehensive → fine)', () => {
const input = { depth: 'comprehensive' };
const { parsed, normalizations } = normalizeLegacyKeys(input);
expect((parsed as Record<string, unknown>).depth).toBeUndefined();
expect(normalizations).toHaveLength(1);
expect(normalizations[0].from).toBe('depth');
// The value in parsed is the mapped granularity value
const gv = (parsed as Record<string, unknown>).granularity
?? (parsed as Record<string, Record<string, unknown>>).planning?.granularity;
expect(gv).toBe('fine');
});
it('migrates top-level depth: quick → coarse', () => {
const input = { depth: 'quick' };
const { parsed, normalizations } = normalizeLegacyKeys(input);
const gv = (parsed as Record<string, unknown>).granularity
?? (parsed as Record<string, Record<string, unknown>>).planning?.granularity;
expect(gv).toBe('coarse');
expect(normalizations[0].from).toBe('depth');
});
it('migrates all four legacy keys in one pass', () => {
const input = {
branching_strategy: 'phase',
sub_repos: ['app1'],
multiRepo: true,
depth: 'comprehensive',
};
const { parsed, normalizations } = normalizeLegacyKeys(input);
expect(normalizations).toHaveLength(4);
// branching_strategy → git.branching_strategy
expect((parsed as Record<string, Record<string, unknown>>).git?.branching_strategy).toBe('phase');
});
it('returns empty normalizations for already-normalized input', () => {
const input = { git: { branching_strategy: 'phase' }, planning: { sub_repos: ['app1'] } };
const { parsed, normalizations } = normalizeLegacyKeys(input);
expect(normalizations).toHaveLength(0);
expect(parsed).toEqual(input);
});
it('is idempotent — running twice produces same result with empty normalizations second time', () => {
const input = { branching_strategy: 'phase' };
const first = normalizeLegacyKeys(input);
const second = normalizeLegacyKeys(first.parsed as Record<string, unknown>);
expect(second.normalizations).toHaveLength(0);
expect(second.parsed).toEqual(first.parsed);
});
it('preserves canonical git.branching_strategy when both top-level and nested exist', () => {
const input = { branching_strategy: 'milestone', git: { branching_strategy: 'phase' } };
const { parsed } = normalizeLegacyKeys(input);
// canonical nested wins
expect((parsed as Record<string, Record<string, unknown>>).git?.branching_strategy).toBe('phase');
});
it('preserves canonical planning.sub_repos when both top-level and nested exist', () => {
const input = { sub_repos: ['legacy'], planning: { sub_repos: null } };
const { parsed } = normalizeLegacyKeys(input);
expect((parsed as Record<string, Record<string, unknown>>).sub_repos).toBeUndefined();
// canonical nested wins even when explicit null is used to unset
expect((parsed as Record<string, Record<string, unknown>>).planning?.sub_repos).toBeNull();
});
});
// ─── mergeDefaults ───────────────────────────────────────────────────────────
describe('mergeDefaults', () => {
it('returns full CONFIG_DEFAULTS for empty input', () => {
const result = mergeDefaults({});
expect(result).toEqual(CONFIG_DEFAULTS);
});
it('merges partial nested input without losing sibling keys', () => {
const partial = { git: { base_branch: 'main' } };
const result = mergeDefaults(partial);
// base_branch from input
expect((result.git as Record<string, unknown>).base_branch).toBe('main');
// sibling from defaults
expect(result.git.branching_strategy).toBe(CONFIG_DEFAULTS.git.branching_strategy);
expect(result.git.phase_branch_template).toBe(CONFIG_DEFAULTS.git.phase_branch_template);
});
it('preserves boolean false values (not overridden by truthy defaults)', () => {
// workflow.research defaults to true; setting false should survive merge
const partial = { workflow: { research: false } };
const result = mergeDefaults(partial);
expect(result.workflow.research).toBe(false);
});
it('preserves explicit null values', () => {
const partial = { project_code: null };
const result = mergeDefaults(partial);
expect(result.project_code).toBeNull();
});
it('user top-level keys win over defaults', () => {
const partial = { model_profile: 'quality' };
const result = mergeDefaults(partial);
expect(result.model_profile).toBe('quality');
});
});
// ─── migrateOnDisk ───────────────────────────────────────────────────────────
describe('migrateOnDisk', () => {
let tmpDir: string;
beforeEach(() => {
tmpDir = makeTmpProject();
});
afterEach(() => {
cleanupDir(tmpDir);
});
it('returns migrated:false, wrote:null for already-normalized config', async () => {
writeConfig(tmpDir, { git: { branching_strategy: 'phase' } });
const report = await migrateOnDisk(tmpDir);
expect(report.migrated).toBe(false);
expect(report.wrote).toBeNull();
expect(report.normalizations).toHaveLength(0);
});
it('returns migrated:true, writes disk when legacy key present', async () => {
writeConfig(tmpDir, { branching_strategy: 'phase' });
const report = await migrateOnDisk(tmpDir);
expect(report.migrated).toBe(true);
expect(report.wrote).not.toBeNull();
expect(report.normalizations.length).toBeGreaterThan(0);
// Verify disk was updated
const onDisk = JSON.parse(readConfigRaw(tmpDir));
expect(onDisk.branching_strategy).toBeUndefined();
expect(onDisk.git?.branching_strategy).toBe('phase');
});
it('returns report shape: { migrated, normalizations, wrote }', async () => {
writeConfig(tmpDir, { branching_strategy: 'milestone' });
const report = await migrateOnDisk(tmpDir);
expect(report).toHaveProperty('migrated');
expect(report).toHaveProperty('normalizations');
expect(report).toHaveProperty('wrote');
});
it('is a no-op when .planning/config.json is missing', async () => {
const dir = mkdtempSync(join(tmpdir(), 'gsd-cfg-nomig-'));
try {
const report = await migrateOnDisk(dir);
expect(report.migrated).toBe(false);
expect(report.wrote).toBeNull();
} finally {
cleanupDir(dir);
}
});
});

View File

@@ -1,325 +0,0 @@
/**
* Configuration Module — single source of truth for config loading,
* legacy-key normalization, defaults merge, and explicit on-disk migration.
*
* Source of truth for both the SDK and (via generator) the CJS side.
* Manifests are read from sdk/shared/*.manifest.json.
*
* Public API:
* loadConfig(cwd, options?) → MergedConfig — pure read, never writes disk
* normalizeLegacyKeys(parsed) → { parsed, normalizations[] } — pure transform
* mergeDefaults(parsed) → MergedConfig — fills in defaults
* migrateOnDisk(cwd) → MigrationReport — explicit, opt-in disk writeback
*/
import { readFileSync, writeFileSync, existsSync, readdirSync } from 'node:fs';
import { join } from 'node:path';
import { fileURLToPath } from 'node:url';
// ─── Manifest imports ─────────────────────────────────────────────────────────
const DEFAULTS_PATH = new URL('../../shared/config-defaults.manifest.json', import.meta.url);
export const CONFIG_DEFAULTS: Record<string, unknown> = JSON.parse(
readFileSync(fileURLToPath(DEFAULTS_PATH), 'utf-8'),
);
const SCHEMA_PATH = new URL('../../shared/config-schema.manifest.json', import.meta.url);
const _schemaManifest: {
validKeys: string[];
runtimeStateKeys: string[];
dynamicKeyPatterns: Array<{ topLevel: string; source: string; description: string }>;
} = JSON.parse(readFileSync(fileURLToPath(SCHEMA_PATH), 'utf-8'));
export const VALID_CONFIG_KEYS: ReadonlySet<string> = new Set(_schemaManifest.validKeys);
export const RUNTIME_STATE_KEYS: ReadonlySet<string> = new Set(_schemaManifest.runtimeStateKeys);
export interface DynamicKeyPattern {
readonly topLevel: string;
readonly source: string;
readonly description: string;
readonly test: (key: string) => boolean;
}
export const DYNAMIC_KEY_PATTERNS: readonly DynamicKeyPattern[] = _schemaManifest.dynamicKeyPatterns.map(
(p) => {
const pattern = new RegExp(p.source);
return {
...p,
test: (key: string) => {
pattern.lastIndex = 0;
return pattern.test(key);
},
};
},
);
// ─── Types ───────────────────────────────────────────────────────────────────
/** Broad merged config type — consumers narrow as needed. */
export type MergedConfig = Record<string, unknown>;
export interface Normalization {
from: string;
to: string;
value: unknown;
requiresFilesystem?: true;
}
export interface NormalizationResult {
parsed: MergedConfig;
normalizations: Normalization[];
}
export interface MigrationReport {
migrated: boolean;
normalizations: Normalization[];
wrote: string | null;
}
export interface LoadConfigOptions {
/** Optional workstream name — routes to .planning/workstreams/<name>/config.json */
workstream?: string;
/** Optional callback to observe normalizations applied during load */
onNormalizations?: (normalizations: Normalization[]) => void;
}
// ─── Depth → Granularity mapping ─────────────────────────────────────────────
const DEPTH_TO_GRANULARITY: Record<string, string> = {
quick: 'coarse',
standard: 'standard',
comprehensive: 'fine',
};
// ─── Internal helpers ─────────────────────────────────────────────────────────
function planningDir(cwd: string, workstream?: string): string {
if (!workstream) return join(cwd, '.planning');
return join(cwd, '.planning', 'workstreams', workstream);
}
function detectSubRepos(cwd: string): string[] {
const results: string[] = [];
try {
const entries = readdirSync(cwd, { withFileTypes: true });
for (const entry of entries) {
if (!entry.isDirectory()) continue;
if (entry.name.startsWith('.') || entry.name === 'node_modules') continue;
const gitPath = join(cwd, entry.name, '.git');
try {
if (existsSync(gitPath)) {
results.push(entry.name);
}
} catch { /* ignore */ }
}
} catch { /* ignore */ }
return results.sort();
}
/**
* Deep-merge two plain config objects. overlay wins on key conflict.
* Explicit null in overlay overrides base (null means "unset this key").
* Arrays are replaced, not merged. undefined in overlay falls back to base.
*/
function deepMergeConfig(base: Record<string, unknown>, overlay: Record<string, unknown>): Record<string, unknown> {
const result: Record<string, unknown> = { ...base };
for (const key of Object.keys(overlay)) {
const ov = overlay[key];
if (ov !== null && ov !== undefined && typeof ov === 'object' && !Array.isArray(ov)) {
const bv = base[key];
if (bv !== null && bv !== undefined && typeof bv === 'object' && !Array.isArray(bv)) {
result[key] = deepMergeConfig(bv as Record<string, unknown>, ov as Record<string, unknown>);
} else {
result[key] = deepMergeConfig({}, ov as Record<string, unknown>);
}
} else {
result[key] = ov;
}
}
return result;
}
// ─── normalizeLegacyKeys ─────────────────────────────────────────────────────
/**
* Pure transform: migrate legacy top-level config keys to their canonical nested locations.
* Returns the normalized parsed object + a list of normalizations applied.
* Idempotent: calling twice returns the same result with empty normalizations second time.
*
* Normalizations applied (in order):
* 1. top-level branching_strategy → git.branching_strategy (canonical wins if both present)
* 2. top-level sub_repos → planning.sub_repos (canonical wins if both present)
* 3. multiRepo: true → planning.sub_repos marker (requiresFilesystem: true)
* 4. top-level depth → granularity (top-level) with mapping quick→coarse/standard→standard/comprehensive→fine
*/
export function normalizeLegacyKeys(parsed: Record<string, unknown>): NormalizationResult {
const result: Record<string, unknown> = { ...parsed };
const normalizations: Normalization[] = [];
// 1. branching_strategy → git.branching_strategy
if (Object.prototype.hasOwnProperty.call(result, 'branching_strategy')) {
const value = result.branching_strategy;
const git = (result.git as Record<string, unknown> | undefined) ?? {};
if (git.branching_strategy === undefined) {
result.git = { ...git, branching_strategy: value };
} else {
// canonical nested wins — just delete the stale top-level
result.git = { ...git };
}
delete result.branching_strategy;
normalizations.push({ from: 'branching_strategy', to: 'git.branching_strategy', value });
}
// 2. top-level sub_repos → planning.sub_repos
if (Object.prototype.hasOwnProperty.call(result, 'sub_repos')) {
const value = result.sub_repos;
const planning = (result.planning as Record<string, unknown> | undefined) ?? {};
if (planning.sub_repos === undefined) {
result.planning = { ...planning, sub_repos: value };
} else {
// canonical nested wins — just drop the stale top-level
result.planning = { ...planning };
}
delete result.sub_repos;
normalizations.push({ from: 'sub_repos', to: 'planning.sub_repos', value });
}
// 3. multiRepo: true → marker (filesystem detection deferred to migrateOnDisk / caller)
if (result.multiRepo === true) {
delete result.multiRepo;
normalizations.push({ from: 'multiRepo', to: 'planning.sub_repos', value: true, requiresFilesystem: true });
}
// 4. top-level depth → granularity
if (Object.prototype.hasOwnProperty.call(result, 'depth') && !Object.prototype.hasOwnProperty.call(result, 'granularity')) {
const rawDepth = result.depth as string;
const mapped = DEPTH_TO_GRANULARITY[rawDepth] ?? rawDepth;
result.granularity = mapped;
delete result.depth;
normalizations.push({ from: 'depth', to: 'granularity', value: mapped });
}
return { parsed: result, normalizations };
}
// ─── mergeDefaults ───────────────────────────────────────────────────────────
/**
* Fill in CONFIG_DEFAULTS where the parsed object lacks values.
* Deep-merges per-section (git, workflow, hooks, agent_skills, planning, ship).
* Boolean false and explicit null are preserved — not overridden by truthy defaults.
*/
export function mergeDefaults(parsed: Record<string, unknown>): MergedConfig {
// Start with a deep clone of defaults, then overlay parsed
const defaults = JSON.parse(JSON.stringify(CONFIG_DEFAULTS)) as Record<string, unknown>;
return deepMergeConfig(defaults, parsed);
}
// ─── loadConfig ──────────────────────────────────────────────────────────────
/**
* Load project config from .planning/config.json (workstream-aware).
* Pure read — never writes disk.
*
* Pipeline: parse JSON → normalizeLegacyKeys → mergeDefaults → return.
*
* Missing file → returns CONFIG_DEFAULTS verbatim.
* Empty file → returns CONFIG_DEFAULTS verbatim.
* Malformed JSON → throws with informative error.
*/
export async function loadConfig(cwd: string, options?: LoadConfigOptions): Promise<MergedConfig> {
const configPath = join(planningDir(cwd, options?.workstream), 'config.json');
let raw: string;
try {
raw = readFileSync(configPath, 'utf-8');
} catch {
// File missing — return defaults
return mergeDefaults({});
}
const trimmed = raw.trim();
if (trimmed === '') {
return mergeDefaults({});
}
let parsed: Record<string, unknown>;
try {
parsed = JSON.parse(trimmed) as Record<string, unknown>;
} catch (err) {
const msg = err instanceof Error ? err.message : String(err);
throw new Error(`Failed to parse config at ${configPath}: ${msg}`);
}
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
throw new Error(`Config at ${configPath} must be a JSON object`);
}
const { parsed: normalized, normalizations } = normalizeLegacyKeys(parsed);
if (options?.onNormalizations && normalizations.length > 0) {
options.onNormalizations(normalizations);
}
return mergeDefaults(normalized);
}
// ─── migrateOnDisk ───────────────────────────────────────────────────────────
/**
* Explicit, opt-in disk writeback.
* Reads raw config, runs normalizeLegacyKeys, writes back only if normalizations are non-empty.
*
* For multiRepo: true entries, also runs filesystem detection to populate planning.sub_repos.
*
* Returns MigrationReport: { migrated, normalizations, wrote }.
*/
export async function migrateOnDisk(cwd: string, workstream?: string): Promise<MigrationReport> {
const configPath = join(planningDir(cwd, workstream), 'config.json');
let raw: string;
try {
raw = readFileSync(configPath, 'utf-8');
} catch {
// File missing — nothing to migrate
return { migrated: false, normalizations: [], wrote: null };
}
const trimmed = raw.trim();
if (trimmed === '') {
return { migrated: false, normalizations: [], wrote: null };
}
let parsed: Record<string, unknown>;
try {
parsed = JSON.parse(trimmed) as Record<string, unknown>;
} catch {
// Malformed — can't migrate
return { migrated: false, normalizations: [], wrote: null };
}
const { parsed: normalized, normalizations } = normalizeLegacyKeys(parsed);
if (normalizations.length === 0) {
return { migrated: false, normalizations: [], wrote: null };
}
// Resolve multiRepo filesystem detection
const result = { ...normalized };
for (const norm of normalizations) {
if (norm.requiresFilesystem) {
const detected = detectSubRepos(cwd);
if (detected.length > 0) {
const planning = (result.planning as Record<string, unknown> | undefined) ?? {};
result.planning = { ...planning, sub_repos: detected, commit_docs: false };
}
}
}
try {
writeFileSync(configPath, JSON.stringify(result, null, 2));
} catch (err) {
const msg = err instanceof Error ? err.message : String(err);
throw new Error(`Failed to write migrated config at ${configPath}: ${msg}`);
}
return { migrated: true, normalizations, wrote: configPath };
}

View File

@@ -1,295 +0,0 @@
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
import { mkdtemp, mkdir, writeFile, rm } from 'node:fs/promises';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import { ContextEngine, PHASE_FILE_MANIFEST } from './context-engine.js';
import { PhaseType } from './types.js';
import type { GSDLogger } from './logger.js';
// ─── Helpers ─────────────────────────────────────────────────────────────────
async function createTempProject(): Promise<string> {
return mkdtemp(join(tmpdir(), 'gsd-ctx-'));
}
async function createPlanningDir(projectDir: string, files: Record<string, string>): Promise<void> {
const planningDir = join(projectDir, '.planning');
await mkdir(planningDir, { recursive: true });
for (const [filename, content] of Object.entries(files)) {
await writeFile(join(planningDir, filename), content, 'utf-8');
}
}
function makeMockLogger(): GSDLogger {
return {
debug: vi.fn(),
info: vi.fn(),
warn: vi.fn(),
error: vi.fn(),
setPhase: vi.fn(),
setPlan: vi.fn(),
setSessionId: vi.fn(),
} as unknown as GSDLogger;
}
// ─── Tests ───────────────────────────────────────────────────────────────────
describe('ContextEngine', () => {
let projectDir: string;
beforeEach(async () => {
projectDir = await createTempProject();
});
afterEach(async () => {
await rm(projectDir, { recursive: true, force: true });
});
describe('resolveContextFiles', () => {
it('returns all files for plan phase when all exist', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '# State\nproject: test',
'ROADMAP.md': '# Roadmap\nphase 01',
'CONTEXT.md': '# Context\nstack: node',
'RESEARCH.md': '# Research\nfindings here',
'REQUIREMENTS.md': '# Requirements\nR1: auth',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Plan);
expect(files.state).toBe('# State\nproject: test');
expect(files.roadmap).toBe('# Roadmap\nphase 01');
expect(files.context).toBe('# Context\nstack: node');
expect(files.research).toBe('# Research\nfindings here');
expect(files.requirements).toBe('# Requirements\nR1: auth');
});
it('returns minimal files for execute phase', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '# State',
'config.json': '{"model":"claude"}',
'ROADMAP.md': '# Roadmap — should not be read',
'CONTEXT.md': '# Context — should not be read',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Execute);
expect(files.state).toBe('# State');
expect(files.config).toBe('{"model":"claude"}');
expect(files.roadmap).toBeUndefined();
expect(files.context).toBeUndefined();
});
it('returns state + roadmap + context for research phase', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '# State',
'ROADMAP.md': '# Roadmap',
'CONTEXT.md': '# Context',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Research);
expect(files.state).toBe('# State');
expect(files.roadmap).toBe('# Roadmap');
expect(files.context).toBe('# Context');
expect(files.requirements).toBeUndefined();
});
it('returns state + roadmap + requirements for verify phase', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '# State',
'ROADMAP.md': '# Roadmap',
'REQUIREMENTS.md': '# Requirements',
'PLAN.md': '# Plan',
'SUMMARY.md': '# Summary',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Verify);
expect(files.state).toBe('# State');
expect(files.roadmap).toBe('# Roadmap');
expect(files.requirements).toBe('# Requirements');
expect(files.plan).toBe('# Plan');
expect(files.summary).toBe('# Summary');
});
it('returns state + optional files for discuss phase', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '# State',
'ROADMAP.md': '# Roadmap',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Discuss);
expect(files.state).toBe('# State');
expect(files.roadmap).toBe('# Roadmap');
expect(files.context).toBeUndefined();
});
it('returns undefined for missing optional files without warning', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '# State',
'ROADMAP.md': '# Roadmap',
'CONTEXT.md': '# Context',
});
const logger = makeMockLogger();
const engine = new ContextEngine(projectDir, logger);
const files = await engine.resolveContextFiles(PhaseType.Plan);
// research and requirements are optional for plan — no warning
expect(files.research).toBeUndefined();
expect(files.requirements).toBeUndefined();
expect(logger.warn).not.toHaveBeenCalled();
});
it('warns for missing required files', async () => {
// Empty .planning dir — STATE.md is required for all phases
await createPlanningDir(projectDir, {});
const logger = makeMockLogger();
const engine = new ContextEngine(projectDir, logger);
await engine.resolveContextFiles(PhaseType.Execute);
expect(logger.warn).toHaveBeenCalledWith(
expect.stringContaining('STATE.md'),
expect.objectContaining({ phase: PhaseType.Execute }),
);
});
it('handles missing .planning directory gracefully', async () => {
// No .planning dir at all
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Execute);
expect(files.state).toBeUndefined();
expect(files.config).toBeUndefined();
});
it('handles empty file content', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Execute);
// Empty string is still defined — the file exists
expect(files.state).toBe('');
});
});
describe('context truncation', () => {
it('truncates files exceeding maxContentLength', async () => {
const largeContent = Array.from({ length: 100 }, (_, i) =>
`## Section ${i}\n\nFirst paragraph.\n\nLong detail ${'x'.repeat(200)}.`
).join('\n\n');
await createPlanningDir(projectDir, {
'STATE.md': '# State',
'ROADMAP.md': '# Roadmap',
'CONTEXT.md': largeContent,
});
const engine = new ContextEngine(projectDir, undefined, { maxContentLength: 500 });
const files = await engine.resolveContextFiles(PhaseType.Plan);
// CONTEXT.md should be truncated
expect(files.context!.length).toBeLessThan(largeContent.length);
expect(files.context).toContain('[...');
});
it('does not truncate files below threshold', async () => {
await createPlanningDir(projectDir, {
'STATE.md': '# State\nproject: test',
'ROADMAP.md': '# Roadmap\nphase 01',
'CONTEXT.md': '# Context\nstack: node',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Plan);
expect(files.context).toBe('# Context\nstack: node');
});
it('never truncates STATE.md (not in truncatable list)', async () => {
const largeState = `# State\n\n${'x'.repeat(20000)}`;
await createPlanningDir(projectDir, {
'STATE.md': largeState,
});
const engine = new ContextEngine(projectDir, undefined, { maxContentLength: 100 });
const files = await engine.resolveContextFiles(PhaseType.Execute);
expect(files.state).toBe(largeState);
});
it('extracts current milestone from ROADMAP.md when state is available', async () => {
const roadmap = `# Roadmap
## Milestone 1: Setup
### Phase 01
Setup content.
## Milestone 2: Build
### Phase 02
Build content.`;
await createPlanningDir(projectDir, {
'STATE.md': 'Current Milestone: Build',
'ROADMAP.md': roadmap,
'CONTEXT.md': '# Context',
});
const engine = new ContextEngine(projectDir);
const files = await engine.resolveContextFiles(PhaseType.Plan);
expect(files.roadmap).toContain('## Milestone 2: Build');
expect(files.roadmap).not.toContain('### Phase 01');
});
it('respects custom truncation options', async () => {
const content = '## Heading\n\nParagraph.\n\nMore.\n' + 'x'.repeat(500);
await createPlanningDir(projectDir, {
'STATE.md': '# State',
'ROADMAP.md': '# Roadmap',
'CONTEXT.md': content,
});
// Low threshold forces truncation
const engine = new ContextEngine(projectDir, undefined, { maxContentLength: 50 });
const files = await engine.resolveContextFiles(PhaseType.Plan);
expect(files.context!.length).toBeLessThan(content.length);
});
});
describe('PHASE_FILE_MANIFEST', () => {
it('covers all phase types', () => {
for (const phase of Object.values(PhaseType)) {
expect(PHASE_FILE_MANIFEST[phase]).toBeDefined();
expect(PHASE_FILE_MANIFEST[phase].length).toBeGreaterThan(0);
}
});
it('execute phase has fewest files', () => {
const executeCount = PHASE_FILE_MANIFEST[PhaseType.Execute].length;
const planCount = PHASE_FILE_MANIFEST[PhaseType.Plan].length;
expect(executeCount).toBeLessThan(planCount);
});
it('every spec has required key, filename, and required flag', () => {
for (const specs of Object.values(PHASE_FILE_MANIFEST)) {
for (const spec of specs) {
expect(spec.key).toBeDefined();
expect(spec.filename).toBeDefined();
expect(typeof spec.required).toBe('boolean');
}
}
});
});
});

View File

@@ -1,170 +0,0 @@
/**
* Context engine — resolves which .planning/ state files exist per phase type.
*
* Different phases need different subsets of context files. The execute phase
* only needs STATE.md + config.json (minimal). Research needs STATE.md +
* ROADMAP.md + CONTEXT.md. Plan needs all files. Verify needs STATE.md +
* ROADMAP.md + REQUIREMENTS.md + PLAN/SUMMARY files.
*
* Context reduction (issue #1614):
* - Large files are truncated to keep prompts cache-friendly
* - ROADMAP.md is narrowed to the current milestone when possible
* - Truncation preserves headings + first paragraph per section
*/
import { readFile, access } from 'node:fs/promises';
import { join } from 'node:path';
import { constants } from 'node:fs';
import type { ContextFiles } from './types.js';
import { PhaseType } from './types.js';
import type { GSDLogger } from './logger.js';
import {
truncateMarkdown,
extractCurrentMilestone,
DEFAULT_TRUNCATION_OPTIONS,
type TruncationOptions,
} from './context-truncation.js';
import { relPlanningPath } from './workstream-utils.js';
// ─── File manifest per phase ─────────────────────────────────────────────────
interface FileSpec {
key: keyof ContextFiles;
filename: string;
required: boolean;
}
/**
* Define which files each phase needs. Required files emit warnings when missing;
* optional files silently return undefined.
*/
const PHASE_FILE_MANIFEST: Record<PhaseType, FileSpec[]> = {
[PhaseType.Execute]: [
{ key: 'state', filename: 'STATE.md', required: true },
{ key: 'config', filename: 'config.json', required: false },
],
[PhaseType.Research]: [
{ key: 'state', filename: 'STATE.md', required: true },
{ key: 'roadmap', filename: 'ROADMAP.md', required: true },
{ key: 'context', filename: 'CONTEXT.md', required: true },
{ key: 'requirements', filename: 'REQUIREMENTS.md', required: false },
],
[PhaseType.Plan]: [
{ key: 'state', filename: 'STATE.md', required: true },
{ key: 'roadmap', filename: 'ROADMAP.md', required: true },
{ key: 'context', filename: 'CONTEXT.md', required: true },
{ key: 'research', filename: 'RESEARCH.md', required: false },
{ key: 'requirements', filename: 'REQUIREMENTS.md', required: false },
],
[PhaseType.Verify]: [
{ key: 'state', filename: 'STATE.md', required: true },
{ key: 'roadmap', filename: 'ROADMAP.md', required: true },
{ key: 'requirements', filename: 'REQUIREMENTS.md', required: false },
{ key: 'plan', filename: 'PLAN.md', required: false },
{ key: 'summary', filename: 'SUMMARY.md', required: false },
],
[PhaseType.Repair]: [
{ key: 'state', filename: 'STATE.md', required: true },
{ key: 'config', filename: 'config.json', required: false },
{ key: 'plan', filename: 'PLAN.md', required: false },
],
[PhaseType.Discuss]: [
{ key: 'state', filename: 'STATE.md', required: true },
{ key: 'roadmap', filename: 'ROADMAP.md', required: false },
{ key: 'context', filename: 'CONTEXT.md', required: false },
],
};
// ─── ContextEngine class ─────────────────────────────────────────────────────
export class ContextEngine {
private readonly planningDir: string;
private readonly logger?: GSDLogger;
private readonly truncation: TruncationOptions;
constructor(projectDir: string, logger?: GSDLogger, truncation?: Partial<TruncationOptions>, workstream?: string) {
this.planningDir = join(projectDir, relPlanningPath(workstream));
this.logger = logger;
this.truncation = { ...DEFAULT_TRUNCATION_OPTIONS, ...truncation };
}
/**
* Resolve context files appropriate for the given phase type.
* Reads each file defined in the phase manifest, returning undefined
* for missing optional files and warning for missing required files.
*
* Files exceeding the truncation threshold are reduced to headings +
* first paragraphs. ROADMAP.md is narrowed to the current milestone.
*/
async resolveContextFiles(phaseType: PhaseType): Promise<ContextFiles> {
const manifest = PHASE_FILE_MANIFEST[phaseType];
const result: ContextFiles = {};
for (const spec of manifest) {
const filePath = join(this.planningDir, spec.filename);
const content = await this.readFileIfExists(filePath);
if (content !== undefined) {
result[spec.key] = content;
} else if (spec.required) {
this.logger?.warn(`Required context file missing for ${phaseType} phase: ${spec.filename}`, {
phase: phaseType,
file: spec.filename,
path: filePath,
});
}
}
// Apply context reduction: milestone extraction then truncation
if (result.roadmap && result.state) {
const before = result.roadmap.length;
result.roadmap = extractCurrentMilestone(result.roadmap, result.state);
if (result.roadmap.length < before) {
this.logger?.debug?.('ROADMAP.md narrowed to current milestone', {
before,
after: result.roadmap.length,
});
}
}
// Truncate oversized files (skip config.json — structured data, not markdown)
const truncatable: Array<{ key: keyof ContextFiles; filename: string }> = [
{ key: 'roadmap', filename: 'ROADMAP.md' },
{ key: 'context', filename: 'CONTEXT.md' },
{ key: 'research', filename: 'RESEARCH.md' },
{ key: 'requirements', filename: 'REQUIREMENTS.md' },
{ key: 'plan', filename: 'PLAN.md' },
{ key: 'summary', filename: 'SUMMARY.md' },
];
for (const { key, filename } of truncatable) {
const raw = result[key];
if (raw && raw.length > this.truncation.maxContentLength) {
const before = raw.length;
result[key] = truncateMarkdown(raw, filename, this.truncation);
this.logger?.debug?.(`${filename} truncated`, {
before,
after: result[key]!.length,
});
}
}
return result;
}
/**
* Check if a file exists and read it. Returns undefined if not found.
*/
private async readFileIfExists(filePath: string): Promise<string | undefined> {
try {
await access(filePath, constants.R_OK);
return await readFile(filePath, 'utf-8');
} catch {
return undefined;
}
}
}
export { PHASE_FILE_MANIFEST };
export type { FileSpec };

View File

@@ -1,163 +0,0 @@
import { describe, it, expect } from 'vitest';
import {
truncateMarkdown,
extractCurrentMilestone,
DEFAULT_TRUNCATION_OPTIONS,
} from './context-truncation.js';
// ─── truncateMarkdown ───────────────────────────────────────────────────────
describe('truncateMarkdown', () => {
it('returns content unchanged when below threshold', () => {
const content = '# Title\n\nShort content.';
const result = truncateMarkdown(content, 'TEST.md');
expect(result).toBe(content);
});
it('truncates content above threshold, keeping headings and first paragraphs', () => {
const sections = [];
for (let i = 0; i < 20; i++) {
sections.push(`## Section ${i}\n\nFirst paragraph of section ${i}.\n\nSecond paragraph with lots of detail.\nMore detail here.\nEven more detail.`);
}
const content = `# Title\n\n${sections.join('\n\n')}`;
const result = truncateMarkdown(content, 'BIG.md', { maxContentLength: 100 });
// Headings preserved
expect(result).toContain('# Title');
expect(result).toContain('## Section 0');
expect(result).toContain('## Section 19');
// First paragraphs preserved
expect(result).toContain('First paragraph of section 0.');
expect(result).toContain('First paragraph of section 19.');
// Second paragraphs omitted
expect(result).not.toContain('Second paragraph');
expect(result).not.toContain('More detail here.');
// Truncation markers present
expect(result).toContain('[...');
expect(result).toContain('lines omitted]');
expect(result).toContain('[Truncated: read .planning/BIG.md for full content]');
});
it('preserves YAML frontmatter entirely', () => {
const content = `---\nphase: "01"\nstatus: active\n---\n\n# Title\n\nParagraph 1.\n\nParagraph 2.\n${'x'.repeat(10000)}`;
const result = truncateMarkdown(content, 'STATE.md', { maxContentLength: 100 });
expect(result).toContain('---\nphase: "01"\nstatus: active\n---');
expect(result).toContain('# Title');
expect(result).toContain('Paragraph 1.');
});
it('is smaller than original when truncated', () => {
const longContent = Array.from({ length: 200 }, (_, i) =>
`## Section ${i}\n\nFirst paragraph.\n\nLong detail paragraph ${'x'.repeat(100)}.`
).join('\n\n');
const result = truncateMarkdown(longContent, 'HUGE.md', { maxContentLength: 100 });
expect(result.length).toBeLessThan(longContent.length);
});
it('handles content with no headings', () => {
const content = `First line.\n\nSecond paragraph.\n\nThird paragraph.\n${'x'.repeat(10000)}`;
const result = truncateMarkdown(content, 'FLAT.md', { maxContentLength: 100 });
// Should still truncate — first paragraph kept
expect(result).toContain('First line.');
expect(result.length).toBeLessThan(content.length);
});
it('default threshold is 8192 characters', () => {
expect(DEFAULT_TRUNCATION_OPTIONS.maxContentLength).toBe(8192);
});
});
// ─── extractCurrentMilestone ────────────────────────────────────────────────
describe('extractCurrentMilestone', () => {
const makeRoadmap = () => `# Project Roadmap
## Milestone 1: Foundation
### Phase 01: Setup
Requirements for setup.
### Phase 02: Core
Requirements for core.
## Milestone 2: Features
### Phase 03: Auth
Requirements for auth.
### Phase 04: API
Requirements for API.
## Milestone 3: Polish
### Phase 05: UI
Requirements for UI.`;
it('returns full roadmap when no state provided', () => {
const roadmap = makeRoadmap();
expect(extractCurrentMilestone(roadmap)).toBe(roadmap);
});
it('returns full roadmap when milestone not found in state', () => {
const roadmap = makeRoadmap();
const state = '# State\nstatus: active';
expect(extractCurrentMilestone(roadmap, state)).toBe(roadmap);
});
it('extracts current milestone section by name', () => {
const roadmap = makeRoadmap();
const state = 'Current Milestone: Features';
const result = extractCurrentMilestone(roadmap, state);
expect(result).toContain('## Milestone 2: Features');
expect(result).toContain('### Phase 03: Auth');
expect(result).toContain('### Phase 04: API');
// Other milestones omitted
expect(result).not.toContain('### Phase 01: Setup');
expect(result).not.toContain('### Phase 05: UI');
expect(result).toContain('other milestone(s) omitted');
});
it('matches milestone name case-insensitively', () => {
const roadmap = makeRoadmap();
const state = 'current milestone: features';
const result = extractCurrentMilestone(roadmap, state);
expect(result).toContain('## Milestone 2: Features');
expect(result).not.toContain('### Phase 01: Setup');
});
it('matches milestone from "milestone:" field in state', () => {
const roadmap = makeRoadmap();
const state = '# State\nmilestone: Foundation\nphase: 01';
const result = extractCurrentMilestone(roadmap, state);
expect(result).toContain('## Milestone 1: Foundation');
expect(result).toContain('### Phase 01: Setup');
expect(result).not.toContain('### Phase 03: Auth');
});
it('matches milestone from Current Position block', () => {
const roadmap = makeRoadmap();
const state = `# State
## Current Position
milestone: Polish
phase: 05`;
const result = extractCurrentMilestone(roadmap, state);
expect(result).toContain('## Milestone 3: Polish');
expect(result).toContain('### Phase 05: UI');
expect(result).not.toContain('### Phase 01: Setup');
});
it('preserves roadmap title in output', () => {
const roadmap = makeRoadmap();
const state = 'Current Milestone: Features';
const result = extractCurrentMilestone(roadmap, state);
expect(result).toContain('# Project Roadmap');
});
});

View File

@@ -1,233 +0,0 @@
/**
* Context truncation — reduces large .planning/ files to cache-friendly sizes.
*
* Two strategies:
* 1. Markdown-aware truncation: keeps headings + first paragraph per section,
* replaces the rest with a pointer to the full file.
* 2. Milestone extraction: pulls only the current milestone from ROADMAP.md.
*
* All functions are pure — no I/O, no side effects.
*/
// ─── Types ──────────────────────────────────────────────────────────────────
export interface TruncationOptions {
/** Max content length in characters before truncation kicks in. Default: 8192 */
maxContentLength: number;
}
export const DEFAULT_TRUNCATION_OPTIONS: TruncationOptions = {
maxContentLength: 8192,
};
// ─── Markdown-aware truncation ──────────────────────────────────────────────
/**
* Truncate markdown content while preserving structure.
*
* Strategy: keep YAML frontmatter, all headings, and the first paragraph under
* each heading. Collapse everything else with a line count summary.
*
* Returns the original content unchanged if below maxContentLength.
*/
export function truncateMarkdown(
content: string,
filename: string,
options: TruncationOptions = DEFAULT_TRUNCATION_OPTIONS,
): string {
if (content.length <= options.maxContentLength) return content;
const lines = content.split('\n');
const kept: string[] = [];
let inFrontmatter = false;
let frontmatterDone = false;
let currentSectionLines = 0;
let paragraphKept = false;
let omittedLines = 0;
let inParagraph = false;
for (let i = 0; i < lines.length; i++) {
const line = lines[i];
// Handle YAML frontmatter (preserve entirely)
if (i === 0 && line.trim() === '---') {
inFrontmatter = true;
kept.push(line);
continue;
}
if (inFrontmatter) {
kept.push(line);
if (line.trim() === '---') {
inFrontmatter = false;
frontmatterDone = true;
}
continue;
}
// Heading — always keep, reset paragraph tracking
if (/^#{1,6}\s/.test(line)) {
if (omittedLines > 0) {
kept.push(`[... ${omittedLines} lines omitted]`);
omittedLines = 0;
}
kept.push(line);
currentSectionLines = 0;
paragraphKept = false;
inParagraph = false;
continue;
}
// Empty line — paragraph boundary
if (line.trim() === '') {
if (inParagraph && !paragraphKept) {
// End of first paragraph — mark it kept
paragraphKept = true;
}
if (!paragraphKept || currentSectionLines === 0) {
kept.push(line);
} else {
omittedLines++;
}
inParagraph = false;
continue;
}
// Content line
currentSectionLines++;
if (!paragraphKept) {
// Still in the first paragraph — keep it
kept.push(line);
inParagraph = true;
} else {
omittedLines++;
}
}
if (omittedLines > 0) {
kept.push(`[... ${omittedLines} lines omitted]`);
}
const totalOmitted = lines.length - kept.length;
if (totalOmitted > 0) {
kept.push('');
kept.push(`[Truncated: read .planning/${filename} for full content]`);
}
return kept.join('\n');
}
// ─── Milestone extraction ───────────────────────────────────────────────────
/**
* Extract the current milestone section from a ROADMAP.md.
*
* Parses STATE.md to find the current milestone name, then extracts only
* that milestone's section from the roadmap. Falls back to full content
* if the milestone can't be identified or found.
*/
export function extractCurrentMilestone(
roadmapContent: string,
stateContent?: string,
): string {
if (!stateContent) return roadmapContent;
// Find current milestone from STATE.md
// Patterns: "Current Milestone: X", "milestone: X", "## Current Position" block
const milestonePatterns = [
/current\s*milestone\s*:\s*(.+)/i,
/^milestone\s*:\s*(.+)/im,
/##\s*current\s*position[\s\S]*?milestone\s*:\s*(.+)/i,
];
let milestoneName: string | undefined;
for (const pattern of milestonePatterns) {
const match = stateContent.match(pattern);
if (match) {
milestoneName = match[1].trim();
break;
}
}
if (!milestoneName) return roadmapContent;
// Find the milestone section in roadmap
// Look for heading containing the milestone name
const lines = roadmapContent.split('\n');
let sectionStart = -1;
let sectionEnd = lines.length;
let sectionHeadingLevel = 0;
for (let i = 0; i < lines.length; i++) {
const headingMatch = lines[i].match(/^(#{1,6})\s+(.+)/);
if (!headingMatch) continue;
const level = headingMatch[1].length;
const title = headingMatch[2];
if (sectionStart === -1) {
// Looking for the milestone heading
if (title.toLowerCase().includes(milestoneName.toLowerCase())) {
sectionStart = i;
sectionHeadingLevel = level;
}
} else {
// Found start — look for next heading at same or higher level
if (level <= sectionHeadingLevel) {
sectionEnd = i;
break;
}
}
}
if (sectionStart === -1) return roadmapContent;
// Extract preamble (everything before first milestone heading at the same level)
const preamble: string[] = [];
for (let i = 0; i < lines.length; i++) {
const headingMatch = lines[i].match(/^(#{1,6})\s/);
if (headingMatch && headingMatch[1].length === sectionHeadingLevel && i !== sectionStart) {
// Hit another milestone-level heading before our section
if (i < sectionStart) {
break; // preamble ends at first milestone heading
}
}
if (i < sectionStart) {
// Keep top-level title and intro
if (i === 0 || lines[i].match(/^#\s/) || !lines[i].match(/^#{1,6}\s/)) {
preamble.push(lines[i]);
}
}
}
const milestoneSection = lines.slice(sectionStart, sectionEnd).join('\n');
const otherMilestones = countOtherMilestones(lines, sectionHeadingLevel, sectionStart);
const result = [
...preamble,
'',
milestoneSection,
];
if (otherMilestones > 0) {
result.push('');
result.push(`[${otherMilestones} other milestone(s) omitted — read .planning/ROADMAP.md for full roadmap]`);
}
return result.join('\n').trim();
}
function countOtherMilestones(
lines: string[],
headingLevel: number,
excludeIndex: number,
): number {
let count = 0;
for (let i = 0; i < lines.length; i++) {
if (i === excludeIndex) continue;
const match = lines[i].match(/^(#{1,6})\s/);
if (match && match[1].length === headingLevel) {
count++;
}
}
return count;
}

View File

@@ -1,246 +0,0 @@
import type { QueryRegistry } from '../query/registry.js';
import { extractField } from '../query/registry.js';
import { normalizeQueryCommand } from '../query/query-command-resolution-strategy.js';
import { runCjsFallbackDispatch } from '../query/query-fallback-executor.js';
import type { QueryDispatchError, QueryDispatchResult } from '../query/query-dispatch-contract.js';
import type { QueryResult } from '../query/utils.js';
import type { QueryNativeDispatchAdapter } from '../query/query-native-dispatch-adapter.js';
import type { CommandTopology, CommandTopologyMatch } from '../query/command-topology.js';
import { unknownCommandError, validationError, fallbackDispatchErrorFromSignal, nativeDispatchErrorFromSignal } from '../query/query-error-taxonomy.js';
import { canUseCjsFallback } from '../query/query-fallback-policy.js';
import { toFailureSignal } from '../query-failure-classification.js';
export interface QueryDispatchDeps {
registry: QueryRegistry;
projectDir: string;
ws?: string;
cjsFallbackEnabled: boolean;
resolveGsdToolsPath: (projectDir: string) => string;
/** @deprecated use topology */
dispatchNative?: (cmd: string, args: string[]) => Promise<QueryResult>;
/** @deprecated use topology */
nativeAdapter?: QueryNativeDispatchAdapter;
topology: CommandTopology;
}
export type DispatchMode = 'native' | 'cjs' | 'error';
export interface DispatchPlan {
mode: DispatchMode;
normalized: { command: string; args: string[]; tokens: string[] };
matched: CommandTopologyMatch | null;
noMatchMessage?: string;
noMatchNormalized?: string;
noMatchAttempted?: string[];
noMatchHints?: string[];
}
export type DispatchSuccessFormat = 'json' | 'text' | undefined;
export interface DispatchInputValidationResult {
queryArgs: string[];
pickField?: string;
error?: QueryDispatchResult;
}
export function dispatchFailure(error: QueryDispatchError, stderr: string[] = []): QueryDispatchResult {
return {
ok: false,
error,
stderr,
exit_code: error.code,
};
}
export function dispatchSuccess(stdout: string, stderr: string[] = []): QueryDispatchResult {
return {
ok: true,
stdout,
stderr,
exit_code: 0,
};
}
export function toDispatchFailure(error: QueryDispatchError, stderr: string[] = []): QueryDispatchResult {
return dispatchFailure(error, stderr);
}
export function mapNativeDispatchError(error: unknown, command: string, args: string[]): QueryDispatchError {
return nativeDispatchErrorFromSignal(toFailureSignal(error), command, args);
}
export function mapFallbackDispatchError(error: unknown, command: string, args: string[]): QueryDispatchError {
return fallbackDispatchErrorFromSignal(toFailureSignal(error), command, args);
}
export function formatPick(data: unknown, pickField?: string): unknown {
if (!pickField) return data;
return extractField(data, pickField);
}
export function formatSuccess(data: unknown, format: DispatchSuccessFormat, pickField?: string): string {
if (format === 'text' && typeof data === 'string') {
if (pickField) {
throw new Error('--pick is not supported for text output');
}
return data.endsWith('\n') ? data : `${data}\n`;
}
const output = formatPick(data, pickField);
if (pickField && typeof output === 'string') {
return output.endsWith('\n') ? output : `${output}\n`;
}
return `${JSON.stringify(output === undefined ? null : output, null, 2)}\n`;
}
export function validateQueryDispatchInput(queryArgv: string[]): DispatchInputValidationResult {
const queryArgs = [...queryArgv];
const pickIdx = queryArgs.indexOf('--pick');
if (pickIdx !== -1) {
if (pickIdx + 1 >= queryArgs.length) {
return {
queryArgs,
error: dispatchFailure(validationError({
message: 'Error: --pick requires a field name',
details: { field: '--pick', reason: 'missing_value' },
})),
};
}
const pickField = queryArgs[pickIdx + 1];
queryArgs.splice(pickIdx, 2);
if (queryArgs.length === 0 || !queryArgs[0]) {
return {
queryArgs,
error: dispatchFailure(validationError({
message: 'Error: "gsd-sdk query" requires a command',
details: { reason: 'missing_command' },
})),
};
}
return { queryArgs, pickField };
}
if (queryArgs.length === 0 || !queryArgs[0]) {
return {
queryArgs,
error: dispatchFailure(validationError({
message: 'Error: "gsd-sdk query" requires a command',
details: { reason: 'missing_command' },
})),
};
}
return { queryArgs };
}
export function planQueryDispatch(
queryArgv: string[],
topology: CommandTopology,
cjsFallbackEnabled: boolean,
): DispatchPlan {
const queryCommand = queryArgv[0];
if (!queryCommand) {
return { mode: 'error', normalized: { command: '', args: [], tokens: [] }, matched: null };
}
const [normCmd, normArgs] = normalizeQueryCommand(queryCommand, queryArgv.slice(1));
const normalizedTokens = [normCmd, ...normArgs];
const resolved = topology.resolve(queryArgv, !cjsFallbackEnabled);
if (resolved.kind === 'match') {
return { mode: 'native', normalized: { command: normCmd, args: normArgs, tokens: normalizedTokens }, matched: resolved };
}
if (cjsFallbackEnabled) {
return { mode: 'cjs', normalized: { command: normCmd, args: normArgs, tokens: normalizedTokens }, matched: null };
}
return {
mode: 'error',
normalized: { command: normCmd, args: normArgs, tokens: normalizedTokens },
matched: null,
noMatchMessage: resolved.message,
noMatchNormalized: resolved.normalized,
noMatchAttempted: resolved.attempted,
noMatchHints: resolved.hints,
};
}
function fail(error: ReturnType<typeof validationError> | ReturnType<typeof unknownCommandError>, stderr: string[] = []): QueryDispatchResult {
return toDispatchFailure(error, stderr);
}
export async function runQueryDispatch(deps: QueryDispatchDeps, queryArgv: string[]): Promise<QueryDispatchResult> {
const validated = validateQueryDispatchInput(queryArgv);
if (validated.error) return validated.error;
const { queryArgs, pickField } = validated;
const plan = planQueryDispatch(queryArgs, deps.topology, deps.cjsFallbackEnabled);
const normCmd = plan.normalized.command;
const normArgs = plan.normalized.args;
if (!normCmd || !String(normCmd).trim()) {
return fail(validationError({ message: 'Error: "gsd-sdk query" requires a command', details: { reason: 'empty_normalized_command' } }));
}
if (plan.mode === 'error') {
return fail(unknownCommandError({
message: plan.noMatchMessage ?? `Error: Unknown command: "${queryArgs[0] ?? normCmd}"`,
normalized: plan.noMatchNormalized ?? [normCmd, ...normArgs].join(' ').trim(),
attempted: plan.noMatchAttempted ?? [],
hints: plan.noMatchHints ?? [],
}));
}
if (plan.mode === 'cjs') {
if (canUseCjsFallback({ cjsFallbackEnabled: deps.cjsFallbackEnabled })) {
try {
const gsdPath = deps.resolveGsdToolsPath(deps.projectDir);
return await runCjsFallbackDispatch({
projectDir: deps.projectDir,
gsdToolsPath: gsdPath,
normCmd,
normArgs,
ws: deps.ws,
pickField,
});
} catch (e) {
return toDispatchFailure(mapFallbackDispatchError(e, normCmd, normArgs));
}
}
return toDispatchFailure(mapFallbackDispatchError(new Error('CJS fallback denied by policy'), normCmd, normArgs));
}
const matched = plan.matched;
if (!matched) {
return toDispatchFailure(mapFallbackDispatchError(new Error('No native match in dispatch plan'), normCmd, normArgs));
}
// #3259: guard — if the invocation contains --help / -h AND the matched
// handler is a mutating command (mutation: true in the command manifest),
// short-circuit to a non-mutating stub. Mutating handlers are not help-aware
// by default (fail-closed). This prevents e.g. `milestone.complete --help`
// from writing milestone artifacts to disk.
const helpFlagPresent = matched.args.some((a) => a === '--help' || a === '-h');
if (helpFlagPresent && matched.mutation) {
return dispatchSuccess(
formatSuccess(
{ help: `Usage: gsd-sdk query ${matched.canonical} [args...]` },
undefined,
),
);
}
const dispatchNative = deps.nativeAdapter
? (cmd: string, args: string[]) => deps.nativeAdapter!.dispatch(cmd, args)
: deps.dispatchNative;
try {
const result = dispatchNative
? await dispatchNative(matched.canonical, matched.args)
: await matched.adapter(matched.args, deps.projectDir, deps.ws);
return dispatchSuccess(formatSuccess(result.data, result.format, pickField));
} catch (e) {
return toDispatchFailure(mapNativeDispatchError(e, matched.canonical, matched.args));
}
}

View File

@@ -1,181 +0,0 @@
/**
* E2E integration test — proves full SDK pipeline:
* parse → prompt → query() → SUMMARY.md
*
* Requires Claude Code CLI (`claude`) installed and authenticated, plus
* opt-in env `GSD_ENABLE_E2E=1`. Skips if env unset or CLI unavailable.
*/
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
import { execSync } from 'node:child_process';
import { mkdtemp, cp, rm, readFile, readdir } from 'node:fs/promises';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';
import { GSD, parsePlanFile, GSDEventType } from './index.js';
import type { GSDEvent } from './index.js';
// ─── CLI availability check ─────────────────────────────────────────────────
let cliAvailable = false;
try {
execSync('which claude', { stdio: 'ignore' });
cliAvailable = true;
} catch {
cliAvailable = false;
}
const e2eEnabled = process.env.GSD_ENABLE_E2E === '1';
const canRunE2E = cliAvailable && e2eEnabled;
const __dirname = fileURLToPath(new URL('.', import.meta.url));
const fixturesDir = join(__dirname, '..', 'test-fixtures');
// ─── Test suite ──────────────────────────────────────────────────────────────
describe.skipIf(!canRunE2E)('E2E: Single plan execution', () => {
let tmpDir: string;
beforeAll(async () => {
tmpDir = await mkdtemp(join(tmpdir(), 'gsd-sdk-e2e-'));
// Copy fixture files to temp directory
await cp(fixturesDir, tmpDir, { recursive: true });
});
afterAll(async () => {
if (tmpDir) {
await rm(tmpDir, { recursive: true, force: true });
}
});
it('executes a single plan and returns a valid PlanResult', async () => {
const gsd = new GSD({ projectDir: tmpDir, maxBudgetUsd: 1.0, maxTurns: 20 });
const result = await gsd.executePlan('sample-plan.md');
expect(result.success).toBe(true);
expect(typeof result.sessionId).toBe('string');
expect(result.sessionId.length).toBeGreaterThan(0);
expect(result.totalCostUsd).toBeGreaterThanOrEqual(0);
expect(result.durationMs).toBeGreaterThan(0);
expect(result.numTurns).toBeGreaterThan(0);
// Verify the plan's task was executed — output.txt should exist
const outputPath = join(tmpDir, 'output.txt');
const outputContent = await readFile(outputPath, 'utf-8');
expect(outputContent).toContain('hello from gsd-sdk');
}, 120_000); // 2 minute timeout for real CLI execution
it('proves session isolation (R014) — different session IDs for sequential runs', async () => {
// Create a second temp dir for isolation proof
const tmpDir2 = await mkdtemp(join(tmpdir(), 'gsd-sdk-e2e-'));
await cp(fixturesDir, tmpDir2, { recursive: true });
try {
const gsd1 = new GSD({ projectDir: tmpDir, maxBudgetUsd: 1.0, maxTurns: 20 });
const gsd2 = new GSD({ projectDir: tmpDir2, maxBudgetUsd: 1.0, maxTurns: 20 });
const result1 = await gsd1.executePlan('sample-plan.md');
const result2 = await gsd2.executePlan('sample-plan.md');
// Different sessions must have different session IDs
expect(result1.sessionId).not.toBe(result2.sessionId);
// Both should track cost independently
expect(result1.totalCostUsd).toBeGreaterThanOrEqual(0);
expect(result2.totalCostUsd).toBeGreaterThanOrEqual(0);
} finally {
await rm(tmpDir2, { recursive: true, force: true });
}
}, 240_000); // 4 minute timeout — two sequential runs
});
describe('E2E: Fixture validation (no CLI required)', () => {
it('fixture PLAN.md is valid and parseable', async () => {
const plan = await parsePlanFile(join(fixturesDir, 'sample-plan.md'));
expect(plan.frontmatter.phase).toBe('01-test');
expect(plan.frontmatter.plan).toBe('01');
expect(plan.frontmatter.type).toBe('execute');
expect(plan.frontmatter.wave).toBe(1);
expect(plan.frontmatter.depends_on).toEqual([]);
expect(plan.frontmatter.files_modified).toEqual(['output.txt']);
expect(plan.frontmatter.autonomous).toBe(true);
expect(plan.frontmatter.requirements).toEqual(['TEST-01']);
expect(plan.frontmatter.must_haves.truths).toEqual(['output.txt exists with expected content']);
expect(plan.objective).toContain('simple output file');
expect(plan.tasks).toHaveLength(1);
expect(plan.tasks[0].name).toBe('Create output file');
expect(plan.tasks[0].type).toBe('auto');
expect(plan.tasks[0].verify).toBe('test -f output.txt');
});
});
describe.skipIf(!canRunE2E)('E2E: Event stream during plan execution (R007)', () => {
let tmpDir: string;
beforeAll(async () => {
tmpDir = await mkdtemp(join(tmpdir(), 'gsd-sdk-e2e-stream-'));
await cp(fixturesDir, tmpDir, { recursive: true });
});
afterAll(async () => {
if (tmpDir) {
await rm(tmpDir, { recursive: true, force: true });
}
});
it('event stream emits events during plan execution (R007)', async () => {
const events: GSDEvent[] = [];
const gsd = new GSD({ projectDir: tmpDir, maxBudgetUsd: 1.0, maxTurns: 20 });
// Subscribe to all events
gsd.onEvent((event) => {
events.push(event);
});
const result = await gsd.executePlan('sample-plan.md');
expect(result.success).toBe(true);
// (a) At least one session_init event received
const initEvents = events.filter(e => e.type === GSDEventType.SessionInit);
expect(initEvents.length).toBeGreaterThanOrEqual(1);
// (b) At least one tool_call event received
const toolCallEvents = events.filter(e => e.type === GSDEventType.ToolCall);
expect(toolCallEvents.length).toBeGreaterThanOrEqual(1);
// (c) Exactly one session_complete event with cost >= 0
const completeEvents = events.filter(e => e.type === GSDEventType.SessionComplete);
expect(completeEvents).toHaveLength(1);
const completeEvent = completeEvents[0]!;
if (completeEvent.type === GSDEventType.SessionComplete) {
expect(completeEvent.totalCostUsd).toBeGreaterThanOrEqual(0);
}
// (d) Events arrived in order: session_init before tool_call before session_complete
const initIdx = events.findIndex(e => e.type === GSDEventType.SessionInit);
const toolCallIdx = events.findIndex(e => e.type === GSDEventType.ToolCall);
const completeIdx = events.findIndex(e => e.type === GSDEventType.SessionComplete);
expect(initIdx).toBeLessThan(toolCallIdx);
expect(toolCallIdx).toBeLessThan(completeIdx);
// Bonus: at least one cost_update event was emitted
const costEvents = events.filter(e => e.type === GSDEventType.CostUpdate);
expect(costEvents.length).toBeGreaterThanOrEqual(1);
}, 120_000);
});
describe('E2E: Error handling', () => {
it('returns failure for nonexistent plan path', async () => {
const tmpDir = await mkdtemp(join(tmpdir(), 'gsd-sdk-e2e-err-'));
try {
const gsd = new GSD({ projectDir: tmpDir });
await expect(gsd.executePlan('nonexistent-plan.md')).rejects.toThrow();
} finally {
await rm(tmpDir, { recursive: true, force: true });
}
});
});

View File

@@ -1 +0,0 @@
export * from './errors/index.js';

View File

@@ -1,72 +0,0 @@
/**
* Error classification system for the GSD SDK.
*
* Provides a taxonomy of error types with semantic exit codes,
* enabling CLI consumers and agents to distinguish between
* validation failures, execution errors, blocked states, and
* interruptions.
*
* @example
* ```typescript
* import { GSDError, ErrorClassification, exitCodeFor } from './errors.js';
*
* throw new GSDError('missing required arg', ErrorClassification.Validation);
* // CLI catch handler: process.exitCode = exitCodeFor(err.classification); // 10
* ```
*/
// ─── Error Classification ───────────────────────────────────────────────────
/** Classifies SDK errors into semantic categories for exit code mapping. */
export enum ErrorClassification {
/** Bad input, missing args, schema violations. Exit code 10. */
Validation = 'validation',
/** Runtime failure, file I/O, parse errors. Exit code 1. */
Execution = 'execution',
/** Dependency missing, phase not found. Exit code 11. */
Blocked = 'blocked',
/** Timeout, signal, user cancel. Exit code 1. */
Interruption = 'interruption',
}
// ─── GSDError ───────────────────────────────────────────────────────────────
/**
* Base error class for the GSD SDK with classification support.
*
* @param message - Human-readable error description
* @param classification - Error category for exit code mapping
*/
export class GSDError extends Error {
readonly name = 'GSDError';
readonly classification: ErrorClassification;
constructor(message: string, classification: ErrorClassification) {
super(message);
this.classification = classification;
}
}
// ─── Exit code mapping ──────────────────────────────────────────────────────
/**
* Maps an error classification to a semantic exit code.
*
* @param classification - The error classification to map
* @returns Numeric exit code: 10 (validation), 11 (blocked), 1 (execution/interruption)
*/
export function exitCodeFor(classification: ErrorClassification): number {
switch (classification) {
case ErrorClassification.Validation:
return 10;
case ErrorClassification.Blocked:
return 11;
case ErrorClassification.Execution:
case ErrorClassification.Interruption:
default:
return 1;
}
}

View File

@@ -1,661 +0,0 @@
import { describe, it, expect, beforeEach, vi } from 'vitest';
import { GSDEventStream } from './event-stream.js';
import {
GSDEventType,
PhaseType,
type GSDEvent,
type GSDSessionInitEvent,
type GSDSessionCompleteEvent,
type GSDSessionErrorEvent,
type GSDAssistantTextEvent,
type GSDToolCallEvent,
type GSDToolProgressEvent,
type GSDToolUseSummaryEvent,
type GSDTaskStartedEvent,
type GSDTaskProgressEvent,
type GSDTaskNotificationEvent,
type GSDAPIRetryEvent,
type GSDRateLimitEvent,
type GSDStatusChangeEvent,
type GSDCompactBoundaryEvent,
type GSDStreamEvent,
type GSDCostUpdateEvent,
type TransportHandler,
} from './types.js';
import type {
SDKMessage,
SDKSystemMessage,
SDKAssistantMessage,
SDKResultSuccess,
SDKResultError,
SDKToolProgressMessage,
SDKToolUseSummaryMessage,
SDKTaskStartedMessage,
SDKTaskProgressMessage,
SDKTaskNotificationMessage,
SDKAPIRetryMessage,
SDKRateLimitEvent,
SDKStatusMessage,
SDKCompactBoundaryMessage,
SDKPartialAssistantMessage,
} from '@anthropic-ai/claude-agent-sdk';
import type { UUID } from 'crypto';
// ─── Helpers ─────────────────────────────────────────────────────────────────
const TEST_UUID = '00000000-0000-0000-0000-000000000000' as UUID;
const TEST_SESSION = 'test-session-1';
function makeSystemInit(): SDKSystemMessage {
return {
type: 'system',
subtype: 'init',
agents: [],
apiKeySource: 'user',
betas: [],
claude_code_version: '1.0.0',
cwd: '/test',
tools: ['Read', 'Write', 'Bash'],
mcp_servers: [],
model: 'claude-sonnet-4-6',
permissionMode: 'bypassPermissions',
slash_commands: [],
output_style: 'text',
skills: [],
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKSystemMessage;
}
function makeAssistantMsg(content: Array<{ type: string; [key: string]: unknown }>): SDKAssistantMessage {
return {
type: 'assistant',
message: {
content,
id: 'msg-1',
type: 'message',
role: 'assistant',
model: 'claude-sonnet-4-6',
stop_reason: 'end_turn',
stop_sequence: null,
usage: { input_tokens: 100, output_tokens: 50 },
} as unknown as SDKAssistantMessage['message'],
parent_tool_use_id: null,
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKAssistantMessage;
}
function makeResultSuccess(costUsd = 0.05): SDKResultSuccess {
return {
type: 'result',
subtype: 'success',
duration_ms: 5000,
duration_api_ms: 4000,
is_error: false,
num_turns: 3,
result: 'Task completed successfully',
stop_reason: 'end_turn',
total_cost_usd: costUsd,
usage: { input_tokens: 1000, output_tokens: 500, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 },
modelUsage: {},
permission_denials: [],
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKResultSuccess;
}
function makeResultError(): SDKResultError {
return {
type: 'result',
subtype: 'error_max_turns',
duration_ms: 10000,
duration_api_ms: 8000,
is_error: true,
num_turns: 50,
stop_reason: null,
total_cost_usd: 2.50,
usage: { input_tokens: 5000, output_tokens: 2000, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 },
modelUsage: {},
permission_denials: [],
errors: ['Max turns exceeded'],
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKResultError;
}
function makeToolProgress(): SDKToolProgressMessage {
return {
type: 'tool_progress',
tool_use_id: 'tu-1',
tool_name: 'Bash',
parent_tool_use_id: null,
elapsed_time_seconds: 5.2,
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKToolProgressMessage;
}
function makeToolUseSummary(): SDKToolUseSummaryMessage {
return {
type: 'tool_use_summary',
summary: 'Ran 3 bash commands',
preceding_tool_use_ids: ['tu-1', 'tu-2', 'tu-3'],
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKToolUseSummaryMessage;
}
function makeTaskStarted(): SDKTaskStartedMessage {
return {
type: 'system',
subtype: 'task_started',
task_id: 'task-1',
description: 'Running test suite',
task_type: 'local_workflow',
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKTaskStartedMessage;
}
function makeTaskProgress(): SDKTaskProgressMessage {
return {
type: 'system',
subtype: 'task_progress',
task_id: 'task-1',
description: 'Running tests',
usage: { total_tokens: 500, tool_uses: 3, duration_ms: 2000 },
last_tool_name: 'Bash',
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKTaskProgressMessage;
}
function makeTaskNotification(): SDKTaskNotificationMessage {
return {
type: 'system',
subtype: 'task_notification',
task_id: 'task-1',
status: 'completed',
output_file: '/tmp/output.txt',
summary: 'All tests passed',
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKTaskNotificationMessage;
}
function makeAPIRetry(): SDKAPIRetryMessage {
return {
type: 'system',
subtype: 'api_retry',
attempt: 2,
max_retries: 5,
retry_delay_ms: 1000,
error_status: 529,
error: 'server_error',
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKAPIRetryMessage;
}
function makeRateLimitEvent(): SDKRateLimitEvent {
return {
type: 'rate_limit_event',
rate_limit_info: {
status: 'allowed_warning',
resetsAt: Date.now() + 60000,
utilization: 0.85,
},
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKRateLimitEvent;
}
function makeStatusMessage(): SDKStatusMessage {
return {
type: 'system',
subtype: 'status',
status: 'compacting',
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKStatusMessage;
}
function makeCompactBoundary(): SDKCompactBoundaryMessage {
return {
type: 'system',
subtype: 'compact_boundary',
compact_metadata: {
trigger: 'auto',
pre_tokens: 95000,
},
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKCompactBoundaryMessage;
}
// ─── SDKMessage → GSDEvent mapping tests ─────────────────────────────────────
describe('GSDEventStream', () => {
let stream: GSDEventStream;
beforeEach(() => {
stream = new GSDEventStream();
});
describe('mapSDKMessage', () => {
it('maps SDKSystemMessage init → SessionInit', () => {
const event = stream.mapSDKMessage(makeSystemInit());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.SessionInit);
const init = event as GSDSessionInitEvent;
expect(init.model).toBe('claude-sonnet-4-6');
expect(init.tools).toEqual(['Read', 'Write', 'Bash']);
expect(init.cwd).toBe('/test');
expect(init.sessionId).toBe(TEST_SESSION);
});
it('maps assistant text blocks → AssistantText', () => {
const msg = makeAssistantMsg([
{ type: 'text', text: 'Hello ' },
{ type: 'text', text: 'world' },
]);
const event = stream.mapSDKMessage(msg);
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.AssistantText);
expect((event as GSDAssistantTextEvent).text).toBe('Hello world');
});
it('maps assistant tool_use blocks → ToolCall', () => {
const msg = makeAssistantMsg([
{ type: 'tool_use', id: 'tu-1', name: 'Read', input: { path: 'test.ts' } },
]);
const event = stream.mapSDKMessage(msg);
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.ToolCall);
const tc = event as GSDToolCallEvent;
expect(tc.toolName).toBe('Read');
expect(tc.toolUseId).toBe('tu-1');
expect(tc.input).toEqual({ path: 'test.ts' });
});
it('handles multi-block assistant messages (text + tool_use)', () => {
const events: GSDEvent[] = [];
stream.on('event', (e: GSDEvent) => events.push(e));
const msg = makeAssistantMsg([
{ type: 'text', text: 'Let me check that.' },
{ type: 'tool_use', id: 'tu-1', name: 'Read', input: { path: 'f.ts' } },
]);
// mapAndEmit will emit the text event directly and return the tool_call
const returned = stream.mapAndEmit(msg);
expect(returned).not.toBeNull();
// Should have received 2 events total
expect(events).toHaveLength(2);
expect(events[0]!.type).toBe(GSDEventType.AssistantText);
expect(events[1]!.type).toBe(GSDEventType.ToolCall);
});
it('maps SDKResultSuccess → SessionComplete', () => {
const event = stream.mapSDKMessage(makeResultSuccess());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.SessionComplete);
const complete = event as GSDSessionCompleteEvent;
expect(complete.success).toBe(true);
expect(complete.totalCostUsd).toBe(0.05);
expect(complete.durationMs).toBe(5000);
expect(complete.numTurns).toBe(3);
expect(complete.result).toBe('Task completed successfully');
});
it('maps SDKResultError → SessionError', () => {
const event = stream.mapSDKMessage(makeResultError());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.SessionError);
const err = event as GSDSessionErrorEvent;
expect(err.success).toBe(false);
expect(err.errorSubtype).toBe('error_max_turns');
expect(err.errors).toContain('Max turns exceeded');
});
it('maps SDKToolProgressMessage → ToolProgress', () => {
const event = stream.mapSDKMessage(makeToolProgress());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.ToolProgress);
const tp = event as GSDToolProgressEvent;
expect(tp.toolName).toBe('Bash');
expect(tp.toolUseId).toBe('tu-1');
expect(tp.elapsedSeconds).toBe(5.2);
});
it('maps SDKToolUseSummaryMessage → ToolUseSummary', () => {
const event = stream.mapSDKMessage(makeToolUseSummary());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.ToolUseSummary);
const tus = event as GSDToolUseSummaryEvent;
expect(tus.summary).toBe('Ran 3 bash commands');
expect(tus.toolUseIds).toEqual(['tu-1', 'tu-2', 'tu-3']);
});
it('maps SDKTaskStartedMessage → TaskStarted', () => {
const event = stream.mapSDKMessage(makeTaskStarted());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.TaskStarted);
const ts = event as GSDTaskStartedEvent;
expect(ts.taskId).toBe('task-1');
expect(ts.description).toBe('Running test suite');
expect(ts.taskType).toBe('local_workflow');
});
it('maps SDKTaskProgressMessage → TaskProgress', () => {
const event = stream.mapSDKMessage(makeTaskProgress());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.TaskProgress);
const tp = event as GSDTaskProgressEvent;
expect(tp.taskId).toBe('task-1');
expect(tp.totalTokens).toBe(500);
expect(tp.toolUses).toBe(3);
expect(tp.lastToolName).toBe('Bash');
});
it('maps SDKTaskNotificationMessage → TaskNotification', () => {
const event = stream.mapSDKMessage(makeTaskNotification());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.TaskNotification);
const tn = event as GSDTaskNotificationEvent;
expect(tn.taskId).toBe('task-1');
expect(tn.status).toBe('completed');
expect(tn.summary).toBe('All tests passed');
});
it('maps SDKAPIRetryMessage → APIRetry', () => {
const event = stream.mapSDKMessage(makeAPIRetry());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.APIRetry);
const retry = event as GSDAPIRetryEvent;
expect(retry.attempt).toBe(2);
expect(retry.maxRetries).toBe(5);
expect(retry.retryDelayMs).toBe(1000);
expect(retry.errorStatus).toBe(529);
});
it('maps SDKRateLimitEvent → RateLimit', () => {
const event = stream.mapSDKMessage(makeRateLimitEvent());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.RateLimit);
const rl = event as GSDRateLimitEvent;
expect(rl.status).toBe('allowed_warning');
expect(rl.utilization).toBe(0.85);
});
it('maps SDKStatusMessage → StatusChange', () => {
const event = stream.mapSDKMessage(makeStatusMessage());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.StatusChange);
expect((event as GSDStatusChangeEvent).status).toBe('compacting');
});
it('maps SDKCompactBoundaryMessage → CompactBoundary', () => {
const event = stream.mapSDKMessage(makeCompactBoundary());
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.CompactBoundary);
const cb = event as GSDCompactBoundaryEvent;
expect(cb.trigger).toBe('auto');
expect(cb.preTokens).toBe(95000);
});
it('returns null for user messages', () => {
const msg = { type: 'user', session_id: TEST_SESSION } as SDKMessage;
expect(stream.mapSDKMessage(msg)).toBeNull();
});
it('returns null for auth_status messages', () => {
const msg = { type: 'auth_status', session_id: TEST_SESSION } as SDKMessage;
expect(stream.mapSDKMessage(msg)).toBeNull();
});
it('returns null for prompt_suggestion messages', () => {
const msg = { type: 'prompt_suggestion', session_id: TEST_SESSION } as SDKMessage;
expect(stream.mapSDKMessage(msg)).toBeNull();
});
it('includes phase and planName context when provided', () => {
const event = stream.mapSDKMessage(makeSystemInit(), {
phase: PhaseType.Execute,
planName: 'feature-plan',
});
expect(event!.phase).toBe(PhaseType.Execute);
expect(event!.planName).toBe('feature-plan');
});
});
// ─── Cost tracking ─────────────────────────────────────────────────────
describe('cost tracking', () => {
it('tracks per-session cost on session_complete', () => {
stream.mapSDKMessage(makeResultSuccess(0.05));
const cost = stream.getCost();
expect(cost.session).toBe(0.05);
expect(cost.cumulative).toBe(0.05);
});
it('accumulates cumulative cost across multiple sessions', () => {
// Session 1
const result1 = makeResultSuccess(0.05);
result1.session_id = 'session-1';
stream.mapSDKMessage(result1);
// Session 2
const result2 = makeResultSuccess(0.10);
result2.session_id = 'session-2';
stream.mapSDKMessage(result2);
const cost = stream.getCost();
// Current session is session-2 (last one updated)
expect(cost.session).toBe(0.10);
expect(cost.cumulative).toBeCloseTo(0.15, 10);
});
it('correctly computes delta when same session updates cost', () => {
// Session reports intermediate cost, then final cost
const result1 = makeResultSuccess(0.03);
stream.mapSDKMessage(result1);
const result2 = makeResultSuccess(0.05);
stream.mapSDKMessage(result2);
const cost = stream.getCost();
expect(cost.session).toBe(0.05);
// Cumulative should be 0.05, not 0.08 (delta was +0.02, not +0.05)
expect(cost.cumulative).toBeCloseTo(0.05, 10);
});
it('tracks error session costs too', () => {
stream.mapSDKMessage(makeResultError());
const cost = stream.getCost();
expect(cost.session).toBe(2.50);
expect(cost.cumulative).toBe(2.50);
});
});
// ─── Transport management ──────────────────────────────────────────────
describe('transport management', () => {
it('delivers events to subscribed transports', () => {
const received: GSDEvent[] = [];
const transport: TransportHandler = {
onEvent: (event) => received.push(event),
close: () => {},
};
stream.addTransport(transport);
stream.mapAndEmit(makeSystemInit());
expect(received).toHaveLength(1);
expect(received[0]!.type).toBe(GSDEventType.SessionInit);
});
it('delivers events to multiple transports', () => {
const received1: GSDEvent[] = [];
const received2: GSDEvent[] = [];
stream.addTransport({
onEvent: (e) => received1.push(e),
close: () => {},
});
stream.addTransport({
onEvent: (e) => received2.push(e),
close: () => {},
});
stream.mapAndEmit(makeSystemInit());
expect(received1).toHaveLength(1);
expect(received2).toHaveLength(1);
});
it('stops delivering events after transport removal', () => {
const received: GSDEvent[] = [];
const transport: TransportHandler = {
onEvent: (e) => received.push(e),
close: () => {},
};
stream.addTransport(transport);
stream.mapAndEmit(makeSystemInit());
expect(received).toHaveLength(1);
stream.removeTransport(transport);
stream.mapAndEmit(makeResultSuccess());
expect(received).toHaveLength(1); // No new events
});
it('survives transport.onEvent() throwing', () => {
const badTransport: TransportHandler = {
onEvent: () => { throw new Error('transport failed'); },
close: () => {},
};
const goodReceived: GSDEvent[] = [];
const goodTransport: TransportHandler = {
onEvent: (e) => goodReceived.push(e),
close: () => {},
};
stream.addTransport(badTransport);
stream.addTransport(goodTransport);
// Should not throw, and good transport still receives events
expect(() => stream.mapAndEmit(makeSystemInit())).not.toThrow();
expect(goodReceived).toHaveLength(1);
});
it('closeAll() calls close on all transports and clears them', () => {
const closeCalled: boolean[] = [];
stream.addTransport({
onEvent: () => {},
close: () => closeCalled.push(true),
});
stream.addTransport({
onEvent: () => {},
close: () => closeCalled.push(true),
});
stream.closeAll();
expect(closeCalled).toHaveLength(2);
// No more deliveries after closeAll
const events: GSDEvent[] = [];
stream.on('event', (e: GSDEvent) => events.push(e));
stream.mapAndEmit(makeSystemInit());
// EventEmitter listeners still work, but transports are gone
expect(events).toHaveLength(1);
});
});
// ─── EventEmitter integration ──────────────────────────────────────────
describe('EventEmitter integration', () => {
it('emits typed events via "event" channel', () => {
const events: GSDEvent[] = [];
stream.on('event', (e: GSDEvent) => events.push(e));
stream.mapAndEmit(makeSystemInit());
stream.mapAndEmit(makeResultSuccess());
expect(events).toHaveLength(2);
expect(events[0]!.type).toBe(GSDEventType.SessionInit);
expect(events[1]!.type).toBe(GSDEventType.SessionComplete);
});
it('emits events on per-type channels', () => {
const initEvents: GSDEvent[] = [];
stream.on(GSDEventType.SessionInit, (e: GSDEvent) => initEvents.push(e));
stream.mapAndEmit(makeSystemInit());
stream.mapAndEmit(makeResultSuccess());
expect(initEvents).toHaveLength(1);
expect(initEvents[0]!.type).toBe(GSDEventType.SessionInit);
});
});
// ─── Stream event mapping ──────────────────────────────────────────────
describe('stream_event mapping', () => {
it('maps SDKPartialAssistantMessage → StreamEvent', () => {
const msg = {
type: 'stream_event' as const,
event: { type: 'content_block_delta' },
parent_tool_use_id: null,
uuid: TEST_UUID,
session_id: TEST_SESSION,
} as SDKPartialAssistantMessage;
const event = stream.mapSDKMessage(msg);
expect(event).not.toBeNull();
expect(event!.type).toBe(GSDEventType.StreamEvent);
expect((event as GSDStreamEvent).event).toEqual({ type: 'content_block_delta' });
});
});
// ─── Empty / edge cases ────────────────────────────────────────────────
describe('edge cases', () => {
it('returns null for assistant messages with empty content', () => {
const msg = makeAssistantMsg([]);
expect(stream.mapSDKMessage(msg)).toBeNull();
});
it('returns null for assistant messages with only empty text', () => {
const msg = makeAssistantMsg([{ type: 'text', text: '' }]);
expect(stream.mapSDKMessage(msg)).toBeNull();
});
it('returns null for unknown system subtypes', () => {
const msg = {
type: 'system',
subtype: 'unknown_future_type',
session_id: TEST_SESSION,
uuid: TEST_UUID,
} as unknown as SDKMessage;
expect(stream.mapSDKMessage(msg)).toBeNull();
});
});
});

View File

@@ -1,441 +0,0 @@
/**
* GSD Event Stream — maps SDKMessage variants to typed GSD events.
*
* Extends EventEmitter to provide a typed event bus. Includes:
* - SDKMessage → GSDEvent mapping
* - Transport management (subscribe/unsubscribe handlers)
* - Per-session cost tracking with cumulative totals
*/
import { EventEmitter } from 'node:events';
import type {
SDKMessage,
SDKResultSuccess,
SDKResultError,
SDKAssistantMessage,
SDKSystemMessage,
SDKToolProgressMessage,
SDKTaskNotificationMessage,
SDKTaskStartedMessage,
SDKTaskProgressMessage,
SDKToolUseSummaryMessage,
SDKRateLimitEvent,
SDKAPIRetryMessage,
SDKStatusMessage,
SDKCompactBoundaryMessage,
SDKPartialAssistantMessage,
} from '@anthropic-ai/claude-agent-sdk';
import {
GSDEventType,
type GSDEvent,
type GSDSessionInitEvent,
type GSDSessionCompleteEvent,
type GSDSessionErrorEvent,
type GSDAssistantTextEvent,
type GSDToolCallEvent,
type GSDToolProgressEvent,
type GSDToolUseSummaryEvent,
type GSDTaskStartedEvent,
type GSDTaskProgressEvent,
type GSDTaskNotificationEvent,
type GSDCostUpdateEvent,
type GSDAPIRetryEvent,
type GSDRateLimitEvent as GSDRateLimitEventType,
type GSDStatusChangeEvent,
type GSDCompactBoundaryEvent,
type GSDStreamEvent,
type TransportHandler,
type CostBucket,
type CostTracker,
type PhaseType,
} from './types.js';
// ─── Mapping context ─────────────────────────────────────────────────────────
export interface EventStreamContext {
phase?: PhaseType;
planName?: string;
}
// ─── GSDEventStream ──────────────────────────────────────────────────────────
export class GSDEventStream extends EventEmitter {
private readonly transports: Set<TransportHandler> = new Set();
private readonly costTracker: CostTracker = {
sessions: new Map(),
cumulativeCostUsd: 0,
};
constructor() {
super();
this.setMaxListeners(20);
}
// ─── Transport management ────────────────────────────────────────────
/** Subscribe a transport handler to receive all events. */
addTransport(handler: TransportHandler): void {
this.transports.add(handler);
}
/** Unsubscribe a transport handler. */
removeTransport(handler: TransportHandler): void {
this.transports.delete(handler);
}
/** Close all transports. */
closeAll(): void {
for (const transport of this.transports) {
try {
transport.close();
} catch {
// Ignore transport close errors
}
}
this.transports.clear();
}
// ─── Event emission ──────────────────────────────────────────────────
/** Emit a typed GSD event to all listeners and transports. */
emitEvent(event: GSDEvent): void {
// Emit via EventEmitter for listener-based consumers
this.emit('event', event);
this.emit(event.type, event);
// Deliver to all transports — wrap in try/catch to prevent
// one bad transport from killing the stream
for (const transport of this.transports) {
try {
transport.onEvent(event);
} catch {
// Silently ignore transport errors
}
}
}
// ─── SDKMessage mapping ──────────────────────────────────────────────
/**
* Map an SDKMessage to a GSDEvent.
* Returns null for non-actionable message types (user messages, replays, etc.).
*/
mapSDKMessage(msg: SDKMessage, context: EventStreamContext = {}): GSDEvent | null {
const base = {
timestamp: new Date().toISOString(),
sessionId: 'session_id' in msg ? (msg.session_id as string) : '',
phase: context.phase,
planName: context.planName,
};
switch (msg.type) {
case 'system':
return this.mapSystemMessage(msg as SDKSystemMessage | SDKAPIRetryMessage | SDKStatusMessage | SDKCompactBoundaryMessage | SDKTaskStartedMessage | SDKTaskProgressMessage | SDKTaskNotificationMessage, base);
case 'assistant':
return this.mapAssistantMessage(msg as SDKAssistantMessage, base);
case 'result':
return this.mapResultMessage(msg as SDKResultSuccess | SDKResultError, base);
case 'tool_progress':
return this.mapToolProgressMessage(msg as SDKToolProgressMessage, base);
case 'tool_use_summary':
return this.mapToolUseSummaryMessage(msg as SDKToolUseSummaryMessage, base);
case 'rate_limit_event':
return this.mapRateLimitMessage(msg as SDKRateLimitEvent, base);
case 'stream_event':
return this.mapStreamEvent(msg as SDKPartialAssistantMessage, base);
// Non-actionable message types — ignore
case 'user':
case 'auth_status':
case 'prompt_suggestion':
return null;
default:
return null;
}
}
/**
* Map an SDKMessage and emit the resulting event (if any).
* Convenience method combining mapSDKMessage + emitEvent.
*/
mapAndEmit(msg: SDKMessage, context: EventStreamContext = {}): GSDEvent | null {
const event = this.mapSDKMessage(msg, context);
if (event) {
this.emitEvent(event);
}
return event;
}
// ─── Cost tracking ───────────────────────────────────────────────────
/** Get current cost totals. */
getCost(): { session: number; cumulative: number } {
const activeId = this.costTracker.activeSessionId;
const sessionCost = activeId
? (this.costTracker.sessions.get(activeId)?.costUsd ?? 0)
: 0;
return {
session: sessionCost,
cumulative: this.costTracker.cumulativeCostUsd,
};
}
/** Update cost for a session. */
private updateCost(sessionId: string, costUsd: number): void {
const existing = this.costTracker.sessions.get(sessionId);
const previousCost = existing?.costUsd ?? 0;
const delta = costUsd - previousCost;
const bucket: CostBucket = { sessionId, costUsd };
this.costTracker.sessions.set(sessionId, bucket);
this.costTracker.activeSessionId = sessionId;
this.costTracker.cumulativeCostUsd += delta;
}
// ─── Private mappers ─────────────────────────────────────────────────
private mapSystemMessage(
msg: SDKSystemMessage | SDKAPIRetryMessage | SDKStatusMessage | SDKCompactBoundaryMessage | SDKTaskStartedMessage | SDKTaskProgressMessage | SDKTaskNotificationMessage,
base: Omit<GSDEvent, 'type'>,
): GSDEvent | null {
// All system messages have a subtype
const subtype = (msg as { subtype: string }).subtype;
switch (subtype) {
case 'init': {
const initMsg = msg as SDKSystemMessage;
return {
...base,
type: GSDEventType.SessionInit,
model: initMsg.model,
tools: initMsg.tools,
cwd: initMsg.cwd,
} as GSDSessionInitEvent;
}
case 'api_retry': {
const retryMsg = msg as SDKAPIRetryMessage;
return {
...base,
type: GSDEventType.APIRetry,
attempt: retryMsg.attempt,
maxRetries: retryMsg.max_retries,
retryDelayMs: retryMsg.retry_delay_ms,
errorStatus: retryMsg.error_status,
} as GSDAPIRetryEvent;
}
case 'status': {
const statusMsg = msg as SDKStatusMessage;
return {
...base,
type: GSDEventType.StatusChange,
status: statusMsg.status,
} as GSDStatusChangeEvent;
}
case 'compact_boundary': {
const compactMsg = msg as SDKCompactBoundaryMessage;
return {
...base,
type: GSDEventType.CompactBoundary,
trigger: compactMsg.compact_metadata.trigger,
preTokens: compactMsg.compact_metadata.pre_tokens,
} as GSDCompactBoundaryEvent;
}
case 'task_started': {
const taskMsg = msg as SDKTaskStartedMessage;
return {
...base,
type: GSDEventType.TaskStarted,
taskId: taskMsg.task_id,
description: taskMsg.description,
taskType: taskMsg.task_type,
} as GSDTaskStartedEvent;
}
case 'task_progress': {
const progressMsg = msg as SDKTaskProgressMessage;
return {
...base,
type: GSDEventType.TaskProgress,
taskId: progressMsg.task_id,
description: progressMsg.description,
totalTokens: progressMsg.usage.total_tokens,
toolUses: progressMsg.usage.tool_uses,
durationMs: progressMsg.usage.duration_ms,
lastToolName: progressMsg.last_tool_name,
} as GSDTaskProgressEvent;
}
case 'task_notification': {
const notifMsg = msg as SDKTaskNotificationMessage;
return {
...base,
type: GSDEventType.TaskNotification,
taskId: notifMsg.task_id,
status: notifMsg.status,
summary: notifMsg.summary,
} as GSDTaskNotificationEvent;
}
// Non-actionable system subtypes
case 'hook_started':
case 'hook_progress':
case 'hook_response':
case 'local_command_output':
case 'session_state_changed':
case 'files_persisted':
case 'elicitation_complete':
return null;
default:
return null;
}
}
private mapAssistantMessage(
msg: SDKAssistantMessage,
base: Omit<GSDEvent, 'type'>,
): GSDEvent | null {
const events: GSDEvent[] = [];
// Extract text blocks — content blocks are a discriminated union with a 'type' field.
// Double-cast via unknown because BetaContentBlock's internal variants don't
// carry an index signature, so TS rejects the direct cast without a widening step.
const content = msg.message.content as unknown as Array<{ type: string; [key: string]: unknown }>;
const textBlocks = content.filter(
(b): b is { type: 'text'; text: string } => b.type === 'text',
);
if (textBlocks.length > 0) {
const text = textBlocks.map(b => b.text).join('');
if (text.length > 0) {
events.push({
...base,
type: GSDEventType.AssistantText,
text,
} as GSDAssistantTextEvent);
}
}
// Extract tool_use blocks
const toolUseBlocks = content.filter(
(b): b is { type: 'tool_use'; id: string; name: string; input: Record<string, unknown> } =>
b.type === 'tool_use',
);
for (const block of toolUseBlocks) {
events.push({
...base,
type: GSDEventType.ToolCall,
toolName: block.name,
toolUseId: block.id,
input: block.input as Record<string, unknown>,
} as GSDToolCallEvent);
}
// Return the first event — for multi-event messages, emit the rest
// via separate emitEvent calls. This preserves the single-return contract
// while still handling multi-block messages.
if (events.length === 0) return null;
if (events.length === 1) return events[0]!;
// For multi-event assistant messages, emit all but the last directly,
// and return the last one for the caller to handle
for (let i = 0; i < events.length - 1; i++) {
this.emitEvent(events[i]!);
}
return events[events.length - 1]!;
}
private mapResultMessage(
msg: SDKResultSuccess | SDKResultError,
base: Omit<GSDEvent, 'type'>,
): GSDEvent {
// Update cost tracking
this.updateCost(msg.session_id, msg.total_cost_usd);
if (msg.subtype === 'success') {
const successMsg = msg as SDKResultSuccess;
return {
...base,
type: GSDEventType.SessionComplete,
success: true,
totalCostUsd: successMsg.total_cost_usd,
durationMs: successMsg.duration_ms,
numTurns: successMsg.num_turns,
result: successMsg.result,
} as GSDSessionCompleteEvent;
}
const errorMsg = msg as SDKResultError;
return {
...base,
type: GSDEventType.SessionError,
success: false,
totalCostUsd: errorMsg.total_cost_usd,
durationMs: errorMsg.duration_ms,
numTurns: errorMsg.num_turns,
errorSubtype: errorMsg.subtype,
errors: errorMsg.errors,
} as GSDSessionErrorEvent;
}
private mapToolProgressMessage(
msg: SDKToolProgressMessage,
base: Omit<GSDEvent, 'type'>,
): GSDToolProgressEvent {
return {
...base,
type: GSDEventType.ToolProgress,
toolName: msg.tool_name,
toolUseId: msg.tool_use_id,
elapsedSeconds: msg.elapsed_time_seconds,
} as GSDToolProgressEvent;
}
private mapToolUseSummaryMessage(
msg: SDKToolUseSummaryMessage,
base: Omit<GSDEvent, 'type'>,
): GSDToolUseSummaryEvent {
return {
...base,
type: GSDEventType.ToolUseSummary,
summary: msg.summary,
toolUseIds: msg.preceding_tool_use_ids,
} as GSDToolUseSummaryEvent;
}
private mapRateLimitMessage(
msg: SDKRateLimitEvent,
base: Omit<GSDEvent, 'type'>,
): GSDRateLimitEventType {
return {
...base,
type: GSDEventType.RateLimit,
status: msg.rate_limit_info.status,
resetsAt: msg.rate_limit_info.resetsAt,
utilization: msg.rate_limit_info.utilization,
} as GSDRateLimitEventType;
}
private mapStreamEvent(
msg: SDKPartialAssistantMessage,
base: Omit<GSDEvent, 'type'>,
): GSDStreamEvent {
return {
...base,
type: GSDEventType.StreamEvent,
event: msg.event,
} as GSDStreamEvent;
}
}

View File

@@ -1,95 +0,0 @@
/**
* Golden test helpers — run `gsd-tools.cjs` as a subprocess and capture JSON or raw stdout.
*
* Used by `golden.integration.test.ts` and `read-only-parity.integration.test.ts` to assert
* SDK `createRegistry()` output matches the legacy CJS CLI.
*/
import { execFile } from 'node:child_process';
import { readFile } from 'node:fs/promises';
import { isAbsolute, join } from 'node:path';
import { resolveGsdToolsPath } from '../gsd-tools.js';
const CAPTURE_TIMEOUT_MS = 120_000;
const MAX_BUFFER = 10 * 1024 * 1024;
function execGsdTools(
projectDir: string,
command: string,
args: string[],
): Promise<{ stdout: string; stderr: string }> {
const script = resolveGsdToolsPath(projectDir);
const fullArgs = [script, command, ...args];
return new Promise((resolve, reject) => {
execFile(
process.execPath,
fullArgs,
{
cwd: projectDir,
maxBuffer: MAX_BUFFER,
timeout: CAPTURE_TIMEOUT_MS,
env: { ...process.env },
},
(err, stdout, stderr) => {
if (err) {
const code = typeof err === 'object' && err && 'code' in err ? String((err as NodeJS.ErrnoException).code) : '';
const stderrStr = stderr?.toString() ?? '';
reject(
new Error(
`gsd-tools failed (exit ${code}): ${stderrStr || (err instanceof Error ? err.message : String(err))}`,
),
);
return;
}
resolve({ stdout: stdout?.toString() ?? '', stderr: stderr?.toString() ?? '' });
},
);
});
}
/** Same `@file:` indirection handling as {@link GSDTools} private parseOutput (cwd = projectDir). */
async function parseGsdToolsJson(raw: string, projectDir: string): Promise<unknown> {
const trimmed = raw.trim();
if (trimmed === '') {
return null;
}
let jsonStr = trimmed;
if (jsonStr.startsWith('@file:')) {
const rel = jsonStr.slice(6).trim();
const filePath = isAbsolute(rel) ? rel : join(projectDir, rel);
try {
jsonStr = await readFile(filePath, 'utf-8');
} catch (err) {
const reason = err instanceof Error ? err.message : String(err);
throw new Error(`Failed to read gsd-tools @file: indirection at "${filePath}": ${reason}`);
}
}
return JSON.parse(jsonStr);
}
/**
* Run `node gsd-tools.cjs <command> [...args]` in `projectDir` and parse stdout as JSON.
*/
export async function captureGsdToolsOutput(
command: string,
args: string[],
projectDir: string,
): Promise<unknown> {
const { stdout } = await execGsdTools(projectDir, command, args);
return parseGsdToolsJson(stdout, projectDir);
}
/**
* Run `node gsd-tools.cjs <command> [...args]` and return raw stdout (no JSON parse).
*/
export async function captureGsdToolsStdout(
command: string,
args: string[],
projectDir: string,
): Promise<string> {
const { stdout } = await execGsdTools(projectDir, command, args);
return stdout;
}

View File

@@ -1 +0,0 @@
{"slug":"my-phase"}

View File

@@ -1,3 +0,0 @@
{"type":"user","userType":"external","message":{"content":"profile sample message one"},"timestamp":1700000000000,"cwd":"/fixture/proj"}
{"type":"assistant","message":{"content":"ok"},"timestamp":1700000000001}
{"type":"user","userType":"external","message":{"content":"profile sample message two"},"timestamp":1700000000002,"cwd":"/fixture/proj"}

View File

@@ -1,26 +0,0 @@
---
phase: "01"
name: Golden Fixture
one-liner: From frontmatter YAML
key-files:
- sdk/src/foo.ts
key-decisions:
- "Auth model: use JWT bearer tokens"
- "Plain decision without colon split"
patterns-established:
- "Repository pattern for data access"
tech-stack:
added:
- vitest
- name: typescript
requirements-completed:
- REQ-GOLD-1
---
# Phase 01: Golden Fixture Summary
**Bold one-liner pulled from body when FM lacks one-liner**
## Section
More body.

View File

@@ -1,15 +0,0 @@
---
status: draft
---
# UAT
## Current Test
number: 1
name: Login flow
expected: |
User can sign in
## Other
Placeholder section after Current Test.

View File

@@ -1,30 +0,0 @@
/**
* Canonical commands exercised by `golden.integration.test.ts` (SDK dispatch vs
* `gsd-tools.cjs` where applicable). Update when adding `describe` blocks there.
*/
export const GOLDEN_INTEGRATION_MAIN_FILE_CANONICALS: readonly string[] = [
'config-get',
'config-set',
'current-timestamp',
'detect-custom-files',
'docs-init',
'find-phase',
'frontmatter.get',
'frontmatter.validate',
'generate-slug',
'init.execute-phase',
'init.plan-phase',
'init.quick',
'init.resume',
'init.verify-work',
'intel.update',
'progress.json',
'roadmap.analyze',
'state.sync',
'state.validate',
'template.select',
'validate.consistency',
'verify.phase-completeness',
'verify.plan-structure',
].sort((a, b) => a.localeCompare(b));

View File

@@ -1,17 +0,0 @@
/**
* Mutation canonicals with explicit subprocess JSON parity vs `gsd-tools.cjs`
* (see `mutation-subprocess.integration.test.ts` when present). Empty until those
* tests land; other mutations rely on `MUTATION_DEFERRED_REASON` in golden-policy.
*/
export const GOLDEN_MUTATION_SUBPROCESS_COVERED: readonly string[] = [
'state.update',
'state.patch',
'state.begin-phase',
'state.sync',
'phase.add',
'phase.add-batch',
'phase.insert',
'phases.clear',
'roadmap.update-plan-progress',
];

View File

@@ -1,8 +0,0 @@
import { describe, it, expect } from 'vitest';
import { verifyGoldenPolicyComplete } from './golden-policy.js';
describe('golden policy', () => {
it('every canonical registry command is integration-covered or excepted', () => {
expect(() => verifyGoldenPolicyComplete()).not.toThrow();
});
});

View File

@@ -1,120 +0,0 @@
/**
* Golden parity policy — every canonical registry command must be either:
* - Listed in `GOLDEN_PARITY_INTEGRATION_COVERED` (subprocess CJS check under `sdk/src/golden/*integration*.test.ts`), or
* - Documented in `GOLDEN_PARITY_EXCEPTIONS` with a stable rationale (mirrored in QUERY-HANDLERS.md § Golden registry coverage matrix).
*/
import { QUERY_MUTATION_COMMANDS } from '../query/index.js';
import { getCanonicalRegistryCommands } from './registry-canonical-commands.js';
import { GOLDEN_INTEGRATION_MAIN_FILE_CANONICALS } from './golden-integration-covered.js';
import { GOLDEN_MUTATION_SUBPROCESS_COVERED } from './golden-mutation-covered.js';
import { readOnlyGoldenCanonicals } from './read-only-golden-rows.js';
/** True if this canonical command participates in mutation event wiring (see QUERY_MUTATION_COMMANDS). */
export function isMutationCanonicalCmd(canonical: string): boolean {
const spaced = canonical.replace(/\./g, ' ');
for (const m of QUERY_MUTATION_COMMANDS) {
if (m === canonical || m === spaced) return true;
}
return false;
}
const MUTATION_DEFERRED_REASON =
'Listed in QUERY_MUTATION_COMMANDS — mutates `.planning/`, git, or profile files. Subprocess golden vs gsd-tools.cjs is covered where a tmp fixture or `--dry-run` exists in golden.integration.test.ts; otherwise handler parity lives in sdk/src/query/*-mutation.test.ts, commit.test.ts, phase-lifecycle.test.ts, workstream.test.ts, intel.test.ts, profile.test.ts, template.test.ts, docs-init.ts, or uat.test.ts as applicable.';
/** Registry commands with no `gsd-tools.cjs` analogue — cannot have subprocess JSON parity. */
const NO_CJS_SUBPROCESS_REASON: Record<string, string> = {
'phases.archive':
'No `gsd-tools.cjs` command for `phases archive` (SDK-only). Covered in sdk/src/query/phase-lifecycle.test.ts.',
'check.config-gates':
'SDK-only decision-routing query (`.planning/research/decision-routing-audit.md` §3.3). Covered in sdk/src/query/config-gates.test.ts.',
'check.phase-ready':
'SDK-only decision-routing query (audit §3.4). Covered in sdk/src/query/phase-ready.test.ts.',
'route.next-action':
'SDK-only decision-routing query (audit §3.1). Covered in sdk/src/query/route-next-action.test.ts.',
'check.auto-mode':
'SDK-only decision-routing query (audit §3.5). Covered in sdk/src/query/check-auto-mode.test.ts.',
'detect.phase-type':
'SDK-only decision-routing query (audit §3.6). Covered in sdk/src/query/detect-phase-type.test.ts.',
'check.completion':
'SDK-only decision-routing query (audit §3.7). Covered in sdk/src/query/check-completion.test.ts.',
'check.gates':
'SDK-only decision-routing query (audit §3.2). Covered in sdk/src/query/check-gates.test.ts.',
'check.verification-status':
'SDK-only decision-routing query (audit §3.8). Covered in sdk/src/query/check-verification-status.test.ts.',
'check.ship-ready':
'SDK-only decision-routing query (audit §3.9). Covered in sdk/src/query/check-ship-ready.test.ts.',
'phase.list-plans':
'SDK-only listing helper for agents (no `gsd-tools.cjs` mirror). Covered in sdk/src/query/phase-list-queries.test.ts.',
'phase.list-artifacts':
'SDK-only artifact enumeration (no CJS mirror). Covered in sdk/src/query/phase-list-queries.test.ts.',
'plan.task-structure':
'SDK-only structured plan parse (no CJS mirror). Covered in sdk/src/query/plan-task-structure.test.ts.',
'requirements.extract-from-plans':
'SDK-only requirements aggregation (no CJS mirror). Covered in sdk/src/query/requirements-extract-from-plans.test.ts.',
'commands':
'SDK-only registry introspection (no gsd-tools.cjs equivalent — the CJS layer has no self-describing verb). Covered in sdk/src/query/commands-list.test.ts. Closes #3121.',
'phase.mvp-mode':
'SDK-only MVP precedence resolver (CLI flag → roadmap → config → false). Centralizes the chain previously duplicated across plan-phase/execute-phase/verify-work/progress workflows. Covered in sdk/src/query/mvp.test.ts.',
'task.is-behavior-adding':
'SDK-only Behavior-Adding Task predicate for the MVP+TDD Gate (tdd=true + <behavior> block + non-test source files). Replaces prose-only specification in references/execute-mvp-tdd.md. Covered in sdk/src/query/mvp.test.ts.',
'user-story.validate':
'SDK-only User Story regex validator. Centralizes /^As a .+, I want to .+, so that .+\\.$/ previously hardcoded in verify-work workflow. Covered in sdk/src/query/mvp.test.ts.',
};
const READ_HANDLER_ONLY_REASON = (cmd: string) =>
`No ` +
'`toEqual` subprocess row yet for this read-only command — handler parity is covered in sdk/src/query/*.test.ts / decomposed-handlers.test.ts; add `captureGsdToolsOutput` + `registry.dispatch` in sdk/src/golden/ when JSON shapes are aligned (see QUERY-HANDLERS.md § Golden registry coverage matrix). Command: `' +
cmd +
'`.';
function buildIntegrationCoveredSet(): Set<string> {
return new Set<string>([
...GOLDEN_INTEGRATION_MAIN_FILE_CANONICALS,
...readOnlyGoldenCanonicals(),
...GOLDEN_MUTATION_SUBPROCESS_COVERED,
]);
}
/**
* Canonical commands with an explicit subprocess JSON check vs gsd-tools.cjs
* (golden.integration.test.ts + read-only-parity.integration.test.ts).
*/
export const GOLDEN_PARITY_INTEGRATION_COVERED = buildIntegrationCoveredSet();
export const GOLDEN_PARITY_EXCEPTIONS: Record<string, string> = buildGoldenParityExceptions();
function buildGoldenParityExceptions(): Record<string, string> {
const out: Record<string, string> = {};
for (const c of getCanonicalRegistryCommands()) {
if (GOLDEN_PARITY_INTEGRATION_COVERED.has(c)) continue;
if (Object.prototype.hasOwnProperty.call(NO_CJS_SUBPROCESS_REASON, c)) {
out[c] = NO_CJS_SUBPROCESS_REASON[c]!;
continue;
}
if (isMutationCanonicalCmd(c)) {
out[c] = MUTATION_DEFERRED_REASON;
} else {
out[c] = READ_HANDLER_ONLY_REASON(c);
}
}
return out;
}
export function verifyGoldenPolicyComplete(): void {
const canon = getCanonicalRegistryCommands();
const missingException: string[] = [];
for (const c of canon) {
if (GOLDEN_PARITY_INTEGRATION_COVERED.has(c)) continue;
if (!Object.prototype.hasOwnProperty.call(GOLDEN_PARITY_EXCEPTIONS, c)) missingException.push(c);
}
if (missingException.length) {
throw new Error(`Missing GOLDEN_PARITY_EXCEPTIONS entry for:\n${missingException.join('\n')}`);
}
const stale: string[] = [];
for (const c of GOLDEN_PARITY_INTEGRATION_COVERED) {
if (!canon.includes(c)) stale.push(c);
}
if (stale.length) {
throw new Error(`Stale GOLDEN_PARITY_INTEGRATION_COVERED entries:\n${stale.join('\n')}`);
}
}

File diff suppressed because it is too large Load Diff

View File

@@ -1,15 +0,0 @@
/**
* Normalize `init quick` payloads for golden parity: CJS runs in a subprocess with a
* different clock than the in-process SDK, so time-derived fields cannot match exactly.
*/
/** Keys derived from `Date` / `quick_id` generation (init.cjs cmdInitQuick). */
export const INIT_QUICK_VOLATILE_KEYS = ['quick_id', 'timestamp', 'branch_name', 'task_dir'] as const;
export function omitInitQuickVolatile(data: Record<string, unknown>): Record<string, unknown> {
const o = { ...data };
for (const k of INIT_QUICK_VOLATILE_KEYS) {
delete o[k];
}
return o;
}

View File

@@ -1,77 +0,0 @@
/**
* Read-only subprocess golden rows: SDK `registry.dispatch` vs `gsd-tools.cjs` JSON on stdout.
* Imported by `read-only-parity.integration.test.ts` and `golden-policy.ts` coverage accounting.
*/
export type JsonParityRow = {
canonical: string;
sdkArgs: string[];
cjs: string;
cjsArgs: string[];
};
/** Repo-relative fixtures (cwd = get-shit-done repo root). */
export const GOLDEN_PLAN = '.planning/phases/09-foundation-and-test-infrastructure/09-01-PLAN.md';
/**
* Strict `toEqual` JSON parity rows verified on this repository.
* (Expand as more handlers are aligned with `gsd-tools.cjs`.)
*/
export const READ_ONLY_JSON_PARITY_ROWS: JsonParityRow[] = [
{ canonical: 'resolve-model', sdkArgs: ['gsd-planner'], cjs: 'resolve-model', cjsArgs: ['gsd-planner'] },
{ canonical: 'phase-plan-index', sdkArgs: ['9'], cjs: 'phase-plan-index', cjsArgs: ['9'] },
{ canonical: 'roadmap.get-phase', sdkArgs: ['9'], cjs: 'roadmap', cjsArgs: ['get-phase', '9'] },
{ canonical: 'list.todos', sdkArgs: [], cjs: 'list-todos', cjsArgs: [] },
{ canonical: 'phase.next-decimal', sdkArgs: ['9'], cjs: 'phase', cjsArgs: ['next-decimal', '9'] },
{ canonical: 'phases.list', sdkArgs: [], cjs: 'phases', cjsArgs: ['list'] },
{ canonical: 'verify.summary', sdkArgs: [GOLDEN_PLAN], cjs: 'verify-summary', cjsArgs: [GOLDEN_PLAN] },
{ canonical: 'verify.path-exists', sdkArgs: ['.planning/STATE.md'], cjs: 'verify-path-exists', cjsArgs: ['.planning/STATE.md'] },
{ canonical: 'verify.artifacts', sdkArgs: [GOLDEN_PLAN], cjs: 'verify', cjsArgs: ['artifacts', GOLDEN_PLAN] },
{ canonical: 'websearch', sdkArgs: ['typescript', '--limit', '1'], cjs: 'websearch', cjsArgs: ['typescript', '--limit', '1'] },
{ canonical: 'workstream.get', sdkArgs: ['default'], cjs: 'workstream', cjsArgs: ['get', 'default'] },
{ canonical: 'workstream.list', sdkArgs: [], cjs: 'workstream', cjsArgs: ['list'] },
{ canonical: 'workstream.status', sdkArgs: ['default'], cjs: 'workstream', cjsArgs: ['status', 'default'] },
{ canonical: 'learnings.list', sdkArgs: [], cjs: 'learnings', cjsArgs: ['list'] },
{ canonical: 'intel.status', sdkArgs: [], cjs: 'intel', cjsArgs: ['status'] },
{ canonical: 'intel.diff', sdkArgs: [], cjs: 'intel', cjsArgs: ['diff'] },
{ canonical: 'intel.validate', sdkArgs: [], cjs: 'intel', cjsArgs: ['validate'] },
{ canonical: 'intel.query', sdkArgs: ['gsd'], cjs: 'intel', cjsArgs: ['query', 'gsd'] },
{
canonical: 'intel.extract-exports',
sdkArgs: ['sdk/src/query/utils.ts'],
cjs: 'intel',
cjsArgs: ['extract-exports', 'sdk/src/query/utils.ts'],
},
{ canonical: 'init.list-workspaces', sdkArgs: [], cjs: 'init', cjsArgs: ['list-workspaces'] },
{ canonical: 'agent-skills', sdkArgs: [], cjs: 'agent-skills', cjsArgs: [] },
{ canonical: 'scan-sessions', sdkArgs: ['--json'], cjs: 'scan-sessions', cjsArgs: ['--json'] },
{ canonical: 'stats.json', sdkArgs: [], cjs: 'stats', cjsArgs: ['json'] },
{ canonical: 'todo.match-phase', sdkArgs: ['9'], cjs: 'todo', cjsArgs: ['match-phase', '9'] },
{ canonical: 'verify.key-links', sdkArgs: [GOLDEN_PLAN], cjs: 'verify', cjsArgs: ['key-links', GOLDEN_PLAN] },
{ canonical: 'verify.schema-drift', sdkArgs: ['9'], cjs: 'verify', cjsArgs: ['schema-drift', '9'] },
{ canonical: 'state-snapshot', sdkArgs: [], cjs: 'state-snapshot', cjsArgs: [] },
{ canonical: 'history.digest', sdkArgs: [], cjs: 'history-digest', cjsArgs: [] },
{ canonical: 'audit-uat', sdkArgs: [], cjs: 'audit-uat', cjsArgs: [] },
{ canonical: 'skill-manifest', sdkArgs: [], cjs: 'skill-manifest', cjsArgs: [] },
{ canonical: 'validate.agents', sdkArgs: [], cjs: 'validate', cjsArgs: ['agents'] },
{
canonical: 'uat.render-checkpoint',
sdkArgs: ['--file', 'sdk/src/golden/fixtures/uat-render-checkpoint-sample.md'],
cjs: 'uat',
cjsArgs: ['render-checkpoint', '--file', 'sdk/src/golden/fixtures/uat-render-checkpoint-sample.md'],
},
];
/** Canonicals from JSON rows plus special-case subprocess tests in read-only-parity integration. */
export function readOnlyGoldenCanonicals(): Set<string> {
const s = new Set<string>(READ_ONLY_JSON_PARITY_ROWS.map((r) => r.canonical));
s.add('verify.commits');
s.add('config-path');
s.add('state.json');
s.add('state.load');
s.add('audit-open');
s.add('state.get');
s.add('summary.extract');
return s;
}

View File

@@ -1,133 +0,0 @@
/**
* Read-only subprocess golden checks (SDK vs gsd-tools.cjs JSON).
* Row data: `read-only-golden-rows.ts`. Policy: `golden-policy.ts`, `QUERY-HANDLERS.md`.
*/
import { describe, it, expect } from 'vitest';
import { captureGsdToolsOutput, captureGsdToolsStdout } from './capture.js';
import { createRegistry } from '../query/index.js';
import { resolve, dirname, normalize } from 'node:path';
import { fileURLToPath } from 'node:url';
import { execSync } from 'node:child_process';
import { READ_ONLY_JSON_PARITY_ROWS } from './read-only-golden-rows.js';
const STABLE_JSON_PARITY_ROWS = READ_ONLY_JSON_PARITY_ROWS.filter(
(row) => row.canonical !== 'scan-sessions' && row.canonical !== 'audit-uat',
);
const __dirname = dirname(fileURLToPath(import.meta.url));
const REPO_ROOT = resolve(__dirname, '..', '..', '..');
describe('Read-only golden parity (JSON toEqual)', () => {
it.each(STABLE_JSON_PARITY_ROWS)('$canonical matches gsd-tools.cjs JSON', async (row) => {
const gsdOutput = await captureGsdToolsOutput(row.cjs, row.cjsArgs, REPO_ROOT);
const registry = createRegistry();
const sdkResult = await registry.dispatch(row.canonical, row.sdkArgs, REPO_ROOT);
expect(sdkResult.data).toEqual(gsdOutput);
});
});
describe('config-path (plain stdout vs SDK { path })', () => {
it('SDK path matches gsd-tools.cjs plain-text stdout', async () => {
const out = await captureGsdToolsStdout('config-path', [], REPO_ROOT);
const registry = createRegistry();
const sdkResult = await registry.dispatch('config-path', [], REPO_ROOT);
const data = sdkResult.data as { path?: string };
expect(data.path).toBeDefined();
expect(normalize(data.path!.trim())).toBe(normalize(out.trim()));
});
});
describe('audit-open golden parity (excluding scanned_at)', () => {
it('SDK JSON matches gsd-tools.cjs except volatile scanned_at', async () => {
const gsdOutput = await captureGsdToolsOutput('audit-open', ['--json'], REPO_ROOT);
const registry = createRegistry();
const sdkResult = await registry.dispatch('audit-open', ['--json'], REPO_ROOT);
const strip = (d: unknown): Record<string, unknown> => {
const o = { ...(d as Record<string, unknown>) };
delete o.scanned_at;
delete o.has_scan_errors;
return o;
};
expect(strip(sdkResult.data)).toEqual(strip(gsdOutput));
});
});
describe('state.json golden parity (excluding last_updated)', () => {
it('SDK rebuilt frontmatter matches gsd-tools.cjs except volatile last_updated', async () => {
const gsdOutput = await captureGsdToolsOutput('state', ['json'], REPO_ROOT);
const registry = createRegistry();
const sdkResult = await registry.dispatch('state.json', [], REPO_ROOT);
const strip = (d: unknown): Record<string, unknown> => {
const o = { ...(d as Record<string, unknown>) };
delete o.last_updated;
return o;
};
expect(strip(sdkResult.data)).toEqual(strip(gsdOutput));
});
});
describe('summary.extract golden parity (with array-of-objects fix)', () => {
it('SDK JSON matches gsd-tools.cjs except for intentional array-of-objects parsing fix', async () => {
const gsdOutput = await captureGsdToolsOutput('summary-extract', ['sdk/src/golden/fixtures/summary-extract-sample.md'], REPO_ROOT);
const registry = createRegistry();
const sdkResult = await registry.dispatch('summary.extract', ['sdk/src/golden/fixtures/summary-extract-sample.md'], REPO_ROOT);
// The SDK correctly parses array-of-objects, whereas CJS parses them as strings.
// Patch the CJS output to reflect the CodeRabbit bugfix.
const patchedGsd = JSON.parse(JSON.stringify(gsdOutput));
if (patchedGsd.tech_added && Array.isArray(patchedGsd.tech_added)) {
patchedGsd.tech_added = patchedGsd.tech_added.map((t: any) =>
t === 'name: typescript' ? { name: 'typescript' } : t
);
}
expect(sdkResult.data).toEqual(patchedGsd);
});
});
describe('state.load golden parity', () => {
it('SDK load payload matches gsd-tools.cjs state load', async () => {
const gsdOutput = await captureGsdToolsOutput('state', ['load'], REPO_ROOT);
const registry = createRegistry();
const sdkResult = await registry.dispatch('state.load', [], REPO_ROOT);
expect(sdkResult.data).toEqual(gsdOutput);
});
});
describe('state.get golden parity', () => {
it('matches full STATE.md when no field (same as `state get` with no section)', async ({ skip }) => {
const registry = createRegistry();
const sdkResult = await registry.dispatch('state.get', [], REPO_ROOT);
// Repo may not have .planning/STATE.md; skip parity in that case.
if ((sdkResult.data as Record<string, unknown>)?.error === 'STATE.md not found') skip();
const gsdOutput = await captureGsdToolsOutput('state', ['get'], REPO_ROOT);
expect(sdkResult.data).toEqual(gsdOutput);
});
it('matches single frontmatter field when `state get <field>`', async ({ skip }) => {
const registry = createRegistry();
const sdkResult = await registry.dispatch('state.get', ['milestone'], REPO_ROOT);
if ((sdkResult.data as Record<string, unknown>)?.error === 'STATE.md not found') skip();
const gsdOutput = await captureGsdToolsOutput('state', ['get', 'milestone'], REPO_ROOT);
expect(sdkResult.data).toEqual(gsdOutput);
});
});
describe('verify.commits golden parity', () => {
it('SDK output matches gsd-tools.cjs for two SHAs', async () => {
const revs = execSync('git rev-list --max-count=2 HEAD', { cwd: REPO_ROOT, encoding: 'utf-8' })
.trim()
.split('\n')
.filter(Boolean);
if (revs.length < 2) {
throw new Error('verify.commits parity requires at least 2 commits in checkout history');
}
const b = revs[0];
const a = revs[1];
const gsdOutput = await captureGsdToolsOutput('verify', ['commits', a, b], REPO_ROOT);
const registry = createRegistry();
const sdkResult = await registry.dispatch('verify.commits', [a, b], REPO_ROOT);
expect(sdkResult.data).toEqual(gsdOutput);
});
});

View File

@@ -1,31 +0,0 @@
/**
* Canonical registry command strings for golden parity — one primary name per unique
* native handler (dedupes dotted vs space-delimited aliases on the same function).
*/
import { createRegistry } from '../query/index.js';
import type { QueryHandler } from '../query/utils.js';
export function getCanonicalRegistryCommands(): string[] {
const registry = createRegistry();
const byHandler = new Map<QueryHandler, string[]>();
for (const cmd of registry.commands()) {
const h = registry.getHandler(cmd);
if (!h) continue;
const list = byHandler.get(h) ?? [];
list.push(cmd);
byHandler.set(h, list);
}
const out: string[] = [];
for (const cmds of byHandler.values()) {
cmds.sort((a, b) => a.localeCompare(b));
const dotted = cmds.find((c) => c.includes('.'));
if (dotted) {
out.push(dotted);
continue;
}
const kebab = cmds.find((c) => c.includes('-'));
out.push(kebab ?? cmds[0]!);
}
return out.sort((a, b) => a.localeCompare(b));
}

View File

@@ -1,21 +0,0 @@
import { describe, expect, it } from 'vitest';
import { GSDToolsError } from './gsd-tools-error.js';
describe('GSDToolsError constructors', () => {
it('builds timeout-classified errors', () => {
const err = GSDToolsError.timeout('timeout', 'state', ['load'], '', 1000);
expect(err.classification).toEqual({ kind: 'timeout', timeoutMs: 1000 });
expect(err.exitCode).toBeNull();
});
it('builds failure-classified errors', () => {
const err = GSDToolsError.failure('boom', 'state', ['load'], 1);
expect(err.classification).toEqual({ kind: 'failure' });
expect(err.exitCode).toBe(1);
});
it('defaults direct constructor to failure classification', () => {
const err = new GSDToolsError('boom', 'state', ['load'], 1, 'stderr');
expect(err.classification).toEqual({ kind: 'failure' });
});
});

View File

@@ -1,65 +0,0 @@
export interface GSDToolsErrorClassification {
kind: 'timeout' | 'failure';
timeoutMs?: number;
}
function timeoutClassification(timeoutMs?: number): GSDToolsErrorClassification {
return timeoutMs === undefined ? { kind: 'timeout' } : { kind: 'timeout', timeoutMs };
}
function failureClassification(): GSDToolsErrorClassification {
return { kind: 'failure' };
}
export class GSDToolsError extends Error {
constructor(
message: string,
public readonly command: string,
public readonly args: string[],
public readonly exitCode: number | null,
public readonly stderr: string,
options?: { cause?: unknown; classification?: GSDToolsErrorClassification },
) {
super(message, options);
this.name = 'GSDToolsError';
this.classification = options?.classification ?? failureClassification();
}
static timeout(
message: string,
command: string,
args: string[],
stderr = '',
timeoutMs?: number,
options?: { cause?: unknown; exitCode?: number | null },
): GSDToolsError {
return new GSDToolsError(
message,
command,
args,
options?.exitCode ?? null,
stderr,
{ cause: options?.cause, classification: timeoutClassification(timeoutMs) },
);
}
static failure(
message: string,
command: string,
args: string[],
exitCode: number | null,
stderr = '',
options?: { cause?: unknown },
): GSDToolsError {
return new GSDToolsError(
message,
command,
args,
exitCode,
stderr,
{ cause: options?.cause, classification: failureClassification() },
);
}
public readonly classification: GSDToolsErrorClassification;
}

View File

@@ -1,472 +0,0 @@
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
import { GSDTools, GSDToolsError, resolveGsdToolsPath } from './gsd-tools.js';
import { setTransportPolicy, clearTransportPolicy } from './gsd-transport-policy.js';
import { mkdir, writeFile, rm } from 'node:fs/promises';
import { existsSync } from 'node:fs';
import { join } from 'node:path';
import { tmpdir, homedir } from 'node:os';
import { fileURLToPath } from 'node:url';
const BUNDLED_GSD_TOOLS_PATH = fileURLToPath(
new URL('../../get-shit-done/bin/gsd-tools.cjs', import.meta.url),
);
describe('GSDTools', () => {
let tmpDir: string;
let fixtureDir: string;
beforeEach(async () => {
tmpDir = join(tmpdir(), `gsd-tools-test-${Date.now()}-${Math.random().toString(36).slice(2)}`);
fixtureDir = join(tmpDir, 'fixtures');
await mkdir(fixtureDir, { recursive: true });
await mkdir(join(tmpDir, '.planning'), { recursive: true });
});
afterEach(async () => {
clearTransportPolicy();
await rm(tmpDir, { recursive: true, force: true });
});
// ─── Helper: create a Node script that outputs something ────────────────
async function createScript(name: string, code: string): Promise<string> {
const scriptPath = join(fixtureDir, name);
await writeFile(scriptPath, code, { mode: 0o755 });
return scriptPath;
}
// ─── exec() tests ──────────────────────────────────────────────────────
describe('exec()', () => {
it('parses valid JSON output', async () => {
// Create a script that ignores args and outputs JSON
const scriptPath = await createScript(
'echo-json.cjs',
`process.stdout.write(JSON.stringify({ status: "ok", count: 42 }));`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.exec('state', ['load']);
expect(result).toEqual({ status: 'ok', count: 42 });
});
it('handles @file: prefix by reading referenced file', async () => {
// Write a large JSON result to a file
const resultFile = join(fixtureDir, 'big-result.json');
const bigData = { items: Array.from({ length: 100 }, (_, i) => ({ id: i })) };
await writeFile(resultFile, JSON.stringify(bigData));
// Script outputs @file: prefix
const scriptPath = await createScript(
'file-ref.cjs',
`process.stdout.write('@file:${resultFile.replace(/\\/g, '\\\\')}');`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.exec('state', ['load']);
expect(result).toEqual(bigData);
});
it('returns null for empty stdout', async () => {
const scriptPath = await createScript(
'empty-output.cjs',
`// outputs nothing`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.exec('state', ['load']);
expect(result).toBeNull();
});
it('throws GSDToolsError on non-zero exit code', async () => {
const scriptPath = await createScript(
'fail.cjs',
`process.stderr.write('something went wrong\\n'); process.exit(1);`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
try {
await tools.exec('state', ['load']);
expect.fail('Should have thrown');
} catch (err) {
expect(err).toBeInstanceOf(GSDToolsError);
const gsdErr = err as GSDToolsError;
expect(gsdErr.command).toBe('state');
expect(gsdErr.args).toEqual(['load']);
expect(gsdErr.stderr).toContain('something went wrong');
expect(gsdErr.exitCode).toBeGreaterThan(0);
}
});
it('throws GSDToolsError with context when gsd-tools.cjs not found', async () => {
const tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: '/nonexistent/path/gsd-tools.cjs',
preferNativeQuery: false,
});
await expect(tools.exec('state', ['load'])).rejects.toThrow(GSDToolsError);
});
it('throws parse error when stdout is non-JSON', async () => {
const scriptPath = await createScript(
'bad-json.cjs',
`process.stdout.write('Not JSON at all');`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
try {
await tools.exec('state', ['load']);
expect.fail('Should have thrown');
} catch (err) {
expect(err).toBeInstanceOf(GSDToolsError);
const gsdErr = err as GSDToolsError;
expect(gsdErr.message).toContain('Failed to parse');
expect(gsdErr.message).toContain('Not JSON at all');
}
});
it('throws when @file: points to nonexistent file', async () => {
const scriptPath = await createScript(
'bad-file-ref.cjs',
`process.stdout.write('@file:/tmp/does-not-exist-${Date.now()}.json');`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
await expect(tools.exec('state', ['load'])).rejects.toThrow(GSDToolsError);
});
it('handles timeout by killing child process', async () => {
const scriptPath = await createScript(
'hang.cjs',
`setTimeout(() => {}, 60000); // hang for 60s`,
);
const tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: scriptPath,
timeoutMs: 500,
preferNativeQuery: false,
});
try {
await tools.exec('state', ['load']);
expect.fail('Should have thrown');
} catch (err) {
expect(err).toBeInstanceOf(GSDToolsError);
const gsdErr = err as GSDToolsError;
expect(gsdErr.message).toContain('timed out');
}
}, 10_000);
it('uses subprocess fallback when native handler throws and policy allows fallback', async () => {
const scriptPath = await createScript(
'fallback-ok.cjs',
`process.stdout.write(JSON.stringify({ from: 'subprocess-fallback' }));`,
);
const tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: scriptPath,
allowFallbackToSubprocess: true,
});
setTransportPolicy('verify.path-exists', { allowFallbackToSubprocess: true });
const result = await tools.exec('verify.path-exists', []);
expect(result).toEqual({ from: 'subprocess-fallback' });
});
it('fails fast in strictSdk mode when command has no native adapter', async () => {
const scriptPath = await createScript(
'strict-should-not-run.cjs',
`process.stdout.write(JSON.stringify({ should: 'not-run' }));`,
);
const tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: scriptPath,
strictSdk: true,
allowFallbackToSubprocess: true,
});
await expect(tools.exec('graphify', [])).rejects.toThrow(
"Strict SDK mode: command 'graphify' has no native adapter",
);
});
it('preserves GSDToolsError contract when native handler throws and fallback disabled', async () => {
const scriptPath = await createScript(
'should-not-run.cjs',
`process.stdout.write(JSON.stringify({ should: 'not-run' }));`,
);
const tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: scriptPath,
allowFallbackToSubprocess: false,
});
setTransportPolicy('verify.path-exists', { allowFallbackToSubprocess: false });
try {
await tools.exec('verify.path-exists', []);
expect.fail('Should have thrown');
} catch (err) {
expect(err).toBeInstanceOf(GSDToolsError);
const gsdErr = err as GSDToolsError;
expect(gsdErr.command).toBe('verify.path-exists');
expect(gsdErr.args).toEqual([]);
expect(gsdErr.stderr).toBe('');
expect(typeof gsdErr.exitCode === 'number').toBe(true);
}
});
});
// ─── Typed method tests ────────────────────────────────────────────────
describe('typed methods', () => {
it('stateLoad() calls exec with correct args', async () => {
const scriptPath = await createScript(
'state-load.cjs',
`
const args = process.argv.slice(2);
// Script receives: state load (no --raw when policy is json)
if (args[0] === 'state' && args[1] === 'load') {
process.stdout.write(JSON.stringify({ phase: '3', status: 'executing' }));
} else {
process.stderr.write('unexpected args: ' + args.join(' '));
process.exit(1);
}
`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.stateLoad();
expect(result).toEqual({ phase: '3', status: 'executing' });
});
it('commit() passes message and optional files', async () => {
const scriptPath = await createScript(
'commit.cjs',
`
const args = process.argv.slice(2);
// commit <msg> --files f1 f2 --raw — returns a git SHA
process.stdout.write('f89ae07');
`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.commit('test message', ['file1.md', 'file2.md']);
expect(result).toBe('f89ae07');
});
it('roadmapAnalyze() calls roadmap analyze', async () => {
const scriptPath = await createScript(
'roadmap.cjs',
`
const args = process.argv.slice(2);
if (args[0] === 'roadmap' && args[1] === 'analyze') {
process.stdout.write(JSON.stringify({ phases: [] }));
} else {
process.exit(1);
}
`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.roadmapAnalyze();
expect(result).toEqual({ phases: [] });
});
it('verifySummary() passes path argument', async () => {
const scriptPath = await createScript(
'verify.cjs',
`
const args = process.argv.slice(2);
if (args[0] === 'verify-summary' && args[1] === '/path/to/SUMMARY.md') {
process.stdout.write('passed');
} else {
process.exit(1);
}
`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.verifySummary('/path/to/SUMMARY.md');
expect(result).toBe('passed');
});
});
// ─── Integration-style test ────────────────────────────────────────────
describe('integration', () => {
it('handles large JSON output (>100KB)', async () => {
const largeArray = Array.from({ length: 5000 }, (_, i) => ({
id: i,
name: `item-${i}`,
data: 'x'.repeat(20),
}));
const largeJson = JSON.stringify(largeArray);
const scriptPath = await createScript(
'large-output.cjs',
`process.stdout.write(${JSON.stringify(largeJson)});`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.exec('state', ['load']);
expect(Array.isArray(result)).toBe(true);
expect((result as unknown[]).length).toBe(5000);
});
});
// ─── initNewProject() tests ────────────────────────────────────────────
describe('initNewProject()', () => {
it('calls init new-project and returns typed result', async () => {
const mockResult = {
researcher_model: 'claude-sonnet-4-6',
synthesizer_model: 'claude-sonnet-4-6',
roadmapper_model: 'claude-sonnet-4-6',
commit_docs: true,
project_exists: false,
has_codebase_map: false,
planning_exists: false,
has_existing_code: false,
has_package_file: false,
is_brownfield: false,
needs_codebase_map: false,
has_git: true,
brave_search_available: false,
firecrawl_available: false,
exa_search_available: false,
project_path: '.planning/PROJECT.md',
project_root: '/tmp/test',
};
const scriptPath = await createScript(
'init-new-project.cjs',
`
const args = process.argv.slice(2);
if (args[0] === 'init' && args[1] === 'new-project') {
process.stdout.write(JSON.stringify(${JSON.stringify(mockResult)}));
} else {
process.stderr.write('unexpected args: ' + args.join(' '));
process.exit(1);
}
`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.initNewProject();
expect(result.researcher_model).toBe('claude-sonnet-4-6');
expect(result.project_exists).toBe(false);
expect(result.has_git).toBe(true);
expect(result.is_brownfield).toBe(false);
expect(result.project_path).toBe('.planning/PROJECT.md');
});
it('propagates errors from gsd-tools', async () => {
const scriptPath = await createScript(
'init-fail.cjs',
`process.stderr.write('init failed\\n'); process.exit(1);`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
await expect(tools.initNewProject()).rejects.toThrow(GSDToolsError);
});
});
// ─── resolveGsdToolsPath() tests ────────────────────────────────────────
describe('resolveGsdToolsPath()', () => {
it('prefers bundled gsd-tools over project .claude when the bundled file exists', async () => {
const localBinDir = join(tmpDir, '.claude', 'get-shit-done', 'bin');
await mkdir(localBinDir, { recursive: true });
await writeFile(join(localBinDir, 'gsd-tools.cjs'), '// stub');
const result = resolveGsdToolsPath(tmpDir);
if (existsSync(BUNDLED_GSD_TOOLS_PATH)) {
expect(result).toBe(BUNDLED_GSD_TOOLS_PATH);
} else {
expect(result).toBe(join(localBinDir, 'gsd-tools.cjs'));
}
});
it('falls back to bundled repo path when repo-local does not exist', () => {
const result = resolveGsdToolsPath(tmpDir);
const expected = existsSync(BUNDLED_GSD_TOOLS_PATH)
? BUNDLED_GSD_TOOLS_PATH
: join(homedir(), '.claude', 'get-shit-done', 'bin', 'gsd-tools.cjs');
expect(result).toBe(expected);
});
it('uses explicit gsdToolsPath when provided (overrides bundled / .claude resolution)', async () => {
const localBinDir = join(tmpDir, '.claude', 'get-shit-done', 'bin');
await mkdir(localBinDir, { recursive: true });
const scriptPath = join(localBinDir, 'gsd-tools.cjs');
await writeFile(
scriptPath,
`process.stdout.write(JSON.stringify({ source: "local" }));`,
{ mode: 0o755 },
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.exec('test', []);
expect(result).toEqual({ source: 'local' });
});
});
// ─── configSet() tests ─────────────────────────────────────────────────
describe('configSet()', () => {
it('calls config-set with key and value args', async () => {
const scriptPath = await createScript(
'config-set.cjs',
`
const args = process.argv.slice(2);
if (args[0] === 'config-set' && args[1] === 'workflow.auto_advance' && args[2] === 'true' && args.includes('--raw')) {
process.stdout.write('workflow.auto_advance=true');
} else {
process.stderr.write('unexpected args: ' + args.join(' '));
process.exit(1);
}
`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.configSet('workflow.auto_advance', 'true');
expect(result).toBe('workflow.auto_advance=true');
});
it('passes string values without coercion', async () => {
const scriptPath = await createScript(
'config-set-str.cjs',
`
const args = process.argv.slice(2);
// config-set mode yolo --raw
process.stdout.write(args[1] + '=' + args[2]);
`,
);
const tools = new GSDTools({ projectDir: tmpDir, gsdToolsPath: scriptPath, preferNativeQuery: false });
const result = await tools.configSet('mode', 'yolo');
expect(result).toBe('mode=yolo');
});
});
});

View File

@@ -1,237 +0,0 @@
/**
* GSD Tools Bridge — programmatic access to GSD planning operations.
*
* By default routes commands through the SDK **query registry** (same handlers as
* `gsd-sdk query`) so `PhaseRunner`, `InitRunner`, and `GSD` share contracts with
* the typed CLI. Runner hot-path helpers (`initPhaseOp`, `phasePlanIndex`,
* `phaseComplete`, `initNewProject`, `configSet`, `commit`) call
* `registry.dispatch()` with canonical keys when native query is active, avoiding
* repeated argv resolution. When a workstream is set, dispatches to `gsd-tools.cjs` so
* workstream env stays aligned with CJS.
*/
import type { InitNewProjectInfo, PhaseOpInfo, PhasePlanIndex, RoadmapAnalysis } from './types.js';
import type { GSDEventStream } from './event-stream.js';
import { toToolsErrorFromUnknown } from './query-tools-error-factory.js';
import { GSDToolsError } from './gsd-tools-error.js';
import type { QueryCommandResolution } from './query/query-command-resolution-strategy.js';
import { resolveGsdToolsPath } from './query-gsd-tools-path.js';
import { createGSDToolsRuntime } from './query-gsd-tools-runtime.js';
import { QueryCommandExecutor } from './query-command-executor.js';
import { QueryHotpathMethods } from './query-hotpath-methods.js';
import { QueryRuntimeBridge, type RuntimeBridgeOptions } from './query-runtime-bridge.js';
export { GSDToolsError } from './gsd-tools-error.js';
// ─── GSDTools class ──────────────────────────────────────────────────────────
const DEFAULT_TIMEOUT_MS = 30_000;
export class GSDTools {
private readonly projectDir: string;
private readonly gsdToolsPath: string;
private readonly timeoutMs: number;
private readonly workstream?: string;
private readonly bridge: QueryRuntimeBridge;
private readonly preferNativeQuery: boolean;
private readonly commandExecutor: QueryCommandExecutor;
private readonly hotpathMethods: QueryHotpathMethods;
constructor(opts: {
projectDir: string;
gsdToolsPath?: string;
timeoutMs?: number;
workstream?: string;
/** When set, mutation handlers emit the same events as `gsd-sdk query`. */
eventStream?: GSDEventStream;
/** Correlation id for mutation events when `eventStream` is set. */
sessionId?: string;
/**
* When true (default), route known commands through the SDK query registry.
* Set false in tests that substitute a mock `gsdToolsPath` script.
*/
preferNativeQuery?: boolean;
/** When true, fail if a command has no native registry adapter. */
strictSdk?: boolean;
/** Explicit subprocess bridge policy. Default false for SDK-native mode. */
allowFallbackToSubprocess?: boolean;
/** Structured runtime bridge dispatch observability callback. */
onDispatchEvent?: RuntimeBridgeOptions['onDispatchEvent'];
}) {
this.projectDir = opts.projectDir;
this.gsdToolsPath =
opts.gsdToolsPath ?? resolveGsdToolsPath(opts.projectDir);
this.timeoutMs = opts.timeoutMs ?? DEFAULT_TIMEOUT_MS;
this.workstream = opts.workstream;
this.preferNativeQuery = opts.preferNativeQuery ?? true;
const runtime = createGSDToolsRuntime({
projectDir: this.projectDir,
gsdToolsPath: this.gsdToolsPath,
timeoutMs: this.timeoutMs,
workstream: this.workstream,
eventStream: opts.eventStream,
sessionId: opts.sessionId,
shouldUseNativeQuery: () => this.shouldUseNativeQuery(),
execJsonFallback: (legacyCommand, legacyArgs) => this.exec(legacyCommand, legacyArgs),
execRawFallback: (legacyCommand, legacyArgs) => this.execRaw(legacyCommand, legacyArgs),
strictSdk: opts.strictSdk,
allowFallbackToSubprocess: opts.allowFallbackToSubprocess,
onDispatchEvent: opts.onDispatchEvent,
});
this.bridge = runtime.bridge;
this.commandExecutor = new QueryCommandExecutor({
nativeMatch: (command, args) => this.nativeMatch(command, args),
execute: async (input) => this.bridge.execute({
legacyCommand: input.legacyCommand,
legacyArgs: input.legacyArgs,
registryCommand: input.registryCommand,
registryArgs: input.registryArgs,
mode: input.mode,
projectDir: this.projectDir,
workstream: this.workstream,
}),
});
this.hotpathMethods = new QueryHotpathMethods({
dispatchNativeHotpath: (legacyCommand, legacyArgs, registryCommand, registryArgs, mode) =>
this.dispatchNativeHotpath(legacyCommand, legacyArgs, registryCommand, registryArgs, mode),
});
}
private shouldUseNativeQuery(): boolean {
return this.preferNativeQuery && !this.workstream;
}
private nativeMatch(command: string, args: string[]): QueryCommandResolution | null {
return this.bridge.resolve(command, args);
}
private async dispatchNativeHotpath(
legacyCommand: string,
legacyArgs: string[],
registryCommand: string,
registryArgs: string[],
mode: 'json' | 'raw',
): Promise<unknown> {
return this.executeWithToolsError(legacyCommand, legacyArgs, () =>
this.bridge.dispatchHotpath(
legacyCommand,
legacyArgs,
registryCommand,
registryArgs,
mode,
));
}
private async executeWithToolsError<T>(command: string, args: string[], work: () => Promise<T>): Promise<T> {
try {
return await work();
} catch (err) {
if (err instanceof GSDToolsError) throw err;
throw toToolsErrorFromUnknown(command, args, err);
}
}
// ─── Core exec ───────────────────────────────────────────────────────────
/**
* Execute a gsd-tools command and return parsed JSON output.
* Handles the `@file:` prefix pattern for large results.
*/
async exec(command: string, args: string[] = []): Promise<unknown> {
return this.executeWithToolsError(command, args, () => this.commandExecutor.exec(command, args, 'json'));
}
// ─── Raw exec (no JSON parsing) ───────────────────────────────────────
/**
* Execute a gsd-tools command and return raw stdout without JSON parsing.
* Use for commands like `config-set` that return plain text, not JSON.
*/
async execRaw(command: string, args: string[] = []): Promise<string> {
return this.executeWithToolsError(command, args, async () => {
const out = await this.commandExecutor.exec(command, args, 'raw');
return typeof out === 'string' ? out : String(out ?? '');
});
}
// ─── Typed convenience methods ─────────────────────────────────────────
async stateLoad(): Promise<unknown> {
return this.exec('state', ['load']);
}
async roadmapAnalyze(): Promise<RoadmapAnalysis> {
return this.exec('roadmap', ['analyze']) as Promise<RoadmapAnalysis>;
}
async phaseComplete(phase: string): Promise<string> {
return this.hotpathMethods.phaseComplete(phase);
}
async commit(message: string, files?: string[]): Promise<string> {
return this.hotpathMethods.commit(message, files);
}
async verifySummary(path: string): Promise<string> {
return this.execRaw('verify-summary', [path]);
}
async initExecutePhase(phase: string): Promise<string> {
return this.execRaw('state', ['begin-phase', '--phase', phase]);
}
/**
* Query phase state from gsd-tools.cjs `init phase-op`.
* Returns a typed PhaseOpInfo describing what exists on disk for this phase.
*/
async initPhaseOp(phaseNumber: string): Promise<PhaseOpInfo> {
return this.hotpathMethods.initPhaseOp(phaseNumber);
}
/**
* Get a config value via the `config-get` surface (CJS and registry use the same key path).
*/
async configGet(key: string): Promise<string | null> {
return this.hotpathMethods.configGet(key);
}
/**
* Begin phase state tracking in gsd-tools.cjs.
*/
async stateBeginPhase(phaseNumber: string): Promise<string> {
return this.execRaw('state', ['begin-phase', '--phase', phaseNumber]);
}
/**
* Get the plan index for a phase, grouping plans into dependency waves.
* Returns typed PhasePlanIndex with wave assignments and completion status.
*/
async phasePlanIndex(phaseNumber: string): Promise<PhasePlanIndex> {
return this.hotpathMethods.phasePlanIndex(phaseNumber);
}
/**
* Query new-project init state from gsd-tools.cjs `init new-project`.
* Returns project metadata, model configs, brownfield detection, etc.
*/
async initNewProject(): Promise<InitNewProjectInfo> {
return this.hotpathMethods.initNewProject();
}
/**
* Set a config value via gsd-tools.cjs `config-set`.
* Handles type coercion (booleans, numbers, JSON) on the gsd-tools side.
* Note: config-set returns `key=value` text, not JSON, so we use execRaw.
*/
async configSet(key: string, value: string): Promise<string> {
return this.hotpathMethods.configSet(key, value);
}
}
export { resolveGsdToolsPath } from './query-gsd-tools-path.js';

View File

@@ -1,34 +0,0 @@
import { describe, it, expect, afterEach } from 'vitest';
import { resolveTransportPolicy, setTransportPolicy, clearTransportPolicy } from './gsd-transport-policy.js';
describe('gsd-transport-policy', () => {
afterEach(() => {
clearTransportPolicy();
});
it('uses legacy-safe defaults for unknown command', () => {
const policy = resolveTransportPolicy('unknown-cmd');
expect(policy.preferNative).toBe(true);
expect(policy.allowFallbackToSubprocess).toBe(true);
expect(policy.outputMode).toBe('json');
});
it('applies built-in raw output override', () => {
const policy = resolveTransportPolicy('config-set');
expect(policy.outputMode).toBe('raw');
expect(policy.allowFallbackToSubprocess).toBe(true);
});
it('applies verify-summary alias raw overrides', () => {
expect(resolveTransportPolicy('verify-summary').outputMode).toBe('raw');
expect(resolveTransportPolicy('verify.summary').outputMode).toBe('raw');
expect(resolveTransportPolicy('verify summary').outputMode).toBe('raw');
});
it('supports per-command override updates', () => {
setTransportPolicy('state', { allowFallbackToSubprocess: false, outputMode: 'raw' });
const policy = resolveTransportPolicy('state');
expect(policy.allowFallbackToSubprocess).toBe(false);
expect(policy.outputMode).toBe('raw');
});
});

View File

@@ -1,48 +0,0 @@
import { TRANSPORT_RAW_COMMANDS } from './query/query-policy-capability.js';
export type TransportMode = 'json' | 'raw';
export interface TransportPolicy {
preferNative: boolean;
allowFallbackToSubprocess: boolean;
outputMode: TransportMode;
}
const DEFAULT_POLICY: TransportPolicy = {
preferNative: true,
allowFallbackToSubprocess: true,
outputMode: 'json',
};
const BUILTIN_COMMAND_POLICY: Record<string, Partial<TransportPolicy>> = Object.fromEntries(
TRANSPORT_RAW_COMMANDS.map((command) => [command, { outputMode: 'raw' as const }]),
);
const COMMAND_POLICY_OVERRIDES: Record<string, Partial<TransportPolicy>> = {};
export function resolveTransportPolicy(command: string): TransportPolicy {
const override = {
...(BUILTIN_COMMAND_POLICY[command] ?? {}),
...(COMMAND_POLICY_OVERRIDES[command] ?? {}),
};
return {
preferNative: override.preferNative ?? DEFAULT_POLICY.preferNative,
allowFallbackToSubprocess:
override.allowFallbackToSubprocess ?? DEFAULT_POLICY.allowFallbackToSubprocess,
outputMode: override.outputMode ?? DEFAULT_POLICY.outputMode,
};
}
export function setTransportPolicy(command: string, override: Partial<TransportPolicy>): void {
COMMAND_POLICY_OVERRIDES[command] = { ...(COMMAND_POLICY_OVERRIDES[command] ?? {}), ...override };
}
export function clearTransportPolicy(command?: string): void {
if (command) {
delete COMMAND_POLICY_OVERRIDES[command];
return;
}
for (const key of Object.keys(COMMAND_POLICY_OVERRIDES)) {
delete COMMAND_POLICY_OVERRIDES[key];
}
}

View File

@@ -1,299 +0,0 @@
import { describe, it, expect, vi } from 'vitest';
import { GSDToolsError } from './gsd-tools-error.js';
import { QueryRegistry } from './query/registry.js';
import { GSDTransport } from './gsd-transport.js';
describe('GSDTransport', () => {
it('uses native adapter when command registered and policy prefers native', async () => {
const registry = new QueryRegistry();
registry.register('state.load', async () => ({ data: { ok: true } }));
const adapters = {
dispatchNative: vi.fn(async () => ({ data: { ok: true } })),
execSubprocessJson: vi.fn(async () => ({ ok: false })),
execSubprocessRaw: vi.fn(async () => 'subprocess'),
};
const transport = new GSDTransport(registry, adapters);
const result = await transport.run({
legacyCommand: 'state',
legacyArgs: ['load'],
registryCommand: 'state.load',
registryArgs: [],
mode: 'json',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
});
expect(result).toEqual({ ok: true });
expect(adapters.dispatchNative).toHaveBeenCalledOnce();
expect(adapters.execSubprocessJson).not.toHaveBeenCalled();
});
it('falls back to subprocess when native throws and policy allows fallback', async () => {
const registry = new QueryRegistry();
registry.register('state.load', async () => ({ data: { ok: true } }));
const adapters = {
dispatchNative: vi.fn(async () => {
throw new Error('native failed');
}),
execSubprocessJson: vi.fn(async () => ({ ok: 'fallback' })),
execSubprocessRaw: vi.fn(async () => 'fallback-raw'),
};
const transport = new GSDTransport(registry, adapters);
const result = await transport.run({
legacyCommand: 'state',
legacyArgs: ['load'],
registryCommand: 'state.load',
registryArgs: [],
mode: 'json',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
});
expect(result).toEqual({ ok: 'fallback' });
expect(adapters.dispatchNative).toHaveBeenCalledOnce();
expect(adapters.execSubprocessJson).toHaveBeenCalledOnce();
});
it('hard-fails when native throws and fallback disabled', async () => {
const registry = new QueryRegistry();
registry.register('state.load', async () => ({ data: { ok: true } }));
const adapters = {
dispatchNative: vi.fn(async () => {
throw new Error('native failed');
}),
execSubprocessJson: vi.fn(async () => ({ ok: 'fallback' })),
execSubprocessRaw: vi.fn(async () => 'fallback-raw'),
};
const transport = new GSDTransport(registry, adapters);
await expect(transport.run({
legacyCommand: 'state',
legacyArgs: ['load'],
registryCommand: 'state.load',
registryArgs: [],
mode: 'json',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: false,
})).rejects.toThrow('native failed');
expect(adapters.execSubprocessJson).not.toHaveBeenCalled();
});
it('does not fallback after timeout-like native error', async () => {
const registry = new QueryRegistry();
registry.register('state.load', async () => ({ data: { ok: true } }));
const adapters = {
dispatchNative: vi.fn(async () => {
throw new Error('gsd-tools timed out after 500ms: state load');
}),
execSubprocessJson: vi.fn(async () => ({ ok: 'fallback' })),
execSubprocessRaw: vi.fn(async () => 'fallback-raw'),
};
const transport = new GSDTransport(registry, adapters);
await expect(transport.run({
legacyCommand: 'state',
legacyArgs: ['load'],
registryCommand: 'state.load',
registryArgs: [],
mode: 'json',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
})).rejects.toThrow('timed out after');
expect(adapters.execSubprocessJson).not.toHaveBeenCalled();
});
it('does not fallback after typed timeout native error', async () => {
const registry = new QueryRegistry();
registry.register('state.load', async () => ({ data: { ok: true } }));
const timeoutError = GSDToolsError.timeout('native timed out', 'state', ['load'], '', 500);
const adapters = {
dispatchNative: vi.fn(async () => {
throw timeoutError;
}),
execSubprocessJson: vi.fn(async () => ({ ok: 'fallback' })),
execSubprocessRaw: vi.fn(async () => 'fallback-raw'),
};
const transport = new GSDTransport(registry, adapters);
await expect(transport.run({
legacyCommand: 'state',
legacyArgs: ['load'],
registryCommand: 'state.load',
registryArgs: [],
mode: 'json',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
})).rejects.toBe(timeoutError);
expect(adapters.execSubprocessJson).not.toHaveBeenCalled();
});
it('formats native raw output via formatNativeRaw when provided', async () => {
const registry = new QueryRegistry();
registry.register('commit', async () => ({ data: { hash: 'abc123' } }));
const adapters = {
dispatchNative: vi.fn(async () => ({ data: { hash: 'abc123' } })),
execSubprocessJson: vi.fn(async () => ({ ok: false })),
execSubprocessRaw: vi.fn(async () => 'subprocess-raw'),
formatNativeRaw: vi.fn(() => 'raw-native-output'),
};
const transport = new GSDTransport(registry, adapters);
const result = await transport.run({
legacyCommand: 'commit',
legacyArgs: ['msg'],
registryCommand: 'commit',
registryArgs: ['msg'],
mode: 'raw',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
});
expect(result).toBe('raw-native-output');
expect(adapters.formatNativeRaw).toHaveBeenCalledOnce();
expect(adapters.execSubprocessRaw).not.toHaveBeenCalled();
});
it('falls back to internal raw formatter when formatNativeRaw missing', async () => {
const registry = new QueryRegistry();
registry.register('commit', async () => ({ data: undefined }));
const adapters = {
dispatchNative: vi.fn(async () => ({ data: undefined })),
execSubprocessJson: vi.fn(async () => ({ ok: false })),
execSubprocessRaw: vi.fn(async () => 'subprocess-raw'),
};
const transport = new GSDTransport(registry, adapters);
const result = await transport.run({
legacyCommand: 'commit',
legacyArgs: ['msg'],
registryCommand: 'commit',
registryArgs: ['msg'],
mode: 'raw',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
});
expect(result).toBe('');
expect(adapters.execSubprocessRaw).not.toHaveBeenCalled();
});
it('routes natively when workstream present (Phase 6 fix)', async () => {
// Phase 6 fix: GSDTransport no longer forces subprocess for workstream-scoped
// requests. The per-request dispatchNative closure (Phase 5.1) correctly
// threads workstream to registry.dispatch(), so native dispatch is used.
const registry = new QueryRegistry();
registry.register('state.load', async () => ({ data: { ok: true } }));
const adapters = {
dispatchNative: vi.fn(async () => ({ data: { ok: true } })),
execSubprocessJson: vi.fn(async () => ({ ok: 'ws-subprocess' })),
execSubprocessRaw: vi.fn(async () => 'ws-subprocess-raw'),
};
const transport = new GSDTransport(registry, adapters);
const result = await transport.run({
legacyCommand: 'state',
legacyArgs: ['load'],
registryCommand: 'state.load',
registryArgs: [],
mode: 'json',
projectDir: '/tmp',
workstream: 'ws-1',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
});
// Native dispatch is used — subprocess is NOT called.
expect(result).toEqual({ ok: true });
expect(adapters.dispatchNative).toHaveBeenCalledOnce();
expect(adapters.execSubprocessJson).not.toHaveBeenCalled();
});
it('fails when command is unregistered and subprocess fallback is disabled', async () => {
const registry = new QueryRegistry();
const adapters = {
dispatchNative: vi.fn(async () => ({ data: { ok: true } })),
execSubprocessJson: vi.fn(async () => ({ ok: 'fallback' })),
execSubprocessRaw: vi.fn(async () => 'fallback-raw'),
};
const transport = new GSDTransport(registry, adapters);
await expect(transport.run({
legacyCommand: 'unknown',
legacyArgs: [],
registryCommand: 'unknown',
registryArgs: [],
mode: 'json',
projectDir: '/tmp',
}, {
preferNative: true,
allowFallbackToSubprocess: false,
})).rejects.toThrow("Subprocess fallback disabled");
expect(adapters.execSubprocessJson).not.toHaveBeenCalled();
});
it('routes natively when workstream present and mode is raw (Phase 6 fix)', async () => {
// Phase 6 fix: workstream no longer forces subprocess. Native dispatch is used
// even in raw mode — formatNativeRaw (if set) handles the output projection.
const registry = new QueryRegistry();
registry.register('commit', async () => ({ data: { hash: 'abc' } }));
const adapters = {
dispatchNative: vi.fn(async () => ({ data: { hash: 'abc' } })),
execSubprocessJson: vi.fn(async () => ({ ok: 'json-subprocess' })),
execSubprocessRaw: vi.fn(async () => 'raw-subprocess'),
};
const transport = new GSDTransport(registry, adapters);
const result = await transport.run({
legacyCommand: 'commit',
legacyArgs: ['msg'],
registryCommand: 'commit',
registryArgs: ['msg'],
mode: 'raw',
projectDir: '/tmp',
workstream: 'ws-1',
}, {
preferNative: true,
allowFallbackToSubprocess: true,
});
// Native dispatch is used — toRaw serializes data to JSON.
expect(typeof result).toBe('string');
expect(adapters.dispatchNative).toHaveBeenCalledOnce();
expect(adapters.execSubprocessRaw).not.toHaveBeenCalled();
expect(adapters.execSubprocessJson).not.toHaveBeenCalled();
});
});

View File

@@ -1,118 +0,0 @@
import type { QueryResult } from './query/utils.js';
import type { QueryRegistry } from './query/registry.js';
import type { TransportMode } from './gsd-transport-policy.js';
import { toFailureSignal } from './query-failure-classification.js';
import { GSDToolsError } from './gsd-tools-error.js';
export interface TransportRequest {
legacyCommand: string;
legacyArgs: string[];
registryCommand: string;
registryArgs: string[];
mode: TransportMode;
projectDir: string;
workstream?: string;
}
export interface TransportAdapters {
dispatchNative: (request: TransportRequest) => Promise<QueryResult>;
execSubprocessJson: (legacyCommand: string, legacyArgs: string[]) => Promise<unknown>;
execSubprocessRaw: (legacyCommand: string, legacyArgs: string[]) => Promise<string>;
formatNativeRaw?: (registryCommand: string, data: unknown) => string;
}
export interface TransportPolicyLike {
preferNative: boolean;
allowFallbackToSubprocess: boolean;
}
export interface TransportDecision {
dispatchMode: 'native' | 'subprocess';
reason?: 'native_not_preferred' | 'native_unregistered' | 'native_failure_fallback';
}
export class GSDTransport {
constructor(
private readonly registry: QueryRegistry,
private readonly adapters: TransportAdapters,
) {}
async run(
request: TransportRequest,
policy: TransportPolicyLike,
onDecision?: (decision: TransportDecision) => void,
): Promise<unknown> {
const useNative = this.shouldUseNative(request, policy);
if (useNative) {
try {
const native = await this.adapters.dispatchNative(request);
onDecision?.({ dispatchMode: 'native' });
return this.projectNativeOutput(request, native.data);
} catch (error) {
if (this.shouldRethrowNativeError(error, policy)) throw error;
onDecision?.({ dispatchMode: 'subprocess', reason: 'native_failure_fallback' });
}
} else {
const reason = this.subprocessReason(request, policy);
if (!policy.allowFallbackToSubprocess && reason === 'native_unregistered') {
throw GSDToolsError.failure(
`Subprocess fallback disabled: command '${request.registryCommand}' cannot run without native dispatch`,
request.legacyCommand,
request.legacyArgs,
null,
);
}
onDecision?.({ dispatchMode: 'subprocess', reason });
}
return this.dispatchSubprocess(request);
}
private shouldUseNative(request: TransportRequest, policy: TransportPolicyLike): boolean {
// Phase 5.0 worker fix: dispatchNative now correctly threads projectDir and
// workstream per-request (see worker.ts dispatchNative closure). Workstream
// commands no longer need to force subprocess — native dispatch handles them.
return policy.preferNative && this.registry.has(request.registryCommand);
}
private subprocessReason(request: TransportRequest, policy: TransportPolicyLike): TransportDecision['reason'] {
if (!policy.preferNative) return 'native_not_preferred';
if (!this.registry.has(request.registryCommand)) return 'native_unregistered';
throw new Error(
`Unexpected subprocess reason state for command '${request.registryCommand}' with preferNative=${String(policy.preferNative)}`,
);
}
private shouldRethrowNativeError(error: unknown, policy: TransportPolicyLike): boolean {
if (!policy.allowFallbackToSubprocess) return true;
// Do not subprocess-fallback after a timed-out native dispatch:
// the timeout does not cancel the native handler, so falling through
// would run the same command twice (double-execution race).
return toFailureSignal(error).kind === 'timeout';
}
private dispatchSubprocess(request: TransportRequest): Promise<unknown> {
if (request.mode === 'raw') {
return this.adapters.execSubprocessRaw(request.legacyCommand, request.legacyArgs);
}
return this.adapters.execSubprocessJson(request.legacyCommand, request.legacyArgs);
}
private projectNativeOutput(request: TransportRequest, data: unknown): unknown {
if (request.mode === 'raw') {
if (this.adapters.formatNativeRaw) {
return this.adapters.formatNativeRaw(request.registryCommand, data).trim();
}
return this.toRaw(data);
}
return data;
}
private toRaw(data: unknown): string {
if (typeof data === 'string') return data.trim();
const json = JSON.stringify(data, null, 2);
if (json == null) return '';
return json.trim();
}
}

View File

@@ -1,815 +0,0 @@
/**
* Complex init composition handlers — the 3 heavyweight init commands
* that require deep filesystem scanning and ROADMAP.md parsing.
*
* Composes existing atomic SDK queries into the same flat JSON bundles
* that CJS init.cjs produces for the new-project, progress, and manager
* workflows.
*
* Port of get-shit-done/bin/lib/init.cjs cmdInitNewProject (lines 296-399),
* cmdInitProgress (lines 1139-1284), cmdInitManager (lines 854-1137).
*
* @example
* ```typescript
* import { initProgress, initManager } from './complex.js';
*
* const result = await initProgress([], '/project');
* // { data: { phases: [...], milestone_version: 'v3.0', ... } }
* ```
*/
import { existsSync, readdirSync, statSync, type Dirent } from 'node:fs';
import { execSync } from 'node:child_process';
import { readFile } from 'node:fs/promises';
import { join, relative } from 'node:path';
import { homedir } from 'node:os';
import { loadConfig } from '../../config.js';
import { resolveModel } from '../../query/config-query.js';
import {
detectRuntime,
planningPaths,
normalizePhaseName,
phaseTokenMatches,
resolveAgentsDir,
toPosixPath,
} from '../../query/helpers.js';
import {
getMilestoneInfo,
extractCurrentMilestone,
extractNextMilestoneSection,
extractPhasesFromSection,
} from '../../query/roadmap.js';
import { agentSkills } from '../../query/skills.js';
import { withProjectRoot } from './composer.js';
import type { QueryHandler } from '../../query/utils.js';
// ─── Internal helpers ──────────────────────────────────────────────────────
/**
* Get model alias string from resolveModel result.
*/
async function getModelAlias(agentType: string, projectDir: string): Promise<string> {
const result = await resolveModel([agentType], projectDir);
const data = result.data as Record<string, unknown>;
return typeof data.model === 'string' ? data.model : 'sonnet';
}
/**
* Check if a file exists at a relative path within projectDir.
*/
function pathExists(base: string, relPath: string): boolean {
return existsSync(join(base, relPath));
}
/**
* Bug #3491: detect whether `base` is inside any git worktree, and if so,
* return the absolute worktree root. Mirrors the CJS `gitWorktreeInfoInternal`
* in get-shit-done/bin/lib/core.cjs — keep these two implementations behaviour-
* identical so the SDK and CJS init handlers emit the same has_git semantics.
*
* Returns { inside, worktreeRoot } — both fall back to false/null on any error
* (git unavailable, not a repo, timeout) so callers see the conservative
* default that preserves pre-fix behaviour for non-git environments.
*/
function gitWorktreeInfo(base: string): { inside: boolean; worktreeRoot: string | null } {
try {
const inside = execSync('git rev-parse --is-inside-work-tree', {
cwd: base,
stdio: ['ignore', 'pipe', 'ignore'],
encoding: 'utf-8',
timeout: 5000,
env: { ...process.env, GIT_TERMINAL_PROMPT: '0' },
}).trim();
if (inside !== 'true') return { inside: false, worktreeRoot: null };
try {
const root = execSync('git rev-parse --show-toplevel', {
cwd: base,
stdio: ['ignore', 'pipe', 'ignore'],
encoding: 'utf-8',
timeout: 5000,
env: { ...process.env, GIT_TERMINAL_PROMPT: '0' },
}).trim();
return { inside: true, worktreeRoot: root || null };
} catch {
return { inside: true, worktreeRoot: null };
}
} catch {
return { inside: false, worktreeRoot: null };
}
}
function detectNestedSubdir(base: string, info: { inside: boolean; worktreeRoot: string | null }): boolean {
if (!info.inside) return false;
try {
const prefix = execSync('git rev-parse --show-prefix', {
cwd: base,
stdio: ['ignore', 'pipe', 'ignore'],
encoding: 'utf-8',
timeout: 5000,
env: { ...process.env, GIT_TERMINAL_PROMPT: '0' },
}).trim().replace(/\\/g, '/');
if (prefix.length > 0) return prefix !== '.' && prefix !== './';
return false;
} catch {}
if (!info.worktreeRoot) return false;
const normalize = (p: string) => p.replace(/\\/g, '/').replace(/\/+$/g, '').toLowerCase();
const root = normalize(info.worktreeRoot);
const cwd = normalize(base);
return root !== cwd;
}
const NEW_PROJECT_REQUIRED_AGENTS = [
'gsd-project-researcher',
'gsd-research-synthesizer',
'gsd-roadmapper',
];
function hasAgentDefinition(agentsDir: string, agent: string): boolean {
return existsSync(join(agentsDir, `${agent}.md`)) ||
existsSync(join(agentsDir, `${agent}.agent.md`));
}
async function resolveAgentSkillPayloadAgents(
requiredAgents: string[],
projectDir: string,
): Promise<string[]> {
const available: string[] = [];
for (const agent of requiredAgents) {
const result = await agentSkills([agent], projectDir);
if (typeof result.data === 'string' && result.data.trim() !== '') {
available.push(agent);
}
}
return available;
}
/**
* Extract ROADMAP checkbox states: `- [x] Phase N` → true, `- [ ] Phase N` → false.
* Shared by initProgress and initManager so both treat ROADMAP as the
* fallback/override source of truth for completion.
*/
function extractCheckboxStates(content: string): Map<string, boolean> {
const states = new Map<string, boolean>();
const pattern = /-\s*\[(x| )\]\s*.*Phase\s+(\d+[A-Z]?(?:\.\d+)*)[:\s]/gi;
let m: RegExpExecArray | null;
while ((m = pattern.exec(content)) !== null) {
states.set(m[2], m[1].toLowerCase() === 'x');
}
return states;
}
/**
* Extract terminal phase markers from ROADMAP phase headings, e.g.
* `(COMPLETE)`, `(SHIPPED ...)`, `(DEFERRED)`, `(SUPERSEDED ...)`.
* These labels mean the phase should not be selected as next pending work.
*/
function extractTerminalStatusLabels(content: string): Set<string> {
const terminal = new Set<string>();
const headingPattern = /#{2,4}\s*Phase\s+(\d+[A-Z]?(?:\.\d+)*)\s*:\s*([^\n]+)/gi;
const terminalRe = /(?:\(|\*\*)\s*(SHIPPED|COMPLETE|DEFERRED|SUPERSEDED|MERGED\s+INTO|FOLDED\s+INTO)\b/i;
let m: RegExpExecArray | null;
while ((m = headingPattern.exec(content)) !== null) {
if (terminalRe.test(m[2])) {
terminal.add(m[1]);
terminal.add(m[1].replace(/^0+/, '') || '0');
}
}
return terminal;
}
/**
* Derive progress-level status from a ROADMAP checkbox when the phase has
* no on-disk directory. Returns 'complete' for `[x]`, 'not_started' otherwise.
* Disk status (when present) always wins — it's more recent truth for in-flight work.
*/
function deriveStatusFromCheckbox(
phaseNum: string,
checkboxStates: Map<string, boolean>,
): 'complete' | 'not_started' {
const stripped = phaseNum.replace(/^0+/, '') || '0';
if (checkboxStates.get(phaseNum) === true) return 'complete';
if (checkboxStates.get(stripped) === true) return 'complete';
return 'not_started';
}
function listPhasePlanAndSummaryCounts(phasePath: string): { plans: string[]; summaries: string[] } {
const phaseFiles = readdirSync(phasePath);
const rootPlans = phaseFiles.filter(f => f.endsWith('-PLAN.md') || f === 'PLAN.md');
const rootSummaries = phaseFiles.filter(f => f.endsWith('-SUMMARY.md') || f === 'SUMMARY.md');
const plansDir = join(phasePath, 'plans');
let nestedPlans: string[] = [];
let nestedSummaries: string[] = [];
if (existsSync(plansDir)) {
const files = readdirSync(plansDir);
nestedPlans = files.filter(f => /^PLAN-\d+.*\.md$/i.test(f));
nestedSummaries = files.filter(f => /^SUMMARY-\d+.*\.md$/i.test(f));
}
return {
plans: rootPlans.concat(nestedPlans),
summaries: rootSummaries.concat(nestedSummaries),
};
}
// ─── initNewProject ───────────────────────────────────────────────────────
/**
* Init handler for new-project workflow.
*
* Detects brownfield state (existing code, package files, git), checks
* search API availability, and resolves project researcher models.
*
* Port of cmdInitNewProject from init.cjs lines 296-399.
*/
export const initNewProject: QueryHandler = async (_args, projectDir, workstream) => {
const config = await loadConfig(projectDir, workstream);
// Detect search API key availability from env vars and ~/.gsd/ files
const gsdHome = join(homedir(), '.gsd');
const hasBraveSearch = !!(
process.env.BRAVE_API_KEY ||
existsSync(join(gsdHome, 'brave_api_key'))
);
const hasFirecrawl = !!(
process.env.FIRECRAWL_API_KEY ||
existsSync(join(gsdHome, 'firecrawl_api_key'))
);
const hasExaSearch = !!(
process.env.EXA_API_KEY ||
existsSync(join(gsdHome, 'exa_api_key'))
);
// Detect existing code (depth-limited scan, no external tools)
const codeExtensions = new Set([
'.ts', '.js', '.py', '.go', '.rs', '.swift', '.java',
'.kt', '.kts', '.c', '.cpp', '.h', '.cs', '.rb', '.php',
'.dart', '.m', '.mm', '.scala', '.groovy', '.lua',
'.r', '.R', '.zig', '.ex', '.exs', '.clj',
]);
const skipDirs = new Set([
'node_modules', '.git', '.planning', '.claude', '.codex',
'__pycache__', 'target', 'dist', 'build',
]);
function findCodeFiles(dir: string, depth: number): boolean {
if (depth > 3) return false;
let entries: Dirent[];
try {
entries = readdirSync(dir, { withFileTypes: true });
} catch {
return false;
}
for (const entry of entries) {
if (entry.isFile()) {
const ext = entry.name.slice(entry.name.lastIndexOf('.'));
if (codeExtensions.has(ext)) return true;
} else if (entry.isDirectory() && !skipDirs.has(entry.name)) {
if (findCodeFiles(join(dir, entry.name), depth + 1)) return true;
}
}
return false;
}
let hasExistingCode = false;
try {
hasExistingCode = findCodeFiles(projectDir, 0);
} catch { /* best-effort */ }
const hasPackageFile =
pathExists(projectDir, 'package.json') ||
pathExists(projectDir, 'requirements.txt') ||
pathExists(projectDir, 'Cargo.toml') ||
pathExists(projectDir, 'go.mod') ||
pathExists(projectDir, 'Package.swift') ||
pathExists(projectDir, 'build.gradle') ||
pathExists(projectDir, 'build.gradle.kts') ||
pathExists(projectDir, 'pom.xml') ||
pathExists(projectDir, 'Gemfile') ||
pathExists(projectDir, 'composer.json') ||
pathExists(projectDir, 'pubspec.yaml') ||
pathExists(projectDir, 'CMakeLists.txt') ||
pathExists(projectDir, 'Makefile') ||
pathExists(projectDir, 'build.zig') ||
pathExists(projectDir, 'mix.exs') ||
pathExists(projectDir, 'project.clj');
const [researcherModel, synthesizerModel, roadmapperModel] = await Promise.all([
getModelAlias('gsd-project-researcher', projectDir),
getModelAlias('gsd-research-synthesizer', projectDir),
getModelAlias('gsd-roadmapper', projectDir),
]);
const runtime = detectRuntime(config as { runtime?: unknown });
const agentsDir = resolveAgentsDir(runtime, projectDir);
const gitInfo = gitWorktreeInfo(projectDir);
const missingRequiredAgents = NEW_PROJECT_REQUIRED_AGENTS.filter(
agent => !hasAgentDefinition(agentsDir, agent),
);
const agentSkillPayloadAgents = await resolveAgentSkillPayloadAgents(
NEW_PROJECT_REQUIRED_AGENTS,
projectDir,
);
const result: Record<string, unknown> = {
researcher_model: researcherModel,
synthesizer_model: synthesizerModel,
roadmapper_model: roadmapperModel,
commit_docs: config.commit_docs,
project_exists: pathExists(projectDir, '.planning/PROJECT.md'),
has_codebase_map: pathExists(projectDir, '.planning/codebase'),
planning_exists: pathExists(projectDir, '.planning'),
has_existing_code: hasExistingCode,
has_package_file: hasPackageFile,
is_brownfield: hasExistingCode || hasPackageFile,
needs_codebase_map:
(hasExistingCode || hasPackageFile) && !pathExists(projectDir, '.planning/codebase'),
// Bug #3491: detect parent worktree to avoid nested .git init.
has_git: gitInfo.inside,
git_worktree_root: gitInfo.worktreeRoot,
in_nested_subdir: detectNestedSubdir(projectDir, gitInfo),
brave_search_available: hasBraveSearch,
firecrawl_available: hasFirecrawl,
exa_search_available: hasExaSearch,
project_path: '.planning/PROJECT.md',
agent_runtime: runtime,
agents_dir: agentsDir,
required_agents: NEW_PROJECT_REQUIRED_AGENTS,
required_agents_installed: missingRequiredAgents.length === 0,
missing_required_agents: missingRequiredAgents,
agent_skill_payloads_available: agentSkillPayloadAgents.length === NEW_PROJECT_REQUIRED_AGENTS.length,
agent_skill_payload_agents: agentSkillPayloadAgents,
};
return { data: withProjectRoot(projectDir, result, config as Record<string, unknown>) };
};
// ─── initProgress ─────────────────────────────────────────────────────────
/**
* Init handler for progress workflow.
*
* Builds phase list with plan/summary counts and paused state detection.
*
* Port of cmdInitProgress from init.cjs lines 1139-1284.
*/
export const initProgress: QueryHandler = async (_args, projectDir, workstream) => {
const config = await loadConfig(projectDir, workstream);
const milestone = await getMilestoneInfo(projectDir, workstream);
const paths = planningPaths(projectDir, workstream);
const phases: Record<string, unknown>[] = [];
let currentPhase: Record<string, unknown> | null = null;
let nextPhase: Record<string, unknown> | null = null;
// Build set of phases from ROADMAP for the current milestone
const roadmapPhaseNames = new Map<string, string>();
const seenPhaseNums = new Set<string>();
let checkboxStates = new Map<string, boolean>();
let terminalLabels = new Set<string>();
try {
const rawRoadmap = await readFile(paths.roadmap, 'utf-8');
const roadmapContent = await extractCurrentMilestone(rawRoadmap, projectDir, workstream);
const headingPattern = /#{2,4}\s*Phase\s+(\d+[A-Z]?(?:\.\d+)*)\s*:\s*([^\n]+)/gi;
let hm: RegExpExecArray | null;
while ((hm = headingPattern.exec(roadmapContent)) !== null) {
const pNum = hm[1];
const pName = hm[2].replace(/\(INSERTED\)/i, '').trim();
roadmapPhaseNames.set(pNum, pName);
}
checkboxStates = extractCheckboxStates(roadmapContent);
terminalLabels = extractTerminalStatusLabels(roadmapContent);
} catch { /* intentionally empty */ }
// Scan phase directories
try {
const entries = readdirSync(paths.phases, { withFileTypes: true });
const dirs = entries
.filter(e => e.isDirectory())
.map(e => e.name)
.sort((a, b) => {
const pa = a.match(/^(\d+[A-Z]?(?:\.\d+)*)/i);
const pb = b.match(/^(\d+[A-Z]?(?:\.\d+)*)/i);
if (!pa || !pb) return a.localeCompare(b);
return parseInt(pa[1], 10) - parseInt(pb[1], 10);
});
for (const dir of dirs) {
const match = dir.match(/^(\d+[A-Z]?(?:\.\d+)*)-?(.*)/i);
const phaseNumber = match ? match[1] : dir;
const phaseName = match && match[2] ? match[2] : null;
seenPhaseNums.add(phaseNumber.replace(/^0+/, '') || '0');
const phasePath = join(paths.phases, dir);
const phaseFiles = readdirSync(phasePath);
const { plans, summaries } = listPhasePlanAndSummaryCounts(phasePath);
const hasResearch = phaseFiles.some(f => f.endsWith('-RESEARCH.md') || f === 'RESEARCH.md');
let status =
summaries.length >= plans.length && plans.length > 0 ? 'complete' :
plans.length > 0 ? 'in_progress' :
hasResearch ? 'researched' : 'pending';
// #2674: align with initManager — a ROADMAP `- [x] Phase N` checkbox
// wins over disk state. A stub phase dir with no SUMMARY is leftover
// scaffolding; the user's explicit [x] is the authoritative signal.
const strippedNum = phaseNumber.replace(/^0+/, '') || '0';
const roadmapComplete =
checkboxStates.get(phaseNumber) === true ||
checkboxStates.get(strippedNum) === true;
if (roadmapComplete && status !== 'complete') {
status = 'complete';
}
if (terminalLabels.has(phaseNumber) || terminalLabels.has(strippedNum)) {
status = 'complete';
}
const phaseInfo: Record<string, unknown> = {
number: phaseNumber,
name: phaseName,
directory: toPosixPath(relative(projectDir, join(paths.phases, dir))),
status,
plan_count: plans.length,
summary_count: summaries.length,
has_research: hasResearch,
};
phases.push(phaseInfo);
if (!currentPhase && (status === 'in_progress' || status === 'researched')) {
currentPhase = phaseInfo;
}
if (!nextPhase && status === 'pending') {
nextPhase = phaseInfo;
}
}
} catch { /* intentionally empty */ }
// Add ROADMAP-only phases not yet on disk. For phases with a ROADMAP
// `[x]` checkbox, treat them as complete (#2646).
for (const [num, name] of roadmapPhaseNames) {
const stripped = num.replace(/^0+/, '') || '0';
if (!seenPhaseNums.has(stripped)) {
const status = deriveStatusFromCheckbox(num, checkboxStates);
const terminalComplete = terminalLabels.has(num) || terminalLabels.has(stripped);
const phaseInfo: Record<string, unknown> = {
number: num,
name: name.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, ''),
directory: null,
status: terminalComplete ? 'complete' : status,
plan_count: 0,
summary_count: 0,
has_research: false,
};
phases.push(phaseInfo);
if (!nextPhase && !currentPhase && phaseInfo.status !== 'complete') {
nextPhase = phaseInfo;
}
}
}
phases.sort((a, b) => parseInt(a.number as string, 10) - parseInt(b.number as string, 10));
// Check paused state in STATE.md
let pausedAt: string | null = null;
try {
const stateContent = await readFile(paths.state, 'utf-8');
const pauseMatch = stateContent.match(/\*\*Paused At:\*\*\s*(.+)/);
if (pauseMatch) pausedAt = pauseMatch[1].trim();
} catch { /* intentionally empty */ }
const result: Record<string, unknown> = {
executor_model: await getModelAlias('gsd-executor', projectDir),
planner_model: await getModelAlias('gsd-planner', projectDir),
commit_docs: config.commit_docs,
milestone_version: milestone.version,
milestone_name: milestone.name,
phases,
phase_count: phases.length,
completed_count: phases.filter(p => p.status === 'complete').length,
in_progress_count: phases.filter(p => p.status === 'in_progress').length,
current_phase: currentPhase,
next_phase: nextPhase,
paused_at: pausedAt,
has_work_in_progress: !!currentPhase,
project_exists: pathExists(projectDir, '.planning/PROJECT.md'),
roadmap_exists: existsSync(paths.roadmap),
state_exists: existsSync(paths.state),
state_path: toPosixPath(relative(projectDir, paths.state)),
roadmap_path: toPosixPath(relative(projectDir, paths.roadmap)),
project_path: '.planning/PROJECT.md',
config_path: toPosixPath(relative(projectDir, paths.config)),
};
return { data: withProjectRoot(projectDir, result, config as Record<string, unknown>) };
};
// ─── initManager ─────────────────────────────────────────────────────────
/**
* Init handler for manager workflow.
*
* Parses ROADMAP.md for all phases, computes disk status, dependency
* graph, and recommended actions per phase.
*
* Port of cmdInitManager from init.cjs lines 854-1137.
*/
export const initManager: QueryHandler = async (_args, projectDir, workstream) => {
const config = await loadConfig(projectDir, workstream);
const milestone = await getMilestoneInfo(projectDir, workstream);
const paths = planningPaths(projectDir, workstream);
let rawContent: string;
try {
rawContent = await readFile(paths.roadmap, 'utf-8');
} catch {
return { data: { error: 'No ROADMAP.md found. Run /gsd-new-milestone first.' } };
}
const content = await extractCurrentMilestone(rawContent, projectDir, workstream);
// Pre-compute directory listing once
let phaseDirEntries: string[] = [];
try {
phaseDirEntries = readdirSync(paths.phases, { withFileTypes: true })
.filter(e => e.isDirectory())
.map(e => e.name);
} catch { /* intentionally empty */ }
// Pre-extract checkbox states in a single pass (shared helper — #2646)
const checkboxStates = extractCheckboxStates(content);
const phasePattern = /#{2,4}\s*Phase\s+(\d+[A-Z]?(?:\.\d+)*)\s*:\s*([^\n]+)/gi;
const phases: Record<string, unknown>[] = [];
let pMatch: RegExpExecArray | null;
while ((pMatch = phasePattern.exec(content)) !== null) {
const phaseNum = pMatch[1];
const phaseName = pMatch[2].replace(/\(INSERTED\)/i, '').trim();
const sectionStart = pMatch.index;
const restOfContent = content.slice(sectionStart);
const nextHeader = restOfContent.match(/\n#{2,4}\s+Phase\s+\d/i);
const sectionEnd = nextHeader ? sectionStart + (nextHeader.index ?? 0) : content.length;
const section = content.slice(sectionStart, sectionEnd);
const goalMatch = section.match(/\*\*Goal(?::\*\*|\*\*:)\s*([^\n]+)/i);
const goal = goalMatch ? goalMatch[1].trim() : null;
const dependsMatch = section.match(/\*\*Depends on(?::\*\*|\*\*:)\s*([^\n]+)/i);
const dependsOn = dependsMatch ? dependsMatch[1].trim() : null;
const normalized = normalizePhaseName(phaseNum);
let diskStatus = 'no_directory';
let planCount = 0;
let summaryCount = 0;
let hasContext = false;
let hasResearch = false;
let lastActivity: string | null = null;
let isActive = false;
try {
const dirMatch = phaseDirEntries.find(d => phaseTokenMatches(d, normalized));
if (dirMatch) {
const fullDir = join(paths.phases, dirMatch);
const phaseFiles = readdirSync(fullDir);
const counts = listPhasePlanAndSummaryCounts(fullDir);
planCount = counts.plans.length;
summaryCount = counts.summaries.length;
hasContext = phaseFiles.some(f => f.endsWith('-CONTEXT.md') || f === 'CONTEXT.md');
hasResearch = phaseFiles.some(f => f.endsWith('-RESEARCH.md') || f === 'RESEARCH.md');
if (summaryCount >= planCount && planCount > 0) diskStatus = 'complete';
else if (summaryCount > 0) diskStatus = 'partial';
else if (planCount > 0) diskStatus = 'planned';
else if (hasResearch) diskStatus = 'researched';
else if (hasContext) diskStatus = 'discussed';
else diskStatus = 'empty';
const now = Date.now();
let newestMtime = 0;
for (const f of phaseFiles) {
try {
const st = statSync(join(fullDir, f));
if (st.mtimeMs > newestMtime) newestMtime = st.mtimeMs;
} catch { /* intentionally empty */ }
}
if (newestMtime > 0) {
lastActivity = new Date(newestMtime).toISOString();
isActive = (now - newestMtime) < 300000; // 5 minutes
}
}
} catch { /* intentionally empty */ }
const roadmapComplete = checkboxStates.get(phaseNum) || false;
if (roadmapComplete && diskStatus !== 'complete') {
diskStatus = 'complete';
}
const MAX_NAME_WIDTH = 20;
const displayName = phaseName.length > MAX_NAME_WIDTH
? phaseName.slice(0, MAX_NAME_WIDTH - 1) + '…'
: phaseName;
phases.push({
number: phaseNum,
name: phaseName,
display_name: displayName,
goal,
depends_on: dependsOn,
disk_status: diskStatus,
has_context: hasContext,
has_research: hasResearch,
plan_count: planCount,
summary_count: summaryCount,
roadmap_complete: roadmapComplete,
last_activity: lastActivity,
is_active: isActive,
});
}
// Dependency satisfaction
const completedNums = new Set(
phases.filter(p => p.disk_status === 'complete').map(p => p.number as string),
);
for (const phase of phases) {
const dependsOnStr = phase.depends_on as string | null;
if (!dependsOnStr || /^none$/i.test(dependsOnStr.trim())) {
phase.deps_satisfied = true;
phase.dep_phases = [];
phase.deps_display = '—';
} else {
const depNums = dependsOnStr.match(/\d+(?:\.\d+)*/g) || [];
phase.deps_satisfied = depNums.every(n => completedNums.has(n));
phase.dep_phases = depNums;
phase.deps_display = depNums.length > 0 ? depNums.join(',') : '—';
}
}
// Bug #2268: mark EVERY undiscussed phase as is_next_to_discuss, not just
// the first one. Multiple independent phases can be discussed in parallel
// — the sliding-window pattern made the manager only recommend one
// discuss action even when callers had free capacity to discuss several.
for (const phase of phases) {
const status = phase.disk_status as string;
phase.is_next_to_discuss = (status === 'empty' || status === 'no_directory');
}
// Check WAITING.json signal
let waitingSignal: unknown = null;
try {
const waitingPath = join(projectDir, '.planning', 'WAITING.json');
if (existsSync(waitingPath)) {
const { readFileSync } = await import('node:fs');
waitingSignal = JSON.parse(readFileSync(waitingPath, 'utf-8'));
}
} catch { /* intentionally empty */ }
// Compute recommended actions
const phaseMap = new Map(phases.map(p => [p.number as string, p]));
function reaches(from: string, to: string, visited = new Set<string>()): boolean {
if (visited.has(from)) return false;
visited.add(from);
const p = phaseMap.get(from);
const depPhases = p?.dep_phases as string[] | undefined;
if (!depPhases || depPhases.length === 0) return false;
if (depPhases.includes(to)) return true;
return depPhases.some(dep => reaches(dep, to, visited));
}
const activeExecuting = phases.filter(p => {
const status = p.disk_status as string;
return status === 'partial' || (status === 'planned' && p.is_active);
});
const activePlanning = phases.filter(p => {
const status = p.disk_status as string;
return p.is_active && (status === 'discussed' || status === 'researched');
});
const recommendedActions: Record<string, unknown>[] = [];
for (const phase of phases) {
const status = phase.disk_status as string;
if (status === 'complete') continue;
if (/^999(?:\.|$)/.test(phase.number as string)) continue;
if (status === 'planned' && phase.deps_satisfied) {
const action = {
phase: phase.number,
phase_name: phase.name,
action: 'execute',
reason: `${phase.plan_count} plans ready, dependencies met`,
command: `/gsd-execute-phase ${phase.number}`,
};
const isAllowed = activeExecuting.length === 0 ||
activeExecuting.every(a => !reaches(phase.number as string, a.number as string) && !reaches(a.number as string, phase.number as string));
if (isAllowed) recommendedActions.push(action);
} else if (status === 'discussed' || status === 'researched') {
const action = {
phase: phase.number,
phase_name: phase.name,
action: 'plan',
reason: 'Context gathered, ready for planning',
command: `/gsd-plan-phase ${phase.number}`,
};
const isAllowed = activePlanning.length === 0 ||
activePlanning.every(a => !reaches(phase.number as string, a.number as string) && !reaches(a.number as string, phase.number as string));
if (isAllowed) recommendedActions.push(action);
} else if ((status === 'empty' || status === 'no_directory') && phase.is_next_to_discuss) {
recommendedActions.push({
phase: phase.number,
phase_name: phase.name,
action: 'discuss',
reason: 'Unblocked, ready to gather context',
command: `/gsd-discuss-phase ${phase.number}`,
});
}
}
const completedCount = phases.filter(p => p.disk_status === 'complete').length;
// ── Next-milestone surface (issue #2497) ───────────────────────────────
// Populate queued_phases + metadata with the milestone immediately after
// the active one, so the /gsd-manager dashboard can preview what's coming
// next without mixing it into the active phases grid. Empty/null when the
// active milestone is the last one in ROADMAP.
let queuedPhases: Record<string, unknown>[] = [];
let queuedMilestoneVersion: string | null = null;
let queuedMilestoneName: string | null = null;
try {
const next = await extractNextMilestoneSection(rawContent, projectDir);
if (next) {
queuedMilestoneVersion = next.version;
queuedMilestoneName = next.name;
queuedPhases = extractPhasesFromSection(next.section).map(p => {
const MAX_NAME_WIDTH = 20;
const display_name = p.name.length > MAX_NAME_WIDTH
? p.name.slice(0, MAX_NAME_WIDTH - 1) + '…'
: p.name;
const depNums = p.depends_on && !/^none$/i.test(p.depends_on.trim())
? (p.depends_on.match(/\d+(?:\.\d+)*/g) || [])
: [];
return {
number: p.number,
name: p.name,
display_name,
goal: p.goal,
depends_on: p.depends_on,
dep_phases: depNums,
deps_display: depNums.length > 0 ? depNums.join(',') : '—',
};
});
}
} catch { /* queued_phases is a non-critical enhancement */ }
// Read manager flags from config
const managerConfig = (config as Record<string, unknown>).manager as Record<string, Record<string, string>> | undefined;
const sanitizeFlags = (raw: unknown): string => {
const val = typeof raw === 'string' ? raw : '';
if (!val) return '';
const tokens = val.split(/\s+/).filter(Boolean);
const safe = tokens.every(t => /^--[a-zA-Z0-9][-a-zA-Z0-9]*$/.test(t) || /^[a-zA-Z0-9][-a-zA-Z0-9_.]*$/.test(t));
return safe ? val : '';
};
const managerFlags = {
discuss: sanitizeFlags(managerConfig?.flags?.discuss),
plan: sanitizeFlags(managerConfig?.flags?.plan),
execute: sanitizeFlags(managerConfig?.flags?.execute),
};
const result: Record<string, unknown> = {
milestone_version: milestone.version,
milestone_name: milestone.name,
phases,
phase_count: phases.length,
completed_count: completedCount,
in_progress_count: phases.filter(p => ['partial', 'planned', 'discussed', 'researched'].includes(p.disk_status as string)).length,
recommended_actions: recommendedActions,
waiting_signal: waitingSignal,
all_complete: completedCount === phases.length && phases.length > 0,
queued_phases: queuedPhases,
queued_milestone_version: queuedMilestoneVersion,
queued_milestone_name: queuedMilestoneName,
project_exists: pathExists(projectDir, '.planning/PROJECT.md'),
roadmap_exists: true,
state_exists: true,
manager_flags: managerFlags,
};
return { data: withProjectRoot(projectDir, result, config as Record<string, unknown>) };
};

File diff suppressed because it is too large Load Diff

View File

@@ -1,28 +0,0 @@
import type { QueryHandler } from '../../query/utils.js';
import {
initExecutePhase, initPlanPhase, initNewMilestone, initQuick,
initIngestDocs, initResume, initVerifyWork, initPhaseOp, initTodos,
initMilestoneOp, initMapCodebase, initNewWorkspace,
initListWorkspaces, initRemoveWorkspace,
} from './composer.js';
import { initNewProject, initProgress, initManager } from './complex.js';
export const INIT_FAMILY_HANDLERS: Readonly<Record<string, QueryHandler>> = {
'init.execute-phase': initExecutePhase,
'init.plan-phase': initPlanPhase,
'init.new-project': initNewProject,
'init.new-milestone': initNewMilestone,
'init.quick': initQuick,
'init.ingest-docs': initIngestDocs,
'init.resume': initResume,
'init.verify-work': initVerifyWork,
'init.phase-op': initPhaseOp,
'init.todos': initTodos,
'init.milestone-op': initMilestoneOp,
'init.map-codebase': initMapCodebase,
'init.progress': initProgress,
'init.manager': initManager,
'init.new-workspace': initNewWorkspace,
'init.list-workspaces': initListWorkspaces,
'init.remove-workspace': initRemoveWorkspace,
};

View File

@@ -1,20 +0,0 @@
import type { QueryHandler } from '../../query/utils.js';
import { phaseListPlans, phaseListArtifacts } from '../../query/phase-list-queries.js';
import { phaseUatPassed } from '../../query/phase-uat-passed.js';
import {
phaseAdd, phaseAddBatch, phaseInsert, phaseRemove, phaseComplete,
phaseScaffold, phaseNextDecimal,
} from '../../query/phase-lifecycle.js';
export const PHASE_FAMILY_HANDLERS: Readonly<Record<string, QueryHandler>> = {
'phase.list-plans': phaseListPlans,
'phase.list-artifacts': phaseListArtifacts,
'phase.uat-passed': phaseUatPassed,
'phase.add': phaseAdd,
'phase.add-batch': phaseAddBatch,
'phase.insert': phaseInsert,
'phase.remove': phaseRemove,
'phase.complete': phaseComplete,
'phase.scaffold': phaseScaffold,
'phase.next-decimal': phaseNextDecimal,
};

View File

@@ -1,8 +0,0 @@
import type { QueryHandler } from '../../query/utils.js';
import { phasesList, phasesClear, phasesArchive } from '../../query/phase-lifecycle.js';
export const PHASES_FAMILY_HANDLERS: Readonly<Record<string, QueryHandler>> = {
'phases.list': phasesList,
'phases.clear': phasesClear,
'phases.archive': phasesArchive,
};

View File

@@ -1,10 +0,0 @@
import type { QueryHandler } from '../../query/utils.js';
import { roadmapAnalyze, roadmapGetPhase, roadmapAnnotateDependencies } from '../../query/roadmap.js';
import { roadmapUpdatePlanProgress } from '../../query/roadmap-update-plan-progress.js';
export const ROADMAP_FAMILY_HANDLERS: Readonly<Record<string, QueryHandler>> = {
'roadmap.analyze': roadmapAnalyze,
'roadmap.get-phase': roadmapGetPhase,
'roadmap.update-plan-progress': roadmapUpdatePlanProgress,
'roadmap.annotate-dependencies': roadmapAnnotateDependencies,
};

View File

@@ -1,35 +0,0 @@
import type { QueryHandler } from '../../query/utils.js';
import { stateProjectLoad } from '../../query/state-project-load.js';
import { stateJson, stateGet } from '../../query/state.js';
import {
stateUpdate, statePatch, stateBeginPhase, stateAdvancePlan,
stateRecordMetric, stateUpdateProgress, stateAddDecision,
stateAddBlocker, stateResolveBlocker, stateRecordSession,
stateSignalWaiting, stateSignalResume, statePlannedPhase,
stateValidate, stateSync, statePrune, stateMilestoneSwitch,
stateAddRoadmapEvolution,
} from '../../query/state-mutation.js';
export const STATE_FAMILY_HANDLERS: Readonly<Record<string, QueryHandler>> = {
'state.load': stateProjectLoad,
'state.json': stateJson,
'state.get': stateGet,
'state.update': stateUpdate,
'state.patch': statePatch,
'state.begin-phase': stateBeginPhase,
'state.advance-plan': stateAdvancePlan,
'state.record-metric': stateRecordMetric,
'state.update-progress': stateUpdateProgress,
'state.add-decision': stateAddDecision,
'state.add-blocker': stateAddBlocker,
'state.resolve-blocker': stateResolveBlocker,
'state.record-session': stateRecordSession,
'state.signal-waiting': stateSignalWaiting,
'state.signal-resume': stateSignalResume,
'state.planned-phase': statePlannedPhase,
'state.validate': stateValidate,
'state.sync': stateSync,
'state.prune': statePrune,
'state.milestone-switch': stateMilestoneSwitch,
'state.add-roadmap-evolution': stateAddRoadmapEvolution,
};

View File

@@ -1,9 +0,0 @@
import type { QueryHandler } from '../../query/utils.js';
import { validateConsistency, validateHealth, validateAgents, validateContext } from '../../query/validate.js';
export const VALIDATE_FAMILY_HANDLERS: Readonly<Record<string, QueryHandler>> = {
'validate.consistency': validateConsistency,
'validate.health': validateHealth,
'validate.agents': validateAgents,
'validate.context': validateContext,
};

View File

@@ -1,18 +0,0 @@
import type { QueryHandler } from '../../query/utils.js';
import {
verifyPlanStructure, verifyPhaseCompleteness, verifyReferences,
verifyCommits, verifyArtifacts, verifySchemaDrift,
} from '../../query/verify.js';
import { verifyKeyLinks } from '../../query/validate.js';
export const VERIFY_FAMILY_HANDLERS: Readonly<Record<string, QueryHandler>> = {
'verify.plan-structure': verifyPlanStructure,
'verify.phase-completeness': verifyPhaseCompleteness,
'verify.references': verifyReferences,
'verify.commits': verifyCommits,
'verify.artifacts': verifyArtifacts,
'verify.key-links': verifyKeyLinks,
'verify.schema-drift': verifySchemaDrift,
// 'verify.codebase-drift' intentionally omitted — out-of-seam CJS-only
// per ADR/PRD 3524 §3 / L160. Router dispatches direct to CJS handler.
};

View File

@@ -1,366 +0,0 @@
/**
* GSD SDK — Public API for running GSD plans programmatically.
*
* The GSD class composes plan parsing, config loading, prompt building,
* and session running into a single `executePlan()` call.
*
* @example
* ```typescript
* import { GSD } from '@opengsd/gsd-sdk';
*
* const gsd = new GSD({ projectDir: '/path/to/project' });
* const result = await gsd.executePlan('.planning/phases/01-auth/01-auth-01-PLAN.md');
*
* if (result.success) {
* console.log(`Plan completed in ${result.durationMs}ms, cost: $${result.totalCostUsd}`);
* } else {
* console.error(`Plan failed: ${result.error?.messages.join(', ')}`);
* }
* ```
*/
import { readFile } from 'node:fs/promises';
import { join, resolve } from 'node:path';
import { homedir } from 'node:os';
import type { GSDOptions, PlanResult, SessionOptions, GSDEvent, TransportHandler, PhaseRunnerOptions, PhaseRunnerResult, MilestoneRunnerOptions, MilestoneRunnerResult, RoadmapPhaseInfo } from './types.js';
import { GSDEventType } from './types.js';
import { parsePlan, parsePlanFile } from './plan-parser.js';
import { loadConfig } from './config.js';
import { GSDTools, resolveGsdToolsPath } from './gsd-tools.js';
import { runPlanSession } from './session-runner.js';
import { buildExecutorPrompt, parseAgentTools } from './prompt-builder.js';
import { GSDEventStream } from './event-stream.js';
import { PhaseRunner } from './phase-runner.js';
import { ContextEngine } from './context-engine.js';
import { PromptFactory } from './phase-prompt.js';
export { PlanningJournal } from './planning-journal.js';
export type { PlanningEvent, PlanningEventActor, PlanningJournalAppendInput } from './planning-journal.js';
export { PlanningRuntime } from './planning-runtime.js';
// ─── GSD class ───────────────────────────────────────────────────────────────
export class GSD {
private readonly projectDir: string;
private readonly gsdToolsPath: string;
private readonly sessionId?: string;
private readonly defaultModel?: string;
private readonly defaultMaxBudgetUsd: number;
private readonly defaultMaxTurns: number;
private readonly autoMode: boolean;
private readonly workstream?: string;
private readonly strictSdk?: boolean;
private readonly allowFallbackToSubprocess?: boolean;
readonly eventStream: GSDEventStream;
constructor(options: GSDOptions) {
this.projectDir = resolve(options.projectDir);
this.gsdToolsPath =
options.gsdToolsPath ?? resolveGsdToolsPath(this.projectDir);
this.sessionId = options.sessionId;
this.defaultModel = options.model;
this.defaultMaxBudgetUsd = options.maxBudgetUsd ?? 5.0;
this.defaultMaxTurns = options.maxTurns ?? 50;
this.autoMode = options.autoMode ?? false;
this.workstream = options.workstream;
this.strictSdk = options.strictSdk;
this.allowFallbackToSubprocess = options.allowFallbackToSubprocess;
this.eventStream = new GSDEventStream();
}
/**
* Execute a single GSD plan file.
*
* Reads the plan from disk, parses it, loads project config,
* optionally reads the agent definition, then runs a query() session.
*
* @param planPath - Path to the PLAN.md file (absolute or relative to projectDir)
* @param options - Per-execution overrides
* @returns PlanResult with cost, duration, success/error status
*/
async executePlan(planPath: string, options?: SessionOptions): Promise<PlanResult> {
// Resolve plan path relative to project dir
const absolutePlanPath = resolve(this.projectDir, planPath);
// Parse the plan
const plan = await parsePlanFile(absolutePlanPath);
// Load project config
const config = await loadConfig(this.projectDir, this.workstream);
// Try to load agent definition for tool restrictions
const agentDef = await this.loadAgentDefinition();
// Merge defaults with per-call options
const sessionOptions: SessionOptions = {
maxTurns: options?.maxTurns ?? this.defaultMaxTurns,
maxBudgetUsd: options?.maxBudgetUsd ?? this.defaultMaxBudgetUsd,
model: options?.model ?? this.defaultModel,
cwd: options?.cwd ?? this.projectDir,
allowedTools: options?.allowedTools,
};
return runPlanSession(plan, config, sessionOptions, agentDef, this.eventStream, {
phase: undefined, // Phase context set by higher-level orchestrators
planName: plan.frontmatter.plan,
});
}
/**
* Subscribe a simple handler to receive all GSD events.
*/
onEvent(handler: (event: GSDEvent) => void): void {
this.eventStream.on('event', handler);
}
/**
* Subscribe a transport handler to receive all GSD events.
* Transports provide structured onEvent/close lifecycle.
*/
addTransport(handler: TransportHandler): void {
this.eventStream.addTransport(handler);
}
/**
* Create a GSDTools instance for state management operations.
*/
createTools(): GSDTools {
return new GSDTools({
projectDir: this.projectDir,
gsdToolsPath: this.gsdToolsPath,
workstream: this.workstream,
eventStream: this.eventStream,
sessionId: this.sessionId,
strictSdk: this.strictSdk,
allowFallbackToSubprocess: this.allowFallbackToSubprocess,
onDispatchEvent: (event) => {
this.eventStream.emitEvent({
type: GSDEventType.StreamEvent,
timestamp: new Date().toISOString(),
sessionId: this.sessionId ?? '',
event,
});
},
});
}
/**
* Run a full phase lifecycle: discuss → research → plan → execute → verify → advance.
*
* Creates the necessary collaborators (GSDTools, PromptFactory, ContextEngine),
* loads project config, instantiates a PhaseRunner, and delegates to `runner.run()`.
*
* @param phaseNumber - The phase number to execute (e.g. "01", "02")
* @param options - Per-phase overrides for budget, turns, model, and callbacks
* @returns PhaseRunnerResult with per-step results, overall success, cost, and timing
*/
async runPhase(phaseNumber: string, options?: PhaseRunnerOptions): Promise<PhaseRunnerResult> {
const tools = this.createTools();
const promptFactory = new PromptFactory({ projectDir: this.projectDir });
const contextEngine = new ContextEngine(this.projectDir, undefined, undefined, this.workstream);
const config = await loadConfig(this.projectDir, this.workstream);
// Auto mode: force auto_advance on and skip_discuss off so self-discuss kicks in
if (this.autoMode) {
config.workflow.auto_advance = true;
config.workflow.skip_discuss = false;
}
const runner = new PhaseRunner({
projectDir: this.projectDir,
tools,
promptFactory,
contextEngine,
eventStream: this.eventStream,
config,
});
return runner.run(phaseNumber, options);
}
/**
* Run a full milestone: discover phases, execute each incomplete one in order,
* re-discover after each completion to catch dynamically inserted phases.
*
* @param prompt - The user prompt describing the milestone goal
* @param options - Per-milestone overrides for budget, turns, model, and callbacks
* @returns MilestoneRunnerResult with per-phase results, overall success, cost, and timing
*/
async run(prompt: string, options?: MilestoneRunnerOptions): Promise<MilestoneRunnerResult> {
const tools = this.createTools();
const startTime = Date.now();
const phaseResults: PhaseRunnerResult[] = [];
let success = true;
// Discover initial phases
const initialAnalysis = await tools.roadmapAnalyze();
const incompletePhases = this.filterAndSortPhases(initialAnalysis.phases);
// Emit MilestoneStart
this.eventStream.emitEvent({
type: GSDEventType.MilestoneStart,
timestamp: new Date().toISOString(),
sessionId: `milestone-${Date.now()}`,
phaseCount: incompletePhases.length,
prompt,
});
// Loop through phases, re-discovering after each completion
let currentPhases = incompletePhases;
while (currentPhases.length > 0) {
const phase = currentPhases[0];
try {
const result = await this.runPhase(phase.number, options);
phaseResults.push(result);
if (!result.success) {
success = false;
break;
}
// Notify callback if present; stop if requested
if (options?.onPhaseComplete) {
const verdict = await options.onPhaseComplete(result, phase);
if (verdict === 'stop') {
break;
}
}
// Re-discover phases to catch dynamically inserted ones
const updatedAnalysis = await tools.roadmapAnalyze();
currentPhases = this.filterAndSortPhases(updatedAnalysis.phases);
} catch (err) {
// Phase threw an unexpected error — record as failure and stop
phaseResults.push({
phaseNumber: phase.number,
phaseName: phase.phase_name,
steps: [],
success: false,
totalCostUsd: 0,
totalDurationMs: 0,
});
success = false;
break;
}
}
const totalCostUsd = phaseResults.reduce((sum, r) => sum + r.totalCostUsd, 0);
const totalDurationMs = Date.now() - startTime;
// Emit MilestoneComplete
this.eventStream.emitEvent({
type: GSDEventType.MilestoneComplete,
timestamp: new Date().toISOString(),
sessionId: `milestone-${Date.now()}`,
success,
totalCostUsd,
totalDurationMs,
phasesCompleted: phaseResults.filter(r => r.success).length,
});
return {
success,
phases: phaseResults,
totalCostUsd,
totalDurationMs,
};
}
/**
* Filter to incomplete phases and sort numerically.
* Uses parseFloat to handle decimal phase numbers (e.g. '5.1').
*/
private filterAndSortPhases(phases: RoadmapPhaseInfo[]): RoadmapPhaseInfo[] {
return phases
.filter(p => !p.roadmap_complete)
.sort((a, b) => parseFloat(a.number) - parseFloat(b.number));
}
/**
* Load the gsd-executor agent definition if available.
* Falls back gracefully — returns undefined if not found.
*/
private async loadAgentDefinition(): Promise<string | undefined> {
const paths = [
// Repo-local GSD installation
join(this.projectDir, '.claude', 'get-shit-done', 'agents', 'gsd-executor.md'),
// Repo-local agents directory
join(this.projectDir, '.claude', 'agents', 'gsd-executor.md'),
// Global home directory
join(homedir(), '.claude', 'agents', 'gsd-executor.md'),
join(this.projectDir, 'agents', 'gsd-executor.md'),
];
for (const p of paths) {
try {
return await readFile(p, 'utf-8');
} catch {
// Not found at this path, try next
}
}
return undefined;
}
}
// ─── Re-exports for advanced usage ──────────────────────────────────────────
export { parsePlan, parsePlanFile } from './plan-parser.js';
export { loadConfig } from './config.js';
export type { GSDConfig } from './config.js';
export { GSDTools, GSDToolsError, resolveGsdToolsPath } from './gsd-tools.js';
export { runPlanSession, runPhaseStepSession } from './session-runner.js';
export { buildExecutorPrompt, parseAgentTools } from './prompt-builder.js';
export type { ExecutorPromptOptions } from './prompt-builder.js';
export * from './types.js';
// S02: Event stream, context, prompt, and logging modules
export { GSDEventStream } from './event-stream.js';
export type { EventStreamContext } from './event-stream.js';
export { ContextEngine, PHASE_FILE_MANIFEST } from './context-engine.js';
export type { FileSpec } from './context-engine.js';
export { truncateMarkdown, extractCurrentMilestone, DEFAULT_TRUNCATION_OPTIONS } from './context-truncation.js';
export type { TruncationOptions } from './context-truncation.js';
export { getToolsForPhase, PHASE_AGENT_MAP, PHASE_DEFAULT_TOOLS } from './tool-scoping.js';
export { checkResearchGate } from './research-gate.js';
export type { ResearchGateResult } from './research-gate.js';
export { PromptFactory, extractBlock, extractSteps, PHASE_WORKFLOW_MAP } from './phase-prompt.js';
export { GSDLogger } from './logger.js';
export type { LogLevel, LogEntry, GSDLoggerOptions } from './logger.js';
// S03: Phase lifecycle state machine
export { PhaseRunner, PhaseRunnerError } from './phase-runner.js';
export type { PhaseRunnerDeps, VerificationOutcome } from './phase-runner.js';
// S05: Transports
export { CLITransport } from './cli-transport.js';
export { WSTransport } from './ws-transport.js';
export type { WSTransportOptions } from './ws-transport.js';
// Query registry argv normalization (matches `gsd-sdk query` and `GSDTools` hot path)
export { createRegistry, normalizeQueryCommand } from './query/index.js';
// Phase UAT predicate — programmatic API surface (#3184)
export {
isPhaseUatPassed,
phaseUatPassed,
REASON_CODE,
ERROR_CODE,
PhaseUatPassedError,
} from './query/phase-uat-passed.js';
export type {
UatReason,
ReasonCode,
ErrorCode,
} from './query/phase-uat-passed.js';
// Workstream utilities
export { validateWorkstreamName, relPlanningPath } from './workstream-utils.js';
// Init workflow
export { InitRunner } from './init-runner.js';
export type { InitRunnerDeps } from './init-runner.js';
export type { InitConfig, InitResult, InitStepResult, InitStepName } from './types.js';

View File

@@ -1,138 +0,0 @@
/**
* E2E integration test — proves InitRunner.run() drives real Agent SDK
* sessions for the gsd-sdk init workflow.
*
* Requires Claude Code CLI (`claude`) installed and authenticated.
* Skips gracefully if CLI is unavailable.
*
* This test proves the headless init pipeline can bootstrap a real project
* without human intervention: setup → config → PROJECT.md → research →
* synthesis → requirements → roadmap.
*/
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
import { execSync } from 'node:child_process';
import { mkdtemp, rm, readFile, stat } from 'node:fs/promises';
import { existsSync } from 'node:fs';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';
import { InitRunner } from './init-runner.js';
import { GSDTools, resolveGsdToolsPath } from './gsd-tools.js';
import { GSDEventStream } from './event-stream.js';
import { GSDEventType } from './types.js';
import type { GSDEvent } from './types.js';
// ─── CLI availability check ─────────────────────────────────────────────────
let cliAvailable = false;
try {
execSync('which claude', { stdio: 'ignore' });
cliAvailable = true;
} catch {
cliAvailable = false;
}
const e2eEnabled = process.env.GSD_ENABLE_E2E === '1';
const __dirname = fileURLToPath(new URL('.', import.meta.url));
const sdkPromptsDir = join(__dirname, '..', 'prompts');
const GSD_TOOLS_PATH = resolveGsdToolsPath(process.cwd());
const gsdToolsAvailable = existsSync(GSD_TOOLS_PATH);
// ─── Test suite ──────────────────────────────────────────────────────────────
describe.skipIf(!cliAvailable || !gsdToolsAvailable || !e2eEnabled)('E2E: InitRunner.run() full workflow', () => {
let tmpDir: string;
let events: GSDEvent[];
beforeAll(async () => {
tmpDir = await mkdtemp(join(tmpdir(), 'gsd-sdk-init-e2e-'));
// Initialize git in the temp dir (required by InitRunner)
execSync('git init', { cwd: tmpDir, stdio: 'ignore' });
execSync('git config user.email "test@test.com"', { cwd: tmpDir, stdio: 'ignore' });
execSync('git config user.name "Test"', { cwd: tmpDir, stdio: 'ignore' });
}, 30_000);
afterAll(async () => {
if (tmpDir) {
await rm(tmpDir, { recursive: true, force: true });
}
});
it('InitRunner.run() bootstraps a project without human intervention', async () => {
events = [];
const eventStream = new GSDEventStream();
eventStream.on('event', (e: GSDEvent) => events.push(e));
const tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: GSD_TOOLS_PATH,
timeoutMs: 30_000,
});
const runner = new InitRunner({
projectDir: tmpDir,
tools,
eventStream,
config: {
maxBudgetPerSession: 1.0,
maxTurnsPerSession: 15,
},
sdkPromptsDir,
});
const result = await runner.run('Build a CLI tool that prints hello world');
// ── Assert: pipeline executed (success OR at least 3+ steps completed) ──
const completedSteps = result.steps.filter(s => s.success);
const pipelineProgressed = result.success || completedSteps.length >= 3;
expect(pipelineProgressed).toBe(true);
// ── Assert: config.json artifact created ──
// config.json is written directly by InitRunner (not by Claude session)
// so it should always exist if the config step succeeded
const configStep = result.steps.find(s => s.step === 'config');
if (configStep?.success) {
const configPath = join(tmpDir, '.planning', 'config.json');
const configStat = await stat(configPath).catch(() => null);
expect(configStat).not.toBeNull();
if (configStat) {
const configContent = JSON.parse(await readFile(configPath, 'utf-8'));
expect(configContent.workflow.auto_advance).toBe(true);
}
}
// ── Assert: PROJECT.md created if project step succeeded ──
const projectStep = result.steps.find(s => s.step === 'project');
if (projectStep?.success) {
const projectPath = join(tmpDir, '.planning', 'PROJECT.md');
const projectStat = await stat(projectPath).catch(() => null);
expect(projectStat).not.toBeNull();
}
// ── Assert: events captured include InitStart and at least one InitStepComplete ──
const initStartEvents = events.filter(e => e.type === GSDEventType.InitStart);
expect(initStartEvents.length).toBe(1);
const stepCompleteEvents = events.filter(e => e.type === GSDEventType.InitStepComplete);
expect(stepCompleteEvents.length).toBeGreaterThanOrEqual(1);
// ── Assert: InitComplete event emitted ──
const initCompleteEvents = events.filter(e => e.type === GSDEventType.InitComplete);
expect(initCompleteEvents.length).toBe(1);
// ── Assert: cost and duration are tracked ──
expect(result.totalDurationMs).toBeGreaterThan(0);
expect(typeof result.totalCostUsd).toBe('number');
// ── Assert: artifacts list is populated ──
if (result.success) {
expect(result.artifacts.length).toBeGreaterThan(0);
expect(result.artifacts).toContain('.planning/config.json');
}
}, 600_000); // 10 minute timeout for the full 7-session init workflow
});

View File

@@ -1,740 +0,0 @@
import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest';
import { mkdir, writeFile, rm, readFile } from 'node:fs/promises';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import { InitRunner } from './init-runner.js';
import type { InitRunnerDeps } from './init-runner.js';
import type {
PlanResult,
SessionUsage,
GSDEvent,
InitNewProjectInfo,
InitStepResult,
} from './types.js';
import { GSDEventType } from './types.js';
// ─── Mock modules ────────────────────────────────────────────────────────────
// Mock session-runner to avoid real SDK calls
vi.mock('./session-runner.js', () => ({
runPhaseStepSession: vi.fn(),
runPlanSession: vi.fn(),
}));
// Mock config loader
vi.mock('./config.js', () => ({
loadConfig: vi.fn().mockResolvedValue({
mode: 'yolo',
model_profile: 'balanced',
}),
CONFIG_DEFAULTS: {},
}));
// Mock fs/promises for template reading (InitRunner reads GSD templates)
// We partially mock — only readFile needs interception for template paths
const originalReadFile = vi.importActual('node:fs/promises').then(m => (m as typeof import('node:fs/promises')).readFile);
import { runPhaseStepSession } from './session-runner.js';
const mockRunSession = vi.mocked(runPhaseStepSession);
// ─── Factory helpers ─────────────────────────────────────────────────────────
function makeUsage(): SessionUsage {
return {
inputTokens: 1000,
outputTokens: 500,
cacheReadInputTokens: 0,
cacheCreationInputTokens: 0,
};
}
function makeSuccessResult(overrides: Partial<PlanResult> = {}): PlanResult {
return {
success: true,
sessionId: `sess-${Date.now()}`,
totalCostUsd: 0.05,
durationMs: 2000,
usage: makeUsage(),
numTurns: 10,
...overrides,
};
}
function makeErrorResult(overrides: Partial<PlanResult> = {}): PlanResult {
return {
success: false,
sessionId: `sess-err-${Date.now()}`,
totalCostUsd: 0.01,
durationMs: 500,
usage: makeUsage(),
numTurns: 2,
error: {
subtype: 'error_during_execution',
messages: ['Session failed'],
},
...overrides,
};
}
function makeProjectInfo(overrides: Partial<InitNewProjectInfo> = {}): InitNewProjectInfo {
return {
researcher_model: 'claude-sonnet-4-6',
synthesizer_model: 'claude-sonnet-4-6',
roadmapper_model: 'claude-sonnet-4-6',
commit_docs: false, // false for tests — no git operations
project_exists: false,
has_codebase_map: false,
planning_exists: false,
has_existing_code: false,
has_package_file: false,
is_brownfield: false,
needs_codebase_map: false,
has_git: true, // skip git init in tests
brave_search_available: false,
firecrawl_available: false,
exa_search_available: false,
project_path: '.planning/PROJECT.md',
...overrides,
};
}
function makeTools(overrides: Record<string, unknown> = {}) {
return {
initNewProject: vi.fn().mockResolvedValue(makeProjectInfo()),
configSet: vi.fn().mockResolvedValue(undefined),
commit: vi.fn().mockResolvedValue(undefined),
exec: vi.fn(),
stateLoad: vi.fn(),
roadmapAnalyze: vi.fn(),
phaseComplete: vi.fn(),
verifySummary: vi.fn(),
initExecutePhase: vi.fn(),
initPhaseOp: vi.fn(),
configGet: vi.fn(),
stateBeginPhase: vi.fn(),
phasePlanIndex: vi.fn(),
...overrides,
} as any;
}
function makeEventStream() {
const events: GSDEvent[] = [];
return {
emitEvent: vi.fn((event: GSDEvent) => events.push(event)),
on: vi.fn(),
emit: vi.fn(),
addTransport: vi.fn(),
events,
} as any;
}
function makeDeps(overrides: Partial<InitRunnerDeps> & { tmpDir: string }): InitRunnerDeps & { events: GSDEvent[] } {
const tools = makeTools();
const eventStream = makeEventStream();
return {
projectDir: overrides.tmpDir,
tools: overrides.tools ?? tools,
eventStream: overrides.eventStream ?? eventStream,
config: overrides.config,
events: eventStream.events,
...(overrides.tools ? {} : {}),
};
}
// ─── Test suite ──────────────────────────────────────────────────────────────
describe('InitRunner', () => {
let tmpDir: string;
beforeEach(async () => {
tmpDir = join(tmpdir(), `init-runner-test-${Date.now()}-${Math.random().toString(36).slice(2)}`);
await mkdir(tmpDir, { recursive: true });
vi.clearAllMocks();
// Default: all sessions succeed
mockRunSession.mockResolvedValue(makeSuccessResult());
});
afterEach(async () => {
await rm(tmpDir, { recursive: true, force: true });
});
// ─── Helpers ─────────────────────────────────────────────────────────────
function createRunner(toolsOverrides: Record<string, unknown> = {}, configOverrides?: Partial<InitRunnerDeps['config']>) {
const tools = makeTools(toolsOverrides);
const eventStream = makeEventStream();
const runner = new InitRunner({
projectDir: tmpDir,
tools,
eventStream,
config: configOverrides as any,
});
return { runner, tools, eventStream, events: eventStream.events as GSDEvent[] };
}
// ─── Core workflow tests ─────────────────────────────────────────────────
it('run() calls initNewProject and validates project_exists === false', async () => {
const { runner, tools } = createRunner();
await runner.run('build a todo app');
expect(tools.initNewProject).toHaveBeenCalledOnce();
});
it('run() returns error result when initNewProject reports project_exists', async () => {
const { runner, tools } = createRunner({
initNewProject: vi.fn().mockResolvedValue(makeProjectInfo({ project_exists: true })),
});
const result = await runner.run('build a todo app');
expect(result.success).toBe(false);
// The setup step should have failed
const setupStep = result.steps.find(s => s.step === 'setup');
expect(setupStep).toBeDefined();
expect(setupStep!.success).toBe(false);
expect(setupStep!.error).toContain('already exists');
});
it('run() writes config.json with auto-mode defaults', async () => {
const { runner } = createRunner();
await runner.run('build a todo app');
// config.json should be written to .planning/config.json in tmpDir
const configPath = join(tmpDir, '.planning', 'config.json');
const content = await readFile(configPath, 'utf-8');
const parsed = JSON.parse(content);
expect(parsed.mode).toBe('yolo');
expect(parsed.parallelization).toBe(true);
expect(parsed.workflow.auto_advance).toBe(true);
});
it('run() calls configSet for auto_advance', async () => {
const { runner, tools } = createRunner();
await runner.run('build a todo app');
expect(tools.configSet).toHaveBeenCalledWith('workflow.auto_advance', 'true');
});
it('run() spawns PROJECT.md synthesis session', async () => {
const { runner } = createRunner();
await runner.run('build a todo app');
// The third session call should be the PROJECT.md synthesis
// Calls: setup (no session), config (no session), project (1st session),
// 4x research, synthesis, requirements, roadmap
// Total: 8 runPhaseStepSession calls
expect(mockRunSession).toHaveBeenCalled();
// First call should be for PROJECT.md (step 3)
const firstCall = mockRunSession.mock.calls[0];
expect(firstCall).toBeDefined();
const prompt = firstCall![0] as string;
expect(prompt).toContain('PROJECT.md');
});
it('run() spawns 4 parallel research sessions via Promise.allSettled', async () => {
const { runner } = createRunner();
await runner.run('build a todo app');
// Count calls that contain the specific "researching the X aspect" pattern
// which uniquely identifies research prompts (vs synthesis/requirements that reference research files)
const researchCalls = mockRunSession.mock.calls.filter(call => {
const prompt = call[0] as string;
return prompt.includes('You are researching the');
});
// Should be exactly 4 research sessions
expect(researchCalls.length).toBe(4);
});
it('run() spawns synthesis session after research completes', async () => {
const { runner } = createRunner();
await runner.run('build a todo app');
// Synthesis call should contain 'Synthesize' or 'SUMMARY'
const synthesisCalls = mockRunSession.mock.calls.filter(call => {
const prompt = call[0] as string;
return prompt.includes('Synthesize') || prompt.includes('SUMMARY.md');
});
expect(synthesisCalls.length).toBeGreaterThanOrEqual(1);
});
it('run() spawns requirements session', async () => {
const { runner } = createRunner();
await runner.run('build a todo app');
const reqCalls = mockRunSession.mock.calls.filter(call => {
const prompt = call[0] as string;
return prompt.includes('REQUIREMENTS.md');
});
expect(reqCalls.length).toBeGreaterThanOrEqual(1);
});
it('run() spawns roadmapper session', async () => {
const { runner } = createRunner();
await runner.run('build a todo app');
const roadmapCalls = mockRunSession.mock.calls.filter(call => {
const prompt = call[0] as string;
return prompt.includes('ROADMAP.md') || prompt.includes('STATE.md');
});
expect(roadmapCalls.length).toBeGreaterThanOrEqual(1);
});
it('run() calls commit after each major step when commit_docs is true', async () => {
const commitFn = vi.fn().mockResolvedValue(undefined);
const { runner } = createRunner({
initNewProject: vi.fn().mockResolvedValue(makeProjectInfo({ commit_docs: true })),
commit: commitFn,
});
await runner.run('build a todo app');
// Should commit: config, PROJECT.md, research, REQUIREMENTS.md, ROADMAP+STATE
expect(commitFn).toHaveBeenCalled();
expect(commitFn.mock.calls.length).toBeGreaterThanOrEqual(4);
});
it('run() does not call commit when commit_docs is false', async () => {
const commitFn = vi.fn().mockResolvedValue(undefined);
const { runner } = createRunner({
initNewProject: vi.fn().mockResolvedValue(makeProjectInfo({ commit_docs: false })),
commit: commitFn,
});
await runner.run('build a todo app');
expect(commitFn).not.toHaveBeenCalled();
});
// ─── Event emission tests ────────────────────────────────────────────────
it('run() emits InitStart and InitComplete events', async () => {
const { runner, events } = createRunner();
await runner.run('build a todo app');
const startEvents = events.filter(e => e.type === GSDEventType.InitStart);
const completeEvents = events.filter(e => e.type === GSDEventType.InitComplete);
expect(startEvents.length).toBe(1);
expect(completeEvents.length).toBe(1);
const start = startEvents[0] as any;
expect(start.projectDir).toBe(tmpDir);
expect(start.input).toBeTruthy();
const complete = completeEvents[0] as any;
expect(complete.success).toBe(true);
expect(complete.totalCostUsd).toBeTypeOf('number');
expect(complete.totalDurationMs).toBeTypeOf('number');
expect(complete.artifactCount).toBeGreaterThan(0);
});
it('run() emits InitStepStart/Complete for each step', async () => {
const { runner, events } = createRunner();
await runner.run('build a todo app');
const stepStarts = events.filter(e => e.type === GSDEventType.InitStepStart);
const stepCompletes = events.filter(e => e.type === GSDEventType.InitStepComplete);
// Steps: setup, config, project, 4x research, synthesis, requirements, roadmap = 10
expect(stepStarts.length).toBe(10);
expect(stepCompletes.length).toBe(10);
// Verify each step start has a matching complete (order may vary for parallel research)
const startSteps = stepStarts.map(e => (e as any).step).sort();
const completeSteps = stepCompletes.map(e => (e as any).step).sort();
expect(startSteps).toEqual(completeSteps);
// Verify expected step names are present
expect(startSteps).toContain('setup');
expect(startSteps).toContain('config');
expect(startSteps).toContain('project');
expect(startSteps).toContain('research-stack');
expect(startSteps).toContain('research-features');
expect(startSteps).toContain('research-architecture');
expect(startSteps).toContain('research-pitfalls');
expect(startSteps).toContain('synthesis');
expect(startSteps).toContain('requirements');
expect(startSteps).toContain('roadmap');
});
it('run() emits InitResearchSpawn before research sessions', async () => {
const { runner, events } = createRunner();
await runner.run('build a todo app');
const spawnEvents = events.filter(e => e.type === GSDEventType.InitResearchSpawn);
expect(spawnEvents.length).toBe(1);
const spawn = spawnEvents[0] as any;
expect(spawn.sessionCount).toBe(4);
expect(spawn.researchTypes).toEqual(['STACK', 'FEATURES', 'ARCHITECTURE', 'PITFALLS']);
});
// ─── Error handling tests ────────────────────────────────────────────────
it('run() returns error when a session fails (partial research success)', async () => {
// Make the STACK research session fail, others succeed
let callCount = 0;
mockRunSession.mockImplementation(async (prompt: string) => {
callCount++;
// First call is PROJECT.md, then 4 research calls
// The 2nd call overall (1st research) should fail
if (callCount === 2) {
return makeErrorResult();
}
return makeSuccessResult();
});
const { runner } = createRunner();
const result = await runner.run('build a todo app');
// Should still complete (partial success allowed for research)
// but overall result indicates research failure
expect(result.success).toBe(false);
// Steps should still exist for all phases
expect(result.steps.length).toBeGreaterThanOrEqual(7);
});
it('run() stops workflow when PROJECT.md synthesis fails', async () => {
// First session (PROJECT.md) fails
mockRunSession.mockResolvedValueOnce(makeErrorResult());
const { runner } = createRunner();
const result = await runner.run('build a todo app');
expect(result.success).toBe(false);
// Should have setup, config, and project steps only
const stepNames = result.steps.map(s => s.step);
expect(stepNames).toContain('setup');
expect(stepNames).toContain('config');
expect(stepNames).toContain('project');
// Should NOT continue to research
expect(stepNames).not.toContain('research-stack');
});
it('run() stops workflow when requirements session fails', async () => {
// Let PROJECT.md and research succeed, but make requirements fail
let sessionCallIndex = 0;
mockRunSession.mockImplementation(async () => {
sessionCallIndex++;
// Calls: 1=PROJECT.md, 2-5=research, 6=synthesis, 7=requirements
if (sessionCallIndex === 7) {
return makeErrorResult();
}
return makeSuccessResult();
});
const { runner } = createRunner();
const result = await runner.run('build a todo app');
expect(result.success).toBe(false);
const stepNames = result.steps.map(s => s.step);
expect(stepNames).toContain('requirements');
// Should NOT continue to roadmap
expect(stepNames).not.toContain('roadmap');
});
// ─── Cost aggregation tests ──────────────────────────────────────────────
it('run() aggregates costs from all sessions', async () => {
const costPerSession = 0.05;
mockRunSession.mockResolvedValue(makeSuccessResult({ totalCostUsd: costPerSession }));
const { runner } = createRunner();
const result = await runner.run('build a todo app');
// 8 total sessions: PROJECT.md + 4 research + synthesis + requirements + roadmap
// Cost from sessions extracted via extractCost, non-session steps (setup/config) are 0
expect(result.totalCostUsd).toBeGreaterThan(0);
expect(result.totalDurationMs).toBeGreaterThan(0);
});
// ─── Artifact tracking tests ─────────────────────────────────────────────
it('run() returns all expected artifacts on success', async () => {
const { runner } = createRunner();
const result = await runner.run('build a todo app');
expect(result.success).toBe(true);
expect(result.artifacts).toContain('.planning/config.json');
expect(result.artifacts).toContain('.planning/PROJECT.md');
expect(result.artifacts).toContain('.planning/research/SUMMARY.md');
expect(result.artifacts).toContain('.planning/REQUIREMENTS.md');
expect(result.artifacts).toContain('.planning/ROADMAP.md');
expect(result.artifacts).toContain('.planning/STATE.md');
});
it('run() includes research artifact paths on success', async () => {
const { runner } = createRunner();
const result = await runner.run('build a todo app');
expect(result.artifacts).toContain('.planning/research/STACK.md');
expect(result.artifacts).toContain('.planning/research/FEATURES.md');
expect(result.artifacts).toContain('.planning/research/ARCHITECTURE.md');
expect(result.artifacts).toContain('.planning/research/PITFALLS.md');
});
// ─── Git init test ─────────────────────────────────────────────────────
it('run() initializes git when has_git is false', async () => {
// We can't easily test git init without mocking execFile deeply,
// but we can verify the tools.initNewProject is called with the result
// and that the workflow continues. Since has_git=true by default in our
// mock, flip it to false and verify the config step still passes.
const { runner } = createRunner({
initNewProject: vi.fn().mockResolvedValue(makeProjectInfo({ has_git: false })),
});
// This will attempt to run `git init` which may or may not exist in test env.
// Since we're in a tmpDir, git init is safe. The test verifies the workflow proceeds.
const result = await runner.run('build a todo app');
// The config step should succeed (git init in tmpDir should work)
const configStep = result.steps.find(s => s.step === 'config');
expect(configStep).toBeDefined();
// Note: if git is not available in CI, this may fail — that's expected
});
// ─── Config passthrough test ─────────────────────────────────────────────
it('constructor accepts config overrides', async () => {
// Set projectInfo model fields to undefined so orchestratorModel is used as fallback
const { runner } = createRunner({
initNewProject: vi.fn().mockResolvedValue(makeProjectInfo({
researcher_model: undefined as any,
synthesizer_model: undefined as any,
roadmapper_model: undefined as any,
})),
}, {
maxBudgetPerSession: 10.0,
maxTurnsPerSession: 50,
orchestratorModel: 'claude-opus-4-6',
});
await runner.run('build a todo app');
// Verify the session runner was called with overridden model
const calls = mockRunSession.mock.calls;
expect(calls.length).toBeGreaterThan(0);
// Check model in options (4th argument, index 3)
const modelsUsed = calls.map(c => {
const options = c[3] as any;
return options?.model;
});
// When projectInfo model is undefined, ?? falls through to orchestratorModel
expect(modelsUsed.some(m => m === 'claude-opus-4-6')).toBe(true);
});
// ─── Session count validation ────────────────────────────────────────────
it('run() calls runPhaseStepSession exactly 8 times on full success', async () => {
const { runner } = createRunner();
await runner.run('build a todo app');
// 1 PROJECT.md + 4 research + 1 synthesis + 1 requirements + 1 roadmap = 8
expect(mockRunSession).toHaveBeenCalledTimes(8);
});
// ─── Headless prompt loading (sdkPromptsDir preference) ──────────────────
describe('sdkPromptsDir preference and sanitizer integration', () => {
let sdkPromptsDir: string;
beforeEach(async () => {
// Create a temp SDK prompts directory with test fixtures
sdkPromptsDir = join(tmpDir, 'sdk-prompts');
await mkdir(join(sdkPromptsDir, 'templates', 'research-project'), { recursive: true });
await mkdir(join(sdkPromptsDir, 'agents'), { recursive: true });
// Write headless templates (with known marker text for assertion)
await writeFile(
join(sdkPromptsDir, 'templates', 'project.md'),
'# PROJECT Template\nSDK_HEADLESS_MARKER_PROJECT\n',
);
await writeFile(
join(sdkPromptsDir, 'templates', 'requirements.md'),
'# REQUIREMENTS Template\nSDK_HEADLESS_MARKER_REQUIREMENTS\n',
);
await writeFile(
join(sdkPromptsDir, 'templates', 'roadmap.md'),
'# ROADMAP Template\nSDK_HEADLESS_MARKER_ROADMAP\n',
);
await writeFile(
join(sdkPromptsDir, 'templates', 'state.md'),
'# STATE Template\nSDK_HEADLESS_MARKER_STATE\n',
);
await writeFile(
join(sdkPromptsDir, 'templates', 'research-project', 'STACK.md'),
'# STACK Template\nSDK_HEADLESS_MARKER_STACK\n',
);
// Write headless agents (with known marker text)
await writeFile(
join(sdkPromptsDir, 'agents', 'gsd-project-researcher.md'),
'# Project Researcher Agent\nSDK_HEADLESS_MARKER_RESEARCHER\n',
);
await writeFile(
join(sdkPromptsDir, 'agents', 'gsd-research-synthesizer.md'),
'# Research Synthesizer Agent\nSDK_HEADLESS_MARKER_SYNTHESIZER\n',
);
await writeFile(
join(sdkPromptsDir, 'agents', 'gsd-roadmapper.md'),
'# Roadmapper Agent\nSDK_HEADLESS_MARKER_ROADMAPPER\n',
);
});
function createRunnerWithSdkPrompts(
toolsOverrides: Record<string, unknown> = {},
configOverrides?: Partial<InitRunnerDeps['config']>,
) {
const tools = makeTools(toolsOverrides);
const eventStream = makeEventStream();
const runner = new InitRunner({
projectDir: tmpDir,
tools,
eventStream,
config: configOverrides as any,
sdkPromptsDir,
});
return { runner, tools, eventStream, events: eventStream.events as GSDEvent[] };
}
it('readGSDFile prefers installed GSD over sdk/prompts/ template', async () => {
const { runner } = createRunnerWithSdkPrompts();
await runner.run('build a todo app');
// The first session call is buildProjectPrompt → reads templates/project.md
// Installed GSD templates (if present) are preferred over SDK bundled copies
const projectPrompt = mockRunSession.mock.calls[0]![0] as string;
// Should contain PROJECT.md creation instruction regardless of source
expect(projectPrompt).toContain('PROJECT.md');
});
it('readAgentFile prefers installed agents over sdk/prompts/agents/', async () => {
const { runner } = createRunnerWithSdkPrompts();
await runner.run('build a todo app');
// Research calls (indices 1-4) use gsd-project-researcher.md agent def
const researchPrompt = mockRunSession.mock.calls[1]![0] as string;
// Should contain research instruction regardless of source
expect(researchPrompt).toContain('You are researching the');
});
it('readGSDFile falls back to GSD-1 when sdk/prompts/ file does not exist', async () => {
// Create an empty sdkPromptsDir — no templates at all
const emptySdkDir = join(tmpDir, 'empty-sdk-prompts');
await mkdir(join(emptySdkDir, 'templates'), { recursive: true });
await mkdir(join(emptySdkDir, 'agents'), { recursive: true });
const tools = makeTools();
const eventStream = makeEventStream();
const runner = new InitRunner({
projectDir: tmpDir,
tools,
eventStream,
sdkPromptsDir: emptySdkDir,
});
await runner.run('build a todo app');
// buildProjectPrompt reads templates/project.md — not found in empty dir,
// falls through to GSD-1 path. If GSD-1 also missing, gets placeholder.
const projectPrompt = mockRunSession.mock.calls[0]![0] as string;
// Should NOT contain our marker (since empty dir was used)
expect(projectPrompt).not.toContain('SDK_HEADLESS_MARKER_PROJECT');
// Should still contain the PROJECT.md synthesis instruction (from the prompt builder)
expect(projectPrompt).toContain('PROJECT.md');
});
it('readAgentFile falls back to GSD-1 when sdk/prompts/agents/ file does not exist', async () => {
// Empty sdkPromptsDir — no agent files
const emptySdkDir = join(tmpDir, 'empty-sdk-agents');
await mkdir(join(emptySdkDir, 'templates', 'research-project'), { recursive: true });
await mkdir(join(emptySdkDir, 'agents'), { recursive: true });
// Write templates so we get past buildProjectPrompt
await writeFile(join(emptySdkDir, 'templates', 'project.md'), '# project\n');
await writeFile(join(emptySdkDir, 'templates', 'research-project', 'STACK.md'), '# stack\n');
await writeFile(join(emptySdkDir, 'templates', 'research-project', 'FEATURES.md'), '# features\n');
await writeFile(join(emptySdkDir, 'templates', 'research-project', 'ARCHITECTURE.md'), '# arch\n');
await writeFile(join(emptySdkDir, 'templates', 'research-project', 'PITFALLS.md'), '# pitfalls\n');
const tools = makeTools();
const eventStream = makeEventStream();
const runner = new InitRunner({
projectDir: tmpDir,
tools,
eventStream,
sdkPromptsDir: emptySdkDir,
});
await runner.run('build a todo app');
// Research prompt uses agent def — not in empty agents dir, falls to GSD-1
const researchPrompt = mockRunSession.mock.calls[1]![0] as string;
// Should NOT contain our marker
expect(researchPrompt).not.toContain('SDK_HEADLESS_MARKER_RESEARCHER');
// Should still have the "researching the" instruction
expect(researchPrompt).toContain('You are researching the');
});
it('buildProjectPrompt output passes through sanitizePrompt (no /gsd: patterns)', async () => {
const { runner } = createRunnerWithSdkPrompts();
await runner.run('build a todo app');
const projectPrompt = mockRunSession.mock.calls[0]![0] as string;
// sanitizePrompt should strip any /gsd: patterns from the assembled prompt
expect(projectPrompt).not.toMatch(/\/gsd:\S+/);
expect(projectPrompt).toContain('PROJECT.md');
});
it('buildResearchPrompt output passes through sanitizePrompt (no /gsd: patterns)', async () => {
const { runner } = createRunnerWithSdkPrompts();
await runner.run('build a todo app');
const researchPrompt = mockRunSession.mock.calls[1]![0] as string;
// sanitizePrompt should strip any /gsd: patterns from the assembled prompt
expect(researchPrompt).not.toMatch(/\/gsd:\S+/);
expect(researchPrompt).toContain('You are researching the');
});
it('buildRoadmapPrompt output passes through sanitizePrompt (no /gsd: patterns)', async () => {
const { runner } = createRunnerWithSdkPrompts();
await runner.run('build a todo app');
// Roadmap prompt is the last session call (index 7)
const roadmapPrompt = mockRunSession.mock.calls[7]![0] as string;
// sanitizePrompt should strip any /gsd: patterns from the assembled prompt
expect(roadmapPrompt).not.toMatch(/\/gsd:\S+/);
});
});
});

View File

@@ -1,734 +0,0 @@
/**
* InitRunner — orchestrates the GSD new-project init workflow.
*
* Workflow: setup → config → PROJECT.md → parallel research (4 sessions)
* → synthesis → requirements → roadmap
*
* Each step calls Agent SDK `query()` via `runPhaseStepSession()` with
* prompts derived from GSD-1 workflow/agent/template files on disk.
*/
import { readFile, writeFile, mkdir } from 'node:fs/promises';
import { join } from 'node:path';
import { fileURLToPath } from 'node:url';
import { execFile } from 'node:child_process';
import type {
InitConfig,
InitResult,
InitStepResult,
InitStepName,
InitNewProjectInfo,
GSDInitStartEvent,
GSDInitStepStartEvent,
GSDInitStepCompleteEvent,
GSDInitCompleteEvent,
GSDInitResearchSpawnEvent,
PlanResult,
} from './types.js';
import { GSDEventType, PhaseStepType } from './types.js';
import type { GSDTools } from './gsd-tools.js';
import type { GSDEventStream } from './event-stream.js';
import { loadConfig } from './config.js';
import { runPhaseStepSession } from './session-runner.js';
import { sanitizePrompt } from './prompt-sanitizer.js';
import { resolveAgentsDir } from './query/helpers.js';
import { resolveLegacyTemplatesDir } from './sdk-package-compatibility.js';
// ─── Constants ───────────────────────────────────────────────────────────────
const GSD_TEMPLATES_DIR = resolveLegacyTemplatesDir();
const GSD_AGENTS_DIR = resolveAgentsDir();
const RESEARCH_TYPES = ['STACK', 'FEATURES', 'ARCHITECTURE', 'PITFALLS'] as const;
type ResearchType = (typeof RESEARCH_TYPES)[number];
const RESEARCH_STEP_MAP: Record<ResearchType, InitStepName> = {
STACK: 'research-stack',
FEATURES: 'research-features',
ARCHITECTURE: 'research-architecture',
PITFALLS: 'research-pitfalls',
};
/** Default config.json written during init for auto-mode projects. */
const AUTO_MODE_CONFIG = {
mode: 'yolo',
parallelization: true,
depth: 'quick',
workflow: {
research: true,
plan_checker: true,
verifier: true,
auto_advance: true,
skip_discuss: false,
},
};
// ─── InitRunner ──────────────────────────────────────────────────────────────
export interface InitRunnerDeps {
projectDir: string;
tools: GSDTools;
eventStream: GSDEventStream;
config?: Partial<InitConfig>;
/** Override for SDK prompts directory. Defaults to package-relative sdk/prompts/. */
sdkPromptsDir?: string;
}
export class InitRunner {
private readonly projectDir: string;
private readonly tools: GSDTools;
private readonly eventStream: GSDEventStream;
private readonly config: InitConfig;
private readonly sessionId: string;
private readonly sdkPromptsDir: string;
constructor(deps: InitRunnerDeps) {
this.projectDir = deps.projectDir;
this.tools = deps.tools;
this.eventStream = deps.eventStream;
this.config = {
maxBudgetPerSession: deps.config?.maxBudgetPerSession ?? 3.0,
maxTurnsPerSession: deps.config?.maxTurnsPerSession ?? 30,
researchModel: deps.config?.researchModel,
orchestratorModel: deps.config?.orchestratorModel,
};
this.sessionId = `init-${Date.now()}`;
// SDK prompts dir: explicit override → package-relative default via import.meta.url
this.sdkPromptsDir =
deps.sdkPromptsDir ??
join(fileURLToPath(new URL('.', import.meta.url)), '..', 'prompts');
}
/**
* Run the full init workflow.
*
* @param input - User input: PRD content, project description, etc.
* @returns InitResult with per-step results, artifacts, and totals.
*/
async run(input: string): Promise<InitResult> {
const startTime = Date.now();
const steps: InitStepResult[] = [];
const artifacts: string[] = [];
this.emitEvent<GSDInitStartEvent>({
type: GSDEventType.InitStart,
input: input.slice(0, 200),
projectDir: this.projectDir,
});
try {
// ── Step 1: Setup — get project metadata ──────────────────────────
const setupResult = await this.runStep('setup', async () => {
const info = await this.tools.initNewProject();
if (info.project_exists) {
throw new Error('Project already exists (.planning/PROJECT.md found). Use a fresh directory or delete .planning/ first.');
}
return info;
});
steps.push(setupResult.stepResult);
if (!setupResult.stepResult.success) {
return this.buildResult(false, steps, artifacts, startTime);
}
const projectInfo = setupResult.value as InitNewProjectInfo;
// ── Step 2: Config — write config.json and init git ───────────────
const configResult = await this.runStep('config', async () => {
// Ensure git is initialized
if (!projectInfo.has_git) {
await this.execGit(['init']);
}
// Ensure .planning/ directory exists
const planningDir = join(this.projectDir, '.planning');
await mkdir(planningDir, { recursive: true });
// Write config.json
const configPath = join(planningDir, 'config.json');
await writeFile(configPath, JSON.stringify(AUTO_MODE_CONFIG, null, 2) + '\n', 'utf-8');
artifacts.push('.planning/config.json');
// Persist auto_advance via gsd-tools (validates & updates state)
await this.tools.configSet('workflow.auto_advance', 'true');
// Commit config
if (projectInfo.commit_docs) {
await this.tools.commit('chore: add project config', ['.planning/config.json']);
}
});
steps.push(configResult.stepResult);
if (!configResult.stepResult.success) {
return this.buildResult(false, steps, artifacts, startTime);
}
// ── Step 3: PROJECT.md — synthesize from input ────────────────────
const projectResult = await this.runStep('project', async () => {
const prompt = await this.buildProjectPrompt(input);
const result = await this.runSession(prompt, projectInfo.researcher_model);
if (!result.success) {
throw new Error(`PROJECT.md synthesis failed: ${result.error?.messages.join(', ') ?? 'unknown error'}`);
}
artifacts.push('.planning/PROJECT.md');
if (projectInfo.commit_docs) {
await this.tools.commit('docs: add PROJECT.md', ['.planning/PROJECT.md']);
}
return result;
});
steps.push(projectResult.stepResult);
if (!projectResult.stepResult.success) {
return this.buildResult(false, steps, artifacts, startTime);
}
// ── Step 4: Parallel research (4 sessions) ───────────────────────
const researchSteps = await this.runParallelResearch(input, projectInfo);
steps.push(...researchSteps);
const researchFailed = researchSteps.some(s => !s.success);
// Add artifacts for successful research files
for (const rs of researchSteps) {
if (rs.success && rs.artifacts) {
artifacts.push(...rs.artifacts);
}
}
if (researchFailed) {
// Continue with partial results — synthesis will work with what's available
// but flag the overall result as partial
}
// ── Step 5: Synthesis — combine research into SUMMARY.md ──────────
const synthResult = await this.runStep('synthesis', async () => {
const prompt = await this.buildSynthesisPrompt();
const result = await this.runSession(prompt, projectInfo.synthesizer_model);
if (!result.success) {
throw new Error(`Research synthesis failed: ${result.error?.messages.join(', ') ?? 'unknown error'}`);
}
artifacts.push('.planning/research/SUMMARY.md');
if (projectInfo.commit_docs) {
await this.tools.commit('docs: add research files', ['.planning/research/']);
}
return result;
});
steps.push(synthResult.stepResult);
if (!synthResult.stepResult.success) {
return this.buildResult(false, steps, artifacts, startTime);
}
// ── Step 6: Requirements — derive from PROJECT + research ─────────
const reqResult = await this.runStep('requirements', async () => {
const prompt = await this.buildRequirementsPrompt();
const result = await this.runSession(prompt, projectInfo.synthesizer_model);
if (!result.success) {
throw new Error(`Requirements generation failed: ${result.error?.messages.join(', ') ?? 'unknown error'}`);
}
artifacts.push('.planning/REQUIREMENTS.md');
if (projectInfo.commit_docs) {
await this.tools.commit('docs: add REQUIREMENTS.md', ['.planning/REQUIREMENTS.md']);
}
return result;
});
steps.push(reqResult.stepResult);
if (!reqResult.stepResult.success) {
return this.buildResult(false, steps, artifacts, startTime);
}
// ── Step 7: Roadmap — create phases + STATE.md ────────────────────
const roadmapResult = await this.runStep('roadmap', async () => {
const prompt = await this.buildRoadmapPrompt();
const result = await this.runSession(prompt, projectInfo.roadmapper_model);
if (!result.success) {
throw new Error(`Roadmap generation failed: ${result.error?.messages.join(', ') ?? 'unknown error'}`);
}
artifacts.push('.planning/ROADMAP.md', '.planning/STATE.md');
if (projectInfo.commit_docs) {
await this.tools.commit('docs: add ROADMAP.md and STATE.md', [
'.planning/ROADMAP.md',
'.planning/STATE.md',
]);
}
return result;
});
steps.push(roadmapResult.stepResult);
if (!roadmapResult.stepResult.success) {
return this.buildResult(false, steps, artifacts, startTime);
}
const success = !researchFailed;
return this.buildResult(success, steps, artifacts, startTime);
} catch (err) {
// Unexpected top-level error
steps.push({
step: 'setup',
success: false,
durationMs: 0,
costUsd: 0,
error: err instanceof Error ? err.message : String(err),
});
return this.buildResult(false, steps, artifacts, startTime);
}
}
// ─── Step execution wrapper ────────────────────────────────────────────────
private async runStep<T>(
step: InitStepName,
fn: () => Promise<T>,
): Promise<{ stepResult: InitStepResult; value?: T }> {
const stepStart = Date.now();
this.emitEvent<GSDInitStepStartEvent>({
type: GSDEventType.InitStepStart,
step,
});
try {
const value = await fn();
const durationMs = Date.now() - stepStart;
const costUsd = this.extractCost(value);
const stepResult: InitStepResult = {
step,
success: true,
durationMs,
costUsd,
};
this.emitEvent<GSDInitStepCompleteEvent>({
type: GSDEventType.InitStepComplete,
step,
success: true,
durationMs,
costUsd,
});
return { stepResult, value };
} catch (err) {
const durationMs = Date.now() - stepStart;
const errorMsg = err instanceof Error ? err.message : String(err);
const stepResult: InitStepResult = {
step,
success: false,
durationMs,
costUsd: 0,
error: errorMsg,
};
this.emitEvent<GSDInitStepCompleteEvent>({
type: GSDEventType.InitStepComplete,
step,
success: false,
durationMs,
costUsd: 0,
error: errorMsg,
});
return { stepResult };
}
}
// ─── Parallel research ─────────────────────────────────────────────────────
private async runParallelResearch(
input: string,
projectInfo: InitNewProjectInfo,
): Promise<InitStepResult[]> {
this.emitEvent<GSDInitResearchSpawnEvent>({
type: GSDEventType.InitResearchSpawn,
sessionCount: RESEARCH_TYPES.length,
researchTypes: [...RESEARCH_TYPES],
});
const promises = RESEARCH_TYPES.map(async (researchType) => {
const step = RESEARCH_STEP_MAP[researchType];
const result = await this.runStep(step, async () => {
const prompt = await this.buildResearchPrompt(researchType, input);
const sessionResult = await this.runSession(prompt, projectInfo.researcher_model);
if (!sessionResult.success) {
throw new Error(
`Research (${researchType}) failed: ${sessionResult.error?.messages.join(', ') ?? 'unknown error'}`,
);
}
return sessionResult;
});
// Attach artifact path on success
if (result.stepResult.success) {
result.stepResult.artifacts = [`.planning/research/${researchType}.md`];
}
return result.stepResult;
});
const results = await Promise.allSettled(promises);
return results.map((r, i) => {
if (r.status === 'fulfilled') {
return r.value;
}
// Promise.allSettled rejection — should not happen since runStep catches,
// but handle defensively
return {
step: RESEARCH_STEP_MAP[RESEARCH_TYPES[i]!]!,
success: false,
durationMs: 0,
costUsd: 0,
error: r.reason instanceof Error ? r.reason.message : String(r.reason),
} satisfies InitStepResult;
});
}
// ─── Prompt builders ───────────────────────────────────────────────────────
/**
* Build the PROJECT.md synthesis prompt.
* Reads the project template and combines with user input.
*/
private async buildProjectPrompt(input: string): Promise<string> {
const template = await this.readGSDFile('templates/project.md');
return sanitizePrompt([
'You are creating the PROJECT.md for a new software project.',
'Write .planning/PROJECT.md based on the template structure below and the user\'s project description.',
'',
'<project_template>',
template,
'</project_template>',
'',
'<user_input>',
input,
'</user_input>',
'',
'Write the file to .planning/PROJECT.md. Follow the template structure but fill in with real content derived from the user input.',
'Be specific and opinionated — make decisions, don\'t list options.',
].join('\n'), this.projectDir);
}
/**
* Build a research prompt for a specific research type.
* Reads the agent definition and research template.
*/
private async buildResearchPrompt(
researchType: ResearchType,
input: string,
): Promise<string> {
const agentDef = await this.readAgentFile('gsd-project-researcher.md');
const template = await this.readGSDFile(`templates/research-project/${researchType}.md`);
// Read PROJECT.md if it exists (it should by now)
let projectContent = '';
try {
projectContent = await readFile(
join(this.projectDir, '.planning', 'PROJECT.md'),
'utf-8',
);
} catch {
// Fall back to raw input if PROJECT.md not yet written
projectContent = input;
}
return sanitizePrompt([
'<agent_definition>',
agentDef,
'</agent_definition>',
'',
`You are researching the ${researchType} aspect of this project.`,
`Write your findings to .planning/research/${researchType}.md`,
'',
'<files_to_read>',
'.planning/PROJECT.md',
'</files_to_read>',
'',
'<project_context>',
projectContent,
'</project_context>',
'',
'<research_template>',
template,
'</research_template>',
'',
`Write .planning/research/${researchType}.md following the template structure.`,
'Be comprehensive but opinionated. "Use X because Y" not "Options are X, Y, Z."',
].join('\n'), this.projectDir);
}
/**
* Build the synthesis prompt.
* Reads synthesizer agent def and all 4 research outputs.
*/
private async buildSynthesisPrompt(): Promise<string> {
const agentDef = await this.readAgentFile('gsd-research-synthesizer.md');
const summaryTemplate = await this.readGSDFile('templates/research-project/SUMMARY.md');
const researchDir = join(this.projectDir, '.planning', 'research');
// Read whatever research files exist
const researchContent: string[] = [];
for (const rt of RESEARCH_TYPES) {
try {
const content = await readFile(join(researchDir, `${rt}.md`), 'utf-8');
researchContent.push(`<research_${rt.toLowerCase()}>\n${content}\n</research_${rt.toLowerCase()}>`);
} catch {
researchContent.push(`<research_${rt.toLowerCase()}>\n(Not available)\n</research_${rt.toLowerCase()}>`);
}
}
return sanitizePrompt([
'<agent_definition>',
agentDef,
'</agent_definition>',
'',
'<files_to_read>',
'.planning/research/STACK.md',
'.planning/research/FEATURES.md',
'.planning/research/ARCHITECTURE.md',
'.planning/research/PITFALLS.md',
'</files_to_read>',
'',
'Synthesize the research files below into .planning/research/SUMMARY.md',
'',
...researchContent,
'',
'<summary_template>',
summaryTemplate,
'</summary_template>',
'',
'Write .planning/research/SUMMARY.md synthesizing all research findings.',
'Also commit all research files: git add .planning/research/ && git commit.',
].join('\n'), this.projectDir);
}
/**
* Build the requirements prompt.
* Reads PROJECT.md + FEATURES.md for requirement derivation.
*/
private async buildRequirementsPrompt(): Promise<string> {
const reqTemplate = await this.readGSDFile('templates/requirements.md');
let projectContent = '';
let featuresContent = '';
try {
projectContent = await readFile(
join(this.projectDir, '.planning', 'PROJECT.md'),
'utf-8',
);
} catch {
// Should not happen at this point
}
try {
featuresContent = await readFile(
join(this.projectDir, '.planning', 'research', 'FEATURES.md'),
'utf-8',
);
} catch {
// Research may have partially failed
}
return sanitizePrompt([
'You are generating REQUIREMENTS.md for this project.',
'Derive requirements from the PROJECT.md and research outputs.',
'Auto-include all table-stakes requirements (auth, error handling, logging, etc.).',
'',
'<project_context>',
projectContent,
'</project_context>',
'',
'<features_research>',
featuresContent || '(Not available)',
'</features_research>',
'',
'<requirements_template>',
reqTemplate,
'</requirements_template>',
'',
'Write .planning/REQUIREMENTS.md following the template structure.',
'Every requirement must be testable and specific. No vague aspirations.',
].join('\n'), this.projectDir);
}
/**
* Build the roadmap prompt.
* Reads PROJECT.md + REQUIREMENTS.md + research/SUMMARY.md + config.json.
*/
private async buildRoadmapPrompt(): Promise<string> {
const agentDef = await this.readAgentFile('gsd-roadmapper.md');
const roadmapTemplate = await this.readGSDFile('templates/roadmap.md');
const stateTemplate = await this.readGSDFile('templates/state.md');
const filesToRead = [
'.planning/PROJECT.md',
'.planning/REQUIREMENTS.md',
'.planning/research/SUMMARY.md',
'.planning/config.json',
];
const fileContents: string[] = [];
for (const fp of filesToRead) {
try {
const content = await readFile(join(this.projectDir, fp), 'utf-8');
fileContents.push(`<file path="${fp}">\n${content}\n</file>`);
} catch {
fileContents.push(`<file path="${fp}">\n(Not available)\n</file>`);
}
}
return sanitizePrompt([
'<agent_definition>',
agentDef,
'</agent_definition>',
'',
'<files_to_read>',
...filesToRead,
'</files_to_read>',
'',
...fileContents,
'',
'<roadmap_template>',
roadmapTemplate,
'</roadmap_template>',
'',
'<state_template>',
stateTemplate,
'</state_template>',
'',
'Create .planning/ROADMAP.md and .planning/STATE.md.',
'ROADMAP.md: Transform requirements into phases. Every v1 requirement maps to exactly one phase.',
'STATE.md: Initialize project state tracking.',
].join('\n'), this.projectDir);
}
// ─── Session execution ─────────────────────────────────────────────────────
/**
* Run a single Agent SDK session via runPhaseStepSession.
*/
private async runSession(prompt: string, modelOverride?: string): Promise<PlanResult> {
const config = await loadConfig(this.projectDir);
return runPhaseStepSession(
prompt,
PhaseStepType.Research, // Research phase gives broadest tool access
config,
{
maxTurns: this.config.maxTurnsPerSession,
maxBudgetUsd: this.config.maxBudgetPerSession,
model: modelOverride ?? this.config.orchestratorModel,
cwd: this.projectDir,
},
this.eventStream,
{ phase: undefined, planName: undefined },
);
}
// ─── File reading helpers ──────────────────────────────────────────────────
/**
* Read a file from the GSD templates directory.
* Tries sdk/prompts/{relativePath} first (headless versions), then
* falls back to GSD-1 originals (~/.claude/get-shit-done/).
*/
private async readGSDFile(relativePath: string): Promise<string> {
// Try installed GSD first (complete, up-to-date versions)
const fullPath = join(GSD_TEMPLATES_DIR, '..', relativePath);
try {
return await readFile(fullPath, 'utf-8');
} catch {
// Not installed, fall through to SDK bundled copies
}
// Fall back to SDK bundled copies
const sdkPath = join(this.sdkPromptsDir, relativePath);
try {
return await readFile(sdkPath, 'utf-8');
} catch {
return `(Template not found: ${relativePath})`;
}
}
/**
* Read an agent definition.
* Tries installed agents first (complete, up-to-date versions), then
* falls back to SDK bundled copies.
*/
private async readAgentFile(filename: string): Promise<string> {
// Try installed agents first (complete, up-to-date versions)
const fullPath = join(GSD_AGENTS_DIR, filename);
try {
return await readFile(fullPath, 'utf-8');
} catch {
// Not installed, fall through to SDK bundled copies
}
// Fall back to SDK bundled copies
const sdkPath = join(this.sdkPromptsDir, 'agents', filename);
try {
return await readFile(sdkPath, 'utf-8');
} catch {
return `(Agent definition not found: ${filename})`;
}
}
// ─── Git helper ────────────────────────────────────────────────────────────
/**
* Execute a git command in the project directory.
*/
private execGit(args: string[]): Promise<string> {
return new Promise((resolve, reject) => {
execFile('git', args, { cwd: this.projectDir }, (error, stdout, stderr) => {
if (error) {
reject(new Error(`git ${args.join(' ')} failed: ${stderr || error.message}`));
return;
}
resolve(stdout.toString());
});
});
}
// ─── Event helpers ─────────────────────────────────────────────────────────
private emitEvent<T extends { type: GSDEventType }>(
partial: Omit<T, 'timestamp' | 'sessionId'> & { type: GSDEventType },
): void {
this.eventStream.emitEvent({
timestamp: new Date().toISOString(),
sessionId: this.sessionId,
...partial,
} as unknown as import('./types.js').GSDEvent);
}
// ─── Result helpers ────────────────────────────────────────────────────────
private buildResult(
success: boolean,
steps: InitStepResult[],
artifacts: string[],
startTime: number,
): InitResult {
const totalCostUsd = steps.reduce((sum, s) => sum + s.costUsd, 0);
const totalDurationMs = Date.now() - startTime;
this.emitEvent<GSDInitCompleteEvent>({
type: GSDEventType.InitComplete,
success,
totalCostUsd,
totalDurationMs,
artifactCount: artifacts.length,
});
return {
success,
steps,
totalCostUsd,
totalDurationMs,
artifacts,
};
}
/**
* Extract cost from a step return value if it's a PlanResult.
*/
private extractCost(value: unknown): number {
if (value && typeof value === 'object' && 'totalCostUsd' in value) {
return (value as PlanResult).totalCostUsd;
}
return 0;
}
}

View File

@@ -1,258 +0,0 @@
/**
* E2E lifecycle integration test — proves GSD.runPhase() drives
* the full phase lifecycle: discuss → research → plan → execute → verify → advance
* after bootstrapping a real project via InitRunner.
*
* This is the capstone proof that `gsd-sdk auto` works end-to-end
* without human intervention. InitRunner bootstraps the project,
* then GSD.runPhase() drives Phase 1 through the complete lifecycle.
*
* Requires Claude Code CLI (`claude`) installed and authenticated.
* Skips gracefully if CLI is unavailable.
*/
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
import { execSync } from 'node:child_process';
import { mkdtemp, rm, readFile, stat, readdir } from 'node:fs/promises';
import { existsSync } from 'node:fs';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';
import { GSD } from './index.js';
import { InitRunner } from './init-runner.js';
import { GSDTools, resolveGsdToolsPath } from './gsd-tools.js';
import { GSDEventStream } from './event-stream.js';
import { GSDEventType, PhaseStepType } from './types.js';
import type { GSDEvent, PhaseRunnerResult, RoadmapAnalysis } from './types.js';
// ─── CLI availability check ─────────────────────────────────────────────────
let cliAvailable = false;
try {
execSync('which claude', { stdio: 'ignore' });
cliAvailable = true;
} catch {
cliAvailable = false;
}
const __dirname = fileURLToPath(new URL('.', import.meta.url));
const sdkPromptsDir = join(__dirname, '..', 'prompts');
const GSD_TOOLS_PATH = resolveGsdToolsPath(process.cwd());
const gsdToolsAvailable = existsSync(GSD_TOOLS_PATH);
// ─── Lifecycle step ordering for monotonicity check ──────────────────────────
const STEP_ORDER: Record<string, number> = {
[PhaseStepType.Discuss]: 0,
[PhaseStepType.Research]: 1,
[PhaseStepType.Plan]: 2,
[PhaseStepType.PlanCheck]: 3,
[PhaseStepType.Execute]: 4,
[PhaseStepType.Verify]: 5,
[PhaseStepType.Advance]: 6,
};
// ─── Test suite ──────────────────────────────────────────────────────────────
describe.skipIf(!cliAvailable || !gsdToolsAvailable)('E2E Lifecycle: InitRunner → GSD.runPhase() full lifecycle', () => {
let tmpDir: string;
let initSuccess: boolean = false;
let phase1Number: string | null = null;
let tools: GSDTools;
// ── Bootstrap: create temp dir, git init, run InitRunner ──────────────
beforeAll(async () => {
tmpDir = await mkdtemp(join(tmpdir(), 'gsd-sdk-lifecycle-e2e-'));
// Git init (required by InitRunner and phase lifecycle)
execSync('git init', { cwd: tmpDir, stdio: 'ignore' });
execSync('git config user.email "test@test.com"', { cwd: tmpDir, stdio: 'ignore' });
execSync('git config user.name "Test"', { cwd: tmpDir, stdio: 'ignore' });
tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: GSD_TOOLS_PATH,
timeoutMs: 30_000,
});
// Run InitRunner to bootstrap the project
const initEventStream = new GSDEventStream();
const initRunner = new InitRunner({
projectDir: tmpDir,
tools,
eventStream: initEventStream,
config: {
maxBudgetPerSession: 1.0,
maxTurnsPerSession: 15,
},
sdkPromptsDir,
});
const initResult = await initRunner.run('Build a CLI tool that converts Celsius to Fahrenheit');
// Mark init as successful if the pipeline progressed enough
const completedSteps = initResult.steps.filter(s => s.success);
initSuccess = initResult.success || completedSteps.length >= 3;
// Discover the first phase number via roadmapAnalyze
if (initSuccess) {
try {
const analysis: RoadmapAnalysis = await tools.roadmapAnalyze();
if (analysis.phases && analysis.phases.length > 0) {
// Sort by phase number and take the first
const sorted = [...analysis.phases].sort(
(a, b) => parseFloat(a.number) - parseFloat(b.number),
);
phase1Number = sorted[0]!.number;
}
} catch {
// If roadmap analyze fails, try scanning the phases dir directly
try {
const phasesDir = join(tmpDir, '.planning', 'phases');
const entries = await readdir(phasesDir);
const phaseEntries = entries
.filter(e => /^\d+/.test(e))
.sort();
if (phaseEntries.length > 0) {
// Extract the phase number (everything before the first dash)
const match = phaseEntries[0]!.match(/^(\d+)/);
if (match) {
phase1Number = match[1]!;
}
}
} catch {
// No phases dir — init didn't create one
}
}
}
}, 600_000); // 10 min for init
afterAll(async () => {
if (tmpDir) {
await rm(tmpDir, { recursive: true, force: true });
}
});
// ── Main lifecycle test ───────────────────────────────────────────────
it('GSD.runPhase() drives Phase 1 through the full lifecycle without human intervention', async () => {
// If init failed, skip — can't test lifecycle without a bootstrapped project
if (!initSuccess) {
console.warn('Skipping lifecycle test: InitRunner did not bootstrap successfully');
return;
}
// Verify ROADMAP.md exists and contains at least one phase
const roadmapPath = join(tmpDir, '.planning', 'ROADMAP.md');
const roadmapStat = await stat(roadmapPath).catch(() => null);
expect(roadmapStat).not.toBeNull();
const roadmapContent = await readFile(roadmapPath, 'utf-8');
expect(roadmapContent.length).toBeGreaterThan(0);
// Verify we discovered a phase number
expect(phase1Number).not.toBeNull();
// Verify the phase exists via initPhaseOp
const phaseOp = await tools.initPhaseOp(phase1Number!);
expect(phaseOp.phase_found).toBe(true);
// Collect all events during the phase lifecycle
const events: GSDEvent[] = [];
// Construct GSD with autoMode: true
const gsd = new GSD({
projectDir: tmpDir,
autoMode: true,
});
gsd.onEvent((e: GSDEvent) => events.push(e));
// Run the discovered first phase with tight budget to minimize cost
const result: PhaseRunnerResult = await gsd.runPhase(phase1Number!, {
maxTurnsPerStep: 10,
maxBudgetPerStep: 0.50,
});
// ── Assert: result.phaseNumber matches the discovered phase ──
expect(result.phaseNumber).toBe(phase1Number);
// ── Assert: result.phaseName is non-empty ──
expect(result.phaseName).toBeTruthy();
expect(result.phaseName.length).toBeGreaterThan(0);
// ── Assert: at least one lifecycle step was attempted ──
expect(result.steps.length).toBeGreaterThanOrEqual(1);
// ── Assert: events include PhaseStart ──
const phaseStartEvents = events.filter(e => e.type === GSDEventType.PhaseStart);
expect(phaseStartEvents.length).toBe(1);
const phaseStart = phaseStartEvents[0]!;
if (phaseStart.type === GSDEventType.PhaseStart) {
expect(phaseStart.phaseNumber).toBe(phase1Number);
expect(phaseStart.phaseName).toBeTruthy();
}
// ── Assert: events include PhaseComplete ──
const phaseCompleteEvents = events.filter(e => e.type === GSDEventType.PhaseComplete);
expect(phaseCompleteEvents.length).toBe(1);
const phaseComplete = phaseCompleteEvents[0]!;
if (phaseComplete.type === GSDEventType.PhaseComplete) {
expect(phaseComplete.phaseNumber).toBe(phase1Number);
expect(typeof phaseComplete.totalCostUsd).toBe('number');
expect(typeof phaseComplete.totalDurationMs).toBe('number');
}
// ── Assert: PhaseStepStart events show step progression ──
const stepStartEvents = events.filter(
(e): e is Extract<GSDEvent, { type: GSDEventType.PhaseStepStart }> =>
e.type === GSDEventType.PhaseStepStart,
);
expect(stepStartEvents.length).toBeGreaterThanOrEqual(1);
// Extract the step types in order
const stepTypesInOrder = stepStartEvents.map(e => e.step);
// Verify monotonic ordering: each step type should have an index >= previous
// Note: gap-closure can re-run plan+execute after verify, so we allow
// monotonicity to break only when verify triggers gap closure.
// For this tight-budget test, full gap closure is unlikely — check basic ordering.
let lastMaxOrder = -1;
for (const stepType of stepTypesInOrder) {
const order = STEP_ORDER[stepType] ?? -1;
// Track the high-water mark — steps should generally progress forward
if (order >= lastMaxOrder) {
lastMaxOrder = order;
}
}
// At least progressed past discuss (order 0) into real work
expect(lastMaxOrder).toBeGreaterThanOrEqual(1);
// ── Assert: at least one step has planResults with cost > 0 (real Agent SDK work) ──
const stepsWithCost = result.steps.filter(s => {
if (!s.planResults) return false;
return s.planResults.some(pr => pr.totalCostUsd > 0);
});
// At least one step should have incurred real cost (proves Agent SDK was invoked)
expect(stepsWithCost.length).toBeGreaterThanOrEqual(1);
// ── Assert: result cost and duration are tracked ──
expect(typeof result.totalCostUsd).toBe('number');
expect(result.totalDurationMs).toBeGreaterThan(0);
// ── Assert: each step result is properly structured ──
for (const step of result.steps) {
expect(Object.values(PhaseStepType)).toContain(step.step);
expect(typeof step.success).toBe('boolean');
expect(typeof step.durationMs).toBe('number');
}
// ── Assert: PhaseStepComplete events match step results ──
const stepCompleteEvents = events.filter(
(e): e is Extract<GSDEvent, { type: GSDEventType.PhaseStepComplete }> =>
e.type === GSDEventType.PhaseStepComplete,
);
// At least as many complete events as step results
expect(stepCompleteEvents.length).toBeGreaterThanOrEqual(result.steps.length);
}, 900_000); // 15 minute timeout: init (~4 min) + phase lifecycle (~10 min)
});

View File

@@ -1,149 +0,0 @@
import { describe, it, expect, beforeEach } from 'vitest';
import { Writable } from 'node:stream';
import { GSDLogger } from './logger.js';
import type { LogEntry } from './logger.js';
import { PhaseType } from './types.js';
// ─── Test output capture ─────────────────────────────────────────────────────
class BufferStream extends Writable {
lines: string[] = [];
_write(chunk: Buffer, _encoding: string, callback: () => void): void {
const str = chunk.toString();
this.lines.push(...str.split('\n').filter(l => l.length > 0));
callback();
}
}
function parseLogEntry(line: string): LogEntry {
return JSON.parse(line) as LogEntry;
}
// ─── Tests ───────────────────────────────────────────────────────────────────
describe('GSDLogger', () => {
let output: BufferStream;
beforeEach(() => {
output = new BufferStream();
});
it('outputs valid JSON on each log call', () => {
const logger = new GSDLogger({ output, level: 'debug' });
logger.info('test message');
expect(output.lines).toHaveLength(1);
expect(() => JSON.parse(output.lines[0]!)).not.toThrow();
});
it('includes required fields: timestamp, level, message', () => {
const logger = new GSDLogger({ output, level: 'debug' });
logger.info('hello world');
const entry = parseLogEntry(output.lines[0]!);
expect(entry.timestamp).toMatch(/^\d{4}-\d{2}-\d{2}T/);
expect(entry.level).toBe('info');
expect(entry.message).toBe('hello world');
});
it('filters messages below minimum log level', () => {
const logger = new GSDLogger({ output, level: 'warn' });
logger.debug('should be dropped');
logger.info('should be dropped');
logger.warn('should appear');
logger.error('should appear');
expect(output.lines).toHaveLength(2);
expect(parseLogEntry(output.lines[0]!).level).toBe('warn');
expect(parseLogEntry(output.lines[1]!).level).toBe('error');
});
it('defaults to info level filtering', () => {
const logger = new GSDLogger({ output });
logger.debug('dropped');
logger.info('kept');
expect(output.lines).toHaveLength(1);
expect(parseLogEntry(output.lines[0]!).level).toBe('info');
});
it('writes to custom output stream', () => {
const customOutput = new BufferStream();
const logger = new GSDLogger({ output: customOutput, level: 'debug' });
logger.info('custom');
expect(customOutput.lines).toHaveLength(1);
expect(output.lines).toHaveLength(0);
});
it('includes phase, plan, and sessionId context when set', () => {
const logger = new GSDLogger({
output,
level: 'debug',
phase: PhaseType.Execute,
plan: 'test-plan',
sessionId: 'sess-123',
});
logger.info('context test');
const entry = parseLogEntry(output.lines[0]!);
expect(entry.phase).toBe('execute');
expect(entry.plan).toBe('test-plan');
expect(entry.sessionId).toBe('sess-123');
});
it('includes extra data when provided', () => {
const logger = new GSDLogger({ output, level: 'debug' });
logger.info('with data', { count: 42, tool: 'Bash' });
const entry = parseLogEntry(output.lines[0]!);
expect(entry.data).toEqual({ count: 42, tool: 'Bash' });
});
it('omits optional fields when not set', () => {
const logger = new GSDLogger({ output, level: 'debug' });
logger.info('minimal');
const entry = parseLogEntry(output.lines[0]!);
expect(entry.phase).toBeUndefined();
expect(entry.plan).toBeUndefined();
expect(entry.sessionId).toBeUndefined();
expect(entry.data).toBeUndefined();
});
it('supports runtime context updates via setters', () => {
const logger = new GSDLogger({ output, level: 'debug' });
logger.info('before');
logger.setPhase(PhaseType.Research);
logger.setPlan('my-plan');
logger.setSessionId('sess-456');
logger.info('after');
const before = parseLogEntry(output.lines[0]!);
const after = parseLogEntry(output.lines[1]!);
expect(before.phase).toBeUndefined();
expect(after.phase).toBe('research');
expect(after.plan).toBe('my-plan');
expect(after.sessionId).toBe('sess-456');
});
it('emits all four log levels correctly', () => {
const logger = new GSDLogger({ output, level: 'debug' });
logger.debug('d');
logger.info('i');
logger.warn('w');
logger.error('e');
expect(output.lines).toHaveLength(4);
expect(parseLogEntry(output.lines[0]!).level).toBe('debug');
expect(parseLogEntry(output.lines[1]!).level).toBe('info');
expect(parseLogEntry(output.lines[2]!).level).toBe('warn');
expect(parseLogEntry(output.lines[3]!).level).toBe('error');
});
});

View File

@@ -1,113 +0,0 @@
/**
* Structured JSON logger for GSD debugging.
*
* Writes structured log entries to stderr (or configurable writable stream).
* This is a debugging facility (R019), separate from the event stream.
*/
import type { Writable } from 'node:stream';
import type { PhaseType } from './types.js';
// ─── Log levels ──────────────────────────────────────────────────────────────
export type LogLevel = 'debug' | 'info' | 'warn' | 'error';
const LOG_LEVEL_PRIORITY: Record<LogLevel, number> = {
debug: 0,
info: 1,
warn: 2,
error: 3,
};
// ─── Log entry ───────────────────────────────────────────────────────────────
export interface LogEntry {
timestamp: string;
level: LogLevel;
phase?: PhaseType;
plan?: string;
sessionId?: string;
message: string;
data?: Record<string, unknown>;
}
// ─── Logger options ──────────────────────────────────────────────────────────
export interface GSDLoggerOptions {
/** Minimum log level to output. Default: 'info'. */
level?: LogLevel;
/** Output stream. Default: process.stderr. */
output?: Writable;
/** Phase context for all log entries. */
phase?: PhaseType;
/** Plan name context for all log entries. */
plan?: string;
/** Session ID context for all log entries. */
sessionId?: string;
}
// ─── Logger class ────────────────────────────────────────────────────────────
export class GSDLogger {
private readonly minLevel: number;
private readonly output: Writable;
private phase?: PhaseType;
private plan?: string;
private sessionId?: string;
constructor(options: GSDLoggerOptions = {}) {
this.minLevel = LOG_LEVEL_PRIORITY[options.level ?? 'info'];
this.output = options.output ?? process.stderr;
this.phase = options.phase;
this.plan = options.plan;
this.sessionId = options.sessionId;
}
/** Set phase context for subsequent log entries. */
setPhase(phase: PhaseType | undefined): void {
this.phase = phase;
}
/** Set plan context for subsequent log entries. */
setPlan(plan: string | undefined): void {
this.plan = plan;
}
/** Set session ID context for subsequent log entries. */
setSessionId(sessionId: string | undefined): void {
this.sessionId = sessionId;
}
debug(message: string, data?: Record<string, unknown>): void {
this.log('debug', message, data);
}
info(message: string, data?: Record<string, unknown>): void {
this.log('info', message, data);
}
warn(message: string, data?: Record<string, unknown>): void {
this.log('warn', message, data);
}
error(message: string, data?: Record<string, unknown>): void {
this.log('error', message, data);
}
private log(level: LogLevel, message: string, data?: Record<string, unknown>): void {
if (LOG_LEVEL_PRIORITY[level] < this.minLevel) return;
const entry: LogEntry = {
timestamp: new Date().toISOString(),
level,
message,
};
if (this.phase !== undefined) entry.phase = this.phase;
if (this.plan !== undefined) entry.plan = this.plan;
if (this.sessionId !== undefined) entry.sessionId = this.sessionId;
if (data !== undefined) entry.data = data;
this.output.write(JSON.stringify(entry) + '\n');
}
}

View File

@@ -1,17 +0,0 @@
import { STATE_COMMAND_MANIFEST } from '../query/command-manifest.state.js';
import { VERIFY_COMMAND_MANIFEST } from '../query/command-manifest.verify.js';
import { INIT_COMMAND_MANIFEST } from '../query/command-manifest.init.js';
import { PHASE_COMMAND_MANIFEST } from '../query/command-manifest.phase.js';
import { PHASES_COMMAND_MANIFEST } from '../query/command-manifest.phases.js';
import { VALIDATE_COMMAND_MANIFEST } from '../query/command-manifest.validate.js';
import { ROADMAP_COMMAND_MANIFEST } from '../query/command-manifest.roadmap.js';
export const COMMAND_MANIFEST = [
...STATE_COMMAND_MANIFEST,
...VERIFY_COMMAND_MANIFEST,
...INIT_COMMAND_MANIFEST,
...PHASE_COMMAND_MANIFEST,
...PHASES_COMMAND_MANIFEST,
...VALIDATE_COMMAND_MANIFEST,
...ROADMAP_COMMAND_MANIFEST,
] as const;

View File

@@ -1,421 +0,0 @@
import { describe, it, expect, vi, beforeEach } from 'vitest';
import type {
PhaseRunnerResult,
RoadmapPhaseInfo,
RoadmapAnalysis,
GSDEvent,
MilestoneRunnerOptions,
} from './types.js';
import { GSDEventType } from './types.js';
// ─── Mock modules ────────────────────────────────────────────────────────────
// Mock the heavy dependencies that GSD constructor + runPhase pull in
vi.mock('./plan-parser.js', () => ({
parsePlan: vi.fn(),
parsePlanFile: vi.fn(),
}));
vi.mock('./config.js', () => ({
loadConfig: vi.fn().mockResolvedValue({
model_profile: 'test-model',
tools: [],
phases: {},
}),
}));
vi.mock('./session-runner.js', () => ({
runPlanSession: vi.fn(),
runPhaseStepSession: vi.fn(),
}));
vi.mock('./prompt-builder.js', () => ({
buildExecutorPrompt: vi.fn(),
parseAgentTools: vi.fn().mockReturnValue([]),
}));
vi.mock('./event-stream.js', () => {
return {
// Use function (not arrow) so `new GSDEventStream()` works under Vitest 4
GSDEventStream: vi.fn(function GSDEventStreamMock() {
return {
emitEvent: vi.fn(),
on: vi.fn(),
emit: vi.fn(),
addTransport: vi.fn(),
};
}),
};
});
vi.mock('./phase-runner.js', () => ({
PhaseRunner: vi.fn(),
PhaseRunnerError: class extends Error {
name = 'PhaseRunnerError';
},
}));
vi.mock('./context-engine.js', () => ({
ContextEngine: vi.fn(),
PHASE_FILE_MANIFEST: [],
}));
vi.mock('./phase-prompt.js', () => ({
PromptFactory: vi.fn(),
extractBlock: vi.fn(),
extractSteps: vi.fn(),
PHASE_WORKFLOW_MAP: {},
}));
vi.mock('./gsd-tools.js', () => ({
// Constructor mock for `new GSDTools(...)` (Vitest 4)
GSDTools: vi.fn(function GSDToolsMock() {
return {
roadmapAnalyze: vi.fn(),
};
}),
GSDToolsError: class extends Error {
name = 'GSDToolsError';
},
resolveGsdToolsPath: vi.fn().mockReturnValue('/mock/gsd-tools.cjs'),
}));
import { GSD } from './index.js';
import { GSDTools } from './gsd-tools.js';
// ─── Helpers ─────────────────────────────────────────────────────────────────
function makePhaseInfo(overrides: Partial<RoadmapPhaseInfo> = {}): RoadmapPhaseInfo {
return {
number: '1',
disk_status: 'not_started',
roadmap_complete: false,
phase_name: 'Auth',
...overrides,
};
}
function makePhaseResult(overrides: Partial<PhaseRunnerResult> = {}): PhaseRunnerResult {
return {
phaseNumber: '1',
phaseName: 'Auth',
steps: [],
success: true,
totalCostUsd: 0.50,
totalDurationMs: 5000,
...overrides,
};
}
function makeAnalysis(phases: RoadmapPhaseInfo[]): RoadmapAnalysis {
return { phases };
}
// ─── Tests ───────────────────────────────────────────────────────────────────
describe('GSD.run()', () => {
let gsd: GSD;
let mockRoadmapAnalyze: ReturnType<typeof vi.fn>;
let events: GSDEvent[];
beforeEach(() => {
vi.clearAllMocks();
gsd = new GSD({ projectDir: '/tmp/test-project' });
events = [];
// Capture emitted events
(gsd.eventStream.emitEvent as ReturnType<typeof vi.fn>).mockImplementation(
(event: GSDEvent) => events.push(event),
);
// Wire mock roadmapAnalyze on the GSDTools instance
mockRoadmapAnalyze = vi.fn();
vi.mocked(GSDTools).mockImplementation(function () {
return {
roadmapAnalyze: mockRoadmapAnalyze,
} as any;
});
});
it('discovers phases and calls runPhase for each incomplete one', async () => {
const phases = [
makePhaseInfo({ number: '1', phase_name: 'Auth', roadmap_complete: false }),
makePhaseInfo({ number: '2', phase_name: 'Dashboard', roadmap_complete: false }),
];
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis(phases)) // initial discovery
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
])) // after phase 1
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: true }),
])); // after phase 2
const runPhaseSpy = vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '1' }))
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '2' }));
const result = await gsd.run('build the app');
expect(result.success).toBe(true);
expect(result.phases).toHaveLength(2);
expect(runPhaseSpy).toHaveBeenCalledTimes(2);
expect(runPhaseSpy).toHaveBeenCalledWith('1', undefined);
expect(runPhaseSpy).toHaveBeenCalledWith('2', undefined);
});
it('skips phases where roadmap_complete === true', async () => {
const phases = [
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
makePhaseInfo({ number: '3', roadmap_complete: true }),
];
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis(phases))
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: true }),
makePhaseInfo({ number: '3', roadmap_complete: true }),
]));
const runPhaseSpy = vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '2' }));
const result = await gsd.run('build it');
expect(result.success).toBe(true);
expect(result.phases).toHaveLength(1);
expect(runPhaseSpy).toHaveBeenCalledTimes(1);
expect(runPhaseSpy).toHaveBeenCalledWith('2', undefined);
});
it('re-discovers phases after each completion to catch dynamically inserted phases', async () => {
// Initially phase 1 and 2 are incomplete
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: false }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]))
// After phase 1, a new phase 1.5 was inserted
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '1.5', phase_name: 'Hotfix', roadmap_complete: false }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]))
// After phase 1.5 completes
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '1.5', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]))
// After phase 2 completes
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '1.5', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: true }),
]));
const runPhaseSpy = vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '1' }))
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '1.5', phaseName: 'Hotfix' }))
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '2' }));
const result = await gsd.run('build it');
expect(result.success).toBe(true);
expect(result.phases).toHaveLength(3);
expect(runPhaseSpy).toHaveBeenCalledTimes(3);
// The dynamically inserted phase 1.5 was executed
expect(runPhaseSpy).toHaveBeenNthCalledWith(2, '1.5', undefined);
});
it('aggregates costs from all phases', async () => {
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: false }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]))
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]))
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: true }),
]));
vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult({ totalCostUsd: 1.25 }))
.mockResolvedValueOnce(makePhaseResult({ totalCostUsd: 0.75 }));
const result = await gsd.run('build it');
expect(result.totalCostUsd).toBeCloseTo(2.0, 2);
});
it('emits MilestoneStart and MilestoneComplete events', async () => {
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: false }),
]))
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
]));
vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult({ totalCostUsd: 0.50 }));
await gsd.run('build it');
const startEvents = events.filter(e => e.type === GSDEventType.MilestoneStart);
const completeEvents = events.filter(e => e.type === GSDEventType.MilestoneComplete);
expect(startEvents).toHaveLength(1);
expect(completeEvents).toHaveLength(1);
const start = startEvents[0] as any;
expect(start.phaseCount).toBe(1);
expect(start.prompt).toBe('build it');
const complete = completeEvents[0] as any;
expect(complete.success).toBe(true);
expect(complete.phasesCompleted).toBe(1);
expect(complete.totalCostUsd).toBeCloseTo(0.50, 2);
});
it('stops on phase failure', async () => {
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: false }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]));
vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '1', success: false }));
const result = await gsd.run('build it');
expect(result.success).toBe(false);
expect(result.phases).toHaveLength(1);
// Phase 2 was never started
});
it('handles empty phase list', async () => {
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([]));
const runPhaseSpy = vi.spyOn(gsd, 'runPhase');
const result = await gsd.run('build it');
expect(result.success).toBe(true);
expect(result.phases).toHaveLength(0);
expect(runPhaseSpy).not.toHaveBeenCalled();
expect(result.totalCostUsd).toBe(0);
});
it('sorts phases numerically, not lexicographically', async () => {
const phases = [
makePhaseInfo({ number: '10', phase_name: 'Ten', roadmap_complete: false }),
makePhaseInfo({ number: '2', phase_name: 'Two', roadmap_complete: false }),
makePhaseInfo({ number: '1.5', phase_name: 'OnePointFive', roadmap_complete: false }),
];
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis(phases))
// After phase 1.5
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1.5', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
makePhaseInfo({ number: '10', roadmap_complete: false }),
]))
// After phase 2
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1.5', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: true }),
makePhaseInfo({ number: '10', roadmap_complete: false }),
]))
// After phase 10
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1.5', roadmap_complete: true }),
makePhaseInfo({ number: '2', roadmap_complete: true }),
makePhaseInfo({ number: '10', roadmap_complete: true }),
]));
const executionOrder: string[] = [];
vi.spyOn(gsd, 'runPhase').mockImplementation(async (phaseNumber: string) => {
executionOrder.push(phaseNumber);
return makePhaseResult({ phaseNumber });
});
await gsd.run('build it');
// Numeric order: 1.5 → 2 → 10 (not lexicographic: "10" < "2")
expect(executionOrder).toEqual(['1.5', '2', '10']);
});
it('handles phase throwing an unexpected error', async () => {
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', phase_name: 'Broken', roadmap_complete: false }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]));
vi.spyOn(gsd, 'runPhase')
.mockRejectedValueOnce(new Error('Unexpected explosion'));
const result = await gsd.run('build it');
expect(result.success).toBe(false);
expect(result.phases).toHaveLength(1);
expect(result.phases[0].success).toBe(false);
expect(result.phases[0].phaseNumber).toBe('1');
});
it('passes MilestoneRunnerOptions through to runPhase', async () => {
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: false }),
]))
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: true }),
]));
const runPhaseSpy = vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult());
const opts: MilestoneRunnerOptions = {
model: 'claude-sonnet-4-6',
maxBudgetPerStep: 2.0,
onPhaseComplete: vi.fn(),
};
await gsd.run('build it', opts);
expect(runPhaseSpy).toHaveBeenCalledWith('1', opts);
});
it('respects onPhaseComplete returning stop', async () => {
mockRoadmapAnalyze
.mockResolvedValueOnce(makeAnalysis([
makePhaseInfo({ number: '1', roadmap_complete: false }),
makePhaseInfo({ number: '2', roadmap_complete: false }),
]));
vi.spyOn(gsd, 'runPhase')
.mockResolvedValueOnce(makePhaseResult({ phaseNumber: '1' }));
const result = await gsd.run('build it', {
onPhaseComplete: async () => 'stop',
});
// Only 1 phase was executed because callback said stop
expect(result.phases).toHaveLength(1);
expect(result.success).toBe(true);
});
});

View File

@@ -1,70 +0,0 @@
import { readFileSync } from 'node:fs';
import { fileURLToPath } from 'node:url';
interface RuntimeTierEntry {
model: string;
reasoning_effort?: string;
}
type RuntimeTierTable = Record<string, Record<string, RuntimeTierEntry | null>>;
interface AgentCatalogEntry {
golden: 'opus' | 'sonnet' | 'haiku';
balanced: 'opus' | 'sonnet' | 'haiku';
budget: 'opus' | 'sonnet' | 'haiku';
phaseType: string;
routingTier: 'light' | 'standard' | 'heavy';
}
interface ModelCatalog {
profiles: string[];
phaseTypes: string[];
adaptiveTierMap: Record<'light' | 'standard' | 'heavy', 'opus' | 'sonnet' | 'haiku'>;
runtimeTierDefaults: RuntimeTierTable;
agents: Record<string, AgentCatalogEntry>;
}
const CATALOG_PATH = new URL('../shared/model-catalog.json', import.meta.url);
export const catalog: ModelCatalog = JSON.parse(readFileSync(fileURLToPath(CATALOG_PATH), 'utf-8'));
export const VALID_PROFILES: string[] = [...catalog.profiles];
export const SUPPORTED_RUNTIMES = Object.keys(catalog.runtimeTierDefaults);
export type Runtime = (typeof SUPPORTED_RUNTIMES)[number];
export const MODEL_PROFILES: Record<string, Record<string, string>> = Object.fromEntries(
Object.entries(catalog.agents).map(([agent, meta]) => [agent, {
quality: meta.golden,
balanced: meta.balanced,
budget: meta.budget,
adaptive: catalog.adaptiveTierMap[meta.routingTier],
}])
);
export const AGENT_TO_PHASE_TYPE: Record<string, string> = Object.fromEntries(
Object.entries(catalog.agents).map(([agent, meta]) => [agent, meta.phaseType])
);
export const AGENT_DEFAULT_TIERS: Record<string, string> = Object.fromEntries(
Object.entries(catalog.agents).map(([agent, meta]) => [agent, meta.routingTier])
);
export function getAgentToModelMapForProfile(normalizedProfile: string): Record<string, string> {
const profile = VALID_PROFILES.includes(normalizedProfile) ? normalizedProfile : 'balanced';
const out: Record<string, string> = {};
for (const [agent, profiles] of Object.entries(MODEL_PROFILES)) {
out[agent] = profile === 'inherit' ? 'inherit' : profiles[profile] ?? profiles.balanced;
}
return out;
}
export function resolveRuntimeTierDefault(runtime: string, alias: 'opus' | 'sonnet' | 'haiku'): RuntimeTierEntry | null {
return catalog.runtimeTierDefaults[runtime]?.[alias] ?? null;
}
export function runtimesWithReasoningEffort(): Set<string> {
return new Set(
Object.entries(catalog.runtimeTierDefaults)
.filter(([, tiers]) => Object.values(tiers).some((entry) => entry && entry.reasoning_effort))
.map(([runtime]) => runtime)
);
}

View File

@@ -1,259 +0,0 @@
/**
* Phase-aware prompt factory — assembles complete prompts for each phase type.
*
* Reads workflow .md + agent .md files from disk (D006), extracts structured
* blocks (<role>, <purpose>, <process>), and composes system prompts with
* injected context files per phase type.
*/
import { readFile } from 'node:fs/promises';
import { join } from 'node:path';
import { fileURLToPath } from 'node:url';
import type { ContextFiles, ParsedPlan } from './types.js';
import { PhaseType } from './types.js';
import { buildExecutorPrompt } from './prompt-builder.js';
import { PHASE_AGENT_MAP } from './tool-scoping.js';
import { sanitizePrompt } from './prompt-sanitizer.js';
import { resolveLegacyInstallDir } from './sdk-package-compatibility.js';
// ─── Workflow file mapping ───────────────────────────────────────────────────
/**
* Maps phase types to their workflow file names.
*/
const PHASE_WORKFLOW_MAP: Record<PhaseType, string> = {
[PhaseType.Execute]: 'execute-plan.md',
[PhaseType.Research]: 'research-phase.md',
[PhaseType.Plan]: 'plan-phase.md',
[PhaseType.Verify]: 'verify-phase.md',
[PhaseType.Discuss]: 'discuss-phase.md',
[PhaseType.Repair]: 'execute-plan.md',
};
// ─── XML block extraction ────────────────────────────────────────────────────
/**
* Extract content from an XML-style block (e.g., <purpose>...</purpose>).
* Returns the trimmed inner content, or empty string if not found.
*/
export function extractBlock(content: string, tagName: string): string {
const regex = new RegExp(`<${tagName}[^>]*>([\\s\\S]*?)<\\/${tagName}>`, 'i');
const match = content.match(regex);
return match ? match[1].trim() : '';
}
/**
* Extract all <step> blocks from a workflow's <process> section.
* Returns an array of step contents with their name attributes.
*/
export function extractSteps(processContent: string): Array<{ name: string; content: string }> {
const steps: Array<{ name: string; content: string }> = [];
const stepRegex = /<step\s+name="([^"]*)"[^>]*>([\s\S]*?)<\/step>/gi;
let match;
while ((match = stepRegex.exec(processContent)) !== null) {
steps.push({
name: match[1],
content: match[2].trim(),
});
}
return steps;
}
// ─── YAML frontmatter stripping ─────────────────────────────────────────────
/**
* Strip YAML frontmatter (---...---) from an agent definition file,
* returning only the markdown/XML content body.
*/
export function stripYamlFrontmatter(content: string): string {
const match = content.match(/^---\s*\n[\s\S]*?\n---\s*\n?([\s\S]*)$/);
return match ? match[1].trim() : content.trim();
}
// ─── PromptFactory class ─────────────────────────────────────────────────────
export class PromptFactory {
private readonly workflowsDir: string;
private readonly agentsDir: string;
private readonly projectAgentsDir?: string;
private readonly sdkPromptsDir: string;
private readonly projectDir?: string;
constructor(options?: {
gsdInstallDir?: string;
agentsDir?: string;
projectAgentsDir?: string;
sdkPromptsDir?: string;
projectDir?: string;
}) {
const gsdInstallDir = options?.gsdInstallDir ?? resolveLegacyInstallDir();
this.workflowsDir = join(gsdInstallDir, 'workflows');
this.agentsDir = options?.agentsDir ?? join(gsdInstallDir, '..', 'agents');
this.projectAgentsDir = options?.projectAgentsDir;
this.projectDir = options?.projectDir;
// SDK prompts dir: explicit override → package-relative default via import.meta.url
this.sdkPromptsDir =
options?.sdkPromptsDir ??
join(fileURLToPath(new URL('.', import.meta.url)), '..', 'prompts');
}
/**
* Build a complete prompt for the given phase type.
*
* For execute phase with a plan, delegates to buildExecutorPrompt().
* For other phases, assembles: role + purpose + process steps + context.
*/
async buildPrompt(
phaseType: PhaseType,
plan: ParsedPlan | null,
contextFiles: ContextFiles,
phaseDir?: string,
): Promise<string> {
// Execute phase with a plan: delegate to existing buildExecutorPrompt
if (phaseType === PhaseType.Execute && plan) {
const agentDef = await this.loadAgentDef(phaseType);
return sanitizePrompt(buildExecutorPrompt(plan, { agentDef, phaseDir }), this.projectDir);
}
// Prompt assembly order is cache-optimized (#1614):
// Stable prefix (deterministic per phase type) → cached by Anthropic at 0.1x cost
// Variable suffix (.planning/ files) → uncached, changes per project/run
const sections: string[] = [];
// ── STABLE PREFIX (cacheable across runs for the same phase type) ──
// ── Full agent definition ──
// Include the complete agent definition (minus YAML frontmatter), not just
// the <role> block. The real agents have critical instructions in sections
// like <philosophy>, <task_breakdown>, <plan_format>, <execution_flow>,
// <scope_estimation>, <context_fidelity>, <checkpoints>, etc.
const agentDef = await this.loadAgentDef(phaseType);
if (agentDef) {
const agentContent = stripYamlFrontmatter(agentDef);
if (agentContent) {
sections.push(`## Agent Instructions\n\n${agentContent}`);
}
}
// ── Workflow purpose + process ──
const workflow = await this.loadWorkflowFile(phaseType);
if (workflow) {
const purpose = extractBlock(workflow, 'purpose');
if (purpose) {
sections.push(`## Purpose\n\n${purpose}`);
}
const process = extractBlock(workflow, 'process');
if (process) {
const steps = extractSteps(process);
if (steps.length > 0) {
const stepBlocks = steps.map((s) => `### ${s.name}\n\n${s.content}`).join('\n\n');
sections.push(`## Process\n\n${stepBlocks}`);
}
}
}
// ── VARIABLE SUFFIX (project-specific, changes per run) ──
// ── Context files ──
const contextSection = this.formatContextFiles(contextFiles);
if (contextSection) {
sections.push(contextSection);
}
return sanitizePrompt(sections.join('\n\n'), this.projectDir);
}
/**
* Load the workflow file for a phase type.
* Tries installed GSD workflows first (the complete, up-to-date versions),
* then falls back to SDK bundled copies only if installed not found.
* Returns the raw content, or undefined if not found.
*/
async loadWorkflowFile(phaseType: PhaseType): Promise<string | undefined> {
const filename = PHASE_WORKFLOW_MAP[phaseType];
// Try installed GSD workflows first (complete versions)
const paths = [
join(this.workflowsDir, filename),
join(this.sdkPromptsDir, 'workflows', filename),
];
for (const p of paths) {
try {
return await readFile(p, 'utf-8');
} catch {
// Not found at this path, try next
}
}
return undefined;
}
/**
* Load the agent definition for a phase type.
* Tries installed agents first (the complete, up-to-date versions),
* then SDK bundled copies as last resort.
* Returns undefined if no agent is mapped or file not found.
*/
async loadAgentDef(phaseType: PhaseType): Promise<string | undefined> {
const agentFilename = PHASE_AGENT_MAP[phaseType];
if (!agentFilename) return undefined;
// Priority: installed agents → project-level → SDK bundled (last resort)
const paths = [
join(this.agentsDir, agentFilename),
];
if (this.projectAgentsDir) {
paths.push(join(this.projectAgentsDir, agentFilename));
}
// SDK bundled copies are last resort only
paths.push(join(this.sdkPromptsDir, 'agents', agentFilename));
for (const p of paths) {
try {
return await readFile(p, 'utf-8');
} catch {
// Not found at this path, try next
}
}
return undefined;
}
/**
* Format context files into a prompt section.
*/
private formatContextFiles(contextFiles: ContextFiles): string | null {
const entries: string[] = [];
const fileLabels: Record<keyof ContextFiles, string> = {
state: 'Project State (STATE.md)',
roadmap: 'Roadmap (ROADMAP.md)',
context: 'Context (CONTEXT.md)',
research: 'Research (RESEARCH.md)',
requirements: 'Requirements (REQUIREMENTS.md)',
config: 'Config (config.json)',
plan: 'Plan (PLAN.md)',
summary: 'Summary (SUMMARY.md)',
};
for (const [key, label] of Object.entries(fileLabels)) {
const content = contextFiles[key as keyof ContextFiles];
if (content) {
entries.push(`### ${label}\n\n${content}`);
}
}
if (entries.length === 0) return null;
return `## Context\n\n${entries.join('\n\n')}`;
}
}
export { PHASE_WORKFLOW_MAP };

View File

@@ -1,377 +0,0 @@
/**
* Integration test — proves PhaseRunner state machine works against real gsd-tools.cjs.
*
* Creates a temp `.planning/` directory structure, instantiates real GSDTools,
* and exercises the state machine. Sessions will fail (no Claude CLI in CI) but
* the state machine's control flow, event emission, and error capture are proven.
*/
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
import { mkdtemp, mkdir, writeFile, rm } from 'node:fs/promises';
import { existsSync } from 'node:fs';
import { join } from 'node:path';
import { tmpdir } from 'node:os';
import { GSDTools, resolveGsdToolsPath } from './gsd-tools.js';
import { PhaseRunner } from './phase-runner.js';
import type { PhaseRunnerDeps } from './phase-runner.js';
import { ContextEngine } from './context-engine.js';
import { PromptFactory } from './phase-prompt.js';
import { GSDEventStream } from './event-stream.js';
import { loadConfig } from './config.js';
import type { GSDEvent } from './types.js';
import { GSDEventType, PhaseStepType } from './types.js';
// ─── Helpers ─────────────────────────────────────────────────────────────────
const GSD_TOOLS_PATH = resolveGsdToolsPath(process.cwd());
const gsdToolsAvailable = existsSync(GSD_TOOLS_PATH);
async function createTempPlanningDir(): Promise<string> {
const tmpDir = await mkdtemp(join(tmpdir(), 'gsd-sdk-phase-int-'));
// Create .planning structure
const planningDir = join(tmpDir, '.planning');
const phasesDir = join(planningDir, 'phases');
const phaseDir = join(phasesDir, '01-integration-test');
await mkdir(phaseDir, { recursive: true });
// config.json
await writeFile(
join(planningDir, 'config.json'),
JSON.stringify({
model_profile: 'balanced',
commit_docs: false,
workflow: {
research: true,
verifier: true,
auto_advance: true,
skip_discuss: false,
},
}),
);
// ROADMAP.md — required for roadmap_exists
await writeFile(join(planningDir, 'ROADMAP.md'), '# Roadmap\n\n## Phase 01: Integration Test\n');
// CONTEXT.md in phase dir — triggers has_context=true → discuss is skipped
await writeFile(
join(phaseDir, 'CONTEXT.md'),
'# Context\n\nThis is an integration test phase with pre-existing context.\n',
);
return tmpDir;
}
// ─── Test suite ──────────────────────────────────────────────────────────────
describe.skipIf(!gsdToolsAvailable)('Integration: PhaseRunner against real gsd-tools.cjs', () => {
let tmpDir: string;
let tools: GSDTools;
beforeAll(async () => {
tmpDir = await createTempPlanningDir();
tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: GSD_TOOLS_PATH,
timeoutMs: 10_000,
});
});
afterAll(async () => {
if (tmpDir) {
await rm(tmpDir, { recursive: true, force: true });
}
});
// ── Test 1: initPhaseOp returns valid PhaseOpInfo ──
it('initPhaseOp returns valid PhaseOpInfo for temp phase', async () => {
const info = await tools.initPhaseOp('01');
expect(info.phase_found).toBe(true);
expect(info.phase_number).toBe('01');
expect(info.phase_name).toBe('integration-test');
expect(info.phase_dir).toBe('.planning/phases/01-integration-test');
expect(info.has_context).toBe(true);
expect(info.has_plans).toBe(false);
expect(info.plan_count).toBe(0);
expect(info.roadmap_exists).toBe(true);
expect(info.planning_exists).toBe(true);
});
it('initPhaseOp returns phase_found=false for nonexistent phase', async () => {
const info = await tools.initPhaseOp('99');
expect(info.phase_found).toBe(false);
expect(info.has_context).toBe(false);
expect(info.plan_count).toBe(0);
});
// ── Test 2: PhaseRunner state machine control flow ──
it('PhaseRunner emits lifecycle events and captures session errors gracefully', { timeout: 300_000 }, async () => {
const eventStream = new GSDEventStream();
const config = await loadConfig(tmpDir);
const contextEngine = new ContextEngine(tmpDir);
const promptFactory = new PromptFactory();
const events: GSDEvent[] = [];
eventStream.on('event', (e: GSDEvent) => events.push(e));
const deps: PhaseRunnerDeps = {
projectDir: tmpDir,
tools,
promptFactory,
contextEngine,
eventStream,
config,
};
const runner = new PhaseRunner(deps);
// Tight budget/turns so each session finishes fast
const result = await runner.run('01', {
maxTurnsPerStep: 2,
maxBudgetPerStep: 0.10,
});
// ── (a) Phase start event emitted ──
const phaseStartEvents = events.filter(e => e.type === GSDEventType.PhaseStart);
expect(phaseStartEvents).toHaveLength(1);
const phaseStart = phaseStartEvents[0]!;
if (phaseStart.type === GSDEventType.PhaseStart) {
expect(phaseStart.phaseNumber).toBe('01');
expect(phaseStart.phaseName).toBe('integration-test');
}
// ── (b) Discuss should be skipped (has_context=true) ──
// No discuss step in results since it was skipped
const discussSteps = result.steps.filter(s => s.step === PhaseStepType.Discuss);
expect(discussSteps).toHaveLength(0);
// ── (c) Step start events emitted for attempted steps ──
const stepStartEvents = events.filter(e => e.type === GSDEventType.PhaseStepStart);
expect(stepStartEvents.length).toBeGreaterThanOrEqual(1);
// ── (d) Step results are properly structured ──
// With CLI available, sessions may succeed or fail depending on budget/turns.
// Either way, each step result must have correct structure.
expect(result.steps.length).toBeGreaterThanOrEqual(1);
for (const step of result.steps) {
expect(Object.values(PhaseStepType)).toContain(step.step);
expect(typeof step.success).toBe('boolean');
expect(typeof step.durationMs).toBe('number');
// Failed steps may or may not have an error message
// (e.g. advance step can fail without explicit error string)
}
// ── (e) Phase complete event emitted ──
const phaseCompleteEvents = events.filter(e => e.type === GSDEventType.PhaseComplete);
expect(phaseCompleteEvents).toHaveLength(1);
// ── (f) Result structure is valid ──
expect(result.phaseNumber).toBe('01');
expect(result.phaseName).toBe('integration-test');
expect(typeof result.totalCostUsd).toBe('number');
expect(typeof result.totalDurationMs).toBe('number');
expect(result.totalDurationMs).toBeGreaterThan(0);
});
// ── Test 3: PhaseRunner with nonexistent phase throws ──
it('PhaseRunner throws PhaseRunnerError for nonexistent phase', async () => {
const eventStream = new GSDEventStream();
const config = await loadConfig(tmpDir);
const contextEngine = new ContextEngine(tmpDir);
const promptFactory = new PromptFactory();
const deps: PhaseRunnerDeps = {
projectDir: tmpDir,
tools,
promptFactory,
contextEngine,
eventStream,
config,
};
const runner = new PhaseRunner(deps);
await expect(runner.run('99')).rejects.toThrow('Phase 99 not found on disk');
});
// ── Test 4: GSD.runPhase() public API delegates correctly ──
it('GSD.runPhase() creates collaborators and delegates to PhaseRunner', { timeout: 300_000 }, async () => {
// Import GSD here to test the public API wiring
const { GSD } = await import('./index.js');
const gsd = new GSD({ projectDir: tmpDir });
const events: GSDEvent[] = [];
gsd.onEvent((e) => events.push(e));
const result = await gsd.runPhase('01', {
maxTurnsPerStep: 2,
maxBudgetPerStep: 0.10,
});
// Proves the full wiring works: GSD → PhaseRunner → GSDTools → gsd-tools.cjs
expect(result.phaseNumber).toBe('01');
expect(result.phaseName).toBe('integration-test');
expect(result.steps.length).toBeGreaterThanOrEqual(1);
expect(events.some(e => e.type === GSDEventType.PhaseStart)).toBe(true);
expect(events.some(e => e.type === GSDEventType.PhaseComplete)).toBe(true);
});
});
// ─── Wave / phasePlanIndex Integration Tests ─────────────────────────────────
/**
* Creates a temp `.planning/` directory with multi-wave plan files.
* - Plans 01 and 02 are wave 1 (parallel)
* - Plan 03 is wave 2 (depends on wave 1)
* - Plan 01 has a SUMMARY.md (marks it as completed)
*/
async function createMultiWavePlanningDir(): Promise<string> {
const tmpDir = await mkdtemp(join(tmpdir(), 'gsd-sdk-wave-int-'));
const planningDir = join(tmpDir, '.planning');
const phaseDir = join(planningDir, 'phases', '01-wave-test');
await mkdir(phaseDir, { recursive: true });
// config.json — with parallelization enabled
await writeFile(
join(planningDir, 'config.json'),
JSON.stringify({
model_profile: 'balanced',
commit_docs: false,
parallelization: true,
workflow: {
research: true,
verifier: true,
auto_advance: true,
skip_discuss: false,
},
}),
);
// ROADMAP.md
await writeFile(join(planningDir, 'ROADMAP.md'), '# Roadmap\n\n## Phase 01: Wave Test\n');
const planTemplate = (id: string, wave: number, dependsOn: string[] = []) => `---
phase: "01"
plan: "${id}"
type: "feature"
wave: ${wave}
depends_on: [${dependsOn.map(d => `"${d}"`).join(', ')}]
files_modified: ["src/${id}.ts"]
autonomous: true
requirements: []
must_haves:
truths: ["${id} exists"]
artifacts: []
key_links: []
---
# Plan: ${id}
<task type="code" name="Create ${id}" files="src/${id}.ts">
<read_first>none</read_first>
<action>Create ${id}</action>
<verify>File exists</verify>
<acceptance_criteria>
- File exists
</acceptance_criteria>
<done>Done</done>
</task>
`;
// Wave 1 plans (parallel)
await writeFile(join(phaseDir, '01-wave-test-01-PLAN.md'), planTemplate('01-wave-test-01', 1));
await writeFile(join(phaseDir, '01-wave-test-02-PLAN.md'), planTemplate('01-wave-test-02', 1));
// Wave 2 plan (depends on wave 1)
await writeFile(
join(phaseDir, '01-wave-test-03-PLAN.md'),
planTemplate('01-wave-test-03', 2, ['01-wave-test-01']),
);
// Summary for plan 01 — marks it as completed
await writeFile(
join(phaseDir, '01-wave-test-01-SUMMARY.md'),
`---\nresult: pass\nplan: "01-wave-test-01"\ncost_usd: 0.01\nduration_ms: 1000\n---\n\n# Summary\n\nAll tasks completed.\n`,
);
return tmpDir;
}
describe.skipIf(!gsdToolsAvailable)('Integration: phasePlanIndex and wave execution', () => {
let tmpDir: string;
let tools: GSDTools;
beforeAll(async () => {
tmpDir = await createMultiWavePlanningDir();
tools = new GSDTools({
projectDir: tmpDir,
gsdToolsPath: GSD_TOOLS_PATH,
timeoutMs: 10_000,
});
});
afterAll(async () => {
if (tmpDir) {
await rm(tmpDir, { recursive: true, force: true });
}
});
it('phasePlanIndex returns typed PhasePlanIndex with correct wave grouping', async () => {
const index = await tools.phasePlanIndex('01');
// 3 plans total
expect(index.plans).toHaveLength(3);
// Wave grouping: wave 1 has 2 plans, wave 2 has 1
expect(index.waves['1']).toHaveLength(2);
expect(index.waves['1']).toContain('01-wave-test-01');
expect(index.waves['1']).toContain('01-wave-test-02');
expect(index.waves['2']).toHaveLength(1);
expect(index.waves['2']).toContain('01-wave-test-03');
// Incomplete: plan 01 has summary so only 02 and 03 are incomplete
expect(index.incomplete).toHaveLength(2);
expect(index.incomplete).toContain('01-wave-test-02');
expect(index.incomplete).toContain('01-wave-test-03');
// All autonomous → no checkpoints
expect(index.has_checkpoints).toBe(false);
// Phase ID correct
expect(index.phase).toBe('01');
});
it('phasePlanIndex marks has_summary correctly per plan', async () => {
const index = await tools.phasePlanIndex('01');
// Plan 01 has a SUMMARY.md on disk
const plan01 = index.plans.find(p => p.id === '01-wave-test-01');
expect(plan01).toBeDefined();
expect(plan01!.has_summary).toBe(true);
// Plans 02 and 03 have no summary
const plan02 = index.plans.find(p => p.id === '01-wave-test-02');
expect(plan02).toBeDefined();
expect(plan02!.has_summary).toBe(false);
const plan03 = index.plans.find(p => p.id === '01-wave-test-03');
expect(plan03).toBeDefined();
expect(plan03!.has_summary).toBe(false);
});
it('phasePlanIndex for nonexistent phase returns empty plans', async () => {
const index = await tools.phasePlanIndex('99');
expect(index.plans).toHaveLength(0);
expect(Object.keys(index.waves)).toHaveLength(0);
expect(index.incomplete).toHaveLength(0);
expect(index.has_checkpoints).toBe(false);
});
});

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -1,579 +0,0 @@
import { describe, it, expect } from 'vitest';
import { parsePlan, parseTasks, extractFrontmatter } from './plan-parser.js';
// ─── Fixtures ────────────────────────────────────────────────────────────────
const FULL_PLAN = `---
phase: 03-features
plan: 01
type: execute
wave: 2
depends_on: [01-01, 01-02]
files_modified: [src/models/user.ts, src/api/users.ts, src/components/UserList.tsx]
autonomous: true
requirements: [R001, R003]
must_haves:
truths:
- "User can see existing messages"
- "User can send a message"
artifacts:
- path: src/components/Chat.tsx
provides: Message list rendering
min_lines: 30
- path: src/app/api/chat/route.ts
provides: Message CRUD operations
key_links:
- from: src/components/Chat.tsx
to: /api/chat
via: fetch in useEffect
pattern: "fetch.*api/chat"
---
<objective>
Implement complete User feature as vertical slice.
Purpose: Self-contained user management that can run parallel to other features.
Output: User model, API endpoints, and UI components.
</objective>
<execution_context>
@~/.claude/get-shit-done/workflows/execute-plan.md
@~/.claude/get-shit-done/templates/summary.md
</execution_context>
<context>
@.planning/PROJECT.md
@.planning/ROADMAP.md
@.planning/STATE.md
# Only include SUMMARY refs if genuinely needed
@src/relevant/source.ts
</context>
<tasks>
<task type="auto">
<name>Task 1: Create User model</name>
<files>src/models/user.ts</files>
<read_first>src/existing/types.ts, src/config/db.ts</read_first>
<action>Define User type with id, email, name, createdAt. Export TypeScript interface.</action>
<verify>tsc --noEmit passes</verify>
<acceptance_criteria>
- User type is exported from src/models/user.ts
- Type includes id, email, name, createdAt fields
</acceptance_criteria>
<done>User type exported and usable</done>
</task>
<task type="auto">
<name>Task 2: Create User API endpoints</name>
<files>src/api/users.ts, src/api/middleware.ts</files>
<action>GET /users (list), GET /users/:id (single), POST /users (create). Use User type from model.</action>
<verify>fetch tests pass for all endpoints</verify>
<done>All CRUD operations work</done>
</task>
<task type="checkpoint:human-verify" gate="blocking">
<name>Verify UI visually</name>
<files>src/components/UserList.tsx</files>
<action>Start dev server and present for review.</action>
<verify>User confirms layout is correct</verify>
<done>Visual verification passed</done>
</task>
</tasks>
<verification>
- [ ] npm run build succeeds
- [ ] API endpoints respond correctly
</verification>
<success_criteria>
- All tasks completed
- User feature works end-to-end
</success_criteria>
`;
const MINIMAL_PLAN = `---
phase: 01-test
plan: 01
type: execute
wave: 1
depends_on: []
files_modified: []
autonomous: true
requirements: []
must_haves:
truths: []
artifacts: []
key_links: []
---
<objective>
Minimal test plan.
</objective>
<tasks>
<task type="auto">
<name>Single task</name>
<files>output.txt</files>
<action>Create output.txt</action>
<verify>test -f output.txt</verify>
<done>File exists</done>
</task>
</tasks>
`;
const MULTILINE_ACTION_PLAN = `---
phase: 02-impl
plan: 01
type: execute
wave: 1
depends_on: []
files_modified: [src/server.ts]
autonomous: true
requirements: [R005]
must_haves:
truths: []
artifacts: []
key_links: []
---
<tasks>
<task type="auto">
<name>Build server with config</name>
<files>src/server.ts</files>
<action>
Create the Express server with the following setup:
1. Import express and configure middleware
2. Add routes for health check and API
3. Configure error handling with proper types:
- ValidationError => 400
- NotFoundError => 404
- Default => 500
Example code structure:
\`\`\`typescript
const app = express();
app.get('/health', (req, res) => {
res.json({ status: 'ok' });
});
\`\`\`
Make sure to handle the edge case where \`req.body\` contains
angle brackets like <script> or XML-like content.
</action>
<verify>npm run build && curl localhost:3000/health</verify>
<done>Server starts and health endpoint returns 200</done>
</task>
</tasks>
`;
// ─── Tests ───────────────────────────────────────────────────────────────────
describe('extractFrontmatter', () => {
it('extracts basic key-value pairs', () => {
const result = extractFrontmatter(FULL_PLAN);
expect(result.phase).toBe('03-features');
expect(result.plan).toBe('01');
expect(result.type).toBe('execute');
});
it('coerces numeric values', () => {
const result = extractFrontmatter(FULL_PLAN);
expect(result.wave).toBe(2);
});
it('coerces boolean values', () => {
const result = extractFrontmatter(FULL_PLAN);
expect(result.autonomous).toBe(true);
});
it('parses inline arrays', () => {
const result = extractFrontmatter(FULL_PLAN);
expect(result.depends_on).toEqual(['01-01', '01-02']);
expect(result.files_modified).toEqual([
'src/models/user.ts',
'src/api/users.ts',
'src/components/UserList.tsx',
]);
expect(result.requirements).toEqual(['R001', 'R003']);
});
it('parses empty inline arrays', () => {
const result = extractFrontmatter(MINIMAL_PLAN);
expect(result.depends_on).toEqual([]);
expect(result.files_modified).toEqual([]);
expect(result.requirements).toEqual([]);
});
it('returns empty object for content without frontmatter', () => {
const result = extractFrontmatter('# Just a heading\nSome content');
expect(result).toEqual({});
});
it('returns empty object for empty string', () => {
const result = extractFrontmatter('');
expect(result).toEqual({});
});
it('returns the LEADING block when body contains markdown horizontal rules', () => {
// Regression: LAST-block semantics picked up body separators as frontmatter (#3240)
const content = [
'---',
'wave: 3',
'autonomous: false',
'phase: 05-hardening',
'---',
'',
'## Section One',
'',
'---',
'',
'## Section Two',
'',
'---',
'',
'body text',
].join('\n');
const result = extractFrontmatter(content);
expect(result.wave).toBe(3);
expect(result.autonomous).toBe(false);
expect(result.phase).toBe('05-hardening');
});
it('returns the LEADING block when body contains embedded YAML in fenced code block', () => {
// Regression: LAST-block semantics matched YAML inside ```yaml fences (#3240)
const content = [
'---',
'wave: 2',
'autonomous: true',
'phase: 04-polish',
'---',
'',
'## Example',
'',
'```yaml',
'---',
'name: example',
'value: 99',
'---',
'```',
'',
'More body text.',
].join('\n');
const result = extractFrontmatter(content);
expect(result.wave).toBe(2);
expect(result.autonomous).toBe(true);
expect(result.phase).toBe('04-polish');
});
});
describe('parsePlan — frontmatter', () => {
it('parses all typed frontmatter fields', () => {
const result = parsePlan(FULL_PLAN);
const fm = result.frontmatter;
expect(fm.phase).toBe('03-features');
expect(fm.plan).toBe('01');
expect(fm.type).toBe('execute');
expect(fm.wave).toBe(2);
expect(fm.depends_on).toEqual(['01-01', '01-02']);
expect(fm.files_modified).toEqual([
'src/models/user.ts',
'src/api/users.ts',
'src/components/UserList.tsx',
]);
expect(fm.autonomous).toBe(true);
expect(fm.requirements).toEqual(['R001', 'R003']);
});
it('parses must_haves.truths', () => {
const result = parsePlan(FULL_PLAN);
expect(result.frontmatter.must_haves.truths).toEqual([
'User can see existing messages',
'User can send a message',
]);
});
it('parses must_haves.artifacts', () => {
const result = parsePlan(FULL_PLAN);
const artifacts = result.frontmatter.must_haves.artifacts;
expect(artifacts).toHaveLength(2);
expect(artifacts[0]).toMatchObject({
path: 'src/components/Chat.tsx',
provides: 'Message list rendering',
min_lines: 30,
});
expect(artifacts[1]).toMatchObject({
path: 'src/app/api/chat/route.ts',
provides: 'Message CRUD operations',
});
});
it('parses must_haves.key_links', () => {
const result = parsePlan(FULL_PLAN);
const links = result.frontmatter.must_haves.key_links;
expect(links).toHaveLength(1);
expect(links[0]).toMatchObject({
from: 'src/components/Chat.tsx',
to: '/api/chat',
via: 'fetch in useEffect',
pattern: 'fetch.*api/chat',
});
});
it('parses empty must_haves', () => {
const result = parsePlan(MINIMAL_PLAN);
expect(result.frontmatter.must_haves).toEqual({
truths: [],
artifacts: [],
key_links: [],
});
});
it('provides defaults for missing frontmatter', () => {
const result = parsePlan('<tasks></tasks>');
expect(result.frontmatter.phase).toBe('');
expect(result.frontmatter.wave).toBe(1);
expect(result.frontmatter.depends_on).toEqual([]);
expect(result.frontmatter.autonomous).toBe(true);
expect(result.frontmatter.must_haves).toEqual({
truths: [],
artifacts: [],
key_links: [],
});
});
});
describe('parsePlan — XML tasks', () => {
it('parses auto tasks', () => {
const result = parsePlan(FULL_PLAN);
expect(result.tasks).toHaveLength(3);
const task1 = result.tasks[0];
expect(task1.type).toBe('auto');
expect(task1.name).toBe('Task 1: Create User model');
expect(task1.files).toEqual(['src/models/user.ts']);
expect(task1.read_first).toEqual(['src/existing/types.ts', 'src/config/db.ts']);
expect(task1.action).toBe(
'Define User type with id, email, name, createdAt. Export TypeScript interface.',
);
expect(task1.verify).toBe('tsc --noEmit passes');
expect(task1.done).toBe('User type exported and usable');
});
it('parses checkpoint tasks', () => {
const result = parsePlan(FULL_PLAN);
const checkpoint = result.tasks[2];
expect(checkpoint.type).toBe('checkpoint:human-verify');
expect(checkpoint.name).toBe('Verify UI visually');
});
it('parses acceptance_criteria list', () => {
const result = parsePlan(FULL_PLAN);
expect(result.tasks[0].acceptance_criteria).toEqual([
'User type is exported from src/models/user.ts',
'Type includes id, email, name, createdAt fields',
]);
});
it('parses multiple files from comma-separated list', () => {
const result = parsePlan(FULL_PLAN);
const task2 = result.tasks[1];
expect(task2.files).toEqual(['src/api/users.ts', 'src/api/middleware.ts']);
});
it('handles missing optional elements', () => {
const result = parsePlan(FULL_PLAN);
const task2 = result.tasks[1];
// Task 2 has no read_first or acceptance_criteria
expect(task2.read_first).toEqual([]);
expect(task2.acceptance_criteria).toEqual([]);
});
it('handles multiline action blocks', () => {
const result = parsePlan(MULTILINE_ACTION_PLAN);
expect(result.tasks).toHaveLength(1);
const task = result.tasks[0];
expect(task.action).toContain('Create the Express server');
expect(task.action).toContain('ValidationError => 400');
expect(task.action).toContain('app.get');
// The angle brackets inside action should be preserved
expect(task.action).toContain('angle brackets like <script>');
});
it('returns empty array for no tasks', () => {
const result = parsePlan('---\nphase: test\n---\n\nNo tasks here.');
expect(result.tasks).toEqual([]);
});
});
describe('parsePlan — sections', () => {
it('extracts objective', () => {
const result = parsePlan(FULL_PLAN);
expect(result.objective).toContain('Implement complete User feature');
expect(result.objective).toContain('Self-contained user management');
});
it('extracts execution_context references', () => {
const result = parsePlan(FULL_PLAN);
expect(result.execution_context).toEqual([
'~/.claude/get-shit-done/workflows/execute-plan.md',
'~/.claude/get-shit-done/templates/summary.md',
]);
});
it('extracts context references (skipping comments)', () => {
const result = parsePlan(FULL_PLAN);
expect(result.context_refs).toEqual([
'.planning/PROJECT.md',
'.planning/ROADMAP.md',
'.planning/STATE.md',
'src/relevant/source.ts',
]);
});
it('returns empty sections for missing blocks', () => {
const result = parsePlan(MINIMAL_PLAN);
expect(result.execution_context).toEqual([]);
// context_refs should be empty when no <context> block
expect(result.context_refs).toEqual([]);
});
});
describe('parsePlan — edge cases', () => {
it('handles empty string input', () => {
const result = parsePlan('');
expect(result.frontmatter.phase).toBe('');
expect(result.tasks).toEqual([]);
expect(result.raw).toBe('');
});
it('handles null-ish input without crashing', () => {
// @ts-expect-error — testing runtime guard
const result = parsePlan(null);
expect(result.tasks).toEqual([]);
expect(result.raw).toBe('');
});
it('handles undefined input without crashing', () => {
// @ts-expect-error — testing runtime guard
const result = parsePlan(undefined);
expect(result.tasks).toEqual([]);
expect(result.raw).toBe('');
});
it('preserves raw content', () => {
const result = parsePlan(MINIMAL_PLAN);
expect(result.raw).toBe(MINIMAL_PLAN);
});
it('handles malformed XML gracefully (unclosed tags)', () => {
const content = `---
phase: test
plan: 01
type: execute
wave: 1
depends_on: []
files_modified: []
autonomous: true
requirements: []
must_haves:
truths: []
artifacts: []
key_links: []
---
<tasks>
<task type="auto">
<name>Broken task</name>
<action>This action is never closed
</tasks>
`;
// Should not throw — just parse what it can
const result = parsePlan(content);
expect(result.tasks).toEqual([]); // Can't match <task>...</task> if malformed
expect(result.frontmatter.phase).toBe('test');
});
it('handles content with only frontmatter', () => {
const content = `---
phase: 01-solo
plan: 01
type: execute
wave: 1
depends_on: []
files_modified: []
autonomous: true
requirements: [R001]
must_haves:
truths: []
artifacts: []
key_links: []
---
`;
const result = parsePlan(content);
expect(result.frontmatter.phase).toBe('01-solo');
expect(result.frontmatter.requirements).toEqual(['R001']);
expect(result.tasks).toEqual([]);
expect(result.objective).toBe('');
});
it('handles code snippets with angle brackets inside action', () => {
const result = parsePlan(MULTILINE_ACTION_PLAN);
const action = result.tasks[0].action;
// The <script> inside the action text should be preserved (it's between <action>...</action>)
expect(action).toContain('<script>');
// TypeScript code block with angle brackets should be preserved
expect(action).toContain("res.json({ status: 'ok' })");
});
it('handles plan with boolean autonomous=false', () => {
const content = `---
phase: test
plan: 01
type: execute
wave: 1
depends_on: []
files_modified: []
autonomous: false
requirements: []
must_haves:
truths: []
artifacts: []
key_links: []
---
`;
const result = parsePlan(content);
expect(result.frontmatter.autonomous).toBe(false);
});
});
describe('parseTasks — standalone', () => {
it('extracts tasks from raw task XML', () => {
const xml = `
<tasks>
<task type="auto">
<name>Do something</name>
<files>a.ts</files>
<action>Build the thing</action>
<verify>npm test</verify>
<done>It works</done>
</task>
</tasks>
`;
const tasks = parseTasks(xml);
expect(tasks).toHaveLength(1);
expect(tasks[0].name).toBe('Do something');
expect(tasks[0].type).toBe('auto');
});
it('defaults task type to auto when attribute missing', () => {
const xml = `<tasks><task><name>No type</name><action>Do it</action></task></tasks>`;
const tasks = parseTasks(xml);
expect(tasks[0].type).toBe('auto');
});
});

View File

@@ -1,431 +0,0 @@
/**
* plan-parser.ts — Parse GSD-1 PLAN.md files into structured data.
*
* Extracts YAML frontmatter, XML task bodies, and markdown sections
* (<objective>, <execution_context>, <context>) from plan files.
*
* Ported from get-shit-done/bin/lib/frontmatter.cjs with TypeScript types.
*/
import { readFile } from 'node:fs/promises';
import type {
PlanFrontmatter,
PlanTask,
ParsedPlan,
MustHaves,
MustHaveArtifact,
MustHaveKeyLink,
} from './types.js';
// ─── YAML frontmatter extraction ─────────────────────────────────────────────
/**
* Extract frontmatter from a PLAN.md content string.
*
* Uses a stack-based parser that handles nested objects, inline arrays,
* multi-line arrays, and boolean/numeric coercion. Ported from the CJS
* reference implementation with the same edge-case coverage.
*
* Anchored at the start of the file — only the leading `---...---` block is
* considered canonical frontmatter. Body `---` separators and embedded YAML
* inside fenced code blocks are never picked up.
*/
export function extractFrontmatter(content: string): Record<string, unknown> {
const frontmatter: Record<string, unknown> = {};
// Anchored at file start — only the leading ---...--- block is canonical frontmatter.
// Body `---` separators and embedded YAML inside fenced code blocks are not matched.
const match = content.match(/^---\r?\n([\s\S]+?)\r?\n---/);
if (!match) return frontmatter;
const yaml = match[1];
const lines = yaml.split(/\r?\n/);
// Stack tracks nested objects: [{obj, key, indent}]
const stack: Array<{ obj: Record<string, unknown> | unknown[]; key: string | null; indent: number }> = [
{ obj: frontmatter, key: null, indent: -1 },
];
for (const line of lines) {
if (line.trim() === '') continue;
const indentMatch = line.match(/^(\s*)/);
const indent = indentMatch ? indentMatch[1].length : 0;
// Pop stack back to appropriate level
while (stack.length > 1 && indent <= stack[stack.length - 1].indent) {
stack.pop();
}
const current = stack[stack.length - 1];
const currentObj = current.obj as Record<string, unknown>;
// Key: value pattern
const keyMatch = line.match(/^(\s*)([a-zA-Z0-9_-]+):\s*(.*)/);
if (keyMatch) {
const key = keyMatch[2];
const value = keyMatch[3].trim();
if (value === '' || value === '[') {
// Key with no value or opening bracket — nested object or array (TBD)
currentObj[key] = value === '[' ? [] : {};
current.key = null;
stack.push({ obj: currentObj[key] as Record<string, unknown>, key: null, indent });
} else if (value.startsWith('[') && value.endsWith(']')) {
// Inline array: key: [a, b, c]
currentObj[key] = value
.slice(1, -1)
.split(',')
.map((s) => s.trim().replace(/^["']|["']$/g, ''))
.filter(Boolean);
current.key = null;
} else {
// Simple key: value — coerce booleans and numbers
const cleanValue = value.replace(/^["']|["']$/g, '');
currentObj[key] = coerceValue(cleanValue);
current.key = null;
}
} else if (line.trim().startsWith('- ')) {
// Array item — could be a plain string or "- key: value" (start of mapping item)
const afterDash = line.trim().slice(2);
const dashKvMatch = afterDash.match(/^([a-zA-Z0-9_-]+):\s*(.*)/);
// Determine the value to push
let itemToPush: unknown;
if (dashKvMatch) {
// "- key: value" → start of a mapping item (object in array)
const obj: Record<string, unknown> = {};
const val = dashKvMatch[2].trim().replace(/^["']|["']$/g, '');
obj[dashKvMatch[1]] = coerceValue(val);
itemToPush = obj;
} else {
const itemValue = afterDash.replace(/^["']|["']$/g, '');
itemToPush = coerceValue(itemValue);
}
// If current context is an empty object, convert to array
if (
typeof current.obj === 'object' &&
!Array.isArray(current.obj) &&
Object.keys(current.obj).length === 0
) {
const parent = stack.length > 1 ? stack[stack.length - 2] : null;
if (parent && typeof parent.obj === 'object' && !Array.isArray(parent.obj)) {
const parentObj = parent.obj as Record<string, unknown>;
for (const k of Object.keys(parentObj)) {
if (parentObj[k] === current.obj) {
parentObj[k] = [itemToPush];
current.obj = parentObj[k] as unknown[];
break;
}
}
}
} else if (Array.isArray(current.obj)) {
current.obj.push(itemToPush);
}
// If we pushed a mapping object, push it onto the stack so subsequent
// indented key-value lines populate the same object
if (dashKvMatch && typeof itemToPush === 'object') {
stack.push({
obj: itemToPush as Record<string, unknown>,
key: null,
indent, // use dash indent so sub-keys (more indented) populate this object
});
}
}
}
return frontmatter;
}
/**
* Coerce string values to appropriate JS types.
* Preserves leading-zero strings (e.g., "01") as strings.
*/
function coerceValue(value: string): unknown {
if (value === 'true') return true;
if (value === 'false') return false;
// Only coerce numbers without leading zeros (01, 007 stay as strings)
if (/^[1-9]\d*$/.test(value) || value === '0') return parseInt(value, 10);
if (/^\d+\.\d+$/.test(value) && !value.startsWith('0')) return parseFloat(value);
return value;
}
// ─── must_haves block parsing ────────────────────────────────────────────────
/**
* Parse the must_haves nested structure from raw frontmatter.
*
* The must_haves field has three sub-keys: truths (string[]),
* artifacts (object[]), and key_links (object[]).
* The stack-based parser above produces these as nested objects
* which need further normalization.
*/
function parseMustHaves(raw: unknown): MustHaves {
const defaults: MustHaves = { truths: [], artifacts: [], key_links: [] };
if (!raw || typeof raw !== 'object') return defaults;
const obj = raw as Record<string, unknown>;
return {
truths: normalizeStringArray(obj.truths),
artifacts: normalizeArtifacts(obj.artifacts),
key_links: normalizeKeyLinks(obj.key_links),
};
}
function normalizeStringArray(val: unknown): string[] {
if (Array.isArray(val)) return val.map(String);
return [];
}
function normalizeArtifacts(val: unknown): MustHaveArtifact[] {
if (!Array.isArray(val)) return [];
return val
.filter((item) => typeof item === 'object' && item !== null)
.map((item) => {
const obj = item as Record<string, unknown>;
return {
path: String(obj.path ?? ''),
provides: String(obj.provides ?? ''),
...(obj.min_lines !== undefined ? { min_lines: Number(obj.min_lines) } : {}),
...(obj.exports !== undefined ? { exports: normalizeStringArray(obj.exports) } : {}),
...(obj.contains !== undefined ? { contains: String(obj.contains) } : {}),
};
});
}
function normalizeKeyLinks(val: unknown): MustHaveKeyLink[] {
if (!Array.isArray(val)) return [];
return val
.filter((item) => typeof item === 'object' && item !== null)
.map((item) => {
const obj = item as Record<string, unknown>;
return {
from: String(obj.from ?? ''),
to: String(obj.to ?? ''),
via: String(obj.via ?? ''),
...(obj.pattern !== undefined ? { pattern: String(obj.pattern) } : {}),
};
});
}
// ─── XML task extraction ─────────────────────────────────────────────────────
/**
* Extract inner text of an XML element from a task body.
* Handles multiline content and trims whitespace.
*/
function extractElement(taskBody: string, tagName: string): string {
const regex = new RegExp(`<${tagName}>([\\s\\S]*?)</${tagName}>`, 'i');
const match = taskBody.match(regex);
return match ? match[1].trim() : '';
}
/**
* Extract the type attribute from a <task> opening tag.
*/
function extractTaskType(taskTag: string): string {
const match = taskTag.match(/type\s*=\s*["']([^"']+)["']/);
return match ? match[1] : 'auto';
}
/**
* Parse XML task blocks from the <tasks> section.
*
* Uses a regex to match <task ...>...</task> blocks, then extracts
* inner elements (name, files, read_first, action, verify,
* acceptance_criteria, done).
*
* Handles:
* - Multiline <action> blocks (including code snippets with angle brackets)
* - Optional elements (missing elements → empty string/array)
* - Both auto and checkpoint task types
*/
export function parseTasks(content: string): PlanTask[] {
const tasks: PlanTask[] = [];
// Extract the <tasks>...</tasks> section first
const tasksSection = content.match(/<tasks>([\s\S]*?)<\/tasks>/i);
const taskContent = tasksSection ? tasksSection[1] : content;
// Match individual task blocks — use a greedy-enough approach
// that handles nested angle brackets in action blocks
const taskRegex = /<task\b([^>]*)>([\s\S]*?)<\/task>/gi;
let taskMatch: RegExpExecArray | null;
while ((taskMatch = taskRegex.exec(taskContent)) !== null) {
const attrs = taskMatch[1];
const body = taskMatch[2];
const type = extractTaskType(attrs);
const name = extractElement(body, 'name');
const filesStr = extractElement(body, 'files');
const readFirstStr = extractElement(body, 'read_first');
const action = extractElement(body, 'action');
const verify = extractElement(body, 'verify');
const done = extractElement(body, 'done');
// Parse acceptance_criteria — can be a block with "- " list items
const acRaw = extractElement(body, 'acceptance_criteria');
const acceptance_criteria = acRaw
? acRaw
.split('\n')
.map((line) => line.trim())
.filter((line) => line.startsWith('- '))
.map((line) => line.slice(2).trim())
: [];
// Parse file lists (comma-separated)
const files = filesStr
? filesStr
.split(',')
.map((f) => f.trim())
.filter(Boolean)
: [];
const read_first = readFirstStr
? readFirstStr
.split(',')
.map((f) => f.trim())
.filter(Boolean)
: [];
tasks.push({
type,
name,
files,
read_first,
action,
verify,
acceptance_criteria,
done,
});
}
return tasks;
}
// ─── Section extraction ──────────────────────────────────────────────────────
/**
* Extract content of a named XML section (e.g., <objective>...</objective>).
*/
function extractSection(content: string, sectionName: string): string {
const regex = new RegExp(`<${sectionName}>([\\s\\S]*?)</${sectionName}>`, 'i');
const match = content.match(regex);
return match ? match[1].trim() : '';
}
/**
* Extract context references from the <context> block.
* Returns an array of file paths (lines starting with @).
*/
function extractContextRefs(content: string): string[] {
const contextBlock = extractSection(content, 'context');
if (!contextBlock) return [];
return contextBlock
.split('\n')
.map((line) => line.trim())
.filter((line) => line.startsWith('@'))
.map((line) => line.slice(1).trim());
}
/**
* Extract execution_context references.
* Returns an array of file paths (lines starting with @).
*/
function extractExecutionContext(content: string): string[] {
const block = extractSection(content, 'execution_context');
if (!block) return [];
return block
.split('\n')
.map((line) => line.trim())
.filter((line) => line.startsWith('@'))
.map((line) => line.slice(1).trim());
}
// ─── Public API ──────────────────────────────────────────────────────────────
/**
* Parse a GSD-1 PLAN.md content string into a structured ParsedPlan.
*
* Extracts:
* - YAML frontmatter (phase, wave, depends_on, must_haves, etc.)
* - <objective> section
* - <execution_context> references
* - <context> file references
* - <task> blocks with all inner elements
*
* Handles edge cases:
* - Empty input → empty frontmatter, no tasks
* - Missing frontmatter → empty object with defaults
* - Malformed XML → partial extraction, no crash
*/
export function parsePlan(content: string): ParsedPlan {
if (!content || typeof content !== 'string') {
return {
frontmatter: createDefaultFrontmatter(),
objective: '',
execution_context: [],
context_refs: [],
tasks: [],
raw: content ?? '',
};
}
const rawFrontmatter = extractFrontmatter(content);
// Build typed frontmatter with defaults
const frontmatter: PlanFrontmatter = {
phase: String(rawFrontmatter.phase ?? ''),
plan: String(rawFrontmatter.plan ?? ''),
type: String(rawFrontmatter.type ?? 'execute'),
wave: Number(rawFrontmatter.wave ?? 1),
depends_on: normalizeStringArray(rawFrontmatter.depends_on),
files_modified: normalizeStringArray(rawFrontmatter.files_modified),
autonomous: rawFrontmatter.autonomous !== false,
requirements: normalizeStringArray(rawFrontmatter.requirements),
must_haves: parseMustHaves(rawFrontmatter.must_haves),
};
// Preserve any extra frontmatter keys
for (const [key, value] of Object.entries(rawFrontmatter)) {
if (!(key in frontmatter)) {
frontmatter[key] = value;
}
}
return {
frontmatter,
objective: extractSection(content, 'objective'),
execution_context: extractExecutionContext(content),
context_refs: extractContextRefs(content),
tasks: parseTasks(content),
raw: content,
};
}
function createDefaultFrontmatter(): PlanFrontmatter {
return {
phase: '',
plan: '',
type: 'execute',
wave: 1,
depends_on: [],
files_modified: [],
autonomous: true,
requirements: [],
must_haves: { truths: [], artifacts: [], key_links: [] },
};
}
/**
* Convenience wrapper — reads a PLAN.md file from disk and parses it.
*/
export async function parsePlanFile(filePath: string): Promise<ParsedPlan> {
const content = await readFile(filePath, 'utf-8');
return parsePlan(content);
}

View File

@@ -1,70 +0,0 @@
import { mkdtemp, readFile } from 'node:fs/promises';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { describe, expect, it } from 'vitest';
import { PlanningJournal } from './planning-journal.js';
describe('PlanningJournal', () => {
it('appends events with monotonic source sequence numbers', async () => {
const dir = await mkdtemp(join(tmpdir(), 'gsd-journal-'));
const journal = new PlanningJournal({ projectDir: dir, sourceId: 'daemon-1', runId: 'run-1' });
const first = await journal.append({
projectId: 'project-1',
type: 'plan.next',
actor: { type: 'agent', id: 'agent-1' },
payload: { itemId: 'item-1' },
idempotencyKey: 'next-1',
});
const second = await journal.append({
projectId: 'project-1',
type: 'plan.done',
actor: { type: 'agent', id: 'agent-1' },
payload: { itemId: 'item-1' },
idempotencyKey: 'done-1',
});
expect(first.source.seq).toBe(1);
expect(second.source.seq).toBe(2);
expect(await journal.readAll()).toHaveLength(2);
});
it('replays an existing event for duplicate idempotency keys', async () => {
const dir = await mkdtemp(join(tmpdir(), 'gsd-journal-'));
const journal = new PlanningJournal({ projectDir: dir, sourceId: 'sdk-1', runId: 'run-1' });
const first = await journal.append({
projectId: 'project-1',
type: 'plan.checkpoint',
actor: { type: 'agent', id: 'agent-1' },
payload: { summary: 'Progress' },
idempotencyKey: 'checkpoint-1',
});
const replay = await journal.append({
projectId: 'project-1',
type: 'plan.checkpoint',
actor: { type: 'agent', id: 'agent-1' },
payload: { summary: 'Progress' },
idempotencyKey: 'checkpoint-1',
});
expect(replay.id).toBe(first.id);
expect(await journal.readAll()).toHaveLength(1);
});
it('writes jsonl under .gsd/journal.jsonl', async () => {
const dir = await mkdtemp(join(tmpdir(), 'gsd-journal-'));
const journal = new PlanningJournal({ projectDir: dir, sourceId: 'sdk-1', runId: 'run-1' });
await journal.append({
projectId: 'project-1',
type: 'plan.status',
actor: { type: 'agent', id: 'agent-1' },
payload: {},
idempotencyKey: 'status-1',
});
const raw = await readFile(join(dir, '.gsd', 'journal.jsonl'), 'utf8');
expect(raw.trim().split('\n')).toHaveLength(1);
expect(JSON.parse(raw).schemaVersion).toBe(1);
});
});

View File

@@ -1,153 +0,0 @@
import { appendFile, mkdir, readFile, rename, writeFile } from 'node:fs/promises';
import { createHash, randomUUID } from 'node:crypto';
import { join } from 'node:path';
export type PlanningEventActor = {
type: 'human' | 'agent' | 'runtime' | 'verifier' | 'system';
id: string;
role?: string;
sessionId?: string;
taskId?: string;
};
export type PlanningEvent = {
id: string;
schemaVersion: 1;
projectionVersion: number;
projectId: string;
source: { id: string; kind: 'sdk' | 'daemon' | 'cloud' | 'import'; seq: number; cursor?: string };
runId: string;
workstreamId?: string;
planId?: string;
itemId?: string;
actor: PlanningEventActor;
authority: 'local' | 'cloud' | 'human_approved' | 'system';
type: string;
idempotencyKey: string;
causationId?: string;
occurredAt: string;
payload: Record<string, unknown>;
evidenceIds: string[];
parentEventIds: string[];
trace: Record<string, unknown>;
requestHash: string;
};
export type PlanningJournalAppendInput = {
projectId: string;
type: string;
actor: PlanningEventActor;
payload: Record<string, unknown>;
idempotencyKey: string;
planId?: string;
itemId?: string;
workstreamId?: string;
evidenceIds?: string[];
parentEventIds?: string[];
causationId?: string;
trace?: Record<string, unknown>;
};
export class PlanningJournal {
private readonly path: string;
constructor(
private readonly options: {
projectDir: string;
sourceId: string;
runId: string;
sourceKind?: 'sdk' | 'daemon' | 'cloud' | 'import';
projectionVersion?: number;
},
) {
this.path = join(options.projectDir, '.gsd', 'journal.jsonl');
}
async append(input: PlanningJournalAppendInput): Promise<PlanningEvent> {
const existing = await this.findByIdempotency(input.idempotencyKey);
const requestHash = hashRequest(input);
if (existing) {
if (existing.requestHash !== requestHash) {
throw new Error(`conflicting idempotency key: ${input.idempotencyKey}`);
}
return existing;
}
const events = await this.readAll();
const event: PlanningEvent = {
id: randomUUID(),
schemaVersion: 1,
projectionVersion: this.options.projectionVersion ?? 1,
projectId: input.projectId,
source: {
id: this.options.sourceId,
kind: this.options.sourceKind ?? 'sdk',
seq: events.filter((candidate) => candidate.source.id === this.options.sourceId).length + 1,
},
runId: this.options.runId,
workstreamId: input.workstreamId,
planId: input.planId,
itemId: input.itemId,
actor: input.actor,
authority: 'local',
type: input.type,
idempotencyKey: input.idempotencyKey,
causationId: input.causationId,
occurredAt: new Date().toISOString(),
payload: input.payload,
evidenceIds: input.evidenceIds ?? [],
parentEventIds: input.parentEventIds ?? [],
trace: input.trace ?? {},
requestHash,
};
await mkdir(join(this.options.projectDir, '.gsd'), { recursive: true });
await appendFile(this.path, `${JSON.stringify(event)}\n`, 'utf8');
return event;
}
async readAll(): Promise<PlanningEvent[]> {
let raw = '';
try {
raw = await readFile(this.path, 'utf8');
} catch {
return [];
}
return raw
.split(/\r?\n/)
.map((line) => line.trim())
.filter(Boolean)
.map((line) => JSON.parse(line) as PlanningEvent);
}
async compact(events: PlanningEvent[]): Promise<void> {
await mkdir(join(this.options.projectDir, '.gsd'), { recursive: true });
const tmp = `${this.path}.tmp`;
await writeFile(
tmp,
events.map((event) => JSON.stringify(event)).join('\n') + (events.length ? '\n' : ''),
'utf8',
);
await rename(tmp, this.path);
}
private async findByIdempotency(idempotencyKey: string): Promise<PlanningEvent | null> {
const events = await this.readAll();
return events.find((event) => event.idempotencyKey === idempotencyKey) ?? null;
}
}
function hashRequest(input: PlanningJournalAppendInput): string {
return createHash('sha256')
.update(
JSON.stringify({
projectId: input.projectId,
type: input.type,
payload: input.payload,
planId: input.planId,
itemId: input.itemId,
actor: input.actor,
}),
)
.digest('hex');
}

View File

@@ -1,29 +0,0 @@
import { mkdtemp } from 'node:fs/promises';
import { tmpdir } from 'node:os';
import { join } from 'node:path';
import { describe, expect, it } from 'vitest';
import { PlanningRuntime } from './planning-runtime.js';
describe('PlanningRuntime', () => {
it('records intent events through the durable journal', async () => {
const dir = await mkdtemp(join(tmpdir(), 'gsd-runtime-'));
const runtime = new PlanningRuntime({
projectDir: dir,
projectId: 'project-1',
runId: 'run-1',
sourceId: 'sdk-1',
actor: { type: 'agent', id: 'agent-1', role: 'executor' },
});
await runtime.status({ idempotencyKey: 'status-1' });
await runtime.next({ idempotencyKey: 'next-1', createPlan: { title: 'Plan', items: [{ title: 'Item' }] } });
await runtime.checkpoint({ idempotencyKey: 'checkpoint-1', summary: 'Progress' });
const events = await runtime.journal.readAll();
expect(events.map((event) => event.type)).toEqual([
'plan.status',
'plan.next',
'plan.checkpoint',
]);
});
});

View File

@@ -1,100 +0,0 @@
import { PlanningJournal, type PlanningEventActor } from './planning-journal.js';
type RuntimeOptions = {
projectDir: string;
projectId: string;
runId: string;
sourceId: string;
actor: PlanningEventActor;
};
type RuntimeMeta = {
idempotencyKey: string;
planId?: string;
itemId?: string;
};
type NextInput = RuntimeMeta & {
selector?: { itemId?: string; titleIncludes?: string };
createPlan?: { title: string; items: Array<{ title: string; description?: string; dependsOn?: string[] }> };
};
type CheckpointInput = RuntimeMeta & {
summary?: string;
subTasks?: Array<{ id?: string; text: string }>;
agentCriteria?: Array<{ id?: string; text: string }>;
criteriaMet?: string[];
blocked?: { reason: string; nextAction?: string };
};
type DoneInput = RuntimeMeta & {
summary: string;
blockers?: string[];
criteriaMet?: string[];
evidenceRefs?: string[];
evidencePolicy?: 'auto' | 'explicit' | 'waive';
evidenceWaiverReason?: string;
advance?: boolean;
};
export class PlanningRuntime {
readonly journal: PlanningJournal;
constructor(private readonly options: RuntimeOptions) {
this.journal = new PlanningJournal({
projectDir: options.projectDir,
sourceId: options.sourceId,
runId: options.runId,
sourceKind: 'sdk',
});
}
status(input: RuntimeMeta) {
return this.record('plan.status', input, {});
}
next(input: NextInput) {
return this.record('plan.next', input, {
selector: input.selector,
createPlan: input.createPlan,
});
}
checkpoint(input: CheckpointInput) {
return this.record('plan.checkpoint', input, {
summary: input.summary,
subTasks: input.subTasks,
agentCriteria: input.agentCriteria,
criteriaMet: input.criteriaMet,
blocked: input.blocked,
});
}
sync(input: RuntimeMeta & { cursor?: string }) {
return this.record('plan.sync', input, { cursor: input.cursor });
}
done(input: DoneInput) {
return this.record('plan.done', input, {
summary: input.summary,
blockers: input.blockers,
criteriaMet: input.criteriaMet,
evidenceRefs: input.evidenceRefs,
evidencePolicy: input.evidencePolicy ?? 'auto',
evidenceWaiverReason: input.evidenceWaiverReason,
advance: input.advance ?? true,
});
}
private record(type: string, input: RuntimeMeta, payload: Record<string, unknown>) {
return this.journal.append({
projectId: this.options.projectId,
type,
actor: this.options.actor,
planId: input.planId,
itemId: input.itemId,
idempotencyKey: input.idempotencyKey,
payload,
});
}
}

View File

@@ -1,318 +0,0 @@
/**
* Unit tests for prompt-builder.ts
*/
import { describe, it, expect } from 'vitest';
import {
buildExecutorPrompt,
parseAgentTools,
parseAgentRole,
DEFAULT_ALLOWED_TOOLS,
} from './prompt-builder.js';
import type { ParsedPlan, PlanFrontmatter, MustHaves } from './types.js';
// ─── Helpers ─────────────────────────────────────────────────────────────────
function makePlan(overrides: Partial<ParsedPlan> = {}): ParsedPlan {
const defaultFrontmatter: PlanFrontmatter = {
phase: '01-auth',
plan: '01',
type: 'execute',
wave: 1,
depends_on: [],
files_modified: [],
autonomous: true,
requirements: ['AUTH-01'],
must_haves: { truths: [], artifacts: [], key_links: [] },
};
return {
frontmatter: { ...defaultFrontmatter, ...overrides.frontmatter },
objective: overrides.objective ?? 'Implement JWT authentication with refresh tokens',
execution_context: overrides.execution_context ?? [],
context_refs: overrides.context_refs ?? [],
tasks: overrides.tasks ?? [
{
type: 'auto',
name: 'Create auth module',
files: ['src/auth.ts'],
read_first: ['src/types.ts'],
action: 'Create the auth module with login and refresh endpoints',
verify: 'npm test -- --filter auth',
acceptance_criteria: ['JWT tokens issued on login', 'Refresh tokens rotate correctly'],
done: 'Auth module created and tests pass',
},
{
type: 'auto',
name: 'Add middleware',
files: ['src/middleware.ts'],
read_first: [],
action: 'Create auth middleware for protected routes',
verify: 'npm test -- --filter middleware',
acceptance_criteria: [],
done: 'Middleware validates JWT on protected routes',
},
],
raw: '',
};
}
const SAMPLE_AGENT_DEF = `---
name: gsd-executor
description: Executes GSD plans
tools: Read, Write, Edit, Bash, Grep, Glob
permissionMode: acceptEdits
---
<role>
You are a GSD plan executor. You execute PLAN.md files atomically.
</role>
<execution_flow>
Some flow content
</execution_flow>`;
// ─── parseAgentTools ─────────────────────────────────────────────────────────
describe('parseAgentTools', () => {
it('extracts tools from agent definition frontmatter', () => {
const tools = parseAgentTools(SAMPLE_AGENT_DEF);
expect(tools).toEqual(['Read', 'Write', 'Edit', 'Bash', 'Grep', 'Glob']);
});
it('returns defaults when no frontmatter found', () => {
const tools = parseAgentTools('Just some text without frontmatter');
expect(tools).toEqual(DEFAULT_ALLOWED_TOOLS);
});
it('returns defaults when frontmatter has no tools key', () => {
const def = `---\nname: test\n---\nContent`;
const tools = parseAgentTools(def);
expect(tools).toEqual(DEFAULT_ALLOWED_TOOLS);
});
it('handles empty tools value', () => {
const def = `---\ntools: \n---`;
const tools = parseAgentTools(def);
expect(tools).toEqual(DEFAULT_ALLOWED_TOOLS);
});
});
// ─── parseAgentRole ──────────────────────────────────────────────────────────
describe('parseAgentRole', () => {
it('extracts role content from agent definition', () => {
const role = parseAgentRole(SAMPLE_AGENT_DEF);
expect(role).toContain('GSD plan executor');
expect(role).toContain('PLAN.md files atomically');
});
it('returns empty string when no role block', () => {
expect(parseAgentRole('No role block here')).toBe('');
});
});
// ─── buildExecutorPrompt ─────────────────────────────────────────────────────
describe('buildExecutorPrompt', () => {
it('includes the objective text', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Implement JWT authentication with refresh tokens');
});
it('includes all task names', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Create auth module');
expect(prompt).toContain('Add middleware');
});
it('includes task actions', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Create the auth module with login and refresh endpoints');
expect(prompt).toContain('Create auth middleware for protected routes');
});
it('includes task verification commands', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('npm test -- --filter auth');
expect(prompt).toContain('npm test -- --filter middleware');
});
it('includes task file references', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('src/auth.ts');
expect(prompt).toContain('src/types.ts');
});
it('includes acceptance criteria', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('JWT tokens issued on login');
expect(prompt).toContain('Refresh tokens rotate correctly');
});
it('includes SUMMARY.md creation instruction with derived filename', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('01-01-SUMMARY.md');
});
it('includes phaseDir in SUMMARY path when provided', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan, { phaseDir: '.planning/phases/01-auth' });
expect(prompt).toContain('.planning/phases/01-auth/01-01-SUMMARY.md');
});
it('uses bare SUMMARY.md when phase/plan numbers missing', () => {
const plan = makePlan({ frontmatter: { phase: '', plan: '', type: 'execute', wave: 1, depends_on: [], files_modified: [], autonomous: true, requirements: [], must_haves: { truths: [], artifacts: [], key_links: [] } } });
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('SUMMARY.md');
expect(prompt).not.toContain('/SUMMARY.md');
});
it('includes sequential execution instruction', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Execute these tasks sequentially');
});
it('handles plan with no tasks gracefully', () => {
const plan = makePlan({ tasks: [] });
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('No tasks defined');
expect(prompt).toContain('SUMMARY.md');
// Should not throw
expect(prompt.length).toBeGreaterThan(0);
});
it('includes context references when present', () => {
const plan = makePlan({
context_refs: ['src/config.ts', 'docs/architecture.md'],
});
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('@src/config.ts');
expect(prompt).toContain('@docs/architecture.md');
expect(prompt).toContain('Read these files for context');
});
it('omits context section when no refs', () => {
const plan = makePlan({ context_refs: [] });
const prompt = buildExecutorPrompt(plan);
expect(prompt).not.toContain('Context Files');
});
it('includes plan metadata', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Phase: 01-auth');
expect(prompt).toContain('Plan: 01');
});
it('includes must-have truths when present', () => {
const plan = makePlan({
frontmatter: {
phase: '01',
plan: '01',
type: 'execute',
wave: 1,
depends_on: [],
files_modified: [],
autonomous: true,
requirements: [],
must_haves: {
truths: ['All endpoints require JWT auth', 'Tokens expire after 15 minutes'],
artifacts: [],
key_links: [],
},
},
});
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('All endpoints require JWT auth');
expect(prompt).toContain('Tokens expire after 15 minutes');
});
it('includes must-have artifacts', () => {
const plan = makePlan({
frontmatter: {
phase: '01',
plan: '01',
type: 'execute',
wave: 1,
depends_on: [],
files_modified: [],
autonomous: true,
requirements: [],
must_haves: {
truths: [],
artifacts: [{ path: 'src/auth.ts', provides: 'JWT auth module' }],
key_links: [],
},
},
});
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('`src/auth.ts`');
expect(prompt).toContain('JWT auth module');
});
it('includes must-have key_links', () => {
const plan = makePlan({
frontmatter: {
phase: '01',
plan: '01',
type: 'execute',
wave: 1,
depends_on: [],
files_modified: [],
autonomous: true,
requirements: [],
must_haves: {
truths: [],
artifacts: [],
key_links: [{ from: 'auth.ts', to: 'middleware.ts', via: 'import' }],
},
},
});
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('auth.ts → middleware.ts via import');
});
it('includes role from agent definition when provided', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan, SAMPLE_AGENT_DEF);
expect(prompt).toContain('## Role');
expect(prompt).toContain('GSD plan executor');
});
it('works without agent definition', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
// Should still produce a valid prompt without role section
expect(prompt).toContain('## Objective');
expect(prompt).toContain('## Tasks');
expect(prompt).not.toContain('## Role');
});
it('provides fallback objective when plan has empty objective', () => {
const plan = makePlan({ objective: '' });
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Execute plan: 01');
});
it('includes done criteria for tasks', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Auth module created and tests pass');
expect(prompt).toContain('Middleware validates JWT on protected routes');
});
it('includes commit instruction in completion section', () => {
const plan = makePlan();
const prompt = buildExecutorPrompt(plan);
expect(prompt).toContain('Commit the SUMMARY.md');
});
});

View File

@@ -1,218 +0,0 @@
/**
* Prompt builder — assembles executor prompts from parsed plans.
*
* Converts a ParsedPlan into a structured prompt that tells the
* executor agent exactly what to do: follow the tasks sequentially,
* verify each one, and produce a SUMMARY.md at the end.
*/
import type { ParsedPlan, PlanTask } from './types.js';
// ─── Constants ───────────────────────────────────────────────────────────────
const DEFAULT_ALLOWED_TOOLS = ['Read', 'Write', 'Edit', 'Bash', 'Grep', 'Glob'];
// ─── Agent definition parsing ────────────────────────────────────────────────
/**
* Extract the tools list from a gsd-executor.md agent definition.
* Falls back to DEFAULT_ALLOWED_TOOLS if parsing fails.
*/
export function parseAgentTools(agentDef: string): string[] {
// Look for "tools:" in the YAML frontmatter
const frontmatterMatch = agentDef.match(/^---\s*\n([\s\S]*?)\n---/);
if (!frontmatterMatch) return DEFAULT_ALLOWED_TOOLS;
const toolsMatch = frontmatterMatch[1].match(/^tools:\s*(.+)$/m);
if (!toolsMatch) return DEFAULT_ALLOWED_TOOLS;
const tools = toolsMatch[1]
.split(',')
.map((t) => t.trim())
.filter(Boolean);
return tools.length > 0 ? tools : DEFAULT_ALLOWED_TOOLS;
}
/**
* Extract the role instructions from a gsd-executor.md agent definition.
* Returns the <role>...</role> block content, or empty string.
*/
export function parseAgentRole(agentDef: string): string {
const match = agentDef.match(/<role>([\s\S]*?)<\/role>/i);
return match ? match[1].trim() : '';
}
// ─── Prompt assembly ─────────────────────────────────────────────────────────
/**
* Format a single task into a prompt block.
*/
function formatTask(task: PlanTask, index: number): string {
const lines: string[] = [];
lines.push(`### Task ${index + 1}: ${task.name}`);
if (task.files.length > 0) {
lines.push(`**Files:** ${task.files.join(', ')}`);
}
if (task.read_first.length > 0) {
lines.push(`**Read first:** ${task.read_first.join(', ')}`);
}
lines.push('');
lines.push('**Action:**');
lines.push(task.action);
if (task.verify) {
lines.push('');
lines.push('**Verify:**');
lines.push(task.verify);
}
if (task.done) {
lines.push('');
lines.push('**Done when:**');
lines.push(task.done);
}
if (task.acceptance_criteria.length > 0) {
lines.push('');
lines.push('**Acceptance criteria:**');
for (const criterion of task.acceptance_criteria) {
lines.push(`- ${criterion}`);
}
}
return lines.join('\n');
}
/**
* Options for buildExecutorPrompt beyond the required plan.
*/
export interface ExecutorPromptOptions {
/** Raw content of gsd-executor.md agent definition. */
agentDef?: string;
/** Phase directory relative to project root (e.g. `.planning/phases/01-auth`). */
phaseDir?: string;
}
/**
* Build the executor prompt from a parsed plan and optional agent definition.
*
* The prompt instructs the executor to:
* 1. Follow the plan tasks sequentially
* 2. Run verification for each task
* 3. Commit each task individually
* 4. Produce a SUMMARY.md file on completion
*
* @param plan - Parsed plan structure from plan-parser
* @param agentDefOrOpts - Raw agent definition string (legacy) or ExecutorPromptOptions
* @returns Assembled prompt string
*/
export function buildExecutorPrompt(plan: ParsedPlan, agentDefOrOpts?: string | ExecutorPromptOptions): string {
const opts: ExecutorPromptOptions = typeof agentDefOrOpts === 'string'
? { agentDef: agentDefOrOpts }
: agentDefOrOpts ?? {};
const { agentDef, phaseDir } = opts;
const sections: string[] = [];
// ── Role instructions from agent definition ──
if (agentDef) {
const role = parseAgentRole(agentDef);
if (role) {
sections.push(`## Role\n\n${role}`);
}
}
// ── Objective ──
if (plan.objective) {
sections.push(`## Objective\n\n${plan.objective}`);
} else {
sections.push(`## Objective\n\nExecute plan: ${plan.frontmatter.plan || plan.frontmatter.phase || 'unnamed'}`);
}
// ── Plan metadata ──
const meta: string[] = [];
if (plan.frontmatter.phase) meta.push(`Phase: ${plan.frontmatter.phase}`);
if (plan.frontmatter.plan) meta.push(`Plan: ${plan.frontmatter.plan}`);
if (plan.frontmatter.type) meta.push(`Type: ${plan.frontmatter.type}`);
if (meta.length > 0) {
sections.push(`## Plan Info\n\n${meta.join('\n')}`);
}
// ── Context references ──
if (plan.context_refs.length > 0) {
const refs = plan.context_refs.map((r) => `- @${r}`).join('\n');
sections.push(`## Context Files\n\nRead these files for context before starting:\n${refs}`);
}
// ── Tasks ──
if (plan.tasks.length > 0) {
const taskBlocks = plan.tasks.map((t, i) => formatTask(t, i)).join('\n\n---\n\n');
sections.push(`## Tasks\n\nExecute these tasks sequentially. For each task: read any referenced files, execute the action, run verification, confirm done criteria, then commit.\n\n${taskBlocks}`);
} else {
sections.push(`## Tasks\n\nNo tasks defined in this plan. Review the objective and determine if any actions are needed.`);
}
// ── Must-haves ──
if (plan.frontmatter.must_haves) {
const mh = plan.frontmatter.must_haves;
const parts: string[] = [];
if (mh.truths.length > 0) {
parts.push('**Truths (invariants):**');
for (const t of mh.truths) {
parts.push(`- ${t}`);
}
}
if (mh.artifacts.length > 0) {
parts.push('**Required artifacts:**');
for (const a of mh.artifacts) {
parts.push(`- \`${a.path}\`: ${a.provides}`);
}
}
if (mh.key_links.length > 0) {
parts.push('**Key links:**');
for (const l of mh.key_links) {
parts.push(`- ${l.from} → ${l.to} via ${l.via}`);
}
}
if (parts.length > 0) {
sections.push(`## Must-Haves\n\n${parts.join('\n')}`);
}
}
// ── Completion instructions ──
// Derive the SUMMARY filename from plan frontmatter (e.g. "01-01-SUMMARY.md")
// Phase may be "01-auth" or "01" — extract leading number, zero-pad to 2 digits.
const phaseNum = (plan.frontmatter.phase || '').match(/^(\d+)/)?.[1] || '';
const planNum = (plan.frontmatter.plan || '').match(/^(\d+)/)?.[1] || '';
const summaryName = phaseNum && planNum
? `${phaseNum.padStart(2, '0')}-${planNum.padStart(2, '0')}-SUMMARY.md`
: 'SUMMARY.md';
const summaryPath = phaseDir
? `${phaseDir}/${summaryName}`
: summaryName;
sections.push(
`## Completion\n\n` +
`After all tasks are complete:\n` +
`1. Run any overall verification or success criteria checks\n` +
`2. Create \`${summaryPath}\` documenting:\n` +
` - One-line summary of what was accomplished\n` +
` - Tasks completed with commit hashes\n` +
` - Any deviations from the plan\n` +
` - Files created or modified\n` +
` - Known issues (if any)\n` +
`3. Commit the SUMMARY.md\n` +
`4. Report completion`,
);
return sections.join('\n\n');
}
export { DEFAULT_ALLOWED_TOOLS };

Some files were not shown because too many files have changed in this diff Show More