From 5fdc950eb7f65dca79569ea8653eaf907c51b26d Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Thu, 30 Apr 2026 01:04:41 -0400 Subject: [PATCH] feat(#2792): namespace meta-skills + keyword-tag descriptions + context utilization guard (#2825) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(#2792): namespace meta-skills retargeted at the post-#2790 surface This branch is now based on #2790's HEAD (the consolidation PR) instead of main, and every routing table targets the consolidated surface so a user routed by a namespace meta-skill never lands at a deleted / folded sub-skill. Cross-PR inconsistencies the original PR #2825 carried (vs #2790): - ns-ideate routed to gsd-note / gsd-add-todo / gsd-add-backlog / gsd-plant-seed → all folded into gsd-capture by #2790. Now routes to gsd-capture (the parent picks the mode from the user's intent). - ns-context routed to gsd-scan and gsd-intel → folded into gsd-map-codebase --fast / --query by #2790. Now routes to those flag forms. - ns-manage routed all workspace intent to gsd-list-workspaces (a list-only entry) → CR also flagged the over-narrow target. #2790 folds into gsd-workspace; routing now points there. - ns-workflow routed to gsd-research-phase → deleted outright by #2790. Removed. - ns-project routed to gsd-plan-milestone-gaps → deleted outright by #2790. Removed. - None of the namespaces previously surfaced #2790's new consolidated skills (gsd-capture, gsd-phase, gsd-config, gsd-workspace, gsd-progress). All five are now reachable through the routers. - extract_learnings → extract-learnings (canonicalized by #2858). Defect fixes within the namespace skills: - Hyphen-form `name:` (gsd-workflow, …) per the canonical naming contract — the colon-form addressed CR's drift complaint. - `Skill` added to allowed-tools on every router. The body instructs "Invoke the matched skill directly using the Skill tool" — without Skill in the permission list the meta-skill cannot route at all. New regression guard in tests/enh-2792-namespace-skills.test.cjs: every gsd-* token in any namespace router's table column resolves to a surviving commands/gsd/*.md file (or to a known consolidated parent for flag-form targets like gsd-map-codebase --fast). This single test would have caught every dead-end route the original PR shipped with. Skill-count cap in tests/enh-2790-skill-consolidation.test.cjs now filters out ns-*.md from its <= 63 cap. Namespace routers are descriptor-only entries, not part of the consolidation surface that cap is policing — they have their own contract in tests/enh-2792-namespace-skills.test.cjs. INVENTORY.md gains a "Namespace Meta-Skills" section with the 6 router rows; INVENTORY-MANIFEST.json gains 6 entries; the headline count moves 59 → 65 to match. Out of scope for this rebase: the gsd-health --context flag (PR #2825 advertised the contract but didn't implement it). That's a separate feature concern and is left untouched here. 5908/5908 on `npm test`. * feat(#2792): implement gsd-health --context utilization guard The original PR #2825 advertised a `--context` flag on gsd-health with a 60%/70% utilization threshold table but never implemented the workflow logic — CR caught it as a contract leak, the rebase deferred it. This commit closes the gap with TDD red/green/refactor. Math layer (pure): - get-shit-done/bin/lib/context-utilization.cjs classifyContextUtilization(tokensUsed, contextWindow) → { percent, state } State boundaries use the exact ratio: < 60% healthy / 60–70% warning / ≥ 70% critical (fracture point) Display percent rounded for humans. Throws TypeError on non-integer or out-of-range inputs. - STATES = Object.freeze({ HEALTHY, WARNING, CRITICAL }) exported so callers reference the names by symbol, not by literal string. SDK CLI integration: - get-shit-done/bin/gsd-tools.cjs `validate context --tokens-used N --context-window M [--json]` routes to the classifier, owns the recommendation copy (the classifier intentionally does not — keeps the renderer free to evolve without touching the math layer or its tests), and uses core.output's rawValue path for the sync-flush guarantee. - sdk/src/query/validate.ts + sdk/src/query/index.ts TypeScript validateContext handler registered at 'validate.context' and 'validate context'. Mirrors the CJS classifier inline (15 lines of arithmetic; not worth a shared cross-language module). User-facing wiring: - commands/gsd/health.md frontmatter advertises --context, body documents the three-state threshold table. - get-shit-done/workflows/health.md adds a `context_check` step that's reached only when --context is set. Step calls `gsd-sdk query validate.context` with self-reported tokensUsed and contextWindow, prints the SDK output verbatim, and ends. Includes a TEXT_MODE plain-text fallback for non-Claude runtimes per #2012. Tests: - tests/context-utilization.test.cjs (17 tests) — pure-function contract: state thresholds at every boundary, percent rounding, input validation, return-shape (no recommendation field — that's the renderer's job). - tests/validate-context.test.cjs (9 tests) — SDK CLI plumbing: arg parsing errors, JSON vs human rendering, recommendation copy pinned per state. - tests/enh-2792-namespace-skills.test.cjs (4 new tests) — markdown contract: --context advertised in argument-hint, threshold table in command body, context_check step exists in workflow, step invokes gsd-sdk query validate.context with both flags. Inventory bookkeeping: - docs/INVENTORY.md "CLI Modules" 31 → 32; new row for context-utilization.cjs. - docs/INVENTORY-MANIFEST.json mirror. 5939/5939 on `npm test`. --- commands/gsd/health.md | 12 +- commands/gsd/ns-context.md | 22 ++ commands/gsd/ns-ideate.md | 23 ++ commands/gsd/ns-manage.md | 28 ++ commands/gsd/ns-project.md | 22 ++ commands/gsd/ns-review.md | 25 ++ commands/gsd/ns-workflow.md | 27 ++ docs/INVENTORY-MANIFEST.json | 9 +- docs/INVENTORY.md | 18 +- get-shit-done/bin/gsd-tools.cjs | 42 ++- get-shit-done/bin/lib/context-utilization.cjs | 47 ++++ get-shit-done/workflows/health.md | 37 ++- sdk/src/query/index.ts | 5 +- sdk/src/query/validate.ts | 56 ++++ tests/context-utilization.test.cjs | 122 +++++++++ tests/enh-2790-skill-consolidation.test.cjs | 11 +- tests/enh-2792-namespace-skills.test.cjs | 255 ++++++++++++++++++ tests/validate-context.test.cjs | 90 +++++++ 18 files changed, 840 insertions(+), 11 deletions(-) create mode 100644 commands/gsd/ns-context.md create mode 100644 commands/gsd/ns-ideate.md create mode 100644 commands/gsd/ns-manage.md create mode 100644 commands/gsd/ns-project.md create mode 100644 commands/gsd/ns-review.md create mode 100644 commands/gsd/ns-workflow.md create mode 100644 get-shit-done/bin/lib/context-utilization.cjs create mode 100644 tests/context-utilization.test.cjs create mode 100644 tests/enh-2792-namespace-skills.test.cjs create mode 100644 tests/validate-context.test.cjs diff --git a/commands/gsd/health.md b/commands/gsd/health.md index 260c83334..b5c449067 100644 --- a/commands/gsd/health.md +++ b/commands/gsd/health.md @@ -1,7 +1,7 @@ --- name: gsd:health description: Diagnose planning directory health and optionally repair issues -argument-hint: [--repair] +argument-hint: "[--repair] [--context]" allowed-tools: - Read - Bash @@ -10,6 +10,14 @@ allowed-tools: --- Validate `.planning/` directory integrity and report actionable issues. Checks for missing files, invalid configurations, inconsistent state, and orphaned plans. + +`--context` runs an orthogonal check: the running session's context utilization. The workflow asks for the model's tokensUsed + contextWindow, calls `gsd-sdk query validate.context`, and renders one of three states: + +| Utilization | State | Action | +|-------------|----------|-------------------------------------------------------| +| < 60% | healthy | no action — context is comfortable | +| 60% – 70% | warning | recommend `/gsd-thread` to start fresh | +| ≥ 70% | critical | reasoning quality may degrade past the fracture point | @@ -18,5 +26,5 @@ Validate `.planning/` directory integrity and report actionable issues. Checks f Execute the health workflow from @~/.claude/get-shit-done/workflows/health.md end-to-end. -Parse --repair flag from arguments and pass to workflow. +Parse `--repair` and `--context` flags from arguments and pass to workflow. diff --git a/commands/gsd/ns-context.md b/commands/gsd/ns-context.md new file mode 100644 index 000000000..3122949d3 --- /dev/null +++ b/commands/gsd/ns-context.md @@ -0,0 +1,22 @@ +--- +name: gsd-context +description: "codebase intelligence | map graphify docs learnings" +argument-hint: "" +allowed-tools: + - Read + - Skill +--- + +Route to the appropriate codebase-intelligence skill based on the user's intent. +`gsd-scan` and `gsd-intel` were folded into `gsd-map-codebase` flags by #2790. + +| User wants | Invoke | +|---|---| +| Map the full codebase structure | gsd-map-codebase | +| Quick lightweight codebase scan | gsd-map-codebase --fast | +| Query mapped intelligence files | gsd-map-codebase --query | +| Generate a knowledge graph | gsd-graphify | +| Update project documentation | gsd-docs-update | +| Extract learnings from a completed phase | gsd-extract-learnings | + +Invoke the matched skill directly using the Skill tool. diff --git a/commands/gsd/ns-ideate.md b/commands/gsd/ns-ideate.md new file mode 100644 index 000000000..988a5c266 --- /dev/null +++ b/commands/gsd/ns-ideate.md @@ -0,0 +1,23 @@ +--- +name: gsd-ideate +description: "exploration capture | explore sketch spike spec capture" +argument-hint: "" +allowed-tools: + - Read + - Skill +--- + +Route to the appropriate exploration / capture skill based on the user's intent. +`gsd-note`, `gsd-add-todo`, `gsd-add-backlog`, and `gsd-plant-seed` were folded +into `gsd-capture` (with `--note`, default, `--backlog`, `--seed` modes) by +#2790. The capture target lists pending todos via `--list`. + +| User wants | Invoke | +|---|---| +| Explore an idea or opportunity | gsd-explore | +| Sketch out a rough design or plan | gsd-sketch | +| Time-boxed technical spike | gsd-spike | +| Write a spec for a phase | gsd-spec-phase | +| Capture a thought (todo / note / backlog / seed) | gsd-capture | + +Invoke the matched skill directly using the Skill tool. diff --git a/commands/gsd/ns-manage.md b/commands/gsd/ns-manage.md new file mode 100644 index 000000000..ffceb32c7 --- /dev/null +++ b/commands/gsd/ns-manage.md @@ -0,0 +1,28 @@ +--- +name: gsd-manage +description: "config workspace | workstreams thread update ship inbox" +argument-hint: "" +allowed-tools: + - Read + - Skill +--- + +Route to the appropriate management skill based on the user's intent. +`gsd-config` (settings + advanced + integrations + profile) and `gsd-workspace` +(new + list + remove) are post-#2790 consolidated entries. + +| User wants | Invoke | +|---|---| +| Configure GSD settings (basic / advanced / integrations / profile) | gsd-config | +| Manage workspaces (create / list / remove) | gsd-workspace | +| Manage parallel workstreams | gsd-workstreams | +| Continue work in a fresh context thread | gsd-thread | +| Pause current work | gsd-pause-work | +| Resume paused work | gsd-resume-work | +| Update the GSD installation | gsd-update | +| Ship completed work | gsd-ship | +| Process inbox items | gsd-inbox | +| Create a clean PR branch | gsd-pr-branch | +| Undo the last GSD action | gsd-undo | + +Invoke the matched skill directly using the Skill tool. diff --git a/commands/gsd/ns-project.md b/commands/gsd/ns-project.md new file mode 100644 index 000000000..68addd737 --- /dev/null +++ b/commands/gsd/ns-project.md @@ -0,0 +1,22 @@ +--- +name: gsd-project +description: "project lifecycle | milestones audits summary" +argument-hint: "" +allowed-tools: + - Read + - Skill +--- + +Route to the appropriate project / milestone skill based on the user's intent. +`gsd-plan-milestone-gaps` was deleted by #2790 — gap planning now happens +inline as part of `gsd-audit-milestone`'s output. + +| User wants | Invoke | +|---|---| +| Start a new project | gsd-new-project | +| Create a new milestone | gsd-new-milestone | +| Complete the current milestone | gsd-complete-milestone | +| Audit a milestone for issues | gsd-audit-milestone | +| Summarize milestone status | gsd-milestone-summary | + +Invoke the matched skill directly using the Skill tool. diff --git a/commands/gsd/ns-review.md b/commands/gsd/ns-review.md new file mode 100644 index 000000000..f4d36d26e --- /dev/null +++ b/commands/gsd/ns-review.md @@ -0,0 +1,25 @@ +--- +name: gsd-review +description: "quality gates | code review debug audit security eval ui" +argument-hint: "" +allowed-tools: + - Read + - Skill +--- + +Route to the appropriate quality / review skill based on the user's intent. +`gsd-code-review-fix` was absorbed by `gsd-code-review --fix` in #2790. + +| User wants | Invoke | +|---|---| +| Review code for quality and correctness | gsd-code-review | +| Auto-fix code review findings | gsd-code-review --fix | +| Audit UAT / acceptance testing | gsd-audit-uat | +| Security review of a phase | gsd-secure-phase | +| Evaluate AI response quality | gsd-eval-review | +| Review UI for design and accessibility | gsd-ui-review | +| Validate phase outputs | gsd-validate-phase | +| Debug a failing feature or error | gsd-debug | +| Forensic investigation of a broken system | gsd-forensics | + +Invoke the matched skill directly using the Skill tool. diff --git a/commands/gsd/ns-workflow.md b/commands/gsd/ns-workflow.md new file mode 100644 index 000000000..680ef0457 --- /dev/null +++ b/commands/gsd/ns-workflow.md @@ -0,0 +1,27 @@ +--- +name: gsd-workflow +description: "workflow | discuss plan execute verify phase progress" +argument-hint: "" +allowed-tools: + - Read + - Skill +--- + +Route to the appropriate phase-pipeline skill based on the user's intent. +Sub-skill names below are post-#2790 consolidated targets — `gsd-phase` +absorbs the former add/insert/remove/edit-phase commands and `gsd-progress` +absorbs the former next/do commands. + +| User wants | Invoke | +|---|---| +| Gather context before planning | gsd-discuss-phase | +| Clarify what a phase delivers | gsd-spec-phase | +| Create a PLAN.md | gsd-plan-phase | +| Execute plans in a phase | gsd-execute-phase | +| Verify built features through UAT | gsd-verify-work | +| Add / insert / remove / edit a phase | gsd-phase | +| Advance to the next logical step | gsd-progress | +| Offload planning to the ultraplan cloud | gsd-ultraplan-phase | +| Cross-AI plan review convergence loop | gsd-plan-review-convergence | + +Invoke the matched skill directly using the Skill tool. diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index 8c5e4c312..18a3aa67d 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -68,6 +68,12 @@ "/gsd-milestone-summary", "/gsd-new-milestone", "/gsd-new-project", + "/gsd-ns-context", + "/gsd-ns-ideate", + "/gsd-ns-manage", + "/gsd-ns-project", + "/gsd-ns-review", + "/gsd-ns-workflow", "/gsd-pause-work", "/gsd-phase", "/gsd-plan-phase", @@ -243,6 +249,7 @@ "commands.cjs", "config-schema.cjs", "config.cjs", + "context-utilization.cjs", "core.cjs", "decisions.cjs", "docs.cjs", @@ -284,4 +291,4 @@ "gsd-workflow-guard.js" ] } -} +} \ No newline at end of file diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index 6c691c721..078f2d4ff 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -54,10 +54,23 @@ Full roster at `agents/gsd-*.md`. The "Primary doc" column flags whether [`docs/ --- -## Commands (59 shipped) +## Commands (65 shipped) Full roster at `commands/gsd/*.md`. The groupings below mirror `docs/COMMANDS.md` section order; each row carries the command name, a one-line role derived from the command's frontmatter `description:`, and a link to the source file. `tests/command-count-sync.test.cjs` locks the count against the filesystem. +### Namespace Meta-Skills + +These six routers are descriptor-only entries that the model picks first; the body of each contains a routing table that points at the correct concrete sub-skill. They exist to keep the eager skill-listing token cost low while the full surface remains reachable. See [#2792](https://github.com/gsd-build/get-shit-done/issues/2792) for the rationale; the routing tables target the post-[#2790](https://github.com/gsd-build/get-shit-done/issues/2790) consolidated surface. + +| Command | Role | Source | +|---------|------|--------| +| `/gsd-ns-workflow` | Phase pipeline router — discuss / plan / execute / verify / phase / progress. | [commands/gsd/ns-workflow.md](../commands/gsd/ns-workflow.md) | +| `/gsd-ns-project` | Project lifecycle router — milestones, audits, summary. | [commands/gsd/ns-project.md](../commands/gsd/ns-project.md) | +| `/gsd-ns-review` | Quality-gate router — code review, debug, audit, security, eval, ui. | [commands/gsd/ns-review.md](../commands/gsd/ns-review.md) | +| `/gsd-ns-context` | Codebase-intelligence router — map, graphify, docs, learnings. | [commands/gsd/ns-context.md](../commands/gsd/ns-context.md) | +| `/gsd-ns-manage` | Management router — config, workspace, workstreams, thread, update, ship, inbox. | [commands/gsd/ns-manage.md](../commands/gsd/ns-manage.md) | +| `/gsd-ns-ideate` | Exploration & capture router — explore, sketch, spike, spec, capture. | [commands/gsd/ns-ideate.md](../commands/gsd/ns-ideate.md) | + ### Core Workflow | Command | Role | Source | @@ -335,7 +348,7 @@ The `gsd-planner` agent is decomposed into a core agent plus reference modules t --- -## CLI Modules (31 shipped) +## CLI Modules (32 shipped) Full listing: `get-shit-done/bin/lib/*.cjs`. @@ -346,6 +359,7 @@ Full listing: `get-shit-done/bin/lib/*.cjs`. | `commands.cjs` | Misc CLI commands (slug, timestamp, todos, scaffolding, stats) | | `config-schema.cjs` | Single source of truth for `VALID_CONFIG_KEYS` and dynamic key patterns; imported by both the validator and the config-schema-docs parity test | | `config.cjs` | `config.json` read/write, section initialization; imports validator from `config-schema.cjs` | +| `context-utilization.cjs` | Pure classifier for `gsd-health --context` — turns (tokensUsed, contextWindow) into a `{ percent, state }` triage result against the 60%/70% fracture-point thresholds (#2792) | | `core.cjs` | Error handling, output formatting, shared utilities, runtime fallbacks | | `decisions.cjs` | Shared parser for CONTEXT.md `` blocks (D-NN entries); used by `gap-checker.cjs` and intended for #2492 plan/verify decision gates | | `docs.cjs` | Docs-update workflow init, Markdown scanning, monorepo detection | diff --git a/get-shit-done/bin/gsd-tools.cjs b/get-shit-done/bin/gsd-tools.cjs index eaa518259..6c7e2ebea 100755 --- a/get-shit-done/bin/gsd-tools.cjs +++ b/get-shit-done/bin/gsd-tools.cjs @@ -792,8 +792,48 @@ async function runCommand(command, args, cwd, raw, defaultValue) { verify.cmdValidateHealth(cwd, { repair: repairFlag, backfill: backfillFlag }, raw); } else if (subcommand === 'agents') { verify.cmdValidateAgents(cwd, raw); + } else if (subcommand === 'context') { + // The model self-reports tokensUsed and contextWindow — the SDK has + // no privileged access to either. Recommendation copy lives here + // (the renderer), not in the classifier, so it can change without + // re-validating the math layer. + const opts = parseNamedArgs(args, ['tokens-used', 'context-window']); + if (opts['tokens-used'] === null) { + error('--tokens-used is required for `validate context`'); + break; + } + if (opts['context-window'] === null) { + error('--context-window is required for `validate context`'); + break; + } + const { classifyContextUtilization, STATES } = require('./lib/context-utilization.cjs'); + const RECOMMENDATIONS = { + [STATES.HEALTHY]: null, + [STATES.WARNING]: 'Context is approaching the fracture zone — consider /gsd-thread to continue in a fresh window.', + [STATES.CRITICAL]: 'Reasoning quality may degrade past 70% utilization (fracture point). Run /gsd-thread now to preserve output quality.', + }; + let classified; + try { + classified = classifyContextUtilization(Number(opts['tokens-used']), Number(opts['context-window'])); + } catch (e) { + // Translate the classifier's TypeError into a CLI-shaped error + // message that names the offending flag. + const flag = /tokensUsed/.test(e.message) ? '--tokens-used' : '--context-window'; + error(`${flag} must be a non-negative integer (window > 0), got the values supplied`); + break; + } + const result = { ...classified, recommendation: RECOMMENDATIONS[classified.state] }; + if (args.includes('--json')) { + core.output(result, raw); + } else { + const lines = [`Context utilization: ${result.percent}% (${result.state})`]; + if (result.recommendation) lines.push(result.recommendation); + // Use core.output's rawValue path for the sync-flush guarantee + // — process.stdout.write can be truncated on process exit. + core.output(result, true, lines.join('\n')); + } } else { - error('Unknown validate subcommand. Available: consistency, health, agents'); + error('Unknown validate subcommand. Available: consistency, health, agents, context'); } break; } diff --git a/get-shit-done/bin/lib/context-utilization.cjs b/get-shit-done/bin/lib/context-utilization.cjs new file mode 100644 index 000000000..ba3ac3975 --- /dev/null +++ b/get-shit-done/bin/lib/context-utilization.cjs @@ -0,0 +1,47 @@ +'use strict'; + +/** + * Context-utilization classifier for `gsd-health --context`. + * + * Pure function. Callers pass tokensUsed + contextWindow; the + * classifier returns the percent and one of three states. Recommendation + * strings are NOT in this module — formatting is the renderer's job + * (see `validate context` in gsd-tools.cjs). That separation lets the + * copy change without touching this module's tests. + * + * Thresholds: + * < 60% healthy no action + * 60–70% warning approaching the fracture zone + * ≥ 70% critical reasoning quality may degrade + * + * State boundaries use the exact ratio. The displayed `percent` is + * rounded for human reading and may differ from the boundary by ±1 in + * edge cases (e.g. 59.999% displays as 60 but classifies as healthy). + */ + +const STATES = Object.freeze({ + HEALTHY: 'healthy', + WARNING: 'warning', + CRITICAL: 'critical', +}); + +function classifyContextUtilization(tokensUsed, contextWindow) { + if (!Number.isInteger(tokensUsed) || tokensUsed < 0) { + throw new TypeError(`tokensUsed must be a non-negative integer, got: ${tokensUsed} (${typeof tokensUsed})`); + } + if (!Number.isInteger(contextWindow) || contextWindow <= 0) { + throw new TypeError(`contextWindow must be a positive integer, got: ${contextWindow} (${typeof contextWindow})`); + } + + const ratio = Math.min(tokensUsed / contextWindow, 1); + const percent = Math.min(Math.round(ratio * 100), 100); + + let state; + if (ratio < 0.60) state = STATES.HEALTHY; + else if (ratio < 0.70) state = STATES.WARNING; + else state = STATES.CRITICAL; + + return { percent, state }; +} + +module.exports = { classifyContextUtilization, STATES }; diff --git a/get-shit-done/workflows/health.md b/get-shit-done/workflows/health.md index df750fd41..ff2c767ad 100644 --- a/get-shit-done/workflows/health.md +++ b/get-shit-done/workflows/health.md @@ -11,18 +11,53 @@ Read all files referenced by the invoking prompt's execution_context before star **Parse arguments:** -Check if `--repair` or `--backfill` flags are present in the command arguments. +Check if `--repair`, `--backfill`, or `--context` flags are present in the command arguments. ``` REPAIR_FLAG="" BACKFILL_FLAG="" +CONTEXT_MODE="" if arguments contain "--repair"; then REPAIR_FLAG="--repair" fi if arguments contain "--backfill"; then BACKFILL_FLAG="--backfill" fi +if arguments contain "--context"; then + CONTEXT_MODE="true" +fi ``` + +If `CONTEXT_MODE` is set, jump to the `context_check` step and skip the +integrity validation steps. The two modes are orthogonal — context utilization +has nothing to do with `.planning/` directory health. + + + +**Run only when `--context` is set.** + +The model running this workflow self-reports the current session's +approximate `tokensUsed` and the active model's `contextWindow`. Use the values +visible in your runtime (Claude Code's `/context` slash command output, or the +model's own session telemetry). If the runtime exposes neither, prompt the user +once via AskUserQuestion for both numbers. + +**TEXT_MODE fallback:** when `text_mode` is true (config or `--text` flag) the +runtime is non-Claude (Codex, Gemini, etc.) and `AskUserQuestion` is not +available — replace the prompt with a plain-text two-question sequence +("Approximate tokens used? Context window size?") and read the answers as +plain text from the user's response. + +```bash +gsd-sdk query validate.context \ + --tokens-used "$TOKENS_USED" \ + --context-window "$CONTEXT_WINDOW" +``` + +The query prints a one-line status (`Context utilization: NN% (state)`) plus +a recommendation line for the warning and critical states. Print the SDK +output verbatim and end the workflow — do **not** mix in `.planning/` +health output, the two modes are independent diagnostics. diff --git a/sdk/src/query/index.ts b/sdk/src/query/index.ts index c296a05a5..f5183c2b2 100644 --- a/sdk/src/query/index.ts +++ b/sdk/src/query/index.ts @@ -42,7 +42,7 @@ import { templateFill, templateSelect } from './template.js'; import { verifyPlanStructure, verifyPhaseCompleteness, verifyArtifacts, verifyCommits, verifyReferences, verifySummary, verifyPathExists } from './verify.js'; import { decisionsParse } from './decisions.js'; import { checkDecisionCoveragePlan, checkDecisionCoverageVerify } from './check-decision-coverage.js'; -import { verifyKeyLinks, validateConsistency, validateHealth, validateAgents } from './validate.js'; +import { verifyKeyLinks, validateConsistency, validateHealth, validateAgents, validateContext } from './validate.js'; import { phaseAdd, phaseAddBatch, phaseInsert, phaseRemove, phaseComplete, phaseScaffold, phasesClear, phasesArchive, @@ -141,6 +141,7 @@ export const QUERY_MUTATION_COMMANDS = new Set([ 'commit', 'check-commit', 'commit-to-subrepo', 'template.fill', 'template.select', 'template select', 'validate.health', 'validate health', + 'validate.context', 'validate context', 'phase.add', 'phase.add-batch', 'phase.insert', 'phase.remove', 'phase.complete', 'phase.scaffold', 'phases.clear', 'phases.archive', 'phase add', 'phase add-batch', 'phase insert', 'phase remove', 'phase complete', @@ -382,6 +383,8 @@ export function createRegistry( registry.register('validate health', validateHealth); registry.register('validate.agents', validateAgents); registry.register('validate agents', validateAgents); + registry.register('validate.context', validateContext); + registry.register('validate context', validateContext); // Decision routing (SDK-only — no `gsd-tools.cjs` mirror yet; see QUERY-HANDLERS.md) registry.register('check.config-gates', checkConfigGates); diff --git a/sdk/src/query/validate.ts b/sdk/src/query/validate.ts index 1c6e35de6..65eef19af 100644 --- a/sdk/src/query/validate.ts +++ b/sdk/src/query/validate.ts @@ -838,3 +838,59 @@ export const validateAgents: QueryHandler = async (_args, _projectDir) => { }, }; }; + +/** + * Classify the running session's context utilization against the + * thresholds documented in #2792: + * < 60% healthy + * 60–70% warning → recommend /gsd-thread + * ≥ 70% critical → reasoning quality may degrade ("fracture point") + * + * Args: --tokens-used --context-window + * + * The model self-reports both numbers — the SDK has no privileged access + * to either. Recommendation copy is owned by this handler (the renderer) + * so it can change without touching the math layer. + * + * Mirror of get-shit-done/bin/lib/context-utilization.cjs (the legacy + * gsd-tools.cjs path uses the CJS module). Keep both in sync. + */ +function parseFlagInt(args: string[], flag: string): number | null { + const idx = args.indexOf(flag); + if (idx === -1 || idx + 1 >= args.length) return null; + const v = Number(args[idx + 1]); + return Number.isInteger(v) ? v : null; +} + +const CONTEXT_RECOMMENDATIONS: Record = { + healthy: null, + warning: 'Context is approaching the fracture zone — consider /gsd-thread to continue in a fresh window.', + critical: 'Reasoning quality may degrade past 70% utilization (fracture point). Run /gsd-thread now to preserve output quality.', +}; + +export const validateContext: QueryHandler = async (args, _projectDir) => { + const tokensUsed = parseFlagInt(args, '--tokens-used'); + const contextWindow = parseFlagInt(args, '--context-window'); + if (tokensUsed === null || tokensUsed < 0) { + throw new GSDError( + '--tokens-used is required for `validate.context`', + ErrorClassification.Validation, + ); + } + if (contextWindow === null || contextWindow <= 0) { + throw new GSDError( + '--context-window is required for `validate.context`', + ErrorClassification.Validation, + ); + } + const ratio = Math.min(tokensUsed / contextWindow, 1); + const percent = Math.min(Math.round(ratio * 100), 100); + const state = ratio < 0.60 ? 'healthy' : ratio < 0.70 ? 'warning' : 'critical'; + return { + data: { + percent, + state, + recommendation: CONTEXT_RECOMMENDATIONS[state], + }, + }; +}; diff --git a/tests/context-utilization.test.cjs b/tests/context-utilization.test.cjs new file mode 100644 index 000000000..7476adac3 --- /dev/null +++ b/tests/context-utilization.test.cjs @@ -0,0 +1,122 @@ +'use strict'; + +/** + * Pure classifier for the gsd-health --context guard. + * + * Thresholds: + * < 60% healthy + * 60–70% warning + * ≥ 70% critical (fracture point) + * + * The classifier is a pure (tokensUsed, contextWindow) → { percent, state } + * function. Recommendation copy is owned by the SDK renderer (see + * tests/validate-context.test.cjs), not this module. + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); + +const { classifyContextUtilization, STATES } = require('../get-shit-done/bin/lib/context-utilization.cjs'); + +describe('STATES constant exposes the three boundary names', () => { + test('exports HEALTHY, WARNING, CRITICAL', () => { + assert.deepStrictEqual( + [STATES.HEALTHY, STATES.WARNING, STATES.CRITICAL], + ['healthy', 'warning', 'critical'], + ); + }); +}); + +describe('classifyContextUtilization — state thresholds', () => { + test('0 tokens used → healthy at 0%', () => { + const r = classifyContextUtilization(0, 200_000); + assert.strictEqual(r.percent, 0); + assert.strictEqual(r.state, STATES.HEALTHY); + }); + + test('just under 60% → healthy (state uses exact ratio, not rounded percent)', () => { + // 119_999 / 200_000 = 59.9995% — rounds to 60 for display, healthy by ratio. + const r = classifyContextUtilization(119_999, 200_000); + assert.strictEqual(r.state, STATES.HEALTHY); + }); + + test('exactly 60% → warning (inclusive lower bound)', () => { + const r = classifyContextUtilization(120_000, 200_000); + assert.strictEqual(r.percent, 60); + assert.strictEqual(r.state, STATES.WARNING); + }); + + test('between 60% and 70% → warning', () => { + const r = classifyContextUtilization(130_000, 200_000); + assert.strictEqual(r.percent, 65); + assert.strictEqual(r.state, STATES.WARNING); + }); + + test('just under 70% → warning', () => { + // 139_999 / 200_000 = 69.9995% — rounds to 70 for display, warning by ratio. + const r = classifyContextUtilization(139_999, 200_000); + assert.strictEqual(r.state, STATES.WARNING); + }); + + test('exactly 70% → critical (fracture point, inclusive lower bound)', () => { + const r = classifyContextUtilization(140_000, 200_000); + assert.strictEqual(r.percent, 70); + assert.strictEqual(r.state, STATES.CRITICAL); + }); + + test('above 70% → critical', () => { + const r = classifyContextUtilization(180_000, 200_000); + assert.strictEqual(r.percent, 90); + assert.strictEqual(r.state, STATES.CRITICAL); + }); + + test('tokensUsed >= contextWindow clamps to 100%', () => { + const r = classifyContextUtilization(250_000, 200_000); + assert.strictEqual(r.percent, 100); + assert.strictEqual(r.state, STATES.CRITICAL); + }); +}); + +describe('classifyContextUtilization — return shape', () => { + test('result is exactly { percent, state } — no recommendation field', () => { + // Recommendation copy lives in the renderer, not the classifier. + // Keeping this contract narrow lets the prose evolve without + // re-validating the math layer. + const r = classifyContextUtilization(100_000, 200_000); + assert.deepStrictEqual(Object.keys(r).sort(), ['percent', 'state']); + }); +}); + +describe('classifyContextUtilization — input validation', () => { + test('negative tokensUsed throws', () => { + assert.throws(() => classifyContextUtilization(-1, 200_000), /tokensUsed/); + }); + + test('non-integer tokensUsed throws', () => { + assert.throws(() => classifyContextUtilization(1.5, 200_000), /tokensUsed/); + }); + + test('zero contextWindow throws', () => { + assert.throws(() => classifyContextUtilization(100, 0), /contextWindow/); + }); + + test('negative contextWindow throws', () => { + assert.throws(() => classifyContextUtilization(100, -1), /contextWindow/); + }); + + test('non-number inputs throw via Number.isInteger', () => { + assert.throws(() => classifyContextUtilization('100', 200_000), /tokensUsed/); + assert.throws(() => classifyContextUtilization(100, '200000'), /contextWindow/); + assert.throws(() => classifyContextUtilization(NaN, 200_000), /tokensUsed/); + assert.throws(() => classifyContextUtilization(100, Infinity), /contextWindow/); + }); +}); + +describe('classifyContextUtilization — percent rounding', () => { + test('display percent rounds; state uses exact ratio', () => { + // 119_998 / 200_000 = 59.999% — display rounds to 60, state stays healthy. + const r = classifyContextUtilization(119_998, 200_000); + assert.strictEqual(r.state, STATES.HEALTHY); + assert.ok([59, 60].includes(r.percent), `expected percent ∈ {59,60}, got ${r.percent}`); + }); +}); diff --git a/tests/enh-2790-skill-consolidation.test.cjs b/tests/enh-2790-skill-consolidation.test.cjs index 1269667a6..da5a80501 100644 --- a/tests/enh-2790-skill-consolidation.test.cjs +++ b/tests/enh-2790-skill-consolidation.test.cjs @@ -250,12 +250,17 @@ describe('settings.md is kept (merged into config entry point or remains standal // Group: Skill count reduced // --------------------------------------------------------------------------- describe('skill count', () => { - test('total files in commands/gsd/*.md is <= 63', () => { - const files = fs.readdirSync(COMMANDS_DIR).filter(f => f.endsWith('.md')); + test('total user-invocable files in commands/gsd/*.md is <= 63', () => { + // Exclude `ns-*.md` namespace meta-skills (#2792) from this cap. + // Those are descriptor-only routers selected first by the model and + // are not part of the consolidation surface this test tracks; their + // own contract is enforced by tests/enh-2792-namespace-skills.test.cjs. + const files = fs.readdirSync(COMMANDS_DIR) + .filter((f) => f.endsWith('.md') && !f.startsWith('ns-')); assert.ok( files.length <= 63, [ - `Expected <= 63 skill files, found ${files.length}.`, + `Expected <= 63 user-invocable skill files, found ${files.length}.`, 'Consolidation target is ~58.', ].join(' '), ); diff --git a/tests/enh-2792-namespace-skills.test.cjs b/tests/enh-2792-namespace-skills.test.cjs new file mode 100644 index 000000000..fedbce47f --- /dev/null +++ b/tests/enh-2792-namespace-skills.test.cjs @@ -0,0 +1,255 @@ +'use strict'; + +// allow-test-rule: source-text-is-the-product +// commands/gsd/*.md files ARE what the runtime loads — testing their +// frontmatter content tests the deployed system-prompt contract. + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const COMMANDS_DIR = path.join(__dirname, '..', 'commands', 'gsd'); + +const NAMESPACE_SKILLS = [ + { file: 'ns-workflow.md', name: 'gsd-workflow' }, + { file: 'ns-project.md', name: 'gsd-project' }, + { file: 'ns-review.md', name: 'gsd-review' }, + { file: 'ns-context.md', name: 'gsd-context' }, + { file: 'ns-manage.md', name: 'gsd-manage' }, + { file: 'ns-ideate.md', name: 'gsd-ideate' }, +]; + +// Route targets named in any namespace body. The cross-reference test below +// asserts that every one of these resolves to a surviving command file or to +// a known consolidated parent (which absorbs flag-form invocations of folded +// skills, e.g. `gsd-map-codebase --fast` for the former `gsd-scan`). +const FLAG_FORM_PARENTS = new Set([ + 'gsd-code-review', // --fix absorbs former gsd-code-review-fix + 'gsd-map-codebase', // --fast absorbs scan, --query absorbs intel +]); + +/** + * Parse the leading YAML frontmatter block of a markdown file into a + * shallow `{ key: value }` map plus the trailing body. Splits on `\r?\n` + * for CRLF tolerance and uses trimmed-line equality for the `---` + * delimiters so whitespace-padded delimiter lines are accepted. + */ +function parseFrontmatter(content) { + const lines = content.split(/\r?\n/); + let openIdx = -1; + let closeIdx = -1; + for (let i = 0; i < lines.length; i += 1) { + if (lines[i].trim() === '---') { + if (openIdx === -1) openIdx = i; + else { closeIdx = i; break; } + } + } + assert.ok(openIdx !== -1 && closeIdx !== -1, 'frontmatter block must be delimited by --- on its own lines'); + const fm = {}; + for (const line of lines.slice(openIdx + 1, closeIdx)) { + const m = line.match(/^([A-Za-z][A-Za-z0-9_-]*):\s*(.*)$/); + if (!m) continue; + const [, key, raw] = m; + const value = raw.trim().replace(/^["']|["']$/g, ''); + fm[key] = value; + } + fm._body = lines.slice(closeIdx + 1).join('\n'); + return fm; +} + +function readNamespaceFile(file) { + const filePath = path.join(COMMANDS_DIR, file); + assert.ok(fs.existsSync(filePath), `${file} must exist at ${filePath}`); + return { filePath, ...parseFrontmatter(fs.readFileSync(filePath, 'utf-8')) }; +} + +// ── Frontmatter contract ─────────────────────────────────────────────── + +describe('Namespace skill files exist with correct name', () => { + for (const { file, name } of NAMESPACE_SKILLS) { + test(`${file} — name field is hyphen-form ${name}`, () => { + const fm = readNamespaceFile(file); + assert.strictEqual( + fm.name, + name, + `name: in ${file} must be ${name} (hyphen form per #2858), got: ${fm.name}`, + ); + }); + } +}); + +describe('Namespace skill descriptions are keyword-tag format', () => { + for (const { file } of NAMESPACE_SKILLS) { + test(`${file} — description ≤ 60 chars`, () => { + const fm = readNamespaceFile(file); + assert.ok( + fm.description.length <= 60, + `${file} description must be ≤ 60 chars, got ${fm.description.length}: ${fm.description}`, + ); + }); + + test(`${file} — description contains a pipe separator`, () => { + const fm = readNamespaceFile(file); + assert.ok( + fm.description.includes('|'), + `${file} description must contain | pipe separator, got: ${fm.description}`, + ); + }); + + test(`${file} — description does not start with prose ("Use " / "This skill")`, () => { + const { description } = readNamespaceFile(file); + assert.ok( + !description.startsWith('Use ') && !description.startsWith('This skill'), + `${file} description must not start with "Use " or "This skill", got: ${description}`, + ); + }); + } +}); + +// ── allowed-tools must include Skill ────────────────────────────────── + +describe('Namespace skills permit Skill execution', () => { + for (const { file } of NAMESPACE_SKILLS) { + test(`${file} — allowed-tools includes Skill`, () => { + const filePath = path.join(COMMANDS_DIR, file); + const raw = fs.readFileSync(filePath, 'utf-8'); + const lines = raw.split(/\r?\n/); + const startIdx = lines.findIndex((l) => l.trim() === 'allowed-tools:'); + assert.ok(startIdx !== -1, `${file} must declare an allowed-tools block`); + const tools = []; + for (let i = startIdx + 1; i < lines.length; i += 1) { + const m = lines[i].match(/^\s+-\s+(\S+)/); + if (!m) break; + tools.push(m[1]); + } + assert.ok( + tools.includes('Skill'), + `${file} body invokes the Skill tool but allowed-tools does not include Skill (got: ${tools.join(', ')})`, + ); + }); + } +}); + +// ── Body contains routing table ─────────────────────────────────────── + +describe('Namespace skill bodies carry a routing table', () => { + for (const { file } of NAMESPACE_SKILLS) { + test(`${file} — body contains "| User wants" table header`, () => { + const fm = readNamespaceFile(file); + const lines = fm._body.split('\n'); + const hasHeader = lines.some((l) => l.includes('| User wants')); + assert.ok(hasHeader, `${file} body must contain a routing table starting with "| User wants"`); + }); + + test(`${file} — body has at least one Invoke target`, () => { + const fm = readNamespaceFile(file); + const hasInvoke = /\bgsd-[a-z-]+/i.test(fm._body); + assert.ok(hasInvoke, `${file} body must reference at least one gsd-* sub-skill`); + }); + } +}); + +// ── Context guard contract on gsd-health ────────────────────────────── +// Asserts the `--context` surface promised by #2792 is wired through to +// both the command frontmatter and the workflow body. The classifier +// itself is covered by tests/context-utilization.test.cjs and the SDK +// CLI by tests/validate-context.test.cjs. + +describe('gsd-health --context flag is wired into command + workflow', () => { + const HEALTH_CMD = path.join(COMMANDS_DIR, 'health.md'); + const HEALTH_WORKFLOW = path.join(__dirname, '..', 'get-shit-done', 'workflows', 'health.md'); + + test('commands/gsd/health.md argument-hint advertises --context', () => { + const raw = fs.readFileSync(HEALTH_CMD, 'utf-8'); + const fm = parseFrontmatter(raw); + assert.ok( + fm['argument-hint'] && fm['argument-hint'].includes('--context'), + `health.md argument-hint must include --context, got: ${fm['argument-hint']}`, + ); + }); + + test('commands/gsd/health.md body documents the three-state utilization table', () => { + const raw = fs.readFileSync(HEALTH_CMD, 'utf-8'); + const body = parseFrontmatter(raw)._body.toLowerCase(); + assert.ok(body.includes('healthy'), 'body must name the healthy state'); + assert.ok(body.includes('warning'), 'body must name the warning state'); + assert.ok(body.includes('critical'), 'body must name the critical state'); + assert.ok( + body.includes('60%') && body.includes('70%'), + 'body must reference the 60% and 70% threshold boundaries', + ); + }); + + test('get-shit-done/workflows/health.md has a context_check step', () => { + const raw = fs.readFileSync(HEALTH_WORKFLOW, 'utf-8'); + assert.match( + raw, + //, + 'workflow must define a branch', + ); + }); + + test('workflow context_check invokes gsd-sdk query validate.context', () => { + const raw = fs.readFileSync(HEALTH_WORKFLOW, 'utf-8'); + // Extract just the context_check step's body so a stray reference + // elsewhere in the file can't satisfy this assertion. + const stepMatch = raw.match(/([\s\S]*?)<\/step>/); + assert.ok(stepMatch, 'context_check step must be a closed ... block'); + const stepBody = stepMatch[1]; + assert.match( + stepBody, + /gsd-sdk\s+query\s+validate\.context/, + 'context_check must call `gsd-sdk query validate.context`', + ); + assert.match(stepBody, /--tokens-used/, 'context_check must pass --tokens-used'); + assert.match(stepBody, /--context-window/, 'context_check must pass --context-window'); + }); +}); + +// ── Cross-reference: every routed sub-skill must exist ───────────────── +// This is the regression guard the original PR lacked. Without it, +// post-#2790 consolidations can quietly invalidate router targets again. + +describe('Namespace router targets resolve to surviving skills', () => { + // Build the post-consolidation surviving set once. + const surviving = new Set(); + for (const f of fs.readdirSync(COMMANDS_DIR)) { + if (!f.endsWith('.md')) continue; + const base = f.replace(/\.md$/, ''); + if (base.startsWith('ns-')) continue; // namespace routers themselves + surviving.add(`gsd-${base}`); + // The PR #2858 rename canonicalized extract_learnings → extract-learnings. + // Until #2790 rebases onto current main, accept either source filename + // as resolving to the canonical hyphenated identifier. + if (base === 'extract_learnings') surviving.add('gsd-extract-learnings'); + } + + for (const { file } of NAMESPACE_SKILLS) { + test(`${file} — every routing target resolves`, () => { + const fm = readNamespaceFile(file); + // Extract every gsd- token that appears in a table-row right column. + // Strip flag suffixes (`gsd-foo --bar` → `gsd-foo`) before resolving. + const targets = new Set(); + for (const line of fm._body.split('\n')) { + // Only consider markdown table data rows: lines that start with `|` + // and have content between pipes. Skip header / separator rows. + if (!line.startsWith('|') || /^\|[\s\-:|]+\|?\s*$/.test(line)) continue; + const cells = line.split('|').map((c) => c.trim()).filter(Boolean); + if (cells.length < 2) continue; + for (const m of cells[cells.length - 1].matchAll(/\bgsd-[a-z][a-z0-9-]*/g)) { + targets.add(m[0]); + } + } + assert.ok(targets.size > 0, `${file} routing table must reference at least one gsd-* target`); + const unresolved = [...targets].filter( + (t) => !surviving.has(t) && !FLAG_FORM_PARENTS.has(t), + ); + assert.deepStrictEqual( + unresolved, + [], + `${file} routes to skills that don't exist in commands/gsd/: ${unresolved.join(', ')}`, + ); + }); + } +}); diff --git a/tests/validate-context.test.cjs b/tests/validate-context.test.cjs new file mode 100644 index 000000000..f4906c521 --- /dev/null +++ b/tests/validate-context.test.cjs @@ -0,0 +1,90 @@ +'use strict'; + +/** + * SDK CLI integration tests for `gsd-tools validate context`. + * + * The pure classifier's behavior is covered by + * tests/context-utilization.test.cjs — these tests focus on what the CLI + * adds on top: argument parsing, JSON vs human-readable rendering, + * recommendation-string formatting, and exit-code semantics. + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const { runGsdTools } = require('./helpers.cjs'); + +describe('gsd-tools validate context — CLI argument errors', () => { + test('missing --tokens-used fails with named flag in stderr', () => { + const r = runGsdTools(['validate', 'context', '--context-window', '200000']); + assert.strictEqual(r.success, false); + assert.match(r.error, /tokens-used/i); + }); + + test('missing --context-window fails with named flag in stderr', () => { + const r = runGsdTools(['validate', 'context', '--tokens-used', '100000']); + assert.strictEqual(r.success, false); + assert.match(r.error, /context-window/i); + }); + + test('non-numeric --tokens-used reports the offending flag', () => { + const r = runGsdTools(['validate', 'context', '--tokens-used', 'abc', '--context-window', '200000']); + assert.strictEqual(r.success, false); + assert.match(r.error, /tokens-used/i); + }); + + test('negative --tokens-used reports the offending flag', () => { + const r = runGsdTools(['validate', 'context', '--tokens-used', '-1', '--context-window', '200000']); + assert.strictEqual(r.success, false); + assert.match(r.error, /tokens-used/i); + }); +}); + +describe('gsd-tools validate context — JSON vs human rendering', () => { + test('--json emits the classifier result plus a recommendation field', () => { + // Single round-trip test confirms (a) classifier integration, + // (b) JSON serialization, and (c) recommendation lookup. Per-state + // classifier behavior is covered by context-utilization.test.cjs. + const r = runGsdTools(['validate', 'context', '--tokens-used', '50000', '--context-window', '200000', '--json']); + assert.strictEqual(r.success, true, `expected success, got: ${r.error}`); + const obj = JSON.parse(r.output); + assert.deepStrictEqual(Object.keys(obj).sort(), ['percent', 'recommendation', 'state']); + assert.strictEqual(obj.percent, 25); + assert.strictEqual(obj.state, 'healthy'); + assert.strictEqual(obj.recommendation, null); + }); + + test('human mode (default) prints percent, state, and recommendation', () => { + const r = runGsdTools(['validate', 'context', '--tokens-used', '140000', '--context-window', '200000']); + assert.strictEqual(r.success, true); + assert.match(r.output, /70%/); + assert.match(r.output, /critical/); + assert.match(r.output, /\/gsd-thread/); + }); + + test('human mode omits the recommendation line for healthy state', () => { + const r = runGsdTools(['validate', 'context', '--tokens-used', '40000', '--context-window', '200000']); + assert.strictEqual(r.success, true); + assert.match(r.output, /20%/); + assert.match(r.output, /healthy/); + assert.doesNotMatch(r.output, /\/gsd-thread/, 'healthy output must not nag the user'); + }); +}); + +describe('gsd-tools validate context — recommendation copy per state', () => { + // The CLI owns the recommendation strings (the classifier does not). + // These tests pin the wording so a regression to the prose is caught. + test('warning state recommends /gsd-thread', () => { + const r = runGsdTools(['validate', 'context', '--tokens-used', '130000', '--context-window', '200000', '--json']); + const obj = JSON.parse(r.output); + assert.strictEqual(obj.state, 'warning'); + assert.match(obj.recommendation, /\/gsd-thread/); + }); + + test('critical state names the fracture-point reasoning risk', () => { + const r = runGsdTools(['validate', 'context', '--tokens-used', '160000', '--context-window', '200000', '--json']); + const obj = JSON.parse(r.output); + assert.strictEqual(obj.state, 'critical'); + assert.match(obj.recommendation, /\/gsd-thread/); + assert.match(obj.recommendation, /reasoning|degrade|fracture/i); + }); +});