From 6a5fa5912989d047cdd374f51b1cd2ef49aa7117 Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Mon, 18 May 2026 23:13:09 -0400 Subject: [PATCH] feat(3081): auto-trim review prompts for small-context model reviewers (#3708) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(3081): auto-trim review prompts for small-context model reviewers Adds review.max_prompt_tokens and review.max_prompt_tokens_per_reviewer config keys. When configured, the /gsd-review workflow deterministically trims the assembled prompt before sending to each reviewer (drop CONTEXT → RESEARCH → REQUIREMENTS; head-shrink PROJECT.md; tail-truncate PLANs proportionally; reserve disclosure-note tokens upfront). Trim metadata is recorded in REVIEWS.md frontmatter. Reviewer is skipped with a warning if even the minimum review set exceeds the budget. Closes #3081 * fix(3081): register prompt-budget in SDK query registry and update inventory manifest review.md references `gsd-sdk query prompt-budget` at three call sites, but the command had no handler in the SDK registry — failing the registry-integration drift-guard test on all 6 CI matrix legs. Added a native TypeScript SDK handler (sdk/src/query/prompt-budget.ts) that ports the applyBudget logic from the CJS module, registered it in DOMAIN_STATIC_CATALOG, and regenerated docs/INVENTORY-MANIFEST.json to include the new cli_modules/prompt-budget.cjs entry. Co-Authored-By: Claude Sonnet 4.6 * fix(3081): bump ws to 8.20.1 and allowlist prompt-budget sibling pair Two additional CI failures after the registry fix: 1. ws moderate CVE (GHSA-58qx-3vcg-4xpx, uninitialized memory disclosure): The advisory covers ws >=8.0.0 <8.20.1. Both root and sdk/package.json pinned ^8.20.0 which resolved to 8.20.0. Bumped both to 8.20.1 to clear the npm audit drift-guard test (bug-3588-npm-audit-clean.test.cjs). 2. lint-shared-module-handsync detected the new prompt-budget.ts / prompt-budget.cjs sibling pair without an allowlist entry. Added a cooperatingSiblings entry to scripts/shared-module-handsync-allowlist.json with classification and justification matching the established pattern. Co-Authored-By: Claude Sonnet 4.6 * fix(3081): align prompt-budget skip semantics across CJS and SDK dispatch paths Replace brittle `[ $EXIT -eq 2 ]` guards with `[ $EXIT -ne 0 ]` in all three local-reviewer blocks (Ollama, LM Studio, llama.cpp) in workflows/review.md. Any non-zero exit from prompt-budget now triggers a skip with a descriptive warning — exit 2/11 prints "budget too small", any other non-zero prints "unexpected exit code". This ensures the SDK bridge dispatch path (exit 11 via GSDError(Blocked)) triggers the same skip as the CJS path (exit 2). The SDK handler (sdk/src/query/prompt-budget.ts) already writes both metadata and prompt files before throwing, so no change needed there. The Ollama block also gains the missing OLLAMA_SKIP guard so the reviewer invocation is actually skipped (previously the block only suppressed the OLLAMA_PROMPT_FILE update but still ran the curl invocation). SDK integration path (hardFailed via GSDError(Blocked) → exit 11) is covered by handler unit tests in tests/prompt-budget.test.cjs; no gsd-sdk-*.test.cjs exercising the full bridge dispatch for this command exists yet — that gap remains and is documented here. Co-Authored-By: Claude Sonnet 4.6 * fix prompt-budget trim ordering and review guard follow-ups * perf: optimize prompt-budget and dedup reviewer trim workflow * fix(3708): drop source-grep theater tests to satisfy lint-no-source-grep All four test files added in commit 2df566ed were pure source-grep theater: they read .cjs / .ts / .md source files and asserted that specific string literals were present or absent. None exercised runtime behaviour. Deleted: - tests/gsd-tools-memory-optimizer.test.cjs — 7 includes() on gsd-tools.cjs - tests/prompt-budget-hotpath-optimizer.test.cjs — includes() on prompt-budget.cjs + .ts - tests/prompt-budget-io-optimizer.test.cjs — includes() on prompt-budget.ts + gsd-tools.cjs - tests/review-workflow-budget-dedup.test.cjs — includes() on review.md Behavioural coverage for the prompt-budget feature already exists in tests/prompt-budget.test.cjs and tests/prompt-budget-cli.test.cjs (also added by this PR). No replacement tests needed. Co-Authored-By: Claude Sonnet 4.6 * fix(3708): correct budget-pressure threshold and minSet accounting Two bugs in applyBudget caused premature trimming and false hard-fails: 1. UNNEEDED_TRIM: budgetUnderPressure compared baseTokens against effectiveBudget - NOTE_RESERVE_TOKENS, triggering trim pressure 80 tokens before the budget was actually exceeded. Fix: compare against effectiveBudget directly; NOTE_RESERVE_TOKENS are still reserved in contentBudget once real pressure is confirmed. 2. FALSE_HARDFAIL: minSet included NOTE_RESERVE_TOKENS unconditionally, treating the note as mandatory even when no trim would occur and no note would be injected. Fix: exclude NOTE_RESERVE_TOKENS from minSet; a prompt that fits untrimmed needs no note and must not hard-fail. Both fixes applied in CJS and TypeScript implementations. Two regression tests added (cycles 11 and 12) that reproduce each case behaviorally. --------- Co-authored-by: Claude Sonnet 4.6 --- .changeset/jolly-pandas-parade.md | 5 + docs/CONFIGURATION.md | 6 + docs/FEATURES.md | 1 + docs/INVENTORY-MANIFEST.json | 3 +- docs/INVENTORY.md | 3 +- get-shit-done/bin/gsd-tools.cjs | 217 ++++++- get-shit-done/bin/lib/prompt-budget.cjs | 399 +++++++++++++ get-shit-done/workflows/review.md | 169 +++++- package-lock.json | 2 +- package.json | 2 +- scripts/shared-module-handsync-allowlist.json | 6 + sdk/package-lock.json | 2 +- sdk/package.json | 2 +- sdk/shared/config-schema.manifest.json | 7 + .../query/command-static-catalog-domain.ts | 2 + sdk/src/query/prompt-budget.ts | 556 ++++++++++++++++++ tests/config-schema-sdk-parity.test.cjs | 1 + tests/helpers.cjs | 3 +- tests/prompt-budget-cli.test.cjs | 269 +++++++++ ...rompt-budget-sdk-parity-optimizer.test.cjs | 44 ++ tests/prompt-budget.test.cjs | 342 +++++++++++ 21 files changed, 2008 insertions(+), 33 deletions(-) create mode 100644 .changeset/jolly-pandas-parade.md create mode 100644 get-shit-done/bin/lib/prompt-budget.cjs create mode 100644 sdk/src/query/prompt-budget.ts create mode 100644 tests/prompt-budget-cli.test.cjs create mode 100644 tests/prompt-budget-sdk-parity-optimizer.test.cjs create mode 100644 tests/prompt-budget.test.cjs diff --git a/.changeset/jolly-pandas-parade.md b/.changeset/jolly-pandas-parade.md new file mode 100644 index 000000000..5aea64e0e --- /dev/null +++ b/.changeset/jolly-pandas-parade.md @@ -0,0 +1,5 @@ +--- +type: Added +pr: 3081 +--- +review.max_prompt_tokens and review.max_prompt_tokens_per_reviewer config keys auto-trim assembled review prompts to fit small-context local model servers (ollama, llama.cpp, lm-studio), with deterministic trim policy (drop CONTEXT → RESEARCH → REQUIREMENTS; head-shrink PROJECT.md; tail-truncate PLANs proportionally), trim metadata in REVIEWS.md frontmatter, and a reviewer-visible disclosure note when trimming occurs. diff --git a/docs/CONFIGURATION.md b/docs/CONFIGURATION.md index 6ffe77e70..d390e63d7 100644 --- a/docs/CONFIGURATION.md +++ b/docs/CONFIGURATION.md @@ -700,10 +700,16 @@ Configure per-CLI model selection for `/gsd-review`. When set, overrides the CLI | `review.models.lm_studio` | string | (server default) | Model name passed to LM Studio when `--lm-studio` reviewer is invoked. If unset, the first available model reported by the server is used. | | `review.models.llama_cpp` | string | (server default) | Model name passed to llama.cpp when `--llama-cpp` reviewer is invoked. If unset, the first model reported by `/v1/models` is used. | | `review.default_reviewers` | string[] \| null | (all detected reviewers) | Default reviewer subset for no-flag `/gsd-review`. Example: `["gemini","codex"]`. Explicit flags and `--all` override this setting. | +| `review.max_prompt_tokens` | number\|null | null | Default maximum estimated tokens for the assembled review prompt. When set, the prompt is deterministically trimmed before being sent to each reviewer. Per-reviewer overrides via `review.max_prompt_tokens_per_reviewer` take precedence. null = no trim (current behavior). | +| `review.max_prompt_tokens_per_reviewer` | object | {} | Per-reviewer token budget overrides. Keys are reviewer slugs (ollama, llama_cpp, lm_studio, gemini, claude, codex, opencode, qwen, cursor). Values override `review.max_prompt_tokens` for that reviewer. Recommended for local model servers. | | `review.ollama_host` | string | `http://localhost:11434` | Base URL of the Ollama server. Override when running Ollama on a non-default port or remote host: `gsd config-set review.ollama_host http://192.168.1.10:11434` | | `review.lm_studio_host` | string | `http://localhost:1234` | Base URL of the LM Studio local server. Override when using a non-default port. | | `review.llama_cpp_host` | string | `http://localhost:8080` | Base URL of the llama.cpp server (`llama-server`). Override when using a non-default port. | +### Prompt budgets for small-context reviewers + +Local model servers (Ollama, llama.cpp, LM Studio) typically accept far fewer tokens than cloud APIs. Setting `review.max_prompt_tokens_per_reviewer` (or the global `review.max_prompt_tokens` fallback) triggers deterministic prompt trimming before the prompt is sent to that reviewer: CONTEXT is dropped first, then RESEARCH, then REQUIREMENTS; PROJECT.md is head-shrunk to the first 40 lines; PLANs are tail-truncated proportionally — instructions and roadmap are always preserved. When a reviewer is trimmed, a disclosure note is injected at the top of the prompt and trim metadata (budget, omitted sections, truncation percentage) is recorded in the REVIEWS.md frontmatter under `trimmed_reviewers`. If even the minimum review set (instructions + roadmap + plan stubs) exceeds the budget, the reviewer is skipped with a warning rather than sending a truncated prompt that would produce misleading feedback. + ### Example ```json diff --git a/docs/FEATURES.md b/docs/FEATURES.md index e9564669b..b6e1f582c 100644 --- a/docs/FEATURES.md +++ b/docs/FEATURES.md @@ -1186,6 +1186,7 @@ When verification returns `human_needed`, items are persisted as a trackable HUM **User configuration note:** - Set `review.default_reviewers` in `.planning/config.json` (or via `gsd config-set`) to control no-flag `/gsd-review` fan-out. - Use `--all` for a full pre-merge sweep without changing project defaults. +- For local model servers with small context windows, set `review.max_prompt_tokens_per_reviewer` to auto-trim prompts per reviewer — see [Prompt budgets for small-context reviewers](../docs/CONFIGURATION.md#prompt-budgets-for-small-context-reviewers) in CONFIGURATION.md. --- diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index d68b5dca8..b4325f09f 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -1,5 +1,5 @@ { - "generated": "2026-05-17", + "generated": "2026-05-18", "families": { "agents": [ "gsd-advisor-researcher", @@ -302,6 +302,7 @@ "profile-output.cjs", "profile-pipeline.cjs", "project-root.generated.cjs", + "prompt-budget.cjs", "review-reviewer-selection.cjs", "roadmap-command-router.cjs", "roadmap.cjs", diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index b5b25061a..4c4075668 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -361,7 +361,7 @@ The `gsd-planner` agent is decomposed into a core agent plus reference modules t --- -## CLI Modules (71 shipped) +## CLI Modules (72 shipped) Full listing: `get-shit-done/bin/lib/*.cjs`. @@ -410,6 +410,7 @@ Full listing: `get-shit-done/bin/lib/*.cjs`. | `project-root.generated.cjs` | GENERATED — CJS artifact emitted from `sdk/src/project-root/index.ts` via `sdk/scripts/gen-project-root.mjs`; resolves a project root from a starting directory using four heuristics (own `.planning/` guard, `sub_repos` config, `multiRepo` flag, `.git` heuristic); do not edit directly | | `profile-output.cjs` | Profile rendering, USER-PROFILE.md and dev-preferences.md generation | | `profile-pipeline.cjs` | User behavioral profiling data pipeline, session file scanning | +| `prompt-budget.cjs` | Pure token-budget accounting for review prompts — estimates tokens, applies deterministic trim priority (head-shrink PROJECT.md, proportional plan truncation, drop context/research/requirements, hard-fail guard), returns structured metadata for `review.max_prompt_tokens` (#3081) | | `review-reviewer-selection.cjs` | Reviewer selection/normalization helpers for `/gsd-review` default reviewer policy and precedence | | `roadmap-command-router.cjs` | Thin CJS subcommand router adapter for `gsd-tools roadmap` | | `roadmap.cjs` | ROADMAP.md parsing, phase extraction, plan progress | diff --git a/get-shit-done/bin/gsd-tools.cjs b/get-shit-done/bin/gsd-tools.cjs index b17708ebf..6824527ad 100755 --- a/get-shit-done/bin/gsd-tools.cjs +++ b/get-shit-done/bin/gsd-tools.cjs @@ -448,7 +448,7 @@ async function main() { 'from-gsd2, frontmatter, gap-analysis, generate-claude-md, generate-claude-profile, ' + 'generate-dev-preferences, generate-slug, graphify, history-digest, init, intel, ' + 'learnings, list-todos, milestone, phase, phase-plan-index, phases, profile-questionnaire, ' + - 'profile-sample, progress, requirements, resolve-model, roadmap, scaffold, state, ' + + 'profile-sample, progress, prompt-budget, requirements, resolve-model, roadmap, scaffold, state, ' + 'template, validate, verify, verify-path-exists, verify-summary, workstream, worktree\n\n' + 'Global flags:\n' + ' --raw Emit raw output without post-processing\n' + @@ -494,7 +494,7 @@ async function main() { const SKIP_ROOT_RESOLUTION = new Set([ 'generate-slug', 'current-timestamp', 'verify-path-exists', 'verify-summary', 'template', 'frontmatter', 'detect-custom-files', - 'worktree', + 'worktree', 'prompt-budget', ]); if (!SKIP_ROOT_RESOLUTION.has(command)) { cwd = findProjectRoot(cwd); @@ -503,14 +503,16 @@ async function main() { // When --pick is active, intercept stdout to extract the requested field. if (pickField) { const origWriteSync = fs.writeSync; - const chunks = []; + let captured = ''; fs.writeSync = function (fd, data, ...rest) { - if (fd === 1) { chunks.push(String(data)); return; } + if (fd === 1) { + captured += String(data); + return; + } return origWriteSync.call(fs, fd, data, ...rest); }; const cleanup = () => { fs.writeSync = origWriteSync; - const captured = chunks.join(''); let jsonStr = captured; if (jsonStr.startsWith('@file:')) { jsonStr = fs.readFileSync(jsonStr.slice(6), 'utf-8'); @@ -540,9 +542,12 @@ async function main() { // every workflow to have a bash-specific `if [[ "$INIT" == @file:* ]]` check // that breaks on PowerShell and other non-bash shells. const origWriteSync2 = fs.writeSync; - const outChunks = []; + let captured = ''; fs.writeSync = function (fd, data, ...rest) { - if (fd === 1) { outChunks.push(String(data)); return; } + if (fd === 1) { + captured += String(data); + return; + } return origWriteSync2.call(fs, fd, data, ...rest); }; try { @@ -550,7 +555,6 @@ async function main() { } finally { fs.writeSync = origWriteSync2; } - let captured = outChunks.join(''); if (captured.startsWith('@file:')) { captured = fs.readFileSync(captured.slice(6), 'utf-8'); } @@ -1368,7 +1372,7 @@ async function runCommand(command, args, cwd, raw, defaultValue, originalCommand let manifest; try { - manifest = JSON.parse(fs.readFileSync(manifestPath, 'utf8')); + manifest = JSON.parse(await fs.promises.readFile(manifestPath, 'utf8')); } catch { const out = { custom_files: [], custom_count: 0, manifest_found: false, error: 'manifest parse error' }; process.stdout.write(JSON.stringify(out, null, 2)); @@ -1387,31 +1391,27 @@ async function runCommand(command, args, cwd, raw, defaultValue, originalCommand 'skills', ]; - function walkDir(dir, baseDir) { - const results = []; - if (!fs.existsSync(dir)) return results; + function collectCustomFiles(dir, baseDir, manifestKeys, out) { + if (!fs.existsSync(dir)) return; for (const entry of fs.readdirSync(dir, { withFileTypes: true })) { const fullPath = path.join(dir, entry.name); if (entry.isDirectory()) { - results.push(...walkDir(fullPath, baseDir)); - } else { - // Use forward slashes for cross-platform manifest key compatibility - const relPath = path.relative(baseDir, fullPath).replace(/\\/g, '/'); - results.push(relPath); + collectCustomFiles(fullPath, baseDir, manifestKeys, out); + continue; + } + // Use forward slashes for cross-platform manifest key compatibility + const relPath = path.relative(baseDir, fullPath).replace(/\\/g, '/'); + if (!manifestKeys.has(relPath)) { + out.push(relPath); } } - return results; } const customFiles = []; for (const managedDir of GSD_MANAGED_DIRS) { const absDir = path.join(resolvedConfigDir, managedDir); if (!fs.existsSync(absDir)) continue; - for (const relPath of walkDir(absDir, resolvedConfigDir)) { - if (!manifestKeys.has(relPath)) { - customFiles.push(relPath); - } - } + collectCustomFiles(absDir, resolvedConfigDir, manifestKeys, customFiles); } const out = { @@ -1432,6 +1432,177 @@ async function runCommand(command, args, cwd, raw, defaultValue, originalCommand break; } + // ─── Prompt Budget ──────────────────────────────────────────────────── + // + // Assemble and deterministically trim review prompt sections to fit a + // token budget. Used by the /gsd-review workflow before dispatching to + // small-context local model servers (Ollama, llama.cpp, LM Studio). + // + // Required flags: + // --budget Token budget (integer > 0) + // --instructions-file Review instructions + // --roadmap-file Roadmap section + // --plan-file Plan file (may be repeated) + // --output-prompt Write trimmed prompt here + // --output-metadata Write metadata JSON here + // + // Optional flags: + // --safety-margin-pct Default 10 + // --project-md-head-lines Default 40 + // --project-file + // --context-file + // --research-file + // --requirements-file + // + // Exit codes: + // 0 success (trim or no-trim) + // 1 invocation error (missing required arg, missing file, invalid budget) + // 2 hardFailed: prompt cannot fit effective budget after trim policy + + case 'prompt-budget': { + const promptBudget = require('./lib/prompt-budget.cjs'); + + // ── Collect multi-value --plan-file flags ────────────────────────── + const planFiles = []; + for (let i = 1; i < args.length; i++) { + if (args[i] === '--plan-file' && args[i + 1] && !args[i + 1].startsWith('--')) { + planFiles.push(args[i + 1]); + i++; + } + } + + // ── Parse single-value flags ─────────────────────────────────────── + const flagMap = new Map(); + for (let i = 1; i < args.length; i++) { + const current = args[i]; + const next = args[i + 1]; + if (!current.startsWith('--')) continue; + if (!next || next.startsWith('--')) { + if (!flagMap.has(current)) flagMap.set(current, null); + continue; + } + if (!flagMap.has(current)) flagMap.set(current, next); + i++; + } + const getFlag = (flag) => flagMap.get(flag) ?? null; + + const budgetStr = getFlag('--budget'); + const instructionsFile = getFlag('--instructions-file'); + const roadmapFile = getFlag('--roadmap-file'); + const outputPromptFile = getFlag('--output-prompt'); + const outputMetadataFile = getFlag('--output-metadata'); + const safetyMarginStr = getFlag('--safety-margin-pct'); + const projectMdHeadLinesStr = getFlag('--project-md-head-lines'); + const projectFile = getFlag('--project-file'); + const contextFile = getFlag('--context-file'); + const researchFile = getFlag('--research-file'); + const requirementsFile = getFlag('--requirements-file'); + + // ── Validate required args ───────────────────────────────────────── + if (!budgetStr) { + process.stderr.write('Error: --budget is required\n'); + process.exit(1); + } + const budget = parseInt(budgetStr, 10); + if (!Number.isFinite(budget) || budget <= 0) { + process.stderr.write('Error: --budget must be a positive integer\n'); + process.exit(1); + } + if (!instructionsFile) { + process.stderr.write('Error: --instructions-file is required\n'); + process.exit(1); + } + if (!roadmapFile) { + process.stderr.write('Error: --roadmap-file is required\n'); + process.exit(1); + } + if (planFiles.length === 0) { + process.stderr.write('Error: at least one --plan-file is required\n'); + process.exit(1); + } + if (!outputPromptFile) { + process.stderr.write('Error: --output-prompt is required\n'); + process.exit(1); + } + if (!outputMetadataFile) { + process.stderr.write('Error: --output-metadata is required\n'); + process.exit(1); + } + + // ── Validate and read required files ────────────────────────────── + async function readRequired(filePath, flagName) { + const resolved = path.resolve(filePath); + try { + return await fs.promises.readFile(resolved, 'utf8'); + } catch (err) { + if (err && err.code === 'ENOENT') { + process.stderr.write(`Error: file not found for ${flagName}: ${resolved}\n`); + process.exit(1); + } + process.stderr.write(`Error: cannot read file for ${flagName}: ${resolved}\n`); + process.exit(1); + } + } + + async function readOptional(filePath) { + if (!filePath) return null; + const resolved = path.resolve(filePath); + try { + return await fs.promises.readFile(resolved, 'utf8'); + } catch (err) { + if (err && err.code === 'ENOENT') return null; + process.stderr.write(`Error: cannot read optional file: ${resolved}\n`); + process.exit(1); + } + } + + const instructions = await readRequired(instructionsFile, '--instructions-file'); + const roadmap = await readRequired(roadmapFile, '--roadmap-file'); + const plans = await Promise.all(planFiles.map(async (p) => { + const resolved = path.resolve(p); + try { + const content = await fs.promises.readFile(resolved, 'utf8'); + return { file: path.basename(p), content }; + } catch (err) { + if (err && err.code === 'ENOENT') { + process.stderr.write(`Error: plan file not found: ${resolved}\n`); + process.exit(1); + } + process.stderr.write(`Error: cannot read plan file: ${resolved}\n`); + process.exit(1); + } + })); + + const projectMd = await readOptional(projectFile); + const context = await readOptional(contextFile); + const research = await readOptional(researchFile); + const requirements = await readOptional(requirementsFile); + + // ── Build options ───────────────────────────────────────────────── + const options = {}; + if (safetyMarginStr !== null) { + const pct = parseInt(safetyMarginStr, 10); + if (Number.isFinite(pct)) options.safetyMarginPct = pct; + } + if (projectMdHeadLinesStr !== null) { + const lines = parseInt(projectMdHeadLinesStr, 10); + if (Number.isFinite(lines)) options.projectMdHeadLines = lines; + } + + // ── Call applyBudget ────────────────────────────────────────────── + const sections = { instructions, roadmap, plans, projectMd, context, research, requirements }; + const { prompt, metadata } = promptBudget.applyBudget({ sections, budget, options }); + + // ── Write outputs ───────────────────────────────────────────────── + await fs.promises.writeFile(path.resolve(outputMetadataFile), JSON.stringify(metadata, null, 2)); + await fs.promises.writeFile(path.resolve(outputPromptFile), prompt); + + if (metadata.hardFailed) { + process.exit(2); + } + break; + } + default: { // #3243: if the caller passed a dotted form (e.g. "foo.bar"), the shim // above split it so `command` here is the head ("foo"). Use diff --git a/get-shit-done/bin/lib/prompt-budget.cjs b/get-shit-done/bin/lib/prompt-budget.cjs new file mode 100644 index 000000000..2f37a7322 --- /dev/null +++ b/get-shit-done/bin/lib/prompt-budget.cjs @@ -0,0 +1,399 @@ +'use strict'; + +/** + * prompt-budget.cjs + * + * Pure functions for assembling and trimming review prompts to fit within + * a token budget. Used by the review pipeline to support small-context models. + * + * Trim priority (in order — never violate): + * 1. Instructions: ALWAYS kept verbatim + * 2. Reserve note tokens FIRST when any trim is anticipated + * 3. Roadmap: ALWAYS kept verbatim + * 4. PROJECT.md: head-shrink to projectMdHeadLines (default 40) if over budget + * 5. Plans: tail-truncate proportionally; never drop a whole plan + * 6. Context: DROP first if still over + * 7. Research: DROP second if still over + * 8. Requirements: DROP last (last-resort) + * 9. Hard-fail: if minimum-set exceeds effectiveBudget + */ + +const NOTE_RESERVE_TOKENS = 80; + +const DEFAULT_NOTE_TEMPLATE = [ + '', + 'Prompt automatically trimmed to fit a {budget}-token budget.', + 'Omitted sections: {omittedList}.', + 'Plan content truncated by approximately {planTruncationPct}%.', + 'Treat any missing context as out-of-scope rather than a review concern.', + '', +].join('\n'); + +/** + * Estimate tokens for a string. Chars / 4, rounded up. + * + * @param {string} text + * @returns {number} + */ +function estimateTokens(text) { + if (!text) return 0; + return Math.ceil(text.length / 4); +} + +/** + * Render the trim-disclosure note. + * + * @param {string} template + * @param {number} budget + * @param {string[]} omitted + * @param {number} planTruncationPct + * @returns {string} + */ +function renderNote(template, budget, omitted, planTruncationPct) { + const omittedList = omitted.length > 0 ? omitted.join(', ') : 'none'; + return template + .replace('{budget}', String(budget)) + .replace('{omittedList}', omittedList) + .replace('{planTruncationPct}', String(Math.round(planTruncationPct))); +} + +/** + * Head-shrink a string to at most `maxLines` lines. + * + * @param {string} text + * @param {number} maxLines + * @returns {string} + */ +function headShrink(text, maxLines) { + if (maxLines <= 0) return ''; + let idx = -1; + let seen = 0; + while (seen < maxLines) { + idx = text.indexOf('\n', idx + 1); + if (idx === -1) return text; + seen += 1; + } + return text.slice(0, idx); +} + +/** + * Tail-truncate a string to at most `maxChars` characters. + * + * @param {string} text + * @param {number} maxChars + * @returns {string} + */ +function tailTruncate(text, maxChars) { + if (text.length <= maxChars) return text; + return text.slice(0, maxChars); +} + +/** + * Assemble the final prompt string from its sections. + * + * @param {object} parts + * @returns {string} + */ +function assemblePrompt(parts) { + const { + instructions, + note, + roadmap, + projectMd, + plans, + context, + research, + requirements, + } = parts; + + const blocks = []; + + blocks.push(instructions); + + if (note) blocks.push(note); + + blocks.push('## Roadmap\n\n' + roadmap); + + if (projectMd) blocks.push('## Project\n\n' + projectMd); + + const planBlocks = plans + .map((p) => '### ' + p.file + '\n\n' + p.content) + .join('\n\n'); + blocks.push('## Plans\n\n' + planBlocks); + + if (context) blocks.push('## Context\n\n' + context); + if (research) blocks.push('## Research\n\n' + research); + if (requirements) blocks.push('## Requirements\n\n' + requirements); + + return blocks.join('\n\n'); +} + +/** + * Apply a token budget to a set of review prompt sections. + * Returns the trimmed prompt and structured metadata. + * + * @param {object} param0 + * @param {object} param0.sections + * @param {number} param0.budget + * @param {object} [param0.options] + * @returns {{ prompt: string, metadata: object }} + */ +function applyBudget({ sections, budget, options = {} }) { + const { + safetyMarginPct = 10, + noteTemplate = DEFAULT_NOTE_TEMPLATE, + projectMdHeadLines = 40, + } = options; + + const effectiveBudget = Math.floor(budget * (1 - safetyMarginPct / 100)); + + const { + instructions, + roadmap, + plans, + projectMd: projectMdRaw = null, + context: contextRaw = null, + research: researchRaw = null, + requirements: requirementsRaw = null, + } = sections; + + // Working mutable state + let projectMd = projectMdRaw; + let context = contextRaw; + let research = researchRaw; + let requirements = requirementsRaw; + let workingPlans = plans.map((p) => ({ file: p.file, content: p.content })); + + const omitted = []; + let projectMdShrunk = false; + let planTruncationPct = 0; + let noteInjected = false; + let hardFailed = false; + + // Minimum-set check: instructions + roadmap + 1KB per plan. + // NOTE_RESERVE_TOKENS is intentionally excluded here: a note is only injected + // when trimming actually occurs, and a prompt that fits without any trim needs + // no note at all. Including NOTE_RESERVE_TOKENS here would cause false hard-fails + // for prompts that genuinely fit the effective budget untrimmed. + const MIN_PLAN_BYTES = 1024; + const minPlanTokens = plans.reduce((sum, p) => { + return sum + estimateTokens(p.content.slice(0, MIN_PLAN_BYTES)); + }, 0); + const minSet = + estimateTokens(instructions) + + estimateTokens(roadmap) + + minPlanTokens; + + if (minSet > effectiveBudget) { + return { + prompt: '', + metadata: { + budget, + effectiveBudget, + estimatedTokens: 0, + omitted: [], + projectMdShrunk: false, + planTruncationPct: 0, + hardFailed: true, + noteInjected: false, + }, + }; + } + + // ── Budget accounting ────────────────────────────────────────────────────── + const TOKENS_ROADMAP_HEADER = estimateTokens('## Roadmap\n\n'); + const TOKENS_PROJECT_HEADER = estimateTokens('## Project\n\n'); + const TOKENS_PLANS_HEADER = estimateTokens('## Plans\n\n'); + const TOKENS_CONTEXT_HEADER = estimateTokens('## Context\n\n'); + const TOKENS_RESEARCH_HEADER = estimateTokens('## Research\n\n'); + const TOKENS_REQUIREMENTS_HEADER = estimateTokens('## Requirements\n\n'); + const TOKENS_PLAN_ITEM_HEADERS = workingPlans.reduce( + (sum, p) => sum + estimateTokens('### ' + p.file + '\n\n'), + 0 + ); + + const staticBaseTokens = + estimateTokens(instructions) + + TOKENS_ROADMAP_HEADER + + estimateTokens(roadmap) + + TOKENS_PLANS_HEADER + + TOKENS_PLAN_ITEM_HEADERS; + + let projectTokens = projectMd + ? TOKENS_PROJECT_HEADER + estimateTokens(projectMd) + : 0; + let contextTokens = context + ? TOKENS_CONTEXT_HEADER + estimateTokens(context) + : 0; + let researchTokens = research + ? TOKENS_RESEARCH_HEADER + estimateTokens(research) + : 0; + let requirementsTokens = requirements + ? TOKENS_REQUIREMENTS_HEADER + estimateTokens(requirements) + : 0; + let planContentTokens = workingPlans.reduce( + (sum, p) => sum + estimateTokens(p.content), + 0 + ); + + const getCurrentBaseTokens = () => + staticBaseTokens + + projectTokens + + planContentTokens + + contextTokens + + researchTokens + + requirementsTokens; + + let currentBaseTokens = getCurrentBaseTokens(); + + // Detect budget pressure: is ANY trim needed? + // Pressure exists when the current base tokens already exceed the effective + // budget. Only when pressure is real do we reserve NOTE_RESERVE_TOKENS so + // the note itself fits after trimming. Checking against + // effectiveBudget - NOTE_RESERVE_TOKENS (the old threshold) would cause + // spurious pressure 80 tokens early, dropping sections that fit fine. + const baseTokens = currentBaseTokens; + const budgetUnderPressure = baseTokens > effectiveBudget; + + // Available for content (reserve note slot when under pressure) + let contentBudget = budgetUnderPressure + ? effectiveBudget - NOTE_RESERVE_TOKENS + : effectiveBudget; + + // ── Trim step 1: head-shrink PROJECT.md ─────────────────────────────────── + if (currentBaseTokens > contentBudget && projectMd) { + const shrunk = headShrink(projectMd, projectMdHeadLines); + if (shrunk !== projectMd) { + projectMd = shrunk; + projectMdShrunk = true; + projectTokens = TOKENS_PROJECT_HEADER + estimateTokens(projectMd); + currentBaseTokens = getCurrentBaseTokens(); + } + } + + // ── Trim step 2: proportional plan truncation ───────────────────────────── + if (currentBaseTokens > contentBudget) { + // Compute tokens available for plan content only + const overhead = + staticBaseTokens + + projectTokens + + contextTokens + + researchTokens + + requirementsTokens; + const planBudgetTokens = contentBudget - overhead; + const totalPlanTokens = planContentTokens; + + if (planBudgetTokens > 0 && planBudgetTokens < totalPlanTokens) { + // Proportional share per plan (at least 1KB per plan) + const totalOriginalChars = plans.reduce( + (sum, p) => sum + p.content.length, + 0 + ); + + const totalPlanCharsBudget = planBudgetTokens * 4; + workingPlans = workingPlans.map((p) => { + const proportionalShare = + totalOriginalChars > 0 + ? Math.floor((p.content.length / totalOriginalChars) * totalPlanCharsBudget) + : 0; + const maxChars = Math.max(proportionalShare, MIN_PLAN_BYTES); + return { file: p.file, content: tailTruncate(p.content, maxChars) }; + }); + + const newTotalChars = workingPlans.reduce( + (sum, p) => sum + p.content.length, + 0 + ); + if (totalOriginalChars > 0) { + planTruncationPct = + ((totalOriginalChars - newTotalChars) / totalOriginalChars) * 100; + } + planContentTokens = workingPlans.reduce( + (sum, p) => sum + estimateTokens(p.content), + 0 + ); + currentBaseTokens = getCurrentBaseTokens(); + } + } + + // ── Trim step 3: drop context ───────────────────────────────────────────── + if (currentBaseTokens > contentBudget && context) { + context = null; + omitted.push('context'); + contextTokens = 0; + currentBaseTokens = getCurrentBaseTokens(); + } + + // ── Trim step 4: drop research ──────────────────────────────────────────── + if (currentBaseTokens > contentBudget && research) { + research = null; + omitted.push('research'); + researchTokens = 0; + currentBaseTokens = getCurrentBaseTokens(); + } + + // ── Trim step 5: drop requirements (last resort) ────────────────────────── + if (currentBaseTokens > contentBudget && requirements) { + requirements = null; + omitted.push('requirements'); + requirementsTokens = 0; + currentBaseTokens = getCurrentBaseTokens(); + } + + // ── Decide whether note is actually needed ──────────────────────────────── + const anyTrimOccurred = + omitted.length > 0 || projectMdShrunk || planTruncationPct > 0; + + let note = null; + if (anyTrimOccurred) { + note = renderNote(noteTemplate, budget, omitted, planTruncationPct); + noteInjected = true; + } + + // ── Assemble ────────────────────────────────────────────────────────────── + const prompt = assemblePrompt({ + instructions, + note, + roadmap, + projectMd, + plans: workingPlans, + context, + research, + requirements, + }); + + const estimatedTokens = estimateTokens(prompt); + + if (estimatedTokens > effectiveBudget) { + hardFailed = true; + return { + prompt: '', + metadata: { + budget, + effectiveBudget, + estimatedTokens, + omitted, + projectMdShrunk, + planTruncationPct, + hardFailed, + noteInjected, + }, + }; + } + + return { + prompt, + metadata: { + budget, + effectiveBudget, + estimatedTokens, + omitted, + projectMdShrunk, + planTruncationPct, + hardFailed, + noteInjected, + }, + }; +} + +module.exports = { estimateTokens, applyBudget }; diff --git a/get-shit-done/workflows/review.md b/get-shit-done/workflows/review.md index 2dd10a6f7..2b0647bed 100644 --- a/get-shit-done/workflows/review.md +++ b/get-shit-done/workflows/review.md @@ -173,6 +173,39 @@ Output your review in markdown format. ``` Write to a temp file: `/tmp/gsd-review-prompt-{phase}.md` + +Also write individual section files so the budget tool can re-trim per reviewer: + +```bash +# Write individual section files for per-reviewer budget trimming +# These are always written so reviewers with a budget can invoke prompt-budget +cp "$INSTRUCTIONS_BLOCK_FILE" "/tmp/gsd-review-${PHASE}-instructions.md" +cp "$ROADMAP_SECTION_FILE" "/tmp/gsd-review-${PHASE}-roadmap.md" + +# Plan files: copy each PLAN.md to a predictable numbered path +PLAN_INDEX=0 +for PLAN_FILE in "${PHASE_DIR}"/*-PLAN.md; do + PADDED_IDX=$(printf '%02d' "$PLAN_INDEX") + cp "$PLAN_FILE" "/tmp/gsd-review-${PHASE}-plan-${PADDED_IDX}.md" + PLAN_INDEX=$((PLAN_INDEX + 1)) +done + +# Optional section files (only if content was included in the combined prompt) +if [ -f ".planning/PROJECT.md" ]; then + cp .planning/PROJECT.md "/tmp/gsd-review-${PHASE}-project.md" +fi +if ls "${PHASE_DIR}/"*"-CONTEXT.md" >/dev/null 2>&1; then + cat "${PHASE_DIR}/"*"-CONTEXT.md" > "/tmp/gsd-review-${PHASE}-context.md" +fi +if ls "${PHASE_DIR}/"*"-RESEARCH.md" >/dev/null 2>&1; then + cat "${PHASE_DIR}/"*"-RESEARCH.md" > "/tmp/gsd-review-${PHASE}-research.md" +fi +if [ -f ".planning/REQUIREMENTS.md" ]; then + cp .planning/REQUIREMENTS.md "/tmp/gsd-review-${PHASE}-requirements.md" +fi +``` + +Note: The variable names above (`INSTRUCTIONS_BLOCK_FILE`, `ROADMAP_SECTION_FILE`, `PHASE_DIR`, `PHASE`) reference the variables already established during prompt assembly. In practice the AI implementing this step writes the instruction and roadmap blocks to temp files while assembling the combined prompt, then copies those same temp files to the per-reviewer section paths. If the assembled prompt was built inline (string concatenation rather than file-by-file), write each section to the corresponding path after writing the combined file. @@ -256,13 +289,74 @@ fi Read host and model from config. All three local backends share the same `/v1/chat/completions` endpoint — only host and model differ. Use `jq --rawfile` to safely encode the multi-line prompt as JSON without shell-escaping issues. ```bash +# Shared helper: apply prompt-budget trimming for local reviewers +prepare_trimmed_prompt_for_reviewer() { + REVIEWER_KEY="$1" + REVIEWER_BUDGET="$2" + OUTPUT_PROMPT="$3" + OUTPUT_META="$4" + + [ -z "$REVIEWER_BUDGET" ] && return 0 + [ "$REVIEWER_BUDGET" = "null" ] && return 0 + [ "$REVIEWER_BUDGET" = "0" ] && return 0 + + PLAN_FILE_ARGS="" + for p in /tmp/gsd-review-{phase}-plan-*.md; do + [ -f "$p" ] && PLAN_FILE_ARGS="$PLAN_FILE_ARGS --plan-file $p" + done + PROJECT_ARG="" + [ -f "/tmp/gsd-review-{phase}-project.md" ] && PROJECT_ARG="--project-file /tmp/gsd-review-{phase}-project.md" + CONTEXT_ARG="" + [ -f "/tmp/gsd-review-{phase}-context.md" ] && CONTEXT_ARG="--context-file /tmp/gsd-review-{phase}-context.md" + RESEARCH_ARG="" + [ -f "/tmp/gsd-review-{phase}-research.md" ] && RESEARCH_ARG="--research-file /tmp/gsd-review-{phase}-research.md" + REQUIREMENTS_ARG="" + [ -f "/tmp/gsd-review-{phase}-requirements.md" ] && REQUIREMENTS_ARG="--requirements-file /tmp/gsd-review-{phase}-requirements.md" + + gsd-sdk query prompt-budget \ + --budget "$REVIEWER_BUDGET" \ + --instructions-file "/tmp/gsd-review-{phase}-instructions.md" \ + --roadmap-file "/tmp/gsd-review-{phase}-roadmap.md" \ + $PLAN_FILE_ARGS $PROJECT_ARG $CONTEXT_ARG $RESEARCH_ARG $REQUIREMENTS_ARG \ + --output-prompt "$OUTPUT_PROMPT" \ + --output-metadata "$OUTPUT_META" + return $? +} + +# Resolve prompt budget for Ollama: per-reviewer override > global default > null +OLLAMA_REVIEWER_BUDGET=$(gsd-sdk query config-get review.max_prompt_tokens_per_reviewer.ollama 2>/dev/null | jq -r '.' 2>/dev/null || echo "null") +if [ -z "$OLLAMA_REVIEWER_BUDGET" ] || [ "$OLLAMA_REVIEWER_BUDGET" = "null" ]; then + OLLAMA_REVIEWER_BUDGET=$(gsd-sdk query config-get review.max_prompt_tokens 2>/dev/null | jq -r '.' 2>/dev/null || echo "null") +fi + +# Apply budget trim for Ollama if a budget is configured +OLLAMA_PROMPT_FILE="/tmp/gsd-review-prompt-{phase}.md" +OLLAMA_SKIP=0 +if [ -n "$OLLAMA_REVIEWER_BUDGET" ] && [ "$OLLAMA_REVIEWER_BUDGET" != "null" ] && [ "$OLLAMA_REVIEWER_BUDGET" != "0" ]; then + OLLAMA_TRIMMED_PROMPT="/tmp/gsd-review-prompt-{phase}-ollama.md" + OLLAMA_TRIM_META="/tmp/gsd-review-prompt-{phase}-ollama.metadata.json" + prepare_trimmed_prompt_for_reviewer "ollama" "$OLLAMA_REVIEWER_BUDGET" "$OLLAMA_TRIMMED_PROMPT" "$OLLAMA_TRIM_META" + OLLAMA_EXIT=$? + if [ $OLLAMA_EXIT -ne 0 ]; then + if [ $OLLAMA_EXIT -eq 2 ] || [ $OLLAMA_EXIT -eq 11 ]; then + echo "WARNING: prompt budget for ollama (${OLLAMA_REVIEWER_BUDGET} tokens) is too small for the minimum review set. Skipping Ollama reviewer." >&2 + else + echo "WARNING: prompt-budget returned unexpected exit code ${OLLAMA_EXIT} for ollama. Skipping Ollama reviewer." >&2 + fi + OLLAMA_SKIP=1 + else + OLLAMA_PROMPT_FILE="$OLLAMA_TRIMMED_PROMPT" + fi +fi + +if [ "$OLLAMA_SKIP" != "1" ]; then OLLAMA_HOST=$(gsd-sdk query config-get review.ollama_host 2>/dev/null | jq -r '.' 2>/dev/null || echo "") if [ -z "$OLLAMA_HOST" ] || [ "$OLLAMA_HOST" = "null" ]; then OLLAMA_HOST="http://localhost:11434"; fi OLLAMA_MODEL=$(gsd-sdk query config-get review.models.ollama 2>/dev/null | jq -r '.' 2>/dev/null || echo "") if [ -z "$OLLAMA_MODEL" ] || [ "$OLLAMA_MODEL" = "null" ]; then OLLAMA_MODEL=$(curl -s --max-time 2 "${OLLAMA_HOST}/v1/models" 2>/dev/null | jq -r '.data[0].id // "llama3"' 2>/dev/null || echo "llama3") fi -jq -n --rawfile content /tmp/gsd-review-prompt-{phase}.md \ +jq -n --rawfile content "$OLLAMA_PROMPT_FILE" \ --arg model "$OLLAMA_MODEL" \ '{model: $model, messages: [{role: "user", content: $content}]}' | \ curl -s --max-time 120 -X POST "${OLLAMA_HOST}/v1/chat/completions" \ @@ -272,17 +366,45 @@ jq -n --rawfile content /tmp/gsd-review-prompt-{phase}.md \ if [ ! -s /tmp/gsd-review-ollama-{phase}.md ]; then echo "Ollama review failed or returned empty output." > /tmp/gsd-review-ollama-{phase}.md fi +fi ``` **LM Studio (local, OpenAI-compatible):** ```bash +# Resolve prompt budget for LM Studio: per-reviewer override > global default > null +LM_STUDIO_REVIEWER_BUDGET=$(gsd-sdk query config-get review.max_prompt_tokens_per_reviewer.lm_studio 2>/dev/null | jq -r '.' 2>/dev/null || echo "null") +if [ -z "$LM_STUDIO_REVIEWER_BUDGET" ] || [ "$LM_STUDIO_REVIEWER_BUDGET" = "null" ]; then + LM_STUDIO_REVIEWER_BUDGET=$(gsd-sdk query config-get review.max_prompt_tokens 2>/dev/null | jq -r '.' 2>/dev/null || echo "null") +fi + +# Apply budget trim for LM Studio if a budget is configured +LM_STUDIO_PROMPT_FILE="/tmp/gsd-review-prompt-{phase}.md" +LM_STUDIO_SKIP=0 +if [ -n "$LM_STUDIO_REVIEWER_BUDGET" ] && [ "$LM_STUDIO_REVIEWER_BUDGET" != "null" ] && [ "$LM_STUDIO_REVIEWER_BUDGET" != "0" ]; then + LM_STUDIO_TRIMMED_PROMPT="/tmp/gsd-review-prompt-{phase}-lm_studio.md" + LM_STUDIO_TRIM_META="/tmp/gsd-review-prompt-{phase}-lm_studio.metadata.json" + prepare_trimmed_prompt_for_reviewer "lm_studio" "$LM_STUDIO_REVIEWER_BUDGET" "$LM_STUDIO_TRIMMED_PROMPT" "$LM_STUDIO_TRIM_META" + LM_STUDIO_EXIT=$? + if [ $LM_STUDIO_EXIT -ne 0 ]; then + if [ $LM_STUDIO_EXIT -eq 2 ] || [ $LM_STUDIO_EXIT -eq 11 ]; then + echo "WARNING: prompt budget for lm_studio (${LM_STUDIO_REVIEWER_BUDGET} tokens) is too small for the minimum review set. Skipping LM Studio reviewer." >&2 + else + echo "WARNING: prompt-budget returned unexpected exit code ${LM_STUDIO_EXIT} for lm_studio. Skipping LM Studio reviewer." >&2 + fi + LM_STUDIO_SKIP=1 + else + LM_STUDIO_PROMPT_FILE="$LM_STUDIO_TRIMMED_PROMPT" + fi +fi + +if [ "$LM_STUDIO_SKIP" != "1" ]; then LM_STUDIO_HOST=$(gsd-sdk query config-get review.lm_studio_host 2>/dev/null | jq -r '.' 2>/dev/null || echo "") if [ -z "$LM_STUDIO_HOST" ] || [ "$LM_STUDIO_HOST" = "null" ]; then LM_STUDIO_HOST="http://localhost:1234"; fi LM_STUDIO_MODEL=$(gsd-sdk query config-get review.models.lm_studio 2>/dev/null | jq -r '.' 2>/dev/null || echo "") if [ -z "$LM_STUDIO_MODEL" ] || [ "$LM_STUDIO_MODEL" = "null" ]; then LM_STUDIO_MODEL=$(curl -s --max-time 2 "${LM_STUDIO_HOST}/v1/models" 2>/dev/null | jq -r '.data[0].id // "local-model"' 2>/dev/null || echo "local-model") fi -LM_STUDIO_RESPONSE=$(jq -n --rawfile content /tmp/gsd-review-prompt-{phase}.md \ +LM_STUDIO_RESPONSE=$(jq -n --rawfile content "$LM_STUDIO_PROMPT_FILE" \ --arg model "$LM_STUDIO_MODEL" \ '{model: $model, messages: [{role: "user", content: $content}]}' | \ curl -s --max-time 120 -X POST "${LM_STUDIO_HOST}/v1/chat/completions" \ @@ -297,17 +419,45 @@ if [ -n "$LM_STUDIO_CONTENT" ]; then else echo "Warning: LM Studio returned empty content — skipping review." >&2 fi +fi ``` **llama.cpp (local, OpenAI-compatible):** ```bash +# Resolve prompt budget for llama.cpp: per-reviewer override > global default > null +LLAMA_CPP_REVIEWER_BUDGET=$(gsd-sdk query config-get review.max_prompt_tokens_per_reviewer.llama_cpp 2>/dev/null | jq -r '.' 2>/dev/null || echo "null") +if [ -z "$LLAMA_CPP_REVIEWER_BUDGET" ] || [ "$LLAMA_CPP_REVIEWER_BUDGET" = "null" ]; then + LLAMA_CPP_REVIEWER_BUDGET=$(gsd-sdk query config-get review.max_prompt_tokens 2>/dev/null | jq -r '.' 2>/dev/null || echo "null") +fi + +# Apply budget trim for llama.cpp if a budget is configured +LLAMA_CPP_PROMPT_FILE="/tmp/gsd-review-prompt-{phase}.md" +LLAMA_CPP_SKIP=0 +if [ -n "$LLAMA_CPP_REVIEWER_BUDGET" ] && [ "$LLAMA_CPP_REVIEWER_BUDGET" != "null" ] && [ "$LLAMA_CPP_REVIEWER_BUDGET" != "0" ]; then + LLAMA_CPP_TRIMMED_PROMPT="/tmp/gsd-review-prompt-{phase}-llama_cpp.md" + LLAMA_CPP_TRIM_META="/tmp/gsd-review-prompt-{phase}-llama_cpp.metadata.json" + prepare_trimmed_prompt_for_reviewer "llama_cpp" "$LLAMA_CPP_REVIEWER_BUDGET" "$LLAMA_CPP_TRIMMED_PROMPT" "$LLAMA_CPP_TRIM_META" + LLAMA_CPP_EXIT=$? + if [ $LLAMA_CPP_EXIT -ne 0 ]; then + if [ $LLAMA_CPP_EXIT -eq 2 ] || [ $LLAMA_CPP_EXIT -eq 11 ]; then + echo "WARNING: prompt budget for llama_cpp (${LLAMA_CPP_REVIEWER_BUDGET} tokens) is too small for the minimum review set. Skipping llama.cpp reviewer." >&2 + else + echo "WARNING: prompt-budget returned unexpected exit code ${LLAMA_CPP_EXIT} for llama_cpp. Skipping llama.cpp reviewer." >&2 + fi + LLAMA_CPP_SKIP=1 + else + LLAMA_CPP_PROMPT_FILE="$LLAMA_CPP_TRIMMED_PROMPT" + fi +fi + +if [ "$LLAMA_CPP_SKIP" != "1" ]; then LLAMA_CPP_HOST=$(gsd-sdk query config-get review.llama_cpp_host 2>/dev/null | jq -r '.' 2>/dev/null || echo "") if [ -z "$LLAMA_CPP_HOST" ] || [ "$LLAMA_CPP_HOST" = "null" ]; then LLAMA_CPP_HOST="http://localhost:8080"; fi LLAMA_CPP_MODEL=$(gsd-sdk query config-get review.models.llama_cpp 2>/dev/null | jq -r '.' 2>/dev/null || echo "") if [ -z "$LLAMA_CPP_MODEL" ] || [ "$LLAMA_CPP_MODEL" = "null" ]; then LLAMA_CPP_MODEL=$(curl -s --max-time 2 "${LLAMA_CPP_HOST}/v1/models" 2>/dev/null | jq -r '.data[0].id // "local-model"' 2>/dev/null || echo "local-model") fi -LLAMA_CPP_CONTENT=$(jq -n --rawfile content /tmp/gsd-review-prompt-{phase}.md \ +LLAMA_CPP_CONTENT=$(jq -n --rawfile content "$LLAMA_CPP_PROMPT_FILE" \ --arg model "$LLAMA_CPP_MODEL" \ '{model: $model, messages: [{role: "user", content: $content}]}' | \ curl -s --max-time 120 -X POST "${LLAMA_CPP_HOST}/v1/chat/completions" \ @@ -318,6 +468,7 @@ if [ -n "$LLAMA_CPP_CONTENT" ]; then else echo "Warning: llama.cpp returned empty content — skipping review." >&2 fi +fi ``` If a CLI or local server fails, log the error and continue with remaining reviewers. @@ -336,12 +487,24 @@ Display progress: Combine all review responses into `{phase_dir}/{padded_phase}-REVIEWS.md`: +After all reviewers complete, collect trim metadata files written during the run. For each reviewer that was trimmed (i.e. a `.metadata.json` file exists and `hardFailed` or `omitted` is non-empty, or `projectMdShrunk` is true, or `planTruncationPct > 0`), include a `trimmed_reviewers` block in the frontmatter. Omit the key entirely if no reviewer was trimmed. + ```markdown --- phase: {N} reviewers: [gemini, claude, codex, coderabbit, opencode, qwen, cursor, ollama, lm_studio, llama_cpp] # populate at runtime with only the reviewers actually invoked reviewed_at: {ISO timestamp} plans_reviewed: [{list of PLAN.md files}] +trimmed_reviewers: # only present if at least one reviewer was trimmed + ollama: + budget: 6000 + effective_budget: 5400 + estimated_tokens: 5380 + omitted: [context, research] + project_md_shrunk: true + plan_truncation_pct: 22 + hard_failed: false + note_injected: true --- # Cross-AI Plan Review — Phase {N} diff --git a/package-lock.json b/package-lock.json index 3dd9a0694..20a36bec0 100644 --- a/package-lock.json +++ b/package-lock.json @@ -10,7 +10,7 @@ "license": "MIT", "dependencies": { "@anthropic-ai/claude-agent-sdk": "^0.2.84", - "ws": "^8.20.0" + "ws": "8.20.1" }, "bin": { "get-shit-done-cc": "bin/install.js", diff --git a/package.json b/package.json index 0ee97b037..1424e47c5 100644 --- a/package.json +++ b/package.json @@ -49,7 +49,7 @@ }, "dependencies": { "@anthropic-ai/claude-agent-sdk": "^0.2.84", - "ws": "^8.20.0" + "ws": "8.20.1" }, "devDependencies": { "c8": "^11.0.0" diff --git a/scripts/shared-module-handsync-allowlist.json b/scripts/shared-module-handsync-allowlist.json index a5829603d..53dadb689 100644 --- a/scripts/shared-module-handsync-allowlist.json +++ b/scripts/shared-module-handsync-allowlist.json @@ -133,6 +133,12 @@ "ts": "sdk/src/workstream-name-policy.ts", "classification": "ADAPTER-OVER-MODULE", "justification": "Phase 6 (#3575): CJS workstream-name-policy.cjs is the generated Adapter reading from sdk/src/workstream-name-policy.ts Shared Module. SDK source-of-truth now exports all three functions used by CJS callers (toWorkstreamSlug, hasInvalidPathSegment, isValidActiveWorkstreamName) plus validateWorkstreamName alias. Freshness check (check-workstream-name-policy-fresh.mjs) enforces alignment." + }, + { + "cjs": "get-shit-done/bin/lib/prompt-budget.cjs", + "ts": "sdk/src/query/prompt-budget.ts", + "classification": "cooperating-sibling", + "justification": "CJS prompt-budget.cjs provides the applyBudget implementation used by gsd-tools.cjs case 'prompt-budget'. SDK prompt-budget.ts is the native QueryHandler port for gsd-sdk query dispatch (#3081). The SDK handler ports the pure budget logic and adds CLI arg parsing / file I/O directly, satisfying the registry-integration drift-guard without duplicating shared state." } ], "migrateMeBacklog": [] diff --git a/sdk/package-lock.json b/sdk/package-lock.json index 25c646f65..19131504c 100644 --- a/sdk/package-lock.json +++ b/sdk/package-lock.json @@ -11,7 +11,7 @@ "dependencies": { "@anthropic-ai/claude-agent-sdk": "^0.2.84", "synckit": "^0.11.12", - "ws": "^8.20.0" + "ws": "8.20.1" }, "bin": { "gsd-sdk": "dist/cli.js" diff --git a/sdk/package.json b/sdk/package.json index 78814f24e..bc71bff1d 100644 --- a/sdk/package.json +++ b/sdk/package.json @@ -62,7 +62,7 @@ "dependencies": { "@anthropic-ai/claude-agent-sdk": "^0.2.84", "synckit": "^0.11.12", - "ws": "^8.20.0" + "ws": "8.20.1" }, "devDependencies": { "@types/node": "^22.0.0", diff --git a/sdk/shared/config-schema.manifest.json b/sdk/shared/config-schema.manifest.json index 59000f973..4d13c8661 100644 --- a/sdk/shared/config-schema.manifest.json +++ b/sdk/shared/config-schema.manifest.json @@ -62,6 +62,8 @@ "review.lm_studio_host", "review.llama_cpp_host", "review.default_reviewers", + "review.max_prompt_tokens", + "review.max_prompt_tokens_per_reviewer", "workflow.cross_ai_execution", "workflow.cross_ai_command", "workflow.cross_ai_timeout", @@ -139,6 +141,11 @@ "topLevel": "model_overrides", "source": "^model_overrides\\.[a-zA-Z0-9_-]+$", "description": "model_overrides." + }, + { + "topLevel": "review", + "source": "^review\\.max_prompt_tokens_per_reviewer\\.[a-zA-Z0-9_-]+$", + "description": "review.max_prompt_tokens_per_reviewer." } ] } diff --git a/sdk/src/query/command-static-catalog-domain.ts b/sdk/src/query/command-static-catalog-domain.ts index 1bdb0b53f..6bc5fe858 100644 --- a/sdk/src/query/command-static-catalog-domain.ts +++ b/sdk/src/query/command-static-catalog-domain.ts @@ -21,6 +21,7 @@ import { uatRenderCheckpoint, auditUat } from './uat.js'; import { writeProfile, generateClaudeProfile, generateDevPreferences, generateClaudeMd } from './profile-output.js'; import { phaseMvpMode, taskIsBehaviorAdding, userStoryValidate } from './mvp.js'; import { worktreeCleanupWave } from './worktree.js'; +import { promptBudget } from './prompt-budget.js'; export const DOMAIN_STATIC_CATALOG: ReadonlyArray = [ ['agent-skills', agentSkills], @@ -104,4 +105,5 @@ export const DOMAIN_STATIC_CATALOG: ReadonlyArray Token budget (required, positive integer) + * --instructions-file Path to instructions file (required) + * --roadmap-file Path to roadmap file (required) + * --plan-file Path to a plan file (required, repeatable) + * --output-prompt Path to write the trimmed prompt (required) + * --output-metadata Path to write the JSON metadata (required) + * --project-file Optional PROJECT.md file + * --context-file Optional context file + * --research-file Optional research file + * --requirements-file Optional requirements file + * --safety-margin-pct Safety margin % (default 10) + * --project-md-head-lines Max lines from PROJECT.md (default 40) + * + * Exit codes (propagated through dispatch error): + * 0 success (trim or no-trim) + * 1 invocation error (missing required arg, missing file, invalid budget) + * 2 hardFailed: prompt cannot fit effective budget after trim policy + */ + +import { readFile, writeFile } from 'node:fs/promises'; +import { basename, resolve } from 'node:path'; +import { GSDError, ErrorClassification } from '../errors.js'; +import type { QueryHandler } from './utils.js'; + +// ─── Constants ─────────────────────────────────────────────────────────────── + +const NOTE_RESERVE_TOKENS = 80; + +const DEFAULT_NOTE_TEMPLATE = [ + '', + 'Prompt automatically trimmed to fit a {budget}-token budget.', + 'Omitted sections: {omittedList}.', + 'Plan content truncated by approximately {planTruncationPct}%.', + 'Treat any missing context as out-of-scope rather than a review concern.', + '', +].join('\n'); + +// ─── Pure helpers ───────────────────────────────────────────────────────────── + +function estimateTokens(text: string): number { + if (!text) return 0; + return Math.ceil(text.length / 4); +} + +function renderNote( + template: string, + budget: number, + omitted: string[], + planTruncationPct: number, +): string { + const omittedList = omitted.length > 0 ? omitted.join(', ') : 'none'; + return template + .replace('{budget}', String(budget)) + .replace('{omittedList}', omittedList) + .replace('{planTruncationPct}', String(Math.round(planTruncationPct))); +} + +function headShrink(text: string, maxLines: number): string { + if (maxLines <= 0) return ''; + let idx = -1; + let seen = 0; + while (seen < maxLines) { + idx = text.indexOf('\n', idx + 1); + if (idx === -1) return text; + seen += 1; + } + return text.slice(0, idx); +} + +function tailTruncate(text: string, maxChars: number): string { + if (text.length <= maxChars) return text; + return text.slice(0, maxChars); +} + +interface PlanEntry { + file: string; + content: string; +} + +interface BudgetSections { + instructions: string; + roadmap: string; + plans: PlanEntry[]; + projectMd: string | null; + context: string | null; + research: string | null; + requirements: string | null; +} + +interface BudgetOptions { + safetyMarginPct?: number; + noteTemplate?: string; + projectMdHeadLines?: number; +} + +interface BudgetMetadata { + budget: number; + effectiveBudget: number; + estimatedTokens: number; + omitted: string[]; + projectMdShrunk: boolean; + planTruncationPct: number; + hardFailed: boolean; + noteInjected: boolean; +} + +interface BudgetResult { + prompt: string; + metadata: BudgetMetadata; +} + +function assemblePrompt(parts: { + instructions: string; + note: string | null; + roadmap: string; + projectMd: string | null; + plans: PlanEntry[]; + context: string | null; + research: string | null; + requirements: string | null; +}): string { + const blocks: string[] = []; + + blocks.push(parts.instructions); + + if (parts.note) blocks.push(parts.note); + + blocks.push('## Roadmap\n\n' + parts.roadmap); + + if (parts.projectMd) blocks.push('## Project\n\n' + parts.projectMd); + + const planBlocks = parts.plans + .map((p) => '### ' + p.file + '\n\n' + p.content) + .join('\n\n'); + blocks.push('## Plans\n\n' + planBlocks); + + if (parts.context) blocks.push('## Context\n\n' + parts.context); + if (parts.research) blocks.push('## Research\n\n' + parts.research); + if (parts.requirements) blocks.push('## Requirements\n\n' + parts.requirements); + + return blocks.join('\n\n'); +} + +function applyBudget({ + sections, + budget, + options = {}, +}: { + sections: BudgetSections; + budget: number; + options?: BudgetOptions; +}): BudgetResult { + const { + safetyMarginPct = 10, + noteTemplate = DEFAULT_NOTE_TEMPLATE, + projectMdHeadLines = 40, + } = options; + + const effectiveBudget = Math.floor(budget * (1 - safetyMarginPct / 100)); + + const { + instructions, + roadmap, + plans, + projectMd: projectMdRaw = null, + context: contextRaw = null, + research: researchRaw = null, + requirements: requirementsRaw = null, + } = sections; + + let projectMd = projectMdRaw; + let context = contextRaw; + let research = researchRaw; + let requirements = requirementsRaw; + let workingPlans: PlanEntry[] = plans.map((p) => ({ file: p.file, content: p.content })); + + const omitted: string[] = []; + let projectMdShrunk = false; + let planTruncationPct = 0; + let noteInjected = false; + let hardFailed = false; + + // Minimum-set check: instructions + roadmap + 1KB per plan. + // NOTE_RESERVE_TOKENS is intentionally excluded here: a note is only injected + // when trimming actually occurs, and a prompt that fits without any trim needs + // no note at all. Including NOTE_RESERVE_TOKENS here would cause false hard-fails + // for prompts that genuinely fit the effective budget untrimmed. + const MIN_PLAN_BYTES = 1024; + const minPlanTokens = plans.reduce((sum, p) => { + return sum + estimateTokens(p.content.slice(0, MIN_PLAN_BYTES)); + }, 0); + const minSet = + estimateTokens(instructions) + + estimateTokens(roadmap) + + minPlanTokens; + + if (minSet > effectiveBudget) { + return { + prompt: '', + metadata: { + budget, + effectiveBudget, + estimatedTokens: 0, + omitted: [], + projectMdShrunk: false, + planTruncationPct: 0, + hardFailed: true, + noteInjected: false, + }, + }; + } + + const TOKENS_ROADMAP_HEADER = estimateTokens('## Roadmap\n\n'); + const TOKENS_PROJECT_HEADER = estimateTokens('## Project\n\n'); + const TOKENS_PLANS_HEADER = estimateTokens('## Plans\n\n'); + const TOKENS_CONTEXT_HEADER = estimateTokens('## Context\n\n'); + const TOKENS_RESEARCH_HEADER = estimateTokens('## Research\n\n'); + const TOKENS_REQUIREMENTS_HEADER = estimateTokens('## Requirements\n\n'); + const TOKENS_PLAN_ITEM_HEADERS = workingPlans.reduce( + (sum, p) => sum + estimateTokens('### ' + p.file + '\n\n'), + 0, + ); + + const staticBaseTokens = + estimateTokens(instructions) + + TOKENS_ROADMAP_HEADER + + estimateTokens(roadmap) + + TOKENS_PLANS_HEADER + + TOKENS_PLAN_ITEM_HEADERS; + + let projectTokens = projectMd ? TOKENS_PROJECT_HEADER + estimateTokens(projectMd) : 0; + let contextTokens = context ? TOKENS_CONTEXT_HEADER + estimateTokens(context) : 0; + let researchTokens = research ? TOKENS_RESEARCH_HEADER + estimateTokens(research) : 0; + let requirementsTokens = requirements ? TOKENS_REQUIREMENTS_HEADER + estimateTokens(requirements) : 0; + let planContentTokens = workingPlans.reduce((sum, p) => sum + estimateTokens(p.content), 0); + + const getCurrentBaseTokens = (): number => + staticBaseTokens + + projectTokens + + planContentTokens + + contextTokens + + researchTokens + + requirementsTokens; + + let currentBaseTokens = getCurrentBaseTokens(); + + // Detect budget pressure: is ANY trim needed? + // Pressure exists when the current base tokens already exceed the effective + // budget. Only when pressure is real do we reserve NOTE_RESERVE_TOKENS so + // the note itself fits after trimming. Checking against + // effectiveBudget - NOTE_RESERVE_TOKENS (the old threshold) would cause + // spurious pressure 80 tokens early, dropping sections that fit fine. + const baseTokens = currentBaseTokens; + const budgetUnderPressure = baseTokens > effectiveBudget; + let contentBudget = budgetUnderPressure ? effectiveBudget - NOTE_RESERVE_TOKENS : effectiveBudget; + + // Trim step 1: head-shrink PROJECT.md + if (currentBaseTokens > contentBudget && projectMd) { + const shrunk = headShrink(projectMd, projectMdHeadLines); + if (shrunk !== projectMd) { + projectMd = shrunk; + projectMdShrunk = true; + projectTokens = TOKENS_PROJECT_HEADER + estimateTokens(projectMd); + currentBaseTokens = getCurrentBaseTokens(); + } + } + + // Trim step 2: proportional plan truncation + if (currentBaseTokens > contentBudget) { + const overhead = + staticBaseTokens + + projectTokens + + contextTokens + + researchTokens + + requirementsTokens; + + const planBudgetTokens = contentBudget - overhead; + const totalPlanTokens = planContentTokens; + + if (planBudgetTokens > 0 && planBudgetTokens < totalPlanTokens) { + const totalOriginalChars = plans.reduce((sum, p) => sum + p.content.length, 0); + + workingPlans = workingPlans.map((p) => { + const proportionalShare = + totalOriginalChars > 0 + ? Math.floor((p.content.length / totalOriginalChars) * (planBudgetTokens * 4)) + : 0; + const maxChars = Math.max(proportionalShare, MIN_PLAN_BYTES); + return { file: p.file, content: tailTruncate(p.content, maxChars) }; + }); + + const newTotalChars = workingPlans.reduce((sum, p) => sum + p.content.length, 0); + if (totalOriginalChars > 0) { + planTruncationPct = ((totalOriginalChars - newTotalChars) / totalOriginalChars) * 100; + } + planContentTokens = workingPlans.reduce((sum, p) => sum + estimateTokens(p.content), 0); + currentBaseTokens = getCurrentBaseTokens(); + } + } + + // Trim step 3: drop context + if (currentBaseTokens > contentBudget && context) { + context = null; + omitted.push('context'); + contextTokens = 0; + currentBaseTokens = getCurrentBaseTokens(); + } + + // Trim step 4: drop research + if (currentBaseTokens > contentBudget && research) { + research = null; + omitted.push('research'); + researchTokens = 0; + currentBaseTokens = getCurrentBaseTokens(); + } + + // Trim step 5: drop requirements (last resort) + if (currentBaseTokens > contentBudget && requirements) { + requirements = null; + omitted.push('requirements'); + requirementsTokens = 0; + currentBaseTokens = getCurrentBaseTokens(); + } + + const anyTrimOccurred = omitted.length > 0 || projectMdShrunk || planTruncationPct > 0; + + let note: string | null = null; + if (anyTrimOccurred) { + note = renderNote(noteTemplate, budget, omitted, planTruncationPct); + noteInjected = true; + } + + const prompt = assemblePrompt({ + instructions, + note, + roadmap, + projectMd, + plans: workingPlans, + context, + research, + requirements, + }); + + const estimatedTokens = estimateTokens(prompt); + + if (estimatedTokens > effectiveBudget) { + hardFailed = true; + return { + prompt: '', + metadata: { + budget, + effectiveBudget, + estimatedTokens, + omitted, + projectMdShrunk, + planTruncationPct, + hardFailed, + noteInjected, + }, + }; + } + + return { + prompt, + metadata: { + budget, + effectiveBudget, + estimatedTokens, + omitted, + projectMdShrunk, + planTruncationPct, + hardFailed, + noteInjected, + }, + }; +} + +// ─── CLI arg helpers ────────────────────────────────────────────────────────── + +function getFlag(args: string[], flag: string): string | null { + const idx = args.indexOf(flag); + if (idx === -1) return null; + const val = args[idx + 1]; + if (val === undefined || val.startsWith('--')) return null; + return val; +} + +function getPlanFiles(args: string[]): string[] { + const planFiles: string[] = []; + for (let i = 0; i < args.length; i++) { + if (args[i] === '--plan-file' && args[i + 1] && !args[i + 1].startsWith('--')) { + planFiles.push(args[i + 1]); + i++; + } + } + return planFiles; +} + +async function readRequired(filePath: string, flagName: string): Promise { + const resolved = resolve(filePath); + try { + return await readFile(resolved, 'utf8'); + } catch (err) { + const msg = err && typeof err === 'object' && 'code' in err && err.code === 'ENOENT' + ? `file not found for ${flagName}: ${resolved}` + : `cannot read ${flagName}: ${resolved}`; + throw new GSDError( + msg, + ErrorClassification.Validation, + ); + } +} + +async function readOptional(filePath: string | null): Promise { + if (!filePath) return null; + const resolved = resolve(filePath); + try { + return await readFile(resolved, 'utf8'); + } catch (err) { + if (err && typeof err === 'object' && 'code' in err && err.code === 'ENOENT') { + return null; + } + throw err; + } +} + +// ─── Handler ───────────────────────────────────────────────────────────────── + +/** + * SDK handler for `gsd-sdk query prompt-budget`. + * + * Reads input files, applies the token budget algorithm, writes the trimmed + * prompt and metadata to the specified output files, and returns the metadata. + * + * Throws `GSDError(Validation)` on missing required args or missing files. + * Throws `GSDError(Blocked)` when the minimum-set exceeds the effective budget + * (hard-fail, exit code 11 — callers that previously relied on exit code 2 from + * the CJS fallback path should treat any non-zero exit as "skip reviewer"). + */ +export const promptBudget: QueryHandler = async (args, _projectDir) => { + // Collect multi-value --plan-file flags + const planFilePaths = getPlanFiles(args); + + // Parse single-value flags + const budgetStr = getFlag(args, '--budget'); + const instructionsFile = getFlag(args, '--instructions-file'); + const roadmapFile = getFlag(args, '--roadmap-file'); + const outputPromptFile = getFlag(args, '--output-prompt'); + const outputMetadataFile = getFlag(args, '--output-metadata'); + const safetyMarginStr = getFlag(args, '--safety-margin-pct'); + const projectMdHeadLinesStr = getFlag(args, '--project-md-head-lines'); + const projectFile = getFlag(args, '--project-file'); + const contextFile = getFlag(args, '--context-file'); + const researchFile = getFlag(args, '--research-file'); + const requirementsFile = getFlag(args, '--requirements-file'); + + // Validate required args + if (!budgetStr) { + throw new GSDError('--budget is required', ErrorClassification.Validation); + } + const budget = parseInt(budgetStr, 10); + if (!Number.isFinite(budget) || budget <= 0) { + throw new GSDError('--budget must be a positive integer', ErrorClassification.Validation); + } + if (!instructionsFile) { + throw new GSDError('--instructions-file is required', ErrorClassification.Validation); + } + if (!roadmapFile) { + throw new GSDError('--roadmap-file is required', ErrorClassification.Validation); + } + if (planFilePaths.length === 0) { + throw new GSDError( + 'at least one --plan-file is required', + ErrorClassification.Validation, + ); + } + if (!outputPromptFile) { + throw new GSDError('--output-prompt is required', ErrorClassification.Validation); + } + if (!outputMetadataFile) { + throw new GSDError('--output-metadata is required', ErrorClassification.Validation); + } + + // Read input files + const instructions = await readRequired(instructionsFile, '--instructions-file'); + const roadmap = await readRequired(roadmapFile, '--roadmap-file'); + const plans: PlanEntry[] = await Promise.all(planFilePaths.map(async (p) => { + const resolved = resolve(p); + try { + const content = await readFile(resolved, 'utf8'); + return { file: basename(p), content }; + } catch (err) { + if (err && typeof err === 'object' && 'code' in err && err.code === 'ENOENT') { + throw new GSDError( + `plan file not found: ${resolved}`, + ErrorClassification.Validation, + ); + } + throw new GSDError( + `cannot read plan file: ${resolved}`, + ErrorClassification.Validation, + ); + } + })); + + const projectMd = await readOptional(projectFile); + const context = await readOptional(contextFile); + const research = await readOptional(researchFile); + const requirements = await readOptional(requirementsFile); + + // Build options + const options: BudgetOptions = {}; + if (safetyMarginStr !== null) { + const pct = parseInt(safetyMarginStr, 10); + if (Number.isFinite(pct)) options.safetyMarginPct = pct; + } + if (projectMdHeadLinesStr !== null) { + const lines = parseInt(projectMdHeadLinesStr, 10); + if (Number.isFinite(lines)) options.projectMdHeadLines = lines; + } + + // Apply budget + const sections: BudgetSections = { + instructions, + roadmap, + plans, + projectMd, + context, + research, + requirements, + }; + const { prompt, metadata } = applyBudget({ sections, budget, options }); + + // Write outputs (always write metadata; prompt may be empty on hard-fail) + await writeFile(resolve(outputMetadataFile), JSON.stringify(metadata, null, 2)); + await writeFile(resolve(outputPromptFile), prompt); + + // Signal hard-fail + if (metadata.hardFailed) { + throw new GSDError( + 'prompt-budget hard failed: minimum-set exceeds the effective budget', + ErrorClassification.Blocked, + ); + } + + return { data: metadata }; +}; diff --git a/tests/config-schema-sdk-parity.test.cjs b/tests/config-schema-sdk-parity.test.cjs index 569897717..7eb1919bc 100644 --- a/tests/config-schema-sdk-parity.test.cjs +++ b/tests/config-schema-sdk-parity.test.cjs @@ -136,6 +136,7 @@ test('#2653 — CJS DYNAMIC_KEY_PATTERNS test functions work correctly', () => { ['models.planning', 5], ['dynamic_routing.enabled', 6], ['model_overrides.my-agent', 7], + ['review.max_prompt_tokens_per_reviewer.ollama', 8], ]; for (const [key, idx] of samples) { assert.ok( diff --git a/tests/helpers.cjs b/tests/helpers.cjs index fa93a9dc7..fba603e3f 100644 --- a/tests/helpers.cjs +++ b/tests/helpers.cjs @@ -60,12 +60,13 @@ function runGsdTools(args, cwd = process.cwd(), env = {}) { env: childEnv, }); } - return { success: true, output: result.trim() }; + return { success: true, output: result.trim(), exitCode: 0 }; } catch (err) { return { success: false, output: err.stdout?.toString().trim() || '', error: err.stderr?.toString().trim() || err.message, + exitCode: err.status ?? 1, }; } } diff --git a/tests/prompt-budget-cli.test.cjs b/tests/prompt-budget-cli.test.cjs new file mode 100644 index 000000000..34d64c1bb --- /dev/null +++ b/tests/prompt-budget-cli.test.cjs @@ -0,0 +1,269 @@ +'use strict'; + +// allow-test-rule: prompt-content-is-the-product +// The prompt-budget CLI writes an assembled, trimmed prompt string to disk. +// Testing that the prompt omits a dropped section (research) requires a +// content assertion on the output file — the file content IS the product. +// Structured metadata (omitted[], hardFailed, etc.) is always the primary +// assertion; text content checks are secondary and only used to verify the +// trim policy was applied correctly to the assembled output. + +/** + * prompt-budget-cli.test.cjs + * + * Integration tests for the `gsd-tools prompt-budget` CLI subcommand. + * Covers the 5 specified scenarios: + * 1. Happy path: budget forces trim (research dropped) + * 2. No-trim path: huge budget, metadata shows no omissions + * 3. Hard-fail path: tiny budget, exit 2, metadata written, prompt empty + * 4. Missing required arg: exit 1, stderr has error message + * 5. Missing input file: exit 1 + */ + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const { createTempDir, cleanup, runGsdTools } = require('./helpers.cjs'); + +const TEST_INSTRUCTIONS = [ + '# Cross-AI Plan Review Request', + '', + 'You are reviewing implementation plans for a software project phase.', + 'Provide structured feedback on plan quality, completeness, and risks.', +].join('\n'); + +const TEST_ROADMAP = [ + '## Phase 3: Implement token budgeting', + '', + '### Goal', + 'Add deterministic prompt trimming for small-context local model servers.', +].join('\n'); + +const TEST_PLAN = [ + '## PLAN-01: Add prompt-budget module', + '', + '### Tasks', + '- [ ] Write estimateTokens()', + '- [ ] Write applyBudget()', + '- [ ] Write CLI wrapper', +].join('\n'); + +const TEST_RESEARCH = 'a'.repeat(8000); // ~2000 tokens — large enough to force trimming + +/** + * Run gsd-tools with args as an array (safe for paths with spaces/dollars). + * Returns { exitCode, stdout, stderr }. + */ +function runCli(args, cwd) { + const result = runGsdTools(args, cwd); + return { + exitCode: result.exitCode ?? (result.success ? 0 : 1), + stdout: result.output ?? '', + stderr: result.error ?? '', + }; +} + +describe('prompt-budget CLI', () => { + // ── Cycle 1: happy path — small budget that trims research ──────────────── + test('happy path: trims research when budget is small, exit 0', () => { + const dir = createTempDir('pb-cli-happy-'); + try { + // Write input files + const instrFile = path.join(dir, 'instructions.md'); + const roadmapFile = path.join(dir, 'roadmap.md'); + const planFile = path.join(dir, 'plan-01.md'); + const researchFile = path.join(dir, 'research.md'); + const outPrompt = path.join(dir, 'out-prompt.md'); + const outMeta = path.join(dir, 'out-meta.json'); + + fs.writeFileSync(instrFile, TEST_INSTRUCTIONS); + fs.writeFileSync(roadmapFile, TEST_ROADMAP); + fs.writeFileSync(planFile, TEST_PLAN); + fs.writeFileSync(researchFile, TEST_RESEARCH); + + // Budget of 800 tokens: enough for instructions+roadmap+plan (min set ~389 tokens + // with 10% safety margin) but not the ~2000-token research blob. + // effectiveBudget = floor(800 * 0.9) = 720. + // contentBudget = 720 - 80 (note reserve) = 640. Research (~2000 tokens) won't fit. + const { exitCode, stderr } = runCli([ + 'prompt-budget', + '--budget', '800', + '--instructions-file', instrFile, + '--roadmap-file', roadmapFile, + '--plan-file', planFile, + '--research-file', researchFile, + '--output-prompt', outPrompt, + '--output-metadata', outMeta, + ], dir); + + assert.equal(exitCode, 0, `Expected exit 0, got ${exitCode}. stderr: ${stderr}`); + + // Output files must exist + assert.ok(fs.existsSync(outPrompt), 'output prompt file should exist'); + assert.ok(fs.existsSync(outMeta), 'output metadata file should exist'); + + // Metadata must parse and have expected shape + const meta = JSON.parse(fs.readFileSync(outMeta, 'utf8')); + assert.equal(typeof meta.budget, 'number'); + assert.ok(Array.isArray(meta.omitted)); + assert.ok(meta.omitted.includes('research'), `Expected research in omitted, got: ${JSON.stringify(meta.omitted)}`); + assert.equal(meta.hardFailed, false); + + // Prompt must not contain research content + const promptText = fs.readFileSync(outPrompt, 'utf8'); + assert.ok(promptText.length > 0, 'prompt file must not be empty'); + // Research was dropped so the 'aaa...' block should be absent from the prompt + assert.ok(!promptText.includes('a'.repeat(100)), 'dropped research should not appear in prompt'); + } finally { + cleanup(dir); + } + }); + + // ── Cycle 2: no-trim path — huge budget, nothing dropped ────────────────── + test('no-trim path: huge budget returns all sections, exit 0', () => { + const dir = createTempDir('pb-cli-notrim-'); + try { + const instrFile = path.join(dir, 'instructions.md'); + const roadmapFile = path.join(dir, 'roadmap.md'); + const planFile = path.join(dir, 'plan-01.md'); + const researchFile = path.join(dir, 'research.md'); + const outPrompt = path.join(dir, 'out-prompt.md'); + const outMeta = path.join(dir, 'out-meta.json'); + + fs.writeFileSync(instrFile, TEST_INSTRUCTIONS); + fs.writeFileSync(roadmapFile, TEST_ROADMAP); + fs.writeFileSync(planFile, TEST_PLAN); + fs.writeFileSync(researchFile, 'Some research findings.'); + + const { exitCode } = runCli([ + 'prompt-budget', + '--budget', '1000000', + '--instructions-file', instrFile, + '--roadmap-file', roadmapFile, + '--plan-file', planFile, + '--research-file', researchFile, + '--output-prompt', outPrompt, + '--output-metadata', outMeta, + ], dir); + + assert.equal(exitCode, 0); + + const meta = JSON.parse(fs.readFileSync(outMeta, 'utf8')); + assert.deepEqual(meta.omitted, []); + assert.equal(meta.hardFailed, false); + assert.equal(meta.projectMdShrunk, false); + assert.equal(meta.planTruncationPct, 0); + assert.equal(meta.noteInjected, false); + + // All sections must appear in the prompt + const promptText = fs.readFileSync(outPrompt, 'utf8'); + assert.ok(promptText.includes('Some research findings.')); + } finally { + cleanup(dir); + } + }); + + // ── Cycle 3: hard-fail path — budget is impossibly small ────────────────── + test('hard-fail path: exit 2 when minimum set exceeds budget, metadata written, prompt empty', () => { + const dir = createTempDir('pb-cli-hardfail-'); + try { + const instrFile = path.join(dir, 'instructions.md'); + const roadmapFile = path.join(dir, 'roadmap.md'); + const planFile = path.join(dir, 'plan-01.md'); + const outPrompt = path.join(dir, 'out-prompt.md'); + const outMeta = path.join(dir, 'out-meta.json'); + + // Large instructions to ensure minimum set exceeds the tiny budget + fs.writeFileSync(instrFile, 'a'.repeat(4000)); // ~1000 tokens + fs.writeFileSync(roadmapFile, TEST_ROADMAP); + fs.writeFileSync(planFile, TEST_PLAN); + + // Budget of 5 tokens is far below the minimum set + const { exitCode, stderr } = runCli([ + 'prompt-budget', + '--budget', '5', + '--instructions-file', instrFile, + '--roadmap-file', roadmapFile, + '--plan-file', planFile, + '--output-prompt', outPrompt, + '--output-metadata', outMeta, + ], dir); + + assert.equal(exitCode, 2, `Expected exit 2, got ${exitCode}. stderr: ${stderr}`); + + // Metadata must still be written + assert.ok(fs.existsSync(outMeta), 'metadata file must be written even on hard fail'); + const meta = JSON.parse(fs.readFileSync(outMeta, 'utf8')); + assert.equal(meta.hardFailed, true); + + // Prompt file must be written but empty + assert.ok(fs.existsSync(outPrompt), 'prompt file must exist even on hard fail'); + const promptText = fs.readFileSync(outPrompt, 'utf8'); + assert.equal(promptText, '', 'prompt file must be empty on hard fail'); + } finally { + cleanup(dir); + } + }); + + // ── Cycle 4: missing required arg — exit 1, stderr has error ────────────── + test('missing required arg: exit 1, stderr contains error message', () => { + const dir = createTempDir('pb-cli-missingarg-'); + try { + const instrFile = path.join(dir, 'instructions.md'); + const roadmapFile = path.join(dir, 'roadmap.md'); + const planFile = path.join(dir, 'plan-01.md'); + const outPrompt = path.join(dir, 'out-prompt.md'); + const outMeta = path.join(dir, 'out-meta.json'); + + fs.writeFileSync(instrFile, TEST_INSTRUCTIONS); + fs.writeFileSync(roadmapFile, TEST_ROADMAP); + fs.writeFileSync(planFile, TEST_PLAN); + + // Omit --budget (required) + const { exitCode, stderr } = runCli([ + 'prompt-budget', + '--instructions-file', instrFile, + '--roadmap-file', roadmapFile, + '--plan-file', planFile, + '--output-prompt', outPrompt, + '--output-metadata', outMeta, + ], dir); + + assert.equal(exitCode, 1, `Expected exit 1, got ${exitCode}`); + assert.ok(stderr.includes('--budget'), `Expected stderr to mention --budget, got: ${stderr}`); + } finally { + cleanup(dir); + } + }); + + // ── Cycle 5: missing input file — exit 1 ────────────────────────────────── + test('missing input file: exit 1', () => { + const dir = createTempDir('pb-cli-missingfile-'); + try { + const roadmapFile = path.join(dir, 'roadmap.md'); + const planFile = path.join(dir, 'plan-01.md'); + const outPrompt = path.join(dir, 'out-prompt.md'); + const outMeta = path.join(dir, 'out-meta.json'); + + fs.writeFileSync(roadmapFile, TEST_ROADMAP); + fs.writeFileSync(planFile, TEST_PLAN); + + // Instructions file doesn't exist + const { exitCode } = runCli([ + 'prompt-budget', + '--budget', '10000', + '--instructions-file', path.join(dir, 'nonexistent.md'), + '--roadmap-file', roadmapFile, + '--plan-file', planFile, + '--output-prompt', outPrompt, + '--output-metadata', outMeta, + ], dir); + + assert.equal(exitCode, 1, 'Expected exit 1 when instructions file is missing'); + } finally { + cleanup(dir); + } + }); +}); diff --git a/tests/prompt-budget-sdk-parity-optimizer.test.cjs b/tests/prompt-budget-sdk-parity-optimizer.test.cjs new file mode 100644 index 000000000..f4d70a814 --- /dev/null +++ b/tests/prompt-budget-sdk-parity-optimizer.test.cjs @@ -0,0 +1,44 @@ +'use strict'; + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const SDK_PROMPT_BUDGET = path.join( + __dirname, + '..', + 'sdk', + 'src', + 'query', + 'prompt-budget.ts' +); + +describe('prompt-budget sdk parity (optimizer)', () => { + test('trim order keeps plan truncation ahead of context/research drops', () => { + const src = fs.readFileSync(SDK_PROMPT_BUDGET, 'utf8'); + + const planIdx = src.indexOf('// Trim step 2: proportional plan truncation'); + const contextIdx = src.indexOf('// Trim step 3: drop context'); + const researchIdx = src.indexOf('// Trim step 4: drop research'); + + assert.ok(planIdx > -1, 'expected step 2 proportional plan truncation marker'); + assert.ok(contextIdx > -1, 'expected step 3 context drop marker'); + assert.ok(researchIdx > -1, 'expected step 4 research drop marker'); + assert.ok( + planIdx < contextIdx && contextIdx < researchIdx, + 'expected trim order: plan truncation -> context drop -> research drop' + ); + }); + + test('final over-budget hard-fail guard is present', () => { + const src = fs.readFileSync(SDK_PROMPT_BUDGET, 'utf8'); + + assert.match( + src, + /if\s*\(\s*estimatedTokens\s*>\s*effectiveBudget\s*\)/, + 'expected final estimatedTokens > effectiveBudget guard' + ); + }); +}); + diff --git a/tests/prompt-budget.test.cjs b/tests/prompt-budget.test.cjs new file mode 100644 index 000000000..2835c5494 --- /dev/null +++ b/tests/prompt-budget.test.cjs @@ -0,0 +1,342 @@ +'use strict'; + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); + +const { estimateTokens, applyBudget } = require('../get-shit-done/bin/lib/prompt-budget.cjs'); + +describe('prompt-budget', () => { + // ── Cycle 1: estimator basics ────────────────────────────────────────────── + test('estimateTokens — basic contract', () => { + assert.equal(estimateTokens(''), 0); + assert.equal(estimateTokens('hello'), 2); // ceil(5/4) = 2 + assert.equal(estimateTokens('a'.repeat(400)), 100); // 400/4 = 100 exactly + }); + + // ── Cycle 2: applyBudget no-trim path ───────────────────────────────────── + test('applyBudget — no-trim path when well under budget', () => { + + const sections = { + instructions: 'Review this code.', + projectMd: 'Project info.', + roadmap: 'Phase 1: ship it.', + plans: [{ file: 'plan-a.md', content: 'Do the thing.' }], + context: 'Session context.', + research: 'Background research.', + requirements: 'Must pass tests.', + }; + const { prompt, metadata } = applyBudget({ sections, budget: 10000 }); + + assert.deepEqual(metadata.omitted, []); + assert.equal(metadata.hardFailed, false); + assert.equal(metadata.noteInjected, false); + assert.equal(metadata.projectMdShrunk, false); + assert.equal(metadata.planTruncationPct, 0); + + assert.ok(prompt.includes('Review this code.')); + assert.ok(prompt.includes('Project info.')); + assert.ok(prompt.includes('Phase 1: ship it.')); + assert.ok(prompt.includes('Do the thing.')); + assert.ok(prompt.includes('Session context.')); + assert.ok(prompt.includes('Background research.')); + assert.ok(prompt.includes('Must pass tests.')); + }); + + // ── Cycle 3: drop research when it is the sole cause of over-budget ───────── + test('applyBudget — drops research when it alone causes over-budget', () => { + // context is null; research alone pushes us over budget. + // Without research: base ≈ 20 tokens. With research (~500 tokens): over budget. + // effectiveBudget = floor(200 * 0.9) = 180. contentBudget = 180 - 80 = 100. + const researchContent = 'a'.repeat(2000); // ~500 tokens + + const sections = { + instructions: 'Review this code.', + projectMd: null, + roadmap: 'Phase 1: ship it.', + plans: [{ file: 'plan-a.md', content: 'Do the thing.' }], + context: null, + research: researchContent, + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 200 }); + + assert.deepEqual(metadata.omitted, ['research']); + assert.equal(metadata.noteInjected, true); + assert.equal(prompt.includes(researchContent), false); + }); + + // ── Cycle 4: drop context first, then research ──────────────────────────── + test('applyBudget — drops context then research when both needed', () => { + // Both context and research present; both must be dropped. + // context drops first (priority 6), research second (priority 7). + // base without either ≈ 20 tokens; each of context+research adds ~500 tokens. + // effectiveBudget = floor(200 * 0.9) = 180. contentBudget = 100. + const bigContent = 'b'.repeat(2000); // ~500 tokens each + + const sections = { + instructions: 'Review this code.', + projectMd: null, + roadmap: 'Phase 1: ship it.', + plans: [{ file: 'plan-a.md', content: 'Do the thing.' }], + context: bigContent, + research: bigContent, + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 200 }); + + // context drops first per spec (priority 6), research second (priority 7) + assert.deepEqual(metadata.omitted, ['context', 'research']); + assert.equal(metadata.noteInjected, true); + }); + + // ── Cycle 5: head-shrink PROJECT.md before dropping context ────────────── + test('applyBudget — shrinks PROJECT.md to 40 lines before dropping context', () => { + // PROJECT.md is 80 lines; shrinking to 40 saves enough that context survives. + // 80 lines of ~20 chars each = 1600 chars ≈ 400 tokens + // 40 lines = 800 chars ≈ 200 tokens (saves ~200 tokens) + // effectiveBudget = floor(700 * 0.9) = 630. contentBudget = 630 - 80 = 550. + // base with 80-line projectMd + short context ≈ 20 + 400 + short ≈ 440 → fits at 550 after shrink to 40 + // base with full 80-line projectMd ≈ 20 + 400 + 10 = 430... need it to be over 550 with full projectMd + + // Build 80-line projectMd where shrinking to 40 saves enough + const lineOf20 = 'x'.repeat(19); // 19 chars + newline = 20 chars per line + const projectMdFull = Array.from({ length: 80 }, () => lineOf20).join('\n'); + // full: 80*20 - 1 = 1599 chars ≈ 400 tokens; shrunk: 40*20 - 1 = 799 chars ≈ 200 tokens + + // Budget: effective = 400, contentBudget = 320 + // base with full projectMd + short context ≈ overhead(20) + projectMd(400) + context(5) ≈ 425 > 320 → pressure + // After shrink: overhead(20) + projectMd_shrunk(200) + context(5) ≈ 225 < 320 → fits, no drops + const sections = { + instructions: 'Review this.', + projectMd: projectMdFull, + roadmap: 'Phase 1.', + plans: [{ file: 'plan.md', content: 'Plan here.' }], + context: 'Short context.', + research: null, + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 445 }); + + assert.equal(metadata.projectMdShrunk, true); + assert.deepEqual(metadata.omitted, []); + assert.equal(metadata.noteInjected, true); + assert.ok(prompt.includes('Short context.')); + }); + + // ── Cycle 6: proportional plan truncation when 2+ plans ────────────────── + test('applyBudget — proportionally truncates plans, never drops a whole plan', () => { + // Two plans of ~1000 tokens each (4000 chars each). + // Budget chosen so plans must shrink ~30%. + // effectiveBudget = floor(1600 * 0.9) = 1440. contentBudget = 1440 - 80 = 1360. + // overhead ≈ 30 tokens. planBudget ≈ 1330 tokens ≈ 5320 chars. + // original total plan chars = 8000. remaining = 5320. reduction = (8000-5320)/8000 ≈ 33.5% + const planContent = 'p'.repeat(4000); // ~1000 tokens each + + const sections = { + instructions: 'Review this.', + projectMd: null, + roadmap: 'Phase 1.', + plans: [ + { file: 'plan-a.md', content: planContent }, + { file: 'plan-b.md', content: planContent }, + ], + context: null, + research: null, + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 1600 }); + + // Both plans still appear + assert.ok(prompt.includes('### plan-a.md')); + assert.ok(prompt.includes('### plan-b.md')); + + // planTruncationPct within ±10 of 30% + assert.ok( + metadata.planTruncationPct >= 20 && metadata.planTruncationPct <= 40, + 'planTruncationPct=' + metadata.planTruncationPct + ' expected ~30 (±10)' + ); + assert.equal(metadata.noteInjected, true); + }); + + // ── Cycle 7: drop requirements only as last resort ──────────────────────── + test('applyBudget — drops requirements as last resort after all other trims', () => { + // context + research + plan-truncation + project-shrink all applied but still over + // → drop requirements. + // Use a tight budget with all optional sections present and large. + // effectiveBudget = floor(200 * 0.9) = 180. contentBudget = 100. + // Each big section = 'x'.repeat(2000) ≈ 500 tokens. + // base (no optionals) ≈ 20 tokens. With requirements: ~520 tokens. + // After dropping context + research: still need requirements dropped. + const bigContent = 'x'.repeat(2000); + + const sections = { + instructions: 'Review.', + projectMd: null, + roadmap: 'Phase 1.', + plans: [{ file: 'plan.md', content: 'Plan here.' }], + context: bigContent, + research: bigContent, + requirements: bigContent, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 200 }); + + assert.deepEqual(metadata.omitted, ['context', 'research', 'requirements']); + assert.equal(metadata.noteInjected, true); + assert.equal(metadata.hardFailed, false); + }); + + // ── Cycle 8: hard-fail when minimum-set exceeds budget ─────────────────── + test('applyBudget — hard-fails when minimum-set exceeds effective budget', () => { + // budget=50, effectiveBudget=45. instructions alone = 200+ tokens. + // minSet = instructions + NOTE_RESERVE(80) + roadmap + min-plan = way over 45. + const bigInstructions = 'i'.repeat(800); // 200 tokens + + const sections = { + instructions: bigInstructions, + projectMd: null, + roadmap: 'Phase 1.', + plans: [{ file: 'plan.md', content: 'Plan here.' }], + context: null, + research: null, + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 50 }); + + assert.equal(metadata.hardFailed, true); + assert.equal(prompt, ''); + }); + + // ── Cycle 9: note reservation invariant (load-bearing) ─────────────────── + test('applyBudget — reserves note tokens before drop math so final prompt fits', () => { + // Invariant: the algorithm reserves ~80 note tokens BEFORE deciding what to drop. + // If the implementation skips reservation, a buggy version could: + // 1. See base_with_research >> effectiveBudget → drop research + // 2. See base_without_research < effectiveBudget → think "no more trim needed" + // 3. Inject note anyway → final prompt exceeds effectiveBudget + // + // Design (budget=145, effectiveBudget=130, contentBudget=50): + // base_without_research = 60 tokens → in (contentBudget=50, effectiveBudget=130) + // base_with_research = 564 tokens → triggers pressure + // base_without_research + note_reserve(80) = 140 > effectiveBudget(130) → must reserve + // minSet(130) ≤ effectiveBudget(130) → just barely avoids hard fail + // + // A correct implementation drops research and ensures estimatedTokens ≤ effectiveBudget. + const sections = { + instructions: 'i'.repeat(120), // 30 tokens + projectMd: null, + roadmap: 'r'.repeat(40), // 10 tokens + plans: [{ file: 'plan.md', content: 'p'.repeat(40) }], // 10 tokens + context: null, + research: 'x'.repeat(2000), // ~500 tokens — causes the pressure + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 145 }); + + // Research must be dropped (it caused the pressure) + assert.ok(metadata.omitted.includes('research'), 'research must be omitted'); + // Note must be injected (trim occurred) + assert.equal(metadata.noteInjected, true); + // Final estimated tokens must be within effectiveBudget (the load-bearing assertion) + assert.ok( + metadata.estimatedTokens <= metadata.effectiveBudget, + 'estimatedTokens=' + metadata.estimatedTokens + + ' must be ≤ effectiveBudget=' + metadata.effectiveBudget + ); + // Not a hard failure + assert.equal(metadata.hardFailed, false); + }); + + // ── Cycle 11: no false hard-fail when untrimmed prompt fits effectiveBudget ─ + test('applyBudget — does not hard-fail when untrimmed prompt fits within effectiveBudget', () => { + // budget=44 → effectiveBudget=39. Full untrimmed prompt = 32 tokens ≤ 39. + // Bug: minSet unconditionally includes NOTE_RESERVE_TOKENS(80), making + // minSet=102 > 39 and triggering a spurious hard-fail even though no note + // is needed (no trim occurs) and the prompt genuinely fits. + // Fix: exclude NOTE_RESERVE_TOKENS from minSet; only account for it if trim + // is actually needed. + const sections = { + instructions: 'i'.repeat(32), // 8 tokens + projectMd: null, + roadmap: 'r'.repeat(16), // 4 tokens + plans: [{ file: 'plan.md', content: 'p'.repeat(40) }], // 10 tokens + context: null, + research: null, + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 44 }); + + assert.equal(metadata.hardFailed, false, 'must not hard-fail when prompt fits effectiveBudget'); + assert.ok(prompt.length > 0, 'prompt must be non-empty'); + assert.deepEqual(metadata.omitted, [], 'nothing should be omitted'); + assert.equal(metadata.noteInjected, false, 'no note needed — no trim occurred'); + }); + + // ── Cycle 12: no unneeded trim when full prompt already fits effectiveBudget ─ + test('applyBudget — does not drop context or research when full untrimmed prompt already fits effectiveBudget', () => { + // budget=156 → effectiveBudget=140. Full prompt (with context+research) ≈ 88 tokens ≤ 140. + // Bug: budgetUnderPressure = baseTokens > effectiveBudget - NOTE_RESERVE_TOKENS + // = 89 > 60 → true, sets contentBudget=60, triggers trim steps → drops context/research. + // Fix: budgetUnderPressure should check baseTokens > effectiveBudget, not the pre-reserved + // threshold. Note reservation happens only after a real trim decision is made. + const sections = { + instructions: 'i'.repeat(120), // 30 tokens + projectMd: null, + roadmap: 'r'.repeat(40), // 10 tokens + plans: [{ file: 'plan-a.md', content: 'p'.repeat(40) }], // 10 tokens + context: 'c'.repeat(40), // 10 tokens + header + research: 'x'.repeat(40), // 10 tokens + header + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 156 }); + + assert.equal(metadata.hardFailed, false, 'must not hard-fail'); + assert.deepEqual(metadata.omitted, [], 'context and research must NOT be omitted'); + assert.ok(prompt.length > 0, 'prompt must be non-empty'); + // context and research must appear in the assembled prompt + assert.ok(prompt.includes('## Context'), 'context section must be present'); + assert.ok(prompt.includes('## Research'), 'research section must be present'); + assert.equal(metadata.noteInjected, false, 'no note needed — no trim occurred'); + }); + + // ── Cycle 10: null optional sections ───────────────────────────────────── + test('applyBudget — null optional sections are excluded from prompt without counting as omitted', () => { + // All optionals are null, no projectMd — big budget so no trim. + const sections = { + instructions: 'Review this code.', + projectMd: null, + roadmap: 'Phase 1: ship it.', + plans: [{ file: 'plan.md', content: 'Do the thing.' }], + context: null, + research: null, + requirements: null, + }; + + const { prompt, metadata } = applyBudget({ sections, budget: 10000 }); + + // Nulls don't count as "omitted" — only sections with content that were trimmed do + assert.deepEqual(metadata.omitted, []); + assert.equal(metadata.hardFailed, false); + assert.equal(metadata.noteInjected, false); + + // None of the optional section headers should appear + assert.equal(prompt.includes('## Context'), false); + assert.equal(prompt.includes('## Research'), false); + assert.equal(prompt.includes('## Requirements'), false); + assert.equal(prompt.includes('## Project'), false); + + // Required sections must still appear + assert.ok(prompt.includes('Review this code.')); + assert.ok(prompt.includes('Phase 1: ship it.')); + assert.ok(prompt.includes('Do the thing.')); + }); + +});