test(#1977): consolidate 22 misc + repo-invariant regression tests

Final epic-#1969 batch. Fold 22 issue-named files: the 4 genuine repo-wide invariant
scans (551-eslint-bin-lib-coverage, bug-3054 stale /gsd-next, bug-3810 no-gsd-sdk-runtime-refs,
feat-3593 cli-negative-universal) into a NEW shared repo-invariants.test.cjs; the other 18 as
singletons into their nearest module suite (model-resolver, codex-config, runtime-converters,
security, state-transition, worktree-safety, roadmap-parser, etc.). Verbatim block-scoped
describe wrappers; 334 subtests conserved 1:1.

Host-env pre-check (B2+B6): the 6 CLI folds into GSD_TEST_MODE-setting hosts (model-resolver/
codex-config/runtime-converters) are benign — each origin independently sets GSD_TEST_MODE=1
itself (idempotent), unlike the B6 real-install case.

Regenerates regression-name allowlist (222->213), ratchets file-count allowlist (state 17->16),
makes 7 relocated allow-test-rule exemptions issue-ref-compliant (ADR-456; prunes stale ids).
Repoints 2 tests/ refs in docs/TESTING-SUITES.md. lint:ci green.

Part of epic #1969. Closes #1977.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Tom Boucher
2026-07-03 03:33:07 -04:00
parent 4c77b46e3d
commit 697cbb1f05
41 changed files with 5972 additions and 5767 deletions

View File

@@ -221,8 +221,8 @@ node scripts/ci-test-scope.cjs --base origin/next --head HEAD
## Test strategy: #443 effort + fast_mode engine
> Feature: unified cross-provider effort and fast_mode knobs (issue #443).
> Test files: `tests/feat-443-effort-fast-mode.test.cjs` (unit),
> `tests/feat-443-effort-fast-mode.integration.test.cjs` (integration).
> Test files: `tests/model-resolver.test.cjs` (unit),
> `tests/model-resolver.test.cjs` (integration).
### Testing pyramid

View File

@@ -18,12 +18,8 @@
"tests/bug-211-launcher-home-fallback.test.cjs :: structural/behavioral regression for the ~/.claude fallback arm in",
"tests/bug-2136-sh-hook-version.test.cjs :: structural-regression-guard",
"tests/bug-2543-gsd-slash-namespace.test.cjs :: structural-regression-guard",
"tests/bug-2559-stale-search-year.test.cjs :: source-text-is-the-product",
"tests/bug-2772-gitmodules-path-intersection.test.cjs :: source-text-is-the-product",
"tests/bug-2808-skill-hyphen-name.test.cjs :: source-text-is-the-product",
"tests/bug-2839-review-fix-transactional-cleanup.test.cjs :: source-text-is-the-product",
"tests/bug-33-settings-model-profile-adaptive.test.cjs :: source-text-is-the-product",
"tests/bug-3384-secondary-defects.test.cjs :: source-text-is-the-product",
"tests/bug-3446-resume-continue-here-discovery.test.cjs :: source-text-is-the-product",
"tests/bug-3491-nested-git-worktree.test.cjs :: source-text-is-the-product",
"tests/bug-3523-cjs-loadconfig-branching-strategy-warning.test.cjs :: validates runtime CLI stdout/stderr warning behavior, not source grep",
@@ -33,14 +29,12 @@
"tests/bug-3683-command-colon-namespace-leak.test.cjs :: source-text-is-the-product",
"tests/bug-3683-workflow-colon-namespace-leak.test.cjs :: source-text-is-the-product",
"tests/bug-3689-resume-glob-nomatch.test.cjs :: source-text-is-the-product",
"tests/bug-3810-no-gsd-sdk-runtime-refs.test.cjs :: source-text-is-the-product",
"tests/bug-444-resolver-local-claude-install.test.cjs :: structural/behavioral regression for the repo-local .claude/ install",
"tests/bug-619-codebase-drift-gate-shim.test.cjs :: source-text-is-the-product",
"tests/bug-622-graphify-optional-graph-html.test.cjs :: source-text-is-the-product",
"tests/bug-630-wave-cleanup-orchestrator-root.test.cjs :: source-text-is-the-product",
"tests/bug-685-windowshide-spawn.test.cjs :: source-text-is-the-product",
"tests/bug-891-non-claude-runtime-home-fallback.test.cjs :: structural/behavioral regression for non-Claude runtime-home",
"tests/bug-patterns-reference.test.cjs :: source-text-is-the-product",
"tests/chain-flag-plan-phase.test.cjs :: source-text-is-the-product",
"tests/changeset-cli.test.cjs :: reads a product workflow .md file (not CJS source) to verify",
"tests/check-update-config-dir.test.cjs :: structural-regression-guard",
@@ -82,7 +76,6 @@
"tests/enh-2789-description-budget.test.cjs :: source-text-is-the-product",
"tests/enh-2790-skill-consolidation.test.cjs :: source-text-is-the-product",
"tests/enh-48-cwd-drift-guard-e2e.test.cjs :: integration-test-input",
"tests/enh-72-business-context.test.cjs :: source-text-is-the-product",
"tests/eslint-rules.test.cjs :: <source-grep reason> must still error",
"tests/eslint-rules.test.cjs :: pending migration",
"tests/eslint-rules.test.cjs :: source-text-is-the-product",
@@ -95,7 +88,6 @@
"tests/execute-phase-worktree-artifacts.test.cjs :: source-text-is-the-product",
"tests/explore-command.test.cjs :: source-text-is-the-product",
"tests/extract-learnings.test.cjs :: source-text-is-the-product",
"tests/feat-2840-issue-driven-orchestration-guide.test.cjs :: structural-IR parser for a docs guide. The .includes()",
"tests/feat-3039-help-tiered.test.cjs :: source-text-is-the-product",
"tests/forensics.test.cjs :: source-text-is-the-product",
"tests/frontmatter-cli.test.cjs :: source-text-is-the-product",

View File

@@ -3,27 +3,19 @@
"bug-1367-claude-local-flat-command-layout.test.cjs",
"bug-1834-sh-hooks-installed.test.cjs",
"bug-1974-context-exhaustion-record.test.cjs",
"bug-21-state-md-template-frontmatter.test.cjs",
"bug-211-launcher-home-fallback.test.cjs",
"bug-2136-sh-hook-version.test.cjs",
"bug-2344-read-guard-claudecode-env.test.cjs",
"bug-2451-context-monitor-over-report.test.cjs",
"bug-2520-read-guard-hook-subprocess-env.test.cjs",
"bug-2543-gsd-slash-namespace.test.cjs",
"bug-2559-stale-search-year.test.cjs",
"bug-260-worktree-path-guard.test.cjs",
"bug-261-worktree-force-add-guard.test.cjs",
"bug-2772-gitmodules-path-intersection.test.cjs",
"bug-2808-skill-hyphen-name.test.cjs",
"bug-2839-review-fix-transactional-cleanup.test.cjs",
"bug-2866-codex-strip-no-trailing-newline.test.cjs",
"bug-2876-skill-frontmatter-quote.test.cjs",
"bug-2916-handle-branching-default-base.test.cjs",
"bug-2995-post-install-script-paths.test.cjs",
"bug-3019-help-passthrough.test.cjs",
"bug-3054-stale-gsd-next-references.test.cjs",
"bug-33-settings-model-profile-adaptive.test.cjs",
"bug-3384-secondary-defects.test.cjs",
"bug-3442-shim-projection-drift-guard.test.cjs",
"bug-3446-resume-continue-here-discovery.test.cjs",
"bug-3491-nested-git-worktree.test.cjs",
@@ -37,7 +29,6 @@
"bug-3683-command-colon-namespace-leak.test.cjs",
"bug-3683-workflow-colon-namespace-leak.test.cjs",
"bug-3689-resume-glob-nomatch.test.cjs",
"bug-3810-no-gsd-sdk-runtime-refs.test.cjs",
"bug-444-resolver-local-claude-install.test.cjs",
"bug-619-codebase-drift-gate-shim.test.cjs",
"bug-622-graphify-optional-graph-html.test.cjs",

View File

@@ -58,7 +58,6 @@
},
"state": {
"files": [
"bug-21-state-md-template-frontmatter.test.cjs",
"state-acquirestatelock-non-eexist.test.cjs",
"state-prune.test.cjs",
"state-rebuild-cli.test.cjs",

View File

@@ -1,102 +0,0 @@
'use strict';
/**
* Regression / migration-gate test for #551 and ADR-457 (TS migration).
*
* ESLint must apply the correct policy to every gsd-core/bin/lib/*.cjs
* file as modules migrate from hand-written CJS to tsc-generated artifacts:
*
* - tsc-generated artifact (has src/<basename>.cts counterpart) → MUST be
* eslint-ignored. We lint the *.cts source instead (ADR-457).
* - Genuinely hand-written (no src/*.cts counterpart) → MUST be linted
* (NOT ignored). Includes scripts-generated package-identity.cjs which
* has no *.cts source.
*
* The test is filesystem-driven — it scans bin/lib at runtime and checks each
* file against the src/ directory, so it stays correct automatically as more
* modules migrate. No hardcoded lists.
*
* ESLint behaviour is verified via ESLint's own `isPathIgnored()` API so the
* test reflects real resolved flat-config precedence, not a textual scan of
* eslint.config.mjs.
*/
const { describe, test, before } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { ESLint } = require('eslint');
const ROOT = path.resolve(__dirname, '..');
const LIB_DIR = path.join(ROOT, 'gsd-core', 'bin', 'lib');
const SRC_DIR = path.join(ROOT, 'src');
/**
* Returns true if the given bin/lib/*.cjs file has a corresponding
* src/<basename>.cts TypeScript source (meaning it is tsc-generated).
*/
function hasTsSource(absPath) {
const base = path.basename(absPath, '.cjs');
return (
fs.existsSync(path.join(SRC_DIR, `${base}.cts`)) ||
fs.existsSync(path.join(SRC_DIR, `${base}.ts`))
);
}
let eslint;
before(() => {
eslint = new ESLint({ cwd: ROOT });
});
describe('ESLint coverage tracks the bin/lib TS migration (ADR-457 / #537)', () => {
/**
* Main invariant: scan every *.cjs in bin/lib and assert the correct ESLint
* policy is applied.
*/
test('each bin/lib/*.cjs is linted xor ignored according to migration state', async () => {
const wronglyIgnored = []; // hand-written but ignored — should be linted
const wronglyLinted = []; // tsc-generated but not ignored — should be ignored
const entries = fs.readdirSync(LIB_DIR).filter((e) => e.endsWith('.cjs'));
for (const entry of entries) {
const abs = path.join(LIB_DIR, entry);
const generated = hasTsSource(abs);
const ignored = await eslint.isPathIgnored(abs);
if (generated && !ignored) {
wronglyLinted.push(entry);
} else if (!generated && ignored) {
wronglyIgnored.push(entry);
}
}
assert.deepEqual(
wronglyLinted,
[],
`tsc-generated bin/lib modules not yet added to ESLint ignore list: ${wronglyLinted.join(', ')}`,
);
assert.deepEqual(
wronglyIgnored,
[],
`Hand-written bin/lib modules silently excluded from ESLint: ${wronglyIgnored.join(', ')}`,
);
});
test('semver-compare.cjs (tsc-generated publish artifact) stays eslint-ignored (ADR-457)', async () => {
const f = path.join(LIB_DIR, 'semver-compare.cjs');
assert.equal(
await eslint.isPathIgnored(f),
true,
'semver-compare.cjs is a tsc-generated publish-time artifact and must stay ignored',
);
});
test('package-identity.cjs (script-generated, no *.cts source) is linted, not ignored (#551)', async () => {
const f = path.join(LIB_DIR, 'package-identity.cjs');
assert.equal(
await eslint.isPathIgnored(f),
false,
'package-identity.cjs has no src/*.cts counterpart and must be linted, not ignored',
);
});
});

View File

@@ -82,3 +82,128 @@ describe('READING: agents with reading blocks use <required_reading>', () => {
});
}
});
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-patterns-reference.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-patterns-reference (consolidation epic #1969 B8 #1977)", () => {
// allow-test-rule: source-text-is-the-product
// Workflow .md / agent .md / command .md / reference .md files — their text
// IS what the runtime loads. Testing text content tests the deployed contract.
// Per CONTRIBUTING.md exception matrix.
/**
* Common Bug Patterns Reference Tests
*
* Structural tests for the common-bug-patterns.md reference file:
* - File exists at expected path
* - Contains expected bug pattern categories (at least 5 of 10)
* - Debugger agent references the file in required_reading
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const REFERENCE_PATH = path.join(
__dirname, '..', 'gsd-core', 'references', 'common-bug-patterns.md'
);
const DEBUGGER_AGENT_PATH = path.join(
__dirname, '..', 'agents', 'gsd-debugger.md'
);
const EXPECTED_CATEGORIES = [
'Off-by-One',
'Null',
'Async',
'State Management',
'Import',
'Environment',
'Data Shape',
'String Handling',
'File System',
'Error Handling',
];
describe('common-bug-patterns.md reference', () => {
test('reference file exists', () => {
assert.ok(
fs.existsSync(REFERENCE_PATH),
`Expected reference file at ${REFERENCE_PATH}`
);
});
test('has title and intro', () => {
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
assert.ok(
content.startsWith('# Common Bug Patterns'),
'File should start with "# Common Bug Patterns" title'
);
assert.ok(
content.includes('---'),
'File should contain --- separator after intro'
);
});
test('contains at least 5 of 10 expected categories', () => {
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
const found = EXPECTED_CATEGORIES.filter(cat =>
content.toLowerCase().includes(cat.toLowerCase())
);
assert.ok(
found.length >= 5,
`Expected at least 5 categories, found ${found.length}: ${found.join(', ')}`
);
});
test('each pattern category has at least one bold bullet item', () => {
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
// Only check sections inside <patterns> block, not <usage>
const patternsBlock = (content.split('<patterns>')[1] || '').split('</patterns>')[0];
const sections = patternsBlock.split(/^## /m).slice(1);
assert.ok(sections.length >= 5, `Expected at least 5 pattern sections, got ${sections.length}`);
for (const section of sections) {
const title = section.split('\n')[0].trim();
const bullets = section.match(/^- \*\*/gm);
assert.ok(
bullets && bullets.length >= 1,
`Pattern section "${title}" should have at least one "- **" bullet item`
);
}
});
});
describe('debugger agent references bug patterns', () => {
test('gsd-debugger.md exists', () => {
assert.ok(
fs.existsSync(DEBUGGER_AGENT_PATH),
`Expected debugger agent at ${DEBUGGER_AGENT_PATH}`
);
});
test('gsd-debugger.md references common-bug-patterns.md', () => {
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
assert.ok(
content.includes('common-bug-patterns.md'),
'Debugger agent should reference common-bug-patterns.md'
);
});
test('reference is inside <required_reading> block', () => {
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
const reqReadMatch = content.match(
/<required_reading>([\s\S]*?)<\/required_reading>/
);
assert.ok(reqReadMatch, 'Debugger agent should have a <required_reading> block');
assert.ok(
reqReadMatch[1].includes('common-bug-patterns.md'),
'common-bug-patterns.md should be inside <required_reading> block'
);
});
});
});
}

View File

@@ -1,187 +0,0 @@
/**
* Regression guard — Bug #21
*
* Both STATE.md template files must include a YAML frontmatter block in their
* "File Template" section so that an AI agent creating .planning/STATE.md from
* the template produces a file that frontmatter consumers can read immediately
* (before the first `state.*` mutation calls syncStateFrontmatter).
*
* Prior to the fix, the template's File Template section began with
* `# Project State` (no frontmatter), leaving the init→first-write window
* without `gsd_state_version`, `status`, or `progress` keys.
*
* Acceptance criteria:
* 1. The template body extracted from each state.md file's File Template code
* block must begin with `---`.
* 2. The frontmatter must contain at minimum: `gsd_state_version` and `status`.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const REPO_ROOT = path.join(__dirname, '..');
const TEMPLATE_PATHS = [
path.join(REPO_ROOT, 'gsd-core', 'templates', 'state.md'),
];
/**
* Extract the content of the first ```markdown ... ``` code block from a
* template file. Returns the raw string (including any leading/trailing
* whitespace within the block).
*
* @param {string} fileContent - Full text of the template file.
* @returns {string} The extracted code block body.
*/
function extractFileTemplate(fileContent) {
const match = fileContent.match(/```markdown\r?\n([\s\S]*?)```/);
assert.ok(match, 'No ```markdown code block found in template file');
return match[1];
}
/**
* Minimal YAML frontmatter parser: returns the set of top-level keys present
* in the first --- ... --- block at the start of `text`. Does not parse nested
* keys — list-valued fields (e.g. `tags: [a, b]`) are recorded only by their
* key name, not their value. Returns an empty Set when the text has no frontmatter.
*
* @param {string} text
* @returns {Set<string>}
*/
function parseFrontmatterKeys(text) {
const keys = new Set();
if (!text.trimStart().startsWith('---')) return keys;
const lines = text.split(/\r?\n/);
let inBlock = false;
for (const line of lines) {
const trimmed = line.trim();
if (!inBlock) {
if (trimmed === '---') { inBlock = true; continue; }
break; // frontmatter must be at the very start
}
if (trimmed === '---') break; // end of block
const colonIdx = trimmed.indexOf(':');
if (colonIdx > 0) {
keys.add(trimmed.slice(0, colonIdx).trim());
}
}
return keys;
}
/**
* Minimal YAML frontmatter parser: returns a plain object of top-level keys
* and their scalar or nested-object values from the first --- ... --- block.
* Handles one level of indented nesting (e.g. progress.total_plans).
* Does not handle YAML lists or multi-line values.
*
* @param {string} text
* @returns {Record<string, any>}
*/
function parseFrontmatter(text) {
const result = {};
if (!text.trimStart().startsWith('---')) return result;
const lines = text.split(/\r?\n/);
let inBlock = false;
let currentKey = null;
for (const line of lines) {
const trimmed = line.trim();
if (!inBlock) {
if (trimmed === '---') { inBlock = true; continue; }
break;
}
if (trimmed === '---') break;
// Detect indented (nested) line: starts with whitespace
if (line.match(/^\s+\S/) && currentKey !== null) {
const colonIdx = trimmed.indexOf(':');
if (colonIdx > 0) {
const subKey = trimmed.slice(0, colonIdx).trim();
const rawVal = trimmed.slice(colonIdx + 1).trim();
const numVal = Number(rawVal);
if (typeof result[currentKey] !== 'object') result[currentKey] = {};
result[currentKey][subKey] = rawVal === '' ? null : (isNaN(numVal) ? rawVal : numVal);
}
} else {
currentKey = null;
const colonIdx = trimmed.indexOf(':');
if (colonIdx > 0) {
const key = trimmed.slice(0, colonIdx).trim();
const rawVal = trimmed.slice(colonIdx + 1).trim();
if (rawVal === '') {
result[key] = {};
currentKey = key;
} else {
const numVal = Number(rawVal);
result[key] = isNaN(numVal) ? rawVal.replace(/^'|'$/g, '') : numVal;
currentKey = null;
}
}
}
}
return result;
}
describe('bug #21 — STATE.md template must carry YAML frontmatter', () => {
for (const templatePath of TEMPLATE_PATHS) {
const label = path.relative(REPO_ROOT, templatePath);
test(`${label} — File Template block starts with frontmatter`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
// The template body must open with a YAML frontmatter delimiter.
assert.ok(
body.trimStart().startsWith('---'),
`${label}: File Template must start with '---' (YAML frontmatter), ` +
`but starts with: ${JSON.stringify(body.slice(0, 60))}`,
);
});
test(`${label} — frontmatter contains gsd_state_version`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
const keys = parseFrontmatterKeys(body.trimStart());
assert.ok(
keys.has('gsd_state_version'),
`${label}: frontmatter must include 'gsd_state_version', found keys: ${[...keys].join(', ')}`,
);
});
test(`${label} — frontmatter contains status`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
const keys = parseFrontmatterKeys(body.trimStart());
assert.ok(
keys.has('status'),
`${label}: frontmatter must include 'status', found keys: ${[...keys].join(', ')}`,
);
});
test(`${label} — progress sub-schema has zeroed total_plans and completed_plans`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
const fm = parseFrontmatter(body.trimStart());
assert.ok(
fm.progress && typeof fm.progress === 'object',
`${label}: frontmatter must include a 'progress' sub-object`,
);
assert.strictEqual(
fm.progress.total_plans,
0,
`${label}: progress.total_plans must be 0 in the template`,
);
assert.strictEqual(
fm.progress.completed_plans,
0,
`${label}: progress.completed_plans must be 0 in the template`,
);
});
}
});

View File

@@ -1,78 +0,0 @@
// allow-test-rule: source-text-is-the-product
// Workflow .md / agent .md / command .md / reference .md files — their text
// IS what the runtime loads. Testing text content tests the deployed contract.
// Per CONTRIBUTING.md exception matrix.
/**
* Bug #2559: Stale document references in Research phase
*
* The gsd-phase-researcher and gsd-project-researcher agents instruct
* WebSearch queries to always include "current year" (or a hardcoded
* year). This biases results toward stale dated content as time passes
* (e.g., a 2024 query run in 2026 returns stale results).
*
* Fix: Remove year-injection instructions from research agent
* WebSearch guidance so searches return current results.
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const PHASE_RESEARCHER = path.join(
__dirname,
'..',
'agents',
'gsd-phase-researcher.md'
);
const PROJECT_RESEARCHER = path.join(
__dirname,
'..',
'agents',
'gsd-project-researcher.md'
);
const FILES = [
{ label: 'gsd-phase-researcher.md', path: PHASE_RESEARCHER },
{ label: 'gsd-project-researcher.md', path: PROJECT_RESEARCHER },
];
describe('research agents do not inject year into web searches (#2559)', () => {
for (const { label, path: filePath } of FILES) {
test(`${label} contains no CURRENT_YEAR placeholder`, () => {
const content = fs.readFileSync(filePath, 'utf-8');
assert.ok(
!/CURRENT_YEAR/.test(content),
`${label} must not contain CURRENT_YEAR placeholder (causes stale-year injection)`
);
});
test(`${label} contains no hardcoded year in web search instructions`, () => {
const content = fs.readFileSync(filePath, 'utf-8');
const match = content.match(/\b20(2[3-9]|[3-9]\d)\b/);
assert.ok(
!match,
`${label} must not contain hardcoded year (found "${match && match[0]}") — biases searches toward stale content`
);
});
test(`${label} does not instruct searches to include year or current year`, () => {
const content = fs.readFileSync(filePath, 'utf-8');
// Match phrases like "include current year", "year in searches",
// "[current year]", "with year", etc.
const patterns = [
/include\s+(?:the\s+)?current\s+year/i,
/current\s+year/i,
/year\s+in\s+(?:searches|queries)/i,
/\[current year\]/i,
];
for (const pat of patterns) {
assert.ok(
!pat.test(content),
`${label} must not instruct year injection (matched /${pat.source}/)`
);
}
});
}
});

View File

@@ -1,171 +0,0 @@
/**
* Regression test for bug #2839
*
* /gsd-code-review-fix cleanup tail is non-transactional. If the agent is
* interrupted (system restart, OOM kill) AFTER the last fix commit but
* BEFORE `git worktree remove`, the worktree is orphaned in
* `git worktree list`, the agent's branch is left with unmerged commits,
* and STATE.md is never advanced. To anyone reading main only, the phase
* looks "ready to plan" while critical fixes sit on a dangling branch.
*
* Fix: introduce a recovery sentinel JSON at
* ${PHASE_DIR}/.review-fix-recovery-pending.json
* The sentinel is written AFTER `git worktree add` succeeds and
* REMOVED only after `git worktree remove` completes, so the cleanup
* tail is transactional from the orchestrator's perspective. If the
* process dies in between, the sentinel is left behind pointing at the
* orphan worktree and branch — a future run, /gsd-resume-work, or
* /gsd-progress can detect and complete the recovery.
*/
'use strict';
// allow-test-rule: source-text-is-the-product
// The gsd-code-fixer agent's working instructions ARE the product — Claude
// follows them at runtime. Structural assertions over the markdown source
// test the deployed contract. See bug-2686 for the same pattern.
const { describe, test, before } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const { parseFrontmatter } = require('./helpers.cjs');
const SENTINEL_NAME = '.review-fix-recovery-pending.json';
function extractStep(content, stepName) {
const re = new RegExp(`<step\\s+name="${stepName}">([\\s\\S]*?)</step>`);
const m = content.match(re);
return m ? m[1] : null;
}
describe('bug-2839: /gsd-code-review-fix cleanup is transactional', () => {
let agentPath;
let agentContent;
let frontmatter;
before(() => {
agentPath = path.join(__dirname, '..', 'agents', 'gsd-code-fixer.md');
assert.ok(fs.existsSync(agentPath), 'agents/gsd-code-fixer.md must exist');
agentContent = fs.readFileSync(agentPath, 'utf-8');
frontmatter = parseFrontmatter(agentContent);
assert.ok(frontmatter, 'agent must have YAML frontmatter');
});
test('agent declares a recovery sentinel filename', () => {
assert.ok(
agentContent.includes(SENTINEL_NAME),
`gsd-code-fixer.md must reference the recovery sentinel ${SENTINEL_NAME} so an interrupted cleanup tail is discoverable (#2839)`
);
});
test('sentinel is written inside setup_worktree, after git worktree add', () => {
const setupStep = extractStep(agentContent, 'setup_worktree');
assert.ok(setupStep, 'setup_worktree step must exist');
assert.ok(
setupStep.includes(SENTINEL_NAME),
`setup_worktree must reference ${SENTINEL_NAME} so the sentinel is created at the start of the run (#2839)`
);
const addPos = setupStep.indexOf('git worktree add');
assert.ok(addPos !== -1, 'setup_worktree must contain `git worktree add`');
// The sentinel WRITE (not just a reference) must come after `git worktree add`.
// Earlier references are allowed (e.g. recovery check for a stale sentinel
// from a prior interrupted run). Look for an explicit write — either a
// shell `>`/`>>` redirection, a `node -e` invocation that uses
// `fs.writeFileSync(...sentinel...)`, or a `Write` tool reference.
const writeIdx = (() => {
const candidates = [
/fs\.writeFileSync\([^)]*sentinel/,
/>\s*"?\$sentinel/,
/>\s*"?\$\{sentinel\}/,
/Write the recovery sentinel/i,
];
let earliest = -1;
for (const re of candidates) {
const m = re.exec(setupStep);
if (m && (earliest === -1 || m.index < earliest)) earliest = m.index;
}
return earliest;
})();
assert.ok(
writeIdx !== -1,
'setup_worktree must explicitly describe writing the sentinel (#2839)'
);
assert.ok(
addPos < writeIdx,
'sentinel must be written AFTER `git worktree add` succeeds (#2839)'
);
});
test('sentinel records worktree path, branch, and padded_phase as JSON fields', () => {
for (const key of ['worktree_path', 'branch', 'padded_phase']) {
assert.ok(
agentContent.includes(key),
`recovery sentinel must record \`${key}\` so a future /gsd-resume-work or /gsd-progress can locate the orphan state (#2839)`
);
}
});
test('sentinel removal happens only AFTER git worktree remove succeeds', () => {
const setupStep = extractStep(agentContent, 'setup_worktree');
assert.ok(setupStep, 'setup_worktree step must exist');
const cleanupAnchor = setupStep.lastIndexOf('Cleanup tail (transactional');
assert.ok(cleanupAnchor !== -1, 'setup_worktree must document cleanup-tail section');
const cleanupSection = setupStep.slice(cleanupAnchor);
const removeIdx = cleanupSection.indexOf('git worktree remove "$wt" --force');
assert.ok(removeIdx !== -1, 'cleanup-tail must remove worktree');
// Within the cleanup-tail section, accept either a literal-filename form
// (`rm -f .../.review-fix-recovery-pending.json`) or a shell-variable form
// referring to the previously-declared `sentinel` variable
// (`rm -f "$sentinel"` / `rm -f "${sentinel}"`).
const escapedName = SENTINEL_NAME.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
const sentinelRemovalRe = new RegExp(
`(rm\\s+(?:-f\\s+)?[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?)|unlink[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?))`
);
const sentinelRemovalMatch = sentinelRemovalRe.exec(cleanupSection);
assert.ok(
sentinelRemovalMatch,
`agent must remove the sentinel file (rm or unlink ${SENTINEL_NAME}) as part of the cleanup tail (#2839)`
);
const sentinelRemovalIdx = sentinelRemovalMatch.index;
assert.ok(
removeIdx < sentinelRemovalIdx,
'cleanup ordering must be: `git worktree remove` BEFORE sentinel removal (#2839)'
);
});
test('agent documents detection of pre-existing sentinel from a prior interrupted run', () => {
const lower = agentContent.toLowerCase();
const mentionsRecovery =
lower.includes('stale sentinel') ||
lower.includes('existing sentinel') ||
lower.includes('previous sentinel') ||
lower.includes('prior run') ||
lower.includes('pre-existing sentinel') ||
lower.includes('recovery');
assert.ok(
mentionsRecovery,
'agent must describe how it handles a pre-existing sentinel from a previous interrupted run (#2839)'
);
});
test('cleanup-tail obligation is documented as transactional / atomic', () => {
const lower = agentContent.toLowerCase();
const mentionsTransactional =
lower.includes('transactional') ||
lower.includes('atomic cleanup') ||
lower.includes('cleanup tail');
assert.ok(
mentionsTransactional,
'agent must document the cleanup tail as transactional/atomic (#2839)'
);
});
});

View File

@@ -1,215 +0,0 @@
/**
* Bug #2866: Codex Installer (RC.7) fails to strip legacy flat hooks if
* trailing newline is missing.
*
* The cleanup regexes in `bin/install.js` matched stale GSD hook blocks
* via `\r?\n` at the end. When a stale block sat at end-of-file without
* a trailing newline (very common — many editors strip them, and the
* legacy installer never wrote one), no shape stripped, the installer
* saw `gsd-check-update` already present, skipped writing the new
* Nested-AoT block, and Codex 0.125+ refused to load with
* "invalid type: map, expected a sequence in `hooks`"
*
* Fix: every shape's terminator is now `(?:\r?\n|$)` so end-of-file
* counts as a valid terminator. The strip logic was lifted into a pure
* helper, `stripStaleGsdHookBlocks(configContent)`, exported from
* `bin/install.js` for direct test coverage.
*
* This test parses `package.json` to require `bin/install.js`
* structurally (not by hardcoded path), then drives each historical
* shape through the helper twice — once with a trailing newline, once
* without — and asserts both are stripped.
*/
'use strict';
process.env.GSD_TEST_MODE = '1';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const path = require('node:path');
const fs = require('node:fs');
const REPO_ROOT = path.join(__dirname, '..');
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
const { stripStaleGsdHookBlocks } = require(installPath);
/**
* Parse the TOML output line-structurally so assertions check shape, not
* substring presence in raw text. Comments are dropped, table headers are
* recorded, and string-valued keys are captured. Sufficient for the small,
* well-formed TOML produced by these tests.
*/
function parseTomlShape(text) {
const tableHeaders = [];
const keys = new Map(); // dotted path → string value (last-write-wins, fine for these inputs)
let currentTable = '';
for (const rawLine of text.split('\n')) {
const line = rawLine.replace(/(?:^|\s)#.*$/, '').trim();
if (!line) continue;
const tableMatch = line.match(/^\[(\[)?([^\]]+)\]?\]$/);
if (tableMatch) {
currentTable = tableMatch[2];
tableHeaders.push((tableMatch[1] ? '[[' : '[') + currentTable + (tableMatch[1] ? ']]' : ']'));
continue;
}
const kvMatch = line.match(/^([A-Za-z_][\w-]*)\s*=\s*(.*)$/);
if (kvMatch) {
const key = currentTable ? `${currentTable}.${kvMatch[1]}` : kvMatch[1];
const value = kvMatch[2].replace(/^"(.*)"$/, '$1');
keys.set(key, value);
}
}
return { tableHeaders, keys };
}
const SHAPES = {
'Shape 1 (legacy gsd-update-check)': [
'# GSD Hooks',
'[[hooks]]',
'event = "SessionStart"',
'command = "node /Users/USER/.codex/hooks/gsd-update-check.js"',
].join('\n'),
'Shape 2 (flat [[hooks]] + gsd-check-update)': [
'# GSD Hooks',
'[[hooks]]',
'event = "SessionStart"',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
'Shape 3 ([[hooks.SessionStart]] without nested .hooks)': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
'Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'',
'[[hooks.SessionStart.hooks]]',
'type = "command"',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
};
describe('bug-2866: stripStaleGsdHookBlocks handles end-of-file without trailing newline', () => {
test('stripStaleGsdHookBlocks is exported from bin/install.js', () => {
assert.strictEqual(typeof stripStaleGsdHookBlocks, 'function',
'bin/install.js must export stripStaleGsdHookBlocks');
});
function assertStripped(out, shape, scenario) {
const shape_ = parseTomlShape(out);
const hooksTable = shape_.tableHeaders.find((h) => /^\[\[?hooks(\.|]\])/.test(h));
assert.strictEqual(hooksTable, undefined,
`(${shape}, ${scenario}) no hooks table header may remain after strip, got tables: ${shape_.tableHeaders.join(', ')}`);
const staleCmd = [...shape_.keys.entries()].find(([_, v]) =>
/gsd-(update-check|check-update)/.test(v));
assert.strictEqual(staleCmd, undefined,
`(${shape}, ${scenario}) no key may carry a stale gsd-*-update command, got: ${staleCmd && staleCmd.join('=')}`);
assert.strictEqual(shape_.keys.get('history.persistence'), 'save-all',
`(${shape}, ${scenario}) history.persistence must be preserved as "save-all"`);
}
for (const [shape, block] of Object.entries(SHAPES)) {
test(`${shape}: stripped when terminated by trailing newline`, () => {
const input = `[history]\npersistence = "save-all"\n${block}\n`;
assertStripped(stripStaleGsdHookBlocks(input), shape, 'with trailing newline');
});
test(`${shape}: stripped when at end-of-file without trailing newline`, () => {
// The reporter's repro: stale block sits at the very end with no \n.
const input = `[history]\npersistence = "save-all"\n${block}`;
assertStripped(stripStaleGsdHookBlocks(input), shape, 'no trailing newline');
});
}
test('returns input unchanged when no GSD hook block is present', () => {
const benign = '[history]\npersistence = "save-all"\n';
const out = stripStaleGsdHookBlocks(benign);
assert.strictEqual(out, benign, 'helper must be a no-op when no GSD reference exists');
const benignShape = parseTomlShape(out);
assert.strictEqual(benignShape.keys.get('history.persistence'), 'save-all',
'parsed shape must preserve history.persistence');
assert.deepStrictEqual(benignShape.tableHeaders, ['[history]'],
'parsed shape must contain only the [history] table');
});
// The structural rewrite (TOML-AST-driven, not regex-driven) must handle
// whitespace and key-ordering variations that the previous regex missed.
// These cases were silently leaked by the old implementation; one
// (V3) actually corrupted the file by leaving an orphaned key=value line
// outside any table.
const VARIATIONS = {
'extra blank line in Shape 4': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'',
'',
'[[hooks.SessionStart.hooks]]',
'type = "command"',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
'keys reordered (command before event in Shape 2)': [
'# GSD Hooks',
'[[hooks]]',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
'event = "SessionStart"',
].join('\n'),
'extra key alongside command (Shape 3 + timeout)': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
'timeout = 5000',
].join('\n'),
'tight whitespace (no spaces around `=`)': [
'# GSD Hooks',
'[[hooks]]',
'event="SessionStart"',
'command="node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
};
for (const [variation, block] of Object.entries(VARIATIONS)) {
test(`variation stripped: ${variation}`, () => {
const input = `[history]\npersistence = "save-all"\n${block}\n`;
assertStripped(stripStaleGsdHookBlocks(input), variation, 'with trailing newline');
});
test(`variation stripped at EOF without trailing newline: ${variation}`, () => {
const input = `[history]\npersistence = "save-all"\n${block}`;
assertStripped(stripStaleGsdHookBlocks(input), variation, 'no trailing newline');
});
}
test('user-authored [[hooks.UserPromptSubmit]] is preserved', () => {
// The structural strip must not touch hook tables that don't carry a
// GSD-managed `gsd-(check-update|update-check).js` command.
const input = [
'[history]',
'persistence = "save-all"',
'[[hooks.UserPromptSubmit]]',
'command = "node /Users/USER/my-hook.js"',
'',
].join('\n');
const out = stripStaleGsdHookBlocks(input);
const shape = parseTomlShape(out);
assert.ok(
shape.tableHeaders.includes('[[hooks.UserPromptSubmit]]'),
`user-authored [[hooks.UserPromptSubmit]] must survive, got: ${shape.tableHeaders.join(', ')}`,
);
assert.strictEqual(
shape.keys.get('hooks.UserPromptSubmit.command'),
'node /Users/USER/my-hook.js',
'user-authored command value must be preserved verbatim',
);
});
test('Shape 4 strip does not leave an orphaned [[hooks.SessionStart]] header', () => {
// Shape 4 is stripped before Shape 3 specifically to avoid this.
const block = SHAPES['Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])'];
const out = stripStaleGsdHookBlocks(`[history]\npersistence = "save-all"\n${block}`);
const outShape = parseTomlShape(out);
const orphan = outShape.tableHeaders.find((h) => /hooks\.SessionStart/.test(h));
assert.strictEqual(orphan, undefined,
`Shape 4 strip must remove the parent [[hooks.SessionStart]] header too, got tables: ${outShape.tableHeaders.join(', ')}`);
});
});

View File

@@ -1,191 +0,0 @@
/**
* Bug #2876: SKILL.md frontmatter parse failure when `description` begins
* with a YAML flow indicator like `[BETA]`.
*
* description: [BETA] Offload plan phase to Claude Code's ultraplan…
*
* YAML 1.2 treats a leading `[` as the start of a flow sequence, so any
* downstream parser (gh-copilot, JetBrains' kit, etc.) fails with
* "Unexpected scalar at node end". The Copilot/Antigravity/Trae/Codebuddy
* skill+agent converters in `bin/install.js` re-emit the description
* unquoted; the Claude variant `yamlQuote(...)`s it. Bring the others
* in line so any value is round-trip-safe regardless of leading char.
*
* The test is structural: it parses each emitted frontmatter into lines
* and asserts the `description` value is a quoted YAML scalar (double or
* single quoted) when the source description starts with a flow indicator.
* It does not regex the bytes for substrings.
*/
'use strict';
process.env.GSD_TEST_MODE = '1';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const path = require('node:path');
const fs = require('node:fs');
const REPO_ROOT = path.join(__dirname, '..');
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
const install = require(installPath);
// Build a minimal Claude command source whose description starts with the
// reporter's exact flow-indicator prefix. Apostrophe in the body forces
// any naive single-quoting to also escape correctly — the canonical
// safe form is `JSON.stringify(...)` (used by yamlQuote).
const REPORTER_DESCRIPTION =
"[BETA] Offload plan phase to Claude Code's ultraplan cloud — drafts remotely while terminal stays free, review in browser with inline comments, import back via /gsd-import. Claude Code only.";
// Use unquoted description in the source frontmatter — that's exactly the
// shape that ships in commands/gsd/*.md when authors paste a description
// without quoting it (see commands/gsd/ultraplan-phase.md). The bug is
// triggered when the converter re-emits this same value to the destination
// runtime without quoting. `extractFrontmatterField` strips a single outer
// quote pair but does not unescape internal characters, so quoting the
// fixture input would actually mask the bug.
function buildClaudeCommand(description) {
return [
'---',
'name: gsd:ultraplan-phase',
`description: ${description}`,
'argument-hint: "[phase-number]"',
'allowed-tools:',
' - Read',
' - Bash',
'---',
'',
'# body',
'',
].join('\n');
}
function buildClaudeAgent(description) {
return [
'---',
'name: gsd-extract-learnings',
`description: ${description}`,
'tools: Read, Bash',
'---',
'',
'# body',
'',
].join('\n');
}
function extractFrontmatter(content) {
// Leading delimiter is `---\n`; closing is the next standalone `---`
// on its own line. Tests parse line-structurally so the assertion
// doesn't drift on whitespace/order changes (per project test-rigor).
assert.ok(content.startsWith('---'), `output must begin with frontmatter, got: ${content.slice(0, 40)}`);
const lines = content.split('\n');
let openIdx = -1;
let closeIdx = -1;
for (let i = 0; i < lines.length; i += 1) {
if (lines[i] === '---') {
if (openIdx === -1) openIdx = i;
else { closeIdx = i; break; }
}
}
assert.ok(openIdx !== -1 && closeIdx !== -1, `output must have a closed frontmatter block, got:\n${content}`);
return lines.slice(openIdx + 1, closeIdx);
}
function findDescriptionLine(frontmatterLines) {
for (const line of frontmatterLines) {
if (line.startsWith('description:')) return line;
}
assert.fail(`no description line found in frontmatter:\n${frontmatterLines.join('\n')}`);
return ''; // unreachable
}
function isQuotedYamlScalar(valueText) {
// YAML safe-quoted scalar: starts with `"` and ends with `"`, OR
// starts with `'` and ends with `'`. This is what `yamlQuote()`
// (JSON.stringify) and the Claude variant of these converters emit.
const trimmed = valueText.trim();
if (trimmed.startsWith('"') && trimmed.endsWith('"')) return true;
if (trimmed.startsWith("'") && trimmed.endsWith("'")) return true;
return false;
}
function parseQuotedYamlValue(valueText) {
const trimmed = valueText.trim();
if (trimmed.startsWith('"')) return JSON.parse(trimmed);
if (trimmed.startsWith("'")) return trimmed.slice(1, -1).replace(/''/g, "'");
return trimmed;
}
function assertDescriptionRoundTrips(emitted, expected, label) {
const fmLines = extractFrontmatter(emitted);
const descLine = findDescriptionLine(fmLines);
const valueText = descLine.slice('description:'.length);
assert.ok(
isQuotedYamlScalar(valueText),
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
);
assert.strictEqual(
parseQuotedYamlValue(valueText),
expected,
`(${label}) description must round-trip through YAML quoting unchanged.`,
);
}
const COMMAND_CONVERTERS = [
{ label: 'convertClaudeCommandToCopilotSkill', fn: (src) => install.convertClaudeCommandToCopilotSkill(src, 'gsd-ultraplan-phase') },
{ label: 'convertClaudeCommandToAntigravitySkill', fn: (src) => install.convertClaudeCommandToAntigravitySkill(src, 'gsd-ultraplan-phase') },
{ label: 'convertClaudeCommandToTraeSkill', fn: (src) => install.convertClaudeCommandToTraeSkill(src, 'gsd-ultraplan-phase') },
{ label: 'convertClaudeCommandToCodebuddySkill', fn: (src) => install.convertClaudeCommandToCodebuddySkill(src, 'gsd-ultraplan-phase') },
];
const AGENT_CONVERTERS = [
{ label: 'convertClaudeAgentToCopilotAgent', fn: (src) => install.convertClaudeAgentToCopilotAgent(src) },
{ label: 'convertClaudeAgentToAntigravityAgent', fn: (src) => install.convertClaudeAgentToAntigravityAgent(src) },
];
// A grab-bag of leading characters that all break unquoted YAML scalar
// parsing per YAML 1.2 §7.3.3 / §6.9. The reporter's case is `[`; the
// rest defend against neighbouring drift.
const FLOW_HOSTILE_PREFIXES = ['[', '{', '*', '&', '!', '|', '>', '%', '@', '`'];
// Some converters (Trae, CodeBuddy) deliberately rewrite "Claude Code"
// in body content to their target runtime name, and the rewrite cuts
// across the description too. That's correct behavior — out of scope for
// the YAML-quoting fix — so for the reporter case we assert only the
// quoting requirement, not byte-equality of the round-tripped value.
function assertDescriptionIsQuoted(emitted, label) {
const fmLines = extractFrontmatter(emitted);
const descLine = findDescriptionLine(fmLines);
const valueText = descLine.slice('description:'.length);
assert.ok(
isQuotedYamlScalar(valueText),
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
);
}
describe('bug-2876: skill+agent converters emit YAML-quoted description', () => {
for (const { label, fn } of COMMAND_CONVERTERS) {
test(`${label}: reporter's "[BETA] ..." description is quoted`, () => {
const out = fn(buildClaudeCommand(REPORTER_DESCRIPTION));
assertDescriptionIsQuoted(out, label);
});
for (const prefix of FLOW_HOSTILE_PREFIXES) {
test(`${label}: leading ${JSON.stringify(prefix)} is quoted`, () => {
// Avoid leading/trailing `'` or `"` in the payload — `extractFrontmatterField`
// strips a single outer quote char of either kind regardless of whether
// the value was actually quoted, which would obscure the round-trip
// assertion. Pre-existing behavior, out of scope for #2876.
const desc = `${prefix} edge-case payload — flow indicator at start`;
const out = fn(buildClaudeCommand(desc));
assertDescriptionRoundTrips(out, desc, `${label} prefix=${prefix}`);
});
}
}
for (const { label, fn } of AGENT_CONVERTERS) {
test(`${label}: reporter-shape "[BETA] ..." description is quoted`, () => {
const out = fn(buildClaudeAgent(REPORTER_DESCRIPTION));
assertDescriptionIsQuoted(out, label);
});
}
});

View File

@@ -1,46 +0,0 @@
'use strict';
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
function walkMd(dir, out = []) {
if (!fs.existsSync(dir)) return out;
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) walkMd(full, out);
else if (entry.name.endsWith('.md')) out.push(full);
}
return out;
}
function extractSlashCommandTokens(markdown) {
const tokenRe = /\/gsd-[a-z0-9-]+/gi;
const tokens = new Set();
let m;
while ((m = tokenRe.exec(markdown)) !== null) {
tokens.add(m[0]);
}
return tokens;
}
describe('bug #3054: user-facing docs should not reference removed /gsd-next command', () => {
test('docs, workflows, and README surfaces use /gsd-progress --next instead', () => {
const root = path.join(__dirname, '..');
const files = [
...walkMd(path.join(root, 'docs')),
...walkMd(path.join(root, 'gsd-core', 'workflows')),
...fs.readdirSync(root).filter((f) => /^README.*\.md$/.test(f)).map((f) => path.join(root, f)),
];
const offenders = [];
for (const file of files) {
const content = fs.readFileSync(file, 'utf8');
const tokens = extractSlashCommandTokens(content);
if (tokens.has('/gsd-next')) offenders.push(path.relative(root, file));
}
assert.deepStrictEqual(offenders, [], `stale /gsd-next references remain in: ${offenders.join(', ')}`);
});
});

View File

@@ -1,151 +0,0 @@
'use strict';
// allow-test-rule: source-text-is-the-product
// The deployed settings.md IS the product — testing its text content tests the deployed contract.
/**
* Regression test for issue #33
*
* model_profile UI shows 4 options, schema has 5 — `adaptive` missing from
* `settings.md` AskUserQuestion.
*
* The schema (gsd-core/bin/shared/model-catalog.json `profiles` array) defines 5 valid
* model_profile values: quality, balanced, budget, adaptive, inherit. The
* settings.md AskUserQuestion block for model_profile originally listed only 4
* options (Quality, Balanced, Budget, Inherit) — `adaptive` was missing.
*
* Fix: the model_profile selection uses a two-question split. Q1 routes between
* Adaptive / Standard-tier / Inherit (3 options). Q2 (only when Q1 = Standard)
* asks Quality / Balanced / Budget. This keeps every individual options array
* within the AskUserQuestion 4-option cap while making all 5 profiles reachable.
*
* Fixes: #33
*/
const { describe, test, before } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const REPO_ROOT = path.join(__dirname, '..');
const SETTINGS_PATH = path.join(REPO_ROOT, 'gsd-core', 'workflows', 'settings.md');
const CATALOG_PATH = path.join(REPO_ROOT, 'gsd-core', 'bin', 'shared', 'model-catalog.json');
/**
* Collect every label: "..." value within a text block, lowercased.
*/
function extractOptionLabels(block) {
const re = /label:\s*"([^"]+)"/g;
const labels = [];
let m;
while ((m = re.exec(block)) !== null) {
labels.push(m[1].toLowerCase());
}
return labels;
}
describe('issue #33: model_profile schema and settings.md UI are in sync', () => {
let catalog;
let settingsContent;
let presentBlock;
before(() => {
catalog = JSON.parse(fs.readFileSync(CATALOG_PATH, 'utf-8'));
settingsContent = fs.readFileSync(SETTINGS_PATH, 'utf-8');
const presentMatch = settingsContent.match(/<step name="present_settings">[\s\S]*?<\/step>/);
assert.ok(presentMatch, 'settings.md must contain a present_settings step');
presentBlock = presentMatch[0];
});
// -- (a) Schema contract ---------------------------------------------------
test('schema includes the adaptive model_profile value', () => {
assert.ok(
Array.isArray(catalog.profiles),
'model-catalog.json must have a "profiles" array'
);
assert.ok(
catalog.profiles.includes('adaptive'),
'model-catalog should include the adaptive profile. Got: [' + catalog.profiles.join(', ') + ']'
);
});
test('schema includes adaptive as a model_profile value', () => {
assert.ok(
catalog.profiles.includes('adaptive'),
'"adaptive" must be in model-catalog.json profiles. Got: [' + catalog.profiles.join(', ') + ']'
);
});
test('schema includes all expected model_profile values', () => {
const expected = ['quality', 'balanced', 'budget', 'adaptive', 'inherit'];
for (const profile of expected) {
assert.ok(
catalog.profiles.includes(profile),
'Schema must include "' + profile + '" in profiles. Got: [' + catalog.profiles.join(', ') + ']'
);
}
});
// -- (b) UI contract — all 5 profiles reachable via present_settings -------
test('present_settings includes Adaptive as a selectable option (#33)', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'adaptive' || l.startsWith('adaptive')),
'Issue #33: present_settings must include an "Adaptive" label in its model_profile AskUserQuestion options so users can select it interactively. Got labels: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Quality as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'quality' || l.startsWith('quality')),
'present_settings must include a "Quality" option. Got: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Balanced as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'balanced' || l.startsWith('balanced')),
'present_settings must include a "Balanced" option. Got: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Budget as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'budget' || l.startsWith('budget')),
'present_settings must include a "Budget" option. Got: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Inherit as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'inherit' || l.startsWith('inherit')),
'present_settings must include an "Inherit" option. Got: [' + labels.join(', ') + ']'
);
});
// -- update_config and confirm steps reference adaptive --------------------
test('update_config step lists adaptive as a valid model_profile value', () => {
const m = settingsContent.match(/<step name="update_config">[\s\S]*?<\/step>/);
assert.ok(m, 'settings.md must have an update_config step');
assert.ok(
m[0].includes('adaptive'),
'update_config step must list "adaptive" as a valid model_profile value'
);
});
test('confirm step table shows adaptive as a possible model profile value', () => {
const m = settingsContent.match(/<step name="confirm">[\s\S]*?<\/step>/);
assert.ok(m, 'settings.md must have a confirm step');
assert.ok(
m[0].includes('adaptive'),
'confirm step must include "adaptive" in the Model Profile row'
);
});
});

View File

@@ -1,76 +0,0 @@
// allow-test-rule: source-text-is-the-product
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const repoRoot = path.resolve(__dirname, '..');
const WORKTREE_BRANCH_CHECK_FRAGMENT = path.join(repoRoot, 'gsd-core', 'references', 'worktree-branch-check.md');
function read(relPath) {
return fs.readFileSync(path.join(repoRoot, relPath), 'utf8');
}
describe('bug #3384: adjacent worktree data-loss guards', () => {
test('worktree cleanup CLI preserves caller cwd instead of resolving project root', () => {
const source = read('gsd-core/bin/gsd-tools.cjs');
const skipSet = source.slice(
source.indexOf('const SKIP_ROOT_RESOLUTION = new Set(['),
source.indexOf('if (!SKIP_ROOT_RESOLUTION.has(command))'),
);
assert.match(skipSet, /'worktree'/);
});
test('diagnose-issues references canonical fragment; fragment is verify-only and fails closed (#48)', () => {
// diagnose-issues.md now references the canonical fragment rather than
// inlining the block. Verify (a) it references the fragment and (b) the
// fragment itself has the correct ordering: symbolic-ref/HEAD assertion and
// ^worktree-agent- allow-list appear before any work, and (c) the fragment
// is verify-only — no destructive self-recovery.
const diagnoseSource = read('gsd-core/workflows/diagnose-issues.md');
assert.ok(
diagnoseSource.includes('worktree-branch-check.md'),
'diagnose-issues.md must reference the canonical worktree-branch-check.md fragment'
);
const fragmentSource = fs.readFileSync(WORKTREE_BRANCH_CHECK_FRAGMENT, 'utf8');
const branchCheck = fragmentSource.indexOf('HEAD_REF=$(git symbolic-ref --quiet HEAD || echo');
const namespaceCheck = fragmentSource.indexOf('^worktree-agent-');
assert.ok(branchCheck > 0, 'canonical fragment must assert HEAD before any work');
assert.ok(namespaceCheck > branchCheck, 'canonical fragment must require disposable worktree-agent branch');
// #48: verify-only — the destructive self-recovery is gone; the fragment fails closed instead.
assert.ok(!fragmentSource.includes('git reset --hard {EXPECTED_BASE}'), 'canonical fragment must not self-recover via reset --hard — orchestrator owns recovery (#48)');
assert.ok(fragmentSource.includes('exit 42'), 'canonical fragment must fail closed with exit 42 on base mismatch (#48)');
});
test('remove-workspace fails closed when git worktree remove fails', () => {
const source = read('gsd-core/workflows/remove-workspace.md');
const init = source.indexOf('REMOVE_FAILED=false');
const loop = source.indexOf('For each repo in the workspace');
const remove = source.indexOf('git worktree remove "$WORKSPACE_PATH/$REPO_NAME"');
assert.doesNotMatch(
source,
/git worktree remove "\$WORKSPACE_PATH\/\$REPO_NAME" 2>&1 \|\| true/,
'worktree removal failures must not be swallowed',
);
assert.ok(init > 0 && init < loop, 'REMOVE_FAILED must initialize once before the per-repo loop');
assert.ok(remove > loop, 'worktree removal should remain inside the per-repo loop');
assert.match(source, /Refusing to delete "\$WORKSPACE_PATH"/);
});
test('validate health warns when worktree inventory cannot be listed', () => {
const source = read('gsd-core/bin/lib/verify.cjs');
// Accept both hand-written dot access and the tsc-compiled bracket form
// (ADR-457: verify.cjs is now emitted from src/verify.cts):
// hand-written: worktreeHealth.reason === 'git_list_failed'
// tsc-compiled: worktreeHealth['reason'] === 'git_list_failed'
const failureBranch = source.search(/worktreeHealth(?:\.reason|\['reason'\]) === 'git_list_failed'/);
const warning = source.indexOf("addIssue('warning', 'W020'", failureBranch);
assert.ok(failureBranch > 0, 'verify health should branch on git_list_failed');
assert.ok(warning > failureBranch, 'git_list_failed should emit W020 degraded-health warning');
});
});

View File

@@ -1,133 +0,0 @@
// allow-test-rule: source-text-is-the-product
// Runtime prompt/hook files are deployed verbatim — their text IS what the
// runtime loads and executes. Asserting that text carries no retired `gsd-sdk`
// reference tests the deployed contract, which no behavioral seam can observe
// (there is no runtime API that enumerates "did any shipped prompt name the
// removed SDK binary").
/**
* Regression guard: no `gsd-sdk` references in runtime-facing surfaces (#339).
*
* The `@opengsd/gsd-sdk` package and its `gsd-sdk` binary were retired (ADR 0174,
* #191). The bulk runtime cleanup is already done — this test locks it in so a
* `gsd-sdk` / `GSD_SDK` reference cannot creep back into a shipped prompt or hook
* and re-introduce drift between the documented surface and the supported
* `gsd-tools` binary.
*
* Scope: runtime surfaces only — the prompts and hooks the installer ships into
* a user's runtime config dir. Explicitly NOT covered here:
* - `bin/install.js` — installer code, not a runtime-deployed prompt/hook
* surface. (It carries zero `gsd-sdk` references today; the SDK-shim
* verification subsystem was removed in #515 and the shim retired in #522.)
* - `gsd-core/bin/` — executable library code, not deployed prompt text; it
* may legitimately reference the SDK retirement in comments.
* - `tests/`, `docs/`, `.changeset/`, CI/lint scripts — legitimately reference
* the SDK retirement as history or detect its stale artifacts.
*
* Complements `tests/gsd-tools-path-refs.test.cjs`, which only catches the
* `gsd-sdk query` binary-invocation form; this catches ANY runtime reference.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const REPO_ROOT = path.join(__dirname, '..');
// Runtime surfaces the installer ships. Each entry is { dir, exts } — dir is
// repo-relative, exts is the set of file extensions whose text is deployed.
const RUNTIME_SURFACES = [
// .md prompts plus the non-.md runtime artifacts this dir also ships:
// _runtime-launcher.snippet.sh (the canonical launcher synced into every hook
// by scripts/sync-runtime-launcher.cjs) and discuss-phase/templates/*.json
// (loaded at runtime by discuss-phase.md). Scanning only .md left these two
// deployed files uncovered. (#691 review)
{ dir: path.join('gsd-core', 'workflows'), exts: ['.md', '.sh', '.json'] },
{ dir: path.join('gsd-core', 'references'), exts: ['.md'] },
// Prompt surfaces the installer deep-copies and the runtime loads via
// `@~/.claude/gsd-core/templates/*.md` anchors in workflows/commands; the
// lone config.json under templates/ ships too. (#691 review)
{ dir: path.join('gsd-core', 'templates'), exts: ['.md', '.json'] },
{ dir: path.join('gsd-core', 'contexts'), exts: ['.md'] },
{ dir: path.join('commands', 'gsd'), exts: ['.md'] },
{ dir: 'agents', exts: ['.md'] },
// Hooks ship as executable text (.js/.cjs/.sh). `hooks/dist/` is a gitignored
// build artifact regenerated from these sources, so scanning the sources is
// sufficient and avoids asserting against generated copies.
{ dir: 'hooks', exts: ['.js', '.cjs', '.sh'], skipDirs: ['dist'] },
];
// Matches every casing/separator variant of the retired SDK token:
// gsd-sdk, gsd_sdk, GSD-SDK, GSD_SDK, etc.
const SDK_REF = /gsd[-_]sdk/i;
/**
* Recursively collect files under `absDir` whose extension is in `exts`,
* skipping any directory name listed in `skipDirs`.
*/
function collectFiles(absDir, exts, skipDirs) {
if (!fs.existsSync(absDir)) return [];
const out = [];
for (const entry of fs.readdirSync(absDir, { withFileTypes: true })) {
if (entry.isDirectory()) {
if (skipDirs.includes(entry.name)) continue;
out.push(...collectFiles(path.join(absDir, entry.name), exts, skipDirs));
} else if (entry.isFile() && exts.includes(path.extname(entry.name))) {
out.push(path.join(absDir, entry.name));
}
}
return out;
}
function rel(file) {
return path.relative(REPO_ROOT, file).split(path.sep).join('/');
}
describe('#339 no gsd-sdk references in runtime surfaces', () => {
test('shipped prompts and hooks carry no retired gsd-sdk reference', () => {
const violations = [];
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
const files = collectFiles(path.join(REPO_ROOT, dir), exts, skipDirs);
for (const file of files) {
const lines = fs.readFileSync(file, 'utf-8').split(/\r?\n/);
for (let i = 0; i < lines.length; i++) {
if (SDK_REF.test(lines[i])) {
violations.push(`${rel(file)}:${i + 1}: ${lines[i].trim()}`);
}
}
}
}
assert.strictEqual(
violations.length,
0,
'Runtime surfaces must not reference the retired gsd-sdk binary/package — ' +
'use gsd-tools instead.\nViolations:\n' + violations.join('\n')
);
});
test('at least one file per configured extension is scanned (guards against an empty sweep)', () => {
// A path typo, directory rename, or stale extension could silently make
// collectFiles() return [] for part of a surface, turning the guard above
// into a no-op that always passes. Checking per-surface isn't enough: for
// gsd-core/workflows the .md files alone keep a per-surface count > 0, so
// dropping .sh/.json would stop covering _runtime-launcher.snippet.sh and
// discuss-phase/templates/*.json while the test stayed green. Assert each
// configured extension actually resolves to scanned files. (#691 review)
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
for (const ext of exts) {
const count = collectFiles(path.join(REPO_ROOT, dir), [ext], skipDirs).length;
assert.ok(
count > 0,
`Runtime surface "${dir}" resolved to 0 "${ext}" files — the path may ` +
'have moved or the extension is stale; update RUNTIME_SURFACES so the ' +
'guard keeps covering it.'
);
}
}
});
});

View File

@@ -1,115 +0,0 @@
// allow-test-rule: source-text-is-the-product
// Workflow .md / agent .md / command .md / reference .md files — their text
// IS what the runtime loads. Testing text content tests the deployed contract.
// Per CONTRIBUTING.md exception matrix.
/**
* Common Bug Patterns Reference Tests
*
* Structural tests for the common-bug-patterns.md reference file:
* - File exists at expected path
* - Contains expected bug pattern categories (at least 5 of 10)
* - Debugger agent references the file in required_reading
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const REFERENCE_PATH = path.join(
__dirname, '..', 'gsd-core', 'references', 'common-bug-patterns.md'
);
const DEBUGGER_AGENT_PATH = path.join(
__dirname, '..', 'agents', 'gsd-debugger.md'
);
const EXPECTED_CATEGORIES = [
'Off-by-One',
'Null',
'Async',
'State Management',
'Import',
'Environment',
'Data Shape',
'String Handling',
'File System',
'Error Handling',
];
describe('common-bug-patterns.md reference', () => {
test('reference file exists', () => {
assert.ok(
fs.existsSync(REFERENCE_PATH),
`Expected reference file at ${REFERENCE_PATH}`
);
});
test('has title and intro', () => {
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
assert.ok(
content.startsWith('# Common Bug Patterns'),
'File should start with "# Common Bug Patterns" title'
);
assert.ok(
content.includes('---'),
'File should contain --- separator after intro'
);
});
test('contains at least 5 of 10 expected categories', () => {
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
const found = EXPECTED_CATEGORIES.filter(cat =>
content.toLowerCase().includes(cat.toLowerCase())
);
assert.ok(
found.length >= 5,
`Expected at least 5 categories, found ${found.length}: ${found.join(', ')}`
);
});
test('each pattern category has at least one bold bullet item', () => {
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
// Only check sections inside <patterns> block, not <usage>
const patternsBlock = (content.split('<patterns>')[1] || '').split('</patterns>')[0];
const sections = patternsBlock.split(/^## /m).slice(1);
assert.ok(sections.length >= 5, `Expected at least 5 pattern sections, got ${sections.length}`);
for (const section of sections) {
const title = section.split('\n')[0].trim();
const bullets = section.match(/^- \*\*/gm);
assert.ok(
bullets && bullets.length >= 1,
`Pattern section "${title}" should have at least one "- **" bullet item`
);
}
});
});
describe('debugger agent references bug patterns', () => {
test('gsd-debugger.md exists', () => {
assert.ok(
fs.existsSync(DEBUGGER_AGENT_PATH),
`Expected debugger agent at ${DEBUGGER_AGENT_PATH}`
);
});
test('gsd-debugger.md references common-bug-patterns.md', () => {
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
assert.ok(
content.includes('common-bug-patterns.md'),
'Debugger agent should reference common-bug-patterns.md'
);
});
test('reference is inside <required_reading> block', () => {
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
const reqReadMatch = content.match(
/<required_reading>([\s\S]*?)<\/required_reading>/
);
assert.ok(reqReadMatch, 'Debugger agent should have a <required_reading> block');
assert.ok(
reqReadMatch[1].includes('common-bug-patterns.md'),
'common-bug-patterns.md should be inside <required_reading> block'
);
});
});

View File

@@ -572,3 +572,184 @@ describe('CR-INTEGRATION: workflow integration points', () => {
`autonomous.md gsd-code-review-fix args missing --auto flag; got args="${fixInvocation.args}"`);
});
});
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-2839-review-fix-transactional-cleanup.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-2839-review-fix-transactional-cleanup (consolidation epic #1969 B8 #1977)", () => {
/**
* Regression test for bug #2839
*
* /gsd-code-review-fix cleanup tail is non-transactional. If the agent is
* interrupted (system restart, OOM kill) AFTER the last fix commit but
* BEFORE `git worktree remove`, the worktree is orphaned in
* `git worktree list`, the agent's branch is left with unmerged commits,
* and STATE.md is never advanced. To anyone reading main only, the phase
* looks "ready to plan" while critical fixes sit on a dangling branch.
*
* Fix: introduce a recovery sentinel JSON at
* ${PHASE_DIR}/.review-fix-recovery-pending.json
* The sentinel is written AFTER `git worktree add` succeeds and
* REMOVED only after `git worktree remove` completes, so the cleanup
* tail is transactional from the orchestrator's perspective. If the
* process dies in between, the sentinel is left behind pointing at the
* orphan worktree and branch — a future run, /gsd-resume-work, or
* /gsd-progress can detect and complete the recovery.
*/
'use strict';
// allow-test-rule: source-text-is-the-product (see #2839)
// The gsd-code-fixer agent's working instructions ARE the product — Claude
// follows them at runtime. Structural assertions over the markdown source
// test the deployed contract. See bug-2686 for the same pattern.
const { describe, test, before } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const { parseFrontmatter } = require('./helpers.cjs');
const SENTINEL_NAME = '.review-fix-recovery-pending.json';
function extractStep(content, stepName) {
const re = new RegExp(`<step\\s+name="${stepName}">([\\s\\S]*?)</step>`);
const m = content.match(re);
return m ? m[1] : null;
}
describe('bug-2839: /gsd-code-review-fix cleanup is transactional', () => {
let agentPath;
let agentContent;
let frontmatter;
before(() => {
agentPath = path.join(__dirname, '..', 'agents', 'gsd-code-fixer.md');
assert.ok(fs.existsSync(agentPath), 'agents/gsd-code-fixer.md must exist');
agentContent = fs.readFileSync(agentPath, 'utf-8');
frontmatter = parseFrontmatter(agentContent);
assert.ok(frontmatter, 'agent must have YAML frontmatter');
});
test('agent declares a recovery sentinel filename', () => {
assert.ok(
agentContent.includes(SENTINEL_NAME),
`gsd-code-fixer.md must reference the recovery sentinel ${SENTINEL_NAME} so an interrupted cleanup tail is discoverable (#2839)`
);
});
test('sentinel is written inside setup_worktree, after git worktree add', () => {
const setupStep = extractStep(agentContent, 'setup_worktree');
assert.ok(setupStep, 'setup_worktree step must exist');
assert.ok(
setupStep.includes(SENTINEL_NAME),
`setup_worktree must reference ${SENTINEL_NAME} so the sentinel is created at the start of the run (#2839)`
);
const addPos = setupStep.indexOf('git worktree add');
assert.ok(addPos !== -1, 'setup_worktree must contain `git worktree add`');
// The sentinel WRITE (not just a reference) must come after `git worktree add`.
// Earlier references are allowed (e.g. recovery check for a stale sentinel
// from a prior interrupted run). Look for an explicit write — either a
// shell `>`/`>>` redirection, a `node -e` invocation that uses
// `fs.writeFileSync(...sentinel...)`, or a `Write` tool reference.
const writeIdx = (() => {
const candidates = [
/fs\.writeFileSync\([^)]*sentinel/,
/>\s*"?\$sentinel/,
/>\s*"?\$\{sentinel\}/,
/Write the recovery sentinel/i,
];
let earliest = -1;
for (const re of candidates) {
const m = re.exec(setupStep);
if (m && (earliest === -1 || m.index < earliest)) earliest = m.index;
}
return earliest;
})();
assert.ok(
writeIdx !== -1,
'setup_worktree must explicitly describe writing the sentinel (#2839)'
);
assert.ok(
addPos < writeIdx,
'sentinel must be written AFTER `git worktree add` succeeds (#2839)'
);
});
test('sentinel records worktree path, branch, and padded_phase as JSON fields', () => {
for (const key of ['worktree_path', 'branch', 'padded_phase']) {
assert.ok(
agentContent.includes(key),
`recovery sentinel must record \`${key}\` so a future /gsd-resume-work or /gsd-progress can locate the orphan state (#2839)`
);
}
});
test('sentinel removal happens only AFTER git worktree remove succeeds', () => {
const setupStep = extractStep(agentContent, 'setup_worktree');
assert.ok(setupStep, 'setup_worktree step must exist');
const cleanupAnchor = setupStep.lastIndexOf('Cleanup tail (transactional');
assert.ok(cleanupAnchor !== -1, 'setup_worktree must document cleanup-tail section');
const cleanupSection = setupStep.slice(cleanupAnchor);
const removeIdx = cleanupSection.indexOf('git worktree remove "$wt" --force');
assert.ok(removeIdx !== -1, 'cleanup-tail must remove worktree');
// Within the cleanup-tail section, accept either a literal-filename form
// (`rm -f .../.review-fix-recovery-pending.json`) or a shell-variable form
// referring to the previously-declared `sentinel` variable
// (`rm -f "$sentinel"` / `rm -f "${sentinel}"`).
const escapedName = SENTINEL_NAME.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
const sentinelRemovalRe = new RegExp(
`(rm\\s+(?:-f\\s+)?[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?)|unlink[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?))`
);
const sentinelRemovalMatch = sentinelRemovalRe.exec(cleanupSection);
assert.ok(
sentinelRemovalMatch,
`agent must remove the sentinel file (rm or unlink ${SENTINEL_NAME}) as part of the cleanup tail (#2839)`
);
const sentinelRemovalIdx = sentinelRemovalMatch.index;
assert.ok(
removeIdx < sentinelRemovalIdx,
'cleanup ordering must be: `git worktree remove` BEFORE sentinel removal (#2839)'
);
});
test('agent documents detection of pre-existing sentinel from a prior interrupted run', () => {
const lower = agentContent.toLowerCase();
const mentionsRecovery =
lower.includes('stale sentinel') ||
lower.includes('existing sentinel') ||
lower.includes('previous sentinel') ||
lower.includes('prior run') ||
lower.includes('pre-existing sentinel') ||
lower.includes('recovery');
assert.ok(
mentionsRecovery,
'agent must describe how it handles a pre-existing sentinel from a previous interrupted run (#2839)'
);
});
test('cleanup-tail obligation is documented as transactional / atomic', () => {
const lower = agentContent.toLowerCase();
const mentionsTransactional =
lower.includes('transactional') ||
lower.includes('atomic cleanup') ||
lower.includes('cleanup tail');
assert.ok(
mentionsTransactional,
'agent must document the cleanup tail as transactional/atomic (#2839)'
);
});
});
});
}

View File

@@ -8252,3 +8252,228 @@ describe('enh-772: reconcileCodexHooksJsonEvent preserves user-owned entries', (
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-2866-codex-strip-no-trailing-newline.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-2866-codex-strip-no-trailing-newline (consolidation epic #1969 B8 #1977)", () => {
/**
* Bug #2866: Codex Installer (RC.7) fails to strip legacy flat hooks if
* trailing newline is missing.
*
* The cleanup regexes in `bin/install.js` matched stale GSD hook blocks
* via `\r?\n` at the end. When a stale block sat at end-of-file without
* a trailing newline (very common — many editors strip them, and the
* legacy installer never wrote one), no shape stripped, the installer
* saw `gsd-check-update` already present, skipped writing the new
* Nested-AoT block, and Codex 0.125+ refused to load with
* "invalid type: map, expected a sequence in `hooks`"
*
* Fix: every shape's terminator is now `(?:\r?\n|$)` so end-of-file
* counts as a valid terminator. The strip logic was lifted into a pure
* helper, `stripStaleGsdHookBlocks(configContent)`, exported from
* `bin/install.js` for direct test coverage.
*
* This test parses `package.json` to require `bin/install.js`
* structurally (not by hardcoded path), then drives each historical
* shape through the helper twice — once with a trailing newline, once
* without — and asserts both are stripped.
*/
'use strict';
process.env.GSD_TEST_MODE = '1';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const path = require('node:path');
const fs = require('node:fs');
const REPO_ROOT = path.join(__dirname, '..');
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
const { stripStaleGsdHookBlocks } = require(installPath);
/**
* Parse the TOML output line-structurally so assertions check shape, not
* substring presence in raw text. Comments are dropped, table headers are
* recorded, and string-valued keys are captured. Sufficient for the small,
* well-formed TOML produced by these tests.
*/
function parseTomlShape(text) {
const tableHeaders = [];
const keys = new Map(); // dotted path → string value (last-write-wins, fine for these inputs)
let currentTable = '';
for (const rawLine of text.split('\n')) {
const line = rawLine.replace(/(?:^|\s)#.*$/, '').trim();
if (!line) continue;
const tableMatch = line.match(/^\[(\[)?([^\]]+)\]?\]$/);
if (tableMatch) {
currentTable = tableMatch[2];
tableHeaders.push((tableMatch[1] ? '[[' : '[') + currentTable + (tableMatch[1] ? ']]' : ']'));
continue;
}
const kvMatch = line.match(/^([A-Za-z_][\w-]*)\s*=\s*(.*)$/);
if (kvMatch) {
const key = currentTable ? `${currentTable}.${kvMatch[1]}` : kvMatch[1];
const value = kvMatch[2].replace(/^"(.*)"$/, '$1');
keys.set(key, value);
}
}
return { tableHeaders, keys };
}
const SHAPES = {
'Shape 1 (legacy gsd-update-check)': [
'# GSD Hooks',
'[[hooks]]',
'event = "SessionStart"',
'command = "node /Users/USER/.codex/hooks/gsd-update-check.js"',
].join('\n'),
'Shape 2 (flat [[hooks]] + gsd-check-update)': [
'# GSD Hooks',
'[[hooks]]',
'event = "SessionStart"',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
'Shape 3 ([[hooks.SessionStart]] without nested .hooks)': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
'Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'',
'[[hooks.SessionStart.hooks]]',
'type = "command"',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
};
describe('bug-2866: stripStaleGsdHookBlocks handles end-of-file without trailing newline', () => {
test('stripStaleGsdHookBlocks is exported from bin/install.js', () => {
assert.strictEqual(typeof stripStaleGsdHookBlocks, 'function',
'bin/install.js must export stripStaleGsdHookBlocks');
});
function assertStripped(out, shape, scenario) {
const shape_ = parseTomlShape(out);
const hooksTable = shape_.tableHeaders.find((h) => /^\[\[?hooks(\.|]\])/.test(h));
assert.strictEqual(hooksTable, undefined,
`(${shape}, ${scenario}) no hooks table header may remain after strip, got tables: ${shape_.tableHeaders.join(', ')}`);
const staleCmd = [...shape_.keys.entries()].find(([_, v]) =>
/gsd-(update-check|check-update)/.test(v));
assert.strictEqual(staleCmd, undefined,
`(${shape}, ${scenario}) no key may carry a stale gsd-*-update command, got: ${staleCmd && staleCmd.join('=')}`);
assert.strictEqual(shape_.keys.get('history.persistence'), 'save-all',
`(${shape}, ${scenario}) history.persistence must be preserved as "save-all"`);
}
for (const [shape, block] of Object.entries(SHAPES)) {
test(`${shape}: stripped when terminated by trailing newline`, () => {
const input = `[history]\npersistence = "save-all"\n${block}\n`;
assertStripped(stripStaleGsdHookBlocks(input), shape, 'with trailing newline');
});
test(`${shape}: stripped when at end-of-file without trailing newline`, () => {
// The reporter's repro: stale block sits at the very end with no \n.
const input = `[history]\npersistence = "save-all"\n${block}`;
assertStripped(stripStaleGsdHookBlocks(input), shape, 'no trailing newline');
});
}
test('returns input unchanged when no GSD hook block is present', () => {
const benign = '[history]\npersistence = "save-all"\n';
const out = stripStaleGsdHookBlocks(benign);
assert.strictEqual(out, benign, 'helper must be a no-op when no GSD reference exists');
const benignShape = parseTomlShape(out);
assert.strictEqual(benignShape.keys.get('history.persistence'), 'save-all',
'parsed shape must preserve history.persistence');
assert.deepStrictEqual(benignShape.tableHeaders, ['[history]'],
'parsed shape must contain only the [history] table');
});
// The structural rewrite (TOML-AST-driven, not regex-driven) must handle
// whitespace and key-ordering variations that the previous regex missed.
// These cases were silently leaked by the old implementation; one
// (V3) actually corrupted the file by leaving an orphaned key=value line
// outside any table.
const VARIATIONS = {
'extra blank line in Shape 4': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'',
'',
'[[hooks.SessionStart.hooks]]',
'type = "command"',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
'keys reordered (command before event in Shape 2)': [
'# GSD Hooks',
'[[hooks]]',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
'event = "SessionStart"',
].join('\n'),
'extra key alongside command (Shape 3 + timeout)': [
'# GSD Hooks',
'[[hooks.SessionStart]]',
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
'timeout = 5000',
].join('\n'),
'tight whitespace (no spaces around `=`)': [
'# GSD Hooks',
'[[hooks]]',
'event="SessionStart"',
'command="node /Users/USER/.codex/hooks/gsd-check-update.js"',
].join('\n'),
};
for (const [variation, block] of Object.entries(VARIATIONS)) {
test(`variation stripped: ${variation}`, () => {
const input = `[history]\npersistence = "save-all"\n${block}\n`;
assertStripped(stripStaleGsdHookBlocks(input), variation, 'with trailing newline');
});
test(`variation stripped at EOF without trailing newline: ${variation}`, () => {
const input = `[history]\npersistence = "save-all"\n${block}`;
assertStripped(stripStaleGsdHookBlocks(input), variation, 'no trailing newline');
});
}
test('user-authored [[hooks.UserPromptSubmit]] is preserved', () => {
// The structural strip must not touch hook tables that don't carry a
// GSD-managed `gsd-(check-update|update-check).js` command.
const input = [
'[history]',
'persistence = "save-all"',
'[[hooks.UserPromptSubmit]]',
'command = "node /Users/USER/my-hook.js"',
'',
].join('\n');
const out = stripStaleGsdHookBlocks(input);
const shape = parseTomlShape(out);
assert.ok(
shape.tableHeaders.includes('[[hooks.UserPromptSubmit]]'),
`user-authored [[hooks.UserPromptSubmit]] must survive, got: ${shape.tableHeaders.join(', ')}`,
);
assert.strictEqual(
shape.keys.get('hooks.UserPromptSubmit.command'),
'node /Users/USER/my-hook.js',
'user-authored command value must be preserved verbatim',
);
});
test('Shape 4 strip does not leave an orphaned [[hooks.SessionStart]] header', () => {
// Shape 4 is stripped before Shape 3 specifically to avoid this.
const block = SHAPES['Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])'];
const out = stripStaleGsdHookBlocks(`[history]\npersistence = "save-all"\n${block}`);
const outShape = parseTomlShape(out);
const orphan = outShape.tableHeaders.find((h) => /hooks\.SessionStart/.test(h));
assert.strictEqual(orphan, undefined,
`Shape 4 strip must remove the parent [[hooks.SessionStart]] header too, got tables: ${outShape.tableHeaders.join(', ')}`);
});
});
});
}

View File

@@ -888,3 +888,643 @@ describe('bug #2950: stale deleted-command references removed from workflow file
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/feat-2840-issue-driven-orchestration-guide.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:feat-2840-issue-driven-orchestration-guide (consolidation epic #1969 B8 #1977)", () => {
/**
* Tests for docs/issue-driven-orchestration.md (#2840).
*
* Structural-IR assertions per CONTRIBUTING.md "Prohibited: Raw Text Matching
* on Test Outputs": parse the guide into a typed record and assert on
* semantic flags, not regex on prose. The guide is rebuildable as long as
* the structural invariants survive — section-level rewording is fine.
*
* Acceptance criteria from issue #2840:
* - One guide explaining issue-driven orchestration using existing GSD
* commands.
* - Concrete end-to-end issue → workspace → plan/execute → verify/review
* → PR flow.
* - Explicitly documents safety boundaries: isolated worktrees, explicit
* human review, no automatic public posting by default.
* - Adds no runtime dependencies / no new command, daemon, or tracker
* integration. (Test-enforced via concept-mapping audit.)
*/
// allow-test-rule: structural-IR parser for a docs guide. The .includes() (see #2840)
// calls below build a typed record (commandsPresent flags, conceptPairs
// flags, nonGoalFlags, safetyFlags); assertions run on those booleans, not
// on raw text. This is the documented escape hatch in
// scripts/lint-no-source-grep.cjs for doc-shape tests.
'use strict';
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const GUIDE_PATH = path.join(__dirname, '..', 'docs', 'issue-driven-orchestration.md');
// ─── Helpers ────────────────────────────────────────────────────────────────
/**
* Extract a section starting at a given heading. Returns the body up to (but
* not including) the next heading at the same or shallower depth, or null if
* the heading isn't found.
*/
function extractSection(content, heading) {
const lines = content.split('\n');
const headingRe = new RegExp(`^(#+)\\s+${heading.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\$&')}\\s*$`);
let start = -1;
let depth = 0;
for (let i = 0; i < lines.length; i++) {
const m = lines[i].match(headingRe);
if (m) {
start = i + 1;
depth = m[1].length;
break;
}
}
if (start < 0) return null;
let end = lines.length;
for (let i = start; i < lines.length; i++) {
const m = lines[i].match(/^(#+)\s+/);
if (m && m[1].length <= depth) {
end = i;
break;
}
}
return lines.slice(start, end).join('\n');
}
/**
* Parse the guide into a typed record. Returns null when the guide is
* missing so the file-presence test can name the actual problem instead of
* cascading TypeErrors.
*/
function parseGuide() {
if (!fs.existsSync(GUIDE_PATH)) return null;
const content = fs.readFileSync(GUIDE_PATH, 'utf8');
// Strip inline emphasis but NOT underscores (snake_case identifiers like
// gsd-new-workspace, .planning/, etc. must survive).
const stripped = content.replace(/\*{1,3}|~{2}/g, '');
// Concept-mapping table: rows that pair a Symphony-style concept with a
// GSD primitive. Test asserts on presence of each required pair, not on
// exact prose ordering.
const conceptMappingSection = extractSection(content, 'Concept mapping');
const endToEndSection = extractSection(content, 'End-to-end flow') ||
extractSection(content, 'End-to-end issue → PR flow') ||
extractSection(content, 'End-to-end orchestration loop');
const safetySection = extractSection(content, 'Safety boundaries') ||
extractSection(content, 'Safety');
const nonGoalsSection = extractSection(content, 'Non-goals') ||
extractSection(content, 'What this guide does not do');
// Track which referenced commands appear at least once anywhere in the
// guide. This prevents drift if /gsd-* command names are renamed.
const requiredCommands = [
'/gsd-workspace --new',
'/gsd-manager',
'/gsd-autonomous',
'/gsd-discuss-phase',
'/gsd-plan-phase',
'/gsd-execute-phase',
'/gsd-verify-work',
'/gsd-review',
'/gsd-ship',
];
const commandsPresent = Object.fromEntries(
requiredCommands.map((c) => [c, content.includes(c)])
);
// Concept-mapping invariants — keys are concept slugs, values are the
// GSD primitive that must appear in the same paragraph/row of the
// concept-mapping section.
const conceptPairs = conceptMappingSection
? {
roadmap: /ROADMAP\.md/.test(conceptMappingSection),
statemd: /STATE\.md/.test(conceptMappingSection),
contextmd: /CONTEXT\.md/.test(conceptMappingSection),
planmd: /PLAN\.md/.test(conceptMappingSection),
workspaceCommand: /\/gsd-workspace\s+--new/.test(conceptMappingSection),
executionCommand:
/\/gsd-manager/.test(conceptMappingSection) ||
/\/gsd-autonomous/.test(conceptMappingSection),
verifyCommand: /\/gsd-verify-work/.test(conceptMappingSection),
reviewCommand: /\/gsd-review/.test(conceptMappingSection),
shipCommand: /\/gsd-ship/.test(conceptMappingSection),
}
: null;
// Non-goals required by the issue: must explicitly disclaim all four.
const nonGoalFlags = nonGoalsSection
? {
noVendoring: /vendor|copy/i.test(nonGoalsSection),
noDaemon: /daemon|polling/i.test(nonGoalsSection),
noTrackerDependency: /tracker.*depend|mandatory.*track/i.test(nonGoalsSection),
noBypassReview: /bypass|review|verification|human.*decision|human gate/i.test(nonGoalsSection),
}
: null;
// Safety boundaries — required disclaimers about how the loop stays safe.
const safetyFlags = safetySection
? {
isolatedWorktrees: /worktree|isolated/i.test(safetySection),
explicitReview: /review|human.*gate|human.*approval/i.test(safetySection),
noAutoPosting: /not.*automatic|no.*auto|explicit.*confirm|user.*confirm|human.*confirm/i.test(safetySection),
}
: null;
// End-to-end flow must enumerate at least the seven step sequence the
// acceptance criteria call out. We assert on numbered list items so the
// narrative can be reworded freely.
const numberedSteps = endToEndSection
? (endToEndSection.match(/^\s*\d+\.\s+/gm) || []).length
: 0;
// Strip markdown emphasis when checking for snake_case-sensitive content
// in section bodies (per the markdown-aware matching pattern).
const strippedConceptMapping = conceptMappingSection
? conceptMappingSection.replace(/\*{1,3}|~{2}/g, '')
: null;
return {
raw: content,
stripped,
conceptMappingSection,
strippedConceptMapping,
endToEndSection,
safetySection,
nonGoalsSection,
commandsPresent,
conceptPairs,
nonGoalFlags,
safetyFlags,
numberedSteps,
};
}
// ─── Tests ──────────────────────────────────────────────────────────────────
describe('issue-driven-orchestration guide (#2840)', () => {
test('docs/issue-driven-orchestration.md exists', () => {
assert.ok(
fs.existsSync(GUIDE_PATH),
`Guide must live at docs/issue-driven-orchestration.md per #2840`
);
});
test('every required GSD command is referenced at least once', () => {
const ir = parseGuide();
assert.ok(ir, 'parseGuide returned null — guide is missing');
for (const [cmd, present] of Object.entries(ir.commandsPresent)) {
assert.ok(present, `guide must reference ${cmd}`);
}
});
test('concept mapping section exists and pairs Symphony concepts with GSD primitives', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.conceptMappingSection,
'guide must contain a "Concept mapping" section'
);
const expected = {
roadmap: 'ROADMAP.md must appear in the concept mapping',
statemd: 'STATE.md must appear in the concept mapping',
contextmd: 'CONTEXT.md must appear in the concept mapping',
planmd: 'PLAN.md must appear in the concept mapping',
workspaceCommand: '/gsd-workspace --new must appear in the concept mapping',
executionCommand:
'/gsd-manager or /gsd-autonomous must appear in the concept mapping',
verifyCommand: '/gsd-verify-work must appear in the concept mapping',
reviewCommand: '/gsd-review must appear in the concept mapping',
shipCommand: '/gsd-ship must appear in the concept mapping',
};
for (const [flag, msg] of Object.entries(expected)) {
assert.equal(ir.conceptPairs[flag], true, msg);
}
});
test('safety boundaries section names isolation, review, and non-auto-posting', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.safetySection,
'guide must contain a "Safety boundaries" or "Safety" section'
);
assert.equal(
ir.safetyFlags.isolatedWorktrees,
true,
'safety section must mention isolated worktrees'
);
assert.equal(
ir.safetyFlags.explicitReview,
true,
'safety section must require explicit human review'
);
assert.equal(
ir.safetyFlags.noAutoPosting,
true,
'safety section must disclaim automatic public posting'
);
});
test('non-goals section disclaims vendoring, daemon, tracker dependency, and gate-bypass', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.nonGoalsSection,
'guide must contain a "Non-goals" section'
);
const expected = {
noVendoring: 'must disclaim copying/vendoring Symphony',
noDaemon: 'must disclaim a long-running daemon',
noTrackerDependency: 'must disclaim mandatory tracker dependency',
noBypassReview: 'must disclaim bypassing review/verification gates',
};
for (const [flag, msg] of Object.entries(expected)) {
assert.equal(ir.nonGoalFlags[flag], true, msg);
}
});
test('end-to-end flow enumerates at least 7 numbered steps (per acceptance criteria)', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.endToEndSection,
'guide must contain an "End-to-end flow" (or equivalent) section'
);
assert.ok(
ir.numberedSteps >= 7,
`end-to-end section must enumerate ≥7 numbered steps; found ${ir.numberedSteps}`
);
});
test('every fenced code block has a language tag (markdownlint MD040)', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
// Pair fence opens; flag any opener with no language tag.
const fences = ir.raw.match(/^```.*$/gm) || [];
const openers = [];
for (let i = 0; i < fences.length; i++) {
// Even index = opener, odd = closer. An opener with empty trailing
// text is MD040.
if (i % 2 === 0) openers.push(fences[i]);
}
const bare = openers.filter((f) => /^```\s*$/.test(f));
assert.equal(
bare.length,
0,
`MD040: ${bare.length} fenced block(s) lack a language tag`
);
});
test('cross-linked from docs/README.md', () => {
const readme = path.join(__dirname, '..', 'docs', 'README.md');
if (!fs.existsSync(readme)) {
// docs/README.md is the discovery surface. Without a cross-link, the
// guide is invisible to users browsing docs/.
return; // tolerate absence; test below ensures FEATURES.md anchor.
}
const txt = fs.readFileSync(readme, 'utf8');
assert.ok(
/issue-driven-orchestration/.test(txt),
'docs/README.md must link to the new guide'
);
});
test('cross-linked from docs/USER-GUIDE.md', () => {
const guide = path.join(__dirname, '..', 'docs', 'USER-GUIDE.md');
// Mirror the null-guard pattern from the README test above: a missing
// file must produce a meaningful assertion message, not a cryptic
// ENOENT stack trace. (CR #3036.)
assert.ok(
fs.existsSync(guide),
'docs/USER-GUIDE.md must exist for cross-link validation'
);
const txt = fs.readFileSync(guide, 'utf8');
assert.ok(
/issue-driven-orchestration/.test(txt),
'docs/USER-GUIDE.md must link to the new guide'
);
});
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/feat-3025-mcp-token-budget-docs.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:feat-3025-mcp-token-budget-docs (consolidation epic #1969 B8 #1977)", () => {
/**
* Documentation regression test for issue #3025 — MCP token-budget guidance.
*
* Verifies that gsd-core/references/context-budget.md contains the
* structural elements the issue requires:
*
* 1. A section explaining MCP/tool schemas as a context-budget concern
* 2. References to the harness-side toggles (enabledMcpjsonServers,
* disabledMcpjsonServers in .claude/settings.json)
* 3. A pre-phase audit checklist (browser/playwright, platform-specific,
* project-specific)
* 4. An explicit note that GSD does NOT manage MCP enablement — this is
* a Claude Code harness concern (with a cross-link)
* 5. Note the interaction with model_profile (compounding levers)
*
* Tests parse the doc into a typed section record (parseMcpSection) and
* assert on flag booleans, not raw text matches. Adheres to
* CONTRIBUTING.md "no-source-grep" — describes invariants, not wording,
* so the prose can be reworded freely as long as the semantics survive.
*
* Companion to docs/USER-GUIDE.md task section, which is exercised by the
* same parser shape (separate test below).
*/
'use strict';
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const ROOT = path.join(__dirname, '..');
const CONTEXT_BUDGET_MD = path.join(ROOT, 'gsd-core', 'references', 'context-budget.md');
const USER_GUIDE_MD = path.join(ROOT, 'docs', 'USER-GUIDE.md');
/**
* Extract the MCP-budget section from a markdown file by header text.
* Returns null if the section is missing. Section runs from the matching
* `## ` header up to the next `## ` header (or EOF).
*/
function extractSection(filePath, headerSubstring) {
const content = fs.readFileSync(filePath, 'utf8');
const lines = content.split(/\r?\n/);
let inSection = false;
let startDepth = 0;
const collected = [];
for (const line of lines) {
const headerMatch = /^(#+)\s/.exec(line);
if (headerMatch) {
const depth = headerMatch[1].length;
if (inSection) {
// Section ends at a header at the same or shallower depth.
// Subsections at deeper depth are part of the section.
if (depth <= startDepth) break;
} else if (line.toLowerCase().includes(headerSubstring.toLowerCase())) {
inSection = true;
startDepth = depth;
}
}
if (inSection) collected.push(line);
}
return collected.length > 0 ? collected.join('\n') : null;
}
/**
* Parse the MCP-budget section into a typed semantic-flag record.
* Each flag answers a single behavioral question that #3025 requires
* the prose to encode.
*/
function parseMcpBudgetSection(section) {
if (!section || typeof section !== 'string') {
return {
ok: false,
sectionLength: 0,
explainsMcpAsBudgetConcern: false,
namesEnabledMcpjsonServers: false,
namesDisabledMcpjsonServers: false,
namesClaudeSettingsJson: false,
includesPrePhaseAudit: false,
auditMentionsBrowserOrPlaywright: false,
auditMentionsPlatformSpecific: false,
auditMentionsCrossProject: false,
explainsHarnessNotGsd: false,
mentionsModelProfileInteraction: false,
crossLinksContextBudget: false,
};
}
// CR follow-up: strip inline markdown emphasis (`**`, `*`, `~~`) and
// backticks before phrase-matching so e.g. "GSD does **not** manage"
// is caught by the primary `gsd does not manage` alternative below.
// WITHOUT this, the markdown-bold breaks the contiguous match and the
// test only passes via the fallback branch (silent dead code).
// Underscores are intentionally NOT stripped — `model_profile` and
// other snake_case identifiers must survive intact so the
// model_profile interaction check still finds them.
const stripped = section.replace(/\*{1,3}|~{2}|`/g, '');
// (1) Explains MCP as budget concern — must mention BOTH "MCP" / "tool
// schema" AND a token/cost framing.
const explainsMcpAsBudgetConcern =
/\bmcp\b|tool schema|tool schemas/i.test(stripped) &&
/\btoken|context budget|per[- ]turn|cost\b/i.test(stripped);
// (2) Names the harness keys verbatim
const namesEnabledMcpjsonServers = /enabledMcpjsonServers/.test(stripped);
const namesDisabledMcpjsonServers = /disabledMcpjsonServers/.test(stripped);
// (3) Names the settings file location
const namesClaudeSettingsJson = /\.claude\/settings\.json/.test(stripped);
// (4) Audit checklist — must mention all three classes the issue
// calls out, plus a "before this phase / pre-phase" framing
const includesPrePhaseAudit =
/audit|checklist|review (your )?mcp|before (starting|beginning) (a |the )?phase/i.test(stripped);
const auditMentionsBrowserOrPlaywright = /\bbrowser\b|playwright/i.test(stripped);
const auditMentionsPlatformSpecific = /platform[- ]specific|mac[- ]?tools|windows[- ]?tools|os[- ]specific/i.test(stripped);
const auditMentionsCrossProject = /(other|different|cross[- ])\s*project|stale (project )?mcp/i.test(stripped);
// (5) Harness vs GSD distinction — must explicitly state GSD doesn't
// own this knob and point at the harness
const explainsHarnessNotGsd =
/(gsd does(?:n[''’]t| not) (own|manage|control)|harness (concern|setting|controlled)|not a gsd (setting|knob))/i.test(stripped);
// (6) Compounding with model_profile
const mentionsModelProfileInteraction =
/model[_ ]profile/i.test(stripped) &&
/compound|multiplier|stack|every[- ]turn|regardless of (which )?model|in addition/i.test(stripped);
// (7) Cross-link to the canonical reference doc — task-guide section
// must point readers at context-budget.md for the full audit. Encoded
// as a named flag (CR follow-up) so the assertion sits alongside the
// other parsed invariants rather than as a one-off inline regex.
const crossLinksContextBudget = /context-budget/i.test(stripped);
return {
ok: true,
sectionLength: section.length,
explainsMcpAsBudgetConcern,
namesEnabledMcpjsonServers,
namesDisabledMcpjsonServers,
namesClaudeSettingsJson,
includesPrePhaseAudit,
auditMentionsBrowserOrPlaywright,
auditMentionsPlatformSpecific,
auditMentionsCrossProject,
explainsHarnessNotGsd,
mentionsModelProfileInteraction,
crossLinksContextBudget,
};
}
// ─── context-budget.md ──────────────────────────────────────────────────────
describe('#3025 context-budget.md: MCP token-budget section exists with required content', () => {
test('the file exists', () => {
assert.ok(fs.existsSync(CONTEXT_BUDGET_MD), `expected file at ${CONTEXT_BUDGET_MD}`);
});
test('has a section header that mentions MCP', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
assert.ok(section, 'must have a `## ...MCP...` heading; section was not found');
});
test('explains MCP/tool schemas as a context-budget concern (#3025 requirement 1)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.explainsMcpAsBudgetConcern, true,
`must explain MCP/tool schemas as a token/context-budget concern; section was:\n${section}`);
});
test('names enabledMcpjsonServers and disabledMcpjsonServers (#3025 requirement 2)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.namesEnabledMcpjsonServers, true,
'section must reference `enabledMcpjsonServers` so users know the exact key');
assert.equal(parsed.namesDisabledMcpjsonServers, true,
'section must reference `disabledMcpjsonServers` for parity');
assert.equal(parsed.namesClaudeSettingsJson, true,
'section must name `.claude/settings.json` as the location of the toggle');
});
test('includes a pre-phase audit checklist with all three classes (#3025 requirement 3)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.includesPrePhaseAudit, true,
'section must include audit/checklist framing for pre-phase MCP review');
assert.equal(parsed.auditMentionsBrowserOrPlaywright, true,
'audit must mention browser/playwright tools as a candidate for disabling');
assert.equal(parsed.auditMentionsPlatformSpecific, true,
'audit must mention platform-specific tools (Mac/Windows/OS-specific)');
assert.equal(parsed.auditMentionsCrossProject, true,
'audit must mention stale/cross-project MCPs from other projects');
});
test('explains GSD does not own MCP enablement — harness concern (#3025 requirement 4)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.explainsHarnessNotGsd, true,
'section must explicitly state GSD does not manage MCP enablement (harness concern)');
});
test('notes interaction with model_profile (compounding levers) (#3025 requirement 5)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.mentionsModelProfileInteraction, true,
'section must note that trimming MCPs compounds with model_profile choice');
});
test('full semantic record matches the #3025 contract — typed snapshot', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
const contract = {
ok: parsed.ok,
explainsMcpAsBudgetConcern: parsed.explainsMcpAsBudgetConcern,
namesEnabledMcpjsonServers: parsed.namesEnabledMcpjsonServers,
namesDisabledMcpjsonServers: parsed.namesDisabledMcpjsonServers,
namesClaudeSettingsJson: parsed.namesClaudeSettingsJson,
includesPrePhaseAudit: parsed.includesPrePhaseAudit,
auditMentionsBrowserOrPlaywright: parsed.auditMentionsBrowserOrPlaywright,
auditMentionsPlatformSpecific: parsed.auditMentionsPlatformSpecific,
auditMentionsCrossProject: parsed.auditMentionsCrossProject,
explainsHarnessNotGsd: parsed.explainsHarnessNotGsd,
mentionsModelProfileInteraction: parsed.mentionsModelProfileInteraction,
};
assert.deepStrictEqual(contract, {
ok: true,
explainsMcpAsBudgetConcern: true,
namesEnabledMcpjsonServers: true,
namesDisabledMcpjsonServers: true,
namesClaudeSettingsJson: true,
includesPrePhaseAudit: true,
auditMentionsBrowserOrPlaywright: true,
auditMentionsPlatformSpecific: true,
auditMentionsCrossProject: true,
explainsHarnessNotGsd: true,
mentionsModelProfileInteraction: true,
}, 'context-budget.md MCP section contract violated');
});
});
// ─── docs/USER-GUIDE.md task section ────────────────────────────────────────
describe('#3025 docs/USER-GUIDE.md: companion task section exists', () => {
test('USER-GUIDE.md has an MCP-trimming task section', () => {
const section = extractSection(USER_GUIDE_MD, 'mcp');
assert.ok(section,
'USER-GUIDE.md must have a `### ...MCP...` task section so users find it via the guide');
});
test('USER-GUIDE.md task section names the harness key and cross-links the reference', () => {
const section = extractSection(USER_GUIDE_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.namesEnabledMcpjsonServers, true,
'task section must mention the harness key by name');
// Cross-link to the reference doc — assert on the parsed flag so
// the invariant lives alongside the other named flags (CR follow-up
// on the no-source-grep standard).
assert.equal(parsed.crossLinksContextBudget, true,
'task section must cross-link to context-budget.md');
});
});
// ─── markdownlint pre-flight (per bundle-docs-with-code skill) ──────────────
describe('#3025 markdownlint pre-flight: MD040 + MD056', () => {
test('every fenced code block in the new MCP section has a language tag (MD040)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
// Guard: extractSection returns null when the section is missing.
// Without this, `section.match(...)` would throw a TypeError instead
// of producing a meaningful assertion failure (CR follow-up).
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD040');
const fences = (section.match(/^```([a-zA-Z0-9_+-]*)?\s*$/gm) || []);
// Pairs of fences open/close; odd-indexed ones close blocks. Every
// OPENING fence must have a language tag. Closing fences are bare ```.
// Walk pairs: even index = opener, odd = closer.
const openers = fences.filter((_, i) => i % 2 === 0);
const missing = openers.filter((line) => /^```\s*$/.test(line));
assert.deepStrictEqual(missing, [],
`every fenced code block opener must have a language tag (MD040). Missing: ${JSON.stringify(missing)}`);
});
test('every markdown table row in the new MCP section has the same column count as its header (MD056)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
// Guard: same null-section concern as MD040 above (CR follow-up).
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD056');
const lines = section.split(/\r?\n/);
// Walk through and detect tables: header row followed by a separator
// (--- pattern) followed by data rows. Count `|` per line.
const issues = [];
for (let i = 0; i < lines.length - 1; i += 1) {
const header = lines[i];
const sep = lines[i + 1];
if (!/^\s*\|.*\|\s*$/.test(header)) continue;
if (!/^\s*\|[\s\-:|]+\|\s*$/.test(sep)) continue;
const headerCols = (header.match(/\|/g) || []).length;
// Walk data rows
for (let j = i + 2; j < lines.length; j += 1) {
const row = lines[j];
if (!/^\s*\|.*\|\s*$/.test(row)) break;
const rowCols = (row.match(/\|/g) || []).length;
if (rowCols !== headerCols) {
issues.push({ line: j, expected: headerCols, actual: rowCols, row });
}
}
}
assert.deepStrictEqual(issues, [],
`table rows must match header column count (MD056). Issues: ${JSON.stringify(issues, null, 2)}`);
});
});
});
}

View File

@@ -1,96 +0,0 @@
// allow-test-rule: source-text-is-the-product
// The PROJECT.md template + complete-milestone workflow .md ARE the product surface
// the runtime loads; asserting on their text tests the deployed contract directly.
/**
* Enhancement #72 — optional Business Context section in the PROJECT.md template.
*
* Contract tests over the product-text surfaces (template + milestone workflow .md):
* the template offers a Business Context section that is explicitly OPTIONAL, capped
* at the four approved one-line fields, and the milestone evolution review treats it
* as conditional so non-business projects that deleted it are never forced to review it.
*/
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const TEMPLATE = path.join(__dirname, '..', 'gsd-core', 'templates', 'project.md');
const COMPLETE_MILESTONE = path.join(__dirname, '..', 'gsd-core', 'workflows', 'complete-milestone.md');
function parseTemplateContract(content) {
const lines = content.split(/\r?\n/);
const lower = content.toLowerCase();
// The Business Context block lives between its heading and the next "## " heading.
const startIdx = lines.findIndex(l => l.trim() === '## Business Context');
let sectionBody = '';
if (startIdx !== -1) {
const rest = lines.slice(startIdx + 1);
const endOffset = rest.findIndex(l => l.startsWith('## '));
sectionBody = (endOffset === -1 ? rest : rest.slice(0, endOffset)).join('\n');
}
const fieldOf = (label) => new RegExp(`^- \\*\\*${label}\\*\\*:`, 'm').test(sectionBody);
return {
hasSection: startIdx !== -1,
// Optional-by-default: an HTML comment tells non-business projects to delete it.
hasOptionalMarker: /<!--\s*OPTIONAL/i.test(sectionBody) && /delete this section/i.test(sectionBody),
fields: {
customer: fieldOf('Customer'),
revenueModel: fieldOf('Revenue model'),
successMetric: fieldOf('Success metric'),
strategyNotes: fieldOf('Strategy notes'),
},
fieldCount: (sectionBody.match(/^- \*\*/gm) || []).length,
// Positioned between Core Value and Requirements.
orderedBetweenCoreValueAndRequirements:
lower.indexOf('## core value') < lower.indexOf('## business context') &&
lower.indexOf('## business context') < lower.indexOf('## requirements'),
hasGuidelinesEntry: /\*\*Business Context:\*\*/.test(content),
};
}
function parseMilestoneContract(content) {
const lower = content.toLowerCase();
const lines = content.split(/\r?\n/);
const reviewLine = lines.find(l =>
l.toLowerCase().includes('business context') &&
(l.toLowerCase().includes('if present') || l.toLowerCase().includes('only if')),
);
return {
mentionsBusinessContext: lower.includes('business context'),
hasConditionalReview: Boolean(reviewLine),
};
}
describe('enhancement #72 — Business Context template section', () => {
const tpl = parseTemplateContract(fs.readFileSync(TEMPLATE, 'utf-8'));
test('template includes a Business Context section', () => {
assert.ok(tpl.hasSection, 'template must contain a "## Business Context" section');
});
test('section is marked OPTIONAL with delete-for-non-business guidance', () => {
assert.ok(tpl.hasOptionalMarker, 'section must carry an OPTIONAL HTML comment telling non-business projects to delete it');
});
test('section carries exactly the four approved one-line fields', () => {
assert.ok(tpl.fields.customer, 'missing **Customer** field');
assert.ok(tpl.fields.revenueModel, 'missing **Revenue model** field');
assert.ok(tpl.fields.successMetric, 'missing **Success metric** field');
assert.ok(tpl.fields.strategyNotes, 'missing **Strategy notes** field');
assert.strictEqual(tpl.fieldCount, 4, 'section is capped at four fields (constraint reference, not a business plan)');
});
test('section is positioned between Core Value and Requirements', () => {
assert.ok(tpl.orderedBetweenCoreValueAndRequirements, 'Business Context must sit between Core Value and Requirements');
});
test('guidelines document the Business Context section', () => {
assert.ok(tpl.hasGuidelinesEntry, 'guidelines block must include a **Business Context:** entry');
});
test('milestone evolution reviews Business Context only when present', () => {
const ms = parseMilestoneContract(fs.readFileSync(COMPLETE_MILESTONE, 'utf-8'));
assert.ok(ms.mentionsBusinessContext, 'complete-milestone must mention Business Context in its review');
assert.ok(ms.hasConditionalReview, 'the Business Context milestone review must be conditional on the section being present');
});
});

View File

@@ -1,320 +0,0 @@
/**
* Tests for docs/issue-driven-orchestration.md (#2840).
*
* Structural-IR assertions per CONTRIBUTING.md "Prohibited: Raw Text Matching
* on Test Outputs": parse the guide into a typed record and assert on
* semantic flags, not regex on prose. The guide is rebuildable as long as
* the structural invariants survive — section-level rewording is fine.
*
* Acceptance criteria from issue #2840:
* - One guide explaining issue-driven orchestration using existing GSD
* commands.
* - Concrete end-to-end issue → workspace → plan/execute → verify/review
* → PR flow.
* - Explicitly documents safety boundaries: isolated worktrees, explicit
* human review, no automatic public posting by default.
* - Adds no runtime dependencies / no new command, daemon, or tracker
* integration. (Test-enforced via concept-mapping audit.)
*/
// allow-test-rule: structural-IR parser for a docs guide. The .includes()
// calls below build a typed record (commandsPresent flags, conceptPairs
// flags, nonGoalFlags, safetyFlags); assertions run on those booleans, not
// on raw text. This is the documented escape hatch in
// scripts/lint-no-source-grep.cjs for doc-shape tests.
'use strict';
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const GUIDE_PATH = path.join(__dirname, '..', 'docs', 'issue-driven-orchestration.md');
// ─── Helpers ────────────────────────────────────────────────────────────────
/**
* Extract a section starting at a given heading. Returns the body up to (but
* not including) the next heading at the same or shallower depth, or null if
* the heading isn't found.
*/
function extractSection(content, heading) {
const lines = content.split('\n');
const headingRe = new RegExp(`^(#+)\\s+${heading.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\$&')}\\s*$`);
let start = -1;
let depth = 0;
for (let i = 0; i < lines.length; i++) {
const m = lines[i].match(headingRe);
if (m) {
start = i + 1;
depth = m[1].length;
break;
}
}
if (start < 0) return null;
let end = lines.length;
for (let i = start; i < lines.length; i++) {
const m = lines[i].match(/^(#+)\s+/);
if (m && m[1].length <= depth) {
end = i;
break;
}
}
return lines.slice(start, end).join('\n');
}
/**
* Parse the guide into a typed record. Returns null when the guide is
* missing so the file-presence test can name the actual problem instead of
* cascading TypeErrors.
*/
function parseGuide() {
if (!fs.existsSync(GUIDE_PATH)) return null;
const content = fs.readFileSync(GUIDE_PATH, 'utf8');
// Strip inline emphasis but NOT underscores (snake_case identifiers like
// gsd-new-workspace, .planning/, etc. must survive).
const stripped = content.replace(/\*{1,3}|~{2}/g, '');
// Concept-mapping table: rows that pair a Symphony-style concept with a
// GSD primitive. Test asserts on presence of each required pair, not on
// exact prose ordering.
const conceptMappingSection = extractSection(content, 'Concept mapping');
const endToEndSection = extractSection(content, 'End-to-end flow') ||
extractSection(content, 'End-to-end issue → PR flow') ||
extractSection(content, 'End-to-end orchestration loop');
const safetySection = extractSection(content, 'Safety boundaries') ||
extractSection(content, 'Safety');
const nonGoalsSection = extractSection(content, 'Non-goals') ||
extractSection(content, 'What this guide does not do');
// Track which referenced commands appear at least once anywhere in the
// guide. This prevents drift if /gsd-* command names are renamed.
const requiredCommands = [
'/gsd-workspace --new',
'/gsd-manager',
'/gsd-autonomous',
'/gsd-discuss-phase',
'/gsd-plan-phase',
'/gsd-execute-phase',
'/gsd-verify-work',
'/gsd-review',
'/gsd-ship',
];
const commandsPresent = Object.fromEntries(
requiredCommands.map((c) => [c, content.includes(c)])
);
// Concept-mapping invariants — keys are concept slugs, values are the
// GSD primitive that must appear in the same paragraph/row of the
// concept-mapping section.
const conceptPairs = conceptMappingSection
? {
roadmap: /ROADMAP\.md/.test(conceptMappingSection),
statemd: /STATE\.md/.test(conceptMappingSection),
contextmd: /CONTEXT\.md/.test(conceptMappingSection),
planmd: /PLAN\.md/.test(conceptMappingSection),
workspaceCommand: /\/gsd-workspace\s+--new/.test(conceptMappingSection),
executionCommand:
/\/gsd-manager/.test(conceptMappingSection) ||
/\/gsd-autonomous/.test(conceptMappingSection),
verifyCommand: /\/gsd-verify-work/.test(conceptMappingSection),
reviewCommand: /\/gsd-review/.test(conceptMappingSection),
shipCommand: /\/gsd-ship/.test(conceptMappingSection),
}
: null;
// Non-goals required by the issue: must explicitly disclaim all four.
const nonGoalFlags = nonGoalsSection
? {
noVendoring: /vendor|copy/i.test(nonGoalsSection),
noDaemon: /daemon|polling/i.test(nonGoalsSection),
noTrackerDependency: /tracker.*depend|mandatory.*track/i.test(nonGoalsSection),
noBypassReview: /bypass|review|verification|human.*decision|human gate/i.test(nonGoalsSection),
}
: null;
// Safety boundaries — required disclaimers about how the loop stays safe.
const safetyFlags = safetySection
? {
isolatedWorktrees: /worktree|isolated/i.test(safetySection),
explicitReview: /review|human.*gate|human.*approval/i.test(safetySection),
noAutoPosting: /not.*automatic|no.*auto|explicit.*confirm|user.*confirm|human.*confirm/i.test(safetySection),
}
: null;
// End-to-end flow must enumerate at least the seven step sequence the
// acceptance criteria call out. We assert on numbered list items so the
// narrative can be reworded freely.
const numberedSteps = endToEndSection
? (endToEndSection.match(/^\s*\d+\.\s+/gm) || []).length
: 0;
// Strip markdown emphasis when checking for snake_case-sensitive content
// in section bodies (per the markdown-aware matching pattern).
const strippedConceptMapping = conceptMappingSection
? conceptMappingSection.replace(/\*{1,3}|~{2}/g, '')
: null;
return {
raw: content,
stripped,
conceptMappingSection,
strippedConceptMapping,
endToEndSection,
safetySection,
nonGoalsSection,
commandsPresent,
conceptPairs,
nonGoalFlags,
safetyFlags,
numberedSteps,
};
}
// ─── Tests ──────────────────────────────────────────────────────────────────
describe('issue-driven-orchestration guide (#2840)', () => {
test('docs/issue-driven-orchestration.md exists', () => {
assert.ok(
fs.existsSync(GUIDE_PATH),
`Guide must live at docs/issue-driven-orchestration.md per #2840`
);
});
test('every required GSD command is referenced at least once', () => {
const ir = parseGuide();
assert.ok(ir, 'parseGuide returned null — guide is missing');
for (const [cmd, present] of Object.entries(ir.commandsPresent)) {
assert.ok(present, `guide must reference ${cmd}`);
}
});
test('concept mapping section exists and pairs Symphony concepts with GSD primitives', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.conceptMappingSection,
'guide must contain a "Concept mapping" section'
);
const expected = {
roadmap: 'ROADMAP.md must appear in the concept mapping',
statemd: 'STATE.md must appear in the concept mapping',
contextmd: 'CONTEXT.md must appear in the concept mapping',
planmd: 'PLAN.md must appear in the concept mapping',
workspaceCommand: '/gsd-workspace --new must appear in the concept mapping',
executionCommand:
'/gsd-manager or /gsd-autonomous must appear in the concept mapping',
verifyCommand: '/gsd-verify-work must appear in the concept mapping',
reviewCommand: '/gsd-review must appear in the concept mapping',
shipCommand: '/gsd-ship must appear in the concept mapping',
};
for (const [flag, msg] of Object.entries(expected)) {
assert.equal(ir.conceptPairs[flag], true, msg);
}
});
test('safety boundaries section names isolation, review, and non-auto-posting', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.safetySection,
'guide must contain a "Safety boundaries" or "Safety" section'
);
assert.equal(
ir.safetyFlags.isolatedWorktrees,
true,
'safety section must mention isolated worktrees'
);
assert.equal(
ir.safetyFlags.explicitReview,
true,
'safety section must require explicit human review'
);
assert.equal(
ir.safetyFlags.noAutoPosting,
true,
'safety section must disclaim automatic public posting'
);
});
test('non-goals section disclaims vendoring, daemon, tracker dependency, and gate-bypass', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.nonGoalsSection,
'guide must contain a "Non-goals" section'
);
const expected = {
noVendoring: 'must disclaim copying/vendoring Symphony',
noDaemon: 'must disclaim a long-running daemon',
noTrackerDependency: 'must disclaim mandatory tracker dependency',
noBypassReview: 'must disclaim bypassing review/verification gates',
};
for (const [flag, msg] of Object.entries(expected)) {
assert.equal(ir.nonGoalFlags[flag], true, msg);
}
});
test('end-to-end flow enumerates at least 7 numbered steps (per acceptance criteria)', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
assert.ok(
ir.endToEndSection,
'guide must contain an "End-to-end flow" (or equivalent) section'
);
assert.ok(
ir.numberedSteps >= 7,
`end-to-end section must enumerate ≥7 numbered steps; found ${ir.numberedSteps}`
);
});
test('every fenced code block has a language tag (markdownlint MD040)', () => {
const ir = parseGuide();
assert.ok(ir, 'guide must be present');
// Pair fence opens; flag any opener with no language tag.
const fences = ir.raw.match(/^```.*$/gm) || [];
const openers = [];
for (let i = 0; i < fences.length; i++) {
// Even index = opener, odd = closer. An opener with empty trailing
// text is MD040.
if (i % 2 === 0) openers.push(fences[i]);
}
const bare = openers.filter((f) => /^```\s*$/.test(f));
assert.equal(
bare.length,
0,
`MD040: ${bare.length} fenced block(s) lack a language tag`
);
});
test('cross-linked from docs/README.md', () => {
const readme = path.join(__dirname, '..', 'docs', 'README.md');
if (!fs.existsSync(readme)) {
// docs/README.md is the discovery surface. Without a cross-link, the
// guide is invisible to users browsing docs/.
return; // tolerate absence; test below ensures FEATURES.md anchor.
}
const txt = fs.readFileSync(readme, 'utf8');
assert.ok(
/issue-driven-orchestration/.test(txt),
'docs/README.md must link to the new guide'
);
});
test('cross-linked from docs/USER-GUIDE.md', () => {
const guide = path.join(__dirname, '..', 'docs', 'USER-GUIDE.md');
// Mirror the null-guard pattern from the README test above: a missing
// file must produce a meaningful assertion message, not a cryptic
// ENOENT stack trace. (CR #3036.)
assert.ok(
fs.existsSync(guide),
'docs/USER-GUIDE.md must exist for cross-link validation'
);
const txt = fs.readFileSync(guide, 'utf8');
assert.ok(
/issue-driven-orchestration/.test(txt),
'docs/USER-GUIDE.md must link to the new guide'
);
});
});

View File

@@ -1,377 +0,0 @@
/**
* Feature test for issue #3023 — per-phase-type model map.
*
* Adds a `models` block to .planning/config.json that accepts phase-type
* keys (planning / discuss / research / execution / verification /
* completion). Resolution precedence:
*
* 1. Per-agent `model_overrides[agent]` (highest)
* 2. Phase-type `models[phase_type]` (NEW)
* 3. Profile table (`model_profile`)
* 4. Runtime default
*
* Tests are typed-IR / structural — assert on the value returned by
* resolveModelInternal, not stdout/grep. Each test seeds a temp project
* with a fixture .planning/config.json and asserts the resolver picks
* the right tier for each agent.
*/
'use strict';
process.env.GSD_TEST_MODE = '1';
const { test, describe, beforeEach, afterEach } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const {
resolveModelInternal,
} = require('../gsd-core/bin/lib/model-resolver.cjs');
const {
AGENT_TO_PHASE_TYPE,
VALID_PHASE_TYPES,
MODEL_PROFILES,
} = require('../gsd-core/bin/lib/model-profiles.cjs');
const { isValidConfigKey } = require('../gsd-core/bin/lib/config-schema.cjs');
const { createTempDir, cleanup } = require('./helpers.cjs');
const makeTmp = (prefix) => createTempDir(`gsd-3023-${prefix}-`);
function writeConfig(projectDir, config) {
const planningDir = path.join(projectDir, '.planning');
fs.mkdirSync(planningDir, { recursive: true });
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
}
function rmr(p) {
cleanup(p);
}
// ─── Schema: AGENT_TO_PHASE_TYPE table + VALID_PHASE_TYPES ──────────────────
describe('#3023 phase-type schema: every agent has a phase-type assignment', () => {
test('AGENT_TO_PHASE_TYPE is exported as a non-empty object', () => {
assert.equal(typeof AGENT_TO_PHASE_TYPE, 'object');
assert.ok(AGENT_TO_PHASE_TYPE !== null);
assert.ok(Object.keys(AGENT_TO_PHASE_TYPE).length > 0);
});
test('VALID_PHASE_TYPES exposes the six named slots from the issue', () => {
// The issue specified exactly these slots. Adding new slots here is a
// schema change that must coordinate with config-schema's dynamic
// pattern and the docs.
assert.deepStrictEqual(
[...VALID_PHASE_TYPES].sort(),
['completion', 'discuss', 'execution', 'planning', 'research', 'verification'].sort()
);
});
test('every agent in MODEL_PROFILES has a phase-type assignment', () => {
const missing = Object.keys(MODEL_PROFILES).filter(
(agent) => !AGENT_TO_PHASE_TYPE[agent]
);
assert.deepStrictEqual(missing, [],
`every agent in MODEL_PROFILES must have a phase-type — missing: ${JSON.stringify(missing)}`);
});
test('every assigned phase-type is one of the six valid slots', () => {
const invalid = Object.entries(AGENT_TO_PHASE_TYPE).filter(
([, phaseType]) => !VALID_PHASE_TYPES.has(phaseType)
);
assert.deepStrictEqual(invalid, [],
`phase-type assignments must use VALID_PHASE_TYPES — invalid: ${JSON.stringify(invalid)}`);
});
});
// ─── Resolver behavior: phase-type drives tier ──────────────────────────────
describe('#3023 resolver: models.<phase_type> overrides profile-based tier', () => {
let projectDir;
beforeEach(() => { projectDir = makeTmp('resolver'); });
afterEach(() => { rmr(projectDir); });
test('phase-type alone — research agents get the phase-type tier, planner gets profile default', () => {
writeConfig(projectDir, {
model_profile: 'balanced',
models: { research: 'haiku' },
});
// gsd-phase-researcher is a research agent — should pick up 'haiku'
// from the phase-type slot, not 'sonnet' from the balanced profile.
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'haiku');
// gsd-codebase-mapper is also research → haiku
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
// gsd-planner is planning, no models.planning set → falls through to
// profile (balanced → opus per MODEL_PROFILES).
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
});
test('per-agent override beats phase-type (acceptance criterion b)', () => {
writeConfig(projectDir, {
model_profile: 'balanced',
models: { research: 'haiku' },
model_overrides: { 'gsd-phase-researcher': 'opus' },
});
// The targeted per-agent override wins for that one agent.
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'opus');
// Other research agents still pick up the phase-type tier.
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
assert.equal(resolveModelInternal(projectDir, 'gsd-research-synthesizer'), 'haiku');
});
test('phase-type beats profile (acceptance criterion c)', () => {
// model_profile=quality would normally make research agents 'opus'.
// models.research='haiku' must win.
writeConfig(projectDir, {
model_profile: 'quality',
models: { research: 'haiku' },
});
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'haiku');
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
// gsd-planner is planning, no slot set, profile=quality → opus.
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
});
test('issue example: opus for planning/discuss/execution, sonnet for research/verification/completion', () => {
writeConfig(projectDir, {
model_profile: 'balanced',
models: {
planning: 'opus',
discuss: 'opus',
execution: 'opus',
research: 'sonnet',
verification: 'sonnet',
completion: 'sonnet',
},
});
// Planning agents → opus
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
// Execution agents → opus
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'opus');
// Research agents → sonnet
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
// Verification agents → sonnet
assert.equal(resolveModelInternal(projectDir, 'gsd-verifier'), 'sonnet');
});
test('phase-type "inherit" is honored (preserves existing inherit semantics)', () => {
writeConfig(projectDir, {
model_profile: 'balanced',
models: { research: 'inherit' },
});
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'inherit');
});
test('empty models block is a no-op (acceptance criterion: backward compat)', () => {
writeConfig(projectDir, {
model_profile: 'balanced',
models: {},
});
// Behavior must match no-models config (balanced profile).
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
});
test('no models block at all is a no-op (acceptance criterion: backward compat)', () => {
writeConfig(projectDir, {
model_profile: 'balanced',
});
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
});
test('unrecognized tier value falls through to profile (typo safety) — CR follow-up', () => {
// The VALID_TIERS guard in resolveModelInternal must reject any value
// that isn't a known tier alias and fall back to the profile tier.
// Without this guard a typo like "haiku3" would pollute the runtime
// resolution chain. Locks the guard in so a future regression that
// removes it is caught.
writeConfig(projectDir, {
model_profile: 'balanced',
models: { research: 'haiku3' }, // typo; not a valid tier alias
});
// Falls back to balanced → sonnet for research agents.
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku',
'gsd-codebase-mapper at balanced is haiku per profile, unaffected by typo');
});
test('full model ID in models.<phase_type> is rejected; falls through to profile — CR follow-up', () => {
// Full IDs are not valid in models.<phase_type>; they belong in
// model_overrides per agent. The guard ensures we don't accidentally
// hand a full ID into the runtime-tier resolution chain.
writeConfig(projectDir, {
model_profile: 'balanced',
models: { research: 'openai/gpt-5' },
});
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
});
// ─── CR Major: phase-type beats inherit profile ─────────────────────────
// Pre-fix bug: model_profile='inherit' + models.execution='opus' returned
// 'inherit' because the profile short-circuit fired BEFORE the phase-type
// override could win, violating the documented precedence where
// models[phase_type] beats model_profile.
test('phase-type override wins over profile=inherit (CR Major) — model resolver', () => {
writeConfig(projectDir, {
model_profile: 'inherit',
models: { execution: 'opus' },
});
// gsd-executor (execution) must get the phase-type opus, not inherit.
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'opus');
});
test('phase-type "haiku" wins over profile=inherit; agents without a slot still inherit', () => {
writeConfig(projectDir, {
model_profile: 'inherit',
models: { research: 'haiku' },
});
// research agents → haiku (phase-type wins)
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'haiku');
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
// planning agent has no slot set → falls through to profile=inherit.
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'inherit');
});
test('profile=inherit with no models block still returns inherit (no regression)', () => {
writeConfig(projectDir, {
model_profile: 'inherit',
});
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'inherit');
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'inherit');
});
test('profile=inherit with models block but agent has no slot → inherit', () => {
writeConfig(projectDir, {
model_profile: 'inherit',
models: { research: 'haiku' },
});
// gsd-executor (execution slot) is not set → falls through to inherit.
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'inherit');
});
});
// ─── #443 Unified effort: resolveEffortInternal + renderEffortForRuntime ────
const { resolveEffortInternal } = require('../gsd-core/bin/lib/model-resolver.cjs');
const { renderEffortForRuntime } = require('../gsd-core/bin/lib/model-catalog.cjs');
describe('#3023 + #443: unified effort resolver (resolveEffortInternal) for Codex', () => {
let projectDir;
beforeEach(() => { projectDir = makeTmp('effort'); });
afterEach(() => { rmr(projectDir); });
test('resolveEffortInternal exported from model-resolver.cjs', () => {
assert.equal(typeof resolveEffortInternal, 'function');
});
test('effort derives from AGENT_DEFAULT_TIERS (routing), not phase-type; gsd-executor is standard → high', () => {
// Under unification, effort is config-driven via routing_tier_defaults.
// gsd-executor has routing tier 'standard' → default effort 'high', regardless
// of models.execution phase-type or model_profile setting.
writeConfig(projectDir, {
runtime: 'codex',
model_profile: 'balanced',
models: { execution: 'opus' },
});
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
// standard tier → 'high' (not 'xhigh' from opus, not 'medium' from old catalog)
assert.equal(eff, 'high');
const rendered = renderEffortForRuntime('codex', eff);
assert.equal(rendered.param, 'model_reasoning_effort');
assert.equal(rendered.value, 'high');
});
test('effort resolves universally even when models.execution=inherit', () => {
// Under unification, models.execution='inherit' does not affect effort resolution.
// Effort always resolves from routing_tier_defaults: gsd-executor (standard) → 'high'.
writeConfig(projectDir, {
runtime: 'codex',
model_profile: 'balanced',
models: { execution: 'inherit' },
});
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
assert.equal(eff, 'high');
const rendered = renderEffortForRuntime('codex', eff);
assert.equal(rendered.param, 'model_reasoning_effort');
assert.equal(rendered.value, 'high');
});
test('per-agent model_overrides does not affect effort (effort is routing-tier-based)', () => {
// Under unification, effort does not check model_overrides.
// gsd-executor (standard tier) → 'high' regardless.
writeConfig(projectDir, {
runtime: 'codex',
model_profile: 'balanced',
models: { execution: 'opus' },
model_overrides: { 'gsd-executor': 'openai/gpt-5' },
});
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
assert.equal(eff, 'high');
const rendered = renderEffortForRuntime('codex', eff);
assert.equal(rendered.param, 'model_reasoning_effort');
assert.equal(rendered.value, 'high');
});
test('Claude runtime: effort is first-class (emits output_config.effort, not null)', () => {
// Under unification, Claude effort is first-class via output_config.effort.
// No `runtime` set → defaults to claude (no runtime key → undefined runtime).
writeConfig(projectDir, {
model_profile: 'balanced',
models: { execution: 'opus' },
});
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
// effort resolves universally; claude render gives output_config.effort
const rendered = renderEffortForRuntime(undefined, eff);
// undefined runtime yields param=null (no runtime key set)
assert.equal(rendered.param, null);
// But if explicitly set to 'claude':
const renderedClaude = renderEffortForRuntime('claude', eff);
assert.equal(renderedClaude.param, 'output_config.effort');
assert.equal(renderedClaude.value, 'high');
});
test('profile=inherit does not affect effort; effort resolves from routing tier', () => {
// Under unification, effort is completely independent of model_profile.
// gsd-executor (standard routing tier) → 'high' even with model_profile='inherit'.
writeConfig(projectDir, {
runtime: 'codex',
model_profile: 'inherit',
models: { execution: 'opus' },
});
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
assert.equal(eff, 'high',
'profile=inherit must not affect effort; standard routing tier → high');
const rendered = renderEffortForRuntime('codex', eff);
assert.equal(rendered.param, 'model_reasoning_effort');
assert.equal(rendered.value, 'high');
});
});
// ─── Schema validation ──────────────────────────────────────────────────────
describe('#3023 config-schema: models.<phase_type> validation', () => {
test('models.planning is a valid config key', () => {
assert.equal(isValidConfigKey('models.planning'), true);
});
test('all six phase-type slots are valid config keys', () => {
for (const slot of ['planning', 'discuss', 'research', 'execution', 'verification', 'completion']) {
assert.equal(isValidConfigKey(`models.${slot}`), true,
`models.${slot} must be a valid config key`);
}
});
test('unknown phase-type is rejected (acceptance criterion d)', () => {
assert.equal(isValidConfigKey('models.deployment'), false,
'unknown phase-type must NOT be accepted');
assert.equal(isValidConfigKey('models.gsd-planner'), false,
'agent name in models.* must NOT be accepted (use model_overrides for agents)');
});
test('models alone (without a slot) is not a valid config-set key', () => {
// Setting the whole block isn't a granular set; users edit JSON directly.
assert.equal(isValidConfigKey('models'), false);
});
});

View File

@@ -1,301 +0,0 @@
/**
* Documentation regression test for issue #3025 — MCP token-budget guidance.
*
* Verifies that gsd-core/references/context-budget.md contains the
* structural elements the issue requires:
*
* 1. A section explaining MCP/tool schemas as a context-budget concern
* 2. References to the harness-side toggles (enabledMcpjsonServers,
* disabledMcpjsonServers in .claude/settings.json)
* 3. A pre-phase audit checklist (browser/playwright, platform-specific,
* project-specific)
* 4. An explicit note that GSD does NOT manage MCP enablement — this is
* a Claude Code harness concern (with a cross-link)
* 5. Note the interaction with model_profile (compounding levers)
*
* Tests parse the doc into a typed section record (parseMcpSection) and
* assert on flag booleans, not raw text matches. Adheres to
* CONTRIBUTING.md "no-source-grep" — describes invariants, not wording,
* so the prose can be reworded freely as long as the semantics survive.
*
* Companion to docs/USER-GUIDE.md task section, which is exercised by the
* same parser shape (separate test below).
*/
'use strict';
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const ROOT = path.join(__dirname, '..');
const CONTEXT_BUDGET_MD = path.join(ROOT, 'gsd-core', 'references', 'context-budget.md');
const USER_GUIDE_MD = path.join(ROOT, 'docs', 'USER-GUIDE.md');
/**
* Extract the MCP-budget section from a markdown file by header text.
* Returns null if the section is missing. Section runs from the matching
* `## ` header up to the next `## ` header (or EOF).
*/
function extractSection(filePath, headerSubstring) {
const content = fs.readFileSync(filePath, 'utf8');
const lines = content.split(/\r?\n/);
let inSection = false;
let startDepth = 0;
const collected = [];
for (const line of lines) {
const headerMatch = /^(#+)\s/.exec(line);
if (headerMatch) {
const depth = headerMatch[1].length;
if (inSection) {
// Section ends at a header at the same or shallower depth.
// Subsections at deeper depth are part of the section.
if (depth <= startDepth) break;
} else if (line.toLowerCase().includes(headerSubstring.toLowerCase())) {
inSection = true;
startDepth = depth;
}
}
if (inSection) collected.push(line);
}
return collected.length > 0 ? collected.join('\n') : null;
}
/**
* Parse the MCP-budget section into a typed semantic-flag record.
* Each flag answers a single behavioral question that #3025 requires
* the prose to encode.
*/
function parseMcpBudgetSection(section) {
if (!section || typeof section !== 'string') {
return {
ok: false,
sectionLength: 0,
explainsMcpAsBudgetConcern: false,
namesEnabledMcpjsonServers: false,
namesDisabledMcpjsonServers: false,
namesClaudeSettingsJson: false,
includesPrePhaseAudit: false,
auditMentionsBrowserOrPlaywright: false,
auditMentionsPlatformSpecific: false,
auditMentionsCrossProject: false,
explainsHarnessNotGsd: false,
mentionsModelProfileInteraction: false,
crossLinksContextBudget: false,
};
}
// CR follow-up: strip inline markdown emphasis (`**`, `*`, `~~`) and
// backticks before phrase-matching so e.g. "GSD does **not** manage"
// is caught by the primary `gsd does not manage` alternative below.
// WITHOUT this, the markdown-bold breaks the contiguous match and the
// test only passes via the fallback branch (silent dead code).
// Underscores are intentionally NOT stripped — `model_profile` and
// other snake_case identifiers must survive intact so the
// model_profile interaction check still finds them.
const stripped = section.replace(/\*{1,3}|~{2}|`/g, '');
// (1) Explains MCP as budget concern — must mention BOTH "MCP" / "tool
// schema" AND a token/cost framing.
const explainsMcpAsBudgetConcern =
/\bmcp\b|tool schema|tool schemas/i.test(stripped) &&
/\btoken|context budget|per[- ]turn|cost\b/i.test(stripped);
// (2) Names the harness keys verbatim
const namesEnabledMcpjsonServers = /enabledMcpjsonServers/.test(stripped);
const namesDisabledMcpjsonServers = /disabledMcpjsonServers/.test(stripped);
// (3) Names the settings file location
const namesClaudeSettingsJson = /\.claude\/settings\.json/.test(stripped);
// (4) Audit checklist — must mention all three classes the issue
// calls out, plus a "before this phase / pre-phase" framing
const includesPrePhaseAudit =
/audit|checklist|review (your )?mcp|before (starting|beginning) (a |the )?phase/i.test(stripped);
const auditMentionsBrowserOrPlaywright = /\bbrowser\b|playwright/i.test(stripped);
const auditMentionsPlatformSpecific = /platform[- ]specific|mac[- ]?tools|windows[- ]?tools|os[- ]specific/i.test(stripped);
const auditMentionsCrossProject = /(other|different|cross[- ])\s*project|stale (project )?mcp/i.test(stripped);
// (5) Harness vs GSD distinction — must explicitly state GSD doesn't
// own this knob and point at the harness
const explainsHarnessNotGsd =
/(gsd does(?:n[''’]t| not) (own|manage|control)|harness (concern|setting|controlled)|not a gsd (setting|knob))/i.test(stripped);
// (6) Compounding with model_profile
const mentionsModelProfileInteraction =
/model[_ ]profile/i.test(stripped) &&
/compound|multiplier|stack|every[- ]turn|regardless of (which )?model|in addition/i.test(stripped);
// (7) Cross-link to the canonical reference doc — task-guide section
// must point readers at context-budget.md for the full audit. Encoded
// as a named flag (CR follow-up) so the assertion sits alongside the
// other parsed invariants rather than as a one-off inline regex.
const crossLinksContextBudget = /context-budget/i.test(stripped);
return {
ok: true,
sectionLength: section.length,
explainsMcpAsBudgetConcern,
namesEnabledMcpjsonServers,
namesDisabledMcpjsonServers,
namesClaudeSettingsJson,
includesPrePhaseAudit,
auditMentionsBrowserOrPlaywright,
auditMentionsPlatformSpecific,
auditMentionsCrossProject,
explainsHarnessNotGsd,
mentionsModelProfileInteraction,
crossLinksContextBudget,
};
}
// ─── context-budget.md ──────────────────────────────────────────────────────
describe('#3025 context-budget.md: MCP token-budget section exists with required content', () => {
test('the file exists', () => {
assert.ok(fs.existsSync(CONTEXT_BUDGET_MD), `expected file at ${CONTEXT_BUDGET_MD}`);
});
test('has a section header that mentions MCP', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
assert.ok(section, 'must have a `## ...MCP...` heading; section was not found');
});
test('explains MCP/tool schemas as a context-budget concern (#3025 requirement 1)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.explainsMcpAsBudgetConcern, true,
`must explain MCP/tool schemas as a token/context-budget concern; section was:\n${section}`);
});
test('names enabledMcpjsonServers and disabledMcpjsonServers (#3025 requirement 2)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.namesEnabledMcpjsonServers, true,
'section must reference `enabledMcpjsonServers` so users know the exact key');
assert.equal(parsed.namesDisabledMcpjsonServers, true,
'section must reference `disabledMcpjsonServers` for parity');
assert.equal(parsed.namesClaudeSettingsJson, true,
'section must name `.claude/settings.json` as the location of the toggle');
});
test('includes a pre-phase audit checklist with all three classes (#3025 requirement 3)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.includesPrePhaseAudit, true,
'section must include audit/checklist framing for pre-phase MCP review');
assert.equal(parsed.auditMentionsBrowserOrPlaywright, true,
'audit must mention browser/playwright tools as a candidate for disabling');
assert.equal(parsed.auditMentionsPlatformSpecific, true,
'audit must mention platform-specific tools (Mac/Windows/OS-specific)');
assert.equal(parsed.auditMentionsCrossProject, true,
'audit must mention stale/cross-project MCPs from other projects');
});
test('explains GSD does not own MCP enablement — harness concern (#3025 requirement 4)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.explainsHarnessNotGsd, true,
'section must explicitly state GSD does not manage MCP enablement (harness concern)');
});
test('notes interaction with model_profile (compounding levers) (#3025 requirement 5)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.mentionsModelProfileInteraction, true,
'section must note that trimming MCPs compounds with model_profile choice');
});
test('full semantic record matches the #3025 contract — typed snapshot', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
const contract = {
ok: parsed.ok,
explainsMcpAsBudgetConcern: parsed.explainsMcpAsBudgetConcern,
namesEnabledMcpjsonServers: parsed.namesEnabledMcpjsonServers,
namesDisabledMcpjsonServers: parsed.namesDisabledMcpjsonServers,
namesClaudeSettingsJson: parsed.namesClaudeSettingsJson,
includesPrePhaseAudit: parsed.includesPrePhaseAudit,
auditMentionsBrowserOrPlaywright: parsed.auditMentionsBrowserOrPlaywright,
auditMentionsPlatformSpecific: parsed.auditMentionsPlatformSpecific,
auditMentionsCrossProject: parsed.auditMentionsCrossProject,
explainsHarnessNotGsd: parsed.explainsHarnessNotGsd,
mentionsModelProfileInteraction: parsed.mentionsModelProfileInteraction,
};
assert.deepStrictEqual(contract, {
ok: true,
explainsMcpAsBudgetConcern: true,
namesEnabledMcpjsonServers: true,
namesDisabledMcpjsonServers: true,
namesClaudeSettingsJson: true,
includesPrePhaseAudit: true,
auditMentionsBrowserOrPlaywright: true,
auditMentionsPlatformSpecific: true,
auditMentionsCrossProject: true,
explainsHarnessNotGsd: true,
mentionsModelProfileInteraction: true,
}, 'context-budget.md MCP section contract violated');
});
});
// ─── docs/USER-GUIDE.md task section ────────────────────────────────────────
describe('#3025 docs/USER-GUIDE.md: companion task section exists', () => {
test('USER-GUIDE.md has an MCP-trimming task section', () => {
const section = extractSection(USER_GUIDE_MD, 'mcp');
assert.ok(section,
'USER-GUIDE.md must have a `### ...MCP...` task section so users find it via the guide');
});
test('USER-GUIDE.md task section names the harness key and cross-links the reference', () => {
const section = extractSection(USER_GUIDE_MD, 'mcp');
const parsed = parseMcpBudgetSection(section);
assert.equal(parsed.namesEnabledMcpjsonServers, true,
'task section must mention the harness key by name');
// Cross-link to the reference doc — assert on the parsed flag so
// the invariant lives alongside the other named flags (CR follow-up
// on the no-source-grep standard).
assert.equal(parsed.crossLinksContextBudget, true,
'task section must cross-link to context-budget.md');
});
});
// ─── markdownlint pre-flight (per bundle-docs-with-code skill) ──────────────
describe('#3025 markdownlint pre-flight: MD040 + MD056', () => {
test('every fenced code block in the new MCP section has a language tag (MD040)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
// Guard: extractSection returns null when the section is missing.
// Without this, `section.match(...)` would throw a TypeError instead
// of producing a meaningful assertion failure (CR follow-up).
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD040');
const fences = (section.match(/^```([a-zA-Z0-9_+-]*)?\s*$/gm) || []);
// Pairs of fences open/close; odd-indexed ones close blocks. Every
// OPENING fence must have a language tag. Closing fences are bare ```.
// Walk pairs: even index = opener, odd = closer.
const openers = fences.filter((_, i) => i % 2 === 0);
const missing = openers.filter((line) => /^```\s*$/.test(line));
assert.deepStrictEqual(missing, [],
`every fenced code block opener must have a language tag (MD040). Missing: ${JSON.stringify(missing)}`);
});
test('every markdown table row in the new MCP section has the same column count as its header (MD056)', () => {
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
// Guard: same null-section concern as MD040 above (CR follow-up).
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD056');
const lines = section.split(/\r?\n/);
// Walk through and detect tables: header row followed by a separator
// (--- pattern) followed by data rows. Count `|` per line.
const issues = [];
for (let i = 0; i < lines.length - 1; i += 1) {
const header = lines[i];
const sep = lines[i + 1];
if (!/^\s*\|.*\|\s*$/.test(header)) continue;
if (!/^\s*\|[\s\-:|]+\|\s*$/.test(sep)) continue;
const headerCols = (header.match(/\|/g) || []).length;
// Walk data rows
for (let j = i + 2; j < lines.length; j += 1) {
const row = lines[j];
if (!/^\s*\|.*\|\s*$/.test(row)) break;
const rowCols = (row.match(/\|/g) || []).length;
if (rowCols !== headerCols) {
issues.push({ line: j, expected: headerCols, actual: rowCols, row });
}
}
}
assert.deepStrictEqual(issues, [],
`table rows must match header column count (MD056). Issues: ${JSON.stringify(issues, null, 2)}`);
});
});

View File

@@ -1,130 +0,0 @@
/**
* Universal CLI negative-matrix sweep across the seven command families
* named in #3593: phase, roadmap, state, config, workstream, init,
* validate.
*
* The full per-case matrix for each family belongs in dedicated files
* (the `config` family is the template — see
* `feat-3593-cli-negative-config.test.cjs`). This file pins a much
* narrower contract that EVERY family must satisfy:
*
* 1. Bare top-level command with no subcommand: must not crash with
* a V8 stack trace, must exit non-zero (or, where the bare form
* is legitimate, return a clean payload).
* 2. Unknown subcommand at command depth: must emit a typed reason
* under --json-errors and must not crash.
* 3. Shell-metacharacter as an argv element: must not be executed.
* Sentinel-file probe in the project temp dir proves this.
*
* Future per-family files will deepen each family's matrix; this file
* is the floor.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { runCli } = require('./helpers/cli-negative.cjs');
const { createTempProject, createTempGitProject, cleanup } = require('./helpers.cjs');
/**
* Each entry names a top-level command, a representative subcommand
* (for the "unknown sub" probe), and the temp-project factory.
*
* The `representativeSubcommand` is not asserted on directly — it
* exists so the test layout reads "for each family, run probe X" and
* so a future maintainer adding a family doesn't need to invent one.
*/
const FAMILIES = [
{ name: 'phase', bare: ['phase'], unknown: ['phase', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'roadmap', bare: ['roadmap'], unknown: ['roadmap', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'state', bare: ['state'], unknown: ['state', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'config', bare: ['config'], unknown: ['config', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'workstream', bare: ['workstream'], unknown: ['workstream', 'this-sub-does-not-exist'], projectFactory: createTempGitProject },
{ name: 'init', bare: ['init'], unknown: ['init', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'validate', bare: ['validate'], unknown: ['validate', 'this-sub-does-not-exist'], projectFactory: createTempProject },
];
describe('feat-3593: bare top-level command does not crash', () => {
for (const fam of FAMILIES) {
test(`${fam.name}: bare invocation exits cleanly without a stack trace`, (t) => {
const projectDir = fam.projectFactory(`cli-neg-univ-bare-${fam.name}-`);
t.after(() => cleanup(projectDir));
const result = runCli(fam.bare, { cwd: projectDir });
assert.equal(result.hasStackTrace, false, `${fam.name} bare must not leak a V8 stack frame`);
assert.equal(result.signal, null, `${fam.name} bare must not be killed by a signal`);
// We do not pin exit code here: some families legitimately treat the
// bare form as a list/status command (status 0); others reject it as
// a usage error (status ≠ 0). What we pin is "no crash."
});
}
});
describe('feat-3593: unknown subcommand emits a typed reason', () => {
for (const fam of FAMILIES) {
test(`${fam.name}: unknown subcommand fails with reason set and no stack trace`, (t) => {
const projectDir = fam.projectFactory(`cli-neg-univ-unk-${fam.name}-`);
t.after(() => cleanup(projectDir));
const result = runCli(fam.unknown, { cwd: projectDir });
assert.notEqual(result.status, 0, `${fam.name} unknown sub must exit non-zero`);
assert.equal(result.hasStackTrace, false, `${fam.name} unknown sub must not leak a stack frame`);
// When --json-errors is on, every typed-failure path lands a non-empty
// reason string from ERROR_REASON. A null reason here means the family
// is using throw/console.error somewhere — a TDD signal to wire that
// failure path through error(msg, ERROR_REASON.X).
assert.equal(typeof result.reason, 'string', `${fam.name}: reason must be a typed enum string (got: ${result.reason})`);
assert.notEqual(result.reason, '', `${fam.name}: reason must be non-empty`);
});
}
});
describe('feat-3593: shell-metacharacter argv values are NOT executed', () => {
for (const fam of FAMILIES) {
test(`${fam.name}: shell-payload as subcommand argv does NOT execute the payload`, (t) => {
const projectDir = fam.projectFactory(`cli-neg-univ-shell-${fam.name}-`);
t.after(() => cleanup(projectDir));
// Place the payload where the subcommand value goes. The payload
// would, if shell-interpreted, create a sentinel file in the
// project dir. Argv-based invocation must treat it as opaque text.
const sentinelPayload = `$(touch ${projectDir}/INJ-${fam.name})`;
const argv = [fam.name, sentinelPayload];
const result = runCli(argv, { cwd: projectDir });
// No stack trace — opaque text must not crash.
assert.equal(result.hasStackTrace, false, `${fam.name}: shell payload must not crash`);
// The sentinel file must NOT exist. We check the project dir's
// listing rather than fs.existsSync of a single path so the test
// surfaces any spelling drift.
const entries = fs.readdirSync(projectDir);
const sentinels = entries.filter((n) => n.startsWith('INJ-'));
assert.deepEqual(
sentinels,
[],
`${fam.name}: shell payload was executed — sentinel files exist: ${sentinels.join(', ')}`,
);
});
}
});
// ─── Cross-family invariants on the global --cwd flag ──────────────────────
test('--cwd with an empty value fails the same way regardless of command family', () => {
for (const fam of FAMILIES) {
const result = runCli(['--cwd', '', ...fam.bare], { cwd: process.cwd() });
assert.notEqual(result.status, 0, `${fam.name}: empty --cwd must fail`);
assert.equal(result.hasStackTrace, false, `${fam.name}: empty --cwd must not crash`);
assert.equal(result.reason, 'usage', `${fam.name}: empty --cwd reason should be 'usage', got: ${result.reason}`);
}
});
test('--cwd pointing at a non-existent path fails uniformly across families', () => {
const nonExistent = path.join(require('os').tmpdir(), 'cli-neg-univ-no-such-' + Date.now() + '-' + Math.random());
assert.equal(fs.existsSync(nonExistent), false, 'pre-check: temp path must not exist');
for (const fam of FAMILIES) {
const result = runCli(['--cwd', nonExistent, ...fam.bare], { cwd: process.cwd() });
assert.notEqual(result.status, 0, `${fam.name}: invalid --cwd must fail`);
assert.equal(result.hasStackTrace, false);
assert.equal(result.reason, 'usage', `${fam.name}: invalid --cwd reason should be 'usage'`);
}
});

View File

@@ -1,275 +0,0 @@
/**
* Adversarial roadmap-parser tests (#3594).
*
* Loads each fixture in `tests/fixtures/adversarial/roadmap/` as the
* project's `.planning/ROADMAP.md` and pins invariants on the public
* `gsd-tools roadmap get-phase <N>` surface — which routes through the
* SDK bridge when available and the CJS handler otherwise.
*
* Per CONTRIBUTING.md §"Testing Standards / Parser and project-file
* inputs", the assertion target is the typed JSON shape the CLI emits,
* not stderr prose. The harness in `tests/helpers/cli-negative.cjs`
* (introduced by #3627 / #3593) is reused here for consistency.
*
* Several fixtures encode known historical regressions:
* - fenced-code-block headings shadowing real phases (#2787)
* - decimal phase prefix collisions (#3537)
* - HTML-comment heading false positives
*
* Pre-existing parser bugs surfaced by these fixtures are NOT fixed in
* this PR — fixing them is out of scope for "add adversarial test
* coverage." Where a fixture exposes a still-open bug, the test
* asserts the *currently observed* behavior with a comment naming the
* open issue, so the flip from RED→GREEN is a one-line change the day
* the real fix lands.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { runCli } = require('./helpers/cli-negative.cjs');
const { createTempProject, cleanup } = require('./helpers.cjs');
const FIXTURE_DIR = path.join(__dirname, 'fixtures', 'adversarial', 'roadmap');
function loadFixture(name) {
return fs.readFileSync(path.join(FIXTURE_DIR, name), 'utf-8');
}
/**
* Create a temp project whose ROADMAP.md is the named fixture's content.
* Returns the project directory; caller is responsible for cleanup.
*/
function projectWithFixture(t, fixtureName) {
const projectDir = createTempProject('roadmap-adv-' + fixtureName.replace(/\W+/g, '-') + '-');
t.after(() => cleanup(projectDir));
fs.writeFileSync(path.join(projectDir, '.planning', 'ROADMAP.md'), loadFixture(fixtureName));
return projectDir;
}
/**
* Run `gsd-tools roadmap get-phase <N>` and parse the JSON payload.
* Returns `{ ok, exit, parsed, raw }` so tests can assert on either
* the exit code or the structured payload.
*/
function getPhase(projectDir, phaseNum) {
// No --json-errors — the get-phase command outputs JSON on success
// via the normal stdout path. Reading the parsed payload is what the
// workflows downstream do, so that's what we test.
const result = runCli(['roadmap', 'get-phase', phaseNum], { cwd: projectDir, jsonErrors: false });
let parsed = null;
try {
parsed = JSON.parse(result.stdout);
} catch {
// Leave parsed null; tests that depend on it must handle that.
}
return {
exit: result.status,
ok: result.status === 0,
parsed,
raw: result.stdout,
stderr: result.stderr,
hasStackTrace: result.hasStackTrace,
};
}
// ─── Fenced code block heading shadowing ────────────────────────────────────
describe('feat-3594: roadmap parser and fenced-code-block headings (#2787)', () => {
test('phase 1 in real prose is found even when ## Phase 999 appears inside a ``` block', (t) => {
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
const result = getPhase(projectDir, '1');
assert.equal(result.hasStackTrace, false, 'no V8 stack trace');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true, 'phase 1 must be found');
assert.equal(result.parsed.phase_number, '1');
assert.match(result.parsed.phase_name, /real phase one/);
});
test('phase 999 inside a fenced block is ignored', (t) => {
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
const result = getPhase(projectDir, '999');
assert.equal(result.hasStackTrace, false, 'no stack trace');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, false, 'phase headings inside fenced blocks must not be parsed');
});
test('fenced example heading does not shadow the real phase details and backlog phase stays unresolved (#1588)', (t) => {
const projectDir = createTempProject('roadmap-1588-');
t.after(() => cleanup(projectDir));
fs.writeFileSync(
path.join(projectDir, '.planning', 'STATE.md'),
[
'---',
'gsd_state_version: 1.0',
'milestone: v1.1',
'status: planning',
'---',
'',
].join('\n')
);
fs.writeFileSync(
path.join(projectDir, '.planning', 'ROADMAP.md'),
[
'# Roadmap',
'',
'<details open>',
'<summary>v1.1 Current (Phases 8-9) - PLANNED</summary>',
'',
'- [ ] **Phase 9: Real Phase**',
'',
'</details>',
'',
'## Phase Details',
'',
'```markdown',
'### Phase 9: Fenced Example Phase',
'**Goal:** This example must not be treated as roadmap structure.',
'```',
'',
'### Phase 9: Real Phase',
'**Goal:** Use the real phase details outside the fenced block.',
'**Requirements:** REAL-01',
'',
'## Backlog',
'',
'### Phase 999.1: Backlog Thing',
'**Goal:** Future backlog item.',
'',
].join('\n')
);
const phase9 = getPhase(projectDir, '9');
assert.ok(phase9.parsed, `expected JSON payload, got: ${phase9.raw}`);
assert.equal(phase9.parsed.found, true, 'phase 9 must be found');
assert.equal(phase9.parsed.phase_name, 'Real Phase');
assert.equal(phase9.parsed.goal, 'Use the real phase details outside the fenced block.');
assert.match(phase9.parsed.section, /REAL-01/, 'real phase section must be returned');
const backlog = getPhase(projectDir, '999.1');
assert.ok(backlog.parsed, `expected JSON payload, got: ${backlog.raw}`);
assert.equal(backlog.parsed.found, false, 'backlog sentinel phase must not resolve as an active roadmap phase');
});
});
// ─── Decimal phase prefix collisions ────────────────────────────────────────
describe('feat-3594: roadmap parser handles decimal phase prefix collisions (#3537)', () => {
test('asking for phase "2" returns the integer phase, NOT phase 2.1 or 2.10', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '2');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '2');
assert.match(result.parsed.phase_name, /integer phase two/);
});
test('asking for phase "2.1" returns the decimal child', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '2.1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '2.1');
assert.match(result.parsed.phase_name, /decimal child/);
});
test('asking for phase "2.10" returns the decimal sibling, NOT phase 2.1', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '2.10');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '2.10');
assert.match(result.parsed.phase_name, /decimal phase 2\.10/);
});
test('asking for phase "21" returns phase 21, NOT phase 2 (prefix-collision guard)', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '21');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '21');
assert.match(result.parsed.phase_name, /phase twenty-one/);
});
});
// ─── Unicode phase titles ───────────────────────────────────────────────────
describe('feat-3594: roadmap parser preserves Unicode phase titles', () => {
test('Japanese title round-trips through phase_name', (t) => {
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
const result = getPhase(projectDir, '1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.phase_name, '日本語フェーズ — initial setup');
});
test('emoji + smart-quote title survives', (t) => {
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
const result = getPhase(projectDir, '2');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.match(result.parsed.phase_name, /🚧/);
assert.match(result.parsed.phase_name, /Émile/);
});
test('Greek-letter title survives', (t) => {
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
const result = getPhase(projectDir, '3');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.phase_name, 'αβγ δεζ ηθι');
});
});
// ─── Repeated phase IDs ─────────────────────────────────────────────────────
describe('feat-3594: roadmap parser handles repeated phase IDs deterministically', () => {
test('two declarations of phase 1: parser returns the FIRST match (current behavior)', (t) => {
const projectDir = projectWithFixture(t, 'repeated-phase-ids.md');
const result = getPhase(projectDir, '1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
// The regex uses `content.match(...)` which returns the FIRST match.
// Pin that — a future change to last-wins or de-dup would fire.
assert.match(result.parsed.phase_name, /first declaration/);
});
});
// ─── HTML comments ──────────────────────────────────────────────────────────
describe('feat-3594: roadmap parser and HTML-commented headings', () => {
test('phase 1 in real prose is found even when phase 998/999 appear inside <!-- ... -->', (t) => {
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
const result = getPhase(projectDir, '1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_name, 'real phase');
});
test('phase 999 inside an HTML comment remains ignored because backlog sentinels never resolve', (t) => {
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
const result = getPhase(projectDir, '999');
assert.equal(result.hasStackTrace, false, 'no stack trace');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, false, 'backlog sentinel phases must not resolve');
});
});
// ─── Cross-corpus invariant ────────────────────────────────────────────────
describe('feat-3594: roadmap parser does not crash on ANY corpus fixture', () => {
const fixtures = fs.readdirSync(FIXTURE_DIR).filter((f) => f.endsWith('.md') && f !== 'README.md');
for (const fixture of fixtures) {
test(`fixture "${fixture}" — get-phase with arbitrary IDs must not crash`, (t) => {
const projectDir = projectWithFixture(t, fixture);
for (const id of ['1', '2', '99', '999', '0', '2.1']) {
const result = getPhase(projectDir, id);
assert.equal(result.hasStackTrace, false, `${fixture} id=${id}: no V8 stack frame allowed`);
// exit status varies (0 for found, non-zero for not-found —
// both are valid). What's pinned: the parser produced SOME output
// (either valid JSON or a clean stderr) without crashing.
}
});
}
});

View File

@@ -1,98 +0,0 @@
'use strict';
// feat(#41): /gsd-ship generate_pr_body emits a TDD Audit table + an aggregate
// `gate_status:` trailer so the per-commit TDD gate trail survives squash-merge.
// These assertions pin the shipped workflow prose in gsd-core/workflows/ship.md.
const fs = require('node:fs');
const path = require('node:path');
const assert = require('node:assert/strict');
const { describe, test } = require('node:test');
const repoRoot = path.resolve(__dirname, '..');
function readRepoFile(relativePath) {
return fs.readFileSync(path.join(repoRoot, relativePath), 'utf8');
}
describe('feat-41: ship.md TDD Audit gate_status extraction', () => {
const workflow = readRepoFile('gsd-core/workflows/ship.md');
test('adds a "## TDD Audit" section to the generated PR body', () => {
assert.match(workflow, /## TDD Audit/);
});
test('extracts gate_status via Git native trailer machinery, not a raw body grep', () => {
assert.match(workflow, /trailers:key=gate_status/);
});
test('scopes the scan to the merge-base..HEAD range', () => {
assert.match(workflow, /merge-base/);
assert.match(workflow, /\.\.HEAD/);
assert.match(workflow, /BASE_BRANCH/);
});
test('excludes merge commits from the audit', () => {
assert.match(workflow, /--no-merges/);
});
test('renders a Test commit / Impl commit / gate_status table', () => {
assert.match(workflow, /Test commit[\s\S]*Impl commit[\s\S]*gate_status/);
});
test('pairs conventional-commit test: rows with their impl commit', () => {
assert.match(workflow, /test:/);
assert.match(workflow, /pair/i);
});
test('escapes pipe characters in commit subjects so the table is not broken', () => {
assert.match(workflow, /[Ee]scape[\s\S]{0,60}\|/);
});
test('counts commits lacking a recognized gate_status trailer as missing', () => {
assert.match(workflow, /missing/);
});
test('is informational and never blocks the ship', () => {
assert.match(workflow, /informational|never block|non-blocking/i);
});
test('emits the aggregate trailer in the exact, stable key order', () => {
assert.match(
workflow,
/gate_status:\s*skill=[^,]*,\s*fallback=[^,]*,\s*exempt=[^,]*,\s*missing=/,
);
});
test('places the aggregate trailer on the final line so squash-merge carries it', () => {
assert.match(workflow, /squash/i);
assert.match(workflow, /final line|last line/i);
});
test('does not disturb the frozen #3167 core section order (Key Decisions precedes the new section)', () => {
assert.match(workflow, /## Key Decisions[\s\S]*## TDD Audit/);
});
// Hardening assertions added after adversarial review.
test('pairs test: rows only with feat:/fix: impl commits, skipping refactor/docs/chore', () => {
assert.match(workflow, /feat:[\s\S]{0,20}fix:/);
assert.match(workflow, /skipping[\s\S]{0,80}(refactor|docs|chore)/i);
});
test('normalizes the gate_status cell to a known token, never raw trailer text', () => {
assert.match(workflow, /normaliz[a-z]*[\s\S]{0,120}missing/i);
assert.match(workflow, /never the raw/i);
});
test('treats a commit with multiple gate_status trailers as missing', () => {
assert.match(workflow, /more than one[\s\S]{0,40}gate_status/i);
});
test('hardens every table cell against pipe/newline injection', () => {
assert.match(workflow, /strip[\s\S]{0,20}\\r/);
});
test('guards record/field delimiters against adversarial commit messages', () => {
assert.match(workflow, /NUL|%x00|delimiter/i);
});
});

View File

@@ -1,718 +0,0 @@
'use strict';
/**
* Architecture-level QA for issue #443 — unified effort + fast_mode engine.
*
* Integration suite (*.integration.test.cjs): cross-module flows that exercise
* real CLI invocations via runGsdTools, the full 33-agent registry, and the
* config round-trip through config-set -> resolve-execution.
*
* INVARIANTS tested here (each is also documented in docs/TESTING-SUITES.md):
*
* (a) CROSS-PROVIDER VALIDITY — renderEffortForRuntime never emits a value
* that the real provider API would 400 on. Ground-truth provider enums are
* defined as local constants (not sourced from the implementation).
*
* (b) PARAM/CHANNEL CONTRACT — each runtime exposes a stable parameter name
* and propagation channel.
*
* (c) RESOLVE-EXECUTION JSON CONTRACT — the CLI command emits a stable JSON
* shape with all required keys and correct types.
*
* (d) TOTALITY across the real 33-agent registry — every agent produces a
* valid effort value; none returns undefined/null.
*
* (e) FAST-MODE HONESTY INVARIANT — claude runtime always reports
* fast_mode_supported=false (emitting fast_mode frontmatter is a silent
* no-op for Claude Code subagents).
*
* (f) PRECEDENCE MATRIX — first-valid-wins for both effort and fast_mode
* cascades, including invalid values correctly falling through.
*
* (g) DYNAMIC-ROUTING COMPOSITION — resolveEffortForTier escalates
* independently of model tier logic; clamps at 'max'; respects
* max_escalations; disabled when escalate_on_failure=false.
*
* (h) CONFIG-TOOLING ROUND-TRIP — config-set accepts all new effort/fast_mode
* key paths (schema validation passes); values survive round-trip through
* resolve-execution.
*/
process.env.GSD_TEST_MODE = '1';
const { describe, test, before, after, beforeEach, afterEach } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { createTempProject, cleanup, runGsdTools } = require('./helpers.cjs');
const {
resolveEffortInternal,
resolveFastModeInternal,
resolveEffortForTier,
VALID_EFFORTS,
} = require('../gsd-core/bin/lib/model-resolver.cjs');
const {
renderEffortForRuntime,
RUNTIMES_WITH_FAST_MODE,
catalog,
} = require('../gsd-core/bin/lib/model-catalog.cjs');
// ─────────────────────────────────────────────────────────────────────────────
// Ground-truth provider enums (defined HERE, not sourced from the implementation).
// These are the exact values the real APIs accept — using a value outside these
// sets would result in a 400 response from the provider.
//
// Sources:
// Anthropic: output_config.effort — https://docs.anthropic.com (Claude API)
// OpenAI: model_reasoning_effort — https://platform.openai.com/docs (Codex)
// ─────────────────────────────────────────────────────────────────────────────
const PROVIDER_EFFORT_ENUMS = {
claude: new Set(['low', 'medium', 'high', 'xhigh', 'max']),
codex: new Set(['minimal', 'low', 'medium', 'high', 'xhigh']),
};
// Helper: write config.json into a temp project
function writeConfig(dir, config) {
const planningDir = path.join(dir, '.planning');
fs.mkdirSync(planningDir, { recursive: true });
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
}
// ─── (a) CROSS-PROVIDER VALIDITY INVARIANT ───────────────────────────────────
describe('#443 integration (a): cross-provider validity invariant', () => {
// For every universal effort × every provider runtime, the rendered value
// must be a member of that provider's real API enum.
test('all VALID_EFFORTS render within provider enums for claude and codex', () => {
for (const universalEffort of VALID_EFFORTS) {
for (const [runtime, providerEnum] of Object.entries(PROVIDER_EFFORT_ENUMS)) {
const rendered = renderEffortForRuntime(runtime, universalEffort);
assert.ok(
providerEnum.has(rendered.value),
`render('${runtime}', '${universalEffort}').value = '${rendered.value}' is NOT in the ` +
`${runtime} provider enum ${[...providerEnum].join('|')} — real API would 400`
);
}
}
});
// Documented clamps must hold exactly
test("render('codex','max').value === 'xhigh' (max is Anthropic-only)", () => {
assert.strictEqual(renderEffortForRuntime('codex', 'max').value, 'xhigh');
});
test("render('claude','minimal').value === 'low' (minimal is Codex-only)", () => {
assert.strictEqual(renderEffortForRuntime('claude', 'minimal').value, 'low');
});
// Common levels must pass through unchanged on BOTH providers
test('common levels (low/medium/high/xhigh) pass through unchanged on claude', () => {
for (const level of ['low', 'medium', 'high', 'xhigh']) {
assert.strictEqual(
renderEffortForRuntime('claude', level).value,
level,
`claude: level '${level}' should pass through unchanged`
);
}
});
test('common levels (low/medium/high/xhigh) pass through unchanged on codex', () => {
for (const level of ['low', 'medium', 'high', 'xhigh']) {
assert.strictEqual(
renderEffortForRuntime('codex', level).value,
level,
`codex: level '${level}' should pass through unchanged`
);
}
});
});
// ─── (b) PARAM/CHANNEL CONTRACT ──────────────────────────────────────────────
describe('#443 integration (b): param/channel contract', () => {
test("claude: param is always 'output_config.effort'", () => {
for (const effort of VALID_EFFORTS) {
const r = renderEffortForRuntime('claude', effort);
assert.strictEqual(r.param, 'output_config.effort',
`claude param must be 'output_config.effort' for effort '${effort}'`);
}
});
test("codex: param is always 'model_reasoning_effort'", () => {
for (const effort of VALID_EFFORTS) {
const r = renderEffortForRuntime('codex', effort);
assert.strictEqual(r.param, 'model_reasoning_effort',
`codex param must be 'model_reasoning_effort' for effort '${effort}'`);
}
});
test('claude channel is stable: frontmatter', () => {
for (const effort of VALID_EFFORTS) {
assert.strictEqual(renderEffortForRuntime('claude', effort).channel, 'frontmatter');
}
});
test('codex channel is stable: api', () => {
for (const effort of VALID_EFFORTS) {
assert.strictEqual(renderEffortForRuntime('codex', effort).channel, 'api');
}
});
test("unknown runtimes (gemini, qwen, 'mystery'): param===null, value passes through", () => {
for (const runtime of ['gemini', 'qwen', 'mystery']) {
for (const effort of VALID_EFFORTS) {
const r = renderEffortForRuntime(runtime, effort);
assert.strictEqual(r.param, null, `${runtime}: param must be null`);
assert.strictEqual(r.channel, null, `${runtime}: channel must be null`);
assert.strictEqual(r.value, effort, `${runtime}: value must pass through unchanged`);
}
}
});
});
// ─── (c) RESOLVE-EXECUTION JSON CONTRACT ─────────────────────────────────────
describe('#443 integration (c): resolve-execution JSON contract', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
function assertFullContract(output, label) {
assert.ok(typeof output.model === 'string' && output.model.length > 0,
`${label}: model must be a non-empty string`);
assert.ok(typeof output.profile === 'string' && output.profile.length > 0,
`${label}: profile must be a non-empty string`);
assert.ok(VALID_EFFORTS.includes(output.effort),
`${label}: effort '${output.effort}' must be a member of VALID_EFFORTS`);
assert.ok(typeof output.effort_rendered === 'string' && output.effort_rendered.length > 0,
`${label}: effort_rendered must be a non-empty string`);
assert.ok(output.effort_param === null || typeof output.effort_param === 'string',
`${label}: effort_param must be string or null`);
assert.ok(output.effort_propagation === null || typeof output.effort_propagation === 'string',
`${label}: effort_propagation must be string or null`);
assert.ok(typeof output.fast_mode === 'boolean',
`${label}: fast_mode must be a boolean`);
assert.ok(typeof output.fast_mode_supported === 'boolean',
`${label}: fast_mode_supported must be a boolean`);
}
test('gsd-planner (default claude runtime): full contract + known-agent shape', () => {
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assertFullContract(output, 'gsd-planner/claude');
assert.strictEqual(output.effort_param, 'output_config.effort');
assert.strictEqual(output.effort_propagation, 'frontmatter');
assert.strictEqual(output.fast_mode_supported, false);
// known agent must NOT have unknown_agent:true
assert.ok(!output.unknown_agent, 'known agent must not have unknown_agent:true');
});
test('codex runtime: full contract + effort_param=model_reasoning_effort', () => {
writeConfig(tmpDir, { runtime: 'codex' });
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assertFullContract(output, 'gsd-planner/codex');
assert.strictEqual(output.effort_param, 'model_reasoning_effort');
assert.strictEqual(output.fast_mode_supported, false);
});
test('gemini runtime: full contract + effort_param===null (no effort wire)', () => {
writeConfig(tmpDir, { runtime: 'gemini' });
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assertFullContract(output, 'gsd-planner/gemini');
assert.strictEqual(output.effort_param, null);
assert.strictEqual(output.effort_propagation, null);
assert.strictEqual(output.fast_mode_supported, false);
});
test('unknown agent: full contract + unknown_agent===true', () => {
const result = runGsdTools(['resolve-execution', 'unknown-agent-xyz'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assertFullContract(output, 'unknown-agent-xyz');
assert.strictEqual(output.unknown_agent, true, 'unknown agent must have unknown_agent:true');
});
});
// ─── (d) TOTALITY across the real 33-agent registry ──────────────────────────
describe('#443 integration (d): totality across real registry', () => {
let tmpDir;
before(() => { tmpDir = createTempProject(); });
after(() => { cleanup(tmpDir); });
const registeredAgents = Object.keys(catalog.agents);
// Confirm we're covering the full registry — snapshot the count so a
// catalog shrink is caught by this assertion.
test(`registry has at least 33 agents (currently ${registeredAgents.length})`, () => {
assert.ok(registeredAgents.length >= 33,
`Expected at least 33 agents in registry, got ${registeredAgents.length}`);
});
test(`all ${registeredAgents.length} agents: resolveEffortInternal returns a VALID_EFFORTS member`, () => {
const effortSet = new Set(VALID_EFFORTS);
const bad = [];
for (const agent of registeredAgents) {
const effort = resolveEffortInternal(tmpDir, agent);
if (effort === undefined || effort === null || !effortSet.has(effort)) {
bad.push(`${agent}: got ${JSON.stringify(effort)}`);
}
}
assert.strictEqual(bad.length, 0,
`Agents with invalid effort:\n${bad.join('\n')}`);
});
test(`all ${registeredAgents.length} agents: resolveFastModeInternal returns strict boolean`, () => {
const bad = [];
for (const agent of registeredAgents) {
const fm = resolveFastModeInternal(tmpDir, agent);
if (typeof fm !== 'boolean') {
bad.push(`${agent}: got ${JSON.stringify(fm)} (${typeof fm})`);
}
}
assert.strictEqual(bad.length, 0,
`Agents with non-boolean fast_mode:\n${bad.join('\n')}`);
});
test(`all ${registeredAgents.length} agents: renderEffortForRuntime('claude', effort) stays in claude enum`, () => {
const claudeEnum = PROVIDER_EFFORT_ENUMS.claude;
const bad = [];
for (const agent of registeredAgents) {
const effort = resolveEffortInternal(tmpDir, agent);
const rendered = renderEffortForRuntime('claude', effort);
if (!claudeEnum.has(rendered.value)) {
bad.push(`${agent}: effort=${effort} rendered=${rendered.value} not in claude enum`);
}
}
assert.strictEqual(bad.length, 0,
`Agents producing invalid claude effort:\n${bad.join('\n')}`);
});
});
// ─── (e) FAST-MODE HONESTY INVARIANT ─────────────────────────────────────────
describe('#443 integration (e): fast-mode honesty invariant', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
// Sample of agents across all tiers to prove the invariant is not agent-specific
const testAgents = ['gsd-planner', 'gsd-executor', 'gsd-codebase-mapper', 'gsd-verifier'];
test('claude runtime: fast_mode_supported is ALWAYS false regardless of fast_mode config', () => {
const configs = [
{},
{ fast_mode: { enabled: true } },
{ fast_mode: { routing_tier_defaults: { heavy: true } } },
{ fast_mode: { agent_overrides: { 'gsd-planner': true } } },
];
for (const config of configs) {
writeConfig(tmpDir, config);
for (const agent of testAgents) {
const result = runGsdTools(['resolve-execution', agent], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed for ${agent}: ${result.error}`);
const output = JSON.parse(result.output);
assert.strictEqual(output.fast_mode_supported, false,
`claude/${agent}: fast_mode_supported must be false (Claude has no per-subagent fast-mode mechanism); config=${JSON.stringify(config)}`);
}
}
});
test("RUNTIMES_WITH_FAST_MODE.has('api') === true (api is the only fast-mode capable runtime)", () => {
assert.ok(RUNTIMES_WITH_FAST_MODE.has('api'),
"RUNTIMES_WITH_FAST_MODE must include 'api' — this is the only runtime with per-call fast_mode support");
});
test("RUNTIMES_WITH_FAST_MODE.has('claude') === false (claude fast-mode is session-level only)", () => {
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('claude'),
"RUNTIMES_WITH_FAST_MODE must NOT include 'claude' — emitting fast_mode frontmatter on a Claude subagent is a silent no-op");
});
test("RUNTIMES_WITH_FAST_MODE.has('codex') === false", () => {
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('codex'),
"codex does not support per-call fast_mode");
});
test("RUNTIMES_WITH_FAST_MODE.has('gemini') === false", () => {
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('gemini'),
"gemini does not support per-call fast_mode");
});
});
// ─── (f) PRECEDENCE MATRIX ───────────────────────────────────────────────────
describe('#443 integration (f): precedence matrix (property/table-driven)', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
// Effort: first-valid-wins from highest precedence to lowest
// 1. opts.override (invocation)
// 2. effort.agent_overrides.<agent>
// 3. effort.routing_tier_defaults.<tier>
// 4. effort.default
// 5. manifest tier default
// 6. hardcoded 'high'
const effortPrecedenceTable = [
{
label: 'layer 1 (invocation override) beats all',
config: {
effort: {
agent_overrides: { 'gsd-planner': 'low' },
routing_tier_defaults: { heavy: 'medium' },
default: 'xhigh',
},
},
opts: { override: 'minimal' },
expected: 'minimal',
},
{
label: 'layer 2 (agent_override) beats tier default and default',
config: {
effort: {
agent_overrides: { 'gsd-planner': 'low' },
routing_tier_defaults: { heavy: 'medium' },
default: 'xhigh',
},
},
opts: {},
expected: 'low',
},
{
label: 'layer 3 (routing_tier_defaults) beats effort.default',
config: {
effort: {
routing_tier_defaults: { heavy: 'medium' },
default: 'xhigh',
},
},
opts: {},
expected: 'medium',
},
{
label: 'layer 4 (effort.default) when no tier default set',
config: {
effort: { default: 'low' },
},
opts: {},
expected: 'low',
},
{
label: 'invalid layer 1 (turbo) falls through to layer 2 (agent_override)',
config: {
effort: { agent_overrides: { 'gsd-planner': 'medium' } },
},
opts: { override: 'turbo' },
expected: 'medium',
},
{
label: 'invalid layer 2 (agent_override=123 numeric) falls through to tier default',
config: {
effort: {
agent_overrides: { 'gsd-planner': 123 },
routing_tier_defaults: { heavy: 'high' },
},
},
opts: {},
expected: 'high',
},
{
label: 'invalid tier default (turbo) falls through to effort.default',
config: {
effort: {
routing_tier_defaults: { heavy: 'turbo' },
default: 'low',
},
},
opts: {},
expected: 'low',
},
];
for (const row of effortPrecedenceTable) {
test(`effort precedence: ${row.label}`, () => {
writeConfig(tmpDir, row.config);
const result = resolveEffortInternal(tmpDir, 'gsd-planner', row.opts);
assert.strictEqual(result, row.expected,
`Expected '${row.expected}', got '${result}' — config: ${JSON.stringify(row.config)}`);
});
}
// fast_mode precedence:
// 1. opts.override (strict boolean only)
// 2. fast_mode.agent_overrides.<agent> (strict boolean only)
// 3. fast_mode.routing_tier_defaults.<tier> (strict boolean only)
// 4. fast_mode.enabled (strict boolean only)
// 5. false
const fastModePrecedenceTable = [
{
label: 'layer 1 (opts.override=false) beats enabled=true',
config: { fast_mode: { enabled: true } },
opts: { override: false },
expected: false,
},
{
label: 'layer 2 (agent_override=true) beats tier default',
config: {
fast_mode: {
agent_overrides: { 'gsd-planner': true },
routing_tier_defaults: { heavy: false },
enabled: false,
},
},
opts: {},
expected: true,
},
{
label: 'layer 3 (tier default=true) beats enabled=false',
config: {
fast_mode: {
routing_tier_defaults: { heavy: true },
enabled: false,
},
},
opts: {},
expected: true,
},
{
label: 'layer 4 (enabled=true) when no tier/agent overrides',
config: { fast_mode: { enabled: true } },
opts: {},
expected: true,
},
{
label: 'layer 5 (default false) when all absent',
config: {},
opts: {},
expected: false,
},
{
label: 'string "true" in opts.override is NOT accepted (falls through)',
config: { fast_mode: { enabled: true } },
// override must be strict boolean; string falls through to next layer
opts: { override: 'true' },
// 'true' as string is not boolean -> falls through to tier default
// gsd-planner is heavy; no tier default set; falls to enabled=true
expected: true,
},
{
label: 'string "true" in agent_overrides is NOT accepted',
config: {
fast_mode: {
agent_overrides: { 'gsd-planner': 'true' },
enabled: false,
},
},
opts: {},
// string 'true' is not boolean -> fall through to tier default -> enabled=false -> false
expected: false,
},
];
for (const row of fastModePrecedenceTable) {
test(`fast_mode precedence: ${row.label}`, () => {
writeConfig(tmpDir, row.config);
const result = resolveFastModeInternal(tmpDir, 'gsd-planner', row.opts);
assert.strictEqual(result, row.expected,
`Expected ${row.expected}, got ${result} — config: ${JSON.stringify(row.config)}`);
});
}
});
// ─── (g) DYNAMIC-ROUTING COMPOSITION ─────────────────────────────────────────
describe('#443 integration (g): dynamic-routing composition', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
const dynamicRoutingBase = {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: true,
max_escalations: 4,
},
effort: { routing_tier_defaults: { light: 'low' } },
};
test('resolveEffortForTier escalates independently of model resolution', () => {
writeConfig(tmpDir, dynamicRoutingBase);
const effort0 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
const effort1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
const effort2 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 2);
assert.strictEqual(effort0, 'low');
assert.strictEqual(effort1, 'medium');
assert.strictEqual(effort2, 'high');
// Verify the effort ladder steps up correctly without asserting model value
// (model timing is a separate concern from effort escalation)
assert.notStrictEqual(effort0, effort1, 'effort should escalate at attempt 1');
assert.notStrictEqual(effort1, effort2, 'effort should escalate at attempt 2');
});
test('escalate_on_failure=false: attempt is ignored for effort', () => {
writeConfig(tmpDir, {
...dynamicRoutingBase,
dynamic_routing: {
...dynamicRoutingBase.dynamic_routing,
escalate_on_failure: false,
},
});
const e0 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
const e1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
const e3 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 3);
assert.strictEqual(e0, e1, 'effort must not escalate when escalate_on_failure=false');
assert.strictEqual(e0, e3, 'effort must not escalate when escalate_on_failure=false');
});
test('escalation clamps at "max" regardless of attempt number', () => {
writeConfig(tmpDir, {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: true,
max_escalations: 99,
},
effort: { default: 'max' },
});
// Any large attempt number — result must never exceed 'max'
const r = resolveEffortForTier(tmpDir, 'gsd-planner', 50);
assert.strictEqual(r, 'max', `Effort must clamp at 'max', got '${r}'`);
const EFFORT_LADDER = VALID_EFFORTS;
const maxIdx = EFFORT_LADDER.indexOf('max');
const rIdx = EFFORT_LADDER.indexOf(r);
assert.ok(rIdx <= maxIdx, 'Effort must not exceed the max position in the ladder');
});
test('respects max_escalations cap: attempt beyond cap gives same as cap', () => {
writeConfig(tmpDir, {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: true,
max_escalations: 1,
},
effort: { routing_tier_defaults: { light: 'low' } },
});
const atCap = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1); // 1 escalation
const beyond = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 5); // capped at 1
assert.strictEqual(atCap, beyond,
'Effort beyond max_escalations must be same as at cap');
assert.strictEqual(atCap, 'medium', 'low + 1 escalation = medium');
});
test('dynamic_routing disabled: resolveEffortForTier ignores attempt', () => {
writeConfig(tmpDir, {
effort: { routing_tier_defaults: { light: 'low' } },
});
const e0 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
const e5 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 5);
assert.strictEqual(e0, e5, 'Effort must not change when dynamic_routing is disabled');
assert.strictEqual(e0, 'low');
});
});
// ─── (h) CONFIG-TOOLING ROUND-TRIP ───────────────────────────────────────────
describe('#443 integration (h): config-tooling round-trip', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
test('config-set effort.default then resolve-execution reflects new value', () => {
const setResult = runGsdTools(['config-set', 'effort.default', 'low'], tmpDir, { HOME: tmpDir });
assert.ok(setResult.success, `config-set effort.default failed: ${setResult.error}`);
const execResult = runGsdTools(['resolve-execution', 'unknown-agent-xyz'], tmpDir, { HOME: tmpDir });
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
const output = JSON.parse(execResult.output);
// unknown agent falls through to effort.default
assert.strictEqual(output.effort, 'low',
`Expected effort='low' after config-set, got '${output.effort}'`);
});
test('config-set effort.routing_tier_defaults.heavy then resolve-execution uses it', () => {
const setResult = runGsdTools(
['config-set', 'effort.routing_tier_defaults.heavy', 'medium'],
tmpDir, { HOME: tmpDir }
);
assert.ok(setResult.success, `config-set failed: ${setResult.error}`);
const execResult = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
const output = JSON.parse(execResult.output);
// gsd-planner is heavy; tier default now overridden to medium
assert.strictEqual(output.effort, 'medium',
`Expected effort='medium' after routing_tier_defaults override, got '${output.effort}'`);
});
test('config-set effort.agent_overrides.<agent> wins over tier default', () => {
// Set tier default first, then per-agent override
runGsdTools(['config-set', 'effort.routing_tier_defaults.heavy', 'medium'], tmpDir, { HOME: tmpDir });
const setResult = runGsdTools(
['config-set', 'effort.agent_overrides.gsd-planner', 'xhigh'],
tmpDir, { HOME: tmpDir }
);
assert.ok(setResult.success, `config-set agent_overrides failed: ${setResult.error}`);
const execResult = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
const output = JSON.parse(execResult.output);
assert.strictEqual(output.effort, 'xhigh',
`Expected agent_overrides to win (xhigh), got '${output.effort}'`);
});
test('config-set fast_mode.enabled true then resolve-execution reflects fast_mode=true', () => {
const setResult = runGsdTools(['config-set', 'fast_mode.enabled', 'true'], tmpDir, { HOME: tmpDir });
assert.ok(setResult.success, `config-set fast_mode.enabled failed: ${setResult.error}`);
const execResult = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
const output = JSON.parse(execResult.output);
assert.strictEqual(output.fast_mode, true,
`Expected fast_mode=true after config-set, got ${output.fast_mode}`);
// fast_mode_supported stays false (claude runtime)
assert.strictEqual(output.fast_mode_supported, false);
});
test('config-set fast_mode.agent_overrides.<agent> true reflects in output', () => {
const setResult = runGsdTools(
['config-set', 'fast_mode.agent_overrides.gsd-codebase-mapper', 'true'],
tmpDir, { HOME: tmpDir }
);
assert.ok(setResult.success, `config-set failed: ${setResult.error}`);
const execResult = runGsdTools(['resolve-execution', 'gsd-codebase-mapper'], tmpDir, { HOME: tmpDir });
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
const output = JSON.parse(execResult.output);
assert.strictEqual(output.fast_mode, true,
`Expected fast_mode=true for agent-specific override`);
});
// Prove the config-set commands accept all the new key namespaces (schema validation)
test('config-set accepts all effort/* and fast_mode/* key namespaces without error', () => {
const keysToTest = [
['effort.default', 'high'],
['effort.routing_tier_defaults.light', 'low'],
['effort.routing_tier_defaults.standard', 'medium'],
['effort.routing_tier_defaults.heavy', 'xhigh'],
['effort.agent_overrides.gsd-executor', 'high'],
['fast_mode.enabled', 'false'],
['fast_mode.routing_tier_defaults.light', 'false'],
['fast_mode.routing_tier_defaults.standard', 'false'],
['fast_mode.routing_tier_defaults.heavy', 'false'],
['fast_mode.agent_overrides.gsd-verifier', 'false'],
];
for (const [key, val] of keysToTest) {
const r = runGsdTools(['config-set', key, val], tmpDir, { HOME: tmpDir });
assert.ok(r.success, `config-set '${key}' '${val}' should succeed, got: ${r.error}`);
}
});
});

View File

@@ -1,862 +0,0 @@
'use strict';
/**
* Feature test for issue #443 — unified cross-provider effort + fast_mode knobs.
*
* Adds config-driven effort (universal ladder: minimal<low<medium<high<xhigh<max)
* and fast_mode knobs. Per-runtime rendering clamps the unique tails:
* - Anthropic/Claude: supports {low,medium,high,xhigh,max}, param=output_config.effort
* - Codex: supports {minimal,low,medium,high,xhigh}, param=model_reasoning_effort
*
* Also adds resolve-execution query which is the superset command including
* effort rendering and fast_mode propagation metadata.
*/
process.env.GSD_TEST_MODE = '1';
const { describe, test, beforeEach, afterEach } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { createTempProject, cleanup, runGsdTools } = require('./helpers.cjs');
const {
resolveEffortInternal,
resolveFastModeInternal,
resolveEffortForTier,
} = require('../gsd-core/bin/lib/model-resolver.cjs');
const {
renderEffortForRuntime,
RUNTIMES_WITH_FAST_MODE,
} = require('../gsd-core/bin/lib/model-catalog.cjs');
const {
injectEffortFrontmatter,
} = require('../bin/install.js');
function writeConfig(dir, config) {
const planningDir = path.join(dir, '.planning');
fs.mkdirSync(planningDir, { recursive: true });
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
}
// ─── Effort cascade ───────────────────────────────────────────────────────────
describe('#443 effort cascade', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
test('no config -> gsd-planner (heavy) defaults to "xhigh" via tier default', () => {
// gsd-planner is heavy tier; manifest default for heavy is xhigh
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
});
test('routing_tier_defaults: light (gsd-codebase-mapper) -> "low"', () => {
// gsd-codebase-mapper routingTier=light, default for light is "low"
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-codebase-mapper'), 'low');
});
test('routing_tier_defaults: standard (gsd-executor) -> "high"', () => {
// gsd-executor routingTier=standard, default for standard is "high"
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-executor'), 'high');
});
test('routing_tier_defaults: heavy (gsd-planner) -> "xhigh"', () => {
// gsd-planner routingTier=heavy, default for heavy is "xhigh"
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
});
test('effort.routing_tier_defaults override beats tier default', () => {
writeConfig(tmpDir, {
effort: { routing_tier_defaults: { heavy: 'medium' } },
});
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'medium');
});
test('effort.agent_overrides beats routing_tier_defaults', () => {
writeConfig(tmpDir, {
effort: {
routing_tier_defaults: { heavy: 'medium' },
agent_overrides: { 'gsd-planner': 'low' },
},
});
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'low');
});
test('opts.override beats agent_overrides', () => {
writeConfig(tmpDir, {
effort: { agent_overrides: { 'gsd-planner': 'low' } },
});
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner', { override: 'minimal' }), 'minimal');
});
test('invalid override falls through to agent_overrides', () => {
writeConfig(tmpDir, {
effort: { agent_overrides: { 'gsd-planner': 'low' } },
});
// 'turbo' is not a valid effort — should fall through to agent_overrides
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner', { override: 'turbo' }), 'low');
});
test('invalid agent_overrides value falls through to routing_tier_defaults', () => {
writeConfig(tmpDir, {
effort: {
agent_overrides: { 'gsd-planner': 123 },
routing_tier_defaults: { heavy: 'medium' },
},
});
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'medium');
});
test('invalid routing_tier_defaults value falls through to effort.default', () => {
writeConfig(tmpDir, {
effort: {
routing_tier_defaults: { heavy: 'turbo' },
default: 'low',
},
});
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'low');
});
test('invalid effort.default falls through to hardcoded "high" (no routing_tier_defaults set)', () => {
writeConfig(tmpDir, {
effort: { default: 'turbo' },
});
// effortCfg set but no routing_tier_defaults; turbo is invalid; fallback = hardcoded 'high'
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'high');
});
test('unknown agent -> uses effort.default', () => {
writeConfig(tmpDir, {
effort: { default: 'medium' },
});
// unknown-agent has no routingTier, so step 3 skipped
assert.strictEqual(resolveEffortInternal(tmpDir, 'unknown-agent-xyz'), 'medium');
});
test('effort.default numeric value (123) ignored, hardcoded "high" fallback', () => {
writeConfig(tmpDir, {
effort: { default: 123 },
});
// effortCfg set, no routing_tier_defaults -> no tier default; numeric ignored -> 'high'
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'high');
});
test('effort block missing entirely -> uses tier default', () => {
// No effort key in config at all
writeConfig(tmpDir, { model_profile: 'balanced' });
// heavy agent: tier default xhigh
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
});
test('effort block is non-object (string) -> effortCfg=null -> uses manifest tier default xhigh', () => {
writeConfig(tmpDir, { effort: 'bad' });
// Non-object effort => effortCfg=null; gsd-planner heavy tier manifest default = xhigh
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
});
test('effort.routing_tier_defaults empty object -> effort.default', () => {
writeConfig(tmpDir, {
effort: { routing_tier_defaults: {}, default: 'low' },
});
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'low');
});
});
// ─── Fast mode cascade ────────────────────────────────────────────────────────
describe('#443 fast_mode cascade', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
test('no config -> defaults to false', () => {
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
});
test('fast_mode.enabled=true -> true when no tier/agent overrides', () => {
writeConfig(tmpDir, { fast_mode: { enabled: true } });
// heavy agent: tier default is false, but enabled=true is layer 4
// tier default for heavy is false (below enabled), so gets enabled=true
// Wait — the cascade is: 1.override 2.agent_overrides 3.tier_defaults 4.enabled 5.false
// For gsd-planner (heavy), tier default is false — falls through to enabled=true
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), true);
});
test('fast_mode.routing_tier_defaults.light=true -> light agent gets true', () => {
writeConfig(tmpDir, {
fast_mode: { routing_tier_defaults: { light: true } },
});
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-codebase-mapper'), true);
});
test('fast_mode.routing_tier_defaults.heavy=false -> heavy agent stays false', () => {
writeConfig(tmpDir, {
fast_mode: { enabled: true, routing_tier_defaults: { heavy: false } },
});
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
});
test('fast_mode.agent_overrides beats routing_tier_defaults', () => {
writeConfig(tmpDir, {
fast_mode: {
routing_tier_defaults: { light: false },
agent_overrides: { 'gsd-codebase-mapper': true },
},
});
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-codebase-mapper'), true);
});
test('opts.override beats agent_overrides', () => {
writeConfig(tmpDir, {
fast_mode: { agent_overrides: { 'gsd-planner': true } },
});
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner', { override: false }), false);
});
test('string "true" NOT accepted as fast_mode override', () => {
writeConfig(tmpDir, {
fast_mode: { agent_overrides: { 'gsd-planner': 'true' } },
});
// string "true" is not boolean -> fall through to tier default or enabled
const result = resolveFastModeInternal(tmpDir, 'gsd-planner');
assert.strictEqual(typeof result, 'boolean');
});
test('string "true" in opts.override NOT accepted', () => {
// opts.override must be strict boolean — string falls through
const result = resolveFastModeInternal(tmpDir, 'gsd-planner', { override: 'true' });
assert.strictEqual(result, false);
});
test('fast_mode block missing entirely -> defaults to false', () => {
writeConfig(tmpDir, { model_profile: 'balanced' });
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
});
test('fast_mode.enabled="yes" (non-boolean) ignored -> false', () => {
writeConfig(tmpDir, { fast_mode: { enabled: 'yes' } });
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
});
test('unknown agent fast_mode -> uses enabled flag', () => {
writeConfig(tmpDir, { fast_mode: { enabled: true } });
assert.strictEqual(resolveFastModeInternal(tmpDir, 'unknown-agent-xyz'), true);
});
});
// ─── Effort escalation (resolveEffortForTier) ─────────────────────────────────
describe('#443 resolveEffortForTier escalation', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
test('dynamic_routing disabled -> attempt ignored, returns base effort', () => {
// gsd-planner heavy -> xhigh baseline
const base = resolveEffortForTier(tmpDir, 'gsd-planner', 0);
const attempt1 = resolveEffortForTier(tmpDir, 'gsd-planner', 1);
assert.strictEqual(base, 'xhigh');
assert.strictEqual(attempt1, 'xhigh'); // no dynamic_routing -> attempt ignored
});
test('dynamic_routing enabled, escalate_on_failure=false -> attempt ignored', () => {
writeConfig(tmpDir, {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: false,
max_escalations: 2,
},
});
const base = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
const attempt1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
assert.strictEqual(base, attempt1);
});
test('dynamic_routing enabled, attempt=1 -> one step up from base', () => {
writeConfig(tmpDir, {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: true,
max_escalations: 2,
},
effort: { routing_tier_defaults: { light: 'low' } },
});
// gsd-codebase-mapper: light -> effort 'low'; attempt=1 -> 'medium'
assert.strictEqual(resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0), 'low');
assert.strictEqual(resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1), 'medium');
});
test('escalation clamps at "max"', () => {
writeConfig(tmpDir, {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: true,
max_escalations: 99,
},
effort: { default: 'xhigh' },
});
// xhigh -> max -> max (clamp)
const result = resolveEffortForTier(tmpDir, 'gsd-planner', 99);
assert.strictEqual(result, 'max');
});
test('respects max_escalations cap', () => {
writeConfig(tmpDir, {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: true,
max_escalations: 1,
},
effort: { routing_tier_defaults: { light: 'low' } },
});
// light: low -> attempt=1 -> medium (but max=1 so can only escalate once)
const at1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
const at2 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 2);
// at2 is capped at 1 escalation, same as at1
assert.strictEqual(at1, at2);
assert.strictEqual(at1, 'medium');
});
});
// ─── Rendering / clamping ──────────────────────────────────────────────────────
describe('#443 renderEffortForRuntime', () => {
test('codex: "max" clamps to "xhigh"', () => {
const r = renderEffortForRuntime('codex', 'max');
assert.strictEqual(r.value, 'xhigh');
assert.strictEqual(r.param, 'model_reasoning_effort');
});
test('codex: common levels passthrough', () => {
assert.strictEqual(renderEffortForRuntime('codex', 'low').value, 'low');
assert.strictEqual(renderEffortForRuntime('codex', 'medium').value, 'medium');
assert.strictEqual(renderEffortForRuntime('codex', 'high').value, 'high');
assert.strictEqual(renderEffortForRuntime('codex', 'xhigh').value, 'xhigh');
});
test('codex: "minimal" passthrough', () => {
assert.strictEqual(renderEffortForRuntime('codex', 'minimal').value, 'minimal');
});
test('claude: "minimal" clamps to "low"', () => {
const r = renderEffortForRuntime('claude', 'minimal');
assert.strictEqual(r.value, 'low');
assert.strictEqual(r.param, 'output_config.effort');
});
test('claude: "max" passthrough (Anthropic-only)', () => {
const r = renderEffortForRuntime('claude', 'max');
assert.strictEqual(r.value, 'max');
assert.strictEqual(r.param, 'output_config.effort');
});
test('claude: common levels passthrough', () => {
assert.strictEqual(renderEffortForRuntime('claude', 'low').value, 'low');
assert.strictEqual(renderEffortForRuntime('claude', 'medium').value, 'medium');
assert.strictEqual(renderEffortForRuntime('claude', 'high').value, 'high');
assert.strictEqual(renderEffortForRuntime('claude', 'xhigh').value, 'xhigh');
});
test('unknown runtime: param is null, value passthrough', () => {
const r = renderEffortForRuntime('unknown-runtime', 'high');
assert.strictEqual(r.param, null);
assert.strictEqual(r.value, 'high');
});
test('RUNTIMES_WITH_FAST_MODE does NOT include "claude"', () => {
// Claude Code has no per-subagent fast-mode mechanism — session-level only
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('claude'),
'claude must NOT be in RUNTIMES_WITH_FAST_MODE — emitting fast_mode frontmatter is a silent no-op');
});
});
// ─── resolve-execution end-to-end ─────────────────────────────────────────────
describe('#443 resolve-execution CLI command', () => {
let tmpDir;
beforeEach(() => {
tmpDir = createTempProject();
// HOME isolation to prevent ~/.gsd/defaults.json bleed
process.env._GSD_TEST_HOME_OVERRIDE = tmpDir;
});
afterEach(() => {
cleanup(tmpDir);
delete process.env._GSD_TEST_HOME_OVERRIDE;
});
test('default (claude) runtime -> effort present, effort_param=output_config.effort, fast_mode_supported=false', () => {
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.ok(output.effort, 'should have effort field');
assert.strictEqual(output.effort_param, 'output_config.effort');
assert.strictEqual(output.fast_mode_supported, false);
assert.ok('fast_mode' in output, 'should have fast_mode field');
assert.ok('model' in output, 'should have model field');
assert.ok('profile' in output, 'should have profile field');
});
test('codex runtime -> effort_param=model_reasoning_effort, max clamps to xhigh, fast_mode_supported=false', () => {
writeConfig(tmpDir, {
runtime: 'codex',
effort: { default: 'max' },
});
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.strictEqual(output.effort_param, 'model_reasoning_effort');
assert.strictEqual(output.effort_rendered, 'xhigh');
// fast_mode_supported: codex does not support fast mode via subagent
assert.strictEqual(output.fast_mode_supported, false);
});
test('--effort flag overrides config effort', () => {
const result = runGsdTools(
['resolve-execution', 'gsd-planner', '--effort', 'low'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.strictEqual(output.effort, 'low');
});
test('--fast-mode flag honored', () => {
const result = runGsdTools(
['resolve-execution', 'gsd-planner', '--fast-mode', 'true'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.strictEqual(output.fast_mode, true);
});
test('--attempt flag triggers escalation', () => {
writeConfig(tmpDir, {
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
escalate_on_failure: true,
max_escalations: 2,
},
effort: { routing_tier_defaults: { light: 'low' } },
});
const result0 = runGsdTools(
['resolve-execution', 'gsd-codebase-mapper', '--attempt', '0'],
tmpDir,
{ HOME: tmpDir }
);
const result1 = runGsdTools(
['resolve-execution', 'gsd-codebase-mapper', '--attempt', '1'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(result0.success && result1.success);
const out0 = JSON.parse(result0.output);
const out1 = JSON.parse(result1.output);
assert.strictEqual(out0.effort, 'low');
assert.strictEqual(out1.effort, 'medium');
});
test('--raw prints effort string', () => {
const result = runGsdTools(
['resolve-execution', 'gsd-planner', '--raw'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(result.success, `Command failed: ${result.error}`);
// Raw output should be the effort string
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
assert.ok(VALID_EFFORTS.includes(result.output.trim()),
`Expected effort string, got: ${result.output}`);
});
test('fails when no agent-type provided', () => {
const result = runGsdTools(['resolve-execution'], tmpDir, { HOME: tmpDir });
assert.ok(!result.success, 'should fail without agent-type');
assert.ok(result.error.includes('agent-type required'), `error: ${result.error}`);
});
test('unknown agent -> unknown_agent=true still emits effort', () => {
const result = runGsdTools(['resolve-execution', 'unknown-agent-xyz'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.strictEqual(output.unknown_agent, true);
assert.ok(output.effort, 'should have effort even for unknown agent');
});
test('emits effort_propagation (channel) field', () => {
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.ok('effort_propagation' in output, 'should have effort_propagation field');
});
});
// ─── resolve-model now emits effort (replaces reasoning_effort) ───────────────
describe('#443 resolve-model emits effort (unified)', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
test('resolve-model on claude runtime emits effort (not null)', () => {
const result = runGsdTools(['resolve-model', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
// effort must be present and valid
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
assert.ok(VALID_EFFORTS.includes(output.effort),
`Expected valid effort, got: ${output.effort}`);
// reasoning_effort must NOT be present (removed)
assert.ok(!Object.prototype.hasOwnProperty.call(output, 'reasoning_effort'),
'resolve-model must not emit reasoning_effort (replaced by effort)');
});
test('resolve-model on codex runtime emits unified effort (not reasoning_effort)', () => {
fs.writeFileSync(
path.join(tmpDir, '.planning', 'config.json'),
JSON.stringify({ runtime: 'codex', model_profile: 'balanced' })
);
const result = runGsdTools(['resolve-model', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
assert.ok(VALID_EFFORTS.includes(output.effort),
`Expected valid effort, got: ${output.effort}`);
assert.ok(!Object.prototype.hasOwnProperty.call(output, 'reasoning_effort'),
'resolve-model must not emit reasoning_effort');
});
});
// ─── QA Matrix — hostile/malformed configs ───────────────────────────────────
describe('#443 QA matrix — malformed effort/fast_mode configs', () => {
let tmpDir;
beforeEach(() => { tmpDir = createTempProject(); });
afterEach(() => { cleanup(tmpDir); });
test('effort.default=123 (numeric) -> gracefully falls through', () => {
writeConfig(tmpDir, { effort: { default: 123 } });
// gsd-planner is heavy, tier default xhigh is used instead
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
assert.ok(typeof result === 'string');
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
assert.ok(VALID_EFFORTS.includes(result));
});
test('fast_mode.enabled="yes" (string) -> ignored, returns false', () => {
writeConfig(tmpDir, { fast_mode: { enabled: 'yes' } });
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
});
test('effort:{} empty block -> uses tier default or hardcoded high', () => {
writeConfig(tmpDir, { effort: {} });
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
assert.ok(VALID_EFFORTS.includes(result));
});
test('fast_mode:{} empty block -> false', () => {
writeConfig(tmpDir, { fast_mode: {} });
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
});
test('effort config is completely absent -> still resolves valid effort', () => {
writeConfig(tmpDir, { model_profile: 'quality' });
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
assert.ok(VALID_EFFORTS.includes(result));
});
test('effort.routing_tier_defaults has boolean value -> falls through', () => {
writeConfig(tmpDir, {
effort: {
routing_tier_defaults: { heavy: true },
default: 'medium',
},
});
// boolean true is not a valid effort -> falls through to default 'medium'
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'medium');
});
test('effort.agent_overrides is non-object -> falls through gracefully', () => {
writeConfig(tmpDir, {
effort: {
agent_overrides: 'not-an-object',
default: 'low',
},
});
// non-object agent_overrides -> skip step 2, use tier default (heavy=xhigh)
// actually heavy tier default kicks in first if no routing_tier_defaults
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
assert.ok(VALID_EFFORTS.includes(result));
});
test('config.json has unknown agent with effort.default set -> uses effort.default', () => {
writeConfig(tmpDir, { effort: { default: 'minimal' } });
assert.strictEqual(resolveEffortInternal(tmpDir, 'completely-unknown-agent-98765'), 'minimal');
});
test('resolve-execution with malformed config does not crash', () => {
writeConfig(tmpDir, {
effort: { default: null, routing_tier_defaults: null },
fast_mode: { enabled: null, agent_overrides: null },
});
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
assert.ok(result.success, `Should not crash with null config values: ${result.error}`);
});
});
// ─── Config schema: new keys are valid ───────────────────────────────────────
describe('#443 config schema: new effort/fast_mode keys valid', () => {
const { isValidConfigKey } = require('../gsd-core/bin/lib/config-schema.cjs');
test('effort.default is a valid config key', () => {
assert.ok(isValidConfigKey('effort.default'), 'effort.default must be valid');
});
test('fast_mode.enabled is a valid config key', () => {
assert.ok(isValidConfigKey('fast_mode.enabled'), 'fast_mode.enabled must be valid');
});
test('effort.routing_tier_defaults.light is valid (dynamic pattern)', () => {
assert.ok(isValidConfigKey('effort.routing_tier_defaults.light'));
});
test('effort.routing_tier_defaults.standard is valid', () => {
assert.ok(isValidConfigKey('effort.routing_tier_defaults.standard'));
});
test('effort.routing_tier_defaults.heavy is valid', () => {
assert.ok(isValidConfigKey('effort.routing_tier_defaults.heavy'));
});
test('effort.agent_overrides.<agent-id> is valid (dynamic pattern)', () => {
assert.ok(isValidConfigKey('effort.agent_overrides.gsd-planner'));
assert.ok(isValidConfigKey('effort.agent_overrides.my-custom-agent'));
});
test('fast_mode.routing_tier_defaults.light is valid', () => {
assert.ok(isValidConfigKey('fast_mode.routing_tier_defaults.light'));
});
test('fast_mode.agent_overrides.<agent-id> is valid', () => {
assert.ok(isValidConfigKey('fast_mode.agent_overrides.gsd-planner'));
});
test('effort.routing_tier_defaults.invalid-tier is NOT valid', () => {
assert.ok(!isValidConfigKey('effort.routing_tier_defaults.super'));
});
});
// ─── resolve-execution arg parsing matrix (Codex adversarial finding #1) ──────
//
// These tests FAIL before the fix: flags-first ordering misroutes the agent.
describe('#443 resolve-execution: deterministic arg parsing (flags-first ordering)', () => {
let tmpDir;
beforeEach(() => {
tmpDir = createTempProject();
process.env._GSD_TEST_HOME_OVERRIDE = tmpDir;
});
afterEach(() => {
cleanup(tmpDir);
delete process.env._GSD_TEST_HOME_OVERRIDE;
});
test('flags-first: --effort low gsd-planner resolves gsd-planner (NOT "low" as agent)', () => {
// BUG: before fix, agentTypeArg = 'low' (first non-dash token) -> unknown_agent:true
const result = runGsdTools(
['resolve-execution', '--effort', 'low', 'gsd-planner'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.ok(!output.unknown_agent, `agent must be resolved (not unknown_agent), got: ${JSON.stringify(output)}`);
assert.strictEqual(output.effort, 'low', `effort should be low, got: ${output.effort}`);
});
test('flags-first: --attempt 1 gsd-codebase-mapper resolves gsd-codebase-mapper (NOT "1" as agent)', () => {
// BUG: before fix, agentTypeArg = '1' -> unknown_agent:true
const result = runGsdTools(
['resolve-execution', '--attempt', '1', 'gsd-codebase-mapper'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(result.success, `Command failed: ${result.error}`);
const output = JSON.parse(result.output);
assert.ok(!output.unknown_agent, `gsd-codebase-mapper must be resolved, got: ${JSON.stringify(output)}`);
});
test('agent-first parity: gsd-planner --effort low produces same effort as flags-first', () => {
const flagsFirst = runGsdTools(
['resolve-execution', '--effort', 'low', 'gsd-planner'],
tmpDir,
{ HOME: tmpDir }
);
const agentFirst = runGsdTools(
['resolve-execution', 'gsd-planner', '--effort', 'low'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(flagsFirst.success && agentFirst.success,
`Both orderings must succeed. flags-first err: ${flagsFirst.error} agent-first err: ${agentFirst.error}`);
const outFF = JSON.parse(flagsFirst.output);
const outAF = JSON.parse(agentFirst.output);
assert.strictEqual(outFF.effort, outAF.effort, 'effort must be identical for both orderings');
assert.strictEqual(outFF.model, outAF.model, 'model must be identical for both orderings');
});
test('error: missing agent (--effort low with no positional) -> non-zero exit, no stack trace', () => {
const result = runGsdTools(
['resolve-execution', '--effort', 'low'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(!result.success, 'must exit non-zero when agent is missing');
assert.ok(!result.error.includes('at '), `error must not contain stack trace, got: ${result.error}`);
assert.ok(result.error.length > 0, 'must emit an error message');
});
test('error: two positional agents -> non-zero exit', () => {
const result = runGsdTools(
['resolve-execution', 'gsd-planner', 'gsd-executor'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(!result.success, 'must exit non-zero when two agents are given');
});
test('error: --attempt notanumber -> non-zero exit, clear error', () => {
const result = runGsdTools(
['resolve-execution', '--attempt', 'notanumber', 'gsd-planner'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(!result.success, 'must exit non-zero for non-integer --attempt');
assert.ok(result.error.length > 0, 'must emit an error message');
});
test('error: trailing --effort (no value) -> non-zero exit', () => {
const result = runGsdTools(
['resolve-execution', 'gsd-planner', '--effort'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(!result.success, 'must exit non-zero for trailing --effort with no value');
assert.ok(result.error.length > 0, 'must emit an error message');
});
test('unknown agent positional -> unknown_agent:true (preserved behavior)', () => {
const result = runGsdTools(
['resolve-execution', 'totally-not-an-agent'],
tmpDir,
{ HOME: tmpDir }
);
assert.ok(result.success, `Should succeed (unknown agent is valid input): ${result.error}`);
const output = JSON.parse(result.output);
assert.strictEqual(output.unknown_agent, true, 'unknown agent must emit unknown_agent:true');
});
});
// ─── injectEffortFrontmatter: newline-agnostic injection (#443 Windows fix) ──
describe('#443 injectEffortFrontmatter: newline-agnostic YAML frontmatter injection', () => {
// LF source (macOS / Linux git checkout) — baseline
test('LF frontmatter: injects effort: before closing ---', () => {
const content = '---\nname: gsd-planner\ndescription: Creates plans\ncolor: blue\n---\nBody here\n';
const result = injectEffortFrontmatter(content, 'xhigh');
assert.notStrictEqual(result, content, 'content should be modified');
assert.match(result, /^effort:\s*xhigh$/m, 'effort: xhigh must be present');
assert.ok(result.includes('\neffort: xhigh\n---\n'), 'effort: must appear before closing --- with LF');
// Closing --- must still be present and intact
assert.ok(result.includes('\n---\n'), 'closing --- must remain with LF');
});
// CRLF source (Windows git checkout with core.autocrlf=true) — the actual bug
test('CRLF frontmatter: injects effort: with CRLF preserved (Windows fix)', () => {
const content = '---\r\nname: gsd-planner\r\ndescription: Creates plans\r\ncolor: blue\r\n---\r\nBody here\r\n';
const result = injectEffortFrontmatter(content, 'xhigh');
assert.notStrictEqual(result, content, 'content should be modified (CRLF source was silently skipped before fix)');
// effort: line must use CRLF, not LF (EOL consistency)
assert.ok(result.includes('effort: xhigh\r\n'), 'effort: line must use CRLF to match surrounding frontmatter');
// Closing --- must use CRLF and remain intact
assert.ok(result.includes('\r\neffort: xhigh\r\n---\r\n'), 'effort: must appear before closing ---\\r\\n with CRLF');
// The effort value must be readable via multiline regex (as the install-wiring assertions do)
assert.match(result, /^effort:\s*xhigh$/m, '/^effort:\\s*xhigh$/m must match in CRLF output');
});
// Idempotency: don't double-insert if effort: already exists
test('idempotent: does NOT insert a second effort: line when already present (LF)', () => {
const content = '---\nname: gsd-planner\neffort: high\n---\nBody\n';
const result = injectEffortFrontmatter(content, 'xhigh');
assert.strictEqual(result, content, 'content must be unchanged when effort: already present');
// Confirm no duplicate
const matches = [...result.matchAll(/^effort:/mg)];
assert.strictEqual(matches.length, 1, 'exactly one effort: key must exist');
});
test('idempotent: does NOT insert a second effort: line when already present (CRLF)', () => {
const content = '---\r\nname: gsd-planner\r\neffort: high\r\n---\r\nBody\r\n';
const result = injectEffortFrontmatter(content, 'xhigh');
assert.strictEqual(result, content, 'content must be unchanged when effort: already present (CRLF)');
});
// No frontmatter — leave unchanged
test('no YAML frontmatter: returns content unchanged', () => {
const content = 'Just a body\nNo frontmatter here\n';
const result = injectEffortFrontmatter(content, 'xhigh');
assert.strictEqual(result, content, 'content without frontmatter must be returned unchanged');
});
// Complex frontmatter with comment lines and color: key (mirrors real agent .md files)
test('complex LF frontmatter (# comment + color:) still injects effort: before ---', () => {
const content = [
'---',
'name: gsd-executor',
'# hooks: see .claude/settings.json',
'description: Executes tasks',
'color: green',
'---',
'Body content here',
'',
].join('\n');
const result = injectEffortFrontmatter(content, 'high');
assert.match(result, /^effort:\s*high$/m, 'effort: high must be present');
assert.ok(result.includes('\neffort: high\n---\n'), 'effort: must appear immediately before closing ---');
// Other frontmatter fields must be untouched
assert.ok(result.includes('color: green'), 'color: must be preserved');
assert.ok(result.includes('# hooks:'), '# comment must be preserved');
});
test('complex CRLF frontmatter (# comment + color:) still injects effort: with CRLF before ---', () => {
const lines = [
'---',
'name: gsd-executor',
'# hooks: see .claude/settings.json',
'description: Executes tasks',
'color: green',
'---',
'Body content here',
'',
];
const content = lines.join('\r\n');
const result = injectEffortFrontmatter(content, 'high');
assert.ok(result.includes('effort: high\r\n'), 'effort: must use CRLF in CRLF file');
assert.ok(result.includes('\r\neffort: high\r\n---\r\n'), 'effort: must appear before closing ---\\r\\n');
assert.ok(result.includes('color: green\r\n'), 'color: must be preserved with CRLF');
});
});

View File

@@ -1,872 +0,0 @@
/**
* Feature test for issue #49 — model_policy presets.
*
* Adds a `model_policy` block to .planning/config.json:
*
* {
* "model_policy": {
* "provider": "anthropic-fable",
* "budget": "high",
* "runtime_tiers": {
* "opencode": {
* "opus": { "model": "anthropic/claude-opus-4-8" }
* }
* }
* }
* }
*
* Resolution precedence in resolveModelInternal (highest → lowest):
* 1. model_overrides[agent] (per-agent full IDs; existing)
* 2. model_policy.runtime_tiers[runtime][tier] (Sub-path A: explicit runtime+tier entry)
* 3. model_policy provider preset + budget (Sub-path B: known-provider catalog lookup)
* 4. model_profile_overrides (legacy runtime-aware overrides)
* 5. resolve_model_ids / profile fallback
*
* Sub-path A (runtime_tiers) fires when config.runtime matches a key inside
* model_policy.runtime_tiers AND that key contains an entry for the resolved tier.
*
* Sub-path B (provider preset) fires when model_policy.provider is a known
* provider AND the catalog contains an entry for (tier, budget) pair.
*
* Both sub-paths return a string model ID. Failures in either sub-path fall
* through cleanly to the next step in the chain.
*
* New config keys accepted by isValidConfigKey:
* - model_policy.provider
* - model_policy.budget
* - model_policy.runtime_tiers.<runtime>.<tier>
*
* Backwards compatibility:
* - model_profile_overrides continues to work when model_policy is absent.
* - When both are set, model_policy wins (fires first).
*
* KNOWN_PROVIDERS is exported from both model-catalog.cjs and core.cjs (re-export).
*
* These tests are written to FAIL before implementation. They use typed-IR /
* structural assertions on resolveModelInternal / resolveModelPolicy / isValidConfigKey
* return values — not stdout / grep.
*/
'use strict';
process.env.GSD_TEST_MODE = '1';
const { test, describe, beforeEach, afterEach } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
// ─── Imports (will fail until implementation exists) ────────────────────────
// resolveModelPolicy is a new internal function that must be exported from core.cjs.
// KNOWN_PROVIDERS must be exported from model-catalog.cjs and re-exported by core.cjs.
const {
resolveModelInternal,
resolveModelPolicy,
resolveModelForTier,
} = require('../gsd-core/bin/lib/model-resolver.cjs');
const {
KNOWN_PROVIDERS,
} = require('../gsd-core/bin/lib/model-catalog.cjs');
// KNOWN_PROVIDERS must also be exported directly from model-catalog.cjs
const modelCatalog = require('../gsd-core/bin/lib/model-catalog.cjs');
const { isValidConfigKey } = require('../gsd-core/bin/lib/config-schema.cjs');
const { createTempDir, cleanup, resetRuntimeWarningCaches } = require('./helpers.cjs');
const makeTmp = (prefix) => createTempDir(`gsd-49-${prefix}-`);
function writeConfig(dir, config) {
const planningDir = path.join(dir, '.planning');
fs.mkdirSync(planningDir, { recursive: true });
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
}
function rmr(p) {
cleanup(p);
}
// ─── resolveModelPolicy unit tests ──────────────────────────────────────────
//
// resolveModelPolicy(config, tier) is the pure resolver that takes a loaded
// config object and a resolved tier string. It returns a string model ID when
// model_policy produces a hit, or null when it falls through.
describe('#49 resolveModelPolicy: null/absent policy returns null', () => {
test('resolveModelPolicy returns null when policy is null or absent', () => {
// policy is null
assert.strictEqual(resolveModelPolicy(null, 'opus'), null);
// policy is undefined
assert.strictEqual(resolveModelPolicy(undefined, 'opus'), null);
// policy is absent (empty object treated as absent)
assert.strictEqual(resolveModelPolicy({}, 'opus'), null);
});
test('resolveModelPolicy returns null when runtime or tier is missing', () => {
const policy = { provider: 'anthropic', budget: 'high' };
// tier is null
assert.strictEqual(resolveModelPolicy(policy, null), null);
// tier is empty string
assert.strictEqual(resolveModelPolicy(policy, ''), null);
// tier is undefined
assert.strictEqual(resolveModelPolicy(policy, undefined), null);
});
});
describe('#49 resolveModelPolicy Sub-path B: provider presets', () => {
test('known provider "anthropic" + tier "opus" + budget "high" returns correct model ID', () => {
// The anthropic preset catalog must contain an entry for opus+high.
// The returned model ID is the high-budget anthropic opus model.
const policy = { provider: 'anthropic', budget: 'high' };
const result = resolveModelPolicy(policy, 'opus');
assert.ok(typeof result === 'string' && result.length > 0,
`expected a non-empty model ID string, got: ${JSON.stringify(result)}`);
assert.strictEqual(result, 'claude-opus-4-8',
`expected anthropic opus/high to resolve to claude-opus-4-8, got: ${result}`);
});
test('known provider "anthropic" + tier "sonnet" + budget "high" preserves Opus 4.8 routing', () => {
const policy = { provider: 'anthropic', budget: 'high' };
const result = resolveModelPolicy(policy, 'sonnet');
assert.strictEqual(result, 'claude-opus-4-8',
`expected anthropic sonnet/high to resolve to claude-opus-4-8, got: ${result}`);
});
test('known provider "anthropic-fable" + tier "opus" + budget "high" resolves to Claude Fable 5', () => {
const policy = { provider: 'anthropic-fable', budget: 'high' };
const result = resolveModelPolicy(policy, 'opus');
assert.strictEqual(result, 'claude-fable-5',
`expected anthropic-fable opus/high to resolve to claude-fable-5, got: ${result}`);
});
test('known provider "anthropic-fable" + tier "haiku" + budget "high" keeps low tier on Sonnet', () => {
const policy = { provider: 'anthropic-fable', budget: 'high' };
const result = resolveModelPolicy(policy, 'haiku');
assert.strictEqual(result, 'claude-sonnet-5',
`expected anthropic-fable haiku/high to resolve to claude-sonnet-5, got: ${result}`);
});
test('known provider "openai" + tier "sonnet" + budget "low" returns model with reasoning_effort from preset', () => {
// The openai preset catalog must contain a sonnet+low entry.
// "openai" maps to a different model family; the entry may include reasoning_effort.
const policy = { provider: 'openai', budget: 'low' };
const result = resolveModelPolicy(policy, 'sonnet');
assert.ok(typeof result === 'string' && result.length > 0,
`expected a non-empty model ID string for openai/sonnet/low, got: ${JSON.stringify(result)}`);
});
test('budget absent defaults to "medium"', () => {
// No "budget" key — defaults to "medium". The anthropic/opus/medium entry must exist.
const policyWithBudget = { provider: 'anthropic', budget: 'medium' };
const policyNoBudget = { provider: 'anthropic' };
const withBudget = resolveModelPolicy(policyWithBudget, 'opus');
const withoutBudget = resolveModelPolicy(policyNoBudget, 'opus');
// Both must return a string (not null)
assert.ok(typeof withBudget === 'string' && withBudget.length > 0,
`expected model from explicit budget:'medium'`);
assert.ok(typeof withoutBudget === 'string' && withoutBudget.length > 0,
`expected model when budget absent (should default to medium)`);
// They must resolve to the same value
assert.strictEqual(withBudget, withoutBudget,
'absent budget must behave identically to explicit "medium"');
});
test('provider "generic" (all null entries) returns null (falls through)', () => {
// provider:'generic' means opaque model IDs — there's no preset catalog for
// generic. Without a runtime_tiers hit, resolveModelPolicy returns null.
const policy = { provider: 'generic', budget: 'high' };
const result = resolveModelPolicy(policy, 'opus');
assert.strictEqual(result, null,
'provider:"generic" with no runtime_tiers must return null (no preset catalog)');
});
test('unknown provider string returns null without throwing', () => {
// A typo like provider:'mistral' must not crash; it degrades gracefully.
const policy = { provider: 'mistral', budget: 'high' };
let result;
assert.doesNotThrow(() => {
result = resolveModelPolicy(policy, 'opus');
}, 'resolveModelPolicy must not throw on unknown provider');
assert.strictEqual(result, null,
'unknown provider with no runtime_tiers must return null');
});
test('known provider + unknown tier returns null', () => {
const policy = { provider: 'anthropic', budget: 'high' };
const result = resolveModelPolicy(policy, 'jumbo');
assert.strictEqual(result, null,
'unknown tier "jumbo" must return null for anthropic provider');
});
test('known provider + known tier + missing budget level returns null', () => {
// The anthropic preset for opus only defines 'high' and 'medium' but NOT 'critical'.
// A missing budget level must fall through (return null) — not crash.
const policy = { provider: 'anthropic', budget: 'critical' };
const result = resolveModelPolicy(policy, 'opus');
assert.strictEqual(result, null,
'missing budget level "critical" must return null without throwing');
});
});
describe('#49 resolveModelPolicy Sub-path A: runtime_tiers', () => {
test('runtime_tiers entry wins over provider preset for same runtime+tier', () => {
// Sub-path A fires first: explicit runtime_tiers entry overrides the
// provider preset catalog. The returned model is the one in runtime_tiers,
// not what the provider preset would have returned.
const policy = {
provider: 'anthropic',
budget: 'high',
runtime: 'opencode',
runtime_tiers: {
opencode: {
opus: { model: 'anthropic/custom-opus-override' },
},
},
};
const result = resolveModelPolicy(policy, 'opus');
assert.strictEqual(result, 'anthropic/custom-opus-override',
'Sub-path A runtime_tiers must win over Sub-path B provider preset');
});
test('runtime_tiers string shorthand normalized to { model } object', () => {
// String shorthand: `{ opencode: { opus: "some-model-id" } }`
// must be normalized to `{ model: "some-model-id" }` so the resolver
// returns the string as-is.
const policy = {
provider: 'anthropic',
budget: 'high',
runtime: 'opencode',
runtime_tiers: {
opencode: {
opus: 'anthropic/string-shorthand-model',
},
},
};
const result = resolveModelPolicy(policy, 'opus');
assert.strictEqual(result, 'anthropic/string-shorthand-model',
'string shorthand in runtime_tiers must be normalized and returned as model ID');
});
test('runtime_tiers partial entry (no matching runtime) falls through to provider preset', () => {
// runtime_tiers has entries for 'copilot' but the active runtime is 'opencode'.
// The miss on runtime_tiers falls through to Sub-path B (provider preset).
const policy = {
provider: 'anthropic',
budget: 'high',
runtime: 'opencode',
runtime_tiers: {
copilot: {
opus: { model: 'some-copilot-model' },
},
},
};
const result = resolveModelPolicy(policy, 'opus');
// Falls through to Sub-path B (anthropic/opus/high) — must not be null.
assert.ok(typeof result === 'string' && result.length > 0,
'runtime_tiers miss must fall through to provider preset, got: ' + JSON.stringify(result));
// And it must NOT be the copilot model
assert.notStrictEqual(result, 'some-copilot-model');
});
});
// ─── resolveModelInternal integration tests ──────────────────────────────────
//
// These tests call resolveModelInternal through a temp project's config.json.
// They verify the full resolution chain including model_policy placement.
describe('#49 resolveModelInternal: model_policy in the resolution chain', () => {
let projectDir;
beforeEach(() => {
projectDir = makeTmp('internal');
resetRuntimeWarningCaches();
});
afterEach(() => {
rmr(projectDir);
resetRuntimeWarningCaches();
});
test('model_policy fires before model_profile_overrides when both are set (model_policy wins)', () => {
// model_policy (Sub-path B: anthropic/opus/high) must win over
// model_profile_overrides when both are present.
// We use a model_profile_overrides entry that would give a DIFFERENT result.
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_policy: {
provider: 'anthropic',
budget: 'high',
},
model_profile_overrides: {
opencode: {
// This legacy override would have returned this model — but model_policy must win.
opus: 'legacy-override-model-should-not-appear',
},
},
});
const result = resolveModelInternal(projectDir, 'gsd-planner');
assert.notStrictEqual(result, 'legacy-override-model-should-not-appear',
'model_policy must fire before model_profile_overrides and win');
assert.ok(typeof result === 'string' && result.length > 0,
'must return a non-empty model ID');
assert.strictEqual(result, 'claude-opus-4-8',
'expected anthropic preset opus/high to resolve to claude-opus-4-8');
});
test('model_policy with provider:"anthropic" + budget:"high" + runtime:"opencode" resolves to preset model', () => {
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality', // gsd-planner quality = opus tier
model_policy: {
provider: 'anthropic',
budget: 'high',
},
});
const result = resolveModelInternal(projectDir, 'gsd-planner');
assert.ok(typeof result === 'string' && result.length > 0,
'expected a non-empty model ID');
assert.strictEqual(result, 'claude-opus-4-8',
'anthropic/opus/high must resolve to claude-opus-4-8');
});
test('model_policy with provider:"anthropic-fable" + budget:"high" resolves to Fable preset model', () => {
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_policy: {
provider: 'anthropic-fable',
budget: 'high',
},
});
const result = resolveModelInternal(projectDir, 'gsd-planner');
assert.strictEqual(result, 'claude-fable-5',
'anthropic-fable/opus/high must resolve to claude-fable-5');
});
test('model_policy is skipped when runtime is absent', () => {
// No `runtime` in config — model_policy fires on any non-null policy
// only when a runtime context is available. Without runtime, the policy
// falls through entirely.
// NOTE: Sub-path B (provider preset) can fire without runtime — it only
// needs tier+budget+provider. Sub-path A requires runtime. This test
// verifies the gating behavior described in the issue: if model_policy
// is present but runtime is absent, provider preset Sub-path B still
// fires (it doesn't need runtime). So "skipped" means the runtime_tiers
// sub-path is skipped but provider preset may still fire.
// The test asserts that resolveModelInternal does not crash and returns
// a string regardless.
writeConfig(projectDir, {
model_profile: 'quality',
model_policy: {
provider: 'anthropic',
budget: 'high',
runtime_tiers: {
opencode: {
opus: { model: 'should-not-appear-no-runtime' },
},
},
},
});
let result;
assert.doesNotThrow(() => {
result = resolveModelInternal(projectDir, 'gsd-planner');
});
assert.ok(typeof result === 'string',
'resolveModelInternal must return a string even when runtime is absent');
// The runtime_tiers entry for opencode must not appear since runtime is absent
assert.notStrictEqual(result, 'should-not-appear-no-runtime',
'runtime_tiers must not fire when config.runtime is absent');
});
test('model_policy provider preset resolves to a Claude alias on runtime:"claude" (#1133)', () => {
writeConfig(projectDir, {
runtime: 'claude',
model_profile: 'balanced',
model_policy: { provider: 'anthropic-fable', budget: 'high' },
});
// gsd-planner -> opus tier; anthropic-fable opus/high = claude-fable-5 -> alias "fable"
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'fable');
});
test('model_policy works with implicit claude runtime (no runtime key) (#1133)', () => {
writeConfig(projectDir, {
model_profile: 'balanced',
model_policy: { provider: 'anthropic-fable', budget: 'high' },
});
// gsd-executor -> sonnet tier; anthropic-fable sonnet/high = claude-fable-5 -> "fable"
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-executor'), 'fable');
});
test('unmappable model_policy ID warns and falls back to the tier alias on claude (#1133)', () => {
resetRuntimeWarningCaches();
writeConfig(projectDir, {
runtime: 'claude',
model_profile: 'balanced',
model_policy: { provider: 'anthropic-fable', budget: 'low' },
});
// gsd-planner -> opus tier; anthropic-fable opus/low = claude-opus-4-5 (no alias) -> fall back to "opus"
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
});
test('model_policy.runtime_tiers applies on runtime:"claude", mapped to alias (#1133)', () => {
writeConfig(projectDir, {
runtime: 'claude',
model_profile: 'balanced',
model_policy: {
provider: 'anthropic',
budget: 'high',
runtime_tiers: { claude: { opus: { model: 'claude-fable-5' } } },
},
});
// gsd-planner -> opus tier; runtime_tiers.claude.opus = claude-fable-5 -> "fable" (was a no-op pre-#1133)
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'fable');
});
test('model_policy maps a built-in catalog model ID to its Claude alias via MODEL_ALIAS_MAP (#1133)', () => {
writeConfig(projectDir, {
runtime: 'claude',
model_profile: 'balanced',
model_policy: {
provider: 'anthropic',
budget: 'high',
runtime_tiers: { claude: { opus: { model: 'claude-opus-4-8' } } },
},
});
// gsd-planner -> opus tier; runtime_tiers.claude.opus = claude-opus-4-8 ->
// reverse of MODEL_ALIAS_MAP -> "opus" (exercises the non-fable reverse-map path)
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
});
test('model_policy still returns full IDs on non-claude runtimes (#1133 regression)', () => {
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'balanced',
model_policy: { provider: 'anthropic-fable', budget: 'high' },
});
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'claude-fable-5');
});
test('model_policy is skipped when tier:"inherit"', () => {
// When the resolved tier is 'inherit', model_policy must not fire.
// This mirrors the existing behavior for runtime-aware resolution.
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'inherit',
model_policy: {
provider: 'anthropic',
budget: 'high',
},
});
const result = resolveModelInternal(projectDir, 'gsd-planner');
// With profile:'inherit', the result must be 'inherit'
assert.strictEqual(result, 'inherit',
'model_policy must not fire when tier is "inherit"; resolveModelInternal must return "inherit"');
});
test('model_profile_overrides still resolves when model_policy is absent (legacy fallback intact)', () => {
// No model_policy — model_profile_overrides must still work exactly as before.
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_profile_overrides: {
opencode: {
opus: 'legacy-overridden-model',
},
},
});
const result = resolveModelInternal(projectDir, 'gsd-planner');
assert.strictEqual(result, 'legacy-overridden-model',
'model_profile_overrides must still win when model_policy is absent');
});
test('model_policy absent + model_profile_overrides set → model_profile_overrides wins (back-compat)', () => {
// Explicit: no model_policy key at all. model_profile_overrides is the only
// custom config. The legacy chain must apply exactly as before this feature.
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'balanced',
model_profile_overrides: {
opencode: {
sonnet: 'back-compat-sonnet-model',
},
},
});
// gsd-executor has balanced/opencode -> sonnet tier
const result = resolveModelInternal(projectDir, 'gsd-executor');
assert.strictEqual(result, 'back-compat-sonnet-model',
'legacy model_profile_overrides must be unaffected when model_policy is absent');
});
test('model_policy present but runtime_tiers empty + provider:"generic" → falls through to model_profile_overrides', () => {
// model_policy is a stub: runtime_tiers is empty ({}), provider is "generic".
// The resolver must fall through all model_policy paths and land on model_profile_overrides.
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_policy: {
provider: 'generic',
budget: 'high',
runtime_tiers: {},
},
model_profile_overrides: {
opencode: {
opus: 'fallthrough-to-legacy',
},
},
});
const result = resolveModelInternal(projectDir, 'gsd-planner');
assert.strictEqual(result, 'fallthrough-to-legacy',
'empty runtime_tiers + generic provider must fall through to model_profile_overrides');
});
});
// ─── Warning emission tests ───────────────────────────────────────────────────
describe('#49 resolveModelInternal: unknown provider warning behavior', () => {
let projectDir;
let origWrite;
let captured;
beforeEach(() => {
projectDir = makeTmp('warnings');
resetRuntimeWarningCaches();
captured = [];
origWrite = process.stderr.write.bind(process.stderr);
process.stderr.write = (chunk) => { captured.push(String(chunk)); return true; };
});
afterEach(() => {
process.stderr.write = origWrite;
rmr(projectDir);
resetRuntimeWarningCaches();
});
test('unknown provider in model_policy → falls through to model_profile_overrides, emits stderr warning once', () => {
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_policy: {
provider: 'mistral',
budget: 'high',
},
model_profile_overrides: {
opencode: {
opus: 'fallback-from-unknown-provider',
},
},
});
const result = resolveModelInternal(projectDir, 'gsd-planner');
// Must fall through to model_profile_overrides
assert.strictEqual(result, 'fallback-from-unknown-provider',
'unknown provider must fall through to model_profile_overrides');
// Must emit at least one stderr warning about the unknown provider
const joined = captured.join('');
assert.match(joined, /model_policy.*provider.*mistral|unknown.*provider.*mistral|mistral.*unknown/i,
'must emit a stderr warning about the unknown provider "mistral"');
});
test('unknown provider warning is deduplicated (emitted only once per config label)', () => {
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_policy: {
provider: 'mistral',
budget: 'high',
},
});
// Call resolveModelInternal multiple times for different agents — the
// warning about the unknown provider must be emitted only once.
resolveModelInternal(projectDir, 'gsd-planner');
resolveModelInternal(projectDir, 'gsd-executor');
resolveModelInternal(projectDir, 'gsd-verifier');
const joined = captured.join('');
// Count occurrences of "mistral" in the warning output
const matches = (joined.match(/mistral/gi) || []).length;
assert.ok(matches >= 1, 'expected at least one warning about "mistral"');
assert.ok(matches <= 2, `warning for unknown provider must be deduplicated — saw ${matches} occurrences`);
});
test('model_policy.runtime_tiers with unknown runtime emits one-shot stderr warning', () => {
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_policy: {
provider: 'anthropic',
budget: 'high',
runtime_tiers: {
unknownrt: {
opus: { model: 'some-model' },
},
},
},
});
resolveModelInternal(projectDir, 'gsd-planner');
const joined = captured.join('');
// Must emit a warning about the unknown runtime key in runtime_tiers
assert.match(joined, /unknownrt|unknown.*runtime|runtime_tiers.*unknown/i,
'must emit a stderr warning about unknown runtime "unknownrt" in model_policy.runtime_tiers');
});
test('model_policy.runtime_tiers with invalid tier name emits one-shot stderr warning', () => {
writeConfig(projectDir, {
runtime: 'opencode',
model_profile: 'quality',
model_policy: {
provider: 'anthropic',
budget: 'high',
runtime_tiers: {
opencode: {
jumbo: { model: 'invalid-tier-model' },
},
},
},
});
resolveModelInternal(projectDir, 'gsd-planner');
const joined = captured.join('');
// Must emit a warning about the invalid tier name "jumbo"
assert.match(joined, /jumbo|invalid.*tier|tier.*invalid|unknown.*tier/i,
'must emit a stderr warning about invalid tier "jumbo" in model_policy.runtime_tiers.opencode');
});
});
// ─── reasoning_effort passthrough tests ──────────────────────────────────────
describe('#49 reasoning_effort in model_policy entries', () => {
let projectDir;
beforeEach(() => { projectDir = makeTmp('effort'); });
afterEach(() => { rmr(projectDir); });
test('reasoning_effort in preset entry is returned as part of the entry object (caller decides whether to emit)', () => {
// When a provider preset includes reasoning_effort (e.g. openai opus/high),
// resolveModelPolicy must return the full entry object (or at minimum the model
// string) without stripping reasoning_effort internally.
// This is checked via the internal resolveModelPolicy function directly.
// The policy object includes a runtime_tiers entry that has reasoning_effort.
const policy = {
provider: 'anthropic',
budget: 'high',
runtime: 'opencode',
runtime_tiers: {
opencode: {
opus: { model: 'anthropic/claude-opus-4-8', reasoning_effort: 'high' },
},
},
};
// resolveModelPolicy must return the model string (at minimum).
// The caller (resolveModelInternal) is responsible for deciding what to
// emit — the resolver just returns the model ID string.
const result = resolveModelPolicy(policy, 'opus');
assert.strictEqual(result, 'anthropic/claude-opus-4-8',
'resolveModelPolicy must return the model string from the runtime_tiers entry');
});
test('reasoning_effort in model_policy.runtime_tiers entry is returned verbatim; renderEffortForRuntime strips it when runtime not in RUNTIMES_WITH_REASONING_EFFORT', () => {
// The renderEffortForRuntime function (already existing) handles the stripping.
// This test verifies the contract: resolveModelPolicy returns the model string,
// and for runtimes not in RUNTIMES_WITH_REASONING_EFFORT, the caller must not
// emit reasoning_effort.
const { renderEffortForRuntime, RUNTIMES_WITH_REASONING_EFFORT } = require('../gsd-core/bin/lib/model-catalog.cjs');
// 'opencode' is NOT in RUNTIMES_WITH_REASONING_EFFORT (only codex has reasoning_effort in catalog)
assert.ok(!RUNTIMES_WITH_REASONING_EFFORT.has('opencode'),
'opencode must not be in RUNTIMES_WITH_REASONING_EFFORT for this test to be meaningful');
// renderEffortForRuntime for a non-effort runtime returns channel:null
const rendered = renderEffortForRuntime('opencode', 'high');
assert.strictEqual(rendered.channel, null,
'renderEffortForRuntime must return channel:null for runtimes not supporting reasoning_effort');
// The resolveModelPolicy function returns just the model string — reasoning_effort
// is stripped at the emit layer, not inside resolveModelPolicy.
const policy = {
runtime: 'opencode',
provider: 'anthropic',
budget: 'high',
runtime_tiers: {
opencode: {
opus: { model: 'anthropic/claude-opus-4-8', reasoning_effort: 'high' },
},
},
};
const result = resolveModelPolicy(policy, 'opus');
assert.strictEqual(result, 'anthropic/claude-opus-4-8',
'resolveModelPolicy must return model string; reasoning_effort is stripped downstream');
});
});
// ─── isValidConfigKey: model_policy.* schema validation ──────────────────────
describe('#49 isValidConfigKey: model_policy.* keys accepted/rejected', () => {
test('isValidConfigKey accepts "model_policy.provider"', () => {
assert.strictEqual(isValidConfigKey('model_policy.provider'), true,
'"model_policy.provider" must be a valid config key');
});
test('isValidConfigKey accepts "model_policy.budget"', () => {
assert.strictEqual(isValidConfigKey('model_policy.budget'), true,
'"model_policy.budget" must be a valid config key');
});
test('isValidConfigKey accepts "model_policy.runtime_tiers.opencode.opus"', () => {
assert.strictEqual(isValidConfigKey('model_policy.runtime_tiers.opencode.opus'), true,
'"model_policy.runtime_tiers.opencode.opus" must be a valid config key');
});
test('isValidConfigKey rejects "model_policy.runtime_tiers.opencode.banana" (invalid tier)', () => {
assert.strictEqual(isValidConfigKey('model_policy.runtime_tiers.opencode.banana'), false,
'"model_policy.runtime_tiers.opencode.banana" must be rejected (banana is not a valid tier)');
});
});
// ─── KNOWN_PROVIDERS export tests ─────────────────────────────────────────────
describe('#49 KNOWN_PROVIDERS exports from model-catalog.cjs', () => {
test('KNOWN_PROVIDERS exported from model-catalog.cjs includes all keys from providerPresets in catalog', () => {
// KNOWN_PROVIDERS must be a Set (or array) exported from model-catalog.cjs.
assert.ok(KNOWN_PROVIDERS != null,
'KNOWN_PROVIDERS must be exported from model-catalog.cjs');
const isIterable = typeof KNOWN_PROVIDERS[Symbol.iterator] === 'function';
assert.ok(isIterable,
'KNOWN_PROVIDERS must be iterable (Set or array)');
const providers = [...KNOWN_PROVIDERS];
assert.ok(providers.length > 0,
'KNOWN_PROVIDERS must not be empty');
// 'anthropic' must be in the set since it is a required provider preset
assert.ok(providers.includes('anthropic'),
'KNOWN_PROVIDERS must include "anthropic"');
assert.ok(providers.includes('anthropic-fable'),
'KNOWN_PROVIDERS must include "anthropic-fable"');
// 'generic' is a special fallback, not a real provider — it must NOT be in KNOWN_PROVIDERS
// (KNOWN_PROVIDERS lists only providers with catalog entries)
assert.ok(!providers.includes('generic'),
'KNOWN_PROVIDERS must not include "generic" (it is not a catalog-backed provider)');
});
test('KNOWN_PROVIDERS from model-catalog.cjs is the canonical export', () => {
// model-catalog.cjs is the canonical source of KNOWN_PROVIDERS.
assert.ok(modelCatalog.KNOWN_PROVIDERS != null,
'KNOWN_PROVIDERS must be exported from model-catalog.cjs');
const fromCatalog = [...modelCatalog.KNOWN_PROVIDERS].sort();
const fromImport = [...KNOWN_PROVIDERS].sort();
assert.deepStrictEqual(fromImport, fromCatalog,
'KNOWN_PROVIDERS imported from model-catalog.cjs must match the module export');
});
});
// ─── resolveModelPolicy: Object.hasOwn prototype-pollution guards ────────────
describe('#49 resolveModelPolicy: prototype-pollution guards', () => {
test('__proto__ as provider returns null without throwing', () => {
assert.strictEqual(resolveModelPolicy({ provider: '__proto__', budget: 'medium' }, 'sonnet'), null);
});
test('constructor as provider returns null without throwing', () => {
assert.strictEqual(resolveModelPolicy({ provider: 'constructor', budget: 'medium' }, 'sonnet'), null);
});
test('__proto__ as budget returns null without throwing', () => {
assert.strictEqual(resolveModelPolicy({ provider: 'openai', budget: '__proto__' }, 'haiku'), null);
});
test('toString as budget returns null without throwing', () => {
assert.strictEqual(resolveModelPolicy({ provider: 'openai', budget: 'toString' }, 'haiku'), null);
});
test('__proto__ as runtime_tiers key returns null without throwing', () => {
const policy = {
runtime: '__proto__',
runtime_tiers: { '__proto__': { haiku: { model: 'evil' } } },
};
assert.strictEqual(resolveModelPolicy(policy, 'haiku'), null);
});
test('__proto__ as tier inside runtime_tiers returns null without throwing', () => {
const policy = {
runtime: 'codex',
runtime_tiers: { codex: { '__proto__': { model: 'evil' } } },
};
assert.strictEqual(resolveModelPolicy(policy, '__proto__'), null);
});
test('valid provider+tier+budget still resolves correctly after guards', () => {
const result = resolveModelPolicy({ provider: 'openai', budget: 'low' }, 'haiku');
assert.ok(typeof result === 'string' && result.length > 0,
'valid openai/haiku/low lookup must still resolve after adding hasOwn guards');
});
});
// ─── resolveModelForTier: model_policy beats dynamic_routing ─────────────────
describe('#49 resolveModelForTier: model_policy beats dynamic_routing', () => {
let tmpDir;
beforeEach(() => { tmpDir = makeTmp('for-tier-'); });
afterEach(() => { rmr(tmpDir); });
test('model_policy wins over dynamic_routing.tier_models when both are set', () => {
writeConfig(tmpDir, {
runtime: 'codex',
model_policy: { provider: 'openai', budget: 'low' },
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
},
});
// model_policy fires before dynamic_routing in resolveModelForTier
const result = resolveModelForTier(tmpDir, 'gsd-executor', 0);
// gsd-executor is standard/sonnet tier; openai+low+sonnet preset model
assert.ok(typeof result === 'string' && result.length > 0,
'model_policy must return a model string');
assert.notStrictEqual(result, 'sonnet',
'dynamic_routing tier alias must not win over model_policy');
});
test('model_overrides still beats model_policy in resolveModelForTier', () => {
writeConfig(tmpDir, {
runtime: 'codex',
model_policy: { provider: 'openai', budget: 'high' },
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
},
model_overrides: { 'gsd-planner': 'custom-model-id' },
});
assert.strictEqual(resolveModelForTier(tmpDir, 'gsd-planner', 0), 'custom-model-id');
});
test('dynamic_routing.tier_models used normally when model_policy absent', () => {
writeConfig(tmpDir, {
runtime: 'codex',
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'my-custom-sonnet', heavy: 'opus' },
},
});
assert.strictEqual(resolveModelForTier(tmpDir, 'gsd-executor', 0), 'my-custom-sonnet');
});
test('model_policy with Claude runtime does not interrupt dynamic_routing', () => {
// model_policy only gates on non-Claude runtimes; with runtime absent/claude,
// dynamic_routing must still work normally.
writeConfig(tmpDir, {
model_policy: { provider: 'openai', budget: 'low' },
dynamic_routing: {
enabled: true,
tier_models: { light: 'haiku', standard: 'my-sonnet', heavy: 'opus' },
},
});
assert.strictEqual(resolveModelForTier(tmpDir, 'gsd-executor', 0), 'my-sonnet');
});
test('model_policy value that is already a bare Claude alias is returned as-is on claude (#1133)', () => {
writeConfig(tmpDir, {
runtime: 'claude',
model_profile: 'balanced',
model_policy: {
provider: 'anthropic',
budget: 'high',
runtime_tiers: { claude: { opus: { model: 'fable' } } },
},
});
// gsd-planner → opus tier; runtime_tiers.claude.opus = "fable" is already a valid alias → "fable"
assert.strictEqual(resolveModelInternal(tmpDir, 'gsd-planner'), 'fable');
});
});

View File

@@ -1,233 +0,0 @@
// allow-test-rule: source-text-is-the-product #1627
// Agent .md / reference .md files — their text IS what the runtime loads.
// Testing text content tests the deployed contract.
// Per CONTRIBUTING.md exception matrix.
/**
* Fix #1627 — ASVS level scaling
*
* Asserts that `workflow.security_asvs_level` now scales both planner
* threat-disposition rigor and auditor verification depth rather than
* being display-only.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const ROOT = path.join(__dirname, '..');
const AGENTS_DIR = path.join(ROOT, 'agents');
const REFS_DIR = path.join(ROOT, 'gsd-core', 'references');
const MANIFEST_PATH = path.join(ROOT, 'docs', 'INVENTORY-MANIFEST.json');
describe('SECURE: ASVS level scaling (#1627)', () => {
// ── 1. New reference file ────────────────────────────────────────────────
describe('security-asvs-levels.md reference', () => {
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
test('file exists', () => {
assert.ok(fs.existsSync(refPath), 'gsd-core/references/security-asvs-levels.md must exist');
});
test('defines all three levels', () => {
const content = fs.readFileSync(refPath, 'utf-8');
assert.ok(content.includes('L1'), 'must define L1');
assert.ok(content.includes('L2'), 'must define L2');
assert.ok(content.includes('L3'), 'must define L3');
});
test('L1 describes opportunistic scope and planner disposition', () => {
const content = fs.readFileSync(refPath, 'utf-8');
assert.ok(
content.toLowerCase().includes('opportunistic'),
'L1 must be described as opportunistic'
);
assert.ok(
content.includes('mitigate') && content.includes('accept'),
'must describe mitigate/accept dispositions'
);
});
test('L2 requires explicit rationale for accepted threats', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// L2 must require documented rationale for accepted risks
assert.ok(
content.includes('rationale') || content.includes('documented'),
'L2 must require documented rationale for accepted threats'
);
});
test('L3 describes deep/comprehensive verification', () => {
const content = fs.readFileSync(refPath, 'utf-8');
const lower = content.toLowerCase();
assert.ok(
lower.includes('deep') || lower.includes('comprehensive') || lower.includes('exhaustive'),
'L3 must describe deep/comprehensive verification'
);
});
test('mentions that higher levels are supersets of lower', () => {
const content = fs.readFileSync(refPath, 'utf-8');
const lower = content.toLowerCase();
assert.ok(
lower.includes('superset') || lower.includes('higher level') || lower.includes('includes all'),
'must note that higher levels are supersets of lower'
);
});
test('describes distinct auditor verification depth for each level', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// All three audit depth keywords should appear
assert.ok(content.includes('grep') || content.includes('PRESENT'), 'L1 audit depth must mention grep/presence check');
assert.ok(content.includes('boundary') || content.includes('addresses'), 'L2 audit depth must mention boundary/addresses');
assert.ok(content.includes('end-to-end') || content.includes('bypass'), 'L3 audit depth must mention end-to-end or bypass check');
});
});
// ── 2. gsd-planner.md — no hardcoded L1 in disposition ──────────────────
describe('gsd-planner.md security disposition', () => {
const plannerPath = path.join(AGENTS_DIR, 'gsd-planner.md');
test('planner security instruction does not hardcode "ASVS L1"', () => {
const content = fs.readFileSync(plannerPath, 'utf-8');
// The old bug: "mitigate if ASVS L1 requires it" — must be gone
assert.ok(
!content.includes('ASVS L1 requires it'),
'planner must not hardcode "ASVS L1 requires it"; it must reference the configured level'
);
});
test('planner references the configured OWASP ASVS level', () => {
const content = fs.readFileSync(plannerPath, 'utf-8');
assert.ok(
content.includes('OWASP ASVS level') || content.includes('configured OWASP'),
'planner must reference the configured OWASP ASVS level'
);
});
test('planner @-references security-asvs-levels.md', () => {
const content = fs.readFileSync(plannerPath, 'utf-8');
assert.ok(
content.includes('security-asvs-levels.md'),
'planner must @-reference security-asvs-levels.md'
);
});
test('planner is under the 49152-char cap', () => {
const content = fs.readFileSync(plannerPath, 'utf-8').replace(/\r\n/g, '\n').replace(/\r/g, '\n');
assert.ok(
content.length < 49152,
`gsd-planner.md must be < 49152 chars (LF-normalized); got ${content.length}`
);
});
});
// ── 3. gsd-security-auditor.md — scaled verification depth ──────────────
describe('gsd-security-auditor.md verification depth', () => {
const auditorPath = path.join(AGENTS_DIR, 'gsd-security-auditor.md');
test('auditor scales verification depth by asvs_level', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
assert.ok(
content.includes('asvs_level') || content.includes('ASVS level'),
'auditor must reference asvs_level to scale verification'
);
});
test('auditor describes L1/L2/L3 depth differences', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
// All three levels must appear in context of depth scaling
assert.ok(content.includes('L1'), 'auditor must mention L1 depth');
assert.ok(content.includes('L2'), 'auditor must mention L2 depth');
assert.ok(content.includes('L3'), 'auditor must mention L3 depth');
});
test('auditor @-references security-asvs-levels.md', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
assert.ok(
content.includes('security-asvs-levels.md'),
'auditor must @-reference security-asvs-levels.md'
);
});
test('auditor still echoes ASVS Level in structured output', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
assert.ok(
content.includes('ASVS Level:') && content.includes('{1/2/3}'),
'auditor must still emit ASVS Level in SECURED/OPEN_THREATS output'
);
});
});
// ── 4. secure-phase.md — ASVS-aware short-circuit ──────────────────────
describe('secure-phase.md short-circuit conditioned on asvs_level', () => {
const wfPath = path.join(ROOT, 'gsd-core', 'workflows', 'secure-phase.md');
test('short-circuit to Step 6 is gated on asvs_level == 1', () => {
const content = fs.readFileSync(wfPath, 'utf-8');
// The condition must reference asvs_level so that L2/L3 don't skip the auditor
assert.ok(
content.includes('asvs_level == 1'),
'secure-phase.md must gate the skip-to-Step-6 short-circuit on asvs_level == 1'
);
});
test('auditor runs at L2/L3 even when threats_open is 0 (asvs_level >= 2 branch present)', () => {
const content = fs.readFileSync(wfPath, 'utf-8');
// The >= 2 branch must explicitly say the auditor is spawned for L2/L3 deep verification
assert.ok(
content.includes('asvs_level >= 2'),
'secure-phase.md must include asvs_level >= 2 branch that does NOT skip the auditor'
);
// The >= 2 branch must make clear the auditor is spawned (not skipped)
assert.ok(
content.includes('L2/L3 deep verification') || content.includes('L2 boundary') || content.includes('L3 end-to-end'),
'secure-phase.md asvs_level >= 2 branch must reference L2/L3 deep verification'
);
});
});
// ── 5. security-asvs-levels.md — L1 medium-severity gap closed ──────────
describe('security-asvs-levels.md L1 medium-severity is specified', () => {
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
test('L1 explicitly handles medium-severity threats (no gap)', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// L1 section must say something about medium-severity
assert.ok(
content.includes('medium-severity') || content.includes('medium severity'),
'L1 must explicitly specify disposition for medium-severity threats (no ambiguity gap)'
);
});
test('L1 medium-severity disposition is conditional (trust-boundary-aware)', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// L1 must distinguish between medium on primary trust boundary vs not
assert.ok(
content.includes('trust boundary') || content.includes('primary trust'),
'L1 medium-severity rule must reference trust boundary to disambiguate disposition'
);
});
});
// ── 6. Inventory manifest ─────────────────────────────────────────────────
describe('inventory manifest', () => {
test('security-asvs-levels.md is registered in INVENTORY-MANIFEST.json', () => {
const manifest = JSON.parse(fs.readFileSync(MANIFEST_PATH, 'utf-8'));
const refs = (manifest.families || {}).references || [];
assert.ok(
refs.includes('security-asvs-levels.md'),
'security-asvs-levels.md must appear in families.references of INVENTORY-MANIFEST.json'
);
});
});
});

View File

@@ -657,3 +657,163 @@ describe('bug #3784: settings.md model profile UI exposes all 5 profiles', () =>
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-33-settings-model-profile-adaptive.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-33-settings-model-profile-adaptive (consolidation epic #1969 B8 #1977)", () => {
'use strict';
// allow-test-rule: source-text-is-the-product (see #33)
// The deployed settings.md IS the product — testing its text content tests the deployed contract.
/**
* Regression test for issue #33
*
* model_profile UI shows 4 options, schema has 5 — `adaptive` missing from
* `settings.md` AskUserQuestion.
*
* The schema (gsd-core/bin/shared/model-catalog.json `profiles` array) defines 5 valid
* model_profile values: quality, balanced, budget, adaptive, inherit. The
* settings.md AskUserQuestion block for model_profile originally listed only 4
* options (Quality, Balanced, Budget, Inherit) — `adaptive` was missing.
*
* Fix: the model_profile selection uses a two-question split. Q1 routes between
* Adaptive / Standard-tier / Inherit (3 options). Q2 (only when Q1 = Standard)
* asks Quality / Balanced / Budget. This keeps every individual options array
* within the AskUserQuestion 4-option cap while making all 5 profiles reachable.
*
* Fixes: #33
*/
const { describe, test, before } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const REPO_ROOT = path.join(__dirname, '..');
const SETTINGS_PATH = path.join(REPO_ROOT, 'gsd-core', 'workflows', 'settings.md');
const CATALOG_PATH = path.join(REPO_ROOT, 'gsd-core', 'bin', 'shared', 'model-catalog.json');
/**
* Collect every label: "..." value within a text block, lowercased.
*/
function extractOptionLabels(block) {
const re = /label:\s*"([^"]+)"/g;
const labels = [];
let m;
while ((m = re.exec(block)) !== null) {
labels.push(m[1].toLowerCase());
}
return labels;
}
describe('issue #33: model_profile schema and settings.md UI are in sync', () => {
let catalog;
let settingsContent;
let presentBlock;
before(() => {
catalog = JSON.parse(fs.readFileSync(CATALOG_PATH, 'utf-8'));
settingsContent = fs.readFileSync(SETTINGS_PATH, 'utf-8');
const presentMatch = settingsContent.match(/<step name="present_settings">[\s\S]*?<\/step>/);
assert.ok(presentMatch, 'settings.md must contain a present_settings step');
presentBlock = presentMatch[0];
});
// -- (a) Schema contract ---------------------------------------------------
test('schema includes the adaptive model_profile value', () => {
assert.ok(
Array.isArray(catalog.profiles),
'model-catalog.json must have a "profiles" array'
);
assert.ok(
catalog.profiles.includes('adaptive'),
'model-catalog should include the adaptive profile. Got: [' + catalog.profiles.join(', ') + ']'
);
});
test('schema includes adaptive as a model_profile value', () => {
assert.ok(
catalog.profiles.includes('adaptive'),
'"adaptive" must be in model-catalog.json profiles. Got: [' + catalog.profiles.join(', ') + ']'
);
});
test('schema includes all expected model_profile values', () => {
const expected = ['quality', 'balanced', 'budget', 'adaptive', 'inherit'];
for (const profile of expected) {
assert.ok(
catalog.profiles.includes(profile),
'Schema must include "' + profile + '" in profiles. Got: [' + catalog.profiles.join(', ') + ']'
);
}
});
// -- (b) UI contract — all 5 profiles reachable via present_settings -------
test('present_settings includes Adaptive as a selectable option (#33)', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'adaptive' || l.startsWith('adaptive')),
'Issue #33: present_settings must include an "Adaptive" label in its model_profile AskUserQuestion options so users can select it interactively. Got labels: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Quality as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'quality' || l.startsWith('quality')),
'present_settings must include a "Quality" option. Got: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Balanced as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'balanced' || l.startsWith('balanced')),
'present_settings must include a "Balanced" option. Got: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Budget as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'budget' || l.startsWith('budget')),
'present_settings must include a "Budget" option. Got: [' + labels.join(', ') + ']'
);
});
test('present_settings includes Inherit as a selectable option', () => {
const labels = extractOptionLabels(presentBlock);
assert.ok(
labels.some(l => l === 'inherit' || l.startsWith('inherit')),
'present_settings must include an "Inherit" option. Got: [' + labels.join(', ') + ']'
);
});
// -- update_config and confirm steps reference adaptive --------------------
test('update_config step lists adaptive as a valid model_profile value', () => {
const m = settingsContent.match(/<step name="update_config">[\s\S]*?<\/step>/);
assert.ok(m, 'settings.md must have an update_config step');
assert.ok(
m[0].includes('adaptive'),
'update_config step must list "adaptive" as a valid model_profile value'
);
});
test('confirm step table shows adaptive as a possible model profile value', () => {
const m = settingsContent.match(/<step name="confirm">[\s\S]*?<\/step>/);
assert.ok(m, 'settings.md must have a confirm step');
assert.ok(
m[0].includes('adaptive'),
'confirm step must include "adaptive" in the Model Profile row'
);
});
});
});
}

View File

@@ -992,3 +992,108 @@ describe('bug #2660: extractOneLinerFromBody', () => {
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/enh-72-business-context.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:enh-72-business-context (consolidation epic #1969 B8 #1977)", () => {
// allow-test-rule: source-text-is-the-product (see #72)
// The PROJECT.md template + complete-milestone workflow .md ARE the product surface
// the runtime loads; asserting on their text tests the deployed contract directly.
/**
* Enhancement #72 — optional Business Context section in the PROJECT.md template.
*
* Contract tests over the product-text surfaces (template + milestone workflow .md):
* the template offers a Business Context section that is explicitly OPTIONAL, capped
* at the four approved one-line fields, and the milestone evolution review treats it
* as conditional so non-business projects that deleted it are never forced to review it.
*/
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const TEMPLATE = path.join(__dirname, '..', 'gsd-core', 'templates', 'project.md');
const COMPLETE_MILESTONE = path.join(__dirname, '..', 'gsd-core', 'workflows', 'complete-milestone.md');
function parseTemplateContract(content) {
const lines = content.split(/\r?\n/);
const lower = content.toLowerCase();
// The Business Context block lives between its heading and the next "## " heading.
const startIdx = lines.findIndex(l => l.trim() === '## Business Context');
let sectionBody = '';
if (startIdx !== -1) {
const rest = lines.slice(startIdx + 1);
const endOffset = rest.findIndex(l => l.startsWith('## '));
sectionBody = (endOffset === -1 ? rest : rest.slice(0, endOffset)).join('\n');
}
const fieldOf = (label) => new RegExp(`^- \\*\\*${label}\\*\\*:`, 'm').test(sectionBody);
return {
hasSection: startIdx !== -1,
// Optional-by-default: an HTML comment tells non-business projects to delete it.
hasOptionalMarker: /<!--\s*OPTIONAL/i.test(sectionBody) && /delete this section/i.test(sectionBody),
fields: {
customer: fieldOf('Customer'),
revenueModel: fieldOf('Revenue model'),
successMetric: fieldOf('Success metric'),
strategyNotes: fieldOf('Strategy notes'),
},
fieldCount: (sectionBody.match(/^- \*\*/gm) || []).length,
// Positioned between Core Value and Requirements.
orderedBetweenCoreValueAndRequirements:
lower.indexOf('## core value') < lower.indexOf('## business context') &&
lower.indexOf('## business context') < lower.indexOf('## requirements'),
hasGuidelinesEntry: /\*\*Business Context:\*\*/.test(content),
};
}
function parseMilestoneContract(content) {
const lower = content.toLowerCase();
const lines = content.split(/\r?\n/);
const reviewLine = lines.find(l =>
l.toLowerCase().includes('business context') &&
(l.toLowerCase().includes('if present') || l.toLowerCase().includes('only if')),
);
return {
mentionsBusinessContext: lower.includes('business context'),
hasConditionalReview: Boolean(reviewLine),
};
}
describe('enhancement #72 — Business Context template section', () => {
const tpl = parseTemplateContract(fs.readFileSync(TEMPLATE, 'utf-8'));
test('template includes a Business Context section', () => {
assert.ok(tpl.hasSection, 'template must contain a "## Business Context" section');
});
test('section is marked OPTIONAL with delete-for-non-business guidance', () => {
assert.ok(tpl.hasOptionalMarker, 'section must carry an OPTIONAL HTML comment telling non-business projects to delete it');
});
test('section carries exactly the four approved one-line fields', () => {
assert.ok(tpl.fields.customer, 'missing **Customer** field');
assert.ok(tpl.fields.revenueModel, 'missing **Revenue model** field');
assert.ok(tpl.fields.successMetric, 'missing **Success metric** field');
assert.ok(tpl.fields.strategyNotes, 'missing **Strategy notes** field');
assert.strictEqual(tpl.fieldCount, 4, 'section is capped at four fields (constraint reference, not a business plan)');
});
test('section is positioned between Core Value and Requirements', () => {
assert.ok(tpl.orderedBetweenCoreValueAndRequirements, 'Business Context must sit between Core Value and Requirements');
});
test('guidelines document the Business Context section', () => {
assert.ok(tpl.hasGuidelinesEntry, 'guidelines block must include a **Business Context:** entry');
});
test('milestone evolution reviews Business Context only when present', () => {
const ms = parseMilestoneContract(fs.readFileSync(COMPLETE_MILESTONE, 'utf-8'));
assert.ok(ms.mentionsBusinessContext, 'complete-milestone must mention Business Context in its review');
assert.ok(ms.hasConditionalReview, 'the Business Context milestone review must be conditional on the section being present');
});
});
});
}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,462 @@
'use strict';
// Repo-wide invariant scans.
//
// Consolidated home (epic #1969, batch B8 #1977) for regression tests that walk
// whole top-level directories (docs/, agents/, commands/, workflows/, references/,
// templates/, bin/lib/, …) asserting a CROSS-CUTTING invariant with no single
// module subject — e.g. "no retired /gsd-next token in any doc", "no gsd-sdk
// reference in any runtime surface", "ESLint coverage tracks the bin/lib migration",
// "every CLI command family fails cleanly on bad input". Each block below is folded
// verbatim from its origin issue-named file and keeps its origin issue number.
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/551-eslint-bin-lib-coverage.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:551-eslint-bin-lib-coverage (consolidation epic #1969 B8 #1977)", () => {
'use strict';
/**
* Regression / migration-gate test for #551 and ADR-457 (TS migration).
*
* ESLint must apply the correct policy to every gsd-core/bin/lib/*.cjs
* file as modules migrate from hand-written CJS to tsc-generated artifacts:
*
* - tsc-generated artifact (has src/<basename>.cts counterpart) → MUST be
* eslint-ignored. We lint the *.cts source instead (ADR-457).
* - Genuinely hand-written (no src/*.cts counterpart) → MUST be linted
* (NOT ignored). Includes scripts-generated package-identity.cjs which
* has no *.cts source.
*
* The test is filesystem-driven — it scans bin/lib at runtime and checks each
* file against the src/ directory, so it stays correct automatically as more
* modules migrate. No hardcoded lists.
*
* ESLint behaviour is verified via ESLint's own `isPathIgnored()` API so the
* test reflects real resolved flat-config precedence, not a textual scan of
* eslint.config.mjs.
*/
const { describe, test, before } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { ESLint } = require('eslint');
const ROOT = path.resolve(__dirname, '..');
const LIB_DIR = path.join(ROOT, 'gsd-core', 'bin', 'lib');
const SRC_DIR = path.join(ROOT, 'src');
/**
* Returns true if the given bin/lib/*.cjs file has a corresponding
* src/<basename>.cts TypeScript source (meaning it is tsc-generated).
*/
function hasTsSource(absPath) {
const base = path.basename(absPath, '.cjs');
return (
fs.existsSync(path.join(SRC_DIR, `${base}.cts`)) ||
fs.existsSync(path.join(SRC_DIR, `${base}.ts`))
);
}
let eslint;
before(() => {
eslint = new ESLint({ cwd: ROOT });
});
describe('ESLint coverage tracks the bin/lib TS migration (ADR-457 / #537)', () => {
/**
* Main invariant: scan every *.cjs in bin/lib and assert the correct ESLint
* policy is applied.
*/
test('each bin/lib/*.cjs is linted xor ignored according to migration state', async () => {
const wronglyIgnored = []; // hand-written but ignored — should be linted
const wronglyLinted = []; // tsc-generated but not ignored — should be ignored
const entries = fs.readdirSync(LIB_DIR).filter((e) => e.endsWith('.cjs'));
for (const entry of entries) {
const abs = path.join(LIB_DIR, entry);
const generated = hasTsSource(abs);
const ignored = await eslint.isPathIgnored(abs);
if (generated && !ignored) {
wronglyLinted.push(entry);
} else if (!generated && ignored) {
wronglyIgnored.push(entry);
}
}
assert.deepEqual(
wronglyLinted,
[],
`tsc-generated bin/lib modules not yet added to ESLint ignore list: ${wronglyLinted.join(', ')}`,
);
assert.deepEqual(
wronglyIgnored,
[],
`Hand-written bin/lib modules silently excluded from ESLint: ${wronglyIgnored.join(', ')}`,
);
});
test('semver-compare.cjs (tsc-generated publish artifact) stays eslint-ignored (ADR-457)', async () => {
const f = path.join(LIB_DIR, 'semver-compare.cjs');
assert.equal(
await eslint.isPathIgnored(f),
true,
'semver-compare.cjs is a tsc-generated publish-time artifact and must stay ignored',
);
});
test('package-identity.cjs (script-generated, no *.cts source) is linted, not ignored (#551)', async () => {
const f = path.join(LIB_DIR, 'package-identity.cjs');
assert.equal(
await eslint.isPathIgnored(f),
false,
'package-identity.cjs has no src/*.cts counterpart and must be linted, not ignored',
);
});
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-3054-stale-gsd-next-references.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-3054-stale-gsd-next-references (consolidation epic #1969 B8 #1977)", () => {
'use strict';
const { test, describe } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
function walkMd(dir, out = []) {
if (!fs.existsSync(dir)) return out;
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
const full = path.join(dir, entry.name);
if (entry.isDirectory()) walkMd(full, out);
else if (entry.name.endsWith('.md')) out.push(full);
}
return out;
}
function extractSlashCommandTokens(markdown) {
const tokenRe = /\/gsd-[a-z0-9-]+/gi;
const tokens = new Set();
let m;
while ((m = tokenRe.exec(markdown)) !== null) {
tokens.add(m[0]);
}
return tokens;
}
describe('bug #3054: user-facing docs should not reference removed /gsd-next command', () => {
test('docs, workflows, and README surfaces use /gsd-progress --next instead', () => {
const root = path.join(__dirname, '..');
const files = [
...walkMd(path.join(root, 'docs')),
...walkMd(path.join(root, 'gsd-core', 'workflows')),
...fs.readdirSync(root).filter((f) => /^README.*\.md$/.test(f)).map((f) => path.join(root, f)),
];
const offenders = [];
for (const file of files) {
const content = fs.readFileSync(file, 'utf8');
const tokens = extractSlashCommandTokens(content);
if (tokens.has('/gsd-next')) offenders.push(path.relative(root, file));
}
assert.deepStrictEqual(offenders, [], `stale /gsd-next references remain in: ${offenders.join(', ')}`);
});
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-3810-no-gsd-sdk-runtime-refs.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-3810-no-gsd-sdk-runtime-refs (consolidation epic #1969 B8 #1977)", () => {
// allow-test-rule: source-text-is-the-product (see #3810)
// Runtime prompt/hook files are deployed verbatim — their text IS what the
// runtime loads and executes. Asserting that text carries no retired `gsd-sdk`
// reference tests the deployed contract, which no behavioral seam can observe
// (there is no runtime API that enumerates "did any shipped prompt name the
// removed SDK binary").
/**
* Regression guard: no `gsd-sdk` references in runtime-facing surfaces (#339).
*
* The `@opengsd/gsd-sdk` package and its `gsd-sdk` binary were retired (ADR 0174,
* #191). The bulk runtime cleanup is already done — this test locks it in so a
* `gsd-sdk` / `GSD_SDK` reference cannot creep back into a shipped prompt or hook
* and re-introduce drift between the documented surface and the supported
* `gsd-tools` binary.
*
* Scope: runtime surfaces only — the prompts and hooks the installer ships into
* a user's runtime config dir. Explicitly NOT covered here:
* - `bin/install.js` — installer code, not a runtime-deployed prompt/hook
* surface. (It carries zero `gsd-sdk` references today; the SDK-shim
* verification subsystem was removed in #515 and the shim retired in #522.)
* - `gsd-core/bin/` — executable library code, not deployed prompt text; it
* may legitimately reference the SDK retirement in comments.
* - `tests/`, `docs/`, `.changeset/`, CI/lint scripts — legitimately reference
* the SDK retirement as history or detect its stale artifacts.
*
* Complements `tests/gsd-tools-path-refs.test.cjs`, which only catches the
* `gsd-sdk query` binary-invocation form; this catches ANY runtime reference.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const REPO_ROOT = path.join(__dirname, '..');
// Runtime surfaces the installer ships. Each entry is { dir, exts } — dir is
// repo-relative, exts is the set of file extensions whose text is deployed.
const RUNTIME_SURFACES = [
// .md prompts plus the non-.md runtime artifacts this dir also ships:
// _runtime-launcher.snippet.sh (the canonical launcher synced into every hook
// by scripts/sync-runtime-launcher.cjs) and discuss-phase/templates/*.json
// (loaded at runtime by discuss-phase.md). Scanning only .md left these two
// deployed files uncovered. (#691 review)
{ dir: path.join('gsd-core', 'workflows'), exts: ['.md', '.sh', '.json'] },
{ dir: path.join('gsd-core', 'references'), exts: ['.md'] },
// Prompt surfaces the installer deep-copies and the runtime loads via
// `@~/.claude/gsd-core/templates/*.md` anchors in workflows/commands; the
// lone config.json under templates/ ships too. (#691 review)
{ dir: path.join('gsd-core', 'templates'), exts: ['.md', '.json'] },
{ dir: path.join('gsd-core', 'contexts'), exts: ['.md'] },
{ dir: path.join('commands', 'gsd'), exts: ['.md'] },
{ dir: 'agents', exts: ['.md'] },
// Hooks ship as executable text (.js/.cjs/.sh). `hooks/dist/` is a gitignored
// build artifact regenerated from these sources, so scanning the sources is
// sufficient and avoids asserting against generated copies.
{ dir: 'hooks', exts: ['.js', '.cjs', '.sh'], skipDirs: ['dist'] },
];
// Matches every casing/separator variant of the retired SDK token:
// gsd-sdk, gsd_sdk, GSD-SDK, GSD_SDK, etc.
const SDK_REF = /gsd[-_]sdk/i;
/**
* Recursively collect files under `absDir` whose extension is in `exts`,
* skipping any directory name listed in `skipDirs`.
*/
function collectFiles(absDir, exts, skipDirs) {
if (!fs.existsSync(absDir)) return [];
const out = [];
for (const entry of fs.readdirSync(absDir, { withFileTypes: true })) {
if (entry.isDirectory()) {
if (skipDirs.includes(entry.name)) continue;
out.push(...collectFiles(path.join(absDir, entry.name), exts, skipDirs));
} else if (entry.isFile() && exts.includes(path.extname(entry.name))) {
out.push(path.join(absDir, entry.name));
}
}
return out;
}
function rel(file) {
return path.relative(REPO_ROOT, file).split(path.sep).join('/');
}
describe('#339 no gsd-sdk references in runtime surfaces', () => {
test('shipped prompts and hooks carry no retired gsd-sdk reference', () => {
const violations = [];
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
const files = collectFiles(path.join(REPO_ROOT, dir), exts, skipDirs);
for (const file of files) {
const lines = fs.readFileSync(file, 'utf-8').split(/\r?\n/);
for (let i = 0; i < lines.length; i++) {
if (SDK_REF.test(lines[i])) {
violations.push(`${rel(file)}:${i + 1}: ${lines[i].trim()}`);
}
}
}
}
assert.strictEqual(
violations.length,
0,
'Runtime surfaces must not reference the retired gsd-sdk binary/package — ' +
'use gsd-tools instead.\nViolations:\n' + violations.join('\n')
);
});
test('at least one file per configured extension is scanned (guards against an empty sweep)', () => {
// A path typo, directory rename, or stale extension could silently make
// collectFiles() return [] for part of a surface, turning the guard above
// into a no-op that always passes. Checking per-surface isn't enough: for
// gsd-core/workflows the .md files alone keep a per-surface count > 0, so
// dropping .sh/.json would stop covering _runtime-launcher.snippet.sh and
// discuss-phase/templates/*.json while the test stayed green. Assert each
// configured extension actually resolves to scanned files. (#691 review)
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
for (const ext of exts) {
const count = collectFiles(path.join(REPO_ROOT, dir), [ext], skipDirs).length;
assert.ok(
count > 0,
`Runtime surface "${dir}" resolved to 0 "${ext}" files — the path may ` +
'have moved or the extension is stale; update RUNTIME_SURFACES so the ' +
'guard keeps covering it.'
);
}
}
});
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/feat-3593-cli-negative-universal.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:feat-3593-cli-negative-universal (consolidation epic #1969 B8 #1977)", () => {
/**
* Universal CLI negative-matrix sweep across the seven command families
* named in #3593: phase, roadmap, state, config, workstream, init,
* validate.
*
* The full per-case matrix for each family belongs in dedicated files
* (the `config` family is the template — see
* `feat-3593-cli-negative-config.test.cjs`). This file pins a much
* narrower contract that EVERY family must satisfy:
*
* 1. Bare top-level command with no subcommand: must not crash with
* a V8 stack trace, must exit non-zero (or, where the bare form
* is legitimate, return a clean payload).
* 2. Unknown subcommand at command depth: must emit a typed reason
* under --json-errors and must not crash.
* 3. Shell-metacharacter as an argv element: must not be executed.
* Sentinel-file probe in the project temp dir proves this.
*
* Future per-family files will deepen each family's matrix; this file
* is the floor.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { runCli } = require('./helpers/cli-negative.cjs');
const { createTempProject, createTempGitProject, cleanup } = require('./helpers.cjs');
/**
* Each entry names a top-level command, a representative subcommand
* (for the "unknown sub" probe), and the temp-project factory.
*
* The `representativeSubcommand` is not asserted on directly — it
* exists so the test layout reads "for each family, run probe X" and
* so a future maintainer adding a family doesn't need to invent one.
*/
const FAMILIES = [
{ name: 'phase', bare: ['phase'], unknown: ['phase', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'roadmap', bare: ['roadmap'], unknown: ['roadmap', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'state', bare: ['state'], unknown: ['state', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'config', bare: ['config'], unknown: ['config', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'workstream', bare: ['workstream'], unknown: ['workstream', 'this-sub-does-not-exist'], projectFactory: createTempGitProject },
{ name: 'init', bare: ['init'], unknown: ['init', 'this-sub-does-not-exist'], projectFactory: createTempProject },
{ name: 'validate', bare: ['validate'], unknown: ['validate', 'this-sub-does-not-exist'], projectFactory: createTempProject },
];
describe('feat-3593: bare top-level command does not crash', () => {
for (const fam of FAMILIES) {
test(`${fam.name}: bare invocation exits cleanly without a stack trace`, (t) => {
const projectDir = fam.projectFactory(`cli-neg-univ-bare-${fam.name}-`);
t.after(() => cleanup(projectDir));
const result = runCli(fam.bare, { cwd: projectDir });
assert.equal(result.hasStackTrace, false, `${fam.name} bare must not leak a V8 stack frame`);
assert.equal(result.signal, null, `${fam.name} bare must not be killed by a signal`);
// We do not pin exit code here: some families legitimately treat the
// bare form as a list/status command (status 0); others reject it as
// a usage error (status ≠ 0). What we pin is "no crash."
});
}
});
describe('feat-3593: unknown subcommand emits a typed reason', () => {
for (const fam of FAMILIES) {
test(`${fam.name}: unknown subcommand fails with reason set and no stack trace`, (t) => {
const projectDir = fam.projectFactory(`cli-neg-univ-unk-${fam.name}-`);
t.after(() => cleanup(projectDir));
const result = runCli(fam.unknown, { cwd: projectDir });
assert.notEqual(result.status, 0, `${fam.name} unknown sub must exit non-zero`);
assert.equal(result.hasStackTrace, false, `${fam.name} unknown sub must not leak a stack frame`);
// When --json-errors is on, every typed-failure path lands a non-empty
// reason string from ERROR_REASON. A null reason here means the family
// is using throw/console.error somewhere — a TDD signal to wire that
// failure path through error(msg, ERROR_REASON.X).
assert.equal(typeof result.reason, 'string', `${fam.name}: reason must be a typed enum string (got: ${result.reason})`);
assert.notEqual(result.reason, '', `${fam.name}: reason must be non-empty`);
});
}
});
describe('feat-3593: shell-metacharacter argv values are NOT executed', () => {
for (const fam of FAMILIES) {
test(`${fam.name}: shell-payload as subcommand argv does NOT execute the payload`, (t) => {
const projectDir = fam.projectFactory(`cli-neg-univ-shell-${fam.name}-`);
t.after(() => cleanup(projectDir));
// Place the payload where the subcommand value goes. The payload
// would, if shell-interpreted, create a sentinel file in the
// project dir. Argv-based invocation must treat it as opaque text.
const sentinelPayload = `$(touch ${projectDir}/INJ-${fam.name})`;
const argv = [fam.name, sentinelPayload];
const result = runCli(argv, { cwd: projectDir });
// No stack trace — opaque text must not crash.
assert.equal(result.hasStackTrace, false, `${fam.name}: shell payload must not crash`);
// The sentinel file must NOT exist. We check the project dir's
// listing rather than fs.existsSync of a single path so the test
// surfaces any spelling drift.
const entries = fs.readdirSync(projectDir);
const sentinels = entries.filter((n) => n.startsWith('INJ-'));
assert.deepEqual(
sentinels,
[],
`${fam.name}: shell payload was executed — sentinel files exist: ${sentinels.join(', ')}`,
);
});
}
});
// ─── Cross-family invariants on the global --cwd flag ──────────────────────
test('--cwd with an empty value fails the same way regardless of command family', () => {
for (const fam of FAMILIES) {
const result = runCli(['--cwd', '', ...fam.bare], { cwd: process.cwd() });
assert.notEqual(result.status, 0, `${fam.name}: empty --cwd must fail`);
assert.equal(result.hasStackTrace, false, `${fam.name}: empty --cwd must not crash`);
assert.equal(result.reason, 'usage', `${fam.name}: empty --cwd reason should be 'usage', got: ${result.reason}`);
}
});
test('--cwd pointing at a non-existent path fails uniformly across families', () => {
const nonExistent = path.join(require('os').tmpdir(), 'cli-neg-univ-no-such-' + Date.now() + '-' + Math.random());
assert.equal(fs.existsSync(nonExistent), false, 'pre-check: temp path must not exist');
for (const fam of FAMILIES) {
const result = runCli(['--cwd', nonExistent, ...fam.bare], { cwd: process.cwd() });
assert.notEqual(result.status, 0, `${fam.name}: invalid --cwd must fail`);
assert.equal(result.hasStackTrace, false);
assert.equal(result.reason, 'usage', `${fam.name}: invalid --cwd reason should be 'usage'`);
}
});
});
}

View File

@@ -432,3 +432,90 @@ describe('gsd-project-researcher agent registration (#2419)', () => {
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-2559-stale-search-year.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-2559-stale-search-year (consolidation epic #1969 B8 #1977)", () => {
// allow-test-rule: source-text-is-the-product (see #2559)
// Workflow .md / agent .md / command .md / reference .md files — their text
// IS what the runtime loads. Testing text content tests the deployed contract.
// Per CONTRIBUTING.md exception matrix.
/**
* Bug #2559: Stale document references in Research phase
*
* The gsd-phase-researcher and gsd-project-researcher agents instruct
* WebSearch queries to always include "current year" (or a hardcoded
* year). This biases results toward stale dated content as time passes
* (e.g., a 2024 query run in 2026 returns stale results).
*
* Fix: Remove year-injection instructions from research agent
* WebSearch guidance so searches return current results.
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const PHASE_RESEARCHER = path.join(
__dirname,
'..',
'agents',
'gsd-phase-researcher.md'
);
const PROJECT_RESEARCHER = path.join(
__dirname,
'..',
'agents',
'gsd-project-researcher.md'
);
const FILES = [
{ label: 'gsd-phase-researcher.md', path: PHASE_RESEARCHER },
{ label: 'gsd-project-researcher.md', path: PROJECT_RESEARCHER },
];
describe('research agents do not inject year into web searches (#2559)', () => {
for (const { label, path: filePath } of FILES) {
test(`${label} contains no CURRENT_YEAR placeholder`, () => {
const content = fs.readFileSync(filePath, 'utf-8');
assert.ok(
!/CURRENT_YEAR/.test(content),
`${label} must not contain CURRENT_YEAR placeholder (causes stale-year injection)`
);
});
test(`${label} contains no hardcoded year in web search instructions`, () => {
const content = fs.readFileSync(filePath, 'utf-8');
const match = content.match(/\b20(2[3-9]|[3-9]\d)\b/);
assert.ok(
!match,
`${label} must not contain hardcoded year (found "${match && match[0]}") — biases searches toward stale content`
);
});
test(`${label} does not instruct searches to include year or current year`, () => {
const content = fs.readFileSync(filePath, 'utf-8');
// Match phrases like "include current year", "year in searches",
// "[current year]", "with year", etc.
const patterns = [
/include\s+(?:the\s+)?current\s+year/i,
/current\s+year/i,
/year\s+in\s+(?:searches|queries)/i,
/\[current year\]/i,
];
for (const pat of patterns) {
assert.ok(
!pat.test(content),
`${label} must not instruct year injection (matched /${pat.source}/)`
);
}
});
}
});
});
}

View File

@@ -1557,3 +1557,287 @@ describe('scanPhasePlans — call-site parity on mixed fixture', () => {
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/feat-3594-parser-adversarial-roadmap.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:feat-3594-parser-adversarial-roadmap (consolidation epic #1969 B8 #1977)", () => {
/**
* Adversarial roadmap-parser tests (#3594).
*
* Loads each fixture in `tests/fixtures/adversarial/roadmap/` as the
* project's `.planning/ROADMAP.md` and pins invariants on the public
* `gsd-tools roadmap get-phase <N>` surface — which routes through the
* SDK bridge when available and the CJS handler otherwise.
*
* Per CONTRIBUTING.md §"Testing Standards / Parser and project-file
* inputs", the assertion target is the typed JSON shape the CLI emits,
* not stderr prose. The harness in `tests/helpers/cli-negative.cjs`
* (introduced by #3627 / #3593) is reused here for consistency.
*
* Several fixtures encode known historical regressions:
* - fenced-code-block headings shadowing real phases (#2787)
* - decimal phase prefix collisions (#3537)
* - HTML-comment heading false positives
*
* Pre-existing parser bugs surfaced by these fixtures are NOT fixed in
* this PR — fixing them is out of scope for "add adversarial test
* coverage." Where a fixture exposes a still-open bug, the test
* asserts the *currently observed* behavior with a comment naming the
* open issue, so the flip from RED→GREEN is a one-line change the day
* the real fix lands.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const { runCli } = require('./helpers/cli-negative.cjs');
const { createTempProject, cleanup } = require('./helpers.cjs');
const FIXTURE_DIR = path.join(__dirname, 'fixtures', 'adversarial', 'roadmap');
function loadFixture(name) {
return fs.readFileSync(path.join(FIXTURE_DIR, name), 'utf-8');
}
/**
* Create a temp project whose ROADMAP.md is the named fixture's content.
* Returns the project directory; caller is responsible for cleanup.
*/
function projectWithFixture(t, fixtureName) {
const projectDir = createTempProject('roadmap-adv-' + fixtureName.replace(/\W+/g, '-') + '-');
t.after(() => cleanup(projectDir));
fs.writeFileSync(path.join(projectDir, '.planning', 'ROADMAP.md'), loadFixture(fixtureName));
return projectDir;
}
/**
* Run `gsd-tools roadmap get-phase <N>` and parse the JSON payload.
* Returns `{ ok, exit, parsed, raw }` so tests can assert on either
* the exit code or the structured payload.
*/
function getPhase(projectDir, phaseNum) {
// No --json-errors — the get-phase command outputs JSON on success
// via the normal stdout path. Reading the parsed payload is what the
// workflows downstream do, so that's what we test.
const result = runCli(['roadmap', 'get-phase', phaseNum], { cwd: projectDir, jsonErrors: false });
let parsed = null;
try {
parsed = JSON.parse(result.stdout);
} catch {
// Leave parsed null; tests that depend on it must handle that.
}
return {
exit: result.status,
ok: result.status === 0,
parsed,
raw: result.stdout,
stderr: result.stderr,
hasStackTrace: result.hasStackTrace,
};
}
// ─── Fenced code block heading shadowing ────────────────────────────────────
describe('feat-3594: roadmap parser and fenced-code-block headings (#2787)', () => {
test('phase 1 in real prose is found even when ## Phase 999 appears inside a ``` block', (t) => {
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
const result = getPhase(projectDir, '1');
assert.equal(result.hasStackTrace, false, 'no V8 stack trace');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true, 'phase 1 must be found');
assert.equal(result.parsed.phase_number, '1');
assert.match(result.parsed.phase_name, /real phase one/);
});
test('phase 999 inside a fenced block is ignored', (t) => {
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
const result = getPhase(projectDir, '999');
assert.equal(result.hasStackTrace, false, 'no stack trace');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, false, 'phase headings inside fenced blocks must not be parsed');
});
test('fenced example heading does not shadow the real phase details and backlog phase stays unresolved (#1588)', (t) => {
const projectDir = createTempProject('roadmap-1588-');
t.after(() => cleanup(projectDir));
fs.writeFileSync(
path.join(projectDir, '.planning', 'STATE.md'),
[
'---',
'gsd_state_version: 1.0',
'milestone: v1.1',
'status: planning',
'---',
'',
].join('\n')
);
fs.writeFileSync(
path.join(projectDir, '.planning', 'ROADMAP.md'),
[
'# Roadmap',
'',
'<details open>',
'<summary>v1.1 Current (Phases 8-9) - PLANNED</summary>',
'',
'- [ ] **Phase 9: Real Phase**',
'',
'</details>',
'',
'## Phase Details',
'',
'```markdown',
'### Phase 9: Fenced Example Phase',
'**Goal:** This example must not be treated as roadmap structure.',
'```',
'',
'### Phase 9: Real Phase',
'**Goal:** Use the real phase details outside the fenced block.',
'**Requirements:** REAL-01',
'',
'## Backlog',
'',
'### Phase 999.1: Backlog Thing',
'**Goal:** Future backlog item.',
'',
].join('\n')
);
const phase9 = getPhase(projectDir, '9');
assert.ok(phase9.parsed, `expected JSON payload, got: ${phase9.raw}`);
assert.equal(phase9.parsed.found, true, 'phase 9 must be found');
assert.equal(phase9.parsed.phase_name, 'Real Phase');
assert.equal(phase9.parsed.goal, 'Use the real phase details outside the fenced block.');
assert.match(phase9.parsed.section, /REAL-01/, 'real phase section must be returned');
const backlog = getPhase(projectDir, '999.1');
assert.ok(backlog.parsed, `expected JSON payload, got: ${backlog.raw}`);
assert.equal(backlog.parsed.found, false, 'backlog sentinel phase must not resolve as an active roadmap phase');
});
});
// ─── Decimal phase prefix collisions ────────────────────────────────────────
describe('feat-3594: roadmap parser handles decimal phase prefix collisions (#3537)', () => {
test('asking for phase "2" returns the integer phase, NOT phase 2.1 or 2.10', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '2');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '2');
assert.match(result.parsed.phase_name, /integer phase two/);
});
test('asking for phase "2.1" returns the decimal child', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '2.1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '2.1');
assert.match(result.parsed.phase_name, /decimal child/);
});
test('asking for phase "2.10" returns the decimal sibling, NOT phase 2.1', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '2.10');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '2.10');
assert.match(result.parsed.phase_name, /decimal phase 2\.10/);
});
test('asking for phase "21" returns phase 21, NOT phase 2 (prefix-collision guard)', (t) => {
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
const result = getPhase(projectDir, '21');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_number, '21');
assert.match(result.parsed.phase_name, /phase twenty-one/);
});
});
// ─── Unicode phase titles ───────────────────────────────────────────────────
describe('feat-3594: roadmap parser preserves Unicode phase titles', () => {
test('Japanese title round-trips through phase_name', (t) => {
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
const result = getPhase(projectDir, '1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.phase_name, '日本語フェーズ — initial setup');
});
test('emoji + smart-quote title survives', (t) => {
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
const result = getPhase(projectDir, '2');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.match(result.parsed.phase_name, /🚧/);
assert.match(result.parsed.phase_name, /Émile/);
});
test('Greek-letter title survives', (t) => {
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
const result = getPhase(projectDir, '3');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.phase_name, 'αβγ δεζ ηθι');
});
});
// ─── Repeated phase IDs ─────────────────────────────────────────────────────
describe('feat-3594: roadmap parser handles repeated phase IDs deterministically', () => {
test('two declarations of phase 1: parser returns the FIRST match (current behavior)', (t) => {
const projectDir = projectWithFixture(t, 'repeated-phase-ids.md');
const result = getPhase(projectDir, '1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
// The regex uses `content.match(...)` which returns the FIRST match.
// Pin that — a future change to last-wins or de-dup would fire.
assert.match(result.parsed.phase_name, /first declaration/);
});
});
// ─── HTML comments ──────────────────────────────────────────────────────────
describe('feat-3594: roadmap parser and HTML-commented headings', () => {
test('phase 1 in real prose is found even when phase 998/999 appear inside <!-- ... -->', (t) => {
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
const result = getPhase(projectDir, '1');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, true);
assert.equal(result.parsed.phase_name, 'real phase');
});
test('phase 999 inside an HTML comment remains ignored because backlog sentinels never resolve', (t) => {
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
const result = getPhase(projectDir, '999');
assert.equal(result.hasStackTrace, false, 'no stack trace');
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
assert.equal(result.parsed.found, false, 'backlog sentinel phases must not resolve');
});
});
// ─── Cross-corpus invariant ────────────────────────────────────────────────
describe('feat-3594: roadmap parser does not crash on ANY corpus fixture', () => {
const fixtures = fs.readdirSync(FIXTURE_DIR).filter((f) => f.endsWith('.md') && f !== 'README.md');
for (const fixture of fixtures) {
test(`fixture "${fixture}" — get-phase with arbitrary IDs must not crash`, (t) => {
const projectDir = projectWithFixture(t, fixture);
for (const id of ['1', '2', '99', '999', '0', '2.1']) {
const result = getPhase(projectDir, id);
assert.equal(result.hasStackTrace, false, `${fixture} id=${id}: no V8 stack frame allowed`);
// exit status varies (0 for found, non-zero for not-found —
// both are valid). What's pinned: the parser produced SOME output
// (either valid JSON or a clean stderr) without crashing.
}
});
}
});
});
}

View File

@@ -1218,3 +1218,203 @@ test('manager.md and autonomous.md no longer contain old "not claude" background
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-2876-skill-frontmatter-quote.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-2876-skill-frontmatter-quote (consolidation epic #1969 B8 #1977)", () => {
/**
* Bug #2876: SKILL.md frontmatter parse failure when `description` begins
* with a YAML flow indicator like `[BETA]`.
*
* description: [BETA] Offload plan phase to Claude Code's ultraplan…
*
* YAML 1.2 treats a leading `[` as the start of a flow sequence, so any
* downstream parser (gh-copilot, JetBrains' kit, etc.) fails with
* "Unexpected scalar at node end". The Copilot/Antigravity/Trae/Codebuddy
* skill+agent converters in `bin/install.js` re-emit the description
* unquoted; the Claude variant `yamlQuote(...)`s it. Bring the others
* in line so any value is round-trip-safe regardless of leading char.
*
* The test is structural: it parses each emitted frontmatter into lines
* and asserts the `description` value is a quoted YAML scalar (double or
* single quoted) when the source description starts with a flow indicator.
* It does not regex the bytes for substrings.
*/
'use strict';
process.env.GSD_TEST_MODE = '1';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const path = require('node:path');
const fs = require('node:fs');
const REPO_ROOT = path.join(__dirname, '..');
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
const install = require(installPath);
// Build a minimal Claude command source whose description starts with the
// reporter's exact flow-indicator prefix. Apostrophe in the body forces
// any naive single-quoting to also escape correctly — the canonical
// safe form is `JSON.stringify(...)` (used by yamlQuote).
const REPORTER_DESCRIPTION =
"[BETA] Offload plan phase to Claude Code's ultraplan cloud — drafts remotely while terminal stays free, review in browser with inline comments, import back via /gsd-import. Claude Code only.";
// Use unquoted description in the source frontmatter — that's exactly the
// shape that ships in commands/gsd/*.md when authors paste a description
// without quoting it (see commands/gsd/ultraplan-phase.md). The bug is
// triggered when the converter re-emits this same value to the destination
// runtime without quoting. `extractFrontmatterField` strips a single outer
// quote pair but does not unescape internal characters, so quoting the
// fixture input would actually mask the bug.
function buildClaudeCommand(description) {
return [
'---',
'name: gsd:ultraplan-phase',
`description: ${description}`,
'argument-hint: "[phase-number]"',
'allowed-tools:',
' - Read',
' - Bash',
'---',
'',
'# body',
'',
].join('\n');
}
function buildClaudeAgent(description) {
return [
'---',
'name: gsd-extract-learnings',
`description: ${description}`,
'tools: Read, Bash',
'---',
'',
'# body',
'',
].join('\n');
}
function extractFrontmatter(content) {
// Leading delimiter is `---\n`; closing is the next standalone `---`
// on its own line. Tests parse line-structurally so the assertion
// doesn't drift on whitespace/order changes (per project test-rigor).
assert.ok(content.startsWith('---'), `output must begin with frontmatter, got: ${content.slice(0, 40)}`);
const lines = content.split('\n');
let openIdx = -1;
let closeIdx = -1;
for (let i = 0; i < lines.length; i += 1) {
if (lines[i] === '---') {
if (openIdx === -1) openIdx = i;
else { closeIdx = i; break; }
}
}
assert.ok(openIdx !== -1 && closeIdx !== -1, `output must have a closed frontmatter block, got:\n${content}`);
return lines.slice(openIdx + 1, closeIdx);
}
function findDescriptionLine(frontmatterLines) {
for (const line of frontmatterLines) {
if (line.startsWith('description:')) return line;
}
assert.fail(`no description line found in frontmatter:\n${frontmatterLines.join('\n')}`);
return ''; // unreachable
}
function isQuotedYamlScalar(valueText) {
// YAML safe-quoted scalar: starts with `"` and ends with `"`, OR
// starts with `'` and ends with `'`. This is what `yamlQuote()`
// (JSON.stringify) and the Claude variant of these converters emit.
const trimmed = valueText.trim();
if (trimmed.startsWith('"') && trimmed.endsWith('"')) return true;
if (trimmed.startsWith("'") && trimmed.endsWith("'")) return true;
return false;
}
function parseQuotedYamlValue(valueText) {
const trimmed = valueText.trim();
if (trimmed.startsWith('"')) return JSON.parse(trimmed);
if (trimmed.startsWith("'")) return trimmed.slice(1, -1).replace(/''/g, "'");
return trimmed;
}
function assertDescriptionRoundTrips(emitted, expected, label) {
const fmLines = extractFrontmatter(emitted);
const descLine = findDescriptionLine(fmLines);
const valueText = descLine.slice('description:'.length);
assert.ok(
isQuotedYamlScalar(valueText),
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
);
assert.strictEqual(
parseQuotedYamlValue(valueText),
expected,
`(${label}) description must round-trip through YAML quoting unchanged.`,
);
}
const COMMAND_CONVERTERS = [
{ label: 'convertClaudeCommandToCopilotSkill', fn: (src) => install.convertClaudeCommandToCopilotSkill(src, 'gsd-ultraplan-phase') },
{ label: 'convertClaudeCommandToAntigravitySkill', fn: (src) => install.convertClaudeCommandToAntigravitySkill(src, 'gsd-ultraplan-phase') },
{ label: 'convertClaudeCommandToTraeSkill', fn: (src) => install.convertClaudeCommandToTraeSkill(src, 'gsd-ultraplan-phase') },
{ label: 'convertClaudeCommandToCodebuddySkill', fn: (src) => install.convertClaudeCommandToCodebuddySkill(src, 'gsd-ultraplan-phase') },
];
const AGENT_CONVERTERS = [
{ label: 'convertClaudeAgentToCopilotAgent', fn: (src) => install.convertClaudeAgentToCopilotAgent(src) },
{ label: 'convertClaudeAgentToAntigravityAgent', fn: (src) => install.convertClaudeAgentToAntigravityAgent(src) },
];
// A grab-bag of leading characters that all break unquoted YAML scalar
// parsing per YAML 1.2 §7.3.3 / §6.9. The reporter's case is `[`; the
// rest defend against neighbouring drift.
const FLOW_HOSTILE_PREFIXES = ['[', '{', '*', '&', '!', '|', '>', '%', '@', '`'];
// Some converters (Trae, CodeBuddy) deliberately rewrite "Claude Code"
// in body content to their target runtime name, and the rewrite cuts
// across the description too. That's correct behavior — out of scope for
// the YAML-quoting fix — so for the reporter case we assert only the
// quoting requirement, not byte-equality of the round-tripped value.
function assertDescriptionIsQuoted(emitted, label) {
const fmLines = extractFrontmatter(emitted);
const descLine = findDescriptionLine(fmLines);
const valueText = descLine.slice('description:'.length);
assert.ok(
isQuotedYamlScalar(valueText),
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
);
}
describe('bug-2876: skill+agent converters emit YAML-quoted description', () => {
for (const { label, fn } of COMMAND_CONVERTERS) {
test(`${label}: reporter's "[BETA] ..." description is quoted`, () => {
const out = fn(buildClaudeCommand(REPORTER_DESCRIPTION));
assertDescriptionIsQuoted(out, label);
});
for (const prefix of FLOW_HOSTILE_PREFIXES) {
test(`${label}: leading ${JSON.stringify(prefix)} is quoted`, () => {
// Avoid leading/trailing `'` or `"` in the payload — `extractFrontmatterField`
// strips a single outer quote char of either kind regardless of whether
// the value was actually quoted, which would obscure the round-trip
// assertion. Pre-existing behavior, out of scope for #2876.
const desc = `${prefix} edge-case payload — flow indicator at start`;
const out = fn(buildClaudeCommand(desc));
assertDescriptionRoundTrips(out, desc, `${label} prefix=${prefix}`);
});
}
}
for (const { label, fn } of AGENT_CONVERTERS) {
test(`${label}: reporter-shape "[BETA] ..." description is quoted`, () => {
const out = fn(buildClaudeAgent(REPORTER_DESCRIPTION));
assertDescriptionIsQuoted(out, label);
});
}
});
});
}

View File

@@ -838,3 +838,246 @@ describe('scanEntropyAnomalies', () => {
assert.equal(result.findings.length, 1, 'only 1 high-entropy paragraph should be flagged');
});
});
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/fix-1627-asvs-level-scaling.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:fix-1627-asvs-level-scaling (consolidation epic #1969 B8 #1977)", () => {
// allow-test-rule: source-text-is-the-product #1627
// Agent .md / reference .md files — their text IS what the runtime loads.
// Testing text content tests the deployed contract.
// Per CONTRIBUTING.md exception matrix.
/**
* Fix #1627 — ASVS level scaling
*
* Asserts that `workflow.security_asvs_level` now scales both planner
* threat-disposition rigor and auditor verification depth rather than
* being display-only.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const ROOT = path.join(__dirname, '..');
const AGENTS_DIR = path.join(ROOT, 'agents');
const REFS_DIR = path.join(ROOT, 'gsd-core', 'references');
const MANIFEST_PATH = path.join(ROOT, 'docs', 'INVENTORY-MANIFEST.json');
describe('SECURE: ASVS level scaling (#1627)', () => {
// ── 1. New reference file ────────────────────────────────────────────────
describe('security-asvs-levels.md reference', () => {
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
test('file exists', () => {
assert.ok(fs.existsSync(refPath), 'gsd-core/references/security-asvs-levels.md must exist');
});
test('defines all three levels', () => {
const content = fs.readFileSync(refPath, 'utf-8');
assert.ok(content.includes('L1'), 'must define L1');
assert.ok(content.includes('L2'), 'must define L2');
assert.ok(content.includes('L3'), 'must define L3');
});
test('L1 describes opportunistic scope and planner disposition', () => {
const content = fs.readFileSync(refPath, 'utf-8');
assert.ok(
content.toLowerCase().includes('opportunistic'),
'L1 must be described as opportunistic'
);
assert.ok(
content.includes('mitigate') && content.includes('accept'),
'must describe mitigate/accept dispositions'
);
});
test('L2 requires explicit rationale for accepted threats', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// L2 must require documented rationale for accepted risks
assert.ok(
content.includes('rationale') || content.includes('documented'),
'L2 must require documented rationale for accepted threats'
);
});
test('L3 describes deep/comprehensive verification', () => {
const content = fs.readFileSync(refPath, 'utf-8');
const lower = content.toLowerCase();
assert.ok(
lower.includes('deep') || lower.includes('comprehensive') || lower.includes('exhaustive'),
'L3 must describe deep/comprehensive verification'
);
});
test('mentions that higher levels are supersets of lower', () => {
const content = fs.readFileSync(refPath, 'utf-8');
const lower = content.toLowerCase();
assert.ok(
lower.includes('superset') || lower.includes('higher level') || lower.includes('includes all'),
'must note that higher levels are supersets of lower'
);
});
test('describes distinct auditor verification depth for each level', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// All three audit depth keywords should appear
assert.ok(content.includes('grep') || content.includes('PRESENT'), 'L1 audit depth must mention grep/presence check');
assert.ok(content.includes('boundary') || content.includes('addresses'), 'L2 audit depth must mention boundary/addresses');
assert.ok(content.includes('end-to-end') || content.includes('bypass'), 'L3 audit depth must mention end-to-end or bypass check');
});
});
// ── 2. gsd-planner.md — no hardcoded L1 in disposition ──────────────────
describe('gsd-planner.md security disposition', () => {
const plannerPath = path.join(AGENTS_DIR, 'gsd-planner.md');
test('planner security instruction does not hardcode "ASVS L1"', () => {
const content = fs.readFileSync(plannerPath, 'utf-8');
// The old bug: "mitigate if ASVS L1 requires it" — must be gone
assert.ok(
!content.includes('ASVS L1 requires it'),
'planner must not hardcode "ASVS L1 requires it"; it must reference the configured level'
);
});
test('planner references the configured OWASP ASVS level', () => {
const content = fs.readFileSync(plannerPath, 'utf-8');
assert.ok(
content.includes('OWASP ASVS level') || content.includes('configured OWASP'),
'planner must reference the configured OWASP ASVS level'
);
});
test('planner @-references security-asvs-levels.md', () => {
const content = fs.readFileSync(plannerPath, 'utf-8');
assert.ok(
content.includes('security-asvs-levels.md'),
'planner must @-reference security-asvs-levels.md'
);
});
test('planner is under the 49152-char cap', () => {
const content = fs.readFileSync(plannerPath, 'utf-8').replace(/\r\n/g, '\n').replace(/\r/g, '\n');
assert.ok(
content.length < 49152,
`gsd-planner.md must be < 49152 chars (LF-normalized); got ${content.length}`
);
});
});
// ── 3. gsd-security-auditor.md — scaled verification depth ──────────────
describe('gsd-security-auditor.md verification depth', () => {
const auditorPath = path.join(AGENTS_DIR, 'gsd-security-auditor.md');
test('auditor scales verification depth by asvs_level', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
assert.ok(
content.includes('asvs_level') || content.includes('ASVS level'),
'auditor must reference asvs_level to scale verification'
);
});
test('auditor describes L1/L2/L3 depth differences', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
// All three levels must appear in context of depth scaling
assert.ok(content.includes('L1'), 'auditor must mention L1 depth');
assert.ok(content.includes('L2'), 'auditor must mention L2 depth');
assert.ok(content.includes('L3'), 'auditor must mention L3 depth');
});
test('auditor @-references security-asvs-levels.md', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
assert.ok(
content.includes('security-asvs-levels.md'),
'auditor must @-reference security-asvs-levels.md'
);
});
test('auditor still echoes ASVS Level in structured output', () => {
const content = fs.readFileSync(auditorPath, 'utf-8');
assert.ok(
content.includes('ASVS Level:') && content.includes('{1/2/3}'),
'auditor must still emit ASVS Level in SECURED/OPEN_THREATS output'
);
});
});
// ── 4. secure-phase.md — ASVS-aware short-circuit ──────────────────────
describe('secure-phase.md short-circuit conditioned on asvs_level', () => {
const wfPath = path.join(ROOT, 'gsd-core', 'workflows', 'secure-phase.md');
test('short-circuit to Step 6 is gated on asvs_level == 1', () => {
const content = fs.readFileSync(wfPath, 'utf-8');
// The condition must reference asvs_level so that L2/L3 don't skip the auditor
assert.ok(
content.includes('asvs_level == 1'),
'secure-phase.md must gate the skip-to-Step-6 short-circuit on asvs_level == 1'
);
});
test('auditor runs at L2/L3 even when threats_open is 0 (asvs_level >= 2 branch present)', () => {
const content = fs.readFileSync(wfPath, 'utf-8');
// The >= 2 branch must explicitly say the auditor is spawned for L2/L3 deep verification
assert.ok(
content.includes('asvs_level >= 2'),
'secure-phase.md must include asvs_level >= 2 branch that does NOT skip the auditor'
);
// The >= 2 branch must make clear the auditor is spawned (not skipped)
assert.ok(
content.includes('L2/L3 deep verification') || content.includes('L2 boundary') || content.includes('L3 end-to-end'),
'secure-phase.md asvs_level >= 2 branch must reference L2/L3 deep verification'
);
});
});
// ── 5. security-asvs-levels.md — L1 medium-severity gap closed ──────────
describe('security-asvs-levels.md L1 medium-severity is specified', () => {
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
test('L1 explicitly handles medium-severity threats (no gap)', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// L1 section must say something about medium-severity
assert.ok(
content.includes('medium-severity') || content.includes('medium severity'),
'L1 must explicitly specify disposition for medium-severity threats (no ambiguity gap)'
);
});
test('L1 medium-severity disposition is conditional (trust-boundary-aware)', () => {
const content = fs.readFileSync(refPath, 'utf-8');
// L1 must distinguish between medium on primary trust boundary vs not
assert.ok(
content.includes('trust boundary') || content.includes('primary trust'),
'L1 medium-severity rule must reference trust boundary to disambiguate disposition'
);
});
});
// ── 6. Inventory manifest ─────────────────────────────────────────────────
describe('inventory manifest', () => {
test('security-asvs-levels.md is registered in INVENTORY-MANIFEST.json', () => {
const manifest = JSON.parse(fs.readFileSync(MANIFEST_PATH, 'utf-8'));
const refs = (manifest.families || {}).references || [];
assert.ok(
refs.includes('security-asvs-levels.md'),
'security-asvs-levels.md must appear in families.references of INVENTORY-MANIFEST.json'
);
});
});
});
});
}

View File

@@ -1313,3 +1313,200 @@ describe('ADR-1769 #1796: applyStatePreservation — table-driven post-sync cons
assert.deepEqual(r.postFm, { status: 'executing', progress: { percent: 10 } });
});
});
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-21-state-md-template-frontmatter.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-21-state-md-template-frontmatter (consolidation epic #1969 B8 #1977)", () => {
/**
* Regression guard — Bug #21
*
* Both STATE.md template files must include a YAML frontmatter block in their
* "File Template" section so that an AI agent creating .planning/STATE.md from
* the template produces a file that frontmatter consumers can read immediately
* (before the first `state.*` mutation calls syncStateFrontmatter).
*
* Prior to the fix, the template's File Template section began with
* `# Project State` (no frontmatter), leaving the init→first-write window
* without `gsd_state_version`, `status`, or `progress` keys.
*
* Acceptance criteria:
* 1. The template body extracted from each state.md file's File Template code
* block must begin with `---`.
* 2. The frontmatter must contain at minimum: `gsd_state_version` and `status`.
*/
'use strict';
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const REPO_ROOT = path.join(__dirname, '..');
const TEMPLATE_PATHS = [
path.join(REPO_ROOT, 'gsd-core', 'templates', 'state.md'),
];
/**
* Extract the content of the first ```markdown ... ``` code block from a
* template file. Returns the raw string (including any leading/trailing
* whitespace within the block).
*
* @param {string} fileContent - Full text of the template file.
* @returns {string} The extracted code block body.
*/
function extractFileTemplate(fileContent) {
const match = fileContent.match(/```markdown\r?\n([\s\S]*?)```/);
assert.ok(match, 'No ```markdown code block found in template file');
return match[1];
}
/**
* Minimal YAML frontmatter parser: returns the set of top-level keys present
* in the first --- ... --- block at the start of `text`. Does not parse nested
* keys — list-valued fields (e.g. `tags: [a, b]`) are recorded only by their
* key name, not their value. Returns an empty Set when the text has no frontmatter.
*
* @param {string} text
* @returns {Set<string>}
*/
function parseFrontmatterKeys(text) {
const keys = new Set();
if (!text.trimStart().startsWith('---')) return keys;
const lines = text.split(/\r?\n/);
let inBlock = false;
for (const line of lines) {
const trimmed = line.trim();
if (!inBlock) {
if (trimmed === '---') { inBlock = true; continue; }
break; // frontmatter must be at the very start
}
if (trimmed === '---') break; // end of block
const colonIdx = trimmed.indexOf(':');
if (colonIdx > 0) {
keys.add(trimmed.slice(0, colonIdx).trim());
}
}
return keys;
}
/**
* Minimal YAML frontmatter parser: returns a plain object of top-level keys
* and their scalar or nested-object values from the first --- ... --- block.
* Handles one level of indented nesting (e.g. progress.total_plans).
* Does not handle YAML lists or multi-line values.
*
* @param {string} text
* @returns {Record<string, any>}
*/
function parseFrontmatter(text) {
const result = {};
if (!text.trimStart().startsWith('---')) return result;
const lines = text.split(/\r?\n/);
let inBlock = false;
let currentKey = null;
for (const line of lines) {
const trimmed = line.trim();
if (!inBlock) {
if (trimmed === '---') { inBlock = true; continue; }
break;
}
if (trimmed === '---') break;
// Detect indented (nested) line: starts with whitespace
if (line.match(/^\s+\S/) && currentKey !== null) {
const colonIdx = trimmed.indexOf(':');
if (colonIdx > 0) {
const subKey = trimmed.slice(0, colonIdx).trim();
const rawVal = trimmed.slice(colonIdx + 1).trim();
const numVal = Number(rawVal);
if (typeof result[currentKey] !== 'object') result[currentKey] = {};
result[currentKey][subKey] = rawVal === '' ? null : (isNaN(numVal) ? rawVal : numVal);
}
} else {
currentKey = null;
const colonIdx = trimmed.indexOf(':');
if (colonIdx > 0) {
const key = trimmed.slice(0, colonIdx).trim();
const rawVal = trimmed.slice(colonIdx + 1).trim();
if (rawVal === '') {
result[key] = {};
currentKey = key;
} else {
const numVal = Number(rawVal);
result[key] = isNaN(numVal) ? rawVal.replace(/^'|'$/g, '') : numVal;
currentKey = null;
}
}
}
}
return result;
}
describe('bug #21 — STATE.md template must carry YAML frontmatter', () => {
for (const templatePath of TEMPLATE_PATHS) {
const label = path.relative(REPO_ROOT, templatePath);
test(`${label} — File Template block starts with frontmatter`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
// The template body must open with a YAML frontmatter delimiter.
assert.ok(
body.trimStart().startsWith('---'),
`${label}: File Template must start with '---' (YAML frontmatter), ` +
`but starts with: ${JSON.stringify(body.slice(0, 60))}`,
);
});
test(`${label} — frontmatter contains gsd_state_version`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
const keys = parseFrontmatterKeys(body.trimStart());
assert.ok(
keys.has('gsd_state_version'),
`${label}: frontmatter must include 'gsd_state_version', found keys: ${[...keys].join(', ')}`,
);
});
test(`${label} — frontmatter contains status`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
const keys = parseFrontmatterKeys(body.trimStart());
assert.ok(
keys.has('status'),
`${label}: frontmatter must include 'status', found keys: ${[...keys].join(', ')}`,
);
});
test(`${label} — progress sub-schema has zeroed total_plans and completed_plans`, () => {
const content = fs.readFileSync(templatePath, 'utf-8');
const body = extractFileTemplate(content);
const fm = parseFrontmatter(body.trimStart());
assert.ok(
fm.progress && typeof fm.progress === 'object',
`${label}: frontmatter must include a 'progress' sub-object`,
);
assert.strictEqual(
fm.progress.total_plans,
0,
`${label}: progress.total_plans must be 0 in the template`,
);
assert.strictEqual(
fm.progress.completed_plans,
0,
`${label}: progress.completed_plans must be 0 in the template`,
);
});
}
});
});
}

View File

@@ -65,3 +65,111 @@ describe('workflow CLI compatibility (#1759)', () => {
);
});
});
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/feat-41-ship-tdd-audit-gate-status.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:feat-41-ship-tdd-audit-gate-status (consolidation epic #1969 B8 #1977)", () => {
'use strict';
// feat(#41): /gsd-ship generate_pr_body emits a TDD Audit table + an aggregate
// `gate_status:` trailer so the per-commit TDD gate trail survives squash-merge.
// These assertions pin the shipped workflow prose in gsd-core/workflows/ship.md.
const fs = require('node:fs');
const path = require('node:path');
const assert = require('node:assert/strict');
const { describe, test } = require('node:test');
const repoRoot = path.resolve(__dirname, '..');
function readRepoFile(relativePath) {
return fs.readFileSync(path.join(repoRoot, relativePath), 'utf8');
}
describe('feat-41: ship.md TDD Audit gate_status extraction', () => {
const workflow = readRepoFile('gsd-core/workflows/ship.md');
test('adds a "## TDD Audit" section to the generated PR body', () => {
assert.match(workflow, /## TDD Audit/);
});
test('extracts gate_status via Git native trailer machinery, not a raw body grep', () => {
assert.match(workflow, /trailers:key=gate_status/);
});
test('scopes the scan to the merge-base..HEAD range', () => {
assert.match(workflow, /merge-base/);
assert.match(workflow, /\.\.HEAD/);
assert.match(workflow, /BASE_BRANCH/);
});
test('excludes merge commits from the audit', () => {
assert.match(workflow, /--no-merges/);
});
test('renders a Test commit / Impl commit / gate_status table', () => {
assert.match(workflow, /Test commit[\s\S]*Impl commit[\s\S]*gate_status/);
});
test('pairs conventional-commit test: rows with their impl commit', () => {
assert.match(workflow, /test:/);
assert.match(workflow, /pair/i);
});
test('escapes pipe characters in commit subjects so the table is not broken', () => {
assert.match(workflow, /[Ee]scape[\s\S]{0,60}\|/);
});
test('counts commits lacking a recognized gate_status trailer as missing', () => {
assert.match(workflow, /missing/);
});
test('is informational and never blocks the ship', () => {
assert.match(workflow, /informational|never block|non-blocking/i);
});
test('emits the aggregate trailer in the exact, stable key order', () => {
assert.match(
workflow,
/gate_status:\s*skill=[^,]*,\s*fallback=[^,]*,\s*exempt=[^,]*,\s*missing=/,
);
});
test('places the aggregate trailer on the final line so squash-merge carries it', () => {
assert.match(workflow, /squash/i);
assert.match(workflow, /final line|last line/i);
});
test('does not disturb the frozen #3167 core section order (Key Decisions precedes the new section)', () => {
assert.match(workflow, /## Key Decisions[\s\S]*## TDD Audit/);
});
// Hardening assertions added after adversarial review.
test('pairs test: rows only with feat:/fix: impl commits, skipping refactor/docs/chore', () => {
assert.match(workflow, /feat:[\s\S]{0,20}fix:/);
assert.match(workflow, /skipping[\s\S]{0,80}(refactor|docs|chore)/i);
});
test('normalizes the gate_status cell to a known token, never raw trailer text', () => {
assert.match(workflow, /normaliz[a-z]*[\s\S]{0,120}missing/i);
assert.match(workflow, /never the raw/i);
});
test('treats a commit with multiple gate_status trailers as missing', () => {
assert.match(workflow, /more than one[\s\S]{0,40}gate_status/i);
});
test('hardens every table cell against pipe/newline injection', () => {
assert.match(workflow, /strip[\s\S]{0,20}\\r/);
});
test('guards record/field delimiters against adversarial commit messages', () => {
assert.match(workflow, /NUL|%x00|delimiter/i);
});
});
});
}

View File

@@ -2575,3 +2575,88 @@ describe('gsd-validate-commit.sh delegates to git-cmd.js', () => {
});
});
}
// ────────────────────────────────────────────────────────────────────────
// Folded from tests/bug-3384-secondary-defects.test.cjs — consolidation epic #1969 (B8 #1977)
// ────────────────────────────────────────────────────────────────────────
{
const { describe: __foldDescribe } = require('node:test');
__foldDescribe("folded:bug-3384-secondary-defects (consolidation epic #1969 B8 #1977)", () => {
// allow-test-rule: source-text-is-the-product (see #3384)
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const repoRoot = path.resolve(__dirname, '..');
const WORKTREE_BRANCH_CHECK_FRAGMENT = path.join(repoRoot, 'gsd-core', 'references', 'worktree-branch-check.md');
function read(relPath) {
return fs.readFileSync(path.join(repoRoot, relPath), 'utf8');
}
describe('bug #3384: adjacent worktree data-loss guards', () => {
test('worktree cleanup CLI preserves caller cwd instead of resolving project root', () => {
const source = read('gsd-core/bin/gsd-tools.cjs');
const skipSet = source.slice(
source.indexOf('const SKIP_ROOT_RESOLUTION = new Set(['),
source.indexOf('if (!SKIP_ROOT_RESOLUTION.has(command))'),
);
assert.match(skipSet, /'worktree'/);
});
test('diagnose-issues references canonical fragment; fragment is verify-only and fails closed (#48)', () => {
// diagnose-issues.md now references the canonical fragment rather than
// inlining the block. Verify (a) it references the fragment and (b) the
// fragment itself has the correct ordering: symbolic-ref/HEAD assertion and
// ^worktree-agent- allow-list appear before any work, and (c) the fragment
// is verify-only — no destructive self-recovery.
const diagnoseSource = read('gsd-core/workflows/diagnose-issues.md');
assert.ok(
diagnoseSource.includes('worktree-branch-check.md'),
'diagnose-issues.md must reference the canonical worktree-branch-check.md fragment'
);
const fragmentSource = fs.readFileSync(WORKTREE_BRANCH_CHECK_FRAGMENT, 'utf8');
const branchCheck = fragmentSource.indexOf('HEAD_REF=$(git symbolic-ref --quiet HEAD || echo');
const namespaceCheck = fragmentSource.indexOf('^worktree-agent-');
assert.ok(branchCheck > 0, 'canonical fragment must assert HEAD before any work');
assert.ok(namespaceCheck > branchCheck, 'canonical fragment must require disposable worktree-agent branch');
// #48: verify-only — the destructive self-recovery is gone; the fragment fails closed instead.
assert.ok(!fragmentSource.includes('git reset --hard {EXPECTED_BASE}'), 'canonical fragment must not self-recover via reset --hard — orchestrator owns recovery (#48)');
assert.ok(fragmentSource.includes('exit 42'), 'canonical fragment must fail closed with exit 42 on base mismatch (#48)');
});
test('remove-workspace fails closed when git worktree remove fails', () => {
const source = read('gsd-core/workflows/remove-workspace.md');
const init = source.indexOf('REMOVE_FAILED=false');
const loop = source.indexOf('For each repo in the workspace');
const remove = source.indexOf('git worktree remove "$WORKSPACE_PATH/$REPO_NAME"');
assert.doesNotMatch(
source,
/git worktree remove "\$WORKSPACE_PATH\/\$REPO_NAME" 2>&1 \|\| true/,
'worktree removal failures must not be swallowed',
);
assert.ok(init > 0 && init < loop, 'REMOVE_FAILED must initialize once before the per-repo loop');
assert.ok(remove > loop, 'worktree removal should remain inside the per-repo loop');
assert.match(source, /Refusing to delete "\$WORKSPACE_PATH"/);
});
test('validate health warns when worktree inventory cannot be listed', () => {
const source = read('gsd-core/bin/lib/verify.cjs');
// Accept both hand-written dot access and the tsc-compiled bracket form
// (ADR-457: verify.cjs is now emitted from src/verify.cts):
// hand-written: worktreeHealth.reason === 'git_list_failed'
// tsc-compiled: worktreeHealth['reason'] === 'git_list_failed'
const failureBranch = source.search(/worktreeHealth(?:\.reason|\['reason'\]) === 'git_list_failed'/);
const warning = source.indexOf("addIssue('warning', 'W020'", failureBranch);
assert.ok(failureBranch > 0, 'verify health should branch on git_list_failed');
assert.ok(warning > failureBranch, 'git_list_failed should emit W020 degraded-health warning');
});
});
});
}