test(#1977): consolidate 22 misc + repo-invariant regression tests
Final epic-#1969 batch. Fold 22 issue-named files: the 4 genuine repo-wide invariant scans (551-eslint-bin-lib-coverage, bug-3054 stale /gsd-next, bug-3810 no-gsd-sdk-runtime-refs, feat-3593 cli-negative-universal) into a NEW shared repo-invariants.test.cjs; the other 18 as singletons into their nearest module suite (model-resolver, codex-config, runtime-converters, security, state-transition, worktree-safety, roadmap-parser, etc.). Verbatim block-scoped describe wrappers; 334 subtests conserved 1:1. Host-env pre-check (B2+B6): the 6 CLI folds into GSD_TEST_MODE-setting hosts (model-resolver/ codex-config/runtime-converters) are benign — each origin independently sets GSD_TEST_MODE=1 itself (idempotent), unlike the B6 real-install case. Regenerates regression-name allowlist (222->213), ratchets file-count allowlist (state 17->16), makes 7 relocated allow-test-rule exemptions issue-ref-compliant (ADR-456; prunes stale ids). Repoints 2 tests/ refs in docs/TESTING-SUITES.md. lint:ci green. Part of epic #1969. Closes #1977. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -221,8 +221,8 @@ node scripts/ci-test-scope.cjs --base origin/next --head HEAD
|
||||
## Test strategy: #443 effort + fast_mode engine
|
||||
|
||||
> Feature: unified cross-provider effort and fast_mode knobs (issue #443).
|
||||
> Test files: `tests/feat-443-effort-fast-mode.test.cjs` (unit),
|
||||
> `tests/feat-443-effort-fast-mode.integration.test.cjs` (integration).
|
||||
> Test files: `tests/model-resolver.test.cjs` (unit),
|
||||
> `tests/model-resolver.test.cjs` (integration).
|
||||
|
||||
### Testing pyramid
|
||||
|
||||
|
||||
@@ -18,12 +18,8 @@
|
||||
"tests/bug-211-launcher-home-fallback.test.cjs :: structural/behavioral regression for the ~/.claude fallback arm in",
|
||||
"tests/bug-2136-sh-hook-version.test.cjs :: structural-regression-guard",
|
||||
"tests/bug-2543-gsd-slash-namespace.test.cjs :: structural-regression-guard",
|
||||
"tests/bug-2559-stale-search-year.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-2772-gitmodules-path-intersection.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-2808-skill-hyphen-name.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-2839-review-fix-transactional-cleanup.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-33-settings-model-profile-adaptive.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-3384-secondary-defects.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-3446-resume-continue-here-discovery.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-3491-nested-git-worktree.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-3523-cjs-loadconfig-branching-strategy-warning.test.cjs :: validates runtime CLI stdout/stderr warning behavior, not source grep",
|
||||
@@ -33,14 +29,12 @@
|
||||
"tests/bug-3683-command-colon-namespace-leak.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-3683-workflow-colon-namespace-leak.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-3689-resume-glob-nomatch.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-3810-no-gsd-sdk-runtime-refs.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-444-resolver-local-claude-install.test.cjs :: structural/behavioral regression for the repo-local .claude/ install",
|
||||
"tests/bug-619-codebase-drift-gate-shim.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-622-graphify-optional-graph-html.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-630-wave-cleanup-orchestrator-root.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-685-windowshide-spawn.test.cjs :: source-text-is-the-product",
|
||||
"tests/bug-891-non-claude-runtime-home-fallback.test.cjs :: structural/behavioral regression for non-Claude runtime-home",
|
||||
"tests/bug-patterns-reference.test.cjs :: source-text-is-the-product",
|
||||
"tests/chain-flag-plan-phase.test.cjs :: source-text-is-the-product",
|
||||
"tests/changeset-cli.test.cjs :: reads a product workflow .md file (not CJS source) to verify",
|
||||
"tests/check-update-config-dir.test.cjs :: structural-regression-guard",
|
||||
@@ -82,7 +76,6 @@
|
||||
"tests/enh-2789-description-budget.test.cjs :: source-text-is-the-product",
|
||||
"tests/enh-2790-skill-consolidation.test.cjs :: source-text-is-the-product",
|
||||
"tests/enh-48-cwd-drift-guard-e2e.test.cjs :: integration-test-input",
|
||||
"tests/enh-72-business-context.test.cjs :: source-text-is-the-product",
|
||||
"tests/eslint-rules.test.cjs :: <source-grep reason> must still error",
|
||||
"tests/eslint-rules.test.cjs :: pending migration",
|
||||
"tests/eslint-rules.test.cjs :: source-text-is-the-product",
|
||||
@@ -95,7 +88,6 @@
|
||||
"tests/execute-phase-worktree-artifacts.test.cjs :: source-text-is-the-product",
|
||||
"tests/explore-command.test.cjs :: source-text-is-the-product",
|
||||
"tests/extract-learnings.test.cjs :: source-text-is-the-product",
|
||||
"tests/feat-2840-issue-driven-orchestration-guide.test.cjs :: structural-IR parser for a docs guide. The .includes()",
|
||||
"tests/feat-3039-help-tiered.test.cjs :: source-text-is-the-product",
|
||||
"tests/forensics.test.cjs :: source-text-is-the-product",
|
||||
"tests/frontmatter-cli.test.cjs :: source-text-is-the-product",
|
||||
|
||||
@@ -3,27 +3,19 @@
|
||||
"bug-1367-claude-local-flat-command-layout.test.cjs",
|
||||
"bug-1834-sh-hooks-installed.test.cjs",
|
||||
"bug-1974-context-exhaustion-record.test.cjs",
|
||||
"bug-21-state-md-template-frontmatter.test.cjs",
|
||||
"bug-211-launcher-home-fallback.test.cjs",
|
||||
"bug-2136-sh-hook-version.test.cjs",
|
||||
"bug-2344-read-guard-claudecode-env.test.cjs",
|
||||
"bug-2451-context-monitor-over-report.test.cjs",
|
||||
"bug-2520-read-guard-hook-subprocess-env.test.cjs",
|
||||
"bug-2543-gsd-slash-namespace.test.cjs",
|
||||
"bug-2559-stale-search-year.test.cjs",
|
||||
"bug-260-worktree-path-guard.test.cjs",
|
||||
"bug-261-worktree-force-add-guard.test.cjs",
|
||||
"bug-2772-gitmodules-path-intersection.test.cjs",
|
||||
"bug-2808-skill-hyphen-name.test.cjs",
|
||||
"bug-2839-review-fix-transactional-cleanup.test.cjs",
|
||||
"bug-2866-codex-strip-no-trailing-newline.test.cjs",
|
||||
"bug-2876-skill-frontmatter-quote.test.cjs",
|
||||
"bug-2916-handle-branching-default-base.test.cjs",
|
||||
"bug-2995-post-install-script-paths.test.cjs",
|
||||
"bug-3019-help-passthrough.test.cjs",
|
||||
"bug-3054-stale-gsd-next-references.test.cjs",
|
||||
"bug-33-settings-model-profile-adaptive.test.cjs",
|
||||
"bug-3384-secondary-defects.test.cjs",
|
||||
"bug-3442-shim-projection-drift-guard.test.cjs",
|
||||
"bug-3446-resume-continue-here-discovery.test.cjs",
|
||||
"bug-3491-nested-git-worktree.test.cjs",
|
||||
@@ -37,7 +29,6 @@
|
||||
"bug-3683-command-colon-namespace-leak.test.cjs",
|
||||
"bug-3683-workflow-colon-namespace-leak.test.cjs",
|
||||
"bug-3689-resume-glob-nomatch.test.cjs",
|
||||
"bug-3810-no-gsd-sdk-runtime-refs.test.cjs",
|
||||
"bug-444-resolver-local-claude-install.test.cjs",
|
||||
"bug-619-codebase-drift-gate-shim.test.cjs",
|
||||
"bug-622-graphify-optional-graph-html.test.cjs",
|
||||
|
||||
@@ -58,7 +58,6 @@
|
||||
},
|
||||
"state": {
|
||||
"files": [
|
||||
"bug-21-state-md-template-frontmatter.test.cjs",
|
||||
"state-acquirestatelock-non-eexist.test.cjs",
|
||||
"state-prune.test.cjs",
|
||||
"state-rebuild-cli.test.cjs",
|
||||
|
||||
@@ -1,102 +0,0 @@
|
||||
'use strict';
|
||||
|
||||
/**
|
||||
* Regression / migration-gate test for #551 and ADR-457 (TS migration).
|
||||
*
|
||||
* ESLint must apply the correct policy to every gsd-core/bin/lib/*.cjs
|
||||
* file as modules migrate from hand-written CJS to tsc-generated artifacts:
|
||||
*
|
||||
* - tsc-generated artifact (has src/<basename>.cts counterpart) → MUST be
|
||||
* eslint-ignored. We lint the *.cts source instead (ADR-457).
|
||||
* - Genuinely hand-written (no src/*.cts counterpart) → MUST be linted
|
||||
* (NOT ignored). Includes scripts-generated package-identity.cjs which
|
||||
* has no *.cts source.
|
||||
*
|
||||
* The test is filesystem-driven — it scans bin/lib at runtime and checks each
|
||||
* file against the src/ directory, so it stays correct automatically as more
|
||||
* modules migrate. No hardcoded lists.
|
||||
*
|
||||
* ESLint behaviour is verified via ESLint's own `isPathIgnored()` API so the
|
||||
* test reflects real resolved flat-config precedence, not a textual scan of
|
||||
* eslint.config.mjs.
|
||||
*/
|
||||
|
||||
const { describe, test, before } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const { ESLint } = require('eslint');
|
||||
|
||||
const ROOT = path.resolve(__dirname, '..');
|
||||
const LIB_DIR = path.join(ROOT, 'gsd-core', 'bin', 'lib');
|
||||
const SRC_DIR = path.join(ROOT, 'src');
|
||||
|
||||
/**
|
||||
* Returns true if the given bin/lib/*.cjs file has a corresponding
|
||||
* src/<basename>.cts TypeScript source (meaning it is tsc-generated).
|
||||
*/
|
||||
function hasTsSource(absPath) {
|
||||
const base = path.basename(absPath, '.cjs');
|
||||
return (
|
||||
fs.existsSync(path.join(SRC_DIR, `${base}.cts`)) ||
|
||||
fs.existsSync(path.join(SRC_DIR, `${base}.ts`))
|
||||
);
|
||||
}
|
||||
|
||||
let eslint;
|
||||
before(() => {
|
||||
eslint = new ESLint({ cwd: ROOT });
|
||||
});
|
||||
|
||||
describe('ESLint coverage tracks the bin/lib TS migration (ADR-457 / #537)', () => {
|
||||
/**
|
||||
* Main invariant: scan every *.cjs in bin/lib and assert the correct ESLint
|
||||
* policy is applied.
|
||||
*/
|
||||
test('each bin/lib/*.cjs is linted xor ignored according to migration state', async () => {
|
||||
const wronglyIgnored = []; // hand-written but ignored — should be linted
|
||||
const wronglyLinted = []; // tsc-generated but not ignored — should be ignored
|
||||
|
||||
const entries = fs.readdirSync(LIB_DIR).filter((e) => e.endsWith('.cjs'));
|
||||
for (const entry of entries) {
|
||||
const abs = path.join(LIB_DIR, entry);
|
||||
const generated = hasTsSource(abs);
|
||||
const ignored = await eslint.isPathIgnored(abs);
|
||||
|
||||
if (generated && !ignored) {
|
||||
wronglyLinted.push(entry);
|
||||
} else if (!generated && ignored) {
|
||||
wronglyIgnored.push(entry);
|
||||
}
|
||||
}
|
||||
|
||||
assert.deepEqual(
|
||||
wronglyLinted,
|
||||
[],
|
||||
`tsc-generated bin/lib modules not yet added to ESLint ignore list: ${wronglyLinted.join(', ')}`,
|
||||
);
|
||||
assert.deepEqual(
|
||||
wronglyIgnored,
|
||||
[],
|
||||
`Hand-written bin/lib modules silently excluded from ESLint: ${wronglyIgnored.join(', ')}`,
|
||||
);
|
||||
});
|
||||
|
||||
test('semver-compare.cjs (tsc-generated publish artifact) stays eslint-ignored (ADR-457)', async () => {
|
||||
const f = path.join(LIB_DIR, 'semver-compare.cjs');
|
||||
assert.equal(
|
||||
await eslint.isPathIgnored(f),
|
||||
true,
|
||||
'semver-compare.cjs is a tsc-generated publish-time artifact and must stay ignored',
|
||||
);
|
||||
});
|
||||
|
||||
test('package-identity.cjs (script-generated, no *.cts source) is linted, not ignored (#551)', async () => {
|
||||
const f = path.join(LIB_DIR, 'package-identity.cjs');
|
||||
assert.equal(
|
||||
await eslint.isPathIgnored(f),
|
||||
false,
|
||||
'package-identity.cjs has no src/*.cts counterpart and must be linted, not ignored',
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -82,3 +82,128 @@ describe('READING: agents with reading blocks use <required_reading>', () => {
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-patterns-reference.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-patterns-reference (consolidation epic #1969 B8 #1977)", () => {
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
// Workflow .md / agent .md / command .md / reference .md files — their text
|
||||
// IS what the runtime loads. Testing text content tests the deployed contract.
|
||||
// Per CONTRIBUTING.md exception matrix.
|
||||
|
||||
/**
|
||||
* Common Bug Patterns Reference Tests
|
||||
*
|
||||
* Structural tests for the common-bug-patterns.md reference file:
|
||||
* - File exists at expected path
|
||||
* - Contains expected bug pattern categories (at least 5 of 10)
|
||||
* - Debugger agent references the file in required_reading
|
||||
*/
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const REFERENCE_PATH = path.join(
|
||||
__dirname, '..', 'gsd-core', 'references', 'common-bug-patterns.md'
|
||||
);
|
||||
const DEBUGGER_AGENT_PATH = path.join(
|
||||
__dirname, '..', 'agents', 'gsd-debugger.md'
|
||||
);
|
||||
|
||||
const EXPECTED_CATEGORIES = [
|
||||
'Off-by-One',
|
||||
'Null',
|
||||
'Async',
|
||||
'State Management',
|
||||
'Import',
|
||||
'Environment',
|
||||
'Data Shape',
|
||||
'String Handling',
|
||||
'File System',
|
||||
'Error Handling',
|
||||
];
|
||||
|
||||
describe('common-bug-patterns.md reference', () => {
|
||||
test('reference file exists', () => {
|
||||
assert.ok(
|
||||
fs.existsSync(REFERENCE_PATH),
|
||||
`Expected reference file at ${REFERENCE_PATH}`
|
||||
);
|
||||
});
|
||||
|
||||
test('has title and intro', () => {
|
||||
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
|
||||
assert.ok(
|
||||
content.startsWith('# Common Bug Patterns'),
|
||||
'File should start with "# Common Bug Patterns" title'
|
||||
);
|
||||
assert.ok(
|
||||
content.includes('---'),
|
||||
'File should contain --- separator after intro'
|
||||
);
|
||||
});
|
||||
|
||||
test('contains at least 5 of 10 expected categories', () => {
|
||||
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
|
||||
const found = EXPECTED_CATEGORIES.filter(cat =>
|
||||
content.toLowerCase().includes(cat.toLowerCase())
|
||||
);
|
||||
assert.ok(
|
||||
found.length >= 5,
|
||||
`Expected at least 5 categories, found ${found.length}: ${found.join(', ')}`
|
||||
);
|
||||
});
|
||||
|
||||
test('each pattern category has at least one bold bullet item', () => {
|
||||
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
|
||||
// Only check sections inside <patterns> block, not <usage>
|
||||
const patternsBlock = (content.split('<patterns>')[1] || '').split('</patterns>')[0];
|
||||
const sections = patternsBlock.split(/^## /m).slice(1);
|
||||
assert.ok(sections.length >= 5, `Expected at least 5 pattern sections, got ${sections.length}`);
|
||||
for (const section of sections) {
|
||||
const title = section.split('\n')[0].trim();
|
||||
const bullets = section.match(/^- \*\*/gm);
|
||||
assert.ok(
|
||||
bullets && bullets.length >= 1,
|
||||
`Pattern section "${title}" should have at least one "- **" bullet item`
|
||||
);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('debugger agent references bug patterns', () => {
|
||||
test('gsd-debugger.md exists', () => {
|
||||
assert.ok(
|
||||
fs.existsSync(DEBUGGER_AGENT_PATH),
|
||||
`Expected debugger agent at ${DEBUGGER_AGENT_PATH}`
|
||||
);
|
||||
});
|
||||
|
||||
test('gsd-debugger.md references common-bug-patterns.md', () => {
|
||||
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('common-bug-patterns.md'),
|
||||
'Debugger agent should reference common-bug-patterns.md'
|
||||
);
|
||||
});
|
||||
|
||||
test('reference is inside <required_reading> block', () => {
|
||||
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
|
||||
const reqReadMatch = content.match(
|
||||
/<required_reading>([\s\S]*?)<\/required_reading>/
|
||||
);
|
||||
assert.ok(reqReadMatch, 'Debugger agent should have a <required_reading> block');
|
||||
assert.ok(
|
||||
reqReadMatch[1].includes('common-bug-patterns.md'),
|
||||
'common-bug-patterns.md should be inside <required_reading> block'
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1,187 +0,0 @@
|
||||
/**
|
||||
* Regression guard — Bug #21
|
||||
*
|
||||
* Both STATE.md template files must include a YAML frontmatter block in their
|
||||
* "File Template" section so that an AI agent creating .planning/STATE.md from
|
||||
* the template produces a file that frontmatter consumers can read immediately
|
||||
* (before the first `state.*` mutation calls syncStateFrontmatter).
|
||||
*
|
||||
* Prior to the fix, the template's File Template section began with
|
||||
* `# Project State` (no frontmatter), leaving the init→first-write window
|
||||
* without `gsd_state_version`, `status`, or `progress` keys.
|
||||
*
|
||||
* Acceptance criteria:
|
||||
* 1. The template body extracted from each state.md file's File Template code
|
||||
* block must begin with `---`.
|
||||
* 2. The frontmatter must contain at minimum: `gsd_state_version` and `status`.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
|
||||
const TEMPLATE_PATHS = [
|
||||
path.join(REPO_ROOT, 'gsd-core', 'templates', 'state.md'),
|
||||
];
|
||||
|
||||
/**
|
||||
* Extract the content of the first ```markdown ... ``` code block from a
|
||||
* template file. Returns the raw string (including any leading/trailing
|
||||
* whitespace within the block).
|
||||
*
|
||||
* @param {string} fileContent - Full text of the template file.
|
||||
* @returns {string} The extracted code block body.
|
||||
*/
|
||||
function extractFileTemplate(fileContent) {
|
||||
const match = fileContent.match(/```markdown\r?\n([\s\S]*?)```/);
|
||||
assert.ok(match, 'No ```markdown code block found in template file');
|
||||
return match[1];
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal YAML frontmatter parser: returns the set of top-level keys present
|
||||
* in the first --- ... --- block at the start of `text`. Does not parse nested
|
||||
* keys — list-valued fields (e.g. `tags: [a, b]`) are recorded only by their
|
||||
* key name, not their value. Returns an empty Set when the text has no frontmatter.
|
||||
*
|
||||
* @param {string} text
|
||||
* @returns {Set<string>}
|
||||
*/
|
||||
function parseFrontmatterKeys(text) {
|
||||
const keys = new Set();
|
||||
if (!text.trimStart().startsWith('---')) return keys;
|
||||
const lines = text.split(/\r?\n/);
|
||||
let inBlock = false;
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!inBlock) {
|
||||
if (trimmed === '---') { inBlock = true; continue; }
|
||||
break; // frontmatter must be at the very start
|
||||
}
|
||||
if (trimmed === '---') break; // end of block
|
||||
const colonIdx = trimmed.indexOf(':');
|
||||
if (colonIdx > 0) {
|
||||
keys.add(trimmed.slice(0, colonIdx).trim());
|
||||
}
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal YAML frontmatter parser: returns a plain object of top-level keys
|
||||
* and their scalar or nested-object values from the first --- ... --- block.
|
||||
* Handles one level of indented nesting (e.g. progress.total_plans).
|
||||
* Does not handle YAML lists or multi-line values.
|
||||
*
|
||||
* @param {string} text
|
||||
* @returns {Record<string, any>}
|
||||
*/
|
||||
function parseFrontmatter(text) {
|
||||
const result = {};
|
||||
if (!text.trimStart().startsWith('---')) return result;
|
||||
const lines = text.split(/\r?\n/);
|
||||
let inBlock = false;
|
||||
let currentKey = null;
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!inBlock) {
|
||||
if (trimmed === '---') { inBlock = true; continue; }
|
||||
break;
|
||||
}
|
||||
if (trimmed === '---') break;
|
||||
// Detect indented (nested) line: starts with whitespace
|
||||
if (line.match(/^\s+\S/) && currentKey !== null) {
|
||||
const colonIdx = trimmed.indexOf(':');
|
||||
if (colonIdx > 0) {
|
||||
const subKey = trimmed.slice(0, colonIdx).trim();
|
||||
const rawVal = trimmed.slice(colonIdx + 1).trim();
|
||||
const numVal = Number(rawVal);
|
||||
if (typeof result[currentKey] !== 'object') result[currentKey] = {};
|
||||
result[currentKey][subKey] = rawVal === '' ? null : (isNaN(numVal) ? rawVal : numVal);
|
||||
}
|
||||
} else {
|
||||
currentKey = null;
|
||||
const colonIdx = trimmed.indexOf(':');
|
||||
if (colonIdx > 0) {
|
||||
const key = trimmed.slice(0, colonIdx).trim();
|
||||
const rawVal = trimmed.slice(colonIdx + 1).trim();
|
||||
if (rawVal === '') {
|
||||
result[key] = {};
|
||||
currentKey = key;
|
||||
} else {
|
||||
const numVal = Number(rawVal);
|
||||
result[key] = isNaN(numVal) ? rawVal.replace(/^'|'$/g, '') : numVal;
|
||||
currentKey = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
describe('bug #21 — STATE.md template must carry YAML frontmatter', () => {
|
||||
for (const templatePath of TEMPLATE_PATHS) {
|
||||
const label = path.relative(REPO_ROOT, templatePath);
|
||||
|
||||
test(`${label} — File Template block starts with frontmatter`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
|
||||
// The template body must open with a YAML frontmatter delimiter.
|
||||
assert.ok(
|
||||
body.trimStart().startsWith('---'),
|
||||
`${label}: File Template must start with '---' (YAML frontmatter), ` +
|
||||
`but starts with: ${JSON.stringify(body.slice(0, 60))}`,
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} — frontmatter contains gsd_state_version`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
const keys = parseFrontmatterKeys(body.trimStart());
|
||||
|
||||
assert.ok(
|
||||
keys.has('gsd_state_version'),
|
||||
`${label}: frontmatter must include 'gsd_state_version', found keys: ${[...keys].join(', ')}`,
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} — frontmatter contains status`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
const keys = parseFrontmatterKeys(body.trimStart());
|
||||
|
||||
assert.ok(
|
||||
keys.has('status'),
|
||||
`${label}: frontmatter must include 'status', found keys: ${[...keys].join(', ')}`,
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} — progress sub-schema has zeroed total_plans and completed_plans`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
const fm = parseFrontmatter(body.trimStart());
|
||||
|
||||
assert.ok(
|
||||
fm.progress && typeof fm.progress === 'object',
|
||||
`${label}: frontmatter must include a 'progress' sub-object`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
fm.progress.total_plans,
|
||||
0,
|
||||
`${label}: progress.total_plans must be 0 in the template`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
fm.progress.completed_plans,
|
||||
0,
|
||||
`${label}: progress.completed_plans must be 0 in the template`,
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
});
|
||||
@@ -1,78 +0,0 @@
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
// Workflow .md / agent .md / command .md / reference .md files — their text
|
||||
// IS what the runtime loads. Testing text content tests the deployed contract.
|
||||
// Per CONTRIBUTING.md exception matrix.
|
||||
|
||||
/**
|
||||
* Bug #2559: Stale document references in Research phase
|
||||
*
|
||||
* The gsd-phase-researcher and gsd-project-researcher agents instruct
|
||||
* WebSearch queries to always include "current year" (or a hardcoded
|
||||
* year). This biases results toward stale dated content as time passes
|
||||
* (e.g., a 2024 query run in 2026 returns stale results).
|
||||
*
|
||||
* Fix: Remove year-injection instructions from research agent
|
||||
* WebSearch guidance so searches return current results.
|
||||
*/
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const PHASE_RESEARCHER = path.join(
|
||||
__dirname,
|
||||
'..',
|
||||
'agents',
|
||||
'gsd-phase-researcher.md'
|
||||
);
|
||||
const PROJECT_RESEARCHER = path.join(
|
||||
__dirname,
|
||||
'..',
|
||||
'agents',
|
||||
'gsd-project-researcher.md'
|
||||
);
|
||||
|
||||
const FILES = [
|
||||
{ label: 'gsd-phase-researcher.md', path: PHASE_RESEARCHER },
|
||||
{ label: 'gsd-project-researcher.md', path: PROJECT_RESEARCHER },
|
||||
];
|
||||
|
||||
describe('research agents do not inject year into web searches (#2559)', () => {
|
||||
for (const { label, path: filePath } of FILES) {
|
||||
test(`${label} contains no CURRENT_YEAR placeholder`, () => {
|
||||
const content = fs.readFileSync(filePath, 'utf-8');
|
||||
assert.ok(
|
||||
!/CURRENT_YEAR/.test(content),
|
||||
`${label} must not contain CURRENT_YEAR placeholder (causes stale-year injection)`
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} contains no hardcoded year in web search instructions`, () => {
|
||||
const content = fs.readFileSync(filePath, 'utf-8');
|
||||
const match = content.match(/\b20(2[3-9]|[3-9]\d)\b/);
|
||||
assert.ok(
|
||||
!match,
|
||||
`${label} must not contain hardcoded year (found "${match && match[0]}") — biases searches toward stale content`
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} does not instruct searches to include year or current year`, () => {
|
||||
const content = fs.readFileSync(filePath, 'utf-8');
|
||||
// Match phrases like "include current year", "year in searches",
|
||||
// "[current year]", "with year", etc.
|
||||
const patterns = [
|
||||
/include\s+(?:the\s+)?current\s+year/i,
|
||||
/current\s+year/i,
|
||||
/year\s+in\s+(?:searches|queries)/i,
|
||||
/\[current year\]/i,
|
||||
];
|
||||
for (const pat of patterns) {
|
||||
assert.ok(
|
||||
!pat.test(content),
|
||||
`${label} must not instruct year injection (matched /${pat.source}/)`
|
||||
);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
@@ -1,171 +0,0 @@
|
||||
/**
|
||||
* Regression test for bug #2839
|
||||
*
|
||||
* /gsd-code-review-fix cleanup tail is non-transactional. If the agent is
|
||||
* interrupted (system restart, OOM kill) AFTER the last fix commit but
|
||||
* BEFORE `git worktree remove`, the worktree is orphaned in
|
||||
* `git worktree list`, the agent's branch is left with unmerged commits,
|
||||
* and STATE.md is never advanced. To anyone reading main only, the phase
|
||||
* looks "ready to plan" while critical fixes sit on a dangling branch.
|
||||
*
|
||||
* Fix: introduce a recovery sentinel JSON at
|
||||
* ${PHASE_DIR}/.review-fix-recovery-pending.json
|
||||
* The sentinel is written AFTER `git worktree add` succeeds and
|
||||
* REMOVED only after `git worktree remove` completes, so the cleanup
|
||||
* tail is transactional from the orchestrator's perspective. If the
|
||||
* process dies in between, the sentinel is left behind pointing at the
|
||||
* orphan worktree and branch — a future run, /gsd-resume-work, or
|
||||
* /gsd-progress can detect and complete the recovery.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
// The gsd-code-fixer agent's working instructions ARE the product — Claude
|
||||
// follows them at runtime. Structural assertions over the markdown source
|
||||
// test the deployed contract. See bug-2686 for the same pattern.
|
||||
|
||||
const { describe, test, before } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const { parseFrontmatter } = require('./helpers.cjs');
|
||||
|
||||
const SENTINEL_NAME = '.review-fix-recovery-pending.json';
|
||||
|
||||
function extractStep(content, stepName) {
|
||||
const re = new RegExp(`<step\\s+name="${stepName}">([\\s\\S]*?)</step>`);
|
||||
const m = content.match(re);
|
||||
return m ? m[1] : null;
|
||||
}
|
||||
|
||||
describe('bug-2839: /gsd-code-review-fix cleanup is transactional', () => {
|
||||
let agentPath;
|
||||
let agentContent;
|
||||
let frontmatter;
|
||||
|
||||
before(() => {
|
||||
agentPath = path.join(__dirname, '..', 'agents', 'gsd-code-fixer.md');
|
||||
assert.ok(fs.existsSync(agentPath), 'agents/gsd-code-fixer.md must exist');
|
||||
agentContent = fs.readFileSync(agentPath, 'utf-8');
|
||||
frontmatter = parseFrontmatter(agentContent);
|
||||
assert.ok(frontmatter, 'agent must have YAML frontmatter');
|
||||
});
|
||||
|
||||
test('agent declares a recovery sentinel filename', () => {
|
||||
assert.ok(
|
||||
agentContent.includes(SENTINEL_NAME),
|
||||
`gsd-code-fixer.md must reference the recovery sentinel ${SENTINEL_NAME} so an interrupted cleanup tail is discoverable (#2839)`
|
||||
);
|
||||
});
|
||||
|
||||
test('sentinel is written inside setup_worktree, after git worktree add', () => {
|
||||
const setupStep = extractStep(agentContent, 'setup_worktree');
|
||||
assert.ok(setupStep, 'setup_worktree step must exist');
|
||||
|
||||
assert.ok(
|
||||
setupStep.includes(SENTINEL_NAME),
|
||||
`setup_worktree must reference ${SENTINEL_NAME} so the sentinel is created at the start of the run (#2839)`
|
||||
);
|
||||
|
||||
const addPos = setupStep.indexOf('git worktree add');
|
||||
assert.ok(addPos !== -1, 'setup_worktree must contain `git worktree add`');
|
||||
|
||||
// The sentinel WRITE (not just a reference) must come after `git worktree add`.
|
||||
// Earlier references are allowed (e.g. recovery check for a stale sentinel
|
||||
// from a prior interrupted run). Look for an explicit write — either a
|
||||
// shell `>`/`>>` redirection, a `node -e` invocation that uses
|
||||
// `fs.writeFileSync(...sentinel...)`, or a `Write` tool reference.
|
||||
const writeIdx = (() => {
|
||||
const candidates = [
|
||||
/fs\.writeFileSync\([^)]*sentinel/,
|
||||
/>\s*"?\$sentinel/,
|
||||
/>\s*"?\$\{sentinel\}/,
|
||||
/Write the recovery sentinel/i,
|
||||
];
|
||||
let earliest = -1;
|
||||
for (const re of candidates) {
|
||||
const m = re.exec(setupStep);
|
||||
if (m && (earliest === -1 || m.index < earliest)) earliest = m.index;
|
||||
}
|
||||
return earliest;
|
||||
})();
|
||||
assert.ok(
|
||||
writeIdx !== -1,
|
||||
'setup_worktree must explicitly describe writing the sentinel (#2839)'
|
||||
);
|
||||
assert.ok(
|
||||
addPos < writeIdx,
|
||||
'sentinel must be written AFTER `git worktree add` succeeds (#2839)'
|
||||
);
|
||||
});
|
||||
|
||||
test('sentinel records worktree path, branch, and padded_phase as JSON fields', () => {
|
||||
for (const key of ['worktree_path', 'branch', 'padded_phase']) {
|
||||
assert.ok(
|
||||
agentContent.includes(key),
|
||||
`recovery sentinel must record \`${key}\` so a future /gsd-resume-work or /gsd-progress can locate the orphan state (#2839)`
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
test('sentinel removal happens only AFTER git worktree remove succeeds', () => {
|
||||
const setupStep = extractStep(agentContent, 'setup_worktree');
|
||||
assert.ok(setupStep, 'setup_worktree step must exist');
|
||||
|
||||
const cleanupAnchor = setupStep.lastIndexOf('Cleanup tail (transactional');
|
||||
assert.ok(cleanupAnchor !== -1, 'setup_worktree must document cleanup-tail section');
|
||||
const cleanupSection = setupStep.slice(cleanupAnchor);
|
||||
|
||||
const removeIdx = cleanupSection.indexOf('git worktree remove "$wt" --force');
|
||||
assert.ok(removeIdx !== -1, 'cleanup-tail must remove worktree');
|
||||
|
||||
// Within the cleanup-tail section, accept either a literal-filename form
|
||||
// (`rm -f .../.review-fix-recovery-pending.json`) or a shell-variable form
|
||||
// referring to the previously-declared `sentinel` variable
|
||||
// (`rm -f "$sentinel"` / `rm -f "${sentinel}"`).
|
||||
const escapedName = SENTINEL_NAME.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
const sentinelRemovalRe = new RegExp(
|
||||
`(rm\\s+(?:-f\\s+)?[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?)|unlink[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?))`
|
||||
);
|
||||
const sentinelRemovalMatch = sentinelRemovalRe.exec(cleanupSection);
|
||||
assert.ok(
|
||||
sentinelRemovalMatch,
|
||||
`agent must remove the sentinel file (rm or unlink ${SENTINEL_NAME}) as part of the cleanup tail (#2839)`
|
||||
);
|
||||
const sentinelRemovalIdx = sentinelRemovalMatch.index;
|
||||
|
||||
assert.ok(
|
||||
removeIdx < sentinelRemovalIdx,
|
||||
'cleanup ordering must be: `git worktree remove` BEFORE sentinel removal (#2839)'
|
||||
);
|
||||
});
|
||||
|
||||
test('agent documents detection of pre-existing sentinel from a prior interrupted run', () => {
|
||||
const lower = agentContent.toLowerCase();
|
||||
const mentionsRecovery =
|
||||
lower.includes('stale sentinel') ||
|
||||
lower.includes('existing sentinel') ||
|
||||
lower.includes('previous sentinel') ||
|
||||
lower.includes('prior run') ||
|
||||
lower.includes('pre-existing sentinel') ||
|
||||
lower.includes('recovery');
|
||||
assert.ok(
|
||||
mentionsRecovery,
|
||||
'agent must describe how it handles a pre-existing sentinel from a previous interrupted run (#2839)'
|
||||
);
|
||||
});
|
||||
|
||||
test('cleanup-tail obligation is documented as transactional / atomic', () => {
|
||||
const lower = agentContent.toLowerCase();
|
||||
const mentionsTransactional =
|
||||
lower.includes('transactional') ||
|
||||
lower.includes('atomic cleanup') ||
|
||||
lower.includes('cleanup tail');
|
||||
assert.ok(
|
||||
mentionsTransactional,
|
||||
'agent must document the cleanup tail as transactional/atomic (#2839)'
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -1,215 +0,0 @@
|
||||
/**
|
||||
* Bug #2866: Codex Installer (RC.7) fails to strip legacy flat hooks if
|
||||
* trailing newline is missing.
|
||||
*
|
||||
* The cleanup regexes in `bin/install.js` matched stale GSD hook blocks
|
||||
* via `\r?\n` at the end. When a stale block sat at end-of-file without
|
||||
* a trailing newline (very common — many editors strip them, and the
|
||||
* legacy installer never wrote one), no shape stripped, the installer
|
||||
* saw `gsd-check-update` already present, skipped writing the new
|
||||
* Nested-AoT block, and Codex 0.125+ refused to load with
|
||||
* "invalid type: map, expected a sequence in `hooks`"
|
||||
*
|
||||
* Fix: every shape's terminator is now `(?:\r?\n|$)` so end-of-file
|
||||
* counts as a valid terminator. The strip logic was lifted into a pure
|
||||
* helper, `stripStaleGsdHookBlocks(configContent)`, exported from
|
||||
* `bin/install.js` for direct test coverage.
|
||||
*
|
||||
* This test parses `package.json` to require `bin/install.js`
|
||||
* structurally (not by hardcoded path), then drives each historical
|
||||
* shape through the helper twice — once with a trailing newline, once
|
||||
* without — and asserts both are stripped.
|
||||
*/
|
||||
'use strict';
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const path = require('node:path');
|
||||
const fs = require('node:fs');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
|
||||
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
|
||||
const { stripStaleGsdHookBlocks } = require(installPath);
|
||||
|
||||
/**
|
||||
* Parse the TOML output line-structurally so assertions check shape, not
|
||||
* substring presence in raw text. Comments are dropped, table headers are
|
||||
* recorded, and string-valued keys are captured. Sufficient for the small,
|
||||
* well-formed TOML produced by these tests.
|
||||
*/
|
||||
function parseTomlShape(text) {
|
||||
const tableHeaders = [];
|
||||
const keys = new Map(); // dotted path → string value (last-write-wins, fine for these inputs)
|
||||
let currentTable = '';
|
||||
for (const rawLine of text.split('\n')) {
|
||||
const line = rawLine.replace(/(?:^|\s)#.*$/, '').trim();
|
||||
if (!line) continue;
|
||||
const tableMatch = line.match(/^\[(\[)?([^\]]+)\]?\]$/);
|
||||
if (tableMatch) {
|
||||
currentTable = tableMatch[2];
|
||||
tableHeaders.push((tableMatch[1] ? '[[' : '[') + currentTable + (tableMatch[1] ? ']]' : ']'));
|
||||
continue;
|
||||
}
|
||||
const kvMatch = line.match(/^([A-Za-z_][\w-]*)\s*=\s*(.*)$/);
|
||||
if (kvMatch) {
|
||||
const key = currentTable ? `${currentTable}.${kvMatch[1]}` : kvMatch[1];
|
||||
const value = kvMatch[2].replace(/^"(.*)"$/, '$1');
|
||||
keys.set(key, value);
|
||||
}
|
||||
}
|
||||
return { tableHeaders, keys };
|
||||
}
|
||||
|
||||
const SHAPES = {
|
||||
'Shape 1 (legacy gsd-update-check)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'event = "SessionStart"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-update-check.js"',
|
||||
].join('\n'),
|
||||
'Shape 2 (flat [[hooks]] + gsd-check-update)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'event = "SessionStart"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
'Shape 3 ([[hooks.SessionStart]] without nested .hooks)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
'Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'',
|
||||
'[[hooks.SessionStart.hooks]]',
|
||||
'type = "command"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
};
|
||||
|
||||
describe('bug-2866: stripStaleGsdHookBlocks handles end-of-file without trailing newline', () => {
|
||||
test('stripStaleGsdHookBlocks is exported from bin/install.js', () => {
|
||||
assert.strictEqual(typeof stripStaleGsdHookBlocks, 'function',
|
||||
'bin/install.js must export stripStaleGsdHookBlocks');
|
||||
});
|
||||
|
||||
function assertStripped(out, shape, scenario) {
|
||||
const shape_ = parseTomlShape(out);
|
||||
const hooksTable = shape_.tableHeaders.find((h) => /^\[\[?hooks(\.|]\])/.test(h));
|
||||
assert.strictEqual(hooksTable, undefined,
|
||||
`(${shape}, ${scenario}) no hooks table header may remain after strip, got tables: ${shape_.tableHeaders.join(', ')}`);
|
||||
const staleCmd = [...shape_.keys.entries()].find(([_, v]) =>
|
||||
/gsd-(update-check|check-update)/.test(v));
|
||||
assert.strictEqual(staleCmd, undefined,
|
||||
`(${shape}, ${scenario}) no key may carry a stale gsd-*-update command, got: ${staleCmd && staleCmd.join('=')}`);
|
||||
assert.strictEqual(shape_.keys.get('history.persistence'), 'save-all',
|
||||
`(${shape}, ${scenario}) history.persistence must be preserved as "save-all"`);
|
||||
}
|
||||
|
||||
for (const [shape, block] of Object.entries(SHAPES)) {
|
||||
test(`${shape}: stripped when terminated by trailing newline`, () => {
|
||||
const input = `[history]\npersistence = "save-all"\n${block}\n`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), shape, 'with trailing newline');
|
||||
});
|
||||
|
||||
test(`${shape}: stripped when at end-of-file without trailing newline`, () => {
|
||||
// The reporter's repro: stale block sits at the very end with no \n.
|
||||
const input = `[history]\npersistence = "save-all"\n${block}`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), shape, 'no trailing newline');
|
||||
});
|
||||
}
|
||||
|
||||
test('returns input unchanged when no GSD hook block is present', () => {
|
||||
const benign = '[history]\npersistence = "save-all"\n';
|
||||
const out = stripStaleGsdHookBlocks(benign);
|
||||
assert.strictEqual(out, benign, 'helper must be a no-op when no GSD reference exists');
|
||||
const benignShape = parseTomlShape(out);
|
||||
assert.strictEqual(benignShape.keys.get('history.persistence'), 'save-all',
|
||||
'parsed shape must preserve history.persistence');
|
||||
assert.deepStrictEqual(benignShape.tableHeaders, ['[history]'],
|
||||
'parsed shape must contain only the [history] table');
|
||||
});
|
||||
|
||||
// The structural rewrite (TOML-AST-driven, not regex-driven) must handle
|
||||
// whitespace and key-ordering variations that the previous regex missed.
|
||||
// These cases were silently leaked by the old implementation; one
|
||||
// (V3) actually corrupted the file by leaving an orphaned key=value line
|
||||
// outside any table.
|
||||
const VARIATIONS = {
|
||||
'extra blank line in Shape 4': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'',
|
||||
'',
|
||||
'[[hooks.SessionStart.hooks]]',
|
||||
'type = "command"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
'keys reordered (command before event in Shape 2)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
'event = "SessionStart"',
|
||||
].join('\n'),
|
||||
'extra key alongside command (Shape 3 + timeout)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
'timeout = 5000',
|
||||
].join('\n'),
|
||||
'tight whitespace (no spaces around `=`)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'event="SessionStart"',
|
||||
'command="node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
};
|
||||
|
||||
for (const [variation, block] of Object.entries(VARIATIONS)) {
|
||||
test(`variation stripped: ${variation}`, () => {
|
||||
const input = `[history]\npersistence = "save-all"\n${block}\n`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), variation, 'with trailing newline');
|
||||
});
|
||||
test(`variation stripped at EOF without trailing newline: ${variation}`, () => {
|
||||
const input = `[history]\npersistence = "save-all"\n${block}`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), variation, 'no trailing newline');
|
||||
});
|
||||
}
|
||||
|
||||
test('user-authored [[hooks.UserPromptSubmit]] is preserved', () => {
|
||||
// The structural strip must not touch hook tables that don't carry a
|
||||
// GSD-managed `gsd-(check-update|update-check).js` command.
|
||||
const input = [
|
||||
'[history]',
|
||||
'persistence = "save-all"',
|
||||
'[[hooks.UserPromptSubmit]]',
|
||||
'command = "node /Users/USER/my-hook.js"',
|
||||
'',
|
||||
].join('\n');
|
||||
const out = stripStaleGsdHookBlocks(input);
|
||||
const shape = parseTomlShape(out);
|
||||
assert.ok(
|
||||
shape.tableHeaders.includes('[[hooks.UserPromptSubmit]]'),
|
||||
`user-authored [[hooks.UserPromptSubmit]] must survive, got: ${shape.tableHeaders.join(', ')}`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
shape.keys.get('hooks.UserPromptSubmit.command'),
|
||||
'node /Users/USER/my-hook.js',
|
||||
'user-authored command value must be preserved verbatim',
|
||||
);
|
||||
});
|
||||
|
||||
test('Shape 4 strip does not leave an orphaned [[hooks.SessionStart]] header', () => {
|
||||
// Shape 4 is stripped before Shape 3 specifically to avoid this.
|
||||
const block = SHAPES['Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])'];
|
||||
const out = stripStaleGsdHookBlocks(`[history]\npersistence = "save-all"\n${block}`);
|
||||
const outShape = parseTomlShape(out);
|
||||
const orphan = outShape.tableHeaders.find((h) => /hooks\.SessionStart/.test(h));
|
||||
assert.strictEqual(orphan, undefined,
|
||||
`Shape 4 strip must remove the parent [[hooks.SessionStart]] header too, got tables: ${outShape.tableHeaders.join(', ')}`);
|
||||
});
|
||||
});
|
||||
@@ -1,191 +0,0 @@
|
||||
/**
|
||||
* Bug #2876: SKILL.md frontmatter parse failure when `description` begins
|
||||
* with a YAML flow indicator like `[BETA]`.
|
||||
*
|
||||
* description: [BETA] Offload plan phase to Claude Code's ultraplan…
|
||||
*
|
||||
* YAML 1.2 treats a leading `[` as the start of a flow sequence, so any
|
||||
* downstream parser (gh-copilot, JetBrains' kit, etc.) fails with
|
||||
* "Unexpected scalar at node end". The Copilot/Antigravity/Trae/Codebuddy
|
||||
* skill+agent converters in `bin/install.js` re-emit the description
|
||||
* unquoted; the Claude variant `yamlQuote(...)`s it. Bring the others
|
||||
* in line so any value is round-trip-safe regardless of leading char.
|
||||
*
|
||||
* The test is structural: it parses each emitted frontmatter into lines
|
||||
* and asserts the `description` value is a quoted YAML scalar (double or
|
||||
* single quoted) when the source description starts with a flow indicator.
|
||||
* It does not regex the bytes for substrings.
|
||||
*/
|
||||
'use strict';
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const path = require('node:path');
|
||||
const fs = require('node:fs');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
|
||||
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
|
||||
const install = require(installPath);
|
||||
|
||||
// Build a minimal Claude command source whose description starts with the
|
||||
// reporter's exact flow-indicator prefix. Apostrophe in the body forces
|
||||
// any naive single-quoting to also escape correctly — the canonical
|
||||
// safe form is `JSON.stringify(...)` (used by yamlQuote).
|
||||
const REPORTER_DESCRIPTION =
|
||||
"[BETA] Offload plan phase to Claude Code's ultraplan cloud — drafts remotely while terminal stays free, review in browser with inline comments, import back via /gsd-import. Claude Code only.";
|
||||
|
||||
// Use unquoted description in the source frontmatter — that's exactly the
|
||||
// shape that ships in commands/gsd/*.md when authors paste a description
|
||||
// without quoting it (see commands/gsd/ultraplan-phase.md). The bug is
|
||||
// triggered when the converter re-emits this same value to the destination
|
||||
// runtime without quoting. `extractFrontmatterField` strips a single outer
|
||||
// quote pair but does not unescape internal characters, so quoting the
|
||||
// fixture input would actually mask the bug.
|
||||
function buildClaudeCommand(description) {
|
||||
return [
|
||||
'---',
|
||||
'name: gsd:ultraplan-phase',
|
||||
`description: ${description}`,
|
||||
'argument-hint: "[phase-number]"',
|
||||
'allowed-tools:',
|
||||
' - Read',
|
||||
' - Bash',
|
||||
'---',
|
||||
'',
|
||||
'# body',
|
||||
'',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
function buildClaudeAgent(description) {
|
||||
return [
|
||||
'---',
|
||||
'name: gsd-extract-learnings',
|
||||
`description: ${description}`,
|
||||
'tools: Read, Bash',
|
||||
'---',
|
||||
'',
|
||||
'# body',
|
||||
'',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
function extractFrontmatter(content) {
|
||||
// Leading delimiter is `---\n`; closing is the next standalone `---`
|
||||
// on its own line. Tests parse line-structurally so the assertion
|
||||
// doesn't drift on whitespace/order changes (per project test-rigor).
|
||||
assert.ok(content.startsWith('---'), `output must begin with frontmatter, got: ${content.slice(0, 40)}`);
|
||||
const lines = content.split('\n');
|
||||
let openIdx = -1;
|
||||
let closeIdx = -1;
|
||||
for (let i = 0; i < lines.length; i += 1) {
|
||||
if (lines[i] === '---') {
|
||||
if (openIdx === -1) openIdx = i;
|
||||
else { closeIdx = i; break; }
|
||||
}
|
||||
}
|
||||
assert.ok(openIdx !== -1 && closeIdx !== -1, `output must have a closed frontmatter block, got:\n${content}`);
|
||||
return lines.slice(openIdx + 1, closeIdx);
|
||||
}
|
||||
|
||||
function findDescriptionLine(frontmatterLines) {
|
||||
for (const line of frontmatterLines) {
|
||||
if (line.startsWith('description:')) return line;
|
||||
}
|
||||
assert.fail(`no description line found in frontmatter:\n${frontmatterLines.join('\n')}`);
|
||||
return ''; // unreachable
|
||||
}
|
||||
|
||||
function isQuotedYamlScalar(valueText) {
|
||||
// YAML safe-quoted scalar: starts with `"` and ends with `"`, OR
|
||||
// starts with `'` and ends with `'`. This is what `yamlQuote()`
|
||||
// (JSON.stringify) and the Claude variant of these converters emit.
|
||||
const trimmed = valueText.trim();
|
||||
if (trimmed.startsWith('"') && trimmed.endsWith('"')) return true;
|
||||
if (trimmed.startsWith("'") && trimmed.endsWith("'")) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
function parseQuotedYamlValue(valueText) {
|
||||
const trimmed = valueText.trim();
|
||||
if (trimmed.startsWith('"')) return JSON.parse(trimmed);
|
||||
if (trimmed.startsWith("'")) return trimmed.slice(1, -1).replace(/''/g, "'");
|
||||
return trimmed;
|
||||
}
|
||||
|
||||
function assertDescriptionRoundTrips(emitted, expected, label) {
|
||||
const fmLines = extractFrontmatter(emitted);
|
||||
const descLine = findDescriptionLine(fmLines);
|
||||
const valueText = descLine.slice('description:'.length);
|
||||
assert.ok(
|
||||
isQuotedYamlScalar(valueText),
|
||||
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
parseQuotedYamlValue(valueText),
|
||||
expected,
|
||||
`(${label}) description must round-trip through YAML quoting unchanged.`,
|
||||
);
|
||||
}
|
||||
|
||||
const COMMAND_CONVERTERS = [
|
||||
{ label: 'convertClaudeCommandToCopilotSkill', fn: (src) => install.convertClaudeCommandToCopilotSkill(src, 'gsd-ultraplan-phase') },
|
||||
{ label: 'convertClaudeCommandToAntigravitySkill', fn: (src) => install.convertClaudeCommandToAntigravitySkill(src, 'gsd-ultraplan-phase') },
|
||||
{ label: 'convertClaudeCommandToTraeSkill', fn: (src) => install.convertClaudeCommandToTraeSkill(src, 'gsd-ultraplan-phase') },
|
||||
{ label: 'convertClaudeCommandToCodebuddySkill', fn: (src) => install.convertClaudeCommandToCodebuddySkill(src, 'gsd-ultraplan-phase') },
|
||||
];
|
||||
|
||||
const AGENT_CONVERTERS = [
|
||||
{ label: 'convertClaudeAgentToCopilotAgent', fn: (src) => install.convertClaudeAgentToCopilotAgent(src) },
|
||||
{ label: 'convertClaudeAgentToAntigravityAgent', fn: (src) => install.convertClaudeAgentToAntigravityAgent(src) },
|
||||
];
|
||||
|
||||
// A grab-bag of leading characters that all break unquoted YAML scalar
|
||||
// parsing per YAML 1.2 §7.3.3 / §6.9. The reporter's case is `[`; the
|
||||
// rest defend against neighbouring drift.
|
||||
const FLOW_HOSTILE_PREFIXES = ['[', '{', '*', '&', '!', '|', '>', '%', '@', '`'];
|
||||
|
||||
// Some converters (Trae, CodeBuddy) deliberately rewrite "Claude Code"
|
||||
// in body content to their target runtime name, and the rewrite cuts
|
||||
// across the description too. That's correct behavior — out of scope for
|
||||
// the YAML-quoting fix — so for the reporter case we assert only the
|
||||
// quoting requirement, not byte-equality of the round-tripped value.
|
||||
function assertDescriptionIsQuoted(emitted, label) {
|
||||
const fmLines = extractFrontmatter(emitted);
|
||||
const descLine = findDescriptionLine(fmLines);
|
||||
const valueText = descLine.slice('description:'.length);
|
||||
assert.ok(
|
||||
isQuotedYamlScalar(valueText),
|
||||
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
|
||||
);
|
||||
}
|
||||
|
||||
describe('bug-2876: skill+agent converters emit YAML-quoted description', () => {
|
||||
for (const { label, fn } of COMMAND_CONVERTERS) {
|
||||
test(`${label}: reporter's "[BETA] ..." description is quoted`, () => {
|
||||
const out = fn(buildClaudeCommand(REPORTER_DESCRIPTION));
|
||||
assertDescriptionIsQuoted(out, label);
|
||||
});
|
||||
for (const prefix of FLOW_HOSTILE_PREFIXES) {
|
||||
test(`${label}: leading ${JSON.stringify(prefix)} is quoted`, () => {
|
||||
// Avoid leading/trailing `'` or `"` in the payload — `extractFrontmatterField`
|
||||
// strips a single outer quote char of either kind regardless of whether
|
||||
// the value was actually quoted, which would obscure the round-trip
|
||||
// assertion. Pre-existing behavior, out of scope for #2876.
|
||||
const desc = `${prefix} edge-case payload — flow indicator at start`;
|
||||
const out = fn(buildClaudeCommand(desc));
|
||||
assertDescriptionRoundTrips(out, desc, `${label} prefix=${prefix}`);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
for (const { label, fn } of AGENT_CONVERTERS) {
|
||||
test(`${label}: reporter-shape "[BETA] ..." description is quoted`, () => {
|
||||
const out = fn(buildClaudeAgent(REPORTER_DESCRIPTION));
|
||||
assertDescriptionIsQuoted(out, label);
|
||||
});
|
||||
}
|
||||
});
|
||||
@@ -1,46 +0,0 @@
|
||||
'use strict';
|
||||
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
function walkMd(dir, out = []) {
|
||||
if (!fs.existsSync(dir)) return out;
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) walkMd(full, out);
|
||||
else if (entry.name.endsWith('.md')) out.push(full);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function extractSlashCommandTokens(markdown) {
|
||||
const tokenRe = /\/gsd-[a-z0-9-]+/gi;
|
||||
const tokens = new Set();
|
||||
let m;
|
||||
while ((m = tokenRe.exec(markdown)) !== null) {
|
||||
tokens.add(m[0]);
|
||||
}
|
||||
return tokens;
|
||||
}
|
||||
|
||||
describe('bug #3054: user-facing docs should not reference removed /gsd-next command', () => {
|
||||
test('docs, workflows, and README surfaces use /gsd-progress --next instead', () => {
|
||||
const root = path.join(__dirname, '..');
|
||||
const files = [
|
||||
...walkMd(path.join(root, 'docs')),
|
||||
...walkMd(path.join(root, 'gsd-core', 'workflows')),
|
||||
...fs.readdirSync(root).filter((f) => /^README.*\.md$/.test(f)).map((f) => path.join(root, f)),
|
||||
];
|
||||
|
||||
const offenders = [];
|
||||
for (const file of files) {
|
||||
const content = fs.readFileSync(file, 'utf8');
|
||||
const tokens = extractSlashCommandTokens(content);
|
||||
if (tokens.has('/gsd-next')) offenders.push(path.relative(root, file));
|
||||
}
|
||||
|
||||
assert.deepStrictEqual(offenders, [], `stale /gsd-next references remain in: ${offenders.join(', ')}`);
|
||||
});
|
||||
});
|
||||
@@ -1,151 +0,0 @@
|
||||
'use strict';
|
||||
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
// The deployed settings.md IS the product — testing its text content tests the deployed contract.
|
||||
|
||||
/**
|
||||
* Regression test for issue #33
|
||||
*
|
||||
* model_profile UI shows 4 options, schema has 5 — `adaptive` missing from
|
||||
* `settings.md` AskUserQuestion.
|
||||
*
|
||||
* The schema (gsd-core/bin/shared/model-catalog.json `profiles` array) defines 5 valid
|
||||
* model_profile values: quality, balanced, budget, adaptive, inherit. The
|
||||
* settings.md AskUserQuestion block for model_profile originally listed only 4
|
||||
* options (Quality, Balanced, Budget, Inherit) — `adaptive` was missing.
|
||||
*
|
||||
* Fix: the model_profile selection uses a two-question split. Q1 routes between
|
||||
* Adaptive / Standard-tier / Inherit (3 options). Q2 (only when Q1 = Standard)
|
||||
* asks Quality / Balanced / Budget. This keeps every individual options array
|
||||
* within the AskUserQuestion 4-option cap while making all 5 profiles reachable.
|
||||
*
|
||||
* Fixes: #33
|
||||
*/
|
||||
|
||||
const { describe, test, before } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
const SETTINGS_PATH = path.join(REPO_ROOT, 'gsd-core', 'workflows', 'settings.md');
|
||||
const CATALOG_PATH = path.join(REPO_ROOT, 'gsd-core', 'bin', 'shared', 'model-catalog.json');
|
||||
|
||||
/**
|
||||
* Collect every label: "..." value within a text block, lowercased.
|
||||
*/
|
||||
function extractOptionLabels(block) {
|
||||
const re = /label:\s*"([^"]+)"/g;
|
||||
const labels = [];
|
||||
let m;
|
||||
while ((m = re.exec(block)) !== null) {
|
||||
labels.push(m[1].toLowerCase());
|
||||
}
|
||||
return labels;
|
||||
}
|
||||
|
||||
describe('issue #33: model_profile schema and settings.md UI are in sync', () => {
|
||||
let catalog;
|
||||
let settingsContent;
|
||||
let presentBlock;
|
||||
|
||||
before(() => {
|
||||
catalog = JSON.parse(fs.readFileSync(CATALOG_PATH, 'utf-8'));
|
||||
settingsContent = fs.readFileSync(SETTINGS_PATH, 'utf-8');
|
||||
const presentMatch = settingsContent.match(/<step name="present_settings">[\s\S]*?<\/step>/);
|
||||
assert.ok(presentMatch, 'settings.md must contain a present_settings step');
|
||||
presentBlock = presentMatch[0];
|
||||
});
|
||||
|
||||
// -- (a) Schema contract ---------------------------------------------------
|
||||
|
||||
test('schema includes the adaptive model_profile value', () => {
|
||||
assert.ok(
|
||||
Array.isArray(catalog.profiles),
|
||||
'model-catalog.json must have a "profiles" array'
|
||||
);
|
||||
assert.ok(
|
||||
catalog.profiles.includes('adaptive'),
|
||||
'model-catalog should include the adaptive profile. Got: [' + catalog.profiles.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('schema includes adaptive as a model_profile value', () => {
|
||||
assert.ok(
|
||||
catalog.profiles.includes('adaptive'),
|
||||
'"adaptive" must be in model-catalog.json profiles. Got: [' + catalog.profiles.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('schema includes all expected model_profile values', () => {
|
||||
const expected = ['quality', 'balanced', 'budget', 'adaptive', 'inherit'];
|
||||
for (const profile of expected) {
|
||||
assert.ok(
|
||||
catalog.profiles.includes(profile),
|
||||
'Schema must include "' + profile + '" in profiles. Got: [' + catalog.profiles.join(', ') + ']'
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
// -- (b) UI contract — all 5 profiles reachable via present_settings -------
|
||||
|
||||
test('present_settings includes Adaptive as a selectable option (#33)', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'adaptive' || l.startsWith('adaptive')),
|
||||
'Issue #33: present_settings must include an "Adaptive" label in its model_profile AskUserQuestion options so users can select it interactively. Got labels: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Quality as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'quality' || l.startsWith('quality')),
|
||||
'present_settings must include a "Quality" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Balanced as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'balanced' || l.startsWith('balanced')),
|
||||
'present_settings must include a "Balanced" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Budget as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'budget' || l.startsWith('budget')),
|
||||
'present_settings must include a "Budget" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Inherit as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'inherit' || l.startsWith('inherit')),
|
||||
'present_settings must include an "Inherit" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
// -- update_config and confirm steps reference adaptive --------------------
|
||||
|
||||
test('update_config step lists adaptive as a valid model_profile value', () => {
|
||||
const m = settingsContent.match(/<step name="update_config">[\s\S]*?<\/step>/);
|
||||
assert.ok(m, 'settings.md must have an update_config step');
|
||||
assert.ok(
|
||||
m[0].includes('adaptive'),
|
||||
'update_config step must list "adaptive" as a valid model_profile value'
|
||||
);
|
||||
});
|
||||
|
||||
test('confirm step table shows adaptive as a possible model profile value', () => {
|
||||
const m = settingsContent.match(/<step name="confirm">[\s\S]*?<\/step>/);
|
||||
assert.ok(m, 'settings.md must have a confirm step');
|
||||
assert.ok(
|
||||
m[0].includes('adaptive'),
|
||||
'confirm step must include "adaptive" in the Model Profile row'
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -1,76 +0,0 @@
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const repoRoot = path.resolve(__dirname, '..');
|
||||
const WORKTREE_BRANCH_CHECK_FRAGMENT = path.join(repoRoot, 'gsd-core', 'references', 'worktree-branch-check.md');
|
||||
|
||||
function read(relPath) {
|
||||
return fs.readFileSync(path.join(repoRoot, relPath), 'utf8');
|
||||
}
|
||||
|
||||
describe('bug #3384: adjacent worktree data-loss guards', () => {
|
||||
test('worktree cleanup CLI preserves caller cwd instead of resolving project root', () => {
|
||||
const source = read('gsd-core/bin/gsd-tools.cjs');
|
||||
const skipSet = source.slice(
|
||||
source.indexOf('const SKIP_ROOT_RESOLUTION = new Set(['),
|
||||
source.indexOf('if (!SKIP_ROOT_RESOLUTION.has(command))'),
|
||||
);
|
||||
|
||||
assert.match(skipSet, /'worktree'/);
|
||||
});
|
||||
|
||||
test('diagnose-issues references canonical fragment; fragment is verify-only and fails closed (#48)', () => {
|
||||
// diagnose-issues.md now references the canonical fragment rather than
|
||||
// inlining the block. Verify (a) it references the fragment and (b) the
|
||||
// fragment itself has the correct ordering: symbolic-ref/HEAD assertion and
|
||||
// ^worktree-agent- allow-list appear before any work, and (c) the fragment
|
||||
// is verify-only — no destructive self-recovery.
|
||||
const diagnoseSource = read('gsd-core/workflows/diagnose-issues.md');
|
||||
assert.ok(
|
||||
diagnoseSource.includes('worktree-branch-check.md'),
|
||||
'diagnose-issues.md must reference the canonical worktree-branch-check.md fragment'
|
||||
);
|
||||
|
||||
const fragmentSource = fs.readFileSync(WORKTREE_BRANCH_CHECK_FRAGMENT, 'utf8');
|
||||
const branchCheck = fragmentSource.indexOf('HEAD_REF=$(git symbolic-ref --quiet HEAD || echo');
|
||||
const namespaceCheck = fragmentSource.indexOf('^worktree-agent-');
|
||||
|
||||
assert.ok(branchCheck > 0, 'canonical fragment must assert HEAD before any work');
|
||||
assert.ok(namespaceCheck > branchCheck, 'canonical fragment must require disposable worktree-agent branch');
|
||||
// #48: verify-only — the destructive self-recovery is gone; the fragment fails closed instead.
|
||||
assert.ok(!fragmentSource.includes('git reset --hard {EXPECTED_BASE}'), 'canonical fragment must not self-recover via reset --hard — orchestrator owns recovery (#48)');
|
||||
assert.ok(fragmentSource.includes('exit 42'), 'canonical fragment must fail closed with exit 42 on base mismatch (#48)');
|
||||
});
|
||||
|
||||
test('remove-workspace fails closed when git worktree remove fails', () => {
|
||||
const source = read('gsd-core/workflows/remove-workspace.md');
|
||||
const init = source.indexOf('REMOVE_FAILED=false');
|
||||
const loop = source.indexOf('For each repo in the workspace');
|
||||
const remove = source.indexOf('git worktree remove "$WORKSPACE_PATH/$REPO_NAME"');
|
||||
|
||||
assert.doesNotMatch(
|
||||
source,
|
||||
/git worktree remove "\$WORKSPACE_PATH\/\$REPO_NAME" 2>&1 \|\| true/,
|
||||
'worktree removal failures must not be swallowed',
|
||||
);
|
||||
assert.ok(init > 0 && init < loop, 'REMOVE_FAILED must initialize once before the per-repo loop');
|
||||
assert.ok(remove > loop, 'worktree removal should remain inside the per-repo loop');
|
||||
assert.match(source, /Refusing to delete "\$WORKSPACE_PATH"/);
|
||||
});
|
||||
|
||||
test('validate health warns when worktree inventory cannot be listed', () => {
|
||||
const source = read('gsd-core/bin/lib/verify.cjs');
|
||||
// Accept both hand-written dot access and the tsc-compiled bracket form
|
||||
// (ADR-457: verify.cjs is now emitted from src/verify.cts):
|
||||
// hand-written: worktreeHealth.reason === 'git_list_failed'
|
||||
// tsc-compiled: worktreeHealth['reason'] === 'git_list_failed'
|
||||
const failureBranch = source.search(/worktreeHealth(?:\.reason|\['reason'\]) === 'git_list_failed'/);
|
||||
const warning = source.indexOf("addIssue('warning', 'W020'", failureBranch);
|
||||
|
||||
assert.ok(failureBranch > 0, 'verify health should branch on git_list_failed');
|
||||
assert.ok(warning > failureBranch, 'git_list_failed should emit W020 degraded-health warning');
|
||||
});
|
||||
});
|
||||
@@ -1,133 +0,0 @@
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
// Runtime prompt/hook files are deployed verbatim — their text IS what the
|
||||
// runtime loads and executes. Asserting that text carries no retired `gsd-sdk`
|
||||
// reference tests the deployed contract, which no behavioral seam can observe
|
||||
// (there is no runtime API that enumerates "did any shipped prompt name the
|
||||
// removed SDK binary").
|
||||
|
||||
/**
|
||||
* Regression guard: no `gsd-sdk` references in runtime-facing surfaces (#339).
|
||||
*
|
||||
* The `@opengsd/gsd-sdk` package and its `gsd-sdk` binary were retired (ADR 0174,
|
||||
* #191). The bulk runtime cleanup is already done — this test locks it in so a
|
||||
* `gsd-sdk` / `GSD_SDK` reference cannot creep back into a shipped prompt or hook
|
||||
* and re-introduce drift between the documented surface and the supported
|
||||
* `gsd-tools` binary.
|
||||
*
|
||||
* Scope: runtime surfaces only — the prompts and hooks the installer ships into
|
||||
* a user's runtime config dir. Explicitly NOT covered here:
|
||||
* - `bin/install.js` — installer code, not a runtime-deployed prompt/hook
|
||||
* surface. (It carries zero `gsd-sdk` references today; the SDK-shim
|
||||
* verification subsystem was removed in #515 and the shim retired in #522.)
|
||||
* - `gsd-core/bin/` — executable library code, not deployed prompt text; it
|
||||
* may legitimately reference the SDK retirement in comments.
|
||||
* - `tests/`, `docs/`, `.changeset/`, CI/lint scripts — legitimately reference
|
||||
* the SDK retirement as history or detect its stale artifacts.
|
||||
*
|
||||
* Complements `tests/gsd-tools-path-refs.test.cjs`, which only catches the
|
||||
* `gsd-sdk query` binary-invocation form; this catches ANY runtime reference.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
|
||||
// Runtime surfaces the installer ships. Each entry is { dir, exts } — dir is
|
||||
// repo-relative, exts is the set of file extensions whose text is deployed.
|
||||
const RUNTIME_SURFACES = [
|
||||
// .md prompts plus the non-.md runtime artifacts this dir also ships:
|
||||
// _runtime-launcher.snippet.sh (the canonical launcher synced into every hook
|
||||
// by scripts/sync-runtime-launcher.cjs) and discuss-phase/templates/*.json
|
||||
// (loaded at runtime by discuss-phase.md). Scanning only .md left these two
|
||||
// deployed files uncovered. (#691 review)
|
||||
{ dir: path.join('gsd-core', 'workflows'), exts: ['.md', '.sh', '.json'] },
|
||||
{ dir: path.join('gsd-core', 'references'), exts: ['.md'] },
|
||||
// Prompt surfaces the installer deep-copies and the runtime loads via
|
||||
// `@~/.claude/gsd-core/templates/*.md` anchors in workflows/commands; the
|
||||
// lone config.json under templates/ ships too. (#691 review)
|
||||
{ dir: path.join('gsd-core', 'templates'), exts: ['.md', '.json'] },
|
||||
{ dir: path.join('gsd-core', 'contexts'), exts: ['.md'] },
|
||||
{ dir: path.join('commands', 'gsd'), exts: ['.md'] },
|
||||
{ dir: 'agents', exts: ['.md'] },
|
||||
// Hooks ship as executable text (.js/.cjs/.sh). `hooks/dist/` is a gitignored
|
||||
// build artifact regenerated from these sources, so scanning the sources is
|
||||
// sufficient and avoids asserting against generated copies.
|
||||
{ dir: 'hooks', exts: ['.js', '.cjs', '.sh'], skipDirs: ['dist'] },
|
||||
];
|
||||
|
||||
// Matches every casing/separator variant of the retired SDK token:
|
||||
// gsd-sdk, gsd_sdk, GSD-SDK, GSD_SDK, etc.
|
||||
const SDK_REF = /gsd[-_]sdk/i;
|
||||
|
||||
/**
|
||||
* Recursively collect files under `absDir` whose extension is in `exts`,
|
||||
* skipping any directory name listed in `skipDirs`.
|
||||
*/
|
||||
function collectFiles(absDir, exts, skipDirs) {
|
||||
if (!fs.existsSync(absDir)) return [];
|
||||
const out = [];
|
||||
for (const entry of fs.readdirSync(absDir, { withFileTypes: true })) {
|
||||
if (entry.isDirectory()) {
|
||||
if (skipDirs.includes(entry.name)) continue;
|
||||
out.push(...collectFiles(path.join(absDir, entry.name), exts, skipDirs));
|
||||
} else if (entry.isFile() && exts.includes(path.extname(entry.name))) {
|
||||
out.push(path.join(absDir, entry.name));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function rel(file) {
|
||||
return path.relative(REPO_ROOT, file).split(path.sep).join('/');
|
||||
}
|
||||
|
||||
describe('#339 no gsd-sdk references in runtime surfaces', () => {
|
||||
test('shipped prompts and hooks carry no retired gsd-sdk reference', () => {
|
||||
const violations = [];
|
||||
|
||||
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
|
||||
const files = collectFiles(path.join(REPO_ROOT, dir), exts, skipDirs);
|
||||
for (const file of files) {
|
||||
const lines = fs.readFileSync(file, 'utf-8').split(/\r?\n/);
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
if (SDK_REF.test(lines[i])) {
|
||||
violations.push(`${rel(file)}:${i + 1}: ${lines[i].trim()}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
assert.strictEqual(
|
||||
violations.length,
|
||||
0,
|
||||
'Runtime surfaces must not reference the retired gsd-sdk binary/package — ' +
|
||||
'use gsd-tools instead.\nViolations:\n' + violations.join('\n')
|
||||
);
|
||||
});
|
||||
|
||||
test('at least one file per configured extension is scanned (guards against an empty sweep)', () => {
|
||||
// A path typo, directory rename, or stale extension could silently make
|
||||
// collectFiles() return [] for part of a surface, turning the guard above
|
||||
// into a no-op that always passes. Checking per-surface isn't enough: for
|
||||
// gsd-core/workflows the .md files alone keep a per-surface count > 0, so
|
||||
// dropping .sh/.json would stop covering _runtime-launcher.snippet.sh and
|
||||
// discuss-phase/templates/*.json while the test stayed green. Assert each
|
||||
// configured extension actually resolves to scanned files. (#691 review)
|
||||
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
|
||||
for (const ext of exts) {
|
||||
const count = collectFiles(path.join(REPO_ROOT, dir), [ext], skipDirs).length;
|
||||
assert.ok(
|
||||
count > 0,
|
||||
`Runtime surface "${dir}" resolved to 0 "${ext}" files — the path may ` +
|
||||
'have moved or the extension is stale; update RUNTIME_SURFACES so the ' +
|
||||
'guard keeps covering it.'
|
||||
);
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -1,115 +0,0 @@
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
// Workflow .md / agent .md / command .md / reference .md files — their text
|
||||
// IS what the runtime loads. Testing text content tests the deployed contract.
|
||||
// Per CONTRIBUTING.md exception matrix.
|
||||
|
||||
/**
|
||||
* Common Bug Patterns Reference Tests
|
||||
*
|
||||
* Structural tests for the common-bug-patterns.md reference file:
|
||||
* - File exists at expected path
|
||||
* - Contains expected bug pattern categories (at least 5 of 10)
|
||||
* - Debugger agent references the file in required_reading
|
||||
*/
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const REFERENCE_PATH = path.join(
|
||||
__dirname, '..', 'gsd-core', 'references', 'common-bug-patterns.md'
|
||||
);
|
||||
const DEBUGGER_AGENT_PATH = path.join(
|
||||
__dirname, '..', 'agents', 'gsd-debugger.md'
|
||||
);
|
||||
|
||||
const EXPECTED_CATEGORIES = [
|
||||
'Off-by-One',
|
||||
'Null',
|
||||
'Async',
|
||||
'State Management',
|
||||
'Import',
|
||||
'Environment',
|
||||
'Data Shape',
|
||||
'String Handling',
|
||||
'File System',
|
||||
'Error Handling',
|
||||
];
|
||||
|
||||
describe('common-bug-patterns.md reference', () => {
|
||||
test('reference file exists', () => {
|
||||
assert.ok(
|
||||
fs.existsSync(REFERENCE_PATH),
|
||||
`Expected reference file at ${REFERENCE_PATH}`
|
||||
);
|
||||
});
|
||||
|
||||
test('has title and intro', () => {
|
||||
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
|
||||
assert.ok(
|
||||
content.startsWith('# Common Bug Patterns'),
|
||||
'File should start with "# Common Bug Patterns" title'
|
||||
);
|
||||
assert.ok(
|
||||
content.includes('---'),
|
||||
'File should contain --- separator after intro'
|
||||
);
|
||||
});
|
||||
|
||||
test('contains at least 5 of 10 expected categories', () => {
|
||||
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
|
||||
const found = EXPECTED_CATEGORIES.filter(cat =>
|
||||
content.toLowerCase().includes(cat.toLowerCase())
|
||||
);
|
||||
assert.ok(
|
||||
found.length >= 5,
|
||||
`Expected at least 5 categories, found ${found.length}: ${found.join(', ')}`
|
||||
);
|
||||
});
|
||||
|
||||
test('each pattern category has at least one bold bullet item', () => {
|
||||
const content = fs.readFileSync(REFERENCE_PATH, 'utf-8');
|
||||
// Only check sections inside <patterns> block, not <usage>
|
||||
const patternsBlock = (content.split('<patterns>')[1] || '').split('</patterns>')[0];
|
||||
const sections = patternsBlock.split(/^## /m).slice(1);
|
||||
assert.ok(sections.length >= 5, `Expected at least 5 pattern sections, got ${sections.length}`);
|
||||
for (const section of sections) {
|
||||
const title = section.split('\n')[0].trim();
|
||||
const bullets = section.match(/^- \*\*/gm);
|
||||
assert.ok(
|
||||
bullets && bullets.length >= 1,
|
||||
`Pattern section "${title}" should have at least one "- **" bullet item`
|
||||
);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('debugger agent references bug patterns', () => {
|
||||
test('gsd-debugger.md exists', () => {
|
||||
assert.ok(
|
||||
fs.existsSync(DEBUGGER_AGENT_PATH),
|
||||
`Expected debugger agent at ${DEBUGGER_AGENT_PATH}`
|
||||
);
|
||||
});
|
||||
|
||||
test('gsd-debugger.md references common-bug-patterns.md', () => {
|
||||
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('common-bug-patterns.md'),
|
||||
'Debugger agent should reference common-bug-patterns.md'
|
||||
);
|
||||
});
|
||||
|
||||
test('reference is inside <required_reading> block', () => {
|
||||
const content = fs.readFileSync(DEBUGGER_AGENT_PATH, 'utf-8');
|
||||
const reqReadMatch = content.match(
|
||||
/<required_reading>([\s\S]*?)<\/required_reading>/
|
||||
);
|
||||
assert.ok(reqReadMatch, 'Debugger agent should have a <required_reading> block');
|
||||
assert.ok(
|
||||
reqReadMatch[1].includes('common-bug-patterns.md'),
|
||||
'common-bug-patterns.md should be inside <required_reading> block'
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -572,3 +572,184 @@ describe('CR-INTEGRATION: workflow integration points', () => {
|
||||
`autonomous.md gsd-code-review-fix args missing --auto flag; got args="${fixInvocation.args}"`);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-2839-review-fix-transactional-cleanup.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-2839-review-fix-transactional-cleanup (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Regression test for bug #2839
|
||||
*
|
||||
* /gsd-code-review-fix cleanup tail is non-transactional. If the agent is
|
||||
* interrupted (system restart, OOM kill) AFTER the last fix commit but
|
||||
* BEFORE `git worktree remove`, the worktree is orphaned in
|
||||
* `git worktree list`, the agent's branch is left with unmerged commits,
|
||||
* and STATE.md is never advanced. To anyone reading main only, the phase
|
||||
* looks "ready to plan" while critical fixes sit on a dangling branch.
|
||||
*
|
||||
* Fix: introduce a recovery sentinel JSON at
|
||||
* ${PHASE_DIR}/.review-fix-recovery-pending.json
|
||||
* The sentinel is written AFTER `git worktree add` succeeds and
|
||||
* REMOVED only after `git worktree remove` completes, so the cleanup
|
||||
* tail is transactional from the orchestrator's perspective. If the
|
||||
* process dies in between, the sentinel is left behind pointing at the
|
||||
* orphan worktree and branch — a future run, /gsd-resume-work, or
|
||||
* /gsd-progress can detect and complete the recovery.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
// allow-test-rule: source-text-is-the-product (see #2839)
|
||||
// The gsd-code-fixer agent's working instructions ARE the product — Claude
|
||||
// follows them at runtime. Structural assertions over the markdown source
|
||||
// test the deployed contract. See bug-2686 for the same pattern.
|
||||
|
||||
const { describe, test, before } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const { parseFrontmatter } = require('./helpers.cjs');
|
||||
|
||||
const SENTINEL_NAME = '.review-fix-recovery-pending.json';
|
||||
|
||||
function extractStep(content, stepName) {
|
||||
const re = new RegExp(`<step\\s+name="${stepName}">([\\s\\S]*?)</step>`);
|
||||
const m = content.match(re);
|
||||
return m ? m[1] : null;
|
||||
}
|
||||
|
||||
describe('bug-2839: /gsd-code-review-fix cleanup is transactional', () => {
|
||||
let agentPath;
|
||||
let agentContent;
|
||||
let frontmatter;
|
||||
|
||||
before(() => {
|
||||
agentPath = path.join(__dirname, '..', 'agents', 'gsd-code-fixer.md');
|
||||
assert.ok(fs.existsSync(agentPath), 'agents/gsd-code-fixer.md must exist');
|
||||
agentContent = fs.readFileSync(agentPath, 'utf-8');
|
||||
frontmatter = parseFrontmatter(agentContent);
|
||||
assert.ok(frontmatter, 'agent must have YAML frontmatter');
|
||||
});
|
||||
|
||||
test('agent declares a recovery sentinel filename', () => {
|
||||
assert.ok(
|
||||
agentContent.includes(SENTINEL_NAME),
|
||||
`gsd-code-fixer.md must reference the recovery sentinel ${SENTINEL_NAME} so an interrupted cleanup tail is discoverable (#2839)`
|
||||
);
|
||||
});
|
||||
|
||||
test('sentinel is written inside setup_worktree, after git worktree add', () => {
|
||||
const setupStep = extractStep(agentContent, 'setup_worktree');
|
||||
assert.ok(setupStep, 'setup_worktree step must exist');
|
||||
|
||||
assert.ok(
|
||||
setupStep.includes(SENTINEL_NAME),
|
||||
`setup_worktree must reference ${SENTINEL_NAME} so the sentinel is created at the start of the run (#2839)`
|
||||
);
|
||||
|
||||
const addPos = setupStep.indexOf('git worktree add');
|
||||
assert.ok(addPos !== -1, 'setup_worktree must contain `git worktree add`');
|
||||
|
||||
// The sentinel WRITE (not just a reference) must come after `git worktree add`.
|
||||
// Earlier references are allowed (e.g. recovery check for a stale sentinel
|
||||
// from a prior interrupted run). Look for an explicit write — either a
|
||||
// shell `>`/`>>` redirection, a `node -e` invocation that uses
|
||||
// `fs.writeFileSync(...sentinel...)`, or a `Write` tool reference.
|
||||
const writeIdx = (() => {
|
||||
const candidates = [
|
||||
/fs\.writeFileSync\([^)]*sentinel/,
|
||||
/>\s*"?\$sentinel/,
|
||||
/>\s*"?\$\{sentinel\}/,
|
||||
/Write the recovery sentinel/i,
|
||||
];
|
||||
let earliest = -1;
|
||||
for (const re of candidates) {
|
||||
const m = re.exec(setupStep);
|
||||
if (m && (earliest === -1 || m.index < earliest)) earliest = m.index;
|
||||
}
|
||||
return earliest;
|
||||
})();
|
||||
assert.ok(
|
||||
writeIdx !== -1,
|
||||
'setup_worktree must explicitly describe writing the sentinel (#2839)'
|
||||
);
|
||||
assert.ok(
|
||||
addPos < writeIdx,
|
||||
'sentinel must be written AFTER `git worktree add` succeeds (#2839)'
|
||||
);
|
||||
});
|
||||
|
||||
test('sentinel records worktree path, branch, and padded_phase as JSON fields', () => {
|
||||
for (const key of ['worktree_path', 'branch', 'padded_phase']) {
|
||||
assert.ok(
|
||||
agentContent.includes(key),
|
||||
`recovery sentinel must record \`${key}\` so a future /gsd-resume-work or /gsd-progress can locate the orphan state (#2839)`
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
test('sentinel removal happens only AFTER git worktree remove succeeds', () => {
|
||||
const setupStep = extractStep(agentContent, 'setup_worktree');
|
||||
assert.ok(setupStep, 'setup_worktree step must exist');
|
||||
|
||||
const cleanupAnchor = setupStep.lastIndexOf('Cleanup tail (transactional');
|
||||
assert.ok(cleanupAnchor !== -1, 'setup_worktree must document cleanup-tail section');
|
||||
const cleanupSection = setupStep.slice(cleanupAnchor);
|
||||
|
||||
const removeIdx = cleanupSection.indexOf('git worktree remove "$wt" --force');
|
||||
assert.ok(removeIdx !== -1, 'cleanup-tail must remove worktree');
|
||||
|
||||
// Within the cleanup-tail section, accept either a literal-filename form
|
||||
// (`rm -f .../.review-fix-recovery-pending.json`) or a shell-variable form
|
||||
// referring to the previously-declared `sentinel` variable
|
||||
// (`rm -f "$sentinel"` / `rm -f "${sentinel}"`).
|
||||
const escapedName = SENTINEL_NAME.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
const sentinelRemovalRe = new RegExp(
|
||||
`(rm\\s+(?:-f\\s+)?[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?)|unlink[^\\n]*(?:${escapedName}|\\$\\{?sentinel\\}?))`
|
||||
);
|
||||
const sentinelRemovalMatch = sentinelRemovalRe.exec(cleanupSection);
|
||||
assert.ok(
|
||||
sentinelRemovalMatch,
|
||||
`agent must remove the sentinel file (rm or unlink ${SENTINEL_NAME}) as part of the cleanup tail (#2839)`
|
||||
);
|
||||
const sentinelRemovalIdx = sentinelRemovalMatch.index;
|
||||
|
||||
assert.ok(
|
||||
removeIdx < sentinelRemovalIdx,
|
||||
'cleanup ordering must be: `git worktree remove` BEFORE sentinel removal (#2839)'
|
||||
);
|
||||
});
|
||||
|
||||
test('agent documents detection of pre-existing sentinel from a prior interrupted run', () => {
|
||||
const lower = agentContent.toLowerCase();
|
||||
const mentionsRecovery =
|
||||
lower.includes('stale sentinel') ||
|
||||
lower.includes('existing sentinel') ||
|
||||
lower.includes('previous sentinel') ||
|
||||
lower.includes('prior run') ||
|
||||
lower.includes('pre-existing sentinel') ||
|
||||
lower.includes('recovery');
|
||||
assert.ok(
|
||||
mentionsRecovery,
|
||||
'agent must describe how it handles a pre-existing sentinel from a previous interrupted run (#2839)'
|
||||
);
|
||||
});
|
||||
|
||||
test('cleanup-tail obligation is documented as transactional / atomic', () => {
|
||||
const lower = agentContent.toLowerCase();
|
||||
const mentionsTransactional =
|
||||
lower.includes('transactional') ||
|
||||
lower.includes('atomic cleanup') ||
|
||||
lower.includes('cleanup tail');
|
||||
assert.ok(
|
||||
mentionsTransactional,
|
||||
'agent must document the cleanup tail as transactional/atomic (#2839)'
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -8252,3 +8252,228 @@ describe('enh-772: reconcileCodexHooksJsonEvent preserves user-owned entries', (
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-2866-codex-strip-no-trailing-newline.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-2866-codex-strip-no-trailing-newline (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Bug #2866: Codex Installer (RC.7) fails to strip legacy flat hooks if
|
||||
* trailing newline is missing.
|
||||
*
|
||||
* The cleanup regexes in `bin/install.js` matched stale GSD hook blocks
|
||||
* via `\r?\n` at the end. When a stale block sat at end-of-file without
|
||||
* a trailing newline (very common — many editors strip them, and the
|
||||
* legacy installer never wrote one), no shape stripped, the installer
|
||||
* saw `gsd-check-update` already present, skipped writing the new
|
||||
* Nested-AoT block, and Codex 0.125+ refused to load with
|
||||
* "invalid type: map, expected a sequence in `hooks`"
|
||||
*
|
||||
* Fix: every shape's terminator is now `(?:\r?\n|$)` so end-of-file
|
||||
* counts as a valid terminator. The strip logic was lifted into a pure
|
||||
* helper, `stripStaleGsdHookBlocks(configContent)`, exported from
|
||||
* `bin/install.js` for direct test coverage.
|
||||
*
|
||||
* This test parses `package.json` to require `bin/install.js`
|
||||
* structurally (not by hardcoded path), then drives each historical
|
||||
* shape through the helper twice — once with a trailing newline, once
|
||||
* without — and asserts both are stripped.
|
||||
*/
|
||||
'use strict';
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const path = require('node:path');
|
||||
const fs = require('node:fs');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
|
||||
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
|
||||
const { stripStaleGsdHookBlocks } = require(installPath);
|
||||
|
||||
/**
|
||||
* Parse the TOML output line-structurally so assertions check shape, not
|
||||
* substring presence in raw text. Comments are dropped, table headers are
|
||||
* recorded, and string-valued keys are captured. Sufficient for the small,
|
||||
* well-formed TOML produced by these tests.
|
||||
*/
|
||||
function parseTomlShape(text) {
|
||||
const tableHeaders = [];
|
||||
const keys = new Map(); // dotted path → string value (last-write-wins, fine for these inputs)
|
||||
let currentTable = '';
|
||||
for (const rawLine of text.split('\n')) {
|
||||
const line = rawLine.replace(/(?:^|\s)#.*$/, '').trim();
|
||||
if (!line) continue;
|
||||
const tableMatch = line.match(/^\[(\[)?([^\]]+)\]?\]$/);
|
||||
if (tableMatch) {
|
||||
currentTable = tableMatch[2];
|
||||
tableHeaders.push((tableMatch[1] ? '[[' : '[') + currentTable + (tableMatch[1] ? ']]' : ']'));
|
||||
continue;
|
||||
}
|
||||
const kvMatch = line.match(/^([A-Za-z_][\w-]*)\s*=\s*(.*)$/);
|
||||
if (kvMatch) {
|
||||
const key = currentTable ? `${currentTable}.${kvMatch[1]}` : kvMatch[1];
|
||||
const value = kvMatch[2].replace(/^"(.*)"$/, '$1');
|
||||
keys.set(key, value);
|
||||
}
|
||||
}
|
||||
return { tableHeaders, keys };
|
||||
}
|
||||
|
||||
const SHAPES = {
|
||||
'Shape 1 (legacy gsd-update-check)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'event = "SessionStart"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-update-check.js"',
|
||||
].join('\n'),
|
||||
'Shape 2 (flat [[hooks]] + gsd-check-update)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'event = "SessionStart"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
'Shape 3 ([[hooks.SessionStart]] without nested .hooks)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
'Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'',
|
||||
'[[hooks.SessionStart.hooks]]',
|
||||
'type = "command"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
};
|
||||
|
||||
describe('bug-2866: stripStaleGsdHookBlocks handles end-of-file without trailing newline', () => {
|
||||
test('stripStaleGsdHookBlocks is exported from bin/install.js', () => {
|
||||
assert.strictEqual(typeof stripStaleGsdHookBlocks, 'function',
|
||||
'bin/install.js must export stripStaleGsdHookBlocks');
|
||||
});
|
||||
|
||||
function assertStripped(out, shape, scenario) {
|
||||
const shape_ = parseTomlShape(out);
|
||||
const hooksTable = shape_.tableHeaders.find((h) => /^\[\[?hooks(\.|]\])/.test(h));
|
||||
assert.strictEqual(hooksTable, undefined,
|
||||
`(${shape}, ${scenario}) no hooks table header may remain after strip, got tables: ${shape_.tableHeaders.join(', ')}`);
|
||||
const staleCmd = [...shape_.keys.entries()].find(([_, v]) =>
|
||||
/gsd-(update-check|check-update)/.test(v));
|
||||
assert.strictEqual(staleCmd, undefined,
|
||||
`(${shape}, ${scenario}) no key may carry a stale gsd-*-update command, got: ${staleCmd && staleCmd.join('=')}`);
|
||||
assert.strictEqual(shape_.keys.get('history.persistence'), 'save-all',
|
||||
`(${shape}, ${scenario}) history.persistence must be preserved as "save-all"`);
|
||||
}
|
||||
|
||||
for (const [shape, block] of Object.entries(SHAPES)) {
|
||||
test(`${shape}: stripped when terminated by trailing newline`, () => {
|
||||
const input = `[history]\npersistence = "save-all"\n${block}\n`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), shape, 'with trailing newline');
|
||||
});
|
||||
|
||||
test(`${shape}: stripped when at end-of-file without trailing newline`, () => {
|
||||
// The reporter's repro: stale block sits at the very end with no \n.
|
||||
const input = `[history]\npersistence = "save-all"\n${block}`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), shape, 'no trailing newline');
|
||||
});
|
||||
}
|
||||
|
||||
test('returns input unchanged when no GSD hook block is present', () => {
|
||||
const benign = '[history]\npersistence = "save-all"\n';
|
||||
const out = stripStaleGsdHookBlocks(benign);
|
||||
assert.strictEqual(out, benign, 'helper must be a no-op when no GSD reference exists');
|
||||
const benignShape = parseTomlShape(out);
|
||||
assert.strictEqual(benignShape.keys.get('history.persistence'), 'save-all',
|
||||
'parsed shape must preserve history.persistence');
|
||||
assert.deepStrictEqual(benignShape.tableHeaders, ['[history]'],
|
||||
'parsed shape must contain only the [history] table');
|
||||
});
|
||||
|
||||
// The structural rewrite (TOML-AST-driven, not regex-driven) must handle
|
||||
// whitespace and key-ordering variations that the previous regex missed.
|
||||
// These cases were silently leaked by the old implementation; one
|
||||
// (V3) actually corrupted the file by leaving an orphaned key=value line
|
||||
// outside any table.
|
||||
const VARIATIONS = {
|
||||
'extra blank line in Shape 4': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'',
|
||||
'',
|
||||
'[[hooks.SessionStart.hooks]]',
|
||||
'type = "command"',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
'keys reordered (command before event in Shape 2)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
'event = "SessionStart"',
|
||||
].join('\n'),
|
||||
'extra key alongside command (Shape 3 + timeout)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks.SessionStart]]',
|
||||
'command = "node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
'timeout = 5000',
|
||||
].join('\n'),
|
||||
'tight whitespace (no spaces around `=`)': [
|
||||
'# GSD Hooks',
|
||||
'[[hooks]]',
|
||||
'event="SessionStart"',
|
||||
'command="node /Users/USER/.codex/hooks/gsd-check-update.js"',
|
||||
].join('\n'),
|
||||
};
|
||||
|
||||
for (const [variation, block] of Object.entries(VARIATIONS)) {
|
||||
test(`variation stripped: ${variation}`, () => {
|
||||
const input = `[history]\npersistence = "save-all"\n${block}\n`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), variation, 'with trailing newline');
|
||||
});
|
||||
test(`variation stripped at EOF without trailing newline: ${variation}`, () => {
|
||||
const input = `[history]\npersistence = "save-all"\n${block}`;
|
||||
assertStripped(stripStaleGsdHookBlocks(input), variation, 'no trailing newline');
|
||||
});
|
||||
}
|
||||
|
||||
test('user-authored [[hooks.UserPromptSubmit]] is preserved', () => {
|
||||
// The structural strip must not touch hook tables that don't carry a
|
||||
// GSD-managed `gsd-(check-update|update-check).js` command.
|
||||
const input = [
|
||||
'[history]',
|
||||
'persistence = "save-all"',
|
||||
'[[hooks.UserPromptSubmit]]',
|
||||
'command = "node /Users/USER/my-hook.js"',
|
||||
'',
|
||||
].join('\n');
|
||||
const out = stripStaleGsdHookBlocks(input);
|
||||
const shape = parseTomlShape(out);
|
||||
assert.ok(
|
||||
shape.tableHeaders.includes('[[hooks.UserPromptSubmit]]'),
|
||||
`user-authored [[hooks.UserPromptSubmit]] must survive, got: ${shape.tableHeaders.join(', ')}`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
shape.keys.get('hooks.UserPromptSubmit.command'),
|
||||
'node /Users/USER/my-hook.js',
|
||||
'user-authored command value must be preserved verbatim',
|
||||
);
|
||||
});
|
||||
|
||||
test('Shape 4 strip does not leave an orphaned [[hooks.SessionStart]] header', () => {
|
||||
// Shape 4 is stripped before Shape 3 specifically to avoid this.
|
||||
const block = SHAPES['Shape 4 (nested [[hooks.SessionStart]] + [[hooks.SessionStart.hooks]])'];
|
||||
const out = stripStaleGsdHookBlocks(`[history]\npersistence = "save-all"\n${block}`);
|
||||
const outShape = parseTomlShape(out);
|
||||
const orphan = outShape.tableHeaders.find((h) => /hooks\.SessionStart/.test(h));
|
||||
assert.strictEqual(orphan, undefined,
|
||||
`Shape 4 strip must remove the parent [[hooks.SessionStart]] header too, got tables: ${outShape.tableHeaders.join(', ')}`);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -888,3 +888,643 @@ describe('bug #2950: stale deleted-command references removed from workflow file
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/feat-2840-issue-driven-orchestration-guide.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:feat-2840-issue-driven-orchestration-guide (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Tests for docs/issue-driven-orchestration.md (#2840).
|
||||
*
|
||||
* Structural-IR assertions per CONTRIBUTING.md "Prohibited: Raw Text Matching
|
||||
* on Test Outputs": parse the guide into a typed record and assert on
|
||||
* semantic flags, not regex on prose. The guide is rebuildable as long as
|
||||
* the structural invariants survive — section-level rewording is fine.
|
||||
*
|
||||
* Acceptance criteria from issue #2840:
|
||||
* - One guide explaining issue-driven orchestration using existing GSD
|
||||
* commands.
|
||||
* - Concrete end-to-end issue → workspace → plan/execute → verify/review
|
||||
* → PR flow.
|
||||
* - Explicitly documents safety boundaries: isolated worktrees, explicit
|
||||
* human review, no automatic public posting by default.
|
||||
* - Adds no runtime dependencies / no new command, daemon, or tracker
|
||||
* integration. (Test-enforced via concept-mapping audit.)
|
||||
*/
|
||||
|
||||
// allow-test-rule: structural-IR parser for a docs guide. The .includes() (see #2840)
|
||||
// calls below build a typed record (commandsPresent flags, conceptPairs
|
||||
// flags, nonGoalFlags, safetyFlags); assertions run on those booleans, not
|
||||
// on raw text. This is the documented escape hatch in
|
||||
// scripts/lint-no-source-grep.cjs for doc-shape tests.
|
||||
|
||||
'use strict';
|
||||
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const GUIDE_PATH = path.join(__dirname, '..', 'docs', 'issue-driven-orchestration.md');
|
||||
|
||||
// ─── Helpers ────────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Extract a section starting at a given heading. Returns the body up to (but
|
||||
* not including) the next heading at the same or shallower depth, or null if
|
||||
* the heading isn't found.
|
||||
*/
|
||||
function extractSection(content, heading) {
|
||||
const lines = content.split('\n');
|
||||
const headingRe = new RegExp(`^(#+)\\s+${heading.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\$&')}\\s*$`);
|
||||
let start = -1;
|
||||
let depth = 0;
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const m = lines[i].match(headingRe);
|
||||
if (m) {
|
||||
start = i + 1;
|
||||
depth = m[1].length;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (start < 0) return null;
|
||||
let end = lines.length;
|
||||
for (let i = start; i < lines.length; i++) {
|
||||
const m = lines[i].match(/^(#+)\s+/);
|
||||
if (m && m[1].length <= depth) {
|
||||
end = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return lines.slice(start, end).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the guide into a typed record. Returns null when the guide is
|
||||
* missing so the file-presence test can name the actual problem instead of
|
||||
* cascading TypeErrors.
|
||||
*/
|
||||
function parseGuide() {
|
||||
if (!fs.existsSync(GUIDE_PATH)) return null;
|
||||
const content = fs.readFileSync(GUIDE_PATH, 'utf8');
|
||||
// Strip inline emphasis but NOT underscores (snake_case identifiers like
|
||||
// gsd-new-workspace, .planning/, etc. must survive).
|
||||
const stripped = content.replace(/\*{1,3}|~{2}/g, '');
|
||||
|
||||
// Concept-mapping table: rows that pair a Symphony-style concept with a
|
||||
// GSD primitive. Test asserts on presence of each required pair, not on
|
||||
// exact prose ordering.
|
||||
const conceptMappingSection = extractSection(content, 'Concept mapping');
|
||||
const endToEndSection = extractSection(content, 'End-to-end flow') ||
|
||||
extractSection(content, 'End-to-end issue → PR flow') ||
|
||||
extractSection(content, 'End-to-end orchestration loop');
|
||||
const safetySection = extractSection(content, 'Safety boundaries') ||
|
||||
extractSection(content, 'Safety');
|
||||
const nonGoalsSection = extractSection(content, 'Non-goals') ||
|
||||
extractSection(content, 'What this guide does not do');
|
||||
|
||||
// Track which referenced commands appear at least once anywhere in the
|
||||
// guide. This prevents drift if /gsd-* command names are renamed.
|
||||
const requiredCommands = [
|
||||
'/gsd-workspace --new',
|
||||
'/gsd-manager',
|
||||
'/gsd-autonomous',
|
||||
'/gsd-discuss-phase',
|
||||
'/gsd-plan-phase',
|
||||
'/gsd-execute-phase',
|
||||
'/gsd-verify-work',
|
||||
'/gsd-review',
|
||||
'/gsd-ship',
|
||||
];
|
||||
const commandsPresent = Object.fromEntries(
|
||||
requiredCommands.map((c) => [c, content.includes(c)])
|
||||
);
|
||||
|
||||
// Concept-mapping invariants — keys are concept slugs, values are the
|
||||
// GSD primitive that must appear in the same paragraph/row of the
|
||||
// concept-mapping section.
|
||||
const conceptPairs = conceptMappingSection
|
||||
? {
|
||||
roadmap: /ROADMAP\.md/.test(conceptMappingSection),
|
||||
statemd: /STATE\.md/.test(conceptMappingSection),
|
||||
contextmd: /CONTEXT\.md/.test(conceptMappingSection),
|
||||
planmd: /PLAN\.md/.test(conceptMappingSection),
|
||||
workspaceCommand: /\/gsd-workspace\s+--new/.test(conceptMappingSection),
|
||||
executionCommand:
|
||||
/\/gsd-manager/.test(conceptMappingSection) ||
|
||||
/\/gsd-autonomous/.test(conceptMappingSection),
|
||||
verifyCommand: /\/gsd-verify-work/.test(conceptMappingSection),
|
||||
reviewCommand: /\/gsd-review/.test(conceptMappingSection),
|
||||
shipCommand: /\/gsd-ship/.test(conceptMappingSection),
|
||||
}
|
||||
: null;
|
||||
|
||||
// Non-goals required by the issue: must explicitly disclaim all four.
|
||||
const nonGoalFlags = nonGoalsSection
|
||||
? {
|
||||
noVendoring: /vendor|copy/i.test(nonGoalsSection),
|
||||
noDaemon: /daemon|polling/i.test(nonGoalsSection),
|
||||
noTrackerDependency: /tracker.*depend|mandatory.*track/i.test(nonGoalsSection),
|
||||
noBypassReview: /bypass|review|verification|human.*decision|human gate/i.test(nonGoalsSection),
|
||||
}
|
||||
: null;
|
||||
|
||||
// Safety boundaries — required disclaimers about how the loop stays safe.
|
||||
const safetyFlags = safetySection
|
||||
? {
|
||||
isolatedWorktrees: /worktree|isolated/i.test(safetySection),
|
||||
explicitReview: /review|human.*gate|human.*approval/i.test(safetySection),
|
||||
noAutoPosting: /not.*automatic|no.*auto|explicit.*confirm|user.*confirm|human.*confirm/i.test(safetySection),
|
||||
}
|
||||
: null;
|
||||
|
||||
// End-to-end flow must enumerate at least the seven step sequence the
|
||||
// acceptance criteria call out. We assert on numbered list items so the
|
||||
// narrative can be reworded freely.
|
||||
const numberedSteps = endToEndSection
|
||||
? (endToEndSection.match(/^\s*\d+\.\s+/gm) || []).length
|
||||
: 0;
|
||||
|
||||
// Strip markdown emphasis when checking for snake_case-sensitive content
|
||||
// in section bodies (per the markdown-aware matching pattern).
|
||||
const strippedConceptMapping = conceptMappingSection
|
||||
? conceptMappingSection.replace(/\*{1,3}|~{2}/g, '')
|
||||
: null;
|
||||
|
||||
return {
|
||||
raw: content,
|
||||
stripped,
|
||||
conceptMappingSection,
|
||||
strippedConceptMapping,
|
||||
endToEndSection,
|
||||
safetySection,
|
||||
nonGoalsSection,
|
||||
commandsPresent,
|
||||
conceptPairs,
|
||||
nonGoalFlags,
|
||||
safetyFlags,
|
||||
numberedSteps,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Tests ──────────────────────────────────────────────────────────────────
|
||||
|
||||
describe('issue-driven-orchestration guide (#2840)', () => {
|
||||
test('docs/issue-driven-orchestration.md exists', () => {
|
||||
assert.ok(
|
||||
fs.existsSync(GUIDE_PATH),
|
||||
`Guide must live at docs/issue-driven-orchestration.md per #2840`
|
||||
);
|
||||
});
|
||||
|
||||
test('every required GSD command is referenced at least once', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'parseGuide returned null — guide is missing');
|
||||
for (const [cmd, present] of Object.entries(ir.commandsPresent)) {
|
||||
assert.ok(present, `guide must reference ${cmd}`);
|
||||
}
|
||||
});
|
||||
|
||||
test('concept mapping section exists and pairs Symphony concepts with GSD primitives', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.conceptMappingSection,
|
||||
'guide must contain a "Concept mapping" section'
|
||||
);
|
||||
const expected = {
|
||||
roadmap: 'ROADMAP.md must appear in the concept mapping',
|
||||
statemd: 'STATE.md must appear in the concept mapping',
|
||||
contextmd: 'CONTEXT.md must appear in the concept mapping',
|
||||
planmd: 'PLAN.md must appear in the concept mapping',
|
||||
workspaceCommand: '/gsd-workspace --new must appear in the concept mapping',
|
||||
executionCommand:
|
||||
'/gsd-manager or /gsd-autonomous must appear in the concept mapping',
|
||||
verifyCommand: '/gsd-verify-work must appear in the concept mapping',
|
||||
reviewCommand: '/gsd-review must appear in the concept mapping',
|
||||
shipCommand: '/gsd-ship must appear in the concept mapping',
|
||||
};
|
||||
for (const [flag, msg] of Object.entries(expected)) {
|
||||
assert.equal(ir.conceptPairs[flag], true, msg);
|
||||
}
|
||||
});
|
||||
|
||||
test('safety boundaries section names isolation, review, and non-auto-posting', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.safetySection,
|
||||
'guide must contain a "Safety boundaries" or "Safety" section'
|
||||
);
|
||||
assert.equal(
|
||||
ir.safetyFlags.isolatedWorktrees,
|
||||
true,
|
||||
'safety section must mention isolated worktrees'
|
||||
);
|
||||
assert.equal(
|
||||
ir.safetyFlags.explicitReview,
|
||||
true,
|
||||
'safety section must require explicit human review'
|
||||
);
|
||||
assert.equal(
|
||||
ir.safetyFlags.noAutoPosting,
|
||||
true,
|
||||
'safety section must disclaim automatic public posting'
|
||||
);
|
||||
});
|
||||
|
||||
test('non-goals section disclaims vendoring, daemon, tracker dependency, and gate-bypass', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.nonGoalsSection,
|
||||
'guide must contain a "Non-goals" section'
|
||||
);
|
||||
const expected = {
|
||||
noVendoring: 'must disclaim copying/vendoring Symphony',
|
||||
noDaemon: 'must disclaim a long-running daemon',
|
||||
noTrackerDependency: 'must disclaim mandatory tracker dependency',
|
||||
noBypassReview: 'must disclaim bypassing review/verification gates',
|
||||
};
|
||||
for (const [flag, msg] of Object.entries(expected)) {
|
||||
assert.equal(ir.nonGoalFlags[flag], true, msg);
|
||||
}
|
||||
});
|
||||
|
||||
test('end-to-end flow enumerates at least 7 numbered steps (per acceptance criteria)', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.endToEndSection,
|
||||
'guide must contain an "End-to-end flow" (or equivalent) section'
|
||||
);
|
||||
assert.ok(
|
||||
ir.numberedSteps >= 7,
|
||||
`end-to-end section must enumerate ≥7 numbered steps; found ${ir.numberedSteps}`
|
||||
);
|
||||
});
|
||||
|
||||
test('every fenced code block has a language tag (markdownlint MD040)', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
// Pair fence opens; flag any opener with no language tag.
|
||||
const fences = ir.raw.match(/^```.*$/gm) || [];
|
||||
const openers = [];
|
||||
for (let i = 0; i < fences.length; i++) {
|
||||
// Even index = opener, odd = closer. An opener with empty trailing
|
||||
// text is MD040.
|
||||
if (i % 2 === 0) openers.push(fences[i]);
|
||||
}
|
||||
const bare = openers.filter((f) => /^```\s*$/.test(f));
|
||||
assert.equal(
|
||||
bare.length,
|
||||
0,
|
||||
`MD040: ${bare.length} fenced block(s) lack a language tag`
|
||||
);
|
||||
});
|
||||
|
||||
test('cross-linked from docs/README.md', () => {
|
||||
const readme = path.join(__dirname, '..', 'docs', 'README.md');
|
||||
if (!fs.existsSync(readme)) {
|
||||
// docs/README.md is the discovery surface. Without a cross-link, the
|
||||
// guide is invisible to users browsing docs/.
|
||||
return; // tolerate absence; test below ensures FEATURES.md anchor.
|
||||
}
|
||||
const txt = fs.readFileSync(readme, 'utf8');
|
||||
assert.ok(
|
||||
/issue-driven-orchestration/.test(txt),
|
||||
'docs/README.md must link to the new guide'
|
||||
);
|
||||
});
|
||||
|
||||
test('cross-linked from docs/USER-GUIDE.md', () => {
|
||||
const guide = path.join(__dirname, '..', 'docs', 'USER-GUIDE.md');
|
||||
// Mirror the null-guard pattern from the README test above: a missing
|
||||
// file must produce a meaningful assertion message, not a cryptic
|
||||
// ENOENT stack trace. (CR #3036.)
|
||||
assert.ok(
|
||||
fs.existsSync(guide),
|
||||
'docs/USER-GUIDE.md must exist for cross-link validation'
|
||||
);
|
||||
const txt = fs.readFileSync(guide, 'utf8');
|
||||
assert.ok(
|
||||
/issue-driven-orchestration/.test(txt),
|
||||
'docs/USER-GUIDE.md must link to the new guide'
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/feat-3025-mcp-token-budget-docs.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:feat-3025-mcp-token-budget-docs (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Documentation regression test for issue #3025 — MCP token-budget guidance.
|
||||
*
|
||||
* Verifies that gsd-core/references/context-budget.md contains the
|
||||
* structural elements the issue requires:
|
||||
*
|
||||
* 1. A section explaining MCP/tool schemas as a context-budget concern
|
||||
* 2. References to the harness-side toggles (enabledMcpjsonServers,
|
||||
* disabledMcpjsonServers in .claude/settings.json)
|
||||
* 3. A pre-phase audit checklist (browser/playwright, platform-specific,
|
||||
* project-specific)
|
||||
* 4. An explicit note that GSD does NOT manage MCP enablement — this is
|
||||
* a Claude Code harness concern (with a cross-link)
|
||||
* 5. Note the interaction with model_profile (compounding levers)
|
||||
*
|
||||
* Tests parse the doc into a typed section record (parseMcpSection) and
|
||||
* assert on flag booleans, not raw text matches. Adheres to
|
||||
* CONTRIBUTING.md "no-source-grep" — describes invariants, not wording,
|
||||
* so the prose can be reworded freely as long as the semantics survive.
|
||||
*
|
||||
* Companion to docs/USER-GUIDE.md task section, which is exercised by the
|
||||
* same parser shape (separate test below).
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const CONTEXT_BUDGET_MD = path.join(ROOT, 'gsd-core', 'references', 'context-budget.md');
|
||||
const USER_GUIDE_MD = path.join(ROOT, 'docs', 'USER-GUIDE.md');
|
||||
|
||||
/**
|
||||
* Extract the MCP-budget section from a markdown file by header text.
|
||||
* Returns null if the section is missing. Section runs from the matching
|
||||
* `## ` header up to the next `## ` header (or EOF).
|
||||
*/
|
||||
function extractSection(filePath, headerSubstring) {
|
||||
const content = fs.readFileSync(filePath, 'utf8');
|
||||
const lines = content.split(/\r?\n/);
|
||||
let inSection = false;
|
||||
let startDepth = 0;
|
||||
const collected = [];
|
||||
for (const line of lines) {
|
||||
const headerMatch = /^(#+)\s/.exec(line);
|
||||
if (headerMatch) {
|
||||
const depth = headerMatch[1].length;
|
||||
if (inSection) {
|
||||
// Section ends at a header at the same or shallower depth.
|
||||
// Subsections at deeper depth are part of the section.
|
||||
if (depth <= startDepth) break;
|
||||
} else if (line.toLowerCase().includes(headerSubstring.toLowerCase())) {
|
||||
inSection = true;
|
||||
startDepth = depth;
|
||||
}
|
||||
}
|
||||
if (inSection) collected.push(line);
|
||||
}
|
||||
return collected.length > 0 ? collected.join('\n') : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the MCP-budget section into a typed semantic-flag record.
|
||||
* Each flag answers a single behavioral question that #3025 requires
|
||||
* the prose to encode.
|
||||
*/
|
||||
function parseMcpBudgetSection(section) {
|
||||
if (!section || typeof section !== 'string') {
|
||||
return {
|
||||
ok: false,
|
||||
sectionLength: 0,
|
||||
explainsMcpAsBudgetConcern: false,
|
||||
namesEnabledMcpjsonServers: false,
|
||||
namesDisabledMcpjsonServers: false,
|
||||
namesClaudeSettingsJson: false,
|
||||
includesPrePhaseAudit: false,
|
||||
auditMentionsBrowserOrPlaywright: false,
|
||||
auditMentionsPlatformSpecific: false,
|
||||
auditMentionsCrossProject: false,
|
||||
explainsHarnessNotGsd: false,
|
||||
mentionsModelProfileInteraction: false,
|
||||
crossLinksContextBudget: false,
|
||||
};
|
||||
}
|
||||
// CR follow-up: strip inline markdown emphasis (`**`, `*`, `~~`) and
|
||||
// backticks before phrase-matching so e.g. "GSD does **not** manage"
|
||||
// is caught by the primary `gsd does not manage` alternative below.
|
||||
// WITHOUT this, the markdown-bold breaks the contiguous match and the
|
||||
// test only passes via the fallback branch (silent dead code).
|
||||
// Underscores are intentionally NOT stripped — `model_profile` and
|
||||
// other snake_case identifiers must survive intact so the
|
||||
// model_profile interaction check still finds them.
|
||||
const stripped = section.replace(/\*{1,3}|~{2}|`/g, '');
|
||||
// (1) Explains MCP as budget concern — must mention BOTH "MCP" / "tool
|
||||
// schema" AND a token/cost framing.
|
||||
const explainsMcpAsBudgetConcern =
|
||||
/\bmcp\b|tool schema|tool schemas/i.test(stripped) &&
|
||||
/\btoken|context budget|per[- ]turn|cost\b/i.test(stripped);
|
||||
// (2) Names the harness keys verbatim
|
||||
const namesEnabledMcpjsonServers = /enabledMcpjsonServers/.test(stripped);
|
||||
const namesDisabledMcpjsonServers = /disabledMcpjsonServers/.test(stripped);
|
||||
// (3) Names the settings file location
|
||||
const namesClaudeSettingsJson = /\.claude\/settings\.json/.test(stripped);
|
||||
// (4) Audit checklist — must mention all three classes the issue
|
||||
// calls out, plus a "before this phase / pre-phase" framing
|
||||
const includesPrePhaseAudit =
|
||||
/audit|checklist|review (your )?mcp|before (starting|beginning) (a |the )?phase/i.test(stripped);
|
||||
const auditMentionsBrowserOrPlaywright = /\bbrowser\b|playwright/i.test(stripped);
|
||||
const auditMentionsPlatformSpecific = /platform[- ]specific|mac[- ]?tools|windows[- ]?tools|os[- ]specific/i.test(stripped);
|
||||
const auditMentionsCrossProject = /(other|different|cross[- ])\s*project|stale (project )?mcp/i.test(stripped);
|
||||
// (5) Harness vs GSD distinction — must explicitly state GSD doesn't
|
||||
// own this knob and point at the harness
|
||||
const explainsHarnessNotGsd =
|
||||
/(gsd does(?:n[''’]t| not) (own|manage|control)|harness (concern|setting|controlled)|not a gsd (setting|knob))/i.test(stripped);
|
||||
// (6) Compounding with model_profile
|
||||
const mentionsModelProfileInteraction =
|
||||
/model[_ ]profile/i.test(stripped) &&
|
||||
/compound|multiplier|stack|every[- ]turn|regardless of (which )?model|in addition/i.test(stripped);
|
||||
// (7) Cross-link to the canonical reference doc — task-guide section
|
||||
// must point readers at context-budget.md for the full audit. Encoded
|
||||
// as a named flag (CR follow-up) so the assertion sits alongside the
|
||||
// other parsed invariants rather than as a one-off inline regex.
|
||||
const crossLinksContextBudget = /context-budget/i.test(stripped);
|
||||
return {
|
||||
ok: true,
|
||||
sectionLength: section.length,
|
||||
explainsMcpAsBudgetConcern,
|
||||
namesEnabledMcpjsonServers,
|
||||
namesDisabledMcpjsonServers,
|
||||
namesClaudeSettingsJson,
|
||||
includesPrePhaseAudit,
|
||||
auditMentionsBrowserOrPlaywright,
|
||||
auditMentionsPlatformSpecific,
|
||||
auditMentionsCrossProject,
|
||||
explainsHarnessNotGsd,
|
||||
mentionsModelProfileInteraction,
|
||||
crossLinksContextBudget,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── context-budget.md ──────────────────────────────────────────────────────
|
||||
|
||||
describe('#3025 context-budget.md: MCP token-budget section exists with required content', () => {
|
||||
test('the file exists', () => {
|
||||
assert.ok(fs.existsSync(CONTEXT_BUDGET_MD), `expected file at ${CONTEXT_BUDGET_MD}`);
|
||||
});
|
||||
|
||||
test('has a section header that mentions MCP', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
assert.ok(section, 'must have a `## ...MCP...` heading; section was not found');
|
||||
});
|
||||
|
||||
test('explains MCP/tool schemas as a context-budget concern (#3025 requirement 1)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.explainsMcpAsBudgetConcern, true,
|
||||
`must explain MCP/tool schemas as a token/context-budget concern; section was:\n${section}`);
|
||||
});
|
||||
|
||||
test('names enabledMcpjsonServers and disabledMcpjsonServers (#3025 requirement 2)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.namesEnabledMcpjsonServers, true,
|
||||
'section must reference `enabledMcpjsonServers` so users know the exact key');
|
||||
assert.equal(parsed.namesDisabledMcpjsonServers, true,
|
||||
'section must reference `disabledMcpjsonServers` for parity');
|
||||
assert.equal(parsed.namesClaudeSettingsJson, true,
|
||||
'section must name `.claude/settings.json` as the location of the toggle');
|
||||
});
|
||||
|
||||
test('includes a pre-phase audit checklist with all three classes (#3025 requirement 3)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.includesPrePhaseAudit, true,
|
||||
'section must include audit/checklist framing for pre-phase MCP review');
|
||||
assert.equal(parsed.auditMentionsBrowserOrPlaywright, true,
|
||||
'audit must mention browser/playwright tools as a candidate for disabling');
|
||||
assert.equal(parsed.auditMentionsPlatformSpecific, true,
|
||||
'audit must mention platform-specific tools (Mac/Windows/OS-specific)');
|
||||
assert.equal(parsed.auditMentionsCrossProject, true,
|
||||
'audit must mention stale/cross-project MCPs from other projects');
|
||||
});
|
||||
|
||||
test('explains GSD does not own MCP enablement — harness concern (#3025 requirement 4)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.explainsHarnessNotGsd, true,
|
||||
'section must explicitly state GSD does not manage MCP enablement (harness concern)');
|
||||
});
|
||||
|
||||
test('notes interaction with model_profile (compounding levers) (#3025 requirement 5)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.mentionsModelProfileInteraction, true,
|
||||
'section must note that trimming MCPs compounds with model_profile choice');
|
||||
});
|
||||
|
||||
test('full semantic record matches the #3025 contract — typed snapshot', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
const contract = {
|
||||
ok: parsed.ok,
|
||||
explainsMcpAsBudgetConcern: parsed.explainsMcpAsBudgetConcern,
|
||||
namesEnabledMcpjsonServers: parsed.namesEnabledMcpjsonServers,
|
||||
namesDisabledMcpjsonServers: parsed.namesDisabledMcpjsonServers,
|
||||
namesClaudeSettingsJson: parsed.namesClaudeSettingsJson,
|
||||
includesPrePhaseAudit: parsed.includesPrePhaseAudit,
|
||||
auditMentionsBrowserOrPlaywright: parsed.auditMentionsBrowserOrPlaywright,
|
||||
auditMentionsPlatformSpecific: parsed.auditMentionsPlatformSpecific,
|
||||
auditMentionsCrossProject: parsed.auditMentionsCrossProject,
|
||||
explainsHarnessNotGsd: parsed.explainsHarnessNotGsd,
|
||||
mentionsModelProfileInteraction: parsed.mentionsModelProfileInteraction,
|
||||
};
|
||||
assert.deepStrictEqual(contract, {
|
||||
ok: true,
|
||||
explainsMcpAsBudgetConcern: true,
|
||||
namesEnabledMcpjsonServers: true,
|
||||
namesDisabledMcpjsonServers: true,
|
||||
namesClaudeSettingsJson: true,
|
||||
includesPrePhaseAudit: true,
|
||||
auditMentionsBrowserOrPlaywright: true,
|
||||
auditMentionsPlatformSpecific: true,
|
||||
auditMentionsCrossProject: true,
|
||||
explainsHarnessNotGsd: true,
|
||||
mentionsModelProfileInteraction: true,
|
||||
}, 'context-budget.md MCP section contract violated');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── docs/USER-GUIDE.md task section ────────────────────────────────────────
|
||||
|
||||
describe('#3025 docs/USER-GUIDE.md: companion task section exists', () => {
|
||||
test('USER-GUIDE.md has an MCP-trimming task section', () => {
|
||||
const section = extractSection(USER_GUIDE_MD, 'mcp');
|
||||
assert.ok(section,
|
||||
'USER-GUIDE.md must have a `### ...MCP...` task section so users find it via the guide');
|
||||
});
|
||||
|
||||
test('USER-GUIDE.md task section names the harness key and cross-links the reference', () => {
|
||||
const section = extractSection(USER_GUIDE_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.namesEnabledMcpjsonServers, true,
|
||||
'task section must mention the harness key by name');
|
||||
// Cross-link to the reference doc — assert on the parsed flag so
|
||||
// the invariant lives alongside the other named flags (CR follow-up
|
||||
// on the no-source-grep standard).
|
||||
assert.equal(parsed.crossLinksContextBudget, true,
|
||||
'task section must cross-link to context-budget.md');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── markdownlint pre-flight (per bundle-docs-with-code skill) ──────────────
|
||||
|
||||
describe('#3025 markdownlint pre-flight: MD040 + MD056', () => {
|
||||
test('every fenced code block in the new MCP section has a language tag (MD040)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
// Guard: extractSection returns null when the section is missing.
|
||||
// Without this, `section.match(...)` would throw a TypeError instead
|
||||
// of producing a meaningful assertion failure (CR follow-up).
|
||||
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD040');
|
||||
const fences = (section.match(/^```([a-zA-Z0-9_+-]*)?\s*$/gm) || []);
|
||||
// Pairs of fences open/close; odd-indexed ones close blocks. Every
|
||||
// OPENING fence must have a language tag. Closing fences are bare ```.
|
||||
// Walk pairs: even index = opener, odd = closer.
|
||||
const openers = fences.filter((_, i) => i % 2 === 0);
|
||||
const missing = openers.filter((line) => /^```\s*$/.test(line));
|
||||
assert.deepStrictEqual(missing, [],
|
||||
`every fenced code block opener must have a language tag (MD040). Missing: ${JSON.stringify(missing)}`);
|
||||
});
|
||||
|
||||
test('every markdown table row in the new MCP section has the same column count as its header (MD056)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
// Guard: same null-section concern as MD040 above (CR follow-up).
|
||||
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD056');
|
||||
const lines = section.split(/\r?\n/);
|
||||
// Walk through and detect tables: header row followed by a separator
|
||||
// (--- pattern) followed by data rows. Count `|` per line.
|
||||
const issues = [];
|
||||
for (let i = 0; i < lines.length - 1; i += 1) {
|
||||
const header = lines[i];
|
||||
const sep = lines[i + 1];
|
||||
if (!/^\s*\|.*\|\s*$/.test(header)) continue;
|
||||
if (!/^\s*\|[\s\-:|]+\|\s*$/.test(sep)) continue;
|
||||
const headerCols = (header.match(/\|/g) || []).length;
|
||||
// Walk data rows
|
||||
for (let j = i + 2; j < lines.length; j += 1) {
|
||||
const row = lines[j];
|
||||
if (!/^\s*\|.*\|\s*$/.test(row)) break;
|
||||
const rowCols = (row.match(/\|/g) || []).length;
|
||||
if (rowCols !== headerCols) {
|
||||
issues.push({ line: j, expected: headerCols, actual: rowCols, row });
|
||||
}
|
||||
}
|
||||
}
|
||||
assert.deepStrictEqual(issues, [],
|
||||
`table rows must match header column count (MD056). Issues: ${JSON.stringify(issues, null, 2)}`);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1,96 +0,0 @@
|
||||
// allow-test-rule: source-text-is-the-product
|
||||
// The PROJECT.md template + complete-milestone workflow .md ARE the product surface
|
||||
// the runtime loads; asserting on their text tests the deployed contract directly.
|
||||
/**
|
||||
* Enhancement #72 — optional Business Context section in the PROJECT.md template.
|
||||
*
|
||||
* Contract tests over the product-text surfaces (template + milestone workflow .md):
|
||||
* the template offers a Business Context section that is explicitly OPTIONAL, capped
|
||||
* at the four approved one-line fields, and the milestone evolution review treats it
|
||||
* as conditional so non-business projects that deleted it are never forced to review it.
|
||||
*/
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const TEMPLATE = path.join(__dirname, '..', 'gsd-core', 'templates', 'project.md');
|
||||
const COMPLETE_MILESTONE = path.join(__dirname, '..', 'gsd-core', 'workflows', 'complete-milestone.md');
|
||||
|
||||
function parseTemplateContract(content) {
|
||||
const lines = content.split(/\r?\n/);
|
||||
const lower = content.toLowerCase();
|
||||
// The Business Context block lives between its heading and the next "## " heading.
|
||||
const startIdx = lines.findIndex(l => l.trim() === '## Business Context');
|
||||
let sectionBody = '';
|
||||
if (startIdx !== -1) {
|
||||
const rest = lines.slice(startIdx + 1);
|
||||
const endOffset = rest.findIndex(l => l.startsWith('## '));
|
||||
sectionBody = (endOffset === -1 ? rest : rest.slice(0, endOffset)).join('\n');
|
||||
}
|
||||
const fieldOf = (label) => new RegExp(`^- \\*\\*${label}\\*\\*:`, 'm').test(sectionBody);
|
||||
return {
|
||||
hasSection: startIdx !== -1,
|
||||
// Optional-by-default: an HTML comment tells non-business projects to delete it.
|
||||
hasOptionalMarker: /<!--\s*OPTIONAL/i.test(sectionBody) && /delete this section/i.test(sectionBody),
|
||||
fields: {
|
||||
customer: fieldOf('Customer'),
|
||||
revenueModel: fieldOf('Revenue model'),
|
||||
successMetric: fieldOf('Success metric'),
|
||||
strategyNotes: fieldOf('Strategy notes'),
|
||||
},
|
||||
fieldCount: (sectionBody.match(/^- \*\*/gm) || []).length,
|
||||
// Positioned between Core Value and Requirements.
|
||||
orderedBetweenCoreValueAndRequirements:
|
||||
lower.indexOf('## core value') < lower.indexOf('## business context') &&
|
||||
lower.indexOf('## business context') < lower.indexOf('## requirements'),
|
||||
hasGuidelinesEntry: /\*\*Business Context:\*\*/.test(content),
|
||||
};
|
||||
}
|
||||
|
||||
function parseMilestoneContract(content) {
|
||||
const lower = content.toLowerCase();
|
||||
const lines = content.split(/\r?\n/);
|
||||
const reviewLine = lines.find(l =>
|
||||
l.toLowerCase().includes('business context') &&
|
||||
(l.toLowerCase().includes('if present') || l.toLowerCase().includes('only if')),
|
||||
);
|
||||
return {
|
||||
mentionsBusinessContext: lower.includes('business context'),
|
||||
hasConditionalReview: Boolean(reviewLine),
|
||||
};
|
||||
}
|
||||
|
||||
describe('enhancement #72 — Business Context template section', () => {
|
||||
const tpl = parseTemplateContract(fs.readFileSync(TEMPLATE, 'utf-8'));
|
||||
|
||||
test('template includes a Business Context section', () => {
|
||||
assert.ok(tpl.hasSection, 'template must contain a "## Business Context" section');
|
||||
});
|
||||
|
||||
test('section is marked OPTIONAL with delete-for-non-business guidance', () => {
|
||||
assert.ok(tpl.hasOptionalMarker, 'section must carry an OPTIONAL HTML comment telling non-business projects to delete it');
|
||||
});
|
||||
|
||||
test('section carries exactly the four approved one-line fields', () => {
|
||||
assert.ok(tpl.fields.customer, 'missing **Customer** field');
|
||||
assert.ok(tpl.fields.revenueModel, 'missing **Revenue model** field');
|
||||
assert.ok(tpl.fields.successMetric, 'missing **Success metric** field');
|
||||
assert.ok(tpl.fields.strategyNotes, 'missing **Strategy notes** field');
|
||||
assert.strictEqual(tpl.fieldCount, 4, 'section is capped at four fields (constraint reference, not a business plan)');
|
||||
});
|
||||
|
||||
test('section is positioned between Core Value and Requirements', () => {
|
||||
assert.ok(tpl.orderedBetweenCoreValueAndRequirements, 'Business Context must sit between Core Value and Requirements');
|
||||
});
|
||||
|
||||
test('guidelines document the Business Context section', () => {
|
||||
assert.ok(tpl.hasGuidelinesEntry, 'guidelines block must include a **Business Context:** entry');
|
||||
});
|
||||
|
||||
test('milestone evolution reviews Business Context only when present', () => {
|
||||
const ms = parseMilestoneContract(fs.readFileSync(COMPLETE_MILESTONE, 'utf-8'));
|
||||
assert.ok(ms.mentionsBusinessContext, 'complete-milestone must mention Business Context in its review');
|
||||
assert.ok(ms.hasConditionalReview, 'the Business Context milestone review must be conditional on the section being present');
|
||||
});
|
||||
});
|
||||
@@ -1,320 +0,0 @@
|
||||
/**
|
||||
* Tests for docs/issue-driven-orchestration.md (#2840).
|
||||
*
|
||||
* Structural-IR assertions per CONTRIBUTING.md "Prohibited: Raw Text Matching
|
||||
* on Test Outputs": parse the guide into a typed record and assert on
|
||||
* semantic flags, not regex on prose. The guide is rebuildable as long as
|
||||
* the structural invariants survive — section-level rewording is fine.
|
||||
*
|
||||
* Acceptance criteria from issue #2840:
|
||||
* - One guide explaining issue-driven orchestration using existing GSD
|
||||
* commands.
|
||||
* - Concrete end-to-end issue → workspace → plan/execute → verify/review
|
||||
* → PR flow.
|
||||
* - Explicitly documents safety boundaries: isolated worktrees, explicit
|
||||
* human review, no automatic public posting by default.
|
||||
* - Adds no runtime dependencies / no new command, daemon, or tracker
|
||||
* integration. (Test-enforced via concept-mapping audit.)
|
||||
*/
|
||||
|
||||
// allow-test-rule: structural-IR parser for a docs guide. The .includes()
|
||||
// calls below build a typed record (commandsPresent flags, conceptPairs
|
||||
// flags, nonGoalFlags, safetyFlags); assertions run on those booleans, not
|
||||
// on raw text. This is the documented escape hatch in
|
||||
// scripts/lint-no-source-grep.cjs for doc-shape tests.
|
||||
|
||||
'use strict';
|
||||
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const GUIDE_PATH = path.join(__dirname, '..', 'docs', 'issue-driven-orchestration.md');
|
||||
|
||||
// ─── Helpers ────────────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Extract a section starting at a given heading. Returns the body up to (but
|
||||
* not including) the next heading at the same or shallower depth, or null if
|
||||
* the heading isn't found.
|
||||
*/
|
||||
function extractSection(content, heading) {
|
||||
const lines = content.split('\n');
|
||||
const headingRe = new RegExp(`^(#+)\\s+${heading.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\$&')}\\s*$`);
|
||||
let start = -1;
|
||||
let depth = 0;
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const m = lines[i].match(headingRe);
|
||||
if (m) {
|
||||
start = i + 1;
|
||||
depth = m[1].length;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (start < 0) return null;
|
||||
let end = lines.length;
|
||||
for (let i = start; i < lines.length; i++) {
|
||||
const m = lines[i].match(/^(#+)\s+/);
|
||||
if (m && m[1].length <= depth) {
|
||||
end = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return lines.slice(start, end).join('\n');
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the guide into a typed record. Returns null when the guide is
|
||||
* missing so the file-presence test can name the actual problem instead of
|
||||
* cascading TypeErrors.
|
||||
*/
|
||||
function parseGuide() {
|
||||
if (!fs.existsSync(GUIDE_PATH)) return null;
|
||||
const content = fs.readFileSync(GUIDE_PATH, 'utf8');
|
||||
// Strip inline emphasis but NOT underscores (snake_case identifiers like
|
||||
// gsd-new-workspace, .planning/, etc. must survive).
|
||||
const stripped = content.replace(/\*{1,3}|~{2}/g, '');
|
||||
|
||||
// Concept-mapping table: rows that pair a Symphony-style concept with a
|
||||
// GSD primitive. Test asserts on presence of each required pair, not on
|
||||
// exact prose ordering.
|
||||
const conceptMappingSection = extractSection(content, 'Concept mapping');
|
||||
const endToEndSection = extractSection(content, 'End-to-end flow') ||
|
||||
extractSection(content, 'End-to-end issue → PR flow') ||
|
||||
extractSection(content, 'End-to-end orchestration loop');
|
||||
const safetySection = extractSection(content, 'Safety boundaries') ||
|
||||
extractSection(content, 'Safety');
|
||||
const nonGoalsSection = extractSection(content, 'Non-goals') ||
|
||||
extractSection(content, 'What this guide does not do');
|
||||
|
||||
// Track which referenced commands appear at least once anywhere in the
|
||||
// guide. This prevents drift if /gsd-* command names are renamed.
|
||||
const requiredCommands = [
|
||||
'/gsd-workspace --new',
|
||||
'/gsd-manager',
|
||||
'/gsd-autonomous',
|
||||
'/gsd-discuss-phase',
|
||||
'/gsd-plan-phase',
|
||||
'/gsd-execute-phase',
|
||||
'/gsd-verify-work',
|
||||
'/gsd-review',
|
||||
'/gsd-ship',
|
||||
];
|
||||
const commandsPresent = Object.fromEntries(
|
||||
requiredCommands.map((c) => [c, content.includes(c)])
|
||||
);
|
||||
|
||||
// Concept-mapping invariants — keys are concept slugs, values are the
|
||||
// GSD primitive that must appear in the same paragraph/row of the
|
||||
// concept-mapping section.
|
||||
const conceptPairs = conceptMappingSection
|
||||
? {
|
||||
roadmap: /ROADMAP\.md/.test(conceptMappingSection),
|
||||
statemd: /STATE\.md/.test(conceptMappingSection),
|
||||
contextmd: /CONTEXT\.md/.test(conceptMappingSection),
|
||||
planmd: /PLAN\.md/.test(conceptMappingSection),
|
||||
workspaceCommand: /\/gsd-workspace\s+--new/.test(conceptMappingSection),
|
||||
executionCommand:
|
||||
/\/gsd-manager/.test(conceptMappingSection) ||
|
||||
/\/gsd-autonomous/.test(conceptMappingSection),
|
||||
verifyCommand: /\/gsd-verify-work/.test(conceptMappingSection),
|
||||
reviewCommand: /\/gsd-review/.test(conceptMappingSection),
|
||||
shipCommand: /\/gsd-ship/.test(conceptMappingSection),
|
||||
}
|
||||
: null;
|
||||
|
||||
// Non-goals required by the issue: must explicitly disclaim all four.
|
||||
const nonGoalFlags = nonGoalsSection
|
||||
? {
|
||||
noVendoring: /vendor|copy/i.test(nonGoalsSection),
|
||||
noDaemon: /daemon|polling/i.test(nonGoalsSection),
|
||||
noTrackerDependency: /tracker.*depend|mandatory.*track/i.test(nonGoalsSection),
|
||||
noBypassReview: /bypass|review|verification|human.*decision|human gate/i.test(nonGoalsSection),
|
||||
}
|
||||
: null;
|
||||
|
||||
// Safety boundaries — required disclaimers about how the loop stays safe.
|
||||
const safetyFlags = safetySection
|
||||
? {
|
||||
isolatedWorktrees: /worktree|isolated/i.test(safetySection),
|
||||
explicitReview: /review|human.*gate|human.*approval/i.test(safetySection),
|
||||
noAutoPosting: /not.*automatic|no.*auto|explicit.*confirm|user.*confirm|human.*confirm/i.test(safetySection),
|
||||
}
|
||||
: null;
|
||||
|
||||
// End-to-end flow must enumerate at least the seven step sequence the
|
||||
// acceptance criteria call out. We assert on numbered list items so the
|
||||
// narrative can be reworded freely.
|
||||
const numberedSteps = endToEndSection
|
||||
? (endToEndSection.match(/^\s*\d+\.\s+/gm) || []).length
|
||||
: 0;
|
||||
|
||||
// Strip markdown emphasis when checking for snake_case-sensitive content
|
||||
// in section bodies (per the markdown-aware matching pattern).
|
||||
const strippedConceptMapping = conceptMappingSection
|
||||
? conceptMappingSection.replace(/\*{1,3}|~{2}/g, '')
|
||||
: null;
|
||||
|
||||
return {
|
||||
raw: content,
|
||||
stripped,
|
||||
conceptMappingSection,
|
||||
strippedConceptMapping,
|
||||
endToEndSection,
|
||||
safetySection,
|
||||
nonGoalsSection,
|
||||
commandsPresent,
|
||||
conceptPairs,
|
||||
nonGoalFlags,
|
||||
safetyFlags,
|
||||
numberedSteps,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Tests ──────────────────────────────────────────────────────────────────
|
||||
|
||||
describe('issue-driven-orchestration guide (#2840)', () => {
|
||||
test('docs/issue-driven-orchestration.md exists', () => {
|
||||
assert.ok(
|
||||
fs.existsSync(GUIDE_PATH),
|
||||
`Guide must live at docs/issue-driven-orchestration.md per #2840`
|
||||
);
|
||||
});
|
||||
|
||||
test('every required GSD command is referenced at least once', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'parseGuide returned null — guide is missing');
|
||||
for (const [cmd, present] of Object.entries(ir.commandsPresent)) {
|
||||
assert.ok(present, `guide must reference ${cmd}`);
|
||||
}
|
||||
});
|
||||
|
||||
test('concept mapping section exists and pairs Symphony concepts with GSD primitives', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.conceptMappingSection,
|
||||
'guide must contain a "Concept mapping" section'
|
||||
);
|
||||
const expected = {
|
||||
roadmap: 'ROADMAP.md must appear in the concept mapping',
|
||||
statemd: 'STATE.md must appear in the concept mapping',
|
||||
contextmd: 'CONTEXT.md must appear in the concept mapping',
|
||||
planmd: 'PLAN.md must appear in the concept mapping',
|
||||
workspaceCommand: '/gsd-workspace --new must appear in the concept mapping',
|
||||
executionCommand:
|
||||
'/gsd-manager or /gsd-autonomous must appear in the concept mapping',
|
||||
verifyCommand: '/gsd-verify-work must appear in the concept mapping',
|
||||
reviewCommand: '/gsd-review must appear in the concept mapping',
|
||||
shipCommand: '/gsd-ship must appear in the concept mapping',
|
||||
};
|
||||
for (const [flag, msg] of Object.entries(expected)) {
|
||||
assert.equal(ir.conceptPairs[flag], true, msg);
|
||||
}
|
||||
});
|
||||
|
||||
test('safety boundaries section names isolation, review, and non-auto-posting', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.safetySection,
|
||||
'guide must contain a "Safety boundaries" or "Safety" section'
|
||||
);
|
||||
assert.equal(
|
||||
ir.safetyFlags.isolatedWorktrees,
|
||||
true,
|
||||
'safety section must mention isolated worktrees'
|
||||
);
|
||||
assert.equal(
|
||||
ir.safetyFlags.explicitReview,
|
||||
true,
|
||||
'safety section must require explicit human review'
|
||||
);
|
||||
assert.equal(
|
||||
ir.safetyFlags.noAutoPosting,
|
||||
true,
|
||||
'safety section must disclaim automatic public posting'
|
||||
);
|
||||
});
|
||||
|
||||
test('non-goals section disclaims vendoring, daemon, tracker dependency, and gate-bypass', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.nonGoalsSection,
|
||||
'guide must contain a "Non-goals" section'
|
||||
);
|
||||
const expected = {
|
||||
noVendoring: 'must disclaim copying/vendoring Symphony',
|
||||
noDaemon: 'must disclaim a long-running daemon',
|
||||
noTrackerDependency: 'must disclaim mandatory tracker dependency',
|
||||
noBypassReview: 'must disclaim bypassing review/verification gates',
|
||||
};
|
||||
for (const [flag, msg] of Object.entries(expected)) {
|
||||
assert.equal(ir.nonGoalFlags[flag], true, msg);
|
||||
}
|
||||
});
|
||||
|
||||
test('end-to-end flow enumerates at least 7 numbered steps (per acceptance criteria)', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
assert.ok(
|
||||
ir.endToEndSection,
|
||||
'guide must contain an "End-to-end flow" (or equivalent) section'
|
||||
);
|
||||
assert.ok(
|
||||
ir.numberedSteps >= 7,
|
||||
`end-to-end section must enumerate ≥7 numbered steps; found ${ir.numberedSteps}`
|
||||
);
|
||||
});
|
||||
|
||||
test('every fenced code block has a language tag (markdownlint MD040)', () => {
|
||||
const ir = parseGuide();
|
||||
assert.ok(ir, 'guide must be present');
|
||||
// Pair fence opens; flag any opener with no language tag.
|
||||
const fences = ir.raw.match(/^```.*$/gm) || [];
|
||||
const openers = [];
|
||||
for (let i = 0; i < fences.length; i++) {
|
||||
// Even index = opener, odd = closer. An opener with empty trailing
|
||||
// text is MD040.
|
||||
if (i % 2 === 0) openers.push(fences[i]);
|
||||
}
|
||||
const bare = openers.filter((f) => /^```\s*$/.test(f));
|
||||
assert.equal(
|
||||
bare.length,
|
||||
0,
|
||||
`MD040: ${bare.length} fenced block(s) lack a language tag`
|
||||
);
|
||||
});
|
||||
|
||||
test('cross-linked from docs/README.md', () => {
|
||||
const readme = path.join(__dirname, '..', 'docs', 'README.md');
|
||||
if (!fs.existsSync(readme)) {
|
||||
// docs/README.md is the discovery surface. Without a cross-link, the
|
||||
// guide is invisible to users browsing docs/.
|
||||
return; // tolerate absence; test below ensures FEATURES.md anchor.
|
||||
}
|
||||
const txt = fs.readFileSync(readme, 'utf8');
|
||||
assert.ok(
|
||||
/issue-driven-orchestration/.test(txt),
|
||||
'docs/README.md must link to the new guide'
|
||||
);
|
||||
});
|
||||
|
||||
test('cross-linked from docs/USER-GUIDE.md', () => {
|
||||
const guide = path.join(__dirname, '..', 'docs', 'USER-GUIDE.md');
|
||||
// Mirror the null-guard pattern from the README test above: a missing
|
||||
// file must produce a meaningful assertion message, not a cryptic
|
||||
// ENOENT stack trace. (CR #3036.)
|
||||
assert.ok(
|
||||
fs.existsSync(guide),
|
||||
'docs/USER-GUIDE.md must exist for cross-link validation'
|
||||
);
|
||||
const txt = fs.readFileSync(guide, 'utf8');
|
||||
assert.ok(
|
||||
/issue-driven-orchestration/.test(txt),
|
||||
'docs/USER-GUIDE.md must link to the new guide'
|
||||
);
|
||||
});
|
||||
});
|
||||
@@ -1,377 +0,0 @@
|
||||
/**
|
||||
* Feature test for issue #3023 — per-phase-type model map.
|
||||
*
|
||||
* Adds a `models` block to .planning/config.json that accepts phase-type
|
||||
* keys (planning / discuss / research / execution / verification /
|
||||
* completion). Resolution precedence:
|
||||
*
|
||||
* 1. Per-agent `model_overrides[agent]` (highest)
|
||||
* 2. Phase-type `models[phase_type]` (NEW)
|
||||
* 3. Profile table (`model_profile`)
|
||||
* 4. Runtime default
|
||||
*
|
||||
* Tests are typed-IR / structural — assert on the value returned by
|
||||
* resolveModelInternal, not stdout/grep. Each test seeds a temp project
|
||||
* with a fixture .planning/config.json and asserts the resolver picks
|
||||
* the right tier for each agent.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { test, describe, beforeEach, afterEach } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const {
|
||||
resolveModelInternal,
|
||||
} = require('../gsd-core/bin/lib/model-resolver.cjs');
|
||||
const {
|
||||
AGENT_TO_PHASE_TYPE,
|
||||
VALID_PHASE_TYPES,
|
||||
MODEL_PROFILES,
|
||||
} = require('../gsd-core/bin/lib/model-profiles.cjs');
|
||||
const { isValidConfigKey } = require('../gsd-core/bin/lib/config-schema.cjs');
|
||||
|
||||
const { createTempDir, cleanup } = require('./helpers.cjs');
|
||||
const makeTmp = (prefix) => createTempDir(`gsd-3023-${prefix}-`);
|
||||
|
||||
function writeConfig(projectDir, config) {
|
||||
const planningDir = path.join(projectDir, '.planning');
|
||||
fs.mkdirSync(planningDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
|
||||
}
|
||||
|
||||
function rmr(p) {
|
||||
cleanup(p);
|
||||
}
|
||||
|
||||
// ─── Schema: AGENT_TO_PHASE_TYPE table + VALID_PHASE_TYPES ──────────────────
|
||||
|
||||
describe('#3023 phase-type schema: every agent has a phase-type assignment', () => {
|
||||
test('AGENT_TO_PHASE_TYPE is exported as a non-empty object', () => {
|
||||
assert.equal(typeof AGENT_TO_PHASE_TYPE, 'object');
|
||||
assert.ok(AGENT_TO_PHASE_TYPE !== null);
|
||||
assert.ok(Object.keys(AGENT_TO_PHASE_TYPE).length > 0);
|
||||
});
|
||||
|
||||
test('VALID_PHASE_TYPES exposes the six named slots from the issue', () => {
|
||||
// The issue specified exactly these slots. Adding new slots here is a
|
||||
// schema change that must coordinate with config-schema's dynamic
|
||||
// pattern and the docs.
|
||||
assert.deepStrictEqual(
|
||||
[...VALID_PHASE_TYPES].sort(),
|
||||
['completion', 'discuss', 'execution', 'planning', 'research', 'verification'].sort()
|
||||
);
|
||||
});
|
||||
|
||||
test('every agent in MODEL_PROFILES has a phase-type assignment', () => {
|
||||
const missing = Object.keys(MODEL_PROFILES).filter(
|
||||
(agent) => !AGENT_TO_PHASE_TYPE[agent]
|
||||
);
|
||||
assert.deepStrictEqual(missing, [],
|
||||
`every agent in MODEL_PROFILES must have a phase-type — missing: ${JSON.stringify(missing)}`);
|
||||
});
|
||||
|
||||
test('every assigned phase-type is one of the six valid slots', () => {
|
||||
const invalid = Object.entries(AGENT_TO_PHASE_TYPE).filter(
|
||||
([, phaseType]) => !VALID_PHASE_TYPES.has(phaseType)
|
||||
);
|
||||
assert.deepStrictEqual(invalid, [],
|
||||
`phase-type assignments must use VALID_PHASE_TYPES — invalid: ${JSON.stringify(invalid)}`);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Resolver behavior: phase-type drives tier ──────────────────────────────
|
||||
|
||||
describe('#3023 resolver: models.<phase_type> overrides profile-based tier', () => {
|
||||
let projectDir;
|
||||
beforeEach(() => { projectDir = makeTmp('resolver'); });
|
||||
afterEach(() => { rmr(projectDir); });
|
||||
|
||||
test('phase-type alone — research agents get the phase-type tier, planner gets profile default', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: { research: 'haiku' },
|
||||
});
|
||||
// gsd-phase-researcher is a research agent — should pick up 'haiku'
|
||||
// from the phase-type slot, not 'sonnet' from the balanced profile.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'haiku');
|
||||
// gsd-codebase-mapper is also research → haiku
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
|
||||
// gsd-planner is planning, no models.planning set → falls through to
|
||||
// profile (balanced → opus per MODEL_PROFILES).
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
|
||||
});
|
||||
|
||||
test('per-agent override beats phase-type (acceptance criterion b)', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: { research: 'haiku' },
|
||||
model_overrides: { 'gsd-phase-researcher': 'opus' },
|
||||
});
|
||||
// The targeted per-agent override wins for that one agent.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'opus');
|
||||
// Other research agents still pick up the phase-type tier.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-research-synthesizer'), 'haiku');
|
||||
});
|
||||
|
||||
test('phase-type beats profile (acceptance criterion c)', () => {
|
||||
// model_profile=quality would normally make research agents 'opus'.
|
||||
// models.research='haiku' must win.
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'quality',
|
||||
models: { research: 'haiku' },
|
||||
});
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'haiku');
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
|
||||
// gsd-planner is planning, no slot set, profile=quality → opus.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
|
||||
});
|
||||
|
||||
test('issue example: opus for planning/discuss/execution, sonnet for research/verification/completion', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: {
|
||||
planning: 'opus',
|
||||
discuss: 'opus',
|
||||
execution: 'opus',
|
||||
research: 'sonnet',
|
||||
verification: 'sonnet',
|
||||
completion: 'sonnet',
|
||||
},
|
||||
});
|
||||
// Planning agents → opus
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
|
||||
// Execution agents → opus
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'opus');
|
||||
// Research agents → sonnet
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
|
||||
// Verification agents → sonnet
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-verifier'), 'sonnet');
|
||||
});
|
||||
|
||||
test('phase-type "inherit" is honored (preserves existing inherit semantics)', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: { research: 'inherit' },
|
||||
});
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'inherit');
|
||||
});
|
||||
|
||||
test('empty models block is a no-op (acceptance criterion: backward compat)', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: {},
|
||||
});
|
||||
// Behavior must match no-models config (balanced profile).
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
|
||||
});
|
||||
|
||||
test('no models block at all is a no-op (acceptance criterion: backward compat)', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
});
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
|
||||
});
|
||||
|
||||
test('unrecognized tier value falls through to profile (typo safety) — CR follow-up', () => {
|
||||
// The VALID_TIERS guard in resolveModelInternal must reject any value
|
||||
// that isn't a known tier alias and fall back to the profile tier.
|
||||
// Without this guard a typo like "haiku3" would pollute the runtime
|
||||
// resolution chain. Locks the guard in so a future regression that
|
||||
// removes it is caught.
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: { research: 'haiku3' }, // typo; not a valid tier alias
|
||||
});
|
||||
// Falls back to balanced → sonnet for research agents.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku',
|
||||
'gsd-codebase-mapper at balanced is haiku per profile, unaffected by typo');
|
||||
});
|
||||
|
||||
test('full model ID in models.<phase_type> is rejected; falls through to profile — CR follow-up', () => {
|
||||
// Full IDs are not valid in models.<phase_type>; they belong in
|
||||
// model_overrides per agent. The guard ensures we don't accidentally
|
||||
// hand a full ID into the runtime-tier resolution chain.
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: { research: 'openai/gpt-5' },
|
||||
});
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'sonnet');
|
||||
});
|
||||
|
||||
// ─── CR Major: phase-type beats inherit profile ─────────────────────────
|
||||
// Pre-fix bug: model_profile='inherit' + models.execution='opus' returned
|
||||
// 'inherit' because the profile short-circuit fired BEFORE the phase-type
|
||||
// override could win, violating the documented precedence where
|
||||
// models[phase_type] beats model_profile.
|
||||
|
||||
test('phase-type override wins over profile=inherit (CR Major) — model resolver', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'inherit',
|
||||
models: { execution: 'opus' },
|
||||
});
|
||||
// gsd-executor (execution) must get the phase-type opus, not inherit.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'opus');
|
||||
});
|
||||
|
||||
test('phase-type "haiku" wins over profile=inherit; agents without a slot still inherit', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'inherit',
|
||||
models: { research: 'haiku' },
|
||||
});
|
||||
// research agents → haiku (phase-type wins)
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'haiku');
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-codebase-mapper'), 'haiku');
|
||||
// planning agent has no slot set → falls through to profile=inherit.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-planner'), 'inherit');
|
||||
});
|
||||
|
||||
test('profile=inherit with no models block still returns inherit (no regression)', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'inherit',
|
||||
});
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'inherit');
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-phase-researcher'), 'inherit');
|
||||
});
|
||||
|
||||
test('profile=inherit with models block but agent has no slot → inherit', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'inherit',
|
||||
models: { research: 'haiku' },
|
||||
});
|
||||
// gsd-executor (execution slot) is not set → falls through to inherit.
|
||||
assert.equal(resolveModelInternal(projectDir, 'gsd-executor'), 'inherit');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── #443 Unified effort: resolveEffortInternal + renderEffortForRuntime ────
|
||||
|
||||
const { resolveEffortInternal } = require('../gsd-core/bin/lib/model-resolver.cjs');
|
||||
const { renderEffortForRuntime } = require('../gsd-core/bin/lib/model-catalog.cjs');
|
||||
|
||||
describe('#3023 + #443: unified effort resolver (resolveEffortInternal) for Codex', () => {
|
||||
let projectDir;
|
||||
beforeEach(() => { projectDir = makeTmp('effort'); });
|
||||
afterEach(() => { rmr(projectDir); });
|
||||
|
||||
test('resolveEffortInternal exported from model-resolver.cjs', () => {
|
||||
assert.equal(typeof resolveEffortInternal, 'function');
|
||||
});
|
||||
|
||||
test('effort derives from AGENT_DEFAULT_TIERS (routing), not phase-type; gsd-executor is standard → high', () => {
|
||||
// Under unification, effort is config-driven via routing_tier_defaults.
|
||||
// gsd-executor has routing tier 'standard' → default effort 'high', regardless
|
||||
// of models.execution phase-type or model_profile setting.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'codex',
|
||||
model_profile: 'balanced',
|
||||
models: { execution: 'opus' },
|
||||
});
|
||||
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
|
||||
// standard tier → 'high' (not 'xhigh' from opus, not 'medium' from old catalog)
|
||||
assert.equal(eff, 'high');
|
||||
const rendered = renderEffortForRuntime('codex', eff);
|
||||
assert.equal(rendered.param, 'model_reasoning_effort');
|
||||
assert.equal(rendered.value, 'high');
|
||||
});
|
||||
|
||||
test('effort resolves universally even when models.execution=inherit', () => {
|
||||
// Under unification, models.execution='inherit' does not affect effort resolution.
|
||||
// Effort always resolves from routing_tier_defaults: gsd-executor (standard) → 'high'.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'codex',
|
||||
model_profile: 'balanced',
|
||||
models: { execution: 'inherit' },
|
||||
});
|
||||
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
|
||||
assert.equal(eff, 'high');
|
||||
const rendered = renderEffortForRuntime('codex', eff);
|
||||
assert.equal(rendered.param, 'model_reasoning_effort');
|
||||
assert.equal(rendered.value, 'high');
|
||||
});
|
||||
|
||||
test('per-agent model_overrides does not affect effort (effort is routing-tier-based)', () => {
|
||||
// Under unification, effort does not check model_overrides.
|
||||
// gsd-executor (standard tier) → 'high' regardless.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'codex',
|
||||
model_profile: 'balanced',
|
||||
models: { execution: 'opus' },
|
||||
model_overrides: { 'gsd-executor': 'openai/gpt-5' },
|
||||
});
|
||||
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
|
||||
assert.equal(eff, 'high');
|
||||
const rendered = renderEffortForRuntime('codex', eff);
|
||||
assert.equal(rendered.param, 'model_reasoning_effort');
|
||||
assert.equal(rendered.value, 'high');
|
||||
});
|
||||
|
||||
test('Claude runtime: effort is first-class (emits output_config.effort, not null)', () => {
|
||||
// Under unification, Claude effort is first-class via output_config.effort.
|
||||
// No `runtime` set → defaults to claude (no runtime key → undefined runtime).
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
models: { execution: 'opus' },
|
||||
});
|
||||
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
|
||||
// effort resolves universally; claude render gives output_config.effort
|
||||
const rendered = renderEffortForRuntime(undefined, eff);
|
||||
// undefined runtime yields param=null (no runtime key set)
|
||||
assert.equal(rendered.param, null);
|
||||
// But if explicitly set to 'claude':
|
||||
const renderedClaude = renderEffortForRuntime('claude', eff);
|
||||
assert.equal(renderedClaude.param, 'output_config.effort');
|
||||
assert.equal(renderedClaude.value, 'high');
|
||||
});
|
||||
|
||||
test('profile=inherit does not affect effort; effort resolves from routing tier', () => {
|
||||
// Under unification, effort is completely independent of model_profile.
|
||||
// gsd-executor (standard routing tier) → 'high' even with model_profile='inherit'.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'codex',
|
||||
model_profile: 'inherit',
|
||||
models: { execution: 'opus' },
|
||||
});
|
||||
const eff = resolveEffortInternal(projectDir, 'gsd-executor');
|
||||
assert.equal(eff, 'high',
|
||||
'profile=inherit must not affect effort; standard routing tier → high');
|
||||
const rendered = renderEffortForRuntime('codex', eff);
|
||||
assert.equal(rendered.param, 'model_reasoning_effort');
|
||||
assert.equal(rendered.value, 'high');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Schema validation ──────────────────────────────────────────────────────
|
||||
|
||||
describe('#3023 config-schema: models.<phase_type> validation', () => {
|
||||
test('models.planning is a valid config key', () => {
|
||||
assert.equal(isValidConfigKey('models.planning'), true);
|
||||
});
|
||||
|
||||
test('all six phase-type slots are valid config keys', () => {
|
||||
for (const slot of ['planning', 'discuss', 'research', 'execution', 'verification', 'completion']) {
|
||||
assert.equal(isValidConfigKey(`models.${slot}`), true,
|
||||
`models.${slot} must be a valid config key`);
|
||||
}
|
||||
});
|
||||
|
||||
test('unknown phase-type is rejected (acceptance criterion d)', () => {
|
||||
assert.equal(isValidConfigKey('models.deployment'), false,
|
||||
'unknown phase-type must NOT be accepted');
|
||||
assert.equal(isValidConfigKey('models.gsd-planner'), false,
|
||||
'agent name in models.* must NOT be accepted (use model_overrides for agents)');
|
||||
});
|
||||
|
||||
test('models alone (without a slot) is not a valid config-set key', () => {
|
||||
// Setting the whole block isn't a granular set; users edit JSON directly.
|
||||
assert.equal(isValidConfigKey('models'), false);
|
||||
});
|
||||
});
|
||||
@@ -1,301 +0,0 @@
|
||||
/**
|
||||
* Documentation regression test for issue #3025 — MCP token-budget guidance.
|
||||
*
|
||||
* Verifies that gsd-core/references/context-budget.md contains the
|
||||
* structural elements the issue requires:
|
||||
*
|
||||
* 1. A section explaining MCP/tool schemas as a context-budget concern
|
||||
* 2. References to the harness-side toggles (enabledMcpjsonServers,
|
||||
* disabledMcpjsonServers in .claude/settings.json)
|
||||
* 3. A pre-phase audit checklist (browser/playwright, platform-specific,
|
||||
* project-specific)
|
||||
* 4. An explicit note that GSD does NOT manage MCP enablement — this is
|
||||
* a Claude Code harness concern (with a cross-link)
|
||||
* 5. Note the interaction with model_profile (compounding levers)
|
||||
*
|
||||
* Tests parse the doc into a typed section record (parseMcpSection) and
|
||||
* assert on flag booleans, not raw text matches. Adheres to
|
||||
* CONTRIBUTING.md "no-source-grep" — describes invariants, not wording,
|
||||
* so the prose can be reworded freely as long as the semantics survive.
|
||||
*
|
||||
* Companion to docs/USER-GUIDE.md task section, which is exercised by the
|
||||
* same parser shape (separate test below).
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const CONTEXT_BUDGET_MD = path.join(ROOT, 'gsd-core', 'references', 'context-budget.md');
|
||||
const USER_GUIDE_MD = path.join(ROOT, 'docs', 'USER-GUIDE.md');
|
||||
|
||||
/**
|
||||
* Extract the MCP-budget section from a markdown file by header text.
|
||||
* Returns null if the section is missing. Section runs from the matching
|
||||
* `## ` header up to the next `## ` header (or EOF).
|
||||
*/
|
||||
function extractSection(filePath, headerSubstring) {
|
||||
const content = fs.readFileSync(filePath, 'utf8');
|
||||
const lines = content.split(/\r?\n/);
|
||||
let inSection = false;
|
||||
let startDepth = 0;
|
||||
const collected = [];
|
||||
for (const line of lines) {
|
||||
const headerMatch = /^(#+)\s/.exec(line);
|
||||
if (headerMatch) {
|
||||
const depth = headerMatch[1].length;
|
||||
if (inSection) {
|
||||
// Section ends at a header at the same or shallower depth.
|
||||
// Subsections at deeper depth are part of the section.
|
||||
if (depth <= startDepth) break;
|
||||
} else if (line.toLowerCase().includes(headerSubstring.toLowerCase())) {
|
||||
inSection = true;
|
||||
startDepth = depth;
|
||||
}
|
||||
}
|
||||
if (inSection) collected.push(line);
|
||||
}
|
||||
return collected.length > 0 ? collected.join('\n') : null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the MCP-budget section into a typed semantic-flag record.
|
||||
* Each flag answers a single behavioral question that #3025 requires
|
||||
* the prose to encode.
|
||||
*/
|
||||
function parseMcpBudgetSection(section) {
|
||||
if (!section || typeof section !== 'string') {
|
||||
return {
|
||||
ok: false,
|
||||
sectionLength: 0,
|
||||
explainsMcpAsBudgetConcern: false,
|
||||
namesEnabledMcpjsonServers: false,
|
||||
namesDisabledMcpjsonServers: false,
|
||||
namesClaudeSettingsJson: false,
|
||||
includesPrePhaseAudit: false,
|
||||
auditMentionsBrowserOrPlaywright: false,
|
||||
auditMentionsPlatformSpecific: false,
|
||||
auditMentionsCrossProject: false,
|
||||
explainsHarnessNotGsd: false,
|
||||
mentionsModelProfileInteraction: false,
|
||||
crossLinksContextBudget: false,
|
||||
};
|
||||
}
|
||||
// CR follow-up: strip inline markdown emphasis (`**`, `*`, `~~`) and
|
||||
// backticks before phrase-matching so e.g. "GSD does **not** manage"
|
||||
// is caught by the primary `gsd does not manage` alternative below.
|
||||
// WITHOUT this, the markdown-bold breaks the contiguous match and the
|
||||
// test only passes via the fallback branch (silent dead code).
|
||||
// Underscores are intentionally NOT stripped — `model_profile` and
|
||||
// other snake_case identifiers must survive intact so the
|
||||
// model_profile interaction check still finds them.
|
||||
const stripped = section.replace(/\*{1,3}|~{2}|`/g, '');
|
||||
// (1) Explains MCP as budget concern — must mention BOTH "MCP" / "tool
|
||||
// schema" AND a token/cost framing.
|
||||
const explainsMcpAsBudgetConcern =
|
||||
/\bmcp\b|tool schema|tool schemas/i.test(stripped) &&
|
||||
/\btoken|context budget|per[- ]turn|cost\b/i.test(stripped);
|
||||
// (2) Names the harness keys verbatim
|
||||
const namesEnabledMcpjsonServers = /enabledMcpjsonServers/.test(stripped);
|
||||
const namesDisabledMcpjsonServers = /disabledMcpjsonServers/.test(stripped);
|
||||
// (3) Names the settings file location
|
||||
const namesClaudeSettingsJson = /\.claude\/settings\.json/.test(stripped);
|
||||
// (4) Audit checklist — must mention all three classes the issue
|
||||
// calls out, plus a "before this phase / pre-phase" framing
|
||||
const includesPrePhaseAudit =
|
||||
/audit|checklist|review (your )?mcp|before (starting|beginning) (a |the )?phase/i.test(stripped);
|
||||
const auditMentionsBrowserOrPlaywright = /\bbrowser\b|playwright/i.test(stripped);
|
||||
const auditMentionsPlatformSpecific = /platform[- ]specific|mac[- ]?tools|windows[- ]?tools|os[- ]specific/i.test(stripped);
|
||||
const auditMentionsCrossProject = /(other|different|cross[- ])\s*project|stale (project )?mcp/i.test(stripped);
|
||||
// (5) Harness vs GSD distinction — must explicitly state GSD doesn't
|
||||
// own this knob and point at the harness
|
||||
const explainsHarnessNotGsd =
|
||||
/(gsd does(?:n[''’]t| not) (own|manage|control)|harness (concern|setting|controlled)|not a gsd (setting|knob))/i.test(stripped);
|
||||
// (6) Compounding with model_profile
|
||||
const mentionsModelProfileInteraction =
|
||||
/model[_ ]profile/i.test(stripped) &&
|
||||
/compound|multiplier|stack|every[- ]turn|regardless of (which )?model|in addition/i.test(stripped);
|
||||
// (7) Cross-link to the canonical reference doc — task-guide section
|
||||
// must point readers at context-budget.md for the full audit. Encoded
|
||||
// as a named flag (CR follow-up) so the assertion sits alongside the
|
||||
// other parsed invariants rather than as a one-off inline regex.
|
||||
const crossLinksContextBudget = /context-budget/i.test(stripped);
|
||||
return {
|
||||
ok: true,
|
||||
sectionLength: section.length,
|
||||
explainsMcpAsBudgetConcern,
|
||||
namesEnabledMcpjsonServers,
|
||||
namesDisabledMcpjsonServers,
|
||||
namesClaudeSettingsJson,
|
||||
includesPrePhaseAudit,
|
||||
auditMentionsBrowserOrPlaywright,
|
||||
auditMentionsPlatformSpecific,
|
||||
auditMentionsCrossProject,
|
||||
explainsHarnessNotGsd,
|
||||
mentionsModelProfileInteraction,
|
||||
crossLinksContextBudget,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── context-budget.md ──────────────────────────────────────────────────────
|
||||
|
||||
describe('#3025 context-budget.md: MCP token-budget section exists with required content', () => {
|
||||
test('the file exists', () => {
|
||||
assert.ok(fs.existsSync(CONTEXT_BUDGET_MD), `expected file at ${CONTEXT_BUDGET_MD}`);
|
||||
});
|
||||
|
||||
test('has a section header that mentions MCP', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
assert.ok(section, 'must have a `## ...MCP...` heading; section was not found');
|
||||
});
|
||||
|
||||
test('explains MCP/tool schemas as a context-budget concern (#3025 requirement 1)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.explainsMcpAsBudgetConcern, true,
|
||||
`must explain MCP/tool schemas as a token/context-budget concern; section was:\n${section}`);
|
||||
});
|
||||
|
||||
test('names enabledMcpjsonServers and disabledMcpjsonServers (#3025 requirement 2)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.namesEnabledMcpjsonServers, true,
|
||||
'section must reference `enabledMcpjsonServers` so users know the exact key');
|
||||
assert.equal(parsed.namesDisabledMcpjsonServers, true,
|
||||
'section must reference `disabledMcpjsonServers` for parity');
|
||||
assert.equal(parsed.namesClaudeSettingsJson, true,
|
||||
'section must name `.claude/settings.json` as the location of the toggle');
|
||||
});
|
||||
|
||||
test('includes a pre-phase audit checklist with all three classes (#3025 requirement 3)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.includesPrePhaseAudit, true,
|
||||
'section must include audit/checklist framing for pre-phase MCP review');
|
||||
assert.equal(parsed.auditMentionsBrowserOrPlaywright, true,
|
||||
'audit must mention browser/playwright tools as a candidate for disabling');
|
||||
assert.equal(parsed.auditMentionsPlatformSpecific, true,
|
||||
'audit must mention platform-specific tools (Mac/Windows/OS-specific)');
|
||||
assert.equal(parsed.auditMentionsCrossProject, true,
|
||||
'audit must mention stale/cross-project MCPs from other projects');
|
||||
});
|
||||
|
||||
test('explains GSD does not own MCP enablement — harness concern (#3025 requirement 4)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.explainsHarnessNotGsd, true,
|
||||
'section must explicitly state GSD does not manage MCP enablement (harness concern)');
|
||||
});
|
||||
|
||||
test('notes interaction with model_profile (compounding levers) (#3025 requirement 5)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.mentionsModelProfileInteraction, true,
|
||||
'section must note that trimming MCPs compounds with model_profile choice');
|
||||
});
|
||||
|
||||
test('full semantic record matches the #3025 contract — typed snapshot', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
const contract = {
|
||||
ok: parsed.ok,
|
||||
explainsMcpAsBudgetConcern: parsed.explainsMcpAsBudgetConcern,
|
||||
namesEnabledMcpjsonServers: parsed.namesEnabledMcpjsonServers,
|
||||
namesDisabledMcpjsonServers: parsed.namesDisabledMcpjsonServers,
|
||||
namesClaudeSettingsJson: parsed.namesClaudeSettingsJson,
|
||||
includesPrePhaseAudit: parsed.includesPrePhaseAudit,
|
||||
auditMentionsBrowserOrPlaywright: parsed.auditMentionsBrowserOrPlaywright,
|
||||
auditMentionsPlatformSpecific: parsed.auditMentionsPlatformSpecific,
|
||||
auditMentionsCrossProject: parsed.auditMentionsCrossProject,
|
||||
explainsHarnessNotGsd: parsed.explainsHarnessNotGsd,
|
||||
mentionsModelProfileInteraction: parsed.mentionsModelProfileInteraction,
|
||||
};
|
||||
assert.deepStrictEqual(contract, {
|
||||
ok: true,
|
||||
explainsMcpAsBudgetConcern: true,
|
||||
namesEnabledMcpjsonServers: true,
|
||||
namesDisabledMcpjsonServers: true,
|
||||
namesClaudeSettingsJson: true,
|
||||
includesPrePhaseAudit: true,
|
||||
auditMentionsBrowserOrPlaywright: true,
|
||||
auditMentionsPlatformSpecific: true,
|
||||
auditMentionsCrossProject: true,
|
||||
explainsHarnessNotGsd: true,
|
||||
mentionsModelProfileInteraction: true,
|
||||
}, 'context-budget.md MCP section contract violated');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── docs/USER-GUIDE.md task section ────────────────────────────────────────
|
||||
|
||||
describe('#3025 docs/USER-GUIDE.md: companion task section exists', () => {
|
||||
test('USER-GUIDE.md has an MCP-trimming task section', () => {
|
||||
const section = extractSection(USER_GUIDE_MD, 'mcp');
|
||||
assert.ok(section,
|
||||
'USER-GUIDE.md must have a `### ...MCP...` task section so users find it via the guide');
|
||||
});
|
||||
|
||||
test('USER-GUIDE.md task section names the harness key and cross-links the reference', () => {
|
||||
const section = extractSection(USER_GUIDE_MD, 'mcp');
|
||||
const parsed = parseMcpBudgetSection(section);
|
||||
assert.equal(parsed.namesEnabledMcpjsonServers, true,
|
||||
'task section must mention the harness key by name');
|
||||
// Cross-link to the reference doc — assert on the parsed flag so
|
||||
// the invariant lives alongside the other named flags (CR follow-up
|
||||
// on the no-source-grep standard).
|
||||
assert.equal(parsed.crossLinksContextBudget, true,
|
||||
'task section must cross-link to context-budget.md');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── markdownlint pre-flight (per bundle-docs-with-code skill) ──────────────
|
||||
|
||||
describe('#3025 markdownlint pre-flight: MD040 + MD056', () => {
|
||||
test('every fenced code block in the new MCP section has a language tag (MD040)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
// Guard: extractSection returns null when the section is missing.
|
||||
// Without this, `section.match(...)` would throw a TypeError instead
|
||||
// of producing a meaningful assertion failure (CR follow-up).
|
||||
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD040');
|
||||
const fences = (section.match(/^```([a-zA-Z0-9_+-]*)?\s*$/gm) || []);
|
||||
// Pairs of fences open/close; odd-indexed ones close blocks. Every
|
||||
// OPENING fence must have a language tag. Closing fences are bare ```.
|
||||
// Walk pairs: even index = opener, odd = closer.
|
||||
const openers = fences.filter((_, i) => i % 2 === 0);
|
||||
const missing = openers.filter((line) => /^```\s*$/.test(line));
|
||||
assert.deepStrictEqual(missing, [],
|
||||
`every fenced code block opener must have a language tag (MD040). Missing: ${JSON.stringify(missing)}`);
|
||||
});
|
||||
|
||||
test('every markdown table row in the new MCP section has the same column count as its header (MD056)', () => {
|
||||
const section = extractSection(CONTEXT_BUDGET_MD, 'mcp');
|
||||
// Guard: same null-section concern as MD040 above (CR follow-up).
|
||||
assert.ok(section, 'MCP section not found in context-budget.md — cannot check MD056');
|
||||
const lines = section.split(/\r?\n/);
|
||||
// Walk through and detect tables: header row followed by a separator
|
||||
// (--- pattern) followed by data rows. Count `|` per line.
|
||||
const issues = [];
|
||||
for (let i = 0; i < lines.length - 1; i += 1) {
|
||||
const header = lines[i];
|
||||
const sep = lines[i + 1];
|
||||
if (!/^\s*\|.*\|\s*$/.test(header)) continue;
|
||||
if (!/^\s*\|[\s\-:|]+\|\s*$/.test(sep)) continue;
|
||||
const headerCols = (header.match(/\|/g) || []).length;
|
||||
// Walk data rows
|
||||
for (let j = i + 2; j < lines.length; j += 1) {
|
||||
const row = lines[j];
|
||||
if (!/^\s*\|.*\|\s*$/.test(row)) break;
|
||||
const rowCols = (row.match(/\|/g) || []).length;
|
||||
if (rowCols !== headerCols) {
|
||||
issues.push({ line: j, expected: headerCols, actual: rowCols, row });
|
||||
}
|
||||
}
|
||||
}
|
||||
assert.deepStrictEqual(issues, [],
|
||||
`table rows must match header column count (MD056). Issues: ${JSON.stringify(issues, null, 2)}`);
|
||||
});
|
||||
});
|
||||
@@ -1,130 +0,0 @@
|
||||
/**
|
||||
* Universal CLI negative-matrix sweep across the seven command families
|
||||
* named in #3593: phase, roadmap, state, config, workstream, init,
|
||||
* validate.
|
||||
*
|
||||
* The full per-case matrix for each family belongs in dedicated files
|
||||
* (the `config` family is the template — see
|
||||
* `feat-3593-cli-negative-config.test.cjs`). This file pins a much
|
||||
* narrower contract that EVERY family must satisfy:
|
||||
*
|
||||
* 1. Bare top-level command with no subcommand: must not crash with
|
||||
* a V8 stack trace, must exit non-zero (or, where the bare form
|
||||
* is legitimate, return a clean payload).
|
||||
* 2. Unknown subcommand at command depth: must emit a typed reason
|
||||
* under --json-errors and must not crash.
|
||||
* 3. Shell-metacharacter as an argv element: must not be executed.
|
||||
* Sentinel-file probe in the project temp dir proves this.
|
||||
*
|
||||
* Future per-family files will deepen each family's matrix; this file
|
||||
* is the floor.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const { runCli } = require('./helpers/cli-negative.cjs');
|
||||
const { createTempProject, createTempGitProject, cleanup } = require('./helpers.cjs');
|
||||
|
||||
/**
|
||||
* Each entry names a top-level command, a representative subcommand
|
||||
* (for the "unknown sub" probe), and the temp-project factory.
|
||||
*
|
||||
* The `representativeSubcommand` is not asserted on directly — it
|
||||
* exists so the test layout reads "for each family, run probe X" and
|
||||
* so a future maintainer adding a family doesn't need to invent one.
|
||||
*/
|
||||
const FAMILIES = [
|
||||
{ name: 'phase', bare: ['phase'], unknown: ['phase', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'roadmap', bare: ['roadmap'], unknown: ['roadmap', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'state', bare: ['state'], unknown: ['state', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'config', bare: ['config'], unknown: ['config', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'workstream', bare: ['workstream'], unknown: ['workstream', 'this-sub-does-not-exist'], projectFactory: createTempGitProject },
|
||||
{ name: 'init', bare: ['init'], unknown: ['init', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'validate', bare: ['validate'], unknown: ['validate', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
];
|
||||
|
||||
describe('feat-3593: bare top-level command does not crash', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
test(`${fam.name}: bare invocation exits cleanly without a stack trace`, (t) => {
|
||||
const projectDir = fam.projectFactory(`cli-neg-univ-bare-${fam.name}-`);
|
||||
t.after(() => cleanup(projectDir));
|
||||
const result = runCli(fam.bare, { cwd: projectDir });
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name} bare must not leak a V8 stack frame`);
|
||||
assert.equal(result.signal, null, `${fam.name} bare must not be killed by a signal`);
|
||||
// We do not pin exit code here: some families legitimately treat the
|
||||
// bare form as a list/status command (status 0); others reject it as
|
||||
// a usage error (status ≠ 0). What we pin is "no crash."
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe('feat-3593: unknown subcommand emits a typed reason', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
test(`${fam.name}: unknown subcommand fails with reason set and no stack trace`, (t) => {
|
||||
const projectDir = fam.projectFactory(`cli-neg-univ-unk-${fam.name}-`);
|
||||
t.after(() => cleanup(projectDir));
|
||||
const result = runCli(fam.unknown, { cwd: projectDir });
|
||||
assert.notEqual(result.status, 0, `${fam.name} unknown sub must exit non-zero`);
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name} unknown sub must not leak a stack frame`);
|
||||
// When --json-errors is on, every typed-failure path lands a non-empty
|
||||
// reason string from ERROR_REASON. A null reason here means the family
|
||||
// is using throw/console.error somewhere — a TDD signal to wire that
|
||||
// failure path through error(msg, ERROR_REASON.X).
|
||||
assert.equal(typeof result.reason, 'string', `${fam.name}: reason must be a typed enum string (got: ${result.reason})`);
|
||||
assert.notEqual(result.reason, '', `${fam.name}: reason must be non-empty`);
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe('feat-3593: shell-metacharacter argv values are NOT executed', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
test(`${fam.name}: shell-payload as subcommand argv does NOT execute the payload`, (t) => {
|
||||
const projectDir = fam.projectFactory(`cli-neg-univ-shell-${fam.name}-`);
|
||||
t.after(() => cleanup(projectDir));
|
||||
// Place the payload where the subcommand value goes. The payload
|
||||
// would, if shell-interpreted, create a sentinel file in the
|
||||
// project dir. Argv-based invocation must treat it as opaque text.
|
||||
const sentinelPayload = `$(touch ${projectDir}/INJ-${fam.name})`;
|
||||
const argv = [fam.name, sentinelPayload];
|
||||
const result = runCli(argv, { cwd: projectDir });
|
||||
// No stack trace — opaque text must not crash.
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name}: shell payload must not crash`);
|
||||
// The sentinel file must NOT exist. We check the project dir's
|
||||
// listing rather than fs.existsSync of a single path so the test
|
||||
// surfaces any spelling drift.
|
||||
const entries = fs.readdirSync(projectDir);
|
||||
const sentinels = entries.filter((n) => n.startsWith('INJ-'));
|
||||
assert.deepEqual(
|
||||
sentinels,
|
||||
[],
|
||||
`${fam.name}: shell payload was executed — sentinel files exist: ${sentinels.join(', ')}`,
|
||||
);
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// ─── Cross-family invariants on the global --cwd flag ──────────────────────
|
||||
|
||||
test('--cwd with an empty value fails the same way regardless of command family', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
const result = runCli(['--cwd', '', ...fam.bare], { cwd: process.cwd() });
|
||||
assert.notEqual(result.status, 0, `${fam.name}: empty --cwd must fail`);
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name}: empty --cwd must not crash`);
|
||||
assert.equal(result.reason, 'usage', `${fam.name}: empty --cwd reason should be 'usage', got: ${result.reason}`);
|
||||
}
|
||||
});
|
||||
|
||||
test('--cwd pointing at a non-existent path fails uniformly across families', () => {
|
||||
const nonExistent = path.join(require('os').tmpdir(), 'cli-neg-univ-no-such-' + Date.now() + '-' + Math.random());
|
||||
assert.equal(fs.existsSync(nonExistent), false, 'pre-check: temp path must not exist');
|
||||
for (const fam of FAMILIES) {
|
||||
const result = runCli(['--cwd', nonExistent, ...fam.bare], { cwd: process.cwd() });
|
||||
assert.notEqual(result.status, 0, `${fam.name}: invalid --cwd must fail`);
|
||||
assert.equal(result.hasStackTrace, false);
|
||||
assert.equal(result.reason, 'usage', `${fam.name}: invalid --cwd reason should be 'usage'`);
|
||||
}
|
||||
});
|
||||
@@ -1,275 +0,0 @@
|
||||
/**
|
||||
* Adversarial roadmap-parser tests (#3594).
|
||||
*
|
||||
* Loads each fixture in `tests/fixtures/adversarial/roadmap/` as the
|
||||
* project's `.planning/ROADMAP.md` and pins invariants on the public
|
||||
* `gsd-tools roadmap get-phase <N>` surface — which routes through the
|
||||
* SDK bridge when available and the CJS handler otherwise.
|
||||
*
|
||||
* Per CONTRIBUTING.md §"Testing Standards / Parser and project-file
|
||||
* inputs", the assertion target is the typed JSON shape the CLI emits,
|
||||
* not stderr prose. The harness in `tests/helpers/cli-negative.cjs`
|
||||
* (introduced by #3627 / #3593) is reused here for consistency.
|
||||
*
|
||||
* Several fixtures encode known historical regressions:
|
||||
* - fenced-code-block headings shadowing real phases (#2787)
|
||||
* - decimal phase prefix collisions (#3537)
|
||||
* - HTML-comment heading false positives
|
||||
*
|
||||
* Pre-existing parser bugs surfaced by these fixtures are NOT fixed in
|
||||
* this PR — fixing them is out of scope for "add adversarial test
|
||||
* coverage." Where a fixture exposes a still-open bug, the test
|
||||
* asserts the *currently observed* behavior with a comment naming the
|
||||
* open issue, so the flip from RED→GREEN is a one-line change the day
|
||||
* the real fix lands.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const { runCli } = require('./helpers/cli-negative.cjs');
|
||||
const { createTempProject, cleanup } = require('./helpers.cjs');
|
||||
|
||||
const FIXTURE_DIR = path.join(__dirname, 'fixtures', 'adversarial', 'roadmap');
|
||||
|
||||
function loadFixture(name) {
|
||||
return fs.readFileSync(path.join(FIXTURE_DIR, name), 'utf-8');
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a temp project whose ROADMAP.md is the named fixture's content.
|
||||
* Returns the project directory; caller is responsible for cleanup.
|
||||
*/
|
||||
function projectWithFixture(t, fixtureName) {
|
||||
const projectDir = createTempProject('roadmap-adv-' + fixtureName.replace(/\W+/g, '-') + '-');
|
||||
t.after(() => cleanup(projectDir));
|
||||
fs.writeFileSync(path.join(projectDir, '.planning', 'ROADMAP.md'), loadFixture(fixtureName));
|
||||
return projectDir;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run `gsd-tools roadmap get-phase <N>` and parse the JSON payload.
|
||||
* Returns `{ ok, exit, parsed, raw }` so tests can assert on either
|
||||
* the exit code or the structured payload.
|
||||
*/
|
||||
function getPhase(projectDir, phaseNum) {
|
||||
// No --json-errors — the get-phase command outputs JSON on success
|
||||
// via the normal stdout path. Reading the parsed payload is what the
|
||||
// workflows downstream do, so that's what we test.
|
||||
const result = runCli(['roadmap', 'get-phase', phaseNum], { cwd: projectDir, jsonErrors: false });
|
||||
let parsed = null;
|
||||
try {
|
||||
parsed = JSON.parse(result.stdout);
|
||||
} catch {
|
||||
// Leave parsed null; tests that depend on it must handle that.
|
||||
}
|
||||
return {
|
||||
exit: result.status,
|
||||
ok: result.status === 0,
|
||||
parsed,
|
||||
raw: result.stdout,
|
||||
stderr: result.stderr,
|
||||
hasStackTrace: result.hasStackTrace,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Fenced code block heading shadowing ────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser and fenced-code-block headings (#2787)', () => {
|
||||
test('phase 1 in real prose is found even when ## Phase 999 appears inside a ``` block', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.equal(result.hasStackTrace, false, 'no V8 stack trace');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true, 'phase 1 must be found');
|
||||
assert.equal(result.parsed.phase_number, '1');
|
||||
assert.match(result.parsed.phase_name, /real phase one/);
|
||||
});
|
||||
|
||||
test('phase 999 inside a fenced block is ignored', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
|
||||
const result = getPhase(projectDir, '999');
|
||||
assert.equal(result.hasStackTrace, false, 'no stack trace');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, false, 'phase headings inside fenced blocks must not be parsed');
|
||||
});
|
||||
|
||||
test('fenced example heading does not shadow the real phase details and backlog phase stays unresolved (#1588)', (t) => {
|
||||
const projectDir = createTempProject('roadmap-1588-');
|
||||
t.after(() => cleanup(projectDir));
|
||||
fs.writeFileSync(
|
||||
path.join(projectDir, '.planning', 'STATE.md'),
|
||||
[
|
||||
'---',
|
||||
'gsd_state_version: 1.0',
|
||||
'milestone: v1.1',
|
||||
'status: planning',
|
||||
'---',
|
||||
'',
|
||||
].join('\n')
|
||||
);
|
||||
fs.writeFileSync(
|
||||
path.join(projectDir, '.planning', 'ROADMAP.md'),
|
||||
[
|
||||
'# Roadmap',
|
||||
'',
|
||||
'<details open>',
|
||||
'<summary>v1.1 Current (Phases 8-9) - PLANNED</summary>',
|
||||
'',
|
||||
'- [ ] **Phase 9: Real Phase**',
|
||||
'',
|
||||
'</details>',
|
||||
'',
|
||||
'## Phase Details',
|
||||
'',
|
||||
'```markdown',
|
||||
'### Phase 9: Fenced Example Phase',
|
||||
'**Goal:** This example must not be treated as roadmap structure.',
|
||||
'```',
|
||||
'',
|
||||
'### Phase 9: Real Phase',
|
||||
'**Goal:** Use the real phase details outside the fenced block.',
|
||||
'**Requirements:** REAL-01',
|
||||
'',
|
||||
'## Backlog',
|
||||
'',
|
||||
'### Phase 999.1: Backlog Thing',
|
||||
'**Goal:** Future backlog item.',
|
||||
'',
|
||||
].join('\n')
|
||||
);
|
||||
|
||||
const phase9 = getPhase(projectDir, '9');
|
||||
assert.ok(phase9.parsed, `expected JSON payload, got: ${phase9.raw}`);
|
||||
assert.equal(phase9.parsed.found, true, 'phase 9 must be found');
|
||||
assert.equal(phase9.parsed.phase_name, 'Real Phase');
|
||||
assert.equal(phase9.parsed.goal, 'Use the real phase details outside the fenced block.');
|
||||
assert.match(phase9.parsed.section, /REAL-01/, 'real phase section must be returned');
|
||||
|
||||
const backlog = getPhase(projectDir, '999.1');
|
||||
assert.ok(backlog.parsed, `expected JSON payload, got: ${backlog.raw}`);
|
||||
assert.equal(backlog.parsed.found, false, 'backlog sentinel phase must not resolve as an active roadmap phase');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Decimal phase prefix collisions ────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser handles decimal phase prefix collisions (#3537)', () => {
|
||||
test('asking for phase "2" returns the integer phase, NOT phase 2.1 or 2.10', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '2');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '2');
|
||||
assert.match(result.parsed.phase_name, /integer phase two/);
|
||||
});
|
||||
|
||||
test('asking for phase "2.1" returns the decimal child', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '2.1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '2.1');
|
||||
assert.match(result.parsed.phase_name, /decimal child/);
|
||||
});
|
||||
|
||||
test('asking for phase "2.10" returns the decimal sibling, NOT phase 2.1', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '2.10');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '2.10');
|
||||
assert.match(result.parsed.phase_name, /decimal phase 2\.10/);
|
||||
});
|
||||
|
||||
test('asking for phase "21" returns phase 21, NOT phase 2 (prefix-collision guard)', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '21');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '21');
|
||||
assert.match(result.parsed.phase_name, /phase twenty-one/);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Unicode phase titles ───────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser preserves Unicode phase titles', () => {
|
||||
test('Japanese title round-trips through phase_name', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.phase_name, '日本語フェーズ — initial setup');
|
||||
});
|
||||
|
||||
test('emoji + smart-quote title survives', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
|
||||
const result = getPhase(projectDir, '2');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.match(result.parsed.phase_name, /🚧/);
|
||||
assert.match(result.parsed.phase_name, /Émile/);
|
||||
});
|
||||
|
||||
test('Greek-letter title survives', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
|
||||
const result = getPhase(projectDir, '3');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.phase_name, 'αβγ δεζ ηθι');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Repeated phase IDs ─────────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser handles repeated phase IDs deterministically', () => {
|
||||
test('two declarations of phase 1: parser returns the FIRST match (current behavior)', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'repeated-phase-ids.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
// The regex uses `content.match(...)` which returns the FIRST match.
|
||||
// Pin that — a future change to last-wins or de-dup would fire.
|
||||
assert.match(result.parsed.phase_name, /first declaration/);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── HTML comments ──────────────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser and HTML-commented headings', () => {
|
||||
test('phase 1 in real prose is found even when phase 998/999 appear inside <!-- ... -->', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_name, 'real phase');
|
||||
});
|
||||
|
||||
test('phase 999 inside an HTML comment remains ignored because backlog sentinels never resolve', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
|
||||
const result = getPhase(projectDir, '999');
|
||||
assert.equal(result.hasStackTrace, false, 'no stack trace');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, false, 'backlog sentinel phases must not resolve');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Cross-corpus invariant ────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser does not crash on ANY corpus fixture', () => {
|
||||
const fixtures = fs.readdirSync(FIXTURE_DIR).filter((f) => f.endsWith('.md') && f !== 'README.md');
|
||||
for (const fixture of fixtures) {
|
||||
test(`fixture "${fixture}" — get-phase with arbitrary IDs must not crash`, (t) => {
|
||||
const projectDir = projectWithFixture(t, fixture);
|
||||
for (const id of ['1', '2', '99', '999', '0', '2.1']) {
|
||||
const result = getPhase(projectDir, id);
|
||||
assert.equal(result.hasStackTrace, false, `${fixture} id=${id}: no V8 stack frame allowed`);
|
||||
// exit status varies (0 for found, non-zero for not-found —
|
||||
// both are valid). What's pinned: the parser produced SOME output
|
||||
// (either valid JSON or a clean stderr) without crashing.
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
@@ -1,98 +0,0 @@
|
||||
'use strict';
|
||||
|
||||
// feat(#41): /gsd-ship generate_pr_body emits a TDD Audit table + an aggregate
|
||||
// `gate_status:` trailer so the per-commit TDD gate trail survives squash-merge.
|
||||
// These assertions pin the shipped workflow prose in gsd-core/workflows/ship.md.
|
||||
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const assert = require('node:assert/strict');
|
||||
const { describe, test } = require('node:test');
|
||||
|
||||
const repoRoot = path.resolve(__dirname, '..');
|
||||
function readRepoFile(relativePath) {
|
||||
return fs.readFileSync(path.join(repoRoot, relativePath), 'utf8');
|
||||
}
|
||||
|
||||
describe('feat-41: ship.md TDD Audit gate_status extraction', () => {
|
||||
const workflow = readRepoFile('gsd-core/workflows/ship.md');
|
||||
|
||||
test('adds a "## TDD Audit" section to the generated PR body', () => {
|
||||
assert.match(workflow, /## TDD Audit/);
|
||||
});
|
||||
|
||||
test('extracts gate_status via Git native trailer machinery, not a raw body grep', () => {
|
||||
assert.match(workflow, /trailers:key=gate_status/);
|
||||
});
|
||||
|
||||
test('scopes the scan to the merge-base..HEAD range', () => {
|
||||
assert.match(workflow, /merge-base/);
|
||||
assert.match(workflow, /\.\.HEAD/);
|
||||
assert.match(workflow, /BASE_BRANCH/);
|
||||
});
|
||||
|
||||
test('excludes merge commits from the audit', () => {
|
||||
assert.match(workflow, /--no-merges/);
|
||||
});
|
||||
|
||||
test('renders a Test commit / Impl commit / gate_status table', () => {
|
||||
assert.match(workflow, /Test commit[\s\S]*Impl commit[\s\S]*gate_status/);
|
||||
});
|
||||
|
||||
test('pairs conventional-commit test: rows with their impl commit', () => {
|
||||
assert.match(workflow, /test:/);
|
||||
assert.match(workflow, /pair/i);
|
||||
});
|
||||
|
||||
test('escapes pipe characters in commit subjects so the table is not broken', () => {
|
||||
assert.match(workflow, /[Ee]scape[\s\S]{0,60}\|/);
|
||||
});
|
||||
|
||||
test('counts commits lacking a recognized gate_status trailer as missing', () => {
|
||||
assert.match(workflow, /missing/);
|
||||
});
|
||||
|
||||
test('is informational and never blocks the ship', () => {
|
||||
assert.match(workflow, /informational|never block|non-blocking/i);
|
||||
});
|
||||
|
||||
test('emits the aggregate trailer in the exact, stable key order', () => {
|
||||
assert.match(
|
||||
workflow,
|
||||
/gate_status:\s*skill=[^,]*,\s*fallback=[^,]*,\s*exempt=[^,]*,\s*missing=/,
|
||||
);
|
||||
});
|
||||
|
||||
test('places the aggregate trailer on the final line so squash-merge carries it', () => {
|
||||
assert.match(workflow, /squash/i);
|
||||
assert.match(workflow, /final line|last line/i);
|
||||
});
|
||||
|
||||
test('does not disturb the frozen #3167 core section order (Key Decisions precedes the new section)', () => {
|
||||
assert.match(workflow, /## Key Decisions[\s\S]*## TDD Audit/);
|
||||
});
|
||||
|
||||
// Hardening assertions added after adversarial review.
|
||||
|
||||
test('pairs test: rows only with feat:/fix: impl commits, skipping refactor/docs/chore', () => {
|
||||
assert.match(workflow, /feat:[\s\S]{0,20}fix:/);
|
||||
assert.match(workflow, /skipping[\s\S]{0,80}(refactor|docs|chore)/i);
|
||||
});
|
||||
|
||||
test('normalizes the gate_status cell to a known token, never raw trailer text', () => {
|
||||
assert.match(workflow, /normaliz[a-z]*[\s\S]{0,120}missing/i);
|
||||
assert.match(workflow, /never the raw/i);
|
||||
});
|
||||
|
||||
test('treats a commit with multiple gate_status trailers as missing', () => {
|
||||
assert.match(workflow, /more than one[\s\S]{0,40}gate_status/i);
|
||||
});
|
||||
|
||||
test('hardens every table cell against pipe/newline injection', () => {
|
||||
assert.match(workflow, /strip[\s\S]{0,20}\\r/);
|
||||
});
|
||||
|
||||
test('guards record/field delimiters against adversarial commit messages', () => {
|
||||
assert.match(workflow, /NUL|%x00|delimiter/i);
|
||||
});
|
||||
});
|
||||
@@ -1,718 +0,0 @@
|
||||
'use strict';
|
||||
|
||||
/**
|
||||
* Architecture-level QA for issue #443 — unified effort + fast_mode engine.
|
||||
*
|
||||
* Integration suite (*.integration.test.cjs): cross-module flows that exercise
|
||||
* real CLI invocations via runGsdTools, the full 33-agent registry, and the
|
||||
* config round-trip through config-set -> resolve-execution.
|
||||
*
|
||||
* INVARIANTS tested here (each is also documented in docs/TESTING-SUITES.md):
|
||||
*
|
||||
* (a) CROSS-PROVIDER VALIDITY — renderEffortForRuntime never emits a value
|
||||
* that the real provider API would 400 on. Ground-truth provider enums are
|
||||
* defined as local constants (not sourced from the implementation).
|
||||
*
|
||||
* (b) PARAM/CHANNEL CONTRACT — each runtime exposes a stable parameter name
|
||||
* and propagation channel.
|
||||
*
|
||||
* (c) RESOLVE-EXECUTION JSON CONTRACT — the CLI command emits a stable JSON
|
||||
* shape with all required keys and correct types.
|
||||
*
|
||||
* (d) TOTALITY across the real 33-agent registry — every agent produces a
|
||||
* valid effort value; none returns undefined/null.
|
||||
*
|
||||
* (e) FAST-MODE HONESTY INVARIANT — claude runtime always reports
|
||||
* fast_mode_supported=false (emitting fast_mode frontmatter is a silent
|
||||
* no-op for Claude Code subagents).
|
||||
*
|
||||
* (f) PRECEDENCE MATRIX — first-valid-wins for both effort and fast_mode
|
||||
* cascades, including invalid values correctly falling through.
|
||||
*
|
||||
* (g) DYNAMIC-ROUTING COMPOSITION — resolveEffortForTier escalates
|
||||
* independently of model tier logic; clamps at 'max'; respects
|
||||
* max_escalations; disabled when escalate_on_failure=false.
|
||||
*
|
||||
* (h) CONFIG-TOOLING ROUND-TRIP — config-set accepts all new effort/fast_mode
|
||||
* key paths (schema validation passes); values survive round-trip through
|
||||
* resolve-execution.
|
||||
*/
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { describe, test, before, after, beforeEach, afterEach } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const { createTempProject, cleanup, runGsdTools } = require('./helpers.cjs');
|
||||
|
||||
const {
|
||||
resolveEffortInternal,
|
||||
resolveFastModeInternal,
|
||||
resolveEffortForTier,
|
||||
VALID_EFFORTS,
|
||||
} = require('../gsd-core/bin/lib/model-resolver.cjs');
|
||||
|
||||
const {
|
||||
renderEffortForRuntime,
|
||||
RUNTIMES_WITH_FAST_MODE,
|
||||
catalog,
|
||||
} = require('../gsd-core/bin/lib/model-catalog.cjs');
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
// Ground-truth provider enums (defined HERE, not sourced from the implementation).
|
||||
// These are the exact values the real APIs accept — using a value outside these
|
||||
// sets would result in a 400 response from the provider.
|
||||
//
|
||||
// Sources:
|
||||
// Anthropic: output_config.effort — https://docs.anthropic.com (Claude API)
|
||||
// OpenAI: model_reasoning_effort — https://platform.openai.com/docs (Codex)
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
const PROVIDER_EFFORT_ENUMS = {
|
||||
claude: new Set(['low', 'medium', 'high', 'xhigh', 'max']),
|
||||
codex: new Set(['minimal', 'low', 'medium', 'high', 'xhigh']),
|
||||
};
|
||||
|
||||
// Helper: write config.json into a temp project
|
||||
function writeConfig(dir, config) {
|
||||
const planningDir = path.join(dir, '.planning');
|
||||
fs.mkdirSync(planningDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
|
||||
}
|
||||
|
||||
// ─── (a) CROSS-PROVIDER VALIDITY INVARIANT ───────────────────────────────────
|
||||
|
||||
describe('#443 integration (a): cross-provider validity invariant', () => {
|
||||
// For every universal effort × every provider runtime, the rendered value
|
||||
// must be a member of that provider's real API enum.
|
||||
test('all VALID_EFFORTS render within provider enums for claude and codex', () => {
|
||||
for (const universalEffort of VALID_EFFORTS) {
|
||||
for (const [runtime, providerEnum] of Object.entries(PROVIDER_EFFORT_ENUMS)) {
|
||||
const rendered = renderEffortForRuntime(runtime, universalEffort);
|
||||
assert.ok(
|
||||
providerEnum.has(rendered.value),
|
||||
`render('${runtime}', '${universalEffort}').value = '${rendered.value}' is NOT in the ` +
|
||||
`${runtime} provider enum ${[...providerEnum].join('|')} — real API would 400`
|
||||
);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// Documented clamps must hold exactly
|
||||
test("render('codex','max').value === 'xhigh' (max is Anthropic-only)", () => {
|
||||
assert.strictEqual(renderEffortForRuntime('codex', 'max').value, 'xhigh');
|
||||
});
|
||||
|
||||
test("render('claude','minimal').value === 'low' (minimal is Codex-only)", () => {
|
||||
assert.strictEqual(renderEffortForRuntime('claude', 'minimal').value, 'low');
|
||||
});
|
||||
|
||||
// Common levels must pass through unchanged on BOTH providers
|
||||
test('common levels (low/medium/high/xhigh) pass through unchanged on claude', () => {
|
||||
for (const level of ['low', 'medium', 'high', 'xhigh']) {
|
||||
assert.strictEqual(
|
||||
renderEffortForRuntime('claude', level).value,
|
||||
level,
|
||||
`claude: level '${level}' should pass through unchanged`
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
test('common levels (low/medium/high/xhigh) pass through unchanged on codex', () => {
|
||||
for (const level of ['low', 'medium', 'high', 'xhigh']) {
|
||||
assert.strictEqual(
|
||||
renderEffortForRuntime('codex', level).value,
|
||||
level,
|
||||
`codex: level '${level}' should pass through unchanged`
|
||||
);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// ─── (b) PARAM/CHANNEL CONTRACT ──────────────────────────────────────────────
|
||||
|
||||
describe('#443 integration (b): param/channel contract', () => {
|
||||
test("claude: param is always 'output_config.effort'", () => {
|
||||
for (const effort of VALID_EFFORTS) {
|
||||
const r = renderEffortForRuntime('claude', effort);
|
||||
assert.strictEqual(r.param, 'output_config.effort',
|
||||
`claude param must be 'output_config.effort' for effort '${effort}'`);
|
||||
}
|
||||
});
|
||||
|
||||
test("codex: param is always 'model_reasoning_effort'", () => {
|
||||
for (const effort of VALID_EFFORTS) {
|
||||
const r = renderEffortForRuntime('codex', effort);
|
||||
assert.strictEqual(r.param, 'model_reasoning_effort',
|
||||
`codex param must be 'model_reasoning_effort' for effort '${effort}'`);
|
||||
}
|
||||
});
|
||||
|
||||
test('claude channel is stable: frontmatter', () => {
|
||||
for (const effort of VALID_EFFORTS) {
|
||||
assert.strictEqual(renderEffortForRuntime('claude', effort).channel, 'frontmatter');
|
||||
}
|
||||
});
|
||||
|
||||
test('codex channel is stable: api', () => {
|
||||
for (const effort of VALID_EFFORTS) {
|
||||
assert.strictEqual(renderEffortForRuntime('codex', effort).channel, 'api');
|
||||
}
|
||||
});
|
||||
|
||||
test("unknown runtimes (gemini, qwen, 'mystery'): param===null, value passes through", () => {
|
||||
for (const runtime of ['gemini', 'qwen', 'mystery']) {
|
||||
for (const effort of VALID_EFFORTS) {
|
||||
const r = renderEffortForRuntime(runtime, effort);
|
||||
assert.strictEqual(r.param, null, `${runtime}: param must be null`);
|
||||
assert.strictEqual(r.channel, null, `${runtime}: channel must be null`);
|
||||
assert.strictEqual(r.value, effort, `${runtime}: value must pass through unchanged`);
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// ─── (c) RESOLVE-EXECUTION JSON CONTRACT ─────────────────────────────────────
|
||||
|
||||
describe('#443 integration (c): resolve-execution JSON contract', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
function assertFullContract(output, label) {
|
||||
assert.ok(typeof output.model === 'string' && output.model.length > 0,
|
||||
`${label}: model must be a non-empty string`);
|
||||
assert.ok(typeof output.profile === 'string' && output.profile.length > 0,
|
||||
`${label}: profile must be a non-empty string`);
|
||||
assert.ok(VALID_EFFORTS.includes(output.effort),
|
||||
`${label}: effort '${output.effort}' must be a member of VALID_EFFORTS`);
|
||||
assert.ok(typeof output.effort_rendered === 'string' && output.effort_rendered.length > 0,
|
||||
`${label}: effort_rendered must be a non-empty string`);
|
||||
assert.ok(output.effort_param === null || typeof output.effort_param === 'string',
|
||||
`${label}: effort_param must be string or null`);
|
||||
assert.ok(output.effort_propagation === null || typeof output.effort_propagation === 'string',
|
||||
`${label}: effort_propagation must be string or null`);
|
||||
assert.ok(typeof output.fast_mode === 'boolean',
|
||||
`${label}: fast_mode must be a boolean`);
|
||||
assert.ok(typeof output.fast_mode_supported === 'boolean',
|
||||
`${label}: fast_mode_supported must be a boolean`);
|
||||
}
|
||||
|
||||
test('gsd-planner (default claude runtime): full contract + known-agent shape', () => {
|
||||
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assertFullContract(output, 'gsd-planner/claude');
|
||||
assert.strictEqual(output.effort_param, 'output_config.effort');
|
||||
assert.strictEqual(output.effort_propagation, 'frontmatter');
|
||||
assert.strictEqual(output.fast_mode_supported, false);
|
||||
// known agent must NOT have unknown_agent:true
|
||||
assert.ok(!output.unknown_agent, 'known agent must not have unknown_agent:true');
|
||||
});
|
||||
|
||||
test('codex runtime: full contract + effort_param=model_reasoning_effort', () => {
|
||||
writeConfig(tmpDir, { runtime: 'codex' });
|
||||
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assertFullContract(output, 'gsd-planner/codex');
|
||||
assert.strictEqual(output.effort_param, 'model_reasoning_effort');
|
||||
assert.strictEqual(output.fast_mode_supported, false);
|
||||
});
|
||||
|
||||
test('gemini runtime: full contract + effort_param===null (no effort wire)', () => {
|
||||
writeConfig(tmpDir, { runtime: 'gemini' });
|
||||
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assertFullContract(output, 'gsd-planner/gemini');
|
||||
assert.strictEqual(output.effort_param, null);
|
||||
assert.strictEqual(output.effort_propagation, null);
|
||||
assert.strictEqual(output.fast_mode_supported, false);
|
||||
});
|
||||
|
||||
test('unknown agent: full contract + unknown_agent===true', () => {
|
||||
const result = runGsdTools(['resolve-execution', 'unknown-agent-xyz'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assertFullContract(output, 'unknown-agent-xyz');
|
||||
assert.strictEqual(output.unknown_agent, true, 'unknown agent must have unknown_agent:true');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── (d) TOTALITY across the real 33-agent registry ──────────────────────────
|
||||
|
||||
describe('#443 integration (d): totality across real registry', () => {
|
||||
let tmpDir;
|
||||
before(() => { tmpDir = createTempProject(); });
|
||||
after(() => { cleanup(tmpDir); });
|
||||
|
||||
const registeredAgents = Object.keys(catalog.agents);
|
||||
// Confirm we're covering the full registry — snapshot the count so a
|
||||
// catalog shrink is caught by this assertion.
|
||||
test(`registry has at least 33 agents (currently ${registeredAgents.length})`, () => {
|
||||
assert.ok(registeredAgents.length >= 33,
|
||||
`Expected at least 33 agents in registry, got ${registeredAgents.length}`);
|
||||
});
|
||||
|
||||
test(`all ${registeredAgents.length} agents: resolveEffortInternal returns a VALID_EFFORTS member`, () => {
|
||||
const effortSet = new Set(VALID_EFFORTS);
|
||||
const bad = [];
|
||||
for (const agent of registeredAgents) {
|
||||
const effort = resolveEffortInternal(tmpDir, agent);
|
||||
if (effort === undefined || effort === null || !effortSet.has(effort)) {
|
||||
bad.push(`${agent}: got ${JSON.stringify(effort)}`);
|
||||
}
|
||||
}
|
||||
assert.strictEqual(bad.length, 0,
|
||||
`Agents with invalid effort:\n${bad.join('\n')}`);
|
||||
});
|
||||
|
||||
test(`all ${registeredAgents.length} agents: resolveFastModeInternal returns strict boolean`, () => {
|
||||
const bad = [];
|
||||
for (const agent of registeredAgents) {
|
||||
const fm = resolveFastModeInternal(tmpDir, agent);
|
||||
if (typeof fm !== 'boolean') {
|
||||
bad.push(`${agent}: got ${JSON.stringify(fm)} (${typeof fm})`);
|
||||
}
|
||||
}
|
||||
assert.strictEqual(bad.length, 0,
|
||||
`Agents with non-boolean fast_mode:\n${bad.join('\n')}`);
|
||||
});
|
||||
|
||||
test(`all ${registeredAgents.length} agents: renderEffortForRuntime('claude', effort) stays in claude enum`, () => {
|
||||
const claudeEnum = PROVIDER_EFFORT_ENUMS.claude;
|
||||
const bad = [];
|
||||
for (const agent of registeredAgents) {
|
||||
const effort = resolveEffortInternal(tmpDir, agent);
|
||||
const rendered = renderEffortForRuntime('claude', effort);
|
||||
if (!claudeEnum.has(rendered.value)) {
|
||||
bad.push(`${agent}: effort=${effort} rendered=${rendered.value} not in claude enum`);
|
||||
}
|
||||
}
|
||||
assert.strictEqual(bad.length, 0,
|
||||
`Agents producing invalid claude effort:\n${bad.join('\n')}`);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── (e) FAST-MODE HONESTY INVARIANT ─────────────────────────────────────────
|
||||
|
||||
describe('#443 integration (e): fast-mode honesty invariant', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
// Sample of agents across all tiers to prove the invariant is not agent-specific
|
||||
const testAgents = ['gsd-planner', 'gsd-executor', 'gsd-codebase-mapper', 'gsd-verifier'];
|
||||
|
||||
test('claude runtime: fast_mode_supported is ALWAYS false regardless of fast_mode config', () => {
|
||||
const configs = [
|
||||
{},
|
||||
{ fast_mode: { enabled: true } },
|
||||
{ fast_mode: { routing_tier_defaults: { heavy: true } } },
|
||||
{ fast_mode: { agent_overrides: { 'gsd-planner': true } } },
|
||||
];
|
||||
for (const config of configs) {
|
||||
writeConfig(tmpDir, config);
|
||||
for (const agent of testAgents) {
|
||||
const result = runGsdTools(['resolve-execution', agent], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed for ${agent}: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.strictEqual(output.fast_mode_supported, false,
|
||||
`claude/${agent}: fast_mode_supported must be false (Claude has no per-subagent fast-mode mechanism); config=${JSON.stringify(config)}`);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
test("RUNTIMES_WITH_FAST_MODE.has('api') === true (api is the only fast-mode capable runtime)", () => {
|
||||
assert.ok(RUNTIMES_WITH_FAST_MODE.has('api'),
|
||||
"RUNTIMES_WITH_FAST_MODE must include 'api' — this is the only runtime with per-call fast_mode support");
|
||||
});
|
||||
|
||||
test("RUNTIMES_WITH_FAST_MODE.has('claude') === false (claude fast-mode is session-level only)", () => {
|
||||
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('claude'),
|
||||
"RUNTIMES_WITH_FAST_MODE must NOT include 'claude' — emitting fast_mode frontmatter on a Claude subagent is a silent no-op");
|
||||
});
|
||||
|
||||
test("RUNTIMES_WITH_FAST_MODE.has('codex') === false", () => {
|
||||
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('codex'),
|
||||
"codex does not support per-call fast_mode");
|
||||
});
|
||||
|
||||
test("RUNTIMES_WITH_FAST_MODE.has('gemini') === false", () => {
|
||||
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('gemini'),
|
||||
"gemini does not support per-call fast_mode");
|
||||
});
|
||||
});
|
||||
|
||||
// ─── (f) PRECEDENCE MATRIX ───────────────────────────────────────────────────
|
||||
|
||||
describe('#443 integration (f): precedence matrix (property/table-driven)', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
// Effort: first-valid-wins from highest precedence to lowest
|
||||
// 1. opts.override (invocation)
|
||||
// 2. effort.agent_overrides.<agent>
|
||||
// 3. effort.routing_tier_defaults.<tier>
|
||||
// 4. effort.default
|
||||
// 5. manifest tier default
|
||||
// 6. hardcoded 'high'
|
||||
const effortPrecedenceTable = [
|
||||
{
|
||||
label: 'layer 1 (invocation override) beats all',
|
||||
config: {
|
||||
effort: {
|
||||
agent_overrides: { 'gsd-planner': 'low' },
|
||||
routing_tier_defaults: { heavy: 'medium' },
|
||||
default: 'xhigh',
|
||||
},
|
||||
},
|
||||
opts: { override: 'minimal' },
|
||||
expected: 'minimal',
|
||||
},
|
||||
{
|
||||
label: 'layer 2 (agent_override) beats tier default and default',
|
||||
config: {
|
||||
effort: {
|
||||
agent_overrides: { 'gsd-planner': 'low' },
|
||||
routing_tier_defaults: { heavy: 'medium' },
|
||||
default: 'xhigh',
|
||||
},
|
||||
},
|
||||
opts: {},
|
||||
expected: 'low',
|
||||
},
|
||||
{
|
||||
label: 'layer 3 (routing_tier_defaults) beats effort.default',
|
||||
config: {
|
||||
effort: {
|
||||
routing_tier_defaults: { heavy: 'medium' },
|
||||
default: 'xhigh',
|
||||
},
|
||||
},
|
||||
opts: {},
|
||||
expected: 'medium',
|
||||
},
|
||||
{
|
||||
label: 'layer 4 (effort.default) when no tier default set',
|
||||
config: {
|
||||
effort: { default: 'low' },
|
||||
},
|
||||
opts: {},
|
||||
expected: 'low',
|
||||
},
|
||||
{
|
||||
label: 'invalid layer 1 (turbo) falls through to layer 2 (agent_override)',
|
||||
config: {
|
||||
effort: { agent_overrides: { 'gsd-planner': 'medium' } },
|
||||
},
|
||||
opts: { override: 'turbo' },
|
||||
expected: 'medium',
|
||||
},
|
||||
{
|
||||
label: 'invalid layer 2 (agent_override=123 numeric) falls through to tier default',
|
||||
config: {
|
||||
effort: {
|
||||
agent_overrides: { 'gsd-planner': 123 },
|
||||
routing_tier_defaults: { heavy: 'high' },
|
||||
},
|
||||
},
|
||||
opts: {},
|
||||
expected: 'high',
|
||||
},
|
||||
{
|
||||
label: 'invalid tier default (turbo) falls through to effort.default',
|
||||
config: {
|
||||
effort: {
|
||||
routing_tier_defaults: { heavy: 'turbo' },
|
||||
default: 'low',
|
||||
},
|
||||
},
|
||||
opts: {},
|
||||
expected: 'low',
|
||||
},
|
||||
];
|
||||
|
||||
for (const row of effortPrecedenceTable) {
|
||||
test(`effort precedence: ${row.label}`, () => {
|
||||
writeConfig(tmpDir, row.config);
|
||||
const result = resolveEffortInternal(tmpDir, 'gsd-planner', row.opts);
|
||||
assert.strictEqual(result, row.expected,
|
||||
`Expected '${row.expected}', got '${result}' — config: ${JSON.stringify(row.config)}`);
|
||||
});
|
||||
}
|
||||
|
||||
// fast_mode precedence:
|
||||
// 1. opts.override (strict boolean only)
|
||||
// 2. fast_mode.agent_overrides.<agent> (strict boolean only)
|
||||
// 3. fast_mode.routing_tier_defaults.<tier> (strict boolean only)
|
||||
// 4. fast_mode.enabled (strict boolean only)
|
||||
// 5. false
|
||||
const fastModePrecedenceTable = [
|
||||
{
|
||||
label: 'layer 1 (opts.override=false) beats enabled=true',
|
||||
config: { fast_mode: { enabled: true } },
|
||||
opts: { override: false },
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
label: 'layer 2 (agent_override=true) beats tier default',
|
||||
config: {
|
||||
fast_mode: {
|
||||
agent_overrides: { 'gsd-planner': true },
|
||||
routing_tier_defaults: { heavy: false },
|
||||
enabled: false,
|
||||
},
|
||||
},
|
||||
opts: {},
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
label: 'layer 3 (tier default=true) beats enabled=false',
|
||||
config: {
|
||||
fast_mode: {
|
||||
routing_tier_defaults: { heavy: true },
|
||||
enabled: false,
|
||||
},
|
||||
},
|
||||
opts: {},
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
label: 'layer 4 (enabled=true) when no tier/agent overrides',
|
||||
config: { fast_mode: { enabled: true } },
|
||||
opts: {},
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
label: 'layer 5 (default false) when all absent',
|
||||
config: {},
|
||||
opts: {},
|
||||
expected: false,
|
||||
},
|
||||
{
|
||||
label: 'string "true" in opts.override is NOT accepted (falls through)',
|
||||
config: { fast_mode: { enabled: true } },
|
||||
// override must be strict boolean; string falls through to next layer
|
||||
opts: { override: 'true' },
|
||||
// 'true' as string is not boolean -> falls through to tier default
|
||||
// gsd-planner is heavy; no tier default set; falls to enabled=true
|
||||
expected: true,
|
||||
},
|
||||
{
|
||||
label: 'string "true" in agent_overrides is NOT accepted',
|
||||
config: {
|
||||
fast_mode: {
|
||||
agent_overrides: { 'gsd-planner': 'true' },
|
||||
enabled: false,
|
||||
},
|
||||
},
|
||||
opts: {},
|
||||
// string 'true' is not boolean -> fall through to tier default -> enabled=false -> false
|
||||
expected: false,
|
||||
},
|
||||
];
|
||||
|
||||
for (const row of fastModePrecedenceTable) {
|
||||
test(`fast_mode precedence: ${row.label}`, () => {
|
||||
writeConfig(tmpDir, row.config);
|
||||
const result = resolveFastModeInternal(tmpDir, 'gsd-planner', row.opts);
|
||||
assert.strictEqual(result, row.expected,
|
||||
`Expected ${row.expected}, got ${result} — config: ${JSON.stringify(row.config)}`);
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// ─── (g) DYNAMIC-ROUTING COMPOSITION ─────────────────────────────────────────
|
||||
|
||||
describe('#443 integration (g): dynamic-routing composition', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
const dynamicRoutingBase = {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: true,
|
||||
max_escalations: 4,
|
||||
},
|
||||
effort: { routing_tier_defaults: { light: 'low' } },
|
||||
};
|
||||
|
||||
test('resolveEffortForTier escalates independently of model resolution', () => {
|
||||
writeConfig(tmpDir, dynamicRoutingBase);
|
||||
const effort0 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
|
||||
const effort1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
|
||||
const effort2 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 2);
|
||||
assert.strictEqual(effort0, 'low');
|
||||
assert.strictEqual(effort1, 'medium');
|
||||
assert.strictEqual(effort2, 'high');
|
||||
// Verify the effort ladder steps up correctly without asserting model value
|
||||
// (model timing is a separate concern from effort escalation)
|
||||
assert.notStrictEqual(effort0, effort1, 'effort should escalate at attempt 1');
|
||||
assert.notStrictEqual(effort1, effort2, 'effort should escalate at attempt 2');
|
||||
});
|
||||
|
||||
test('escalate_on_failure=false: attempt is ignored for effort', () => {
|
||||
writeConfig(tmpDir, {
|
||||
...dynamicRoutingBase,
|
||||
dynamic_routing: {
|
||||
...dynamicRoutingBase.dynamic_routing,
|
||||
escalate_on_failure: false,
|
||||
},
|
||||
});
|
||||
const e0 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
|
||||
const e1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
|
||||
const e3 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 3);
|
||||
assert.strictEqual(e0, e1, 'effort must not escalate when escalate_on_failure=false');
|
||||
assert.strictEqual(e0, e3, 'effort must not escalate when escalate_on_failure=false');
|
||||
});
|
||||
|
||||
test('escalation clamps at "max" regardless of attempt number', () => {
|
||||
writeConfig(tmpDir, {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: true,
|
||||
max_escalations: 99,
|
||||
},
|
||||
effort: { default: 'max' },
|
||||
});
|
||||
// Any large attempt number — result must never exceed 'max'
|
||||
const r = resolveEffortForTier(tmpDir, 'gsd-planner', 50);
|
||||
assert.strictEqual(r, 'max', `Effort must clamp at 'max', got '${r}'`);
|
||||
const EFFORT_LADDER = VALID_EFFORTS;
|
||||
const maxIdx = EFFORT_LADDER.indexOf('max');
|
||||
const rIdx = EFFORT_LADDER.indexOf(r);
|
||||
assert.ok(rIdx <= maxIdx, 'Effort must not exceed the max position in the ladder');
|
||||
});
|
||||
|
||||
test('respects max_escalations cap: attempt beyond cap gives same as cap', () => {
|
||||
writeConfig(tmpDir, {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: true,
|
||||
max_escalations: 1,
|
||||
},
|
||||
effort: { routing_tier_defaults: { light: 'low' } },
|
||||
});
|
||||
const atCap = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1); // 1 escalation
|
||||
const beyond = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 5); // capped at 1
|
||||
assert.strictEqual(atCap, beyond,
|
||||
'Effort beyond max_escalations must be same as at cap');
|
||||
assert.strictEqual(atCap, 'medium', 'low + 1 escalation = medium');
|
||||
});
|
||||
|
||||
test('dynamic_routing disabled: resolveEffortForTier ignores attempt', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { routing_tier_defaults: { light: 'low' } },
|
||||
});
|
||||
const e0 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
|
||||
const e5 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 5);
|
||||
assert.strictEqual(e0, e5, 'Effort must not change when dynamic_routing is disabled');
|
||||
assert.strictEqual(e0, 'low');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── (h) CONFIG-TOOLING ROUND-TRIP ───────────────────────────────────────────
|
||||
|
||||
describe('#443 integration (h): config-tooling round-trip', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
test('config-set effort.default then resolve-execution reflects new value', () => {
|
||||
const setResult = runGsdTools(['config-set', 'effort.default', 'low'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(setResult.success, `config-set effort.default failed: ${setResult.error}`);
|
||||
|
||||
const execResult = runGsdTools(['resolve-execution', 'unknown-agent-xyz'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
|
||||
const output = JSON.parse(execResult.output);
|
||||
// unknown agent falls through to effort.default
|
||||
assert.strictEqual(output.effort, 'low',
|
||||
`Expected effort='low' after config-set, got '${output.effort}'`);
|
||||
});
|
||||
|
||||
test('config-set effort.routing_tier_defaults.heavy then resolve-execution uses it', () => {
|
||||
const setResult = runGsdTools(
|
||||
['config-set', 'effort.routing_tier_defaults.heavy', 'medium'],
|
||||
tmpDir, { HOME: tmpDir }
|
||||
);
|
||||
assert.ok(setResult.success, `config-set failed: ${setResult.error}`);
|
||||
|
||||
const execResult = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
|
||||
const output = JSON.parse(execResult.output);
|
||||
// gsd-planner is heavy; tier default now overridden to medium
|
||||
assert.strictEqual(output.effort, 'medium',
|
||||
`Expected effort='medium' after routing_tier_defaults override, got '${output.effort}'`);
|
||||
});
|
||||
|
||||
test('config-set effort.agent_overrides.<agent> wins over tier default', () => {
|
||||
// Set tier default first, then per-agent override
|
||||
runGsdTools(['config-set', 'effort.routing_tier_defaults.heavy', 'medium'], tmpDir, { HOME: tmpDir });
|
||||
const setResult = runGsdTools(
|
||||
['config-set', 'effort.agent_overrides.gsd-planner', 'xhigh'],
|
||||
tmpDir, { HOME: tmpDir }
|
||||
);
|
||||
assert.ok(setResult.success, `config-set agent_overrides failed: ${setResult.error}`);
|
||||
|
||||
const execResult = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
|
||||
const output = JSON.parse(execResult.output);
|
||||
assert.strictEqual(output.effort, 'xhigh',
|
||||
`Expected agent_overrides to win (xhigh), got '${output.effort}'`);
|
||||
});
|
||||
|
||||
test('config-set fast_mode.enabled true then resolve-execution reflects fast_mode=true', () => {
|
||||
const setResult = runGsdTools(['config-set', 'fast_mode.enabled', 'true'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(setResult.success, `config-set fast_mode.enabled failed: ${setResult.error}`);
|
||||
|
||||
const execResult = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
|
||||
const output = JSON.parse(execResult.output);
|
||||
assert.strictEqual(output.fast_mode, true,
|
||||
`Expected fast_mode=true after config-set, got ${output.fast_mode}`);
|
||||
// fast_mode_supported stays false (claude runtime)
|
||||
assert.strictEqual(output.fast_mode_supported, false);
|
||||
});
|
||||
|
||||
test('config-set fast_mode.agent_overrides.<agent> true reflects in output', () => {
|
||||
const setResult = runGsdTools(
|
||||
['config-set', 'fast_mode.agent_overrides.gsd-codebase-mapper', 'true'],
|
||||
tmpDir, { HOME: tmpDir }
|
||||
);
|
||||
assert.ok(setResult.success, `config-set failed: ${setResult.error}`);
|
||||
|
||||
const execResult = runGsdTools(['resolve-execution', 'gsd-codebase-mapper'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(execResult.success, `resolve-execution failed: ${execResult.error}`);
|
||||
const output = JSON.parse(execResult.output);
|
||||
assert.strictEqual(output.fast_mode, true,
|
||||
`Expected fast_mode=true for agent-specific override`);
|
||||
});
|
||||
|
||||
// Prove the config-set commands accept all the new key namespaces (schema validation)
|
||||
test('config-set accepts all effort/* and fast_mode/* key namespaces without error', () => {
|
||||
const keysToTest = [
|
||||
['effort.default', 'high'],
|
||||
['effort.routing_tier_defaults.light', 'low'],
|
||||
['effort.routing_tier_defaults.standard', 'medium'],
|
||||
['effort.routing_tier_defaults.heavy', 'xhigh'],
|
||||
['effort.agent_overrides.gsd-executor', 'high'],
|
||||
['fast_mode.enabled', 'false'],
|
||||
['fast_mode.routing_tier_defaults.light', 'false'],
|
||||
['fast_mode.routing_tier_defaults.standard', 'false'],
|
||||
['fast_mode.routing_tier_defaults.heavy', 'false'],
|
||||
['fast_mode.agent_overrides.gsd-verifier', 'false'],
|
||||
];
|
||||
for (const [key, val] of keysToTest) {
|
||||
const r = runGsdTools(['config-set', key, val], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(r.success, `config-set '${key}' '${val}' should succeed, got: ${r.error}`);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -1,862 +0,0 @@
|
||||
'use strict';
|
||||
|
||||
/**
|
||||
* Feature test for issue #443 — unified cross-provider effort + fast_mode knobs.
|
||||
*
|
||||
* Adds config-driven effort (universal ladder: minimal<low<medium<high<xhigh<max)
|
||||
* and fast_mode knobs. Per-runtime rendering clamps the unique tails:
|
||||
* - Anthropic/Claude: supports {low,medium,high,xhigh,max}, param=output_config.effort
|
||||
* - Codex: supports {minimal,low,medium,high,xhigh}, param=model_reasoning_effort
|
||||
*
|
||||
* Also adds resolve-execution query which is the superset command including
|
||||
* effort rendering and fast_mode propagation metadata.
|
||||
*/
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { describe, test, beforeEach, afterEach } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const { createTempProject, cleanup, runGsdTools } = require('./helpers.cjs');
|
||||
|
||||
const {
|
||||
resolveEffortInternal,
|
||||
resolveFastModeInternal,
|
||||
resolveEffortForTier,
|
||||
} = require('../gsd-core/bin/lib/model-resolver.cjs');
|
||||
|
||||
const {
|
||||
renderEffortForRuntime,
|
||||
RUNTIMES_WITH_FAST_MODE,
|
||||
} = require('../gsd-core/bin/lib/model-catalog.cjs');
|
||||
|
||||
const {
|
||||
injectEffortFrontmatter,
|
||||
} = require('../bin/install.js');
|
||||
|
||||
function writeConfig(dir, config) {
|
||||
const planningDir = path.join(dir, '.planning');
|
||||
fs.mkdirSync(planningDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
|
||||
}
|
||||
|
||||
// ─── Effort cascade ───────────────────────────────────────────────────────────
|
||||
|
||||
describe('#443 effort cascade', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
test('no config -> gsd-planner (heavy) defaults to "xhigh" via tier default', () => {
|
||||
// gsd-planner is heavy tier; manifest default for heavy is xhigh
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
|
||||
});
|
||||
|
||||
test('routing_tier_defaults: light (gsd-codebase-mapper) -> "low"', () => {
|
||||
// gsd-codebase-mapper routingTier=light, default for light is "low"
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-codebase-mapper'), 'low');
|
||||
});
|
||||
|
||||
test('routing_tier_defaults: standard (gsd-executor) -> "high"', () => {
|
||||
// gsd-executor routingTier=standard, default for standard is "high"
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-executor'), 'high');
|
||||
});
|
||||
|
||||
test('routing_tier_defaults: heavy (gsd-planner) -> "xhigh"', () => {
|
||||
// gsd-planner routingTier=heavy, default for heavy is "xhigh"
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
|
||||
});
|
||||
|
||||
test('effort.routing_tier_defaults override beats tier default', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { routing_tier_defaults: { heavy: 'medium' } },
|
||||
});
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'medium');
|
||||
});
|
||||
|
||||
test('effort.agent_overrides beats routing_tier_defaults', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: {
|
||||
routing_tier_defaults: { heavy: 'medium' },
|
||||
agent_overrides: { 'gsd-planner': 'low' },
|
||||
},
|
||||
});
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'low');
|
||||
});
|
||||
|
||||
test('opts.override beats agent_overrides', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { agent_overrides: { 'gsd-planner': 'low' } },
|
||||
});
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner', { override: 'minimal' }), 'minimal');
|
||||
});
|
||||
|
||||
test('invalid override falls through to agent_overrides', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { agent_overrides: { 'gsd-planner': 'low' } },
|
||||
});
|
||||
// 'turbo' is not a valid effort — should fall through to agent_overrides
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner', { override: 'turbo' }), 'low');
|
||||
});
|
||||
|
||||
test('invalid agent_overrides value falls through to routing_tier_defaults', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: {
|
||||
agent_overrides: { 'gsd-planner': 123 },
|
||||
routing_tier_defaults: { heavy: 'medium' },
|
||||
},
|
||||
});
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'medium');
|
||||
});
|
||||
|
||||
test('invalid routing_tier_defaults value falls through to effort.default', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: {
|
||||
routing_tier_defaults: { heavy: 'turbo' },
|
||||
default: 'low',
|
||||
},
|
||||
});
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'low');
|
||||
});
|
||||
|
||||
test('invalid effort.default falls through to hardcoded "high" (no routing_tier_defaults set)', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { default: 'turbo' },
|
||||
});
|
||||
// effortCfg set but no routing_tier_defaults; turbo is invalid; fallback = hardcoded 'high'
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'high');
|
||||
});
|
||||
|
||||
test('unknown agent -> uses effort.default', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { default: 'medium' },
|
||||
});
|
||||
// unknown-agent has no routingTier, so step 3 skipped
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'unknown-agent-xyz'), 'medium');
|
||||
});
|
||||
|
||||
test('effort.default numeric value (123) ignored, hardcoded "high" fallback', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { default: 123 },
|
||||
});
|
||||
// effortCfg set, no routing_tier_defaults -> no tier default; numeric ignored -> 'high'
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'high');
|
||||
});
|
||||
|
||||
test('effort block missing entirely -> uses tier default', () => {
|
||||
// No effort key in config at all
|
||||
writeConfig(tmpDir, { model_profile: 'balanced' });
|
||||
// heavy agent: tier default xhigh
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
|
||||
});
|
||||
|
||||
test('effort block is non-object (string) -> effortCfg=null -> uses manifest tier default xhigh', () => {
|
||||
writeConfig(tmpDir, { effort: 'bad' });
|
||||
// Non-object effort => effortCfg=null; gsd-planner heavy tier manifest default = xhigh
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'xhigh');
|
||||
});
|
||||
|
||||
test('effort.routing_tier_defaults empty object -> effort.default', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { routing_tier_defaults: {}, default: 'low' },
|
||||
});
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'low');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Fast mode cascade ────────────────────────────────────────────────────────
|
||||
|
||||
describe('#443 fast_mode cascade', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
test('no config -> defaults to false', () => {
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
|
||||
});
|
||||
|
||||
test('fast_mode.enabled=true -> true when no tier/agent overrides', () => {
|
||||
writeConfig(tmpDir, { fast_mode: { enabled: true } });
|
||||
// heavy agent: tier default is false, but enabled=true is layer 4
|
||||
// tier default for heavy is false (below enabled), so gets enabled=true
|
||||
// Wait — the cascade is: 1.override 2.agent_overrides 3.tier_defaults 4.enabled 5.false
|
||||
// For gsd-planner (heavy), tier default is false — falls through to enabled=true
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), true);
|
||||
});
|
||||
|
||||
test('fast_mode.routing_tier_defaults.light=true -> light agent gets true', () => {
|
||||
writeConfig(tmpDir, {
|
||||
fast_mode: { routing_tier_defaults: { light: true } },
|
||||
});
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-codebase-mapper'), true);
|
||||
});
|
||||
|
||||
test('fast_mode.routing_tier_defaults.heavy=false -> heavy agent stays false', () => {
|
||||
writeConfig(tmpDir, {
|
||||
fast_mode: { enabled: true, routing_tier_defaults: { heavy: false } },
|
||||
});
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
|
||||
});
|
||||
|
||||
test('fast_mode.agent_overrides beats routing_tier_defaults', () => {
|
||||
writeConfig(tmpDir, {
|
||||
fast_mode: {
|
||||
routing_tier_defaults: { light: false },
|
||||
agent_overrides: { 'gsd-codebase-mapper': true },
|
||||
},
|
||||
});
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-codebase-mapper'), true);
|
||||
});
|
||||
|
||||
test('opts.override beats agent_overrides', () => {
|
||||
writeConfig(tmpDir, {
|
||||
fast_mode: { agent_overrides: { 'gsd-planner': true } },
|
||||
});
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner', { override: false }), false);
|
||||
});
|
||||
|
||||
test('string "true" NOT accepted as fast_mode override', () => {
|
||||
writeConfig(tmpDir, {
|
||||
fast_mode: { agent_overrides: { 'gsd-planner': 'true' } },
|
||||
});
|
||||
// string "true" is not boolean -> fall through to tier default or enabled
|
||||
const result = resolveFastModeInternal(tmpDir, 'gsd-planner');
|
||||
assert.strictEqual(typeof result, 'boolean');
|
||||
});
|
||||
|
||||
test('string "true" in opts.override NOT accepted', () => {
|
||||
// opts.override must be strict boolean — string falls through
|
||||
const result = resolveFastModeInternal(tmpDir, 'gsd-planner', { override: 'true' });
|
||||
assert.strictEqual(result, false);
|
||||
});
|
||||
|
||||
test('fast_mode block missing entirely -> defaults to false', () => {
|
||||
writeConfig(tmpDir, { model_profile: 'balanced' });
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
|
||||
});
|
||||
|
||||
test('fast_mode.enabled="yes" (non-boolean) ignored -> false', () => {
|
||||
writeConfig(tmpDir, { fast_mode: { enabled: 'yes' } });
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
|
||||
});
|
||||
|
||||
test('unknown agent fast_mode -> uses enabled flag', () => {
|
||||
writeConfig(tmpDir, { fast_mode: { enabled: true } });
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'unknown-agent-xyz'), true);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Effort escalation (resolveEffortForTier) ─────────────────────────────────
|
||||
|
||||
describe('#443 resolveEffortForTier escalation', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
test('dynamic_routing disabled -> attempt ignored, returns base effort', () => {
|
||||
// gsd-planner heavy -> xhigh baseline
|
||||
const base = resolveEffortForTier(tmpDir, 'gsd-planner', 0);
|
||||
const attempt1 = resolveEffortForTier(tmpDir, 'gsd-planner', 1);
|
||||
assert.strictEqual(base, 'xhigh');
|
||||
assert.strictEqual(attempt1, 'xhigh'); // no dynamic_routing -> attempt ignored
|
||||
});
|
||||
|
||||
test('dynamic_routing enabled, escalate_on_failure=false -> attempt ignored', () => {
|
||||
writeConfig(tmpDir, {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: false,
|
||||
max_escalations: 2,
|
||||
},
|
||||
});
|
||||
const base = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0);
|
||||
const attempt1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
|
||||
assert.strictEqual(base, attempt1);
|
||||
});
|
||||
|
||||
test('dynamic_routing enabled, attempt=1 -> one step up from base', () => {
|
||||
writeConfig(tmpDir, {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: true,
|
||||
max_escalations: 2,
|
||||
},
|
||||
effort: { routing_tier_defaults: { light: 'low' } },
|
||||
});
|
||||
// gsd-codebase-mapper: light -> effort 'low'; attempt=1 -> 'medium'
|
||||
assert.strictEqual(resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 0), 'low');
|
||||
assert.strictEqual(resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1), 'medium');
|
||||
});
|
||||
|
||||
test('escalation clamps at "max"', () => {
|
||||
writeConfig(tmpDir, {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: true,
|
||||
max_escalations: 99,
|
||||
},
|
||||
effort: { default: 'xhigh' },
|
||||
});
|
||||
// xhigh -> max -> max (clamp)
|
||||
const result = resolveEffortForTier(tmpDir, 'gsd-planner', 99);
|
||||
assert.strictEqual(result, 'max');
|
||||
});
|
||||
|
||||
test('respects max_escalations cap', () => {
|
||||
writeConfig(tmpDir, {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: true,
|
||||
max_escalations: 1,
|
||||
},
|
||||
effort: { routing_tier_defaults: { light: 'low' } },
|
||||
});
|
||||
// light: low -> attempt=1 -> medium (but max=1 so can only escalate once)
|
||||
const at1 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 1);
|
||||
const at2 = resolveEffortForTier(tmpDir, 'gsd-codebase-mapper', 2);
|
||||
// at2 is capped at 1 escalation, same as at1
|
||||
assert.strictEqual(at1, at2);
|
||||
assert.strictEqual(at1, 'medium');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Rendering / clamping ──────────────────────────────────────────────────────
|
||||
|
||||
describe('#443 renderEffortForRuntime', () => {
|
||||
test('codex: "max" clamps to "xhigh"', () => {
|
||||
const r = renderEffortForRuntime('codex', 'max');
|
||||
assert.strictEqual(r.value, 'xhigh');
|
||||
assert.strictEqual(r.param, 'model_reasoning_effort');
|
||||
});
|
||||
|
||||
test('codex: common levels passthrough', () => {
|
||||
assert.strictEqual(renderEffortForRuntime('codex', 'low').value, 'low');
|
||||
assert.strictEqual(renderEffortForRuntime('codex', 'medium').value, 'medium');
|
||||
assert.strictEqual(renderEffortForRuntime('codex', 'high').value, 'high');
|
||||
assert.strictEqual(renderEffortForRuntime('codex', 'xhigh').value, 'xhigh');
|
||||
});
|
||||
|
||||
test('codex: "minimal" passthrough', () => {
|
||||
assert.strictEqual(renderEffortForRuntime('codex', 'minimal').value, 'minimal');
|
||||
});
|
||||
|
||||
test('claude: "minimal" clamps to "low"', () => {
|
||||
const r = renderEffortForRuntime('claude', 'minimal');
|
||||
assert.strictEqual(r.value, 'low');
|
||||
assert.strictEqual(r.param, 'output_config.effort');
|
||||
});
|
||||
|
||||
test('claude: "max" passthrough (Anthropic-only)', () => {
|
||||
const r = renderEffortForRuntime('claude', 'max');
|
||||
assert.strictEqual(r.value, 'max');
|
||||
assert.strictEqual(r.param, 'output_config.effort');
|
||||
});
|
||||
|
||||
test('claude: common levels passthrough', () => {
|
||||
assert.strictEqual(renderEffortForRuntime('claude', 'low').value, 'low');
|
||||
assert.strictEqual(renderEffortForRuntime('claude', 'medium').value, 'medium');
|
||||
assert.strictEqual(renderEffortForRuntime('claude', 'high').value, 'high');
|
||||
assert.strictEqual(renderEffortForRuntime('claude', 'xhigh').value, 'xhigh');
|
||||
});
|
||||
|
||||
test('unknown runtime: param is null, value passthrough', () => {
|
||||
const r = renderEffortForRuntime('unknown-runtime', 'high');
|
||||
assert.strictEqual(r.param, null);
|
||||
assert.strictEqual(r.value, 'high');
|
||||
});
|
||||
|
||||
test('RUNTIMES_WITH_FAST_MODE does NOT include "claude"', () => {
|
||||
// Claude Code has no per-subagent fast-mode mechanism — session-level only
|
||||
assert.ok(!RUNTIMES_WITH_FAST_MODE.has('claude'),
|
||||
'claude must NOT be in RUNTIMES_WITH_FAST_MODE — emitting fast_mode frontmatter is a silent no-op');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── resolve-execution end-to-end ─────────────────────────────────────────────
|
||||
|
||||
describe('#443 resolve-execution CLI command', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => {
|
||||
tmpDir = createTempProject();
|
||||
// HOME isolation to prevent ~/.gsd/defaults.json bleed
|
||||
process.env._GSD_TEST_HOME_OVERRIDE = tmpDir;
|
||||
});
|
||||
afterEach(() => {
|
||||
cleanup(tmpDir);
|
||||
delete process.env._GSD_TEST_HOME_OVERRIDE;
|
||||
});
|
||||
|
||||
test('default (claude) runtime -> effort present, effort_param=output_config.effort, fast_mode_supported=false', () => {
|
||||
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.ok(output.effort, 'should have effort field');
|
||||
assert.strictEqual(output.effort_param, 'output_config.effort');
|
||||
assert.strictEqual(output.fast_mode_supported, false);
|
||||
assert.ok('fast_mode' in output, 'should have fast_mode field');
|
||||
assert.ok('model' in output, 'should have model field');
|
||||
assert.ok('profile' in output, 'should have profile field');
|
||||
});
|
||||
|
||||
test('codex runtime -> effort_param=model_reasoning_effort, max clamps to xhigh, fast_mode_supported=false', () => {
|
||||
writeConfig(tmpDir, {
|
||||
runtime: 'codex',
|
||||
effort: { default: 'max' },
|
||||
});
|
||||
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.strictEqual(output.effort_param, 'model_reasoning_effort');
|
||||
assert.strictEqual(output.effort_rendered, 'xhigh');
|
||||
// fast_mode_supported: codex does not support fast mode via subagent
|
||||
assert.strictEqual(output.fast_mode_supported, false);
|
||||
});
|
||||
|
||||
test('--effort flag overrides config effort', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', 'gsd-planner', '--effort', 'low'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.strictEqual(output.effort, 'low');
|
||||
});
|
||||
|
||||
test('--fast-mode flag honored', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', 'gsd-planner', '--fast-mode', 'true'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.strictEqual(output.fast_mode, true);
|
||||
});
|
||||
|
||||
test('--attempt flag triggers escalation', () => {
|
||||
writeConfig(tmpDir, {
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
escalate_on_failure: true,
|
||||
max_escalations: 2,
|
||||
},
|
||||
effort: { routing_tier_defaults: { light: 'low' } },
|
||||
});
|
||||
const result0 = runGsdTools(
|
||||
['resolve-execution', 'gsd-codebase-mapper', '--attempt', '0'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
const result1 = runGsdTools(
|
||||
['resolve-execution', 'gsd-codebase-mapper', '--attempt', '1'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(result0.success && result1.success);
|
||||
const out0 = JSON.parse(result0.output);
|
||||
const out1 = JSON.parse(result1.output);
|
||||
assert.strictEqual(out0.effort, 'low');
|
||||
assert.strictEqual(out1.effort, 'medium');
|
||||
});
|
||||
|
||||
test('--raw prints effort string', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', 'gsd-planner', '--raw'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
// Raw output should be the effort string
|
||||
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
|
||||
assert.ok(VALID_EFFORTS.includes(result.output.trim()),
|
||||
`Expected effort string, got: ${result.output}`);
|
||||
});
|
||||
|
||||
test('fails when no agent-type provided', () => {
|
||||
const result = runGsdTools(['resolve-execution'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(!result.success, 'should fail without agent-type');
|
||||
assert.ok(result.error.includes('agent-type required'), `error: ${result.error}`);
|
||||
});
|
||||
|
||||
test('unknown agent -> unknown_agent=true still emits effort', () => {
|
||||
const result = runGsdTools(['resolve-execution', 'unknown-agent-xyz'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.strictEqual(output.unknown_agent, true);
|
||||
assert.ok(output.effort, 'should have effort even for unknown agent');
|
||||
});
|
||||
|
||||
test('emits effort_propagation (channel) field', () => {
|
||||
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.ok('effort_propagation' in output, 'should have effort_propagation field');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── resolve-model now emits effort (replaces reasoning_effort) ───────────────
|
||||
|
||||
describe('#443 resolve-model emits effort (unified)', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
test('resolve-model on claude runtime emits effort (not null)', () => {
|
||||
const result = runGsdTools(['resolve-model', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
// effort must be present and valid
|
||||
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
|
||||
assert.ok(VALID_EFFORTS.includes(output.effort),
|
||||
`Expected valid effort, got: ${output.effort}`);
|
||||
// reasoning_effort must NOT be present (removed)
|
||||
assert.ok(!Object.prototype.hasOwnProperty.call(output, 'reasoning_effort'),
|
||||
'resolve-model must not emit reasoning_effort (replaced by effort)');
|
||||
});
|
||||
|
||||
test('resolve-model on codex runtime emits unified effort (not reasoning_effort)', () => {
|
||||
fs.writeFileSync(
|
||||
path.join(tmpDir, '.planning', 'config.json'),
|
||||
JSON.stringify({ runtime: 'codex', model_profile: 'balanced' })
|
||||
);
|
||||
const result = runGsdTools(['resolve-model', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
|
||||
assert.ok(VALID_EFFORTS.includes(output.effort),
|
||||
`Expected valid effort, got: ${output.effort}`);
|
||||
assert.ok(!Object.prototype.hasOwnProperty.call(output, 'reasoning_effort'),
|
||||
'resolve-model must not emit reasoning_effort');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── QA Matrix — hostile/malformed configs ───────────────────────────────────
|
||||
|
||||
describe('#443 QA matrix — malformed effort/fast_mode configs', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = createTempProject(); });
|
||||
afterEach(() => { cleanup(tmpDir); });
|
||||
|
||||
test('effort.default=123 (numeric) -> gracefully falls through', () => {
|
||||
writeConfig(tmpDir, { effort: { default: 123 } });
|
||||
// gsd-planner is heavy, tier default xhigh is used instead
|
||||
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
|
||||
assert.ok(typeof result === 'string');
|
||||
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
|
||||
assert.ok(VALID_EFFORTS.includes(result));
|
||||
});
|
||||
|
||||
test('fast_mode.enabled="yes" (string) -> ignored, returns false', () => {
|
||||
writeConfig(tmpDir, { fast_mode: { enabled: 'yes' } });
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
|
||||
});
|
||||
|
||||
test('effort:{} empty block -> uses tier default or hardcoded high', () => {
|
||||
writeConfig(tmpDir, { effort: {} });
|
||||
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
|
||||
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
|
||||
assert.ok(VALID_EFFORTS.includes(result));
|
||||
});
|
||||
|
||||
test('fast_mode:{} empty block -> false', () => {
|
||||
writeConfig(tmpDir, { fast_mode: {} });
|
||||
assert.strictEqual(resolveFastModeInternal(tmpDir, 'gsd-planner'), false);
|
||||
});
|
||||
|
||||
test('effort config is completely absent -> still resolves valid effort', () => {
|
||||
writeConfig(tmpDir, { model_profile: 'quality' });
|
||||
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
|
||||
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
|
||||
assert.ok(VALID_EFFORTS.includes(result));
|
||||
});
|
||||
|
||||
test('effort.routing_tier_defaults has boolean value -> falls through', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: {
|
||||
routing_tier_defaults: { heavy: true },
|
||||
default: 'medium',
|
||||
},
|
||||
});
|
||||
// boolean true is not a valid effort -> falls through to default 'medium'
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'gsd-planner'), 'medium');
|
||||
});
|
||||
|
||||
test('effort.agent_overrides is non-object -> falls through gracefully', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: {
|
||||
agent_overrides: 'not-an-object',
|
||||
default: 'low',
|
||||
},
|
||||
});
|
||||
// non-object agent_overrides -> skip step 2, use tier default (heavy=xhigh)
|
||||
// actually heavy tier default kicks in first if no routing_tier_defaults
|
||||
const result = resolveEffortInternal(tmpDir, 'gsd-planner');
|
||||
const VALID_EFFORTS = ['minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
|
||||
assert.ok(VALID_EFFORTS.includes(result));
|
||||
});
|
||||
|
||||
test('config.json has unknown agent with effort.default set -> uses effort.default', () => {
|
||||
writeConfig(tmpDir, { effort: { default: 'minimal' } });
|
||||
assert.strictEqual(resolveEffortInternal(tmpDir, 'completely-unknown-agent-98765'), 'minimal');
|
||||
});
|
||||
|
||||
test('resolve-execution with malformed config does not crash', () => {
|
||||
writeConfig(tmpDir, {
|
||||
effort: { default: null, routing_tier_defaults: null },
|
||||
fast_mode: { enabled: null, agent_overrides: null },
|
||||
});
|
||||
const result = runGsdTools(['resolve-execution', 'gsd-planner'], tmpDir, { HOME: tmpDir });
|
||||
assert.ok(result.success, `Should not crash with null config values: ${result.error}`);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Config schema: new keys are valid ───────────────────────────────────────
|
||||
|
||||
describe('#443 config schema: new effort/fast_mode keys valid', () => {
|
||||
const { isValidConfigKey } = require('../gsd-core/bin/lib/config-schema.cjs');
|
||||
|
||||
test('effort.default is a valid config key', () => {
|
||||
assert.ok(isValidConfigKey('effort.default'), 'effort.default must be valid');
|
||||
});
|
||||
|
||||
test('fast_mode.enabled is a valid config key', () => {
|
||||
assert.ok(isValidConfigKey('fast_mode.enabled'), 'fast_mode.enabled must be valid');
|
||||
});
|
||||
|
||||
test('effort.routing_tier_defaults.light is valid (dynamic pattern)', () => {
|
||||
assert.ok(isValidConfigKey('effort.routing_tier_defaults.light'));
|
||||
});
|
||||
|
||||
test('effort.routing_tier_defaults.standard is valid', () => {
|
||||
assert.ok(isValidConfigKey('effort.routing_tier_defaults.standard'));
|
||||
});
|
||||
|
||||
test('effort.routing_tier_defaults.heavy is valid', () => {
|
||||
assert.ok(isValidConfigKey('effort.routing_tier_defaults.heavy'));
|
||||
});
|
||||
|
||||
test('effort.agent_overrides.<agent-id> is valid (dynamic pattern)', () => {
|
||||
assert.ok(isValidConfigKey('effort.agent_overrides.gsd-planner'));
|
||||
assert.ok(isValidConfigKey('effort.agent_overrides.my-custom-agent'));
|
||||
});
|
||||
|
||||
test('fast_mode.routing_tier_defaults.light is valid', () => {
|
||||
assert.ok(isValidConfigKey('fast_mode.routing_tier_defaults.light'));
|
||||
});
|
||||
|
||||
test('fast_mode.agent_overrides.<agent-id> is valid', () => {
|
||||
assert.ok(isValidConfigKey('fast_mode.agent_overrides.gsd-planner'));
|
||||
});
|
||||
|
||||
test('effort.routing_tier_defaults.invalid-tier is NOT valid', () => {
|
||||
assert.ok(!isValidConfigKey('effort.routing_tier_defaults.super'));
|
||||
});
|
||||
});
|
||||
|
||||
// ─── resolve-execution arg parsing matrix (Codex adversarial finding #1) ──────
|
||||
//
|
||||
// These tests FAIL before the fix: flags-first ordering misroutes the agent.
|
||||
|
||||
describe('#443 resolve-execution: deterministic arg parsing (flags-first ordering)', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => {
|
||||
tmpDir = createTempProject();
|
||||
process.env._GSD_TEST_HOME_OVERRIDE = tmpDir;
|
||||
});
|
||||
afterEach(() => {
|
||||
cleanup(tmpDir);
|
||||
delete process.env._GSD_TEST_HOME_OVERRIDE;
|
||||
});
|
||||
|
||||
test('flags-first: --effort low gsd-planner resolves gsd-planner (NOT "low" as agent)', () => {
|
||||
// BUG: before fix, agentTypeArg = 'low' (first non-dash token) -> unknown_agent:true
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', '--effort', 'low', 'gsd-planner'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.ok(!output.unknown_agent, `agent must be resolved (not unknown_agent), got: ${JSON.stringify(output)}`);
|
||||
assert.strictEqual(output.effort, 'low', `effort should be low, got: ${output.effort}`);
|
||||
});
|
||||
|
||||
test('flags-first: --attempt 1 gsd-codebase-mapper resolves gsd-codebase-mapper (NOT "1" as agent)', () => {
|
||||
// BUG: before fix, agentTypeArg = '1' -> unknown_agent:true
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', '--attempt', '1', 'gsd-codebase-mapper'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(result.success, `Command failed: ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.ok(!output.unknown_agent, `gsd-codebase-mapper must be resolved, got: ${JSON.stringify(output)}`);
|
||||
});
|
||||
|
||||
test('agent-first parity: gsd-planner --effort low produces same effort as flags-first', () => {
|
||||
const flagsFirst = runGsdTools(
|
||||
['resolve-execution', '--effort', 'low', 'gsd-planner'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
const agentFirst = runGsdTools(
|
||||
['resolve-execution', 'gsd-planner', '--effort', 'low'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(flagsFirst.success && agentFirst.success,
|
||||
`Both orderings must succeed. flags-first err: ${flagsFirst.error} agent-first err: ${agentFirst.error}`);
|
||||
const outFF = JSON.parse(flagsFirst.output);
|
||||
const outAF = JSON.parse(agentFirst.output);
|
||||
assert.strictEqual(outFF.effort, outAF.effort, 'effort must be identical for both orderings');
|
||||
assert.strictEqual(outFF.model, outAF.model, 'model must be identical for both orderings');
|
||||
});
|
||||
|
||||
test('error: missing agent (--effort low with no positional) -> non-zero exit, no stack trace', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', '--effort', 'low'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(!result.success, 'must exit non-zero when agent is missing');
|
||||
assert.ok(!result.error.includes('at '), `error must not contain stack trace, got: ${result.error}`);
|
||||
assert.ok(result.error.length > 0, 'must emit an error message');
|
||||
});
|
||||
|
||||
test('error: two positional agents -> non-zero exit', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', 'gsd-planner', 'gsd-executor'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(!result.success, 'must exit non-zero when two agents are given');
|
||||
});
|
||||
|
||||
test('error: --attempt notanumber -> non-zero exit, clear error', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', '--attempt', 'notanumber', 'gsd-planner'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(!result.success, 'must exit non-zero for non-integer --attempt');
|
||||
assert.ok(result.error.length > 0, 'must emit an error message');
|
||||
});
|
||||
|
||||
test('error: trailing --effort (no value) -> non-zero exit', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', 'gsd-planner', '--effort'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(!result.success, 'must exit non-zero for trailing --effort with no value');
|
||||
assert.ok(result.error.length > 0, 'must emit an error message');
|
||||
});
|
||||
|
||||
test('unknown agent positional -> unknown_agent:true (preserved behavior)', () => {
|
||||
const result = runGsdTools(
|
||||
['resolve-execution', 'totally-not-an-agent'],
|
||||
tmpDir,
|
||||
{ HOME: tmpDir }
|
||||
);
|
||||
assert.ok(result.success, `Should succeed (unknown agent is valid input): ${result.error}`);
|
||||
const output = JSON.parse(result.output);
|
||||
assert.strictEqual(output.unknown_agent, true, 'unknown agent must emit unknown_agent:true');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── injectEffortFrontmatter: newline-agnostic injection (#443 Windows fix) ──
|
||||
|
||||
describe('#443 injectEffortFrontmatter: newline-agnostic YAML frontmatter injection', () => {
|
||||
// LF source (macOS / Linux git checkout) — baseline
|
||||
test('LF frontmatter: injects effort: before closing ---', () => {
|
||||
const content = '---\nname: gsd-planner\ndescription: Creates plans\ncolor: blue\n---\nBody here\n';
|
||||
const result = injectEffortFrontmatter(content, 'xhigh');
|
||||
assert.notStrictEqual(result, content, 'content should be modified');
|
||||
assert.match(result, /^effort:\s*xhigh$/m, 'effort: xhigh must be present');
|
||||
assert.ok(result.includes('\neffort: xhigh\n---\n'), 'effort: must appear before closing --- with LF');
|
||||
// Closing --- must still be present and intact
|
||||
assert.ok(result.includes('\n---\n'), 'closing --- must remain with LF');
|
||||
});
|
||||
|
||||
// CRLF source (Windows git checkout with core.autocrlf=true) — the actual bug
|
||||
test('CRLF frontmatter: injects effort: with CRLF preserved (Windows fix)', () => {
|
||||
const content = '---\r\nname: gsd-planner\r\ndescription: Creates plans\r\ncolor: blue\r\n---\r\nBody here\r\n';
|
||||
const result = injectEffortFrontmatter(content, 'xhigh');
|
||||
assert.notStrictEqual(result, content, 'content should be modified (CRLF source was silently skipped before fix)');
|
||||
// effort: line must use CRLF, not LF (EOL consistency)
|
||||
assert.ok(result.includes('effort: xhigh\r\n'), 'effort: line must use CRLF to match surrounding frontmatter');
|
||||
// Closing --- must use CRLF and remain intact
|
||||
assert.ok(result.includes('\r\neffort: xhigh\r\n---\r\n'), 'effort: must appear before closing ---\\r\\n with CRLF');
|
||||
// The effort value must be readable via multiline regex (as the install-wiring assertions do)
|
||||
assert.match(result, /^effort:\s*xhigh$/m, '/^effort:\\s*xhigh$/m must match in CRLF output');
|
||||
});
|
||||
|
||||
// Idempotency: don't double-insert if effort: already exists
|
||||
test('idempotent: does NOT insert a second effort: line when already present (LF)', () => {
|
||||
const content = '---\nname: gsd-planner\neffort: high\n---\nBody\n';
|
||||
const result = injectEffortFrontmatter(content, 'xhigh');
|
||||
assert.strictEqual(result, content, 'content must be unchanged when effort: already present');
|
||||
// Confirm no duplicate
|
||||
const matches = [...result.matchAll(/^effort:/mg)];
|
||||
assert.strictEqual(matches.length, 1, 'exactly one effort: key must exist');
|
||||
});
|
||||
|
||||
test('idempotent: does NOT insert a second effort: line when already present (CRLF)', () => {
|
||||
const content = '---\r\nname: gsd-planner\r\neffort: high\r\n---\r\nBody\r\n';
|
||||
const result = injectEffortFrontmatter(content, 'xhigh');
|
||||
assert.strictEqual(result, content, 'content must be unchanged when effort: already present (CRLF)');
|
||||
});
|
||||
|
||||
// No frontmatter — leave unchanged
|
||||
test('no YAML frontmatter: returns content unchanged', () => {
|
||||
const content = 'Just a body\nNo frontmatter here\n';
|
||||
const result = injectEffortFrontmatter(content, 'xhigh');
|
||||
assert.strictEqual(result, content, 'content without frontmatter must be returned unchanged');
|
||||
});
|
||||
|
||||
// Complex frontmatter with comment lines and color: key (mirrors real agent .md files)
|
||||
test('complex LF frontmatter (# comment + color:) still injects effort: before ---', () => {
|
||||
const content = [
|
||||
'---',
|
||||
'name: gsd-executor',
|
||||
'# hooks: see .claude/settings.json',
|
||||
'description: Executes tasks',
|
||||
'color: green',
|
||||
'---',
|
||||
'Body content here',
|
||||
'',
|
||||
].join('\n');
|
||||
const result = injectEffortFrontmatter(content, 'high');
|
||||
assert.match(result, /^effort:\s*high$/m, 'effort: high must be present');
|
||||
assert.ok(result.includes('\neffort: high\n---\n'), 'effort: must appear immediately before closing ---');
|
||||
// Other frontmatter fields must be untouched
|
||||
assert.ok(result.includes('color: green'), 'color: must be preserved');
|
||||
assert.ok(result.includes('# hooks:'), '# comment must be preserved');
|
||||
});
|
||||
|
||||
test('complex CRLF frontmatter (# comment + color:) still injects effort: with CRLF before ---', () => {
|
||||
const lines = [
|
||||
'---',
|
||||
'name: gsd-executor',
|
||||
'# hooks: see .claude/settings.json',
|
||||
'description: Executes tasks',
|
||||
'color: green',
|
||||
'---',
|
||||
'Body content here',
|
||||
'',
|
||||
];
|
||||
const content = lines.join('\r\n');
|
||||
const result = injectEffortFrontmatter(content, 'high');
|
||||
assert.ok(result.includes('effort: high\r\n'), 'effort: must use CRLF in CRLF file');
|
||||
assert.ok(result.includes('\r\neffort: high\r\n---\r\n'), 'effort: must appear before closing ---\\r\\n');
|
||||
assert.ok(result.includes('color: green\r\n'), 'color: must be preserved with CRLF');
|
||||
});
|
||||
});
|
||||
@@ -1,872 +0,0 @@
|
||||
/**
|
||||
* Feature test for issue #49 — model_policy presets.
|
||||
*
|
||||
* Adds a `model_policy` block to .planning/config.json:
|
||||
*
|
||||
* {
|
||||
* "model_policy": {
|
||||
* "provider": "anthropic-fable",
|
||||
* "budget": "high",
|
||||
* "runtime_tiers": {
|
||||
* "opencode": {
|
||||
* "opus": { "model": "anthropic/claude-opus-4-8" }
|
||||
* }
|
||||
* }
|
||||
* }
|
||||
* }
|
||||
*
|
||||
* Resolution precedence in resolveModelInternal (highest → lowest):
|
||||
* 1. model_overrides[agent] (per-agent full IDs; existing)
|
||||
* 2. model_policy.runtime_tiers[runtime][tier] (Sub-path A: explicit runtime+tier entry)
|
||||
* 3. model_policy provider preset + budget (Sub-path B: known-provider catalog lookup)
|
||||
* 4. model_profile_overrides (legacy runtime-aware overrides)
|
||||
* 5. resolve_model_ids / profile fallback
|
||||
*
|
||||
* Sub-path A (runtime_tiers) fires when config.runtime matches a key inside
|
||||
* model_policy.runtime_tiers AND that key contains an entry for the resolved tier.
|
||||
*
|
||||
* Sub-path B (provider preset) fires when model_policy.provider is a known
|
||||
* provider AND the catalog contains an entry for (tier, budget) pair.
|
||||
*
|
||||
* Both sub-paths return a string model ID. Failures in either sub-path fall
|
||||
* through cleanly to the next step in the chain.
|
||||
*
|
||||
* New config keys accepted by isValidConfigKey:
|
||||
* - model_policy.provider
|
||||
* - model_policy.budget
|
||||
* - model_policy.runtime_tiers.<runtime>.<tier>
|
||||
*
|
||||
* Backwards compatibility:
|
||||
* - model_profile_overrides continues to work when model_policy is absent.
|
||||
* - When both are set, model_policy wins (fires first).
|
||||
*
|
||||
* KNOWN_PROVIDERS is exported from both model-catalog.cjs and core.cjs (re-export).
|
||||
*
|
||||
* These tests are written to FAIL before implementation. They use typed-IR /
|
||||
* structural assertions on resolveModelInternal / resolveModelPolicy / isValidConfigKey
|
||||
* return values — not stdout / grep.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { test, describe, beforeEach, afterEach } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
// ─── Imports (will fail until implementation exists) ────────────────────────
|
||||
// resolveModelPolicy is a new internal function that must be exported from core.cjs.
|
||||
// KNOWN_PROVIDERS must be exported from model-catalog.cjs and re-exported by core.cjs.
|
||||
const {
|
||||
resolveModelInternal,
|
||||
resolveModelPolicy,
|
||||
resolveModelForTier,
|
||||
} = require('../gsd-core/bin/lib/model-resolver.cjs');
|
||||
const {
|
||||
KNOWN_PROVIDERS,
|
||||
} = require('../gsd-core/bin/lib/model-catalog.cjs');
|
||||
|
||||
// KNOWN_PROVIDERS must also be exported directly from model-catalog.cjs
|
||||
const modelCatalog = require('../gsd-core/bin/lib/model-catalog.cjs');
|
||||
|
||||
const { isValidConfigKey } = require('../gsd-core/bin/lib/config-schema.cjs');
|
||||
const { createTempDir, cleanup, resetRuntimeWarningCaches } = require('./helpers.cjs');
|
||||
|
||||
const makeTmp = (prefix) => createTempDir(`gsd-49-${prefix}-`);
|
||||
|
||||
function writeConfig(dir, config) {
|
||||
const planningDir = path.join(dir, '.planning');
|
||||
fs.mkdirSync(planningDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(planningDir, 'config.json'), JSON.stringify(config, null, 2));
|
||||
}
|
||||
|
||||
function rmr(p) {
|
||||
cleanup(p);
|
||||
}
|
||||
|
||||
// ─── resolveModelPolicy unit tests ──────────────────────────────────────────
|
||||
//
|
||||
// resolveModelPolicy(config, tier) is the pure resolver that takes a loaded
|
||||
// config object and a resolved tier string. It returns a string model ID when
|
||||
// model_policy produces a hit, or null when it falls through.
|
||||
|
||||
describe('#49 resolveModelPolicy: null/absent policy returns null', () => {
|
||||
test('resolveModelPolicy returns null when policy is null or absent', () => {
|
||||
// policy is null
|
||||
assert.strictEqual(resolveModelPolicy(null, 'opus'), null);
|
||||
// policy is undefined
|
||||
assert.strictEqual(resolveModelPolicy(undefined, 'opus'), null);
|
||||
// policy is absent (empty object treated as absent)
|
||||
assert.strictEqual(resolveModelPolicy({}, 'opus'), null);
|
||||
});
|
||||
|
||||
test('resolveModelPolicy returns null when runtime or tier is missing', () => {
|
||||
const policy = { provider: 'anthropic', budget: 'high' };
|
||||
// tier is null
|
||||
assert.strictEqual(resolveModelPolicy(policy, null), null);
|
||||
// tier is empty string
|
||||
assert.strictEqual(resolveModelPolicy(policy, ''), null);
|
||||
// tier is undefined
|
||||
assert.strictEqual(resolveModelPolicy(policy, undefined), null);
|
||||
});
|
||||
});
|
||||
|
||||
describe('#49 resolveModelPolicy Sub-path B: provider presets', () => {
|
||||
test('known provider "anthropic" + tier "opus" + budget "high" returns correct model ID', () => {
|
||||
// The anthropic preset catalog must contain an entry for opus+high.
|
||||
// The returned model ID is the high-budget anthropic opus model.
|
||||
const policy = { provider: 'anthropic', budget: 'high' };
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.ok(typeof result === 'string' && result.length > 0,
|
||||
`expected a non-empty model ID string, got: ${JSON.stringify(result)}`);
|
||||
assert.strictEqual(result, 'claude-opus-4-8',
|
||||
`expected anthropic opus/high to resolve to claude-opus-4-8, got: ${result}`);
|
||||
});
|
||||
|
||||
test('known provider "anthropic" + tier "sonnet" + budget "high" preserves Opus 4.8 routing', () => {
|
||||
const policy = { provider: 'anthropic', budget: 'high' };
|
||||
const result = resolveModelPolicy(policy, 'sonnet');
|
||||
assert.strictEqual(result, 'claude-opus-4-8',
|
||||
`expected anthropic sonnet/high to resolve to claude-opus-4-8, got: ${result}`);
|
||||
});
|
||||
|
||||
test('known provider "anthropic-fable" + tier "opus" + budget "high" resolves to Claude Fable 5', () => {
|
||||
const policy = { provider: 'anthropic-fable', budget: 'high' };
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.strictEqual(result, 'claude-fable-5',
|
||||
`expected anthropic-fable opus/high to resolve to claude-fable-5, got: ${result}`);
|
||||
});
|
||||
|
||||
test('known provider "anthropic-fable" + tier "haiku" + budget "high" keeps low tier on Sonnet', () => {
|
||||
const policy = { provider: 'anthropic-fable', budget: 'high' };
|
||||
const result = resolveModelPolicy(policy, 'haiku');
|
||||
assert.strictEqual(result, 'claude-sonnet-5',
|
||||
`expected anthropic-fable haiku/high to resolve to claude-sonnet-5, got: ${result}`);
|
||||
});
|
||||
|
||||
test('known provider "openai" + tier "sonnet" + budget "low" returns model with reasoning_effort from preset', () => {
|
||||
// The openai preset catalog must contain a sonnet+low entry.
|
||||
// "openai" maps to a different model family; the entry may include reasoning_effort.
|
||||
const policy = { provider: 'openai', budget: 'low' };
|
||||
const result = resolveModelPolicy(policy, 'sonnet');
|
||||
assert.ok(typeof result === 'string' && result.length > 0,
|
||||
`expected a non-empty model ID string for openai/sonnet/low, got: ${JSON.stringify(result)}`);
|
||||
});
|
||||
|
||||
test('budget absent defaults to "medium"', () => {
|
||||
// No "budget" key — defaults to "medium". The anthropic/opus/medium entry must exist.
|
||||
const policyWithBudget = { provider: 'anthropic', budget: 'medium' };
|
||||
const policyNoBudget = { provider: 'anthropic' };
|
||||
const withBudget = resolveModelPolicy(policyWithBudget, 'opus');
|
||||
const withoutBudget = resolveModelPolicy(policyNoBudget, 'opus');
|
||||
// Both must return a string (not null)
|
||||
assert.ok(typeof withBudget === 'string' && withBudget.length > 0,
|
||||
`expected model from explicit budget:'medium'`);
|
||||
assert.ok(typeof withoutBudget === 'string' && withoutBudget.length > 0,
|
||||
`expected model when budget absent (should default to medium)`);
|
||||
// They must resolve to the same value
|
||||
assert.strictEqual(withBudget, withoutBudget,
|
||||
'absent budget must behave identically to explicit "medium"');
|
||||
});
|
||||
|
||||
test('provider "generic" (all null entries) returns null (falls through)', () => {
|
||||
// provider:'generic' means opaque model IDs — there's no preset catalog for
|
||||
// generic. Without a runtime_tiers hit, resolveModelPolicy returns null.
|
||||
const policy = { provider: 'generic', budget: 'high' };
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.strictEqual(result, null,
|
||||
'provider:"generic" with no runtime_tiers must return null (no preset catalog)');
|
||||
});
|
||||
|
||||
test('unknown provider string returns null without throwing', () => {
|
||||
// A typo like provider:'mistral' must not crash; it degrades gracefully.
|
||||
const policy = { provider: 'mistral', budget: 'high' };
|
||||
let result;
|
||||
assert.doesNotThrow(() => {
|
||||
result = resolveModelPolicy(policy, 'opus');
|
||||
}, 'resolveModelPolicy must not throw on unknown provider');
|
||||
assert.strictEqual(result, null,
|
||||
'unknown provider with no runtime_tiers must return null');
|
||||
});
|
||||
|
||||
test('known provider + unknown tier returns null', () => {
|
||||
const policy = { provider: 'anthropic', budget: 'high' };
|
||||
const result = resolveModelPolicy(policy, 'jumbo');
|
||||
assert.strictEqual(result, null,
|
||||
'unknown tier "jumbo" must return null for anthropic provider');
|
||||
});
|
||||
|
||||
test('known provider + known tier + missing budget level returns null', () => {
|
||||
// The anthropic preset for opus only defines 'high' and 'medium' but NOT 'critical'.
|
||||
// A missing budget level must fall through (return null) — not crash.
|
||||
const policy = { provider: 'anthropic', budget: 'critical' };
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.strictEqual(result, null,
|
||||
'missing budget level "critical" must return null without throwing');
|
||||
});
|
||||
});
|
||||
|
||||
describe('#49 resolveModelPolicy Sub-path A: runtime_tiers', () => {
|
||||
test('runtime_tiers entry wins over provider preset for same runtime+tier', () => {
|
||||
// Sub-path A fires first: explicit runtime_tiers entry overrides the
|
||||
// provider preset catalog. The returned model is the one in runtime_tiers,
|
||||
// not what the provider preset would have returned.
|
||||
const policy = {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime: 'opencode',
|
||||
runtime_tiers: {
|
||||
opencode: {
|
||||
opus: { model: 'anthropic/custom-opus-override' },
|
||||
},
|
||||
},
|
||||
};
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.strictEqual(result, 'anthropic/custom-opus-override',
|
||||
'Sub-path A runtime_tiers must win over Sub-path B provider preset');
|
||||
});
|
||||
|
||||
test('runtime_tiers string shorthand normalized to { model } object', () => {
|
||||
// String shorthand: `{ opencode: { opus: "some-model-id" } }`
|
||||
// must be normalized to `{ model: "some-model-id" }` so the resolver
|
||||
// returns the string as-is.
|
||||
const policy = {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime: 'opencode',
|
||||
runtime_tiers: {
|
||||
opencode: {
|
||||
opus: 'anthropic/string-shorthand-model',
|
||||
},
|
||||
},
|
||||
};
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.strictEqual(result, 'anthropic/string-shorthand-model',
|
||||
'string shorthand in runtime_tiers must be normalized and returned as model ID');
|
||||
});
|
||||
|
||||
test('runtime_tiers partial entry (no matching runtime) falls through to provider preset', () => {
|
||||
// runtime_tiers has entries for 'copilot' but the active runtime is 'opencode'.
|
||||
// The miss on runtime_tiers falls through to Sub-path B (provider preset).
|
||||
const policy = {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime: 'opencode',
|
||||
runtime_tiers: {
|
||||
copilot: {
|
||||
opus: { model: 'some-copilot-model' },
|
||||
},
|
||||
},
|
||||
};
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
// Falls through to Sub-path B (anthropic/opus/high) — must not be null.
|
||||
assert.ok(typeof result === 'string' && result.length > 0,
|
||||
'runtime_tiers miss must fall through to provider preset, got: ' + JSON.stringify(result));
|
||||
// And it must NOT be the copilot model
|
||||
assert.notStrictEqual(result, 'some-copilot-model');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── resolveModelInternal integration tests ──────────────────────────────────
|
||||
//
|
||||
// These tests call resolveModelInternal through a temp project's config.json.
|
||||
// They verify the full resolution chain including model_policy placement.
|
||||
|
||||
describe('#49 resolveModelInternal: model_policy in the resolution chain', () => {
|
||||
let projectDir;
|
||||
beforeEach(() => {
|
||||
projectDir = makeTmp('internal');
|
||||
resetRuntimeWarningCaches();
|
||||
});
|
||||
afterEach(() => {
|
||||
rmr(projectDir);
|
||||
resetRuntimeWarningCaches();
|
||||
});
|
||||
|
||||
test('model_policy fires before model_profile_overrides when both are set (model_policy wins)', () => {
|
||||
// model_policy (Sub-path B: anthropic/opus/high) must win over
|
||||
// model_profile_overrides when both are present.
|
||||
// We use a model_profile_overrides entry that would give a DIFFERENT result.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
},
|
||||
model_profile_overrides: {
|
||||
opencode: {
|
||||
// This legacy override would have returned this model — but model_policy must win.
|
||||
opus: 'legacy-override-model-should-not-appear',
|
||||
},
|
||||
},
|
||||
});
|
||||
const result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
assert.notStrictEqual(result, 'legacy-override-model-should-not-appear',
|
||||
'model_policy must fire before model_profile_overrides and win');
|
||||
assert.ok(typeof result === 'string' && result.length > 0,
|
||||
'must return a non-empty model ID');
|
||||
assert.strictEqual(result, 'claude-opus-4-8',
|
||||
'expected anthropic preset opus/high to resolve to claude-opus-4-8');
|
||||
});
|
||||
|
||||
test('model_policy with provider:"anthropic" + budget:"high" + runtime:"opencode" resolves to preset model', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality', // gsd-planner quality = opus tier
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
},
|
||||
});
|
||||
const result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
assert.ok(typeof result === 'string' && result.length > 0,
|
||||
'expected a non-empty model ID');
|
||||
assert.strictEqual(result, 'claude-opus-4-8',
|
||||
'anthropic/opus/high must resolve to claude-opus-4-8');
|
||||
});
|
||||
|
||||
test('model_policy with provider:"anthropic-fable" + budget:"high" resolves to Fable preset model', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'anthropic-fable',
|
||||
budget: 'high',
|
||||
},
|
||||
});
|
||||
const result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
assert.strictEqual(result, 'claude-fable-5',
|
||||
'anthropic-fable/opus/high must resolve to claude-fable-5');
|
||||
});
|
||||
|
||||
test('model_policy is skipped when runtime is absent', () => {
|
||||
// No `runtime` in config — model_policy fires on any non-null policy
|
||||
// only when a runtime context is available. Without runtime, the policy
|
||||
// falls through entirely.
|
||||
// NOTE: Sub-path B (provider preset) can fire without runtime — it only
|
||||
// needs tier+budget+provider. Sub-path A requires runtime. This test
|
||||
// verifies the gating behavior described in the issue: if model_policy
|
||||
// is present but runtime is absent, provider preset Sub-path B still
|
||||
// fires (it doesn't need runtime). So "skipped" means the runtime_tiers
|
||||
// sub-path is skipped but provider preset may still fire.
|
||||
// The test asserts that resolveModelInternal does not crash and returns
|
||||
// a string regardless.
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime_tiers: {
|
||||
opencode: {
|
||||
opus: { model: 'should-not-appear-no-runtime' },
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
let result;
|
||||
assert.doesNotThrow(() => {
|
||||
result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
});
|
||||
assert.ok(typeof result === 'string',
|
||||
'resolveModelInternal must return a string even when runtime is absent');
|
||||
// The runtime_tiers entry for opencode must not appear since runtime is absent
|
||||
assert.notStrictEqual(result, 'should-not-appear-no-runtime',
|
||||
'runtime_tiers must not fire when config.runtime is absent');
|
||||
});
|
||||
|
||||
test('model_policy provider preset resolves to a Claude alias on runtime:"claude" (#1133)', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'claude',
|
||||
model_profile: 'balanced',
|
||||
model_policy: { provider: 'anthropic-fable', budget: 'high' },
|
||||
});
|
||||
// gsd-planner -> opus tier; anthropic-fable opus/high = claude-fable-5 -> alias "fable"
|
||||
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'fable');
|
||||
});
|
||||
|
||||
test('model_policy works with implicit claude runtime (no runtime key) (#1133)', () => {
|
||||
writeConfig(projectDir, {
|
||||
model_profile: 'balanced',
|
||||
model_policy: { provider: 'anthropic-fable', budget: 'high' },
|
||||
});
|
||||
// gsd-executor -> sonnet tier; anthropic-fable sonnet/high = claude-fable-5 -> "fable"
|
||||
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-executor'), 'fable');
|
||||
});
|
||||
|
||||
test('unmappable model_policy ID warns and falls back to the tier alias on claude (#1133)', () => {
|
||||
resetRuntimeWarningCaches();
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'claude',
|
||||
model_profile: 'balanced',
|
||||
model_policy: { provider: 'anthropic-fable', budget: 'low' },
|
||||
});
|
||||
// gsd-planner -> opus tier; anthropic-fable opus/low = claude-opus-4-5 (no alias) -> fall back to "opus"
|
||||
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
|
||||
});
|
||||
|
||||
test('model_policy.runtime_tiers applies on runtime:"claude", mapped to alias (#1133)', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'claude',
|
||||
model_profile: 'balanced',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime_tiers: { claude: { opus: { model: 'claude-fable-5' } } },
|
||||
},
|
||||
});
|
||||
// gsd-planner -> opus tier; runtime_tiers.claude.opus = claude-fable-5 -> "fable" (was a no-op pre-#1133)
|
||||
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'fable');
|
||||
});
|
||||
|
||||
test('model_policy maps a built-in catalog model ID to its Claude alias via MODEL_ALIAS_MAP (#1133)', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'claude',
|
||||
model_profile: 'balanced',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime_tiers: { claude: { opus: { model: 'claude-opus-4-8' } } },
|
||||
},
|
||||
});
|
||||
// gsd-planner -> opus tier; runtime_tiers.claude.opus = claude-opus-4-8 ->
|
||||
// reverse of MODEL_ALIAS_MAP -> "opus" (exercises the non-fable reverse-map path)
|
||||
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'opus');
|
||||
});
|
||||
|
||||
test('model_policy still returns full IDs on non-claude runtimes (#1133 regression)', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'balanced',
|
||||
model_policy: { provider: 'anthropic-fable', budget: 'high' },
|
||||
});
|
||||
assert.strictEqual(resolveModelInternal(projectDir, 'gsd-planner'), 'claude-fable-5');
|
||||
});
|
||||
|
||||
test('model_policy is skipped when tier:"inherit"', () => {
|
||||
// When the resolved tier is 'inherit', model_policy must not fire.
|
||||
// This mirrors the existing behavior for runtime-aware resolution.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'inherit',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
},
|
||||
});
|
||||
const result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
// With profile:'inherit', the result must be 'inherit'
|
||||
assert.strictEqual(result, 'inherit',
|
||||
'model_policy must not fire when tier is "inherit"; resolveModelInternal must return "inherit"');
|
||||
});
|
||||
|
||||
test('model_profile_overrides still resolves when model_policy is absent (legacy fallback intact)', () => {
|
||||
// No model_policy — model_profile_overrides must still work exactly as before.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_profile_overrides: {
|
||||
opencode: {
|
||||
opus: 'legacy-overridden-model',
|
||||
},
|
||||
},
|
||||
});
|
||||
const result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
assert.strictEqual(result, 'legacy-overridden-model',
|
||||
'model_profile_overrides must still win when model_policy is absent');
|
||||
});
|
||||
|
||||
test('model_policy absent + model_profile_overrides set → model_profile_overrides wins (back-compat)', () => {
|
||||
// Explicit: no model_policy key at all. model_profile_overrides is the only
|
||||
// custom config. The legacy chain must apply exactly as before this feature.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'balanced',
|
||||
model_profile_overrides: {
|
||||
opencode: {
|
||||
sonnet: 'back-compat-sonnet-model',
|
||||
},
|
||||
},
|
||||
});
|
||||
// gsd-executor has balanced/opencode -> sonnet tier
|
||||
const result = resolveModelInternal(projectDir, 'gsd-executor');
|
||||
assert.strictEqual(result, 'back-compat-sonnet-model',
|
||||
'legacy model_profile_overrides must be unaffected when model_policy is absent');
|
||||
});
|
||||
|
||||
test('model_policy present but runtime_tiers empty + provider:"generic" → falls through to model_profile_overrides', () => {
|
||||
// model_policy is a stub: runtime_tiers is empty ({}), provider is "generic".
|
||||
// The resolver must fall through all model_policy paths and land on model_profile_overrides.
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'generic',
|
||||
budget: 'high',
|
||||
runtime_tiers: {},
|
||||
},
|
||||
model_profile_overrides: {
|
||||
opencode: {
|
||||
opus: 'fallthrough-to-legacy',
|
||||
},
|
||||
},
|
||||
});
|
||||
const result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
assert.strictEqual(result, 'fallthrough-to-legacy',
|
||||
'empty runtime_tiers + generic provider must fall through to model_profile_overrides');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Warning emission tests ───────────────────────────────────────────────────
|
||||
|
||||
describe('#49 resolveModelInternal: unknown provider warning behavior', () => {
|
||||
let projectDir;
|
||||
let origWrite;
|
||||
let captured;
|
||||
|
||||
beforeEach(() => {
|
||||
projectDir = makeTmp('warnings');
|
||||
resetRuntimeWarningCaches();
|
||||
captured = [];
|
||||
origWrite = process.stderr.write.bind(process.stderr);
|
||||
process.stderr.write = (chunk) => { captured.push(String(chunk)); return true; };
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
process.stderr.write = origWrite;
|
||||
rmr(projectDir);
|
||||
resetRuntimeWarningCaches();
|
||||
});
|
||||
|
||||
test('unknown provider in model_policy → falls through to model_profile_overrides, emits stderr warning once', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'mistral',
|
||||
budget: 'high',
|
||||
},
|
||||
model_profile_overrides: {
|
||||
opencode: {
|
||||
opus: 'fallback-from-unknown-provider',
|
||||
},
|
||||
},
|
||||
});
|
||||
const result = resolveModelInternal(projectDir, 'gsd-planner');
|
||||
// Must fall through to model_profile_overrides
|
||||
assert.strictEqual(result, 'fallback-from-unknown-provider',
|
||||
'unknown provider must fall through to model_profile_overrides');
|
||||
// Must emit at least one stderr warning about the unknown provider
|
||||
const joined = captured.join('');
|
||||
assert.match(joined, /model_policy.*provider.*mistral|unknown.*provider.*mistral|mistral.*unknown/i,
|
||||
'must emit a stderr warning about the unknown provider "mistral"');
|
||||
});
|
||||
|
||||
test('unknown provider warning is deduplicated (emitted only once per config label)', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'mistral',
|
||||
budget: 'high',
|
||||
},
|
||||
});
|
||||
// Call resolveModelInternal multiple times for different agents — the
|
||||
// warning about the unknown provider must be emitted only once.
|
||||
resolveModelInternal(projectDir, 'gsd-planner');
|
||||
resolveModelInternal(projectDir, 'gsd-executor');
|
||||
resolveModelInternal(projectDir, 'gsd-verifier');
|
||||
const joined = captured.join('');
|
||||
// Count occurrences of "mistral" in the warning output
|
||||
const matches = (joined.match(/mistral/gi) || []).length;
|
||||
assert.ok(matches >= 1, 'expected at least one warning about "mistral"');
|
||||
assert.ok(matches <= 2, `warning for unknown provider must be deduplicated — saw ${matches} occurrences`);
|
||||
});
|
||||
|
||||
test('model_policy.runtime_tiers with unknown runtime emits one-shot stderr warning', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime_tiers: {
|
||||
unknownrt: {
|
||||
opus: { model: 'some-model' },
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
resolveModelInternal(projectDir, 'gsd-planner');
|
||||
const joined = captured.join('');
|
||||
// Must emit a warning about the unknown runtime key in runtime_tiers
|
||||
assert.match(joined, /unknownrt|unknown.*runtime|runtime_tiers.*unknown/i,
|
||||
'must emit a stderr warning about unknown runtime "unknownrt" in model_policy.runtime_tiers');
|
||||
});
|
||||
|
||||
test('model_policy.runtime_tiers with invalid tier name emits one-shot stderr warning', () => {
|
||||
writeConfig(projectDir, {
|
||||
runtime: 'opencode',
|
||||
model_profile: 'quality',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime_tiers: {
|
||||
opencode: {
|
||||
jumbo: { model: 'invalid-tier-model' },
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
resolveModelInternal(projectDir, 'gsd-planner');
|
||||
const joined = captured.join('');
|
||||
// Must emit a warning about the invalid tier name "jumbo"
|
||||
assert.match(joined, /jumbo|invalid.*tier|tier.*invalid|unknown.*tier/i,
|
||||
'must emit a stderr warning about invalid tier "jumbo" in model_policy.runtime_tiers.opencode');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── reasoning_effort passthrough tests ──────────────────────────────────────
|
||||
|
||||
describe('#49 reasoning_effort in model_policy entries', () => {
|
||||
let projectDir;
|
||||
beforeEach(() => { projectDir = makeTmp('effort'); });
|
||||
afterEach(() => { rmr(projectDir); });
|
||||
|
||||
test('reasoning_effort in preset entry is returned as part of the entry object (caller decides whether to emit)', () => {
|
||||
// When a provider preset includes reasoning_effort (e.g. openai opus/high),
|
||||
// resolveModelPolicy must return the full entry object (or at minimum the model
|
||||
// string) without stripping reasoning_effort internally.
|
||||
// This is checked via the internal resolveModelPolicy function directly.
|
||||
// The policy object includes a runtime_tiers entry that has reasoning_effort.
|
||||
const policy = {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime: 'opencode',
|
||||
runtime_tiers: {
|
||||
opencode: {
|
||||
opus: { model: 'anthropic/claude-opus-4-8', reasoning_effort: 'high' },
|
||||
},
|
||||
},
|
||||
};
|
||||
// resolveModelPolicy must return the model string (at minimum).
|
||||
// The caller (resolveModelInternal) is responsible for deciding what to
|
||||
// emit — the resolver just returns the model ID string.
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.strictEqual(result, 'anthropic/claude-opus-4-8',
|
||||
'resolveModelPolicy must return the model string from the runtime_tiers entry');
|
||||
});
|
||||
|
||||
test('reasoning_effort in model_policy.runtime_tiers entry is returned verbatim; renderEffortForRuntime strips it when runtime not in RUNTIMES_WITH_REASONING_EFFORT', () => {
|
||||
// The renderEffortForRuntime function (already existing) handles the stripping.
|
||||
// This test verifies the contract: resolveModelPolicy returns the model string,
|
||||
// and for runtimes not in RUNTIMES_WITH_REASONING_EFFORT, the caller must not
|
||||
// emit reasoning_effort.
|
||||
const { renderEffortForRuntime, RUNTIMES_WITH_REASONING_EFFORT } = require('../gsd-core/bin/lib/model-catalog.cjs');
|
||||
|
||||
// 'opencode' is NOT in RUNTIMES_WITH_REASONING_EFFORT (only codex has reasoning_effort in catalog)
|
||||
assert.ok(!RUNTIMES_WITH_REASONING_EFFORT.has('opencode'),
|
||||
'opencode must not be in RUNTIMES_WITH_REASONING_EFFORT for this test to be meaningful');
|
||||
|
||||
// renderEffortForRuntime for a non-effort runtime returns channel:null
|
||||
const rendered = renderEffortForRuntime('opencode', 'high');
|
||||
assert.strictEqual(rendered.channel, null,
|
||||
'renderEffortForRuntime must return channel:null for runtimes not supporting reasoning_effort');
|
||||
|
||||
// The resolveModelPolicy function returns just the model string — reasoning_effort
|
||||
// is stripped at the emit layer, not inside resolveModelPolicy.
|
||||
const policy = {
|
||||
runtime: 'opencode',
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime_tiers: {
|
||||
opencode: {
|
||||
opus: { model: 'anthropic/claude-opus-4-8', reasoning_effort: 'high' },
|
||||
},
|
||||
},
|
||||
};
|
||||
const result = resolveModelPolicy(policy, 'opus');
|
||||
assert.strictEqual(result, 'anthropic/claude-opus-4-8',
|
||||
'resolveModelPolicy must return model string; reasoning_effort is stripped downstream');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── isValidConfigKey: model_policy.* schema validation ──────────────────────
|
||||
|
||||
describe('#49 isValidConfigKey: model_policy.* keys accepted/rejected', () => {
|
||||
test('isValidConfigKey accepts "model_policy.provider"', () => {
|
||||
assert.strictEqual(isValidConfigKey('model_policy.provider'), true,
|
||||
'"model_policy.provider" must be a valid config key');
|
||||
});
|
||||
|
||||
test('isValidConfigKey accepts "model_policy.budget"', () => {
|
||||
assert.strictEqual(isValidConfigKey('model_policy.budget'), true,
|
||||
'"model_policy.budget" must be a valid config key');
|
||||
});
|
||||
|
||||
test('isValidConfigKey accepts "model_policy.runtime_tiers.opencode.opus"', () => {
|
||||
assert.strictEqual(isValidConfigKey('model_policy.runtime_tiers.opencode.opus'), true,
|
||||
'"model_policy.runtime_tiers.opencode.opus" must be a valid config key');
|
||||
});
|
||||
|
||||
test('isValidConfigKey rejects "model_policy.runtime_tiers.opencode.banana" (invalid tier)', () => {
|
||||
assert.strictEqual(isValidConfigKey('model_policy.runtime_tiers.opencode.banana'), false,
|
||||
'"model_policy.runtime_tiers.opencode.banana" must be rejected (banana is not a valid tier)');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── KNOWN_PROVIDERS export tests ─────────────────────────────────────────────
|
||||
|
||||
describe('#49 KNOWN_PROVIDERS exports from model-catalog.cjs', () => {
|
||||
test('KNOWN_PROVIDERS exported from model-catalog.cjs includes all keys from providerPresets in catalog', () => {
|
||||
// KNOWN_PROVIDERS must be a Set (or array) exported from model-catalog.cjs.
|
||||
assert.ok(KNOWN_PROVIDERS != null,
|
||||
'KNOWN_PROVIDERS must be exported from model-catalog.cjs');
|
||||
const isIterable = typeof KNOWN_PROVIDERS[Symbol.iterator] === 'function';
|
||||
assert.ok(isIterable,
|
||||
'KNOWN_PROVIDERS must be iterable (Set or array)');
|
||||
const providers = [...KNOWN_PROVIDERS];
|
||||
assert.ok(providers.length > 0,
|
||||
'KNOWN_PROVIDERS must not be empty');
|
||||
// 'anthropic' must be in the set since it is a required provider preset
|
||||
assert.ok(providers.includes('anthropic'),
|
||||
'KNOWN_PROVIDERS must include "anthropic"');
|
||||
assert.ok(providers.includes('anthropic-fable'),
|
||||
'KNOWN_PROVIDERS must include "anthropic-fable"');
|
||||
// 'generic' is a special fallback, not a real provider — it must NOT be in KNOWN_PROVIDERS
|
||||
// (KNOWN_PROVIDERS lists only providers with catalog entries)
|
||||
assert.ok(!providers.includes('generic'),
|
||||
'KNOWN_PROVIDERS must not include "generic" (it is not a catalog-backed provider)');
|
||||
});
|
||||
|
||||
test('KNOWN_PROVIDERS from model-catalog.cjs is the canonical export', () => {
|
||||
// model-catalog.cjs is the canonical source of KNOWN_PROVIDERS.
|
||||
assert.ok(modelCatalog.KNOWN_PROVIDERS != null,
|
||||
'KNOWN_PROVIDERS must be exported from model-catalog.cjs');
|
||||
const fromCatalog = [...modelCatalog.KNOWN_PROVIDERS].sort();
|
||||
const fromImport = [...KNOWN_PROVIDERS].sort();
|
||||
assert.deepStrictEqual(fromImport, fromCatalog,
|
||||
'KNOWN_PROVIDERS imported from model-catalog.cjs must match the module export');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── resolveModelPolicy: Object.hasOwn prototype-pollution guards ────────────
|
||||
|
||||
describe('#49 resolveModelPolicy: prototype-pollution guards', () => {
|
||||
test('__proto__ as provider returns null without throwing', () => {
|
||||
assert.strictEqual(resolveModelPolicy({ provider: '__proto__', budget: 'medium' }, 'sonnet'), null);
|
||||
});
|
||||
|
||||
test('constructor as provider returns null without throwing', () => {
|
||||
assert.strictEqual(resolveModelPolicy({ provider: 'constructor', budget: 'medium' }, 'sonnet'), null);
|
||||
});
|
||||
|
||||
test('__proto__ as budget returns null without throwing', () => {
|
||||
assert.strictEqual(resolveModelPolicy({ provider: 'openai', budget: '__proto__' }, 'haiku'), null);
|
||||
});
|
||||
|
||||
test('toString as budget returns null without throwing', () => {
|
||||
assert.strictEqual(resolveModelPolicy({ provider: 'openai', budget: 'toString' }, 'haiku'), null);
|
||||
});
|
||||
|
||||
test('__proto__ as runtime_tiers key returns null without throwing', () => {
|
||||
const policy = {
|
||||
runtime: '__proto__',
|
||||
runtime_tiers: { '__proto__': { haiku: { model: 'evil' } } },
|
||||
};
|
||||
assert.strictEqual(resolveModelPolicy(policy, 'haiku'), null);
|
||||
});
|
||||
|
||||
test('__proto__ as tier inside runtime_tiers returns null without throwing', () => {
|
||||
const policy = {
|
||||
runtime: 'codex',
|
||||
runtime_tiers: { codex: { '__proto__': { model: 'evil' } } },
|
||||
};
|
||||
assert.strictEqual(resolveModelPolicy(policy, '__proto__'), null);
|
||||
});
|
||||
|
||||
test('valid provider+tier+budget still resolves correctly after guards', () => {
|
||||
const result = resolveModelPolicy({ provider: 'openai', budget: 'low' }, 'haiku');
|
||||
assert.ok(typeof result === 'string' && result.length > 0,
|
||||
'valid openai/haiku/low lookup must still resolve after adding hasOwn guards');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── resolveModelForTier: model_policy beats dynamic_routing ─────────────────
|
||||
|
||||
describe('#49 resolveModelForTier: model_policy beats dynamic_routing', () => {
|
||||
let tmpDir;
|
||||
beforeEach(() => { tmpDir = makeTmp('for-tier-'); });
|
||||
afterEach(() => { rmr(tmpDir); });
|
||||
|
||||
test('model_policy wins over dynamic_routing.tier_models when both are set', () => {
|
||||
writeConfig(tmpDir, {
|
||||
runtime: 'codex',
|
||||
model_policy: { provider: 'openai', budget: 'low' },
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
},
|
||||
});
|
||||
// model_policy fires before dynamic_routing in resolveModelForTier
|
||||
const result = resolveModelForTier(tmpDir, 'gsd-executor', 0);
|
||||
// gsd-executor is standard/sonnet tier; openai+low+sonnet preset model
|
||||
assert.ok(typeof result === 'string' && result.length > 0,
|
||||
'model_policy must return a model string');
|
||||
assert.notStrictEqual(result, 'sonnet',
|
||||
'dynamic_routing tier alias must not win over model_policy');
|
||||
});
|
||||
|
||||
test('model_overrides still beats model_policy in resolveModelForTier', () => {
|
||||
writeConfig(tmpDir, {
|
||||
runtime: 'codex',
|
||||
model_policy: { provider: 'openai', budget: 'high' },
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'sonnet', heavy: 'opus' },
|
||||
},
|
||||
model_overrides: { 'gsd-planner': 'custom-model-id' },
|
||||
});
|
||||
assert.strictEqual(resolveModelForTier(tmpDir, 'gsd-planner', 0), 'custom-model-id');
|
||||
});
|
||||
|
||||
test('dynamic_routing.tier_models used normally when model_policy absent', () => {
|
||||
writeConfig(tmpDir, {
|
||||
runtime: 'codex',
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'my-custom-sonnet', heavy: 'opus' },
|
||||
},
|
||||
});
|
||||
assert.strictEqual(resolveModelForTier(tmpDir, 'gsd-executor', 0), 'my-custom-sonnet');
|
||||
});
|
||||
|
||||
test('model_policy with Claude runtime does not interrupt dynamic_routing', () => {
|
||||
// model_policy only gates on non-Claude runtimes; with runtime absent/claude,
|
||||
// dynamic_routing must still work normally.
|
||||
writeConfig(tmpDir, {
|
||||
model_policy: { provider: 'openai', budget: 'low' },
|
||||
dynamic_routing: {
|
||||
enabled: true,
|
||||
tier_models: { light: 'haiku', standard: 'my-sonnet', heavy: 'opus' },
|
||||
},
|
||||
});
|
||||
assert.strictEqual(resolveModelForTier(tmpDir, 'gsd-executor', 0), 'my-sonnet');
|
||||
});
|
||||
|
||||
test('model_policy value that is already a bare Claude alias is returned as-is on claude (#1133)', () => {
|
||||
writeConfig(tmpDir, {
|
||||
runtime: 'claude',
|
||||
model_profile: 'balanced',
|
||||
model_policy: {
|
||||
provider: 'anthropic',
|
||||
budget: 'high',
|
||||
runtime_tiers: { claude: { opus: { model: 'fable' } } },
|
||||
},
|
||||
});
|
||||
// gsd-planner → opus tier; runtime_tiers.claude.opus = "fable" is already a valid alias → "fable"
|
||||
assert.strictEqual(resolveModelInternal(tmpDir, 'gsd-planner'), 'fable');
|
||||
});
|
||||
});
|
||||
@@ -1,233 +0,0 @@
|
||||
// allow-test-rule: source-text-is-the-product #1627
|
||||
// Agent .md / reference .md files — their text IS what the runtime loads.
|
||||
// Testing text content tests the deployed contract.
|
||||
// Per CONTRIBUTING.md exception matrix.
|
||||
|
||||
/**
|
||||
* Fix #1627 — ASVS level scaling
|
||||
*
|
||||
* Asserts that `workflow.security_asvs_level` now scales both planner
|
||||
* threat-disposition rigor and auditor verification depth rather than
|
||||
* being display-only.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const AGENTS_DIR = path.join(ROOT, 'agents');
|
||||
const REFS_DIR = path.join(ROOT, 'gsd-core', 'references');
|
||||
const MANIFEST_PATH = path.join(ROOT, 'docs', 'INVENTORY-MANIFEST.json');
|
||||
|
||||
describe('SECURE: ASVS level scaling (#1627)', () => {
|
||||
// ── 1. New reference file ────────────────────────────────────────────────
|
||||
|
||||
describe('security-asvs-levels.md reference', () => {
|
||||
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
|
||||
|
||||
test('file exists', () => {
|
||||
assert.ok(fs.existsSync(refPath), 'gsd-core/references/security-asvs-levels.md must exist');
|
||||
});
|
||||
|
||||
test('defines all three levels', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
assert.ok(content.includes('L1'), 'must define L1');
|
||||
assert.ok(content.includes('L2'), 'must define L2');
|
||||
assert.ok(content.includes('L3'), 'must define L3');
|
||||
});
|
||||
|
||||
test('L1 describes opportunistic scope and planner disposition', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.toLowerCase().includes('opportunistic'),
|
||||
'L1 must be described as opportunistic'
|
||||
);
|
||||
assert.ok(
|
||||
content.includes('mitigate') && content.includes('accept'),
|
||||
'must describe mitigate/accept dispositions'
|
||||
);
|
||||
});
|
||||
|
||||
test('L2 requires explicit rationale for accepted threats', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// L2 must require documented rationale for accepted risks
|
||||
assert.ok(
|
||||
content.includes('rationale') || content.includes('documented'),
|
||||
'L2 must require documented rationale for accepted threats'
|
||||
);
|
||||
});
|
||||
|
||||
test('L3 describes deep/comprehensive verification', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
const lower = content.toLowerCase();
|
||||
assert.ok(
|
||||
lower.includes('deep') || lower.includes('comprehensive') || lower.includes('exhaustive'),
|
||||
'L3 must describe deep/comprehensive verification'
|
||||
);
|
||||
});
|
||||
|
||||
test('mentions that higher levels are supersets of lower', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
const lower = content.toLowerCase();
|
||||
assert.ok(
|
||||
lower.includes('superset') || lower.includes('higher level') || lower.includes('includes all'),
|
||||
'must note that higher levels are supersets of lower'
|
||||
);
|
||||
});
|
||||
|
||||
test('describes distinct auditor verification depth for each level', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// All three audit depth keywords should appear
|
||||
assert.ok(content.includes('grep') || content.includes('PRESENT'), 'L1 audit depth must mention grep/presence check');
|
||||
assert.ok(content.includes('boundary') || content.includes('addresses'), 'L2 audit depth must mention boundary/addresses');
|
||||
assert.ok(content.includes('end-to-end') || content.includes('bypass'), 'L3 audit depth must mention end-to-end or bypass check');
|
||||
});
|
||||
});
|
||||
|
||||
// ── 2. gsd-planner.md — no hardcoded L1 in disposition ──────────────────
|
||||
|
||||
describe('gsd-planner.md security disposition', () => {
|
||||
const plannerPath = path.join(AGENTS_DIR, 'gsd-planner.md');
|
||||
|
||||
test('planner security instruction does not hardcode "ASVS L1"', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8');
|
||||
// The old bug: "mitigate if ASVS L1 requires it" — must be gone
|
||||
assert.ok(
|
||||
!content.includes('ASVS L1 requires it'),
|
||||
'planner must not hardcode "ASVS L1 requires it"; it must reference the configured level'
|
||||
);
|
||||
});
|
||||
|
||||
test('planner references the configured OWASP ASVS level', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('OWASP ASVS level') || content.includes('configured OWASP'),
|
||||
'planner must reference the configured OWASP ASVS level'
|
||||
);
|
||||
});
|
||||
|
||||
test('planner @-references security-asvs-levels.md', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('security-asvs-levels.md'),
|
||||
'planner must @-reference security-asvs-levels.md'
|
||||
);
|
||||
});
|
||||
|
||||
test('planner is under the 49152-char cap', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8').replace(/\r\n/g, '\n').replace(/\r/g, '\n');
|
||||
assert.ok(
|
||||
content.length < 49152,
|
||||
`gsd-planner.md must be < 49152 chars (LF-normalized); got ${content.length}`
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 3. gsd-security-auditor.md — scaled verification depth ──────────────
|
||||
|
||||
describe('gsd-security-auditor.md verification depth', () => {
|
||||
const auditorPath = path.join(AGENTS_DIR, 'gsd-security-auditor.md');
|
||||
|
||||
test('auditor scales verification depth by asvs_level', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('asvs_level') || content.includes('ASVS level'),
|
||||
'auditor must reference asvs_level to scale verification'
|
||||
);
|
||||
});
|
||||
|
||||
test('auditor describes L1/L2/L3 depth differences', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
// All three levels must appear in context of depth scaling
|
||||
assert.ok(content.includes('L1'), 'auditor must mention L1 depth');
|
||||
assert.ok(content.includes('L2'), 'auditor must mention L2 depth');
|
||||
assert.ok(content.includes('L3'), 'auditor must mention L3 depth');
|
||||
});
|
||||
|
||||
test('auditor @-references security-asvs-levels.md', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('security-asvs-levels.md'),
|
||||
'auditor must @-reference security-asvs-levels.md'
|
||||
);
|
||||
});
|
||||
|
||||
test('auditor still echoes ASVS Level in structured output', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('ASVS Level:') && content.includes('{1/2/3}'),
|
||||
'auditor must still emit ASVS Level in SECURED/OPEN_THREATS output'
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 4. secure-phase.md — ASVS-aware short-circuit ──────────────────────
|
||||
|
||||
describe('secure-phase.md short-circuit conditioned on asvs_level', () => {
|
||||
const wfPath = path.join(ROOT, 'gsd-core', 'workflows', 'secure-phase.md');
|
||||
|
||||
test('short-circuit to Step 6 is gated on asvs_level == 1', () => {
|
||||
const content = fs.readFileSync(wfPath, 'utf-8');
|
||||
// The condition must reference asvs_level so that L2/L3 don't skip the auditor
|
||||
assert.ok(
|
||||
content.includes('asvs_level == 1'),
|
||||
'secure-phase.md must gate the skip-to-Step-6 short-circuit on asvs_level == 1'
|
||||
);
|
||||
});
|
||||
|
||||
test('auditor runs at L2/L3 even when threats_open is 0 (asvs_level >= 2 branch present)', () => {
|
||||
const content = fs.readFileSync(wfPath, 'utf-8');
|
||||
// The >= 2 branch must explicitly say the auditor is spawned for L2/L3 deep verification
|
||||
assert.ok(
|
||||
content.includes('asvs_level >= 2'),
|
||||
'secure-phase.md must include asvs_level >= 2 branch that does NOT skip the auditor'
|
||||
);
|
||||
// The >= 2 branch must make clear the auditor is spawned (not skipped)
|
||||
assert.ok(
|
||||
content.includes('L2/L3 deep verification') || content.includes('L2 boundary') || content.includes('L3 end-to-end'),
|
||||
'secure-phase.md asvs_level >= 2 branch must reference L2/L3 deep verification'
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 5. security-asvs-levels.md — L1 medium-severity gap closed ──────────
|
||||
|
||||
describe('security-asvs-levels.md L1 medium-severity is specified', () => {
|
||||
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
|
||||
|
||||
test('L1 explicitly handles medium-severity threats (no gap)', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// L1 section must say something about medium-severity
|
||||
assert.ok(
|
||||
content.includes('medium-severity') || content.includes('medium severity'),
|
||||
'L1 must explicitly specify disposition for medium-severity threats (no ambiguity gap)'
|
||||
);
|
||||
});
|
||||
|
||||
test('L1 medium-severity disposition is conditional (trust-boundary-aware)', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// L1 must distinguish between medium on primary trust boundary vs not
|
||||
assert.ok(
|
||||
content.includes('trust boundary') || content.includes('primary trust'),
|
||||
'L1 medium-severity rule must reference trust boundary to disambiguate disposition'
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 6. Inventory manifest ─────────────────────────────────────────────────
|
||||
|
||||
describe('inventory manifest', () => {
|
||||
test('security-asvs-levels.md is registered in INVENTORY-MANIFEST.json', () => {
|
||||
const manifest = JSON.parse(fs.readFileSync(MANIFEST_PATH, 'utf-8'));
|
||||
const refs = (manifest.families || {}).references || [];
|
||||
assert.ok(
|
||||
refs.includes('security-asvs-levels.md'),
|
||||
'security-asvs-levels.md must appear in families.references of INVENTORY-MANIFEST.json'
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -657,3 +657,163 @@ describe('bug #3784: settings.md model profile UI exposes all 5 profiles', () =>
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-33-settings-model-profile-adaptive.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-33-settings-model-profile-adaptive (consolidation epic #1969 B8 #1977)", () => {
|
||||
'use strict';
|
||||
|
||||
// allow-test-rule: source-text-is-the-product (see #33)
|
||||
// The deployed settings.md IS the product — testing its text content tests the deployed contract.
|
||||
|
||||
/**
|
||||
* Regression test for issue #33
|
||||
*
|
||||
* model_profile UI shows 4 options, schema has 5 — `adaptive` missing from
|
||||
* `settings.md` AskUserQuestion.
|
||||
*
|
||||
* The schema (gsd-core/bin/shared/model-catalog.json `profiles` array) defines 5 valid
|
||||
* model_profile values: quality, balanced, budget, adaptive, inherit. The
|
||||
* settings.md AskUserQuestion block for model_profile originally listed only 4
|
||||
* options (Quality, Balanced, Budget, Inherit) — `adaptive` was missing.
|
||||
*
|
||||
* Fix: the model_profile selection uses a two-question split. Q1 routes between
|
||||
* Adaptive / Standard-tier / Inherit (3 options). Q2 (only when Q1 = Standard)
|
||||
* asks Quality / Balanced / Budget. This keeps every individual options array
|
||||
* within the AskUserQuestion 4-option cap while making all 5 profiles reachable.
|
||||
*
|
||||
* Fixes: #33
|
||||
*/
|
||||
|
||||
const { describe, test, before } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
const SETTINGS_PATH = path.join(REPO_ROOT, 'gsd-core', 'workflows', 'settings.md');
|
||||
const CATALOG_PATH = path.join(REPO_ROOT, 'gsd-core', 'bin', 'shared', 'model-catalog.json');
|
||||
|
||||
/**
|
||||
* Collect every label: "..." value within a text block, lowercased.
|
||||
*/
|
||||
function extractOptionLabels(block) {
|
||||
const re = /label:\s*"([^"]+)"/g;
|
||||
const labels = [];
|
||||
let m;
|
||||
while ((m = re.exec(block)) !== null) {
|
||||
labels.push(m[1].toLowerCase());
|
||||
}
|
||||
return labels;
|
||||
}
|
||||
|
||||
describe('issue #33: model_profile schema and settings.md UI are in sync', () => {
|
||||
let catalog;
|
||||
let settingsContent;
|
||||
let presentBlock;
|
||||
|
||||
before(() => {
|
||||
catalog = JSON.parse(fs.readFileSync(CATALOG_PATH, 'utf-8'));
|
||||
settingsContent = fs.readFileSync(SETTINGS_PATH, 'utf-8');
|
||||
const presentMatch = settingsContent.match(/<step name="present_settings">[\s\S]*?<\/step>/);
|
||||
assert.ok(presentMatch, 'settings.md must contain a present_settings step');
|
||||
presentBlock = presentMatch[0];
|
||||
});
|
||||
|
||||
// -- (a) Schema contract ---------------------------------------------------
|
||||
|
||||
test('schema includes the adaptive model_profile value', () => {
|
||||
assert.ok(
|
||||
Array.isArray(catalog.profiles),
|
||||
'model-catalog.json must have a "profiles" array'
|
||||
);
|
||||
assert.ok(
|
||||
catalog.profiles.includes('adaptive'),
|
||||
'model-catalog should include the adaptive profile. Got: [' + catalog.profiles.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('schema includes adaptive as a model_profile value', () => {
|
||||
assert.ok(
|
||||
catalog.profiles.includes('adaptive'),
|
||||
'"adaptive" must be in model-catalog.json profiles. Got: [' + catalog.profiles.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('schema includes all expected model_profile values', () => {
|
||||
const expected = ['quality', 'balanced', 'budget', 'adaptive', 'inherit'];
|
||||
for (const profile of expected) {
|
||||
assert.ok(
|
||||
catalog.profiles.includes(profile),
|
||||
'Schema must include "' + profile + '" in profiles. Got: [' + catalog.profiles.join(', ') + ']'
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
// -- (b) UI contract — all 5 profiles reachable via present_settings -------
|
||||
|
||||
test('present_settings includes Adaptive as a selectable option (#33)', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'adaptive' || l.startsWith('adaptive')),
|
||||
'Issue #33: present_settings must include an "Adaptive" label in its model_profile AskUserQuestion options so users can select it interactively. Got labels: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Quality as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'quality' || l.startsWith('quality')),
|
||||
'present_settings must include a "Quality" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Balanced as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'balanced' || l.startsWith('balanced')),
|
||||
'present_settings must include a "Balanced" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Budget as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'budget' || l.startsWith('budget')),
|
||||
'present_settings must include a "Budget" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
test('present_settings includes Inherit as a selectable option', () => {
|
||||
const labels = extractOptionLabels(presentBlock);
|
||||
assert.ok(
|
||||
labels.some(l => l === 'inherit' || l.startsWith('inherit')),
|
||||
'present_settings must include an "Inherit" option. Got: [' + labels.join(', ') + ']'
|
||||
);
|
||||
});
|
||||
|
||||
// -- update_config and confirm steps reference adaptive --------------------
|
||||
|
||||
test('update_config step lists adaptive as a valid model_profile value', () => {
|
||||
const m = settingsContent.match(/<step name="update_config">[\s\S]*?<\/step>/);
|
||||
assert.ok(m, 'settings.md must have an update_config step');
|
||||
assert.ok(
|
||||
m[0].includes('adaptive'),
|
||||
'update_config step must list "adaptive" as a valid model_profile value'
|
||||
);
|
||||
});
|
||||
|
||||
test('confirm step table shows adaptive as a possible model profile value', () => {
|
||||
const m = settingsContent.match(/<step name="confirm">[\s\S]*?<\/step>/);
|
||||
assert.ok(m, 'settings.md must have a confirm step');
|
||||
assert.ok(
|
||||
m[0].includes('adaptive'),
|
||||
'confirm step must include "adaptive" in the Model Profile row'
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -992,3 +992,108 @@ describe('bug #2660: extractOneLinerFromBody', () => {
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/enh-72-business-context.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:enh-72-business-context (consolidation epic #1969 B8 #1977)", () => {
|
||||
// allow-test-rule: source-text-is-the-product (see #72)
|
||||
// The PROJECT.md template + complete-milestone workflow .md ARE the product surface
|
||||
// the runtime loads; asserting on their text tests the deployed contract directly.
|
||||
/**
|
||||
* Enhancement #72 — optional Business Context section in the PROJECT.md template.
|
||||
*
|
||||
* Contract tests over the product-text surfaces (template + milestone workflow .md):
|
||||
* the template offers a Business Context section that is explicitly OPTIONAL, capped
|
||||
* at the four approved one-line fields, and the milestone evolution review treats it
|
||||
* as conditional so non-business projects that deleted it are never forced to review it.
|
||||
*/
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const TEMPLATE = path.join(__dirname, '..', 'gsd-core', 'templates', 'project.md');
|
||||
const COMPLETE_MILESTONE = path.join(__dirname, '..', 'gsd-core', 'workflows', 'complete-milestone.md');
|
||||
|
||||
function parseTemplateContract(content) {
|
||||
const lines = content.split(/\r?\n/);
|
||||
const lower = content.toLowerCase();
|
||||
// The Business Context block lives between its heading and the next "## " heading.
|
||||
const startIdx = lines.findIndex(l => l.trim() === '## Business Context');
|
||||
let sectionBody = '';
|
||||
if (startIdx !== -1) {
|
||||
const rest = lines.slice(startIdx + 1);
|
||||
const endOffset = rest.findIndex(l => l.startsWith('## '));
|
||||
sectionBody = (endOffset === -1 ? rest : rest.slice(0, endOffset)).join('\n');
|
||||
}
|
||||
const fieldOf = (label) => new RegExp(`^- \\*\\*${label}\\*\\*:`, 'm').test(sectionBody);
|
||||
return {
|
||||
hasSection: startIdx !== -1,
|
||||
// Optional-by-default: an HTML comment tells non-business projects to delete it.
|
||||
hasOptionalMarker: /<!--\s*OPTIONAL/i.test(sectionBody) && /delete this section/i.test(sectionBody),
|
||||
fields: {
|
||||
customer: fieldOf('Customer'),
|
||||
revenueModel: fieldOf('Revenue model'),
|
||||
successMetric: fieldOf('Success metric'),
|
||||
strategyNotes: fieldOf('Strategy notes'),
|
||||
},
|
||||
fieldCount: (sectionBody.match(/^- \*\*/gm) || []).length,
|
||||
// Positioned between Core Value and Requirements.
|
||||
orderedBetweenCoreValueAndRequirements:
|
||||
lower.indexOf('## core value') < lower.indexOf('## business context') &&
|
||||
lower.indexOf('## business context') < lower.indexOf('## requirements'),
|
||||
hasGuidelinesEntry: /\*\*Business Context:\*\*/.test(content),
|
||||
};
|
||||
}
|
||||
|
||||
function parseMilestoneContract(content) {
|
||||
const lower = content.toLowerCase();
|
||||
const lines = content.split(/\r?\n/);
|
||||
const reviewLine = lines.find(l =>
|
||||
l.toLowerCase().includes('business context') &&
|
||||
(l.toLowerCase().includes('if present') || l.toLowerCase().includes('only if')),
|
||||
);
|
||||
return {
|
||||
mentionsBusinessContext: lower.includes('business context'),
|
||||
hasConditionalReview: Boolean(reviewLine),
|
||||
};
|
||||
}
|
||||
|
||||
describe('enhancement #72 — Business Context template section', () => {
|
||||
const tpl = parseTemplateContract(fs.readFileSync(TEMPLATE, 'utf-8'));
|
||||
|
||||
test('template includes a Business Context section', () => {
|
||||
assert.ok(tpl.hasSection, 'template must contain a "## Business Context" section');
|
||||
});
|
||||
|
||||
test('section is marked OPTIONAL with delete-for-non-business guidance', () => {
|
||||
assert.ok(tpl.hasOptionalMarker, 'section must carry an OPTIONAL HTML comment telling non-business projects to delete it');
|
||||
});
|
||||
|
||||
test('section carries exactly the four approved one-line fields', () => {
|
||||
assert.ok(tpl.fields.customer, 'missing **Customer** field');
|
||||
assert.ok(tpl.fields.revenueModel, 'missing **Revenue model** field');
|
||||
assert.ok(tpl.fields.successMetric, 'missing **Success metric** field');
|
||||
assert.ok(tpl.fields.strategyNotes, 'missing **Strategy notes** field');
|
||||
assert.strictEqual(tpl.fieldCount, 4, 'section is capped at four fields (constraint reference, not a business plan)');
|
||||
});
|
||||
|
||||
test('section is positioned between Core Value and Requirements', () => {
|
||||
assert.ok(tpl.orderedBetweenCoreValueAndRequirements, 'Business Context must sit between Core Value and Requirements');
|
||||
});
|
||||
|
||||
test('guidelines document the Business Context section', () => {
|
||||
assert.ok(tpl.hasGuidelinesEntry, 'guidelines block must include a **Business Context:** entry');
|
||||
});
|
||||
|
||||
test('milestone evolution reviews Business Context only when present', () => {
|
||||
const ms = parseMilestoneContract(fs.readFileSync(COMPLETE_MILESTONE, 'utf-8'));
|
||||
assert.ok(ms.mentionsBusinessContext, 'complete-milestone must mention Business Context in its review');
|
||||
assert.ok(ms.hasConditionalReview, 'the Business Context milestone review must be conditional on the section being present');
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
462
tests/repo-invariants.test.cjs
Normal file
462
tests/repo-invariants.test.cjs
Normal file
@@ -0,0 +1,462 @@
|
||||
'use strict';
|
||||
|
||||
// Repo-wide invariant scans.
|
||||
//
|
||||
// Consolidated home (epic #1969, batch B8 #1977) for regression tests that walk
|
||||
// whole top-level directories (docs/, agents/, commands/, workflows/, references/,
|
||||
// templates/, bin/lib/, …) asserting a CROSS-CUTTING invariant with no single
|
||||
// module subject — e.g. "no retired /gsd-next token in any doc", "no gsd-sdk
|
||||
// reference in any runtime surface", "ESLint coverage tracks the bin/lib migration",
|
||||
// "every CLI command family fails cleanly on bad input". Each block below is folded
|
||||
// verbatim from its origin issue-named file and keeps its origin issue number.
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/551-eslint-bin-lib-coverage.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:551-eslint-bin-lib-coverage (consolidation epic #1969 B8 #1977)", () => {
|
||||
'use strict';
|
||||
|
||||
/**
|
||||
* Regression / migration-gate test for #551 and ADR-457 (TS migration).
|
||||
*
|
||||
* ESLint must apply the correct policy to every gsd-core/bin/lib/*.cjs
|
||||
* file as modules migrate from hand-written CJS to tsc-generated artifacts:
|
||||
*
|
||||
* - tsc-generated artifact (has src/<basename>.cts counterpart) → MUST be
|
||||
* eslint-ignored. We lint the *.cts source instead (ADR-457).
|
||||
* - Genuinely hand-written (no src/*.cts counterpart) → MUST be linted
|
||||
* (NOT ignored). Includes scripts-generated package-identity.cjs which
|
||||
* has no *.cts source.
|
||||
*
|
||||
* The test is filesystem-driven — it scans bin/lib at runtime and checks each
|
||||
* file against the src/ directory, so it stays correct automatically as more
|
||||
* modules migrate. No hardcoded lists.
|
||||
*
|
||||
* ESLint behaviour is verified via ESLint's own `isPathIgnored()` API so the
|
||||
* test reflects real resolved flat-config precedence, not a textual scan of
|
||||
* eslint.config.mjs.
|
||||
*/
|
||||
|
||||
const { describe, test, before } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const { ESLint } = require('eslint');
|
||||
|
||||
const ROOT = path.resolve(__dirname, '..');
|
||||
const LIB_DIR = path.join(ROOT, 'gsd-core', 'bin', 'lib');
|
||||
const SRC_DIR = path.join(ROOT, 'src');
|
||||
|
||||
/**
|
||||
* Returns true if the given bin/lib/*.cjs file has a corresponding
|
||||
* src/<basename>.cts TypeScript source (meaning it is tsc-generated).
|
||||
*/
|
||||
function hasTsSource(absPath) {
|
||||
const base = path.basename(absPath, '.cjs');
|
||||
return (
|
||||
fs.existsSync(path.join(SRC_DIR, `${base}.cts`)) ||
|
||||
fs.existsSync(path.join(SRC_DIR, `${base}.ts`))
|
||||
);
|
||||
}
|
||||
|
||||
let eslint;
|
||||
before(() => {
|
||||
eslint = new ESLint({ cwd: ROOT });
|
||||
});
|
||||
|
||||
describe('ESLint coverage tracks the bin/lib TS migration (ADR-457 / #537)', () => {
|
||||
/**
|
||||
* Main invariant: scan every *.cjs in bin/lib and assert the correct ESLint
|
||||
* policy is applied.
|
||||
*/
|
||||
test('each bin/lib/*.cjs is linted xor ignored according to migration state', async () => {
|
||||
const wronglyIgnored = []; // hand-written but ignored — should be linted
|
||||
const wronglyLinted = []; // tsc-generated but not ignored — should be ignored
|
||||
|
||||
const entries = fs.readdirSync(LIB_DIR).filter((e) => e.endsWith('.cjs'));
|
||||
for (const entry of entries) {
|
||||
const abs = path.join(LIB_DIR, entry);
|
||||
const generated = hasTsSource(abs);
|
||||
const ignored = await eslint.isPathIgnored(abs);
|
||||
|
||||
if (generated && !ignored) {
|
||||
wronglyLinted.push(entry);
|
||||
} else if (!generated && ignored) {
|
||||
wronglyIgnored.push(entry);
|
||||
}
|
||||
}
|
||||
|
||||
assert.deepEqual(
|
||||
wronglyLinted,
|
||||
[],
|
||||
`tsc-generated bin/lib modules not yet added to ESLint ignore list: ${wronglyLinted.join(', ')}`,
|
||||
);
|
||||
assert.deepEqual(
|
||||
wronglyIgnored,
|
||||
[],
|
||||
`Hand-written bin/lib modules silently excluded from ESLint: ${wronglyIgnored.join(', ')}`,
|
||||
);
|
||||
});
|
||||
|
||||
test('semver-compare.cjs (tsc-generated publish artifact) stays eslint-ignored (ADR-457)', async () => {
|
||||
const f = path.join(LIB_DIR, 'semver-compare.cjs');
|
||||
assert.equal(
|
||||
await eslint.isPathIgnored(f),
|
||||
true,
|
||||
'semver-compare.cjs is a tsc-generated publish-time artifact and must stay ignored',
|
||||
);
|
||||
});
|
||||
|
||||
test('package-identity.cjs (script-generated, no *.cts source) is linted, not ignored (#551)', async () => {
|
||||
const f = path.join(LIB_DIR, 'package-identity.cjs');
|
||||
assert.equal(
|
||||
await eslint.isPathIgnored(f),
|
||||
false,
|
||||
'package-identity.cjs has no src/*.cts counterpart and must be linted, not ignored',
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-3054-stale-gsd-next-references.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-3054-stale-gsd-next-references (consolidation epic #1969 B8 #1977)", () => {
|
||||
'use strict';
|
||||
|
||||
const { test, describe } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
function walkMd(dir, out = []) {
|
||||
if (!fs.existsSync(dir)) return out;
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) walkMd(full, out);
|
||||
else if (entry.name.endsWith('.md')) out.push(full);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function extractSlashCommandTokens(markdown) {
|
||||
const tokenRe = /\/gsd-[a-z0-9-]+/gi;
|
||||
const tokens = new Set();
|
||||
let m;
|
||||
while ((m = tokenRe.exec(markdown)) !== null) {
|
||||
tokens.add(m[0]);
|
||||
}
|
||||
return tokens;
|
||||
}
|
||||
|
||||
describe('bug #3054: user-facing docs should not reference removed /gsd-next command', () => {
|
||||
test('docs, workflows, and README surfaces use /gsd-progress --next instead', () => {
|
||||
const root = path.join(__dirname, '..');
|
||||
const files = [
|
||||
...walkMd(path.join(root, 'docs')),
|
||||
...walkMd(path.join(root, 'gsd-core', 'workflows')),
|
||||
...fs.readdirSync(root).filter((f) => /^README.*\.md$/.test(f)).map((f) => path.join(root, f)),
|
||||
];
|
||||
|
||||
const offenders = [];
|
||||
for (const file of files) {
|
||||
const content = fs.readFileSync(file, 'utf8');
|
||||
const tokens = extractSlashCommandTokens(content);
|
||||
if (tokens.has('/gsd-next')) offenders.push(path.relative(root, file));
|
||||
}
|
||||
|
||||
assert.deepStrictEqual(offenders, [], `stale /gsd-next references remain in: ${offenders.join(', ')}`);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-3810-no-gsd-sdk-runtime-refs.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-3810-no-gsd-sdk-runtime-refs (consolidation epic #1969 B8 #1977)", () => {
|
||||
// allow-test-rule: source-text-is-the-product (see #3810)
|
||||
// Runtime prompt/hook files are deployed verbatim — their text IS what the
|
||||
// runtime loads and executes. Asserting that text carries no retired `gsd-sdk`
|
||||
// reference tests the deployed contract, which no behavioral seam can observe
|
||||
// (there is no runtime API that enumerates "did any shipped prompt name the
|
||||
// removed SDK binary").
|
||||
|
||||
/**
|
||||
* Regression guard: no `gsd-sdk` references in runtime-facing surfaces (#339).
|
||||
*
|
||||
* The `@opengsd/gsd-sdk` package and its `gsd-sdk` binary were retired (ADR 0174,
|
||||
* #191). The bulk runtime cleanup is already done — this test locks it in so a
|
||||
* `gsd-sdk` / `GSD_SDK` reference cannot creep back into a shipped prompt or hook
|
||||
* and re-introduce drift between the documented surface and the supported
|
||||
* `gsd-tools` binary.
|
||||
*
|
||||
* Scope: runtime surfaces only — the prompts and hooks the installer ships into
|
||||
* a user's runtime config dir. Explicitly NOT covered here:
|
||||
* - `bin/install.js` — installer code, not a runtime-deployed prompt/hook
|
||||
* surface. (It carries zero `gsd-sdk` references today; the SDK-shim
|
||||
* verification subsystem was removed in #515 and the shim retired in #522.)
|
||||
* - `gsd-core/bin/` — executable library code, not deployed prompt text; it
|
||||
* may legitimately reference the SDK retirement in comments.
|
||||
* - `tests/`, `docs/`, `.changeset/`, CI/lint scripts — legitimately reference
|
||||
* the SDK retirement as history or detect its stale artifacts.
|
||||
*
|
||||
* Complements `tests/gsd-tools-path-refs.test.cjs`, which only catches the
|
||||
* `gsd-sdk query` binary-invocation form; this catches ANY runtime reference.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
|
||||
// Runtime surfaces the installer ships. Each entry is { dir, exts } — dir is
|
||||
// repo-relative, exts is the set of file extensions whose text is deployed.
|
||||
const RUNTIME_SURFACES = [
|
||||
// .md prompts plus the non-.md runtime artifacts this dir also ships:
|
||||
// _runtime-launcher.snippet.sh (the canonical launcher synced into every hook
|
||||
// by scripts/sync-runtime-launcher.cjs) and discuss-phase/templates/*.json
|
||||
// (loaded at runtime by discuss-phase.md). Scanning only .md left these two
|
||||
// deployed files uncovered. (#691 review)
|
||||
{ dir: path.join('gsd-core', 'workflows'), exts: ['.md', '.sh', '.json'] },
|
||||
{ dir: path.join('gsd-core', 'references'), exts: ['.md'] },
|
||||
// Prompt surfaces the installer deep-copies and the runtime loads via
|
||||
// `@~/.claude/gsd-core/templates/*.md` anchors in workflows/commands; the
|
||||
// lone config.json under templates/ ships too. (#691 review)
|
||||
{ dir: path.join('gsd-core', 'templates'), exts: ['.md', '.json'] },
|
||||
{ dir: path.join('gsd-core', 'contexts'), exts: ['.md'] },
|
||||
{ dir: path.join('commands', 'gsd'), exts: ['.md'] },
|
||||
{ dir: 'agents', exts: ['.md'] },
|
||||
// Hooks ship as executable text (.js/.cjs/.sh). `hooks/dist/` is a gitignored
|
||||
// build artifact regenerated from these sources, so scanning the sources is
|
||||
// sufficient and avoids asserting against generated copies.
|
||||
{ dir: 'hooks', exts: ['.js', '.cjs', '.sh'], skipDirs: ['dist'] },
|
||||
];
|
||||
|
||||
// Matches every casing/separator variant of the retired SDK token:
|
||||
// gsd-sdk, gsd_sdk, GSD-SDK, GSD_SDK, etc.
|
||||
const SDK_REF = /gsd[-_]sdk/i;
|
||||
|
||||
/**
|
||||
* Recursively collect files under `absDir` whose extension is in `exts`,
|
||||
* skipping any directory name listed in `skipDirs`.
|
||||
*/
|
||||
function collectFiles(absDir, exts, skipDirs) {
|
||||
if (!fs.existsSync(absDir)) return [];
|
||||
const out = [];
|
||||
for (const entry of fs.readdirSync(absDir, { withFileTypes: true })) {
|
||||
if (entry.isDirectory()) {
|
||||
if (skipDirs.includes(entry.name)) continue;
|
||||
out.push(...collectFiles(path.join(absDir, entry.name), exts, skipDirs));
|
||||
} else if (entry.isFile() && exts.includes(path.extname(entry.name))) {
|
||||
out.push(path.join(absDir, entry.name));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function rel(file) {
|
||||
return path.relative(REPO_ROOT, file).split(path.sep).join('/');
|
||||
}
|
||||
|
||||
describe('#339 no gsd-sdk references in runtime surfaces', () => {
|
||||
test('shipped prompts and hooks carry no retired gsd-sdk reference', () => {
|
||||
const violations = [];
|
||||
|
||||
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
|
||||
const files = collectFiles(path.join(REPO_ROOT, dir), exts, skipDirs);
|
||||
for (const file of files) {
|
||||
const lines = fs.readFileSync(file, 'utf-8').split(/\r?\n/);
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
if (SDK_REF.test(lines[i])) {
|
||||
violations.push(`${rel(file)}:${i + 1}: ${lines[i].trim()}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
assert.strictEqual(
|
||||
violations.length,
|
||||
0,
|
||||
'Runtime surfaces must not reference the retired gsd-sdk binary/package — ' +
|
||||
'use gsd-tools instead.\nViolations:\n' + violations.join('\n')
|
||||
);
|
||||
});
|
||||
|
||||
test('at least one file per configured extension is scanned (guards against an empty sweep)', () => {
|
||||
// A path typo, directory rename, or stale extension could silently make
|
||||
// collectFiles() return [] for part of a surface, turning the guard above
|
||||
// into a no-op that always passes. Checking per-surface isn't enough: for
|
||||
// gsd-core/workflows the .md files alone keep a per-surface count > 0, so
|
||||
// dropping .sh/.json would stop covering _runtime-launcher.snippet.sh and
|
||||
// discuss-phase/templates/*.json while the test stayed green. Assert each
|
||||
// configured extension actually resolves to scanned files. (#691 review)
|
||||
for (const { dir, exts, skipDirs = [] } of RUNTIME_SURFACES) {
|
||||
for (const ext of exts) {
|
||||
const count = collectFiles(path.join(REPO_ROOT, dir), [ext], skipDirs).length;
|
||||
assert.ok(
|
||||
count > 0,
|
||||
`Runtime surface "${dir}" resolved to 0 "${ext}" files — the path may ` +
|
||||
'have moved or the extension is stale; update RUNTIME_SURFACES so the ' +
|
||||
'guard keeps covering it.'
|
||||
);
|
||||
}
|
||||
}
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/feat-3593-cli-negative-universal.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:feat-3593-cli-negative-universal (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Universal CLI negative-matrix sweep across the seven command families
|
||||
* named in #3593: phase, roadmap, state, config, workstream, init,
|
||||
* validate.
|
||||
*
|
||||
* The full per-case matrix for each family belongs in dedicated files
|
||||
* (the `config` family is the template — see
|
||||
* `feat-3593-cli-negative-config.test.cjs`). This file pins a much
|
||||
* narrower contract that EVERY family must satisfy:
|
||||
*
|
||||
* 1. Bare top-level command with no subcommand: must not crash with
|
||||
* a V8 stack trace, must exit non-zero (or, where the bare form
|
||||
* is legitimate, return a clean payload).
|
||||
* 2. Unknown subcommand at command depth: must emit a typed reason
|
||||
* under --json-errors and must not crash.
|
||||
* 3. Shell-metacharacter as an argv element: must not be executed.
|
||||
* Sentinel-file probe in the project temp dir proves this.
|
||||
*
|
||||
* Future per-family files will deepen each family's matrix; this file
|
||||
* is the floor.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const { runCli } = require('./helpers/cli-negative.cjs');
|
||||
const { createTempProject, createTempGitProject, cleanup } = require('./helpers.cjs');
|
||||
|
||||
/**
|
||||
* Each entry names a top-level command, a representative subcommand
|
||||
* (for the "unknown sub" probe), and the temp-project factory.
|
||||
*
|
||||
* The `representativeSubcommand` is not asserted on directly — it
|
||||
* exists so the test layout reads "for each family, run probe X" and
|
||||
* so a future maintainer adding a family doesn't need to invent one.
|
||||
*/
|
||||
const FAMILIES = [
|
||||
{ name: 'phase', bare: ['phase'], unknown: ['phase', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'roadmap', bare: ['roadmap'], unknown: ['roadmap', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'state', bare: ['state'], unknown: ['state', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'config', bare: ['config'], unknown: ['config', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'workstream', bare: ['workstream'], unknown: ['workstream', 'this-sub-does-not-exist'], projectFactory: createTempGitProject },
|
||||
{ name: 'init', bare: ['init'], unknown: ['init', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
{ name: 'validate', bare: ['validate'], unknown: ['validate', 'this-sub-does-not-exist'], projectFactory: createTempProject },
|
||||
];
|
||||
|
||||
describe('feat-3593: bare top-level command does not crash', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
test(`${fam.name}: bare invocation exits cleanly without a stack trace`, (t) => {
|
||||
const projectDir = fam.projectFactory(`cli-neg-univ-bare-${fam.name}-`);
|
||||
t.after(() => cleanup(projectDir));
|
||||
const result = runCli(fam.bare, { cwd: projectDir });
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name} bare must not leak a V8 stack frame`);
|
||||
assert.equal(result.signal, null, `${fam.name} bare must not be killed by a signal`);
|
||||
// We do not pin exit code here: some families legitimately treat the
|
||||
// bare form as a list/status command (status 0); others reject it as
|
||||
// a usage error (status ≠ 0). What we pin is "no crash."
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe('feat-3593: unknown subcommand emits a typed reason', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
test(`${fam.name}: unknown subcommand fails with reason set and no stack trace`, (t) => {
|
||||
const projectDir = fam.projectFactory(`cli-neg-univ-unk-${fam.name}-`);
|
||||
t.after(() => cleanup(projectDir));
|
||||
const result = runCli(fam.unknown, { cwd: projectDir });
|
||||
assert.notEqual(result.status, 0, `${fam.name} unknown sub must exit non-zero`);
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name} unknown sub must not leak a stack frame`);
|
||||
// When --json-errors is on, every typed-failure path lands a non-empty
|
||||
// reason string from ERROR_REASON. A null reason here means the family
|
||||
// is using throw/console.error somewhere — a TDD signal to wire that
|
||||
// failure path through error(msg, ERROR_REASON.X).
|
||||
assert.equal(typeof result.reason, 'string', `${fam.name}: reason must be a typed enum string (got: ${result.reason})`);
|
||||
assert.notEqual(result.reason, '', `${fam.name}: reason must be non-empty`);
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe('feat-3593: shell-metacharacter argv values are NOT executed', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
test(`${fam.name}: shell-payload as subcommand argv does NOT execute the payload`, (t) => {
|
||||
const projectDir = fam.projectFactory(`cli-neg-univ-shell-${fam.name}-`);
|
||||
t.after(() => cleanup(projectDir));
|
||||
// Place the payload where the subcommand value goes. The payload
|
||||
// would, if shell-interpreted, create a sentinel file in the
|
||||
// project dir. Argv-based invocation must treat it as opaque text.
|
||||
const sentinelPayload = `$(touch ${projectDir}/INJ-${fam.name})`;
|
||||
const argv = [fam.name, sentinelPayload];
|
||||
const result = runCli(argv, { cwd: projectDir });
|
||||
// No stack trace — opaque text must not crash.
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name}: shell payload must not crash`);
|
||||
// The sentinel file must NOT exist. We check the project dir's
|
||||
// listing rather than fs.existsSync of a single path so the test
|
||||
// surfaces any spelling drift.
|
||||
const entries = fs.readdirSync(projectDir);
|
||||
const sentinels = entries.filter((n) => n.startsWith('INJ-'));
|
||||
assert.deepEqual(
|
||||
sentinels,
|
||||
[],
|
||||
`${fam.name}: shell payload was executed — sentinel files exist: ${sentinels.join(', ')}`,
|
||||
);
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// ─── Cross-family invariants on the global --cwd flag ──────────────────────
|
||||
|
||||
test('--cwd with an empty value fails the same way regardless of command family', () => {
|
||||
for (const fam of FAMILIES) {
|
||||
const result = runCli(['--cwd', '', ...fam.bare], { cwd: process.cwd() });
|
||||
assert.notEqual(result.status, 0, `${fam.name}: empty --cwd must fail`);
|
||||
assert.equal(result.hasStackTrace, false, `${fam.name}: empty --cwd must not crash`);
|
||||
assert.equal(result.reason, 'usage', `${fam.name}: empty --cwd reason should be 'usage', got: ${result.reason}`);
|
||||
}
|
||||
});
|
||||
|
||||
test('--cwd pointing at a non-existent path fails uniformly across families', () => {
|
||||
const nonExistent = path.join(require('os').tmpdir(), 'cli-neg-univ-no-such-' + Date.now() + '-' + Math.random());
|
||||
assert.equal(fs.existsSync(nonExistent), false, 'pre-check: temp path must not exist');
|
||||
for (const fam of FAMILIES) {
|
||||
const result = runCli(['--cwd', nonExistent, ...fam.bare], { cwd: process.cwd() });
|
||||
assert.notEqual(result.status, 0, `${fam.name}: invalid --cwd must fail`);
|
||||
assert.equal(result.hasStackTrace, false);
|
||||
assert.equal(result.reason, 'usage', `${fam.name}: invalid --cwd reason should be 'usage'`);
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
@@ -432,3 +432,90 @@ describe('gsd-project-researcher agent registration (#2419)', () => {
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-2559-stale-search-year.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-2559-stale-search-year (consolidation epic #1969 B8 #1977)", () => {
|
||||
// allow-test-rule: source-text-is-the-product (see #2559)
|
||||
// Workflow .md / agent .md / command .md / reference .md files — their text
|
||||
// IS what the runtime loads. Testing text content tests the deployed contract.
|
||||
// Per CONTRIBUTING.md exception matrix.
|
||||
|
||||
/**
|
||||
* Bug #2559: Stale document references in Research phase
|
||||
*
|
||||
* The gsd-phase-researcher and gsd-project-researcher agents instruct
|
||||
* WebSearch queries to always include "current year" (or a hardcoded
|
||||
* year). This biases results toward stale dated content as time passes
|
||||
* (e.g., a 2024 query run in 2026 returns stale results).
|
||||
*
|
||||
* Fix: Remove year-injection instructions from research agent
|
||||
* WebSearch guidance so searches return current results.
|
||||
*/
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const PHASE_RESEARCHER = path.join(
|
||||
__dirname,
|
||||
'..',
|
||||
'agents',
|
||||
'gsd-phase-researcher.md'
|
||||
);
|
||||
const PROJECT_RESEARCHER = path.join(
|
||||
__dirname,
|
||||
'..',
|
||||
'agents',
|
||||
'gsd-project-researcher.md'
|
||||
);
|
||||
|
||||
const FILES = [
|
||||
{ label: 'gsd-phase-researcher.md', path: PHASE_RESEARCHER },
|
||||
{ label: 'gsd-project-researcher.md', path: PROJECT_RESEARCHER },
|
||||
];
|
||||
|
||||
describe('research agents do not inject year into web searches (#2559)', () => {
|
||||
for (const { label, path: filePath } of FILES) {
|
||||
test(`${label} contains no CURRENT_YEAR placeholder`, () => {
|
||||
const content = fs.readFileSync(filePath, 'utf-8');
|
||||
assert.ok(
|
||||
!/CURRENT_YEAR/.test(content),
|
||||
`${label} must not contain CURRENT_YEAR placeholder (causes stale-year injection)`
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} contains no hardcoded year in web search instructions`, () => {
|
||||
const content = fs.readFileSync(filePath, 'utf-8');
|
||||
const match = content.match(/\b20(2[3-9]|[3-9]\d)\b/);
|
||||
assert.ok(
|
||||
!match,
|
||||
`${label} must not contain hardcoded year (found "${match && match[0]}") — biases searches toward stale content`
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} does not instruct searches to include year or current year`, () => {
|
||||
const content = fs.readFileSync(filePath, 'utf-8');
|
||||
// Match phrases like "include current year", "year in searches",
|
||||
// "[current year]", "with year", etc.
|
||||
const patterns = [
|
||||
/include\s+(?:the\s+)?current\s+year/i,
|
||||
/current\s+year/i,
|
||||
/year\s+in\s+(?:searches|queries)/i,
|
||||
/\[current year\]/i,
|
||||
];
|
||||
for (const pat of patterns) {
|
||||
assert.ok(
|
||||
!pat.test(content),
|
||||
`${label} must not instruct year injection (matched /${pat.source}/)`
|
||||
);
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1557,3 +1557,287 @@ describe('scanPhasePlans — call-site parity on mixed fixture', () => {
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/feat-3594-parser-adversarial-roadmap.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:feat-3594-parser-adversarial-roadmap (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Adversarial roadmap-parser tests (#3594).
|
||||
*
|
||||
* Loads each fixture in `tests/fixtures/adversarial/roadmap/` as the
|
||||
* project's `.planning/ROADMAP.md` and pins invariants on the public
|
||||
* `gsd-tools roadmap get-phase <N>` surface — which routes through the
|
||||
* SDK bridge when available and the CJS handler otherwise.
|
||||
*
|
||||
* Per CONTRIBUTING.md §"Testing Standards / Parser and project-file
|
||||
* inputs", the assertion target is the typed JSON shape the CLI emits,
|
||||
* not stderr prose. The harness in `tests/helpers/cli-negative.cjs`
|
||||
* (introduced by #3627 / #3593) is reused here for consistency.
|
||||
*
|
||||
* Several fixtures encode known historical regressions:
|
||||
* - fenced-code-block headings shadowing real phases (#2787)
|
||||
* - decimal phase prefix collisions (#3537)
|
||||
* - HTML-comment heading false positives
|
||||
*
|
||||
* Pre-existing parser bugs surfaced by these fixtures are NOT fixed in
|
||||
* this PR — fixing them is out of scope for "add adversarial test
|
||||
* coverage." Where a fixture exposes a still-open bug, the test
|
||||
* asserts the *currently observed* behavior with a comment naming the
|
||||
* open issue, so the flip from RED→GREEN is a one-line change the day
|
||||
* the real fix lands.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const { runCli } = require('./helpers/cli-negative.cjs');
|
||||
const { createTempProject, cleanup } = require('./helpers.cjs');
|
||||
|
||||
const FIXTURE_DIR = path.join(__dirname, 'fixtures', 'adversarial', 'roadmap');
|
||||
|
||||
function loadFixture(name) {
|
||||
return fs.readFileSync(path.join(FIXTURE_DIR, name), 'utf-8');
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a temp project whose ROADMAP.md is the named fixture's content.
|
||||
* Returns the project directory; caller is responsible for cleanup.
|
||||
*/
|
||||
function projectWithFixture(t, fixtureName) {
|
||||
const projectDir = createTempProject('roadmap-adv-' + fixtureName.replace(/\W+/g, '-') + '-');
|
||||
t.after(() => cleanup(projectDir));
|
||||
fs.writeFileSync(path.join(projectDir, '.planning', 'ROADMAP.md'), loadFixture(fixtureName));
|
||||
return projectDir;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run `gsd-tools roadmap get-phase <N>` and parse the JSON payload.
|
||||
* Returns `{ ok, exit, parsed, raw }` so tests can assert on either
|
||||
* the exit code or the structured payload.
|
||||
*/
|
||||
function getPhase(projectDir, phaseNum) {
|
||||
// No --json-errors — the get-phase command outputs JSON on success
|
||||
// via the normal stdout path. Reading the parsed payload is what the
|
||||
// workflows downstream do, so that's what we test.
|
||||
const result = runCli(['roadmap', 'get-phase', phaseNum], { cwd: projectDir, jsonErrors: false });
|
||||
let parsed = null;
|
||||
try {
|
||||
parsed = JSON.parse(result.stdout);
|
||||
} catch {
|
||||
// Leave parsed null; tests that depend on it must handle that.
|
||||
}
|
||||
return {
|
||||
exit: result.status,
|
||||
ok: result.status === 0,
|
||||
parsed,
|
||||
raw: result.stdout,
|
||||
stderr: result.stderr,
|
||||
hasStackTrace: result.hasStackTrace,
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Fenced code block heading shadowing ────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser and fenced-code-block headings (#2787)', () => {
|
||||
test('phase 1 in real prose is found even when ## Phase 999 appears inside a ``` block', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.equal(result.hasStackTrace, false, 'no V8 stack trace');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true, 'phase 1 must be found');
|
||||
assert.equal(result.parsed.phase_number, '1');
|
||||
assert.match(result.parsed.phase_name, /real phase one/);
|
||||
});
|
||||
|
||||
test('phase 999 inside a fenced block is ignored', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'phase-heading-inside-fenced-code.md');
|
||||
const result = getPhase(projectDir, '999');
|
||||
assert.equal(result.hasStackTrace, false, 'no stack trace');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, false, 'phase headings inside fenced blocks must not be parsed');
|
||||
});
|
||||
|
||||
test('fenced example heading does not shadow the real phase details and backlog phase stays unresolved (#1588)', (t) => {
|
||||
const projectDir = createTempProject('roadmap-1588-');
|
||||
t.after(() => cleanup(projectDir));
|
||||
fs.writeFileSync(
|
||||
path.join(projectDir, '.planning', 'STATE.md'),
|
||||
[
|
||||
'---',
|
||||
'gsd_state_version: 1.0',
|
||||
'milestone: v1.1',
|
||||
'status: planning',
|
||||
'---',
|
||||
'',
|
||||
].join('\n')
|
||||
);
|
||||
fs.writeFileSync(
|
||||
path.join(projectDir, '.planning', 'ROADMAP.md'),
|
||||
[
|
||||
'# Roadmap',
|
||||
'',
|
||||
'<details open>',
|
||||
'<summary>v1.1 Current (Phases 8-9) - PLANNED</summary>',
|
||||
'',
|
||||
'- [ ] **Phase 9: Real Phase**',
|
||||
'',
|
||||
'</details>',
|
||||
'',
|
||||
'## Phase Details',
|
||||
'',
|
||||
'```markdown',
|
||||
'### Phase 9: Fenced Example Phase',
|
||||
'**Goal:** This example must not be treated as roadmap structure.',
|
||||
'```',
|
||||
'',
|
||||
'### Phase 9: Real Phase',
|
||||
'**Goal:** Use the real phase details outside the fenced block.',
|
||||
'**Requirements:** REAL-01',
|
||||
'',
|
||||
'## Backlog',
|
||||
'',
|
||||
'### Phase 999.1: Backlog Thing',
|
||||
'**Goal:** Future backlog item.',
|
||||
'',
|
||||
].join('\n')
|
||||
);
|
||||
|
||||
const phase9 = getPhase(projectDir, '9');
|
||||
assert.ok(phase9.parsed, `expected JSON payload, got: ${phase9.raw}`);
|
||||
assert.equal(phase9.parsed.found, true, 'phase 9 must be found');
|
||||
assert.equal(phase9.parsed.phase_name, 'Real Phase');
|
||||
assert.equal(phase9.parsed.goal, 'Use the real phase details outside the fenced block.');
|
||||
assert.match(phase9.parsed.section, /REAL-01/, 'real phase section must be returned');
|
||||
|
||||
const backlog = getPhase(projectDir, '999.1');
|
||||
assert.ok(backlog.parsed, `expected JSON payload, got: ${backlog.raw}`);
|
||||
assert.equal(backlog.parsed.found, false, 'backlog sentinel phase must not resolve as an active roadmap phase');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Decimal phase prefix collisions ────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser handles decimal phase prefix collisions (#3537)', () => {
|
||||
test('asking for phase "2" returns the integer phase, NOT phase 2.1 or 2.10', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '2');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '2');
|
||||
assert.match(result.parsed.phase_name, /integer phase two/);
|
||||
});
|
||||
|
||||
test('asking for phase "2.1" returns the decimal child', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '2.1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '2.1');
|
||||
assert.match(result.parsed.phase_name, /decimal child/);
|
||||
});
|
||||
|
||||
test('asking for phase "2.10" returns the decimal sibling, NOT phase 2.1', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '2.10');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '2.10');
|
||||
assert.match(result.parsed.phase_name, /decimal phase 2\.10/);
|
||||
});
|
||||
|
||||
test('asking for phase "21" returns phase 21, NOT phase 2 (prefix-collision guard)', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'decimal-phase-mixed.md');
|
||||
const result = getPhase(projectDir, '21');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_number, '21');
|
||||
assert.match(result.parsed.phase_name, /phase twenty-one/);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Unicode phase titles ───────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser preserves Unicode phase titles', () => {
|
||||
test('Japanese title round-trips through phase_name', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.phase_name, '日本語フェーズ — initial setup');
|
||||
});
|
||||
|
||||
test('emoji + smart-quote title survives', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
|
||||
const result = getPhase(projectDir, '2');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.match(result.parsed.phase_name, /🚧/);
|
||||
assert.match(result.parsed.phase_name, /Émile/);
|
||||
});
|
||||
|
||||
test('Greek-letter title survives', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'unicode-phase-titles.md');
|
||||
const result = getPhase(projectDir, '3');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.phase_name, 'αβγ δεζ ηθι');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Repeated phase IDs ─────────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser handles repeated phase IDs deterministically', () => {
|
||||
test('two declarations of phase 1: parser returns the FIRST match (current behavior)', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'repeated-phase-ids.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
// The regex uses `content.match(...)` which returns the FIRST match.
|
||||
// Pin that — a future change to last-wins or de-dup would fire.
|
||||
assert.match(result.parsed.phase_name, /first declaration/);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── HTML comments ──────────────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser and HTML-commented headings', () => {
|
||||
test('phase 1 in real prose is found even when phase 998/999 appear inside <!-- ... -->', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
|
||||
const result = getPhase(projectDir, '1');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, true);
|
||||
assert.equal(result.parsed.phase_name, 'real phase');
|
||||
});
|
||||
|
||||
test('phase 999 inside an HTML comment remains ignored because backlog sentinels never resolve', (t) => {
|
||||
const projectDir = projectWithFixture(t, 'markdown-headings-inside-html-comment.md');
|
||||
const result = getPhase(projectDir, '999');
|
||||
assert.equal(result.hasStackTrace, false, 'no stack trace');
|
||||
assert.ok(result.parsed, `expected JSON payload, got: ${result.raw}`);
|
||||
assert.equal(result.parsed.found, false, 'backlog sentinel phases must not resolve');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Cross-corpus invariant ────────────────────────────────────────────────
|
||||
|
||||
describe('feat-3594: roadmap parser does not crash on ANY corpus fixture', () => {
|
||||
const fixtures = fs.readdirSync(FIXTURE_DIR).filter((f) => f.endsWith('.md') && f !== 'README.md');
|
||||
for (const fixture of fixtures) {
|
||||
test(`fixture "${fixture}" — get-phase with arbitrary IDs must not crash`, (t) => {
|
||||
const projectDir = projectWithFixture(t, fixture);
|
||||
for (const id of ['1', '2', '99', '999', '0', '2.1']) {
|
||||
const result = getPhase(projectDir, id);
|
||||
assert.equal(result.hasStackTrace, false, `${fixture} id=${id}: no V8 stack frame allowed`);
|
||||
// exit status varies (0 for found, non-zero for not-found —
|
||||
// both are valid). What's pinned: the parser produced SOME output
|
||||
// (either valid JSON or a clean stderr) without crashing.
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1218,3 +1218,203 @@ test('manager.md and autonomous.md no longer contain old "not claude" background
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-2876-skill-frontmatter-quote.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-2876-skill-frontmatter-quote (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Bug #2876: SKILL.md frontmatter parse failure when `description` begins
|
||||
* with a YAML flow indicator like `[BETA]`.
|
||||
*
|
||||
* description: [BETA] Offload plan phase to Claude Code's ultraplan…
|
||||
*
|
||||
* YAML 1.2 treats a leading `[` as the start of a flow sequence, so any
|
||||
* downstream parser (gh-copilot, JetBrains' kit, etc.) fails with
|
||||
* "Unexpected scalar at node end". The Copilot/Antigravity/Trae/Codebuddy
|
||||
* skill+agent converters in `bin/install.js` re-emit the description
|
||||
* unquoted; the Claude variant `yamlQuote(...)`s it. Bring the others
|
||||
* in line so any value is round-trip-safe regardless of leading char.
|
||||
*
|
||||
* The test is structural: it parses each emitted frontmatter into lines
|
||||
* and asserts the `description` value is a quoted YAML scalar (double or
|
||||
* single quoted) when the source description starts with a flow indicator.
|
||||
* It does not regex the bytes for substrings.
|
||||
*/
|
||||
'use strict';
|
||||
|
||||
process.env.GSD_TEST_MODE = '1';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const path = require('node:path');
|
||||
const fs = require('node:fs');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
const pkg = JSON.parse(fs.readFileSync(path.join(REPO_ROOT, 'package.json'), 'utf-8'));
|
||||
const installPath = path.resolve(REPO_ROOT, pkg.bin['gsd-core']);
|
||||
const install = require(installPath);
|
||||
|
||||
// Build a minimal Claude command source whose description starts with the
|
||||
// reporter's exact flow-indicator prefix. Apostrophe in the body forces
|
||||
// any naive single-quoting to also escape correctly — the canonical
|
||||
// safe form is `JSON.stringify(...)` (used by yamlQuote).
|
||||
const REPORTER_DESCRIPTION =
|
||||
"[BETA] Offload plan phase to Claude Code's ultraplan cloud — drafts remotely while terminal stays free, review in browser with inline comments, import back via /gsd-import. Claude Code only.";
|
||||
|
||||
// Use unquoted description in the source frontmatter — that's exactly the
|
||||
// shape that ships in commands/gsd/*.md when authors paste a description
|
||||
// without quoting it (see commands/gsd/ultraplan-phase.md). The bug is
|
||||
// triggered when the converter re-emits this same value to the destination
|
||||
// runtime without quoting. `extractFrontmatterField` strips a single outer
|
||||
// quote pair but does not unescape internal characters, so quoting the
|
||||
// fixture input would actually mask the bug.
|
||||
function buildClaudeCommand(description) {
|
||||
return [
|
||||
'---',
|
||||
'name: gsd:ultraplan-phase',
|
||||
`description: ${description}`,
|
||||
'argument-hint: "[phase-number]"',
|
||||
'allowed-tools:',
|
||||
' - Read',
|
||||
' - Bash',
|
||||
'---',
|
||||
'',
|
||||
'# body',
|
||||
'',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
function buildClaudeAgent(description) {
|
||||
return [
|
||||
'---',
|
||||
'name: gsd-extract-learnings',
|
||||
`description: ${description}`,
|
||||
'tools: Read, Bash',
|
||||
'---',
|
||||
'',
|
||||
'# body',
|
||||
'',
|
||||
].join('\n');
|
||||
}
|
||||
|
||||
function extractFrontmatter(content) {
|
||||
// Leading delimiter is `---\n`; closing is the next standalone `---`
|
||||
// on its own line. Tests parse line-structurally so the assertion
|
||||
// doesn't drift on whitespace/order changes (per project test-rigor).
|
||||
assert.ok(content.startsWith('---'), `output must begin with frontmatter, got: ${content.slice(0, 40)}`);
|
||||
const lines = content.split('\n');
|
||||
let openIdx = -1;
|
||||
let closeIdx = -1;
|
||||
for (let i = 0; i < lines.length; i += 1) {
|
||||
if (lines[i] === '---') {
|
||||
if (openIdx === -1) openIdx = i;
|
||||
else { closeIdx = i; break; }
|
||||
}
|
||||
}
|
||||
assert.ok(openIdx !== -1 && closeIdx !== -1, `output must have a closed frontmatter block, got:\n${content}`);
|
||||
return lines.slice(openIdx + 1, closeIdx);
|
||||
}
|
||||
|
||||
function findDescriptionLine(frontmatterLines) {
|
||||
for (const line of frontmatterLines) {
|
||||
if (line.startsWith('description:')) return line;
|
||||
}
|
||||
assert.fail(`no description line found in frontmatter:\n${frontmatterLines.join('\n')}`);
|
||||
return ''; // unreachable
|
||||
}
|
||||
|
||||
function isQuotedYamlScalar(valueText) {
|
||||
// YAML safe-quoted scalar: starts with `"` and ends with `"`, OR
|
||||
// starts with `'` and ends with `'`. This is what `yamlQuote()`
|
||||
// (JSON.stringify) and the Claude variant of these converters emit.
|
||||
const trimmed = valueText.trim();
|
||||
if (trimmed.startsWith('"') && trimmed.endsWith('"')) return true;
|
||||
if (trimmed.startsWith("'") && trimmed.endsWith("'")) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
function parseQuotedYamlValue(valueText) {
|
||||
const trimmed = valueText.trim();
|
||||
if (trimmed.startsWith('"')) return JSON.parse(trimmed);
|
||||
if (trimmed.startsWith("'")) return trimmed.slice(1, -1).replace(/''/g, "'");
|
||||
return trimmed;
|
||||
}
|
||||
|
||||
function assertDescriptionRoundTrips(emitted, expected, label) {
|
||||
const fmLines = extractFrontmatter(emitted);
|
||||
const descLine = findDescriptionLine(fmLines);
|
||||
const valueText = descLine.slice('description:'.length);
|
||||
assert.ok(
|
||||
isQuotedYamlScalar(valueText),
|
||||
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
parseQuotedYamlValue(valueText),
|
||||
expected,
|
||||
`(${label}) description must round-trip through YAML quoting unchanged.`,
|
||||
);
|
||||
}
|
||||
|
||||
const COMMAND_CONVERTERS = [
|
||||
{ label: 'convertClaudeCommandToCopilotSkill', fn: (src) => install.convertClaudeCommandToCopilotSkill(src, 'gsd-ultraplan-phase') },
|
||||
{ label: 'convertClaudeCommandToAntigravitySkill', fn: (src) => install.convertClaudeCommandToAntigravitySkill(src, 'gsd-ultraplan-phase') },
|
||||
{ label: 'convertClaudeCommandToTraeSkill', fn: (src) => install.convertClaudeCommandToTraeSkill(src, 'gsd-ultraplan-phase') },
|
||||
{ label: 'convertClaudeCommandToCodebuddySkill', fn: (src) => install.convertClaudeCommandToCodebuddySkill(src, 'gsd-ultraplan-phase') },
|
||||
];
|
||||
|
||||
const AGENT_CONVERTERS = [
|
||||
{ label: 'convertClaudeAgentToCopilotAgent', fn: (src) => install.convertClaudeAgentToCopilotAgent(src) },
|
||||
{ label: 'convertClaudeAgentToAntigravityAgent', fn: (src) => install.convertClaudeAgentToAntigravityAgent(src) },
|
||||
];
|
||||
|
||||
// A grab-bag of leading characters that all break unquoted YAML scalar
|
||||
// parsing per YAML 1.2 §7.3.3 / §6.9. The reporter's case is `[`; the
|
||||
// rest defend against neighbouring drift.
|
||||
const FLOW_HOSTILE_PREFIXES = ['[', '{', '*', '&', '!', '|', '>', '%', '@', '`'];
|
||||
|
||||
// Some converters (Trae, CodeBuddy) deliberately rewrite "Claude Code"
|
||||
// in body content to their target runtime name, and the rewrite cuts
|
||||
// across the description too. That's correct behavior — out of scope for
|
||||
// the YAML-quoting fix — so for the reporter case we assert only the
|
||||
// quoting requirement, not byte-equality of the round-tripped value.
|
||||
function assertDescriptionIsQuoted(emitted, label) {
|
||||
const fmLines = extractFrontmatter(emitted);
|
||||
const descLine = findDescriptionLine(fmLines);
|
||||
const valueText = descLine.slice('description:'.length);
|
||||
assert.ok(
|
||||
isQuotedYamlScalar(valueText),
|
||||
`(${label}) description must be a quoted YAML scalar (parser-safe for leading flow indicators). Got line: ${descLine}`,
|
||||
);
|
||||
}
|
||||
|
||||
describe('bug-2876: skill+agent converters emit YAML-quoted description', () => {
|
||||
for (const { label, fn } of COMMAND_CONVERTERS) {
|
||||
test(`${label}: reporter's "[BETA] ..." description is quoted`, () => {
|
||||
const out = fn(buildClaudeCommand(REPORTER_DESCRIPTION));
|
||||
assertDescriptionIsQuoted(out, label);
|
||||
});
|
||||
for (const prefix of FLOW_HOSTILE_PREFIXES) {
|
||||
test(`${label}: leading ${JSON.stringify(prefix)} is quoted`, () => {
|
||||
// Avoid leading/trailing `'` or `"` in the payload — `extractFrontmatterField`
|
||||
// strips a single outer quote char of either kind regardless of whether
|
||||
// the value was actually quoted, which would obscure the round-trip
|
||||
// assertion. Pre-existing behavior, out of scope for #2876.
|
||||
const desc = `${prefix} edge-case payload — flow indicator at start`;
|
||||
const out = fn(buildClaudeCommand(desc));
|
||||
assertDescriptionRoundTrips(out, desc, `${label} prefix=${prefix}`);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
for (const { label, fn } of AGENT_CONVERTERS) {
|
||||
test(`${label}: reporter-shape "[BETA] ..." description is quoted`, () => {
|
||||
const out = fn(buildClaudeAgent(REPORTER_DESCRIPTION));
|
||||
assertDescriptionIsQuoted(out, label);
|
||||
});
|
||||
}
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -838,3 +838,246 @@ describe('scanEntropyAnomalies', () => {
|
||||
assert.equal(result.findings.length, 1, 'only 1 high-entropy paragraph should be flagged');
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/fix-1627-asvs-level-scaling.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:fix-1627-asvs-level-scaling (consolidation epic #1969 B8 #1977)", () => {
|
||||
// allow-test-rule: source-text-is-the-product #1627
|
||||
// Agent .md / reference .md files — their text IS what the runtime loads.
|
||||
// Testing text content tests the deployed contract.
|
||||
// Per CONTRIBUTING.md exception matrix.
|
||||
|
||||
/**
|
||||
* Fix #1627 — ASVS level scaling
|
||||
*
|
||||
* Asserts that `workflow.security_asvs_level` now scales both planner
|
||||
* threat-disposition rigor and auditor verification depth rather than
|
||||
* being display-only.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const AGENTS_DIR = path.join(ROOT, 'agents');
|
||||
const REFS_DIR = path.join(ROOT, 'gsd-core', 'references');
|
||||
const MANIFEST_PATH = path.join(ROOT, 'docs', 'INVENTORY-MANIFEST.json');
|
||||
|
||||
describe('SECURE: ASVS level scaling (#1627)', () => {
|
||||
// ── 1. New reference file ────────────────────────────────────────────────
|
||||
|
||||
describe('security-asvs-levels.md reference', () => {
|
||||
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
|
||||
|
||||
test('file exists', () => {
|
||||
assert.ok(fs.existsSync(refPath), 'gsd-core/references/security-asvs-levels.md must exist');
|
||||
});
|
||||
|
||||
test('defines all three levels', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
assert.ok(content.includes('L1'), 'must define L1');
|
||||
assert.ok(content.includes('L2'), 'must define L2');
|
||||
assert.ok(content.includes('L3'), 'must define L3');
|
||||
});
|
||||
|
||||
test('L1 describes opportunistic scope and planner disposition', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.toLowerCase().includes('opportunistic'),
|
||||
'L1 must be described as opportunistic'
|
||||
);
|
||||
assert.ok(
|
||||
content.includes('mitigate') && content.includes('accept'),
|
||||
'must describe mitigate/accept dispositions'
|
||||
);
|
||||
});
|
||||
|
||||
test('L2 requires explicit rationale for accepted threats', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// L2 must require documented rationale for accepted risks
|
||||
assert.ok(
|
||||
content.includes('rationale') || content.includes('documented'),
|
||||
'L2 must require documented rationale for accepted threats'
|
||||
);
|
||||
});
|
||||
|
||||
test('L3 describes deep/comprehensive verification', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
const lower = content.toLowerCase();
|
||||
assert.ok(
|
||||
lower.includes('deep') || lower.includes('comprehensive') || lower.includes('exhaustive'),
|
||||
'L3 must describe deep/comprehensive verification'
|
||||
);
|
||||
});
|
||||
|
||||
test('mentions that higher levels are supersets of lower', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
const lower = content.toLowerCase();
|
||||
assert.ok(
|
||||
lower.includes('superset') || lower.includes('higher level') || lower.includes('includes all'),
|
||||
'must note that higher levels are supersets of lower'
|
||||
);
|
||||
});
|
||||
|
||||
test('describes distinct auditor verification depth for each level', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// All three audit depth keywords should appear
|
||||
assert.ok(content.includes('grep') || content.includes('PRESENT'), 'L1 audit depth must mention grep/presence check');
|
||||
assert.ok(content.includes('boundary') || content.includes('addresses'), 'L2 audit depth must mention boundary/addresses');
|
||||
assert.ok(content.includes('end-to-end') || content.includes('bypass'), 'L3 audit depth must mention end-to-end or bypass check');
|
||||
});
|
||||
});
|
||||
|
||||
// ── 2. gsd-planner.md — no hardcoded L1 in disposition ──────────────────
|
||||
|
||||
describe('gsd-planner.md security disposition', () => {
|
||||
const plannerPath = path.join(AGENTS_DIR, 'gsd-planner.md');
|
||||
|
||||
test('planner security instruction does not hardcode "ASVS L1"', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8');
|
||||
// The old bug: "mitigate if ASVS L1 requires it" — must be gone
|
||||
assert.ok(
|
||||
!content.includes('ASVS L1 requires it'),
|
||||
'planner must not hardcode "ASVS L1 requires it"; it must reference the configured level'
|
||||
);
|
||||
});
|
||||
|
||||
test('planner references the configured OWASP ASVS level', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('OWASP ASVS level') || content.includes('configured OWASP'),
|
||||
'planner must reference the configured OWASP ASVS level'
|
||||
);
|
||||
});
|
||||
|
||||
test('planner @-references security-asvs-levels.md', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('security-asvs-levels.md'),
|
||||
'planner must @-reference security-asvs-levels.md'
|
||||
);
|
||||
});
|
||||
|
||||
test('planner is under the 49152-char cap', () => {
|
||||
const content = fs.readFileSync(plannerPath, 'utf-8').replace(/\r\n/g, '\n').replace(/\r/g, '\n');
|
||||
assert.ok(
|
||||
content.length < 49152,
|
||||
`gsd-planner.md must be < 49152 chars (LF-normalized); got ${content.length}`
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 3. gsd-security-auditor.md — scaled verification depth ──────────────
|
||||
|
||||
describe('gsd-security-auditor.md verification depth', () => {
|
||||
const auditorPath = path.join(AGENTS_DIR, 'gsd-security-auditor.md');
|
||||
|
||||
test('auditor scales verification depth by asvs_level', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('asvs_level') || content.includes('ASVS level'),
|
||||
'auditor must reference asvs_level to scale verification'
|
||||
);
|
||||
});
|
||||
|
||||
test('auditor describes L1/L2/L3 depth differences', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
// All three levels must appear in context of depth scaling
|
||||
assert.ok(content.includes('L1'), 'auditor must mention L1 depth');
|
||||
assert.ok(content.includes('L2'), 'auditor must mention L2 depth');
|
||||
assert.ok(content.includes('L3'), 'auditor must mention L3 depth');
|
||||
});
|
||||
|
||||
test('auditor @-references security-asvs-levels.md', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('security-asvs-levels.md'),
|
||||
'auditor must @-reference security-asvs-levels.md'
|
||||
);
|
||||
});
|
||||
|
||||
test('auditor still echoes ASVS Level in structured output', () => {
|
||||
const content = fs.readFileSync(auditorPath, 'utf-8');
|
||||
assert.ok(
|
||||
content.includes('ASVS Level:') && content.includes('{1/2/3}'),
|
||||
'auditor must still emit ASVS Level in SECURED/OPEN_THREATS output'
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 4. secure-phase.md — ASVS-aware short-circuit ──────────────────────
|
||||
|
||||
describe('secure-phase.md short-circuit conditioned on asvs_level', () => {
|
||||
const wfPath = path.join(ROOT, 'gsd-core', 'workflows', 'secure-phase.md');
|
||||
|
||||
test('short-circuit to Step 6 is gated on asvs_level == 1', () => {
|
||||
const content = fs.readFileSync(wfPath, 'utf-8');
|
||||
// The condition must reference asvs_level so that L2/L3 don't skip the auditor
|
||||
assert.ok(
|
||||
content.includes('asvs_level == 1'),
|
||||
'secure-phase.md must gate the skip-to-Step-6 short-circuit on asvs_level == 1'
|
||||
);
|
||||
});
|
||||
|
||||
test('auditor runs at L2/L3 even when threats_open is 0 (asvs_level >= 2 branch present)', () => {
|
||||
const content = fs.readFileSync(wfPath, 'utf-8');
|
||||
// The >= 2 branch must explicitly say the auditor is spawned for L2/L3 deep verification
|
||||
assert.ok(
|
||||
content.includes('asvs_level >= 2'),
|
||||
'secure-phase.md must include asvs_level >= 2 branch that does NOT skip the auditor'
|
||||
);
|
||||
// The >= 2 branch must make clear the auditor is spawned (not skipped)
|
||||
assert.ok(
|
||||
content.includes('L2/L3 deep verification') || content.includes('L2 boundary') || content.includes('L3 end-to-end'),
|
||||
'secure-phase.md asvs_level >= 2 branch must reference L2/L3 deep verification'
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 5. security-asvs-levels.md — L1 medium-severity gap closed ──────────
|
||||
|
||||
describe('security-asvs-levels.md L1 medium-severity is specified', () => {
|
||||
const refPath = path.join(REFS_DIR, 'security-asvs-levels.md');
|
||||
|
||||
test('L1 explicitly handles medium-severity threats (no gap)', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// L1 section must say something about medium-severity
|
||||
assert.ok(
|
||||
content.includes('medium-severity') || content.includes('medium severity'),
|
||||
'L1 must explicitly specify disposition for medium-severity threats (no ambiguity gap)'
|
||||
);
|
||||
});
|
||||
|
||||
test('L1 medium-severity disposition is conditional (trust-boundary-aware)', () => {
|
||||
const content = fs.readFileSync(refPath, 'utf-8');
|
||||
// L1 must distinguish between medium on primary trust boundary vs not
|
||||
assert.ok(
|
||||
content.includes('trust boundary') || content.includes('primary trust'),
|
||||
'L1 medium-severity rule must reference trust boundary to disambiguate disposition'
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
// ── 6. Inventory manifest ─────────────────────────────────────────────────
|
||||
|
||||
describe('inventory manifest', () => {
|
||||
test('security-asvs-levels.md is registered in INVENTORY-MANIFEST.json', () => {
|
||||
const manifest = JSON.parse(fs.readFileSync(MANIFEST_PATH, 'utf-8'));
|
||||
const refs = (manifest.families || {}).references || [];
|
||||
assert.ok(
|
||||
refs.includes('security-asvs-levels.md'),
|
||||
'security-asvs-levels.md must appear in families.references of INVENTORY-MANIFEST.json'
|
||||
);
|
||||
});
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1313,3 +1313,200 @@ describe('ADR-1769 #1796: applyStatePreservation — table-driven post-sync cons
|
||||
assert.deepEqual(r.postFm, { status: 'executing', progress: { percent: 10 } });
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-21-state-md-template-frontmatter.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-21-state-md-template-frontmatter (consolidation epic #1969 B8 #1977)", () => {
|
||||
/**
|
||||
* Regression guard — Bug #21
|
||||
*
|
||||
* Both STATE.md template files must include a YAML frontmatter block in their
|
||||
* "File Template" section so that an AI agent creating .planning/STATE.md from
|
||||
* the template produces a file that frontmatter consumers can read immediately
|
||||
* (before the first `state.*` mutation calls syncStateFrontmatter).
|
||||
*
|
||||
* Prior to the fix, the template's File Template section began with
|
||||
* `# Project State` (no frontmatter), leaving the init→first-write window
|
||||
* without `gsd_state_version`, `status`, or `progress` keys.
|
||||
*
|
||||
* Acceptance criteria:
|
||||
* 1. The template body extracted from each state.md file's File Template code
|
||||
* block must begin with `---`.
|
||||
* 2. The frontmatter must contain at minimum: `gsd_state_version` and `status`.
|
||||
*/
|
||||
|
||||
'use strict';
|
||||
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const REPO_ROOT = path.join(__dirname, '..');
|
||||
|
||||
const TEMPLATE_PATHS = [
|
||||
path.join(REPO_ROOT, 'gsd-core', 'templates', 'state.md'),
|
||||
];
|
||||
|
||||
/**
|
||||
* Extract the content of the first ```markdown ... ``` code block from a
|
||||
* template file. Returns the raw string (including any leading/trailing
|
||||
* whitespace within the block).
|
||||
*
|
||||
* @param {string} fileContent - Full text of the template file.
|
||||
* @returns {string} The extracted code block body.
|
||||
*/
|
||||
function extractFileTemplate(fileContent) {
|
||||
const match = fileContent.match(/```markdown\r?\n([\s\S]*?)```/);
|
||||
assert.ok(match, 'No ```markdown code block found in template file');
|
||||
return match[1];
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal YAML frontmatter parser: returns the set of top-level keys present
|
||||
* in the first --- ... --- block at the start of `text`. Does not parse nested
|
||||
* keys — list-valued fields (e.g. `tags: [a, b]`) are recorded only by their
|
||||
* key name, not their value. Returns an empty Set when the text has no frontmatter.
|
||||
*
|
||||
* @param {string} text
|
||||
* @returns {Set<string>}
|
||||
*/
|
||||
function parseFrontmatterKeys(text) {
|
||||
const keys = new Set();
|
||||
if (!text.trimStart().startsWith('---')) return keys;
|
||||
const lines = text.split(/\r?\n/);
|
||||
let inBlock = false;
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!inBlock) {
|
||||
if (trimmed === '---') { inBlock = true; continue; }
|
||||
break; // frontmatter must be at the very start
|
||||
}
|
||||
if (trimmed === '---') break; // end of block
|
||||
const colonIdx = trimmed.indexOf(':');
|
||||
if (colonIdx > 0) {
|
||||
keys.add(trimmed.slice(0, colonIdx).trim());
|
||||
}
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal YAML frontmatter parser: returns a plain object of top-level keys
|
||||
* and their scalar or nested-object values from the first --- ... --- block.
|
||||
* Handles one level of indented nesting (e.g. progress.total_plans).
|
||||
* Does not handle YAML lists or multi-line values.
|
||||
*
|
||||
* @param {string} text
|
||||
* @returns {Record<string, any>}
|
||||
*/
|
||||
function parseFrontmatter(text) {
|
||||
const result = {};
|
||||
if (!text.trimStart().startsWith('---')) return result;
|
||||
const lines = text.split(/\r?\n/);
|
||||
let inBlock = false;
|
||||
let currentKey = null;
|
||||
for (const line of lines) {
|
||||
const trimmed = line.trim();
|
||||
if (!inBlock) {
|
||||
if (trimmed === '---') { inBlock = true; continue; }
|
||||
break;
|
||||
}
|
||||
if (trimmed === '---') break;
|
||||
// Detect indented (nested) line: starts with whitespace
|
||||
if (line.match(/^\s+\S/) && currentKey !== null) {
|
||||
const colonIdx = trimmed.indexOf(':');
|
||||
if (colonIdx > 0) {
|
||||
const subKey = trimmed.slice(0, colonIdx).trim();
|
||||
const rawVal = trimmed.slice(colonIdx + 1).trim();
|
||||
const numVal = Number(rawVal);
|
||||
if (typeof result[currentKey] !== 'object') result[currentKey] = {};
|
||||
result[currentKey][subKey] = rawVal === '' ? null : (isNaN(numVal) ? rawVal : numVal);
|
||||
}
|
||||
} else {
|
||||
currentKey = null;
|
||||
const colonIdx = trimmed.indexOf(':');
|
||||
if (colonIdx > 0) {
|
||||
const key = trimmed.slice(0, colonIdx).trim();
|
||||
const rawVal = trimmed.slice(colonIdx + 1).trim();
|
||||
if (rawVal === '') {
|
||||
result[key] = {};
|
||||
currentKey = key;
|
||||
} else {
|
||||
const numVal = Number(rawVal);
|
||||
result[key] = isNaN(numVal) ? rawVal.replace(/^'|'$/g, '') : numVal;
|
||||
currentKey = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
describe('bug #21 — STATE.md template must carry YAML frontmatter', () => {
|
||||
for (const templatePath of TEMPLATE_PATHS) {
|
||||
const label = path.relative(REPO_ROOT, templatePath);
|
||||
|
||||
test(`${label} — File Template block starts with frontmatter`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
|
||||
// The template body must open with a YAML frontmatter delimiter.
|
||||
assert.ok(
|
||||
body.trimStart().startsWith('---'),
|
||||
`${label}: File Template must start with '---' (YAML frontmatter), ` +
|
||||
`but starts with: ${JSON.stringify(body.slice(0, 60))}`,
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} — frontmatter contains gsd_state_version`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
const keys = parseFrontmatterKeys(body.trimStart());
|
||||
|
||||
assert.ok(
|
||||
keys.has('gsd_state_version'),
|
||||
`${label}: frontmatter must include 'gsd_state_version', found keys: ${[...keys].join(', ')}`,
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} — frontmatter contains status`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
const keys = parseFrontmatterKeys(body.trimStart());
|
||||
|
||||
assert.ok(
|
||||
keys.has('status'),
|
||||
`${label}: frontmatter must include 'status', found keys: ${[...keys].join(', ')}`,
|
||||
);
|
||||
});
|
||||
|
||||
test(`${label} — progress sub-schema has zeroed total_plans and completed_plans`, () => {
|
||||
const content = fs.readFileSync(templatePath, 'utf-8');
|
||||
const body = extractFileTemplate(content);
|
||||
const fm = parseFrontmatter(body.trimStart());
|
||||
|
||||
assert.ok(
|
||||
fm.progress && typeof fm.progress === 'object',
|
||||
`${label}: frontmatter must include a 'progress' sub-object`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
fm.progress.total_plans,
|
||||
0,
|
||||
`${label}: progress.total_plans must be 0 in the template`,
|
||||
);
|
||||
assert.strictEqual(
|
||||
fm.progress.completed_plans,
|
||||
0,
|
||||
`${label}: progress.completed_plans must be 0 in the template`,
|
||||
);
|
||||
});
|
||||
}
|
||||
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -65,3 +65,111 @@ describe('workflow CLI compatibility (#1759)', () => {
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/feat-41-ship-tdd-audit-gate-status.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:feat-41-ship-tdd-audit-gate-status (consolidation epic #1969 B8 #1977)", () => {
|
||||
'use strict';
|
||||
|
||||
// feat(#41): /gsd-ship generate_pr_body emits a TDD Audit table + an aggregate
|
||||
// `gate_status:` trailer so the per-commit TDD gate trail survives squash-merge.
|
||||
// These assertions pin the shipped workflow prose in gsd-core/workflows/ship.md.
|
||||
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
const assert = require('node:assert/strict');
|
||||
const { describe, test } = require('node:test');
|
||||
|
||||
const repoRoot = path.resolve(__dirname, '..');
|
||||
function readRepoFile(relativePath) {
|
||||
return fs.readFileSync(path.join(repoRoot, relativePath), 'utf8');
|
||||
}
|
||||
|
||||
describe('feat-41: ship.md TDD Audit gate_status extraction', () => {
|
||||
const workflow = readRepoFile('gsd-core/workflows/ship.md');
|
||||
|
||||
test('adds a "## TDD Audit" section to the generated PR body', () => {
|
||||
assert.match(workflow, /## TDD Audit/);
|
||||
});
|
||||
|
||||
test('extracts gate_status via Git native trailer machinery, not a raw body grep', () => {
|
||||
assert.match(workflow, /trailers:key=gate_status/);
|
||||
});
|
||||
|
||||
test('scopes the scan to the merge-base..HEAD range', () => {
|
||||
assert.match(workflow, /merge-base/);
|
||||
assert.match(workflow, /\.\.HEAD/);
|
||||
assert.match(workflow, /BASE_BRANCH/);
|
||||
});
|
||||
|
||||
test('excludes merge commits from the audit', () => {
|
||||
assert.match(workflow, /--no-merges/);
|
||||
});
|
||||
|
||||
test('renders a Test commit / Impl commit / gate_status table', () => {
|
||||
assert.match(workflow, /Test commit[\s\S]*Impl commit[\s\S]*gate_status/);
|
||||
});
|
||||
|
||||
test('pairs conventional-commit test: rows with their impl commit', () => {
|
||||
assert.match(workflow, /test:/);
|
||||
assert.match(workflow, /pair/i);
|
||||
});
|
||||
|
||||
test('escapes pipe characters in commit subjects so the table is not broken', () => {
|
||||
assert.match(workflow, /[Ee]scape[\s\S]{0,60}\|/);
|
||||
});
|
||||
|
||||
test('counts commits lacking a recognized gate_status trailer as missing', () => {
|
||||
assert.match(workflow, /missing/);
|
||||
});
|
||||
|
||||
test('is informational and never blocks the ship', () => {
|
||||
assert.match(workflow, /informational|never block|non-blocking/i);
|
||||
});
|
||||
|
||||
test('emits the aggregate trailer in the exact, stable key order', () => {
|
||||
assert.match(
|
||||
workflow,
|
||||
/gate_status:\s*skill=[^,]*,\s*fallback=[^,]*,\s*exempt=[^,]*,\s*missing=/,
|
||||
);
|
||||
});
|
||||
|
||||
test('places the aggregate trailer on the final line so squash-merge carries it', () => {
|
||||
assert.match(workflow, /squash/i);
|
||||
assert.match(workflow, /final line|last line/i);
|
||||
});
|
||||
|
||||
test('does not disturb the frozen #3167 core section order (Key Decisions precedes the new section)', () => {
|
||||
assert.match(workflow, /## Key Decisions[\s\S]*## TDD Audit/);
|
||||
});
|
||||
|
||||
// Hardening assertions added after adversarial review.
|
||||
|
||||
test('pairs test: rows only with feat:/fix: impl commits, skipping refactor/docs/chore', () => {
|
||||
assert.match(workflow, /feat:[\s\S]{0,20}fix:/);
|
||||
assert.match(workflow, /skipping[\s\S]{0,80}(refactor|docs|chore)/i);
|
||||
});
|
||||
|
||||
test('normalizes the gate_status cell to a known token, never raw trailer text', () => {
|
||||
assert.match(workflow, /normaliz[a-z]*[\s\S]{0,120}missing/i);
|
||||
assert.match(workflow, /never the raw/i);
|
||||
});
|
||||
|
||||
test('treats a commit with multiple gate_status trailers as missing', () => {
|
||||
assert.match(workflow, /more than one[\s\S]{0,40}gate_status/i);
|
||||
});
|
||||
|
||||
test('hardens every table cell against pipe/newline injection', () => {
|
||||
assert.match(workflow, /strip[\s\S]{0,20}\\r/);
|
||||
});
|
||||
|
||||
test('guards record/field delimiters against adversarial commit messages', () => {
|
||||
assert.match(workflow, /NUL|%x00|delimiter/i);
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
@@ -2575,3 +2575,88 @@ describe('gsd-validate-commit.sh delegates to git-cmd.js', () => {
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Folded from tests/bug-3384-secondary-defects.test.cjs — consolidation epic #1969 (B8 #1977)
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
{
|
||||
const { describe: __foldDescribe } = require('node:test');
|
||||
__foldDescribe("folded:bug-3384-secondary-defects (consolidation epic #1969 B8 #1977)", () => {
|
||||
// allow-test-rule: source-text-is-the-product (see #3384)
|
||||
const { describe, test } = require('node:test');
|
||||
const assert = require('node:assert/strict');
|
||||
const fs = require('node:fs');
|
||||
const path = require('node:path');
|
||||
|
||||
const repoRoot = path.resolve(__dirname, '..');
|
||||
const WORKTREE_BRANCH_CHECK_FRAGMENT = path.join(repoRoot, 'gsd-core', 'references', 'worktree-branch-check.md');
|
||||
|
||||
function read(relPath) {
|
||||
return fs.readFileSync(path.join(repoRoot, relPath), 'utf8');
|
||||
}
|
||||
|
||||
describe('bug #3384: adjacent worktree data-loss guards', () => {
|
||||
test('worktree cleanup CLI preserves caller cwd instead of resolving project root', () => {
|
||||
const source = read('gsd-core/bin/gsd-tools.cjs');
|
||||
const skipSet = source.slice(
|
||||
source.indexOf('const SKIP_ROOT_RESOLUTION = new Set(['),
|
||||
source.indexOf('if (!SKIP_ROOT_RESOLUTION.has(command))'),
|
||||
);
|
||||
|
||||
assert.match(skipSet, /'worktree'/);
|
||||
});
|
||||
|
||||
test('diagnose-issues references canonical fragment; fragment is verify-only and fails closed (#48)', () => {
|
||||
// diagnose-issues.md now references the canonical fragment rather than
|
||||
// inlining the block. Verify (a) it references the fragment and (b) the
|
||||
// fragment itself has the correct ordering: symbolic-ref/HEAD assertion and
|
||||
// ^worktree-agent- allow-list appear before any work, and (c) the fragment
|
||||
// is verify-only — no destructive self-recovery.
|
||||
const diagnoseSource = read('gsd-core/workflows/diagnose-issues.md');
|
||||
assert.ok(
|
||||
diagnoseSource.includes('worktree-branch-check.md'),
|
||||
'diagnose-issues.md must reference the canonical worktree-branch-check.md fragment'
|
||||
);
|
||||
|
||||
const fragmentSource = fs.readFileSync(WORKTREE_BRANCH_CHECK_FRAGMENT, 'utf8');
|
||||
const branchCheck = fragmentSource.indexOf('HEAD_REF=$(git symbolic-ref --quiet HEAD || echo');
|
||||
const namespaceCheck = fragmentSource.indexOf('^worktree-agent-');
|
||||
|
||||
assert.ok(branchCheck > 0, 'canonical fragment must assert HEAD before any work');
|
||||
assert.ok(namespaceCheck > branchCheck, 'canonical fragment must require disposable worktree-agent branch');
|
||||
// #48: verify-only — the destructive self-recovery is gone; the fragment fails closed instead.
|
||||
assert.ok(!fragmentSource.includes('git reset --hard {EXPECTED_BASE}'), 'canonical fragment must not self-recover via reset --hard — orchestrator owns recovery (#48)');
|
||||
assert.ok(fragmentSource.includes('exit 42'), 'canonical fragment must fail closed with exit 42 on base mismatch (#48)');
|
||||
});
|
||||
|
||||
test('remove-workspace fails closed when git worktree remove fails', () => {
|
||||
const source = read('gsd-core/workflows/remove-workspace.md');
|
||||
const init = source.indexOf('REMOVE_FAILED=false');
|
||||
const loop = source.indexOf('For each repo in the workspace');
|
||||
const remove = source.indexOf('git worktree remove "$WORKSPACE_PATH/$REPO_NAME"');
|
||||
|
||||
assert.doesNotMatch(
|
||||
source,
|
||||
/git worktree remove "\$WORKSPACE_PATH\/\$REPO_NAME" 2>&1 \|\| true/,
|
||||
'worktree removal failures must not be swallowed',
|
||||
);
|
||||
assert.ok(init > 0 && init < loop, 'REMOVE_FAILED must initialize once before the per-repo loop');
|
||||
assert.ok(remove > loop, 'worktree removal should remain inside the per-repo loop');
|
||||
assert.match(source, /Refusing to delete "\$WORKSPACE_PATH"/);
|
||||
});
|
||||
|
||||
test('validate health warns when worktree inventory cannot be listed', () => {
|
||||
const source = read('gsd-core/bin/lib/verify.cjs');
|
||||
// Accept both hand-written dot access and the tsc-compiled bracket form
|
||||
// (ADR-457: verify.cjs is now emitted from src/verify.cts):
|
||||
// hand-written: worktreeHealth.reason === 'git_list_failed'
|
||||
// tsc-compiled: worktreeHealth['reason'] === 'git_list_failed'
|
||||
const failureBranch = source.search(/worktreeHealth(?:\.reason|\['reason'\]) === 'git_list_failed'/);
|
||||
const warning = source.indexOf("addIssue('warning', 'W020'", failureBranch);
|
||||
|
||||
assert.ok(failureBranch > 0, 'verify health should branch on git_list_failed');
|
||||
assert.ok(warning > failureBranch, 'git_list_failed should emit W020 degraded-health warning');
|
||||
});
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user