TEST_ENV_BASE blanks session-identity variables but none of the three that
decide WHERE a child process writes: CLAUDE_CONFIG_DIR, GSD_RUNTIME and
CODEX_HOME. The config-home resolver is env-first (runtime-homes.cts, the
dot-home case consults the env var before the home-derived fallback), so an
ambient CLAUDE_CONFIG_DIR in the developer's shell beats a call site that
sandboxes only HOME. The suite then writes into the developer's real config
directory -- including a registered skill under <configDir>/skills/ whose
body carries behavioural directives that load into later sessions.
Blank all three alongside the session-identity vars. `...env` still spreads
last, so the five call sites that already constrain these locally keep
winning with their explicit values.
TEST_ENV_BASE is re-declared in nine files, so the three lines are added
nine times rather than once. Consolidating the nine into a single exported
constant -- and fixing the TERM_SESSION / TERM_SESSION_ID drift between the
copies -- is deliberately left out of this change; see the PR body.
One call site needed adjusting. capability-state.test.cjs's
`capability state --runtime claude` CLI test passed no env at all and
compared the CHILD's resolved config dir against the PARENT process's
getGlobalConfigDir('claude'). That agreed only because the child inherited
the developer's ambient CLAUDE_CONFIG_DIR -- i.e. it passed *because of*
the leak. It now redirects both runtime homes into the sandbox and asserts
against values the test controls, so it is hermetic with the variable set
or unset.
Regression case folded into the owning module's test file rather than a new
bug-NNNN file, per scripts/lint-regression-test-names.cjs. It sets the
variable on the PARENT process, which is the actual vector; setting it in
the per-call env argument would exercise a path that was never broken.
258 lines
11 KiB
JavaScript
258 lines
11 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* Representative-corpus gate tests (#2371).
|
|
*
|
|
* Every fixture under tests/fixtures/representative/ is verbatim (or a
|
|
* minimal faithful subset) of a real reported artifact — never invented to
|
|
* match a gate's own grammar. See tests/fixtures/representative/README.md
|
|
* for the full rationale and CONTRIBUTING.md's "Fixture provenance" rule.
|
|
*
|
|
* Each gate is driven through its real CLI entrypoint (gate-verdict
|
|
* altitude), matching the established pattern in
|
|
* tests/api-coverage-gate-e2e.test.cjs and tests/decisions.test.cjs — never
|
|
* the parser function called in isolation.
|
|
*
|
|
* Three fixtures (across api-coverage-detector, api-coverage-matrix,
|
|
* decision-coverage-guard) encode gates that are still open bugs (#2365,
|
|
* #2366, #2347). For those, MANIFEST.json carries BOTH the correct target
|
|
* verdict (`expected*` — what the fix must produce) and the exact CURRENT
|
|
* observed verdict (`currentBuggyOutput` — what today's code actually
|
|
* returns). The test asserts against `currentBuggyOutput`: an honest,
|
|
* non-vacuous characterization of today's known-broken reality, not a fake
|
|
* pass. This assertion WILL fail, loudly, the moment the underlying bug is
|
|
* fixed and the gate starts returning something other than the pinned
|
|
* buggy value — at which point whoever's fix landed must update the
|
|
* assertion to check `expected*` instead (and can delete `currentBuggyOutput`).
|
|
*
|
|
* Why not node:test's `todo` option: this repo's own test-runner
|
|
* (gsd-test / gsd-test-runner v1.6.2) has no concept of it. Its JSONL
|
|
* result parser (internal/pipeline/parse.go's parseJSONL, gsd-test-runner
|
|
* repo) only recognizes `kind: "pass" | "fail"` — verified directly against
|
|
* that source — so a `{ todo: true }` test whose body throws is still
|
|
* counted as a real failure in the tool's own verdict. Characterization
|
|
* (assert the known-current value) sidesteps this because the test
|
|
* genuinely passes today; it needs no runner-level "expected failure"
|
|
* feature at all.
|
|
*
|
|
* The audit-uat corpus (#2286, fixed by #2317) has no currentBuggyOutput:
|
|
* it already asserts the correct behavior directly, because the bug is
|
|
* already fixed — proof the methodology works end to end, not just a
|
|
* record of gaps.
|
|
*/
|
|
|
|
const { describe, test, afterEach } = require('node:test');
|
|
const assert = require('node:assert/strict');
|
|
const fs = require('node:fs');
|
|
const os = require('node:os');
|
|
const path = require('node:path');
|
|
const { execFileSync } = require('node:child_process');
|
|
|
|
const { cleanup } = require('./helpers.cjs');
|
|
|
|
const TOOLS_PATH = path.join(__dirname, '..', 'gsd-core', 'bin', 'gsd-tools.cjs');
|
|
const FIXTURES_ROOT = path.join(__dirname, 'fixtures', 'representative');
|
|
|
|
const TEST_ENV_BASE = {
|
|
GSD_SESSION_KEY: '',
|
|
CODEX_THREAD_ID: '',
|
|
CLAUDE_SESSION_ID: '',
|
|
CLAUDE_CODE_SSE_PORT: '',
|
|
OPENCODE_SESSION_ID: '',
|
|
GEMINI_SESSION_ID: '',
|
|
CURSOR_SESSION_ID: '',
|
|
WINDSURF_SESSION_ID: '',
|
|
TERM_SESSION: '',
|
|
WT_SESSION: '',
|
|
TMUX_PANE: '',
|
|
ZELLIJ_SESSION_NAME: '',
|
|
TTY: '',
|
|
SSH_TTY: '',
|
|
// Config-LOCATION vars. Distinct in kind from the session-identity vars
|
|
// above: these decide WHERE a child writes, so leaving them ambient lets a
|
|
// test that sandboxes HOME still escape into the developer's real config dir.
|
|
CLAUDE_CONFIG_DIR: '',
|
|
GSD_RUNTIME: '',
|
|
CODEX_HOME: '',
|
|
};
|
|
|
|
function runTools(args, cwd) {
|
|
try {
|
|
const stdout = execFileSync(process.execPath, [TOOLS_PATH, ...args], {
|
|
cwd,
|
|
encoding: 'utf-8',
|
|
env: { ...process.env, ...TEST_ENV_BASE },
|
|
timeout: 60000,
|
|
});
|
|
return { success: true, output: stdout.trim(), error: '' };
|
|
} catch (err) {
|
|
return {
|
|
success: false,
|
|
output: err.stdout?.toString().trim() || '',
|
|
error: err.stderr?.toString().trim() || err.message,
|
|
};
|
|
}
|
|
}
|
|
|
|
function readManifest(gateDir) {
|
|
const raw = fs.readFileSync(path.join(FIXTURES_ROOT, gateDir, 'MANIFEST.json'), 'utf8');
|
|
return JSON.parse(raw);
|
|
}
|
|
|
|
function readFixture(gateDir, file) {
|
|
return fs.readFileSync(path.join(FIXTURES_ROOT, gateDir, file), 'utf8');
|
|
}
|
|
|
|
function makeProject() {
|
|
const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gsd-repcorpus-'));
|
|
fs.mkdirSync(path.join(tmpDir, '.planning', 'phases'), { recursive: true });
|
|
fs.writeFileSync(path.join(tmpDir, '.planning', 'config.json'), '{}', 'utf8');
|
|
return tmpDir;
|
|
}
|
|
|
|
function makePhaseDir(projectDir, slug) {
|
|
const dir = path.join(projectDir, '.planning', 'phases', slug);
|
|
fs.mkdirSync(dir, { recursive: true });
|
|
return dir;
|
|
}
|
|
|
|
// ─── api-coverage-detector (#2365) ────────────────────────────────────────────
|
|
|
|
describe('representative corpus — api-coverage detector (#2365)', () => {
|
|
let tmpDir;
|
|
afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } });
|
|
|
|
const manifest = readManifest('api-coverage-detector');
|
|
|
|
for (const fx of manifest.fixtures) {
|
|
const label = fx.currentBuggyOutput ? `${fx.file} → currently detected:true (#2365)` : `${fx.file} → detected:false`;
|
|
test(label, () => {
|
|
tmpDir = makeProject();
|
|
const phaseDir = makePhaseDir(tmpDir, '01-repcorpus');
|
|
const body = readFixture('api-coverage-detector', fx.file);
|
|
fs.writeFileSync(path.join(phaseDir, '01-PLAN.md'), `# Plan\n${body}\n`, 'utf8');
|
|
|
|
const r = runTools(['check', 'api-coverage.verify-pre', phaseDir, '--raw'], tmpDir);
|
|
assert.ok(r.success, `gate should succeed (JSON). stderr: ${r.error}`);
|
|
const j = JSON.parse(r.output);
|
|
|
|
if (fx.currentBuggyOutput) {
|
|
assert.strictEqual(j.detected, fx.currentBuggyOutput.detected,
|
|
`${fx.file}: expected today's known-buggy detected:${fx.currentBuggyOutput.detected}, got ${JSON.stringify(j)}. ` +
|
|
`If this now differs, #2365 may be fixed — check against expectedDetected:${fx.expectedDetected} instead.`);
|
|
assert.strictEqual(j.signals?.[0]?.verb, fx.currentBuggyOutput.signal.verb, `${fx.file}: signal.verb`);
|
|
assert.strictEqual(j.signals?.[0]?.noun, fx.currentBuggyOutput.signal.noun, `${fx.file}: signal.noun`);
|
|
} else {
|
|
assert.strictEqual(j.detected, fx.expectedDetected,
|
|
`${fx.file}: expected detected:${fx.expectedDetected}, got ${JSON.stringify(j)}`);
|
|
}
|
|
});
|
|
}
|
|
});
|
|
|
|
// ─── api-coverage-matrix (#2366) ──────────────────────────────────────────────
|
|
|
|
describe('representative corpus — api-coverage matrix (#2366)', () => {
|
|
let tmpDir;
|
|
afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } });
|
|
|
|
const manifest = readManifest('api-coverage-matrix');
|
|
|
|
for (const fx of manifest.fixtures) {
|
|
const label = `${fx.file} → exactly the canonical rows, 0 errors`;
|
|
test(label, () => {
|
|
tmpDir = makeProject();
|
|
const phaseDir = makePhaseDir(tmpDir, '01-repcorpus');
|
|
fs.writeFileSync(path.join(phaseDir, '01-PLAN.md'), `# Plan\n${fx.pairedPlan}\n`, 'utf8');
|
|
fs.writeFileSync(path.join(phaseDir, 'COVERAGE.md'), readFixture('api-coverage-matrix', fx.file), 'utf8');
|
|
|
|
const r = runTools(['check', 'api-coverage.verify-pre', phaseDir, '--raw'], tmpDir);
|
|
assert.ok(r.success, `gate should succeed (JSON). stderr: ${r.error}`);
|
|
const j = JSON.parse(r.output);
|
|
|
|
assert.strictEqual(j.block, fx.expectedBlock, `${fx.file}: block. Got ${JSON.stringify(j)}`);
|
|
assert.deepStrictEqual(j.counts, fx.expectedCounts, `${fx.file}: counts. Got ${JSON.stringify(j)}`);
|
|
assert.strictEqual((j.errors || []).length, fx.expectedErrorCount,
|
|
`${fx.file}: errors. Got ${JSON.stringify(j.errors)}`);
|
|
});
|
|
}
|
|
});
|
|
|
|
// ─── audit-uat (#2286, fixed by #2317 — asserts correct behavior directly) ────
|
|
|
|
describe('representative corpus — audit-uat (#2286, fixed by #2317)', () => {
|
|
let tmpDir;
|
|
afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } });
|
|
|
|
const manifest = readManifest('audit-uat');
|
|
|
|
test('Gaps-section + human-verification-frontmatter fixtures both surface as real items', () => {
|
|
tmpDir = makeProject();
|
|
const phaseDir = makePhaseDir(tmpDir, '01-repcorpus');
|
|
for (const fx of manifest.fixtures) {
|
|
fs.writeFileSync(
|
|
path.join(phaseDir, `01${fx.filenameSuffix}`),
|
|
readFixture('audit-uat', fx.file),
|
|
'utf8',
|
|
);
|
|
}
|
|
|
|
const r = runTools(['audit-uat', '--raw'], tmpDir);
|
|
assert.ok(r.success, `audit-uat should succeed. stderr: ${r.error}`);
|
|
const j = JSON.parse(r.output);
|
|
assert.ok(
|
|
j.summary.total_items >= manifest.expectedTotalItems,
|
|
`expected total_items >= ${manifest.expectedTotalItems}, got ${JSON.stringify(j.summary)}`,
|
|
);
|
|
// Per-fixture check (not just the aggregate): a regression that moves
|
|
// items between files while preserving the total would slip past the
|
|
// total_items check above but not this one.
|
|
for (const fx of manifest.fixtures) {
|
|
const fileName = `01${fx.filenameSuffix}`;
|
|
const fileResult = j.results.find((r2) => r2.file === fileName);
|
|
assert.ok(fileResult, `expected a result entry for ${fileName}, got ${JSON.stringify(j.results)}`);
|
|
assert.ok(
|
|
fileResult.items.length >= fx.expectedMinItems,
|
|
`${fileName}: expected items.length >= ${fx.expectedMinItems}, got ${fileResult.items.length}`,
|
|
);
|
|
}
|
|
});
|
|
});
|
|
|
|
// ─── decision-coverage-guard (#2347) ──────────────────────────────────────────
|
|
|
|
describe('representative corpus — decision-coverage guard (#2347)', () => {
|
|
let tmpDir;
|
|
afterEach(() => { if (tmpDir) { cleanup(tmpDir); tmpDir = null; } });
|
|
|
|
const manifest = readManifest('decision-coverage-guard');
|
|
|
|
for (const fx of manifest.fixtures) {
|
|
const label = fx.currentBuggyOutput
|
|
? `${fx.file} → currently passed:true, skipped:true (#2347)`
|
|
: `${fx.file} → outcome could-not-parse, passed:false`;
|
|
test(label, () => {
|
|
tmpDir = makeProject();
|
|
const phaseDir = makePhaseDir(tmpDir, '01-repcorpus');
|
|
const contextPath = path.join(phaseDir, 'CONTEXT.md');
|
|
fs.writeFileSync(contextPath, readFixture('decision-coverage-guard', fx.file), 'utf8');
|
|
|
|
const r = runTools(['query', 'check.decision-coverage-plan', phaseDir, contextPath], tmpDir);
|
|
assert.ok(r.success, `gate should succeed (JSON). stderr: ${r.error}`);
|
|
const j = JSON.parse(r.output);
|
|
|
|
if (fx.currentBuggyOutput) {
|
|
assert.strictEqual(j.passed, fx.currentBuggyOutput.passed,
|
|
`${fx.file}: expected today's known-buggy passed:${fx.currentBuggyOutput.passed}, got ${JSON.stringify(j)}. ` +
|
|
`If this now differs, #2347 may be fixed — check against expectedPassed:${fx.expectedPassed} instead.`);
|
|
assert.strictEqual(j.skipped, fx.currentBuggyOutput.skipped, `${fx.file}: skipped`);
|
|
assert.strictEqual(j.reason, fx.currentBuggyOutput.reason, `${fx.file}: reason`);
|
|
assert.strictEqual(j.total, fx.currentBuggyOutput.total, `${fx.file}: total`);
|
|
} else {
|
|
assert.strictEqual(j.passed, fx.expectedPassed, `${fx.file}: passed. Got ${JSON.stringify(j)}`);
|
|
assert.strictEqual(j.reason, fx.expectedReason, `${fx.file}: reason. Got ${JSON.stringify(j)}`);
|
|
}
|
|
});
|
|
}
|
|
});
|