Files
msd-core/tests/few-shot-calibration.test.cjs
Tom Boucher ca3be82f71 fix(3597): clear residual Windows test failures + add ratchet lint guard
Two more diagnostic passes (clusters J: residuals in already-touched files,
K: 12 untouched files) plus a production-code path fix and a new
ratchet-style lint guard.

## Test-only fixes (14 files)

bug-1736, bug-2248, bug-2698 — replace inline 1s-budget rmSync with the
shared cleanup() helper (5s budget, 20×250ms retries). The earlier
inline maxRetries:10 / retryDelay:100 wasn't enough to absorb Windows
Defender's deferred-scan handle hold on cold runners.

bug-2256, skill-manifest — also override USERPROFILE alongside HOME in
beforeEach/runGsdTools calls. os.homedir() reads USERPROFILE on win32,
so HOME-only stubs leak the runner's real home into the SUT.

bug-2784, bug-3608, enh-2500, enh-2790, few-shot-calibration,
gsd-settings-advanced — CRLF tolerance: literal \n in regexes against
file content (frontmatter anchors, bash-fence regex, multi-line
numbered-list captures, awk-block extractors) becomes \r?\n; split('\n')
becomes split(/\r?\n/). Windows checkout with autocrlf=true puts \r
before every \n; .+ doesn't match \r in JS regex by default.

bug-2966 — three-part fix to extractStepRun (CRLF split), awk regex
(\r?\n), and conflict-marker parser (rawLine + \r$ strip).

bug-2969, config — normalize separators on test assertions where the
SUT correctly emits \ on win32 but the test compares against /.

prompt-injection-scan — normalize relPath via replace(/\\/g, '/') before
ALLOWLIST.has() lookup. ALLOWLIST keys are POSIX; path.relative returns
backslashes on win32 → falsely scans allowlisted security module → trips
the boundary-tag detector on its own legitimate detection code.

prune-orphaned-worktrees — use the existing canonicalPath +
listedWorktreePaths(repoDir).has(...) helpers instead of substring
matching the raw path. git stores long-form canonical paths
(runneradmin), but mkdtempSync returns 8.3 short-form (RUNNER~1) on
Windows runners; plain string compare misses every entry.

## Production-code fix (1 file)

get-shit-done/bin/lib/init.cjs — bug-3491 in_nested_subdir computation
canonicalizes both worktreeRoot and cwd via fs.realpathSync.native +
path.relative before declaring "nested." Windows runner cwd (8.3 short
name) vs git's --show-toplevel (long form, forward slashes) made the
raw string compare always say true even at the worktree root, breaking
the "init new-project at worktree root" subtest.

## New ratchet lint guard

tests/windows-test-parity-guard.test.cjs — scans tests/ for 7
anti-patterns that drove the Windows failure clusters. Each rule has a
baseline count snapshot from this PR; the test fails when a new
occurrence appears (count grows above baseline), ratcheting down as
existing offenders are fixed. Patterns covered:

  G1 split('\n') after readFileSync (use /\r?\n/)
  G2 ```bash\n fence regex (use ```bash\r?\n)
  G3 ^---\n frontmatter anchor (use ^---\r?\n)
  G4 hardcoded "/tmp/..." literal passed to fs.* (use os.tmpdir())
  G5 bare 'npm' to execFileSync without {shell:true} on win32
  G6 process.env.HOME stub with no USERPROFILE
  G7 fs.rmSync({recursive,force}) without maxRetries

Future Windows-parity regressions get caught at PR time rather than
five iterations into a CI loop.

Validated: holodeck (ubuntu docker) 11232/0 pass (count +8 = the 7
new ratchet tests + parent describe).

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-16 12:39:02 -04:00

147 lines
7.6 KiB
JavaScript

const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fs = require('fs');
const path = require('path');
const REFS_DIR = path.join(__dirname, '..', 'get-shit-done', 'references', 'few-shot-examples');
const AGENTS_DIR = path.join(__dirname, '..', 'agents');
// ── Helpers ────────────────────────────────────────────────────────
function readFile(filePath) {
return fs.readFileSync(filePath, 'utf-8');
}
function countPattern(content, pattern) {
const matches = content.match(pattern);
return matches ? matches.length : 0;
}
// ── File existence ─────────────────────────────────────────────────
describe('few-shot calibration examples', () => {
describe('reference files exist', () => {
test('plan-checker.md exists in references/few-shot-examples/', () => {
assert.ok(fs.existsSync(path.join(REFS_DIR, 'plan-checker.md')));
});
test('verifier.md exists in references/few-shot-examples/', () => {
assert.ok(fs.existsSync(path.join(REFS_DIR, 'verifier.md')));
});
});
// ── Version/format metadata ────────────────────────────────────
describe('frontmatter metadata', () => {
test('plan-checker.md has version and component in frontmatter', () => {
const content = readFile(path.join(REFS_DIR, 'plan-checker.md'));
assert.match(content, /^---\r?\n/);
assert.match(content, /component:\s*plan-checker/);
assert.match(content, /version:\s*\d+/);
assert.match(content, /last_calibrated:\s*\d{4}-\d{2}-\d{2}/);
});
test('verifier.md has version and component in frontmatter', () => {
const content = readFile(path.join(REFS_DIR, 'verifier.md'));
assert.match(content, /^---\r?\n/);
assert.match(content, /component:\s*verifier/);
assert.match(content, /version:\s*\d+/);
assert.match(content, /last_calibrated:\s*\d{4}-\d{2}-\d{2}/);
});
// Version difference is intentional: plan-checker was calibrated first (v1),
// verifier later with updated format (v2) including calibration_source field.
test('version metadata values are present and numeric', () => {
const pcContent = readFile(path.join(REFS_DIR, 'plan-checker.md'));
const vContent = readFile(path.join(REFS_DIR, 'verifier.md'));
const pcVersion = pcContent.match(/version:\s*(\d+)/);
const vVersion = vContent.match(/version:\s*(\d+)/);
assert.ok(pcVersion, 'plan-checker.md must have a numeric version');
assert.ok(vVersion, 'verifier.md must have a numeric version');
});
});
// ── Example counts ─────────────────────────────────────────────
describe('example counts', () => {
test('plan-checker.md contains exactly 4 examples (2 positive, 2 negative)', () => {
const content = readFile(path.join(REFS_DIR, 'plan-checker.md'));
const totalExamples = countPattern(content, /^### Example \d+/gm);
assert.strictEqual(totalExamples, 4);
// Verify section breakdown
const positiveSection = content.indexOf('## Positive Examples');
const negativeSection = content.indexOf('## Negative Examples');
assert.ok(positiveSection >= 0, 'must have Positive Examples section');
assert.ok(negativeSection >= 0, 'must have Negative Examples section');
assert.ok(positiveSection < negativeSection, 'positive examples come before negative');
});
test('verifier.md contains exactly 7 examples (5 positive, 2 negative)', () => {
const content = readFile(path.join(REFS_DIR, 'verifier.md'));
const totalExamples = countPattern(content, /^### Example \d+/gm);
assert.strictEqual(totalExamples, 7);
const positiveSection = content.indexOf('## Positive Examples');
const negativeSection = content.indexOf('## Negative Examples');
assert.ok(positiveSection >= 0, 'must have Positive Examples section');
assert.ok(negativeSection >= 0, 'must have Negative Examples section');
assert.ok(positiveSection < negativeSection, 'positive examples come before negative');
});
});
// ── WHY annotations ────────────────────────────────────────────
describe('WHY annotations', () => {
test('every plan-checker example has a WHY annotation', () => {
const content = readFile(path.join(REFS_DIR, 'plan-checker.md'));
const exampleCount = countPattern(content, /^### Example \d+/gm);
const whyCount = countPattern(content, /^\*\*Why this is (good|bad):\*\*/gm);
assert.strictEqual(whyCount, exampleCount,
`expected ${exampleCount} WHY annotations, found ${whyCount}`);
});
test('every verifier example has a WHY annotation', () => {
const content = readFile(path.join(REFS_DIR, 'verifier.md'));
const exampleCount = countPattern(content, /^### Example \d+/gm);
const whyCount = countPattern(content, /^\*\*Why this is (good|bad):\*\*/gm);
assert.strictEqual(whyCount, exampleCount,
`expected ${exampleCount} WHY annotations, found ${whyCount}`);
});
});
// ── Agent reference lines ──────────────────────────────────────
describe('agent files reference few-shot examples', () => {
test('gsd-plan-checker.md contains reference to plan-checker few-shot examples', () => {
const content = readFile(path.join(AGENTS_DIR, 'gsd-plan-checker.md'));
assert.match(content, /@~\/\.claude\/get-shit-done\/references\/few-shot-examples\/plan-checker\.md/);
});
test('gsd-verifier.md contains reference to verifier few-shot examples', () => {
const content = readFile(path.join(AGENTS_DIR, 'gsd-verifier.md'));
assert.match(content, /@~\/\.claude\/get-shit-done\/references\/few-shot-examples\/verifier\.md/);
});
});
// ── Content structure ──────────────────────────────────────────
describe('content structure', () => {
test('plan-checker examples include input/output pairs', () => {
const content = readFile(path.join(REFS_DIR, 'plan-checker.md'));
const inputCount = countPattern(content, /^\*\*Input:\*\*/gm);
const outputCount = countPattern(content, /^\*\*Output:\*\*/gm);
assert.ok(inputCount >= 4, `expected at least 4 Input blocks, found ${inputCount}`);
assert.ok(outputCount >= 4, `expected at least 4 Output blocks, found ${outputCount}`);
});
test('verifier examples include input/output pairs', () => {
const content = readFile(path.join(REFS_DIR, 'verifier.md'));
const inputCount = countPattern(content, /^\*\*Input:\*\*/gm);
const outputCount = countPattern(content, /^\*\*Output:\*\*/gm);
assert.ok(inputCount >= 7, `expected at least 7 Input blocks, found ${inputCount}`);
assert.ok(outputCount >= 7, `expected at least 7 Output blocks, found ${outputCount}`);
});
test('verifier.md includes calibration-derived gap patterns table', () => {
const content = readFile(path.join(REFS_DIR, 'verifier.md'));
assert.match(content, /## Calibration-Derived Gap Patterns/);
assert.match(content, /Missing wiring/);
assert.match(content, /Missing tests/);
});
});
});