test(tests): antagonistic tier — fast-check property tests + Stryker mutation testing (PR-gated) (#461)

* feat(#454): add antagonistic tier — fast-check property tests + Stryker mutation config

Adds property-based testing (fast-check v4) and mutation testing scaffolding
(Stryker v8) as the antagonistic validation tier for lib/*.cjs pure logic.

## Files added

### Shared setup
- tests/helpers/fast-check-setup.cjs — configureGlobal({ numRuns:200, seed:42 })
  for deterministic CI; override locally with GSD_FC_SEED

### Property test suites (node:test + fast-check)
- tests/context-utilization.property.test.cjs — boundary at 60%/70% thresholds
  (exact Math.ceil boundary, not Math.floor), TypeError on all invalid inputs,
  overflow clamping to 100%/critical, shape invariants (7 tests, all pass)
- tests/prompt-budget.property.test.cjs — estimateTokens monotonicity + ceil(len/4)
  exactness; applyBudget shape invariant, instructions/roadmap verbatim, budget
  envelope, omit tracking (11 tests, all pass)
- tests/frontmatter.property.test.cjs — extractFrontmatter/reconstructFrontmatter/
  spliceFrontmatter never-throw + type shape + splice→extract round-trip (9 tests)
- tests/adr-parser.property.test.cjs — shouldRejectAdrStatus boundary (3 statuses
  only), parseAdrMarkdown shape + title trim invariant (discovered: parser trims
  trailing whitespace) (9 tests, all pass)
- tests/config-schema.property.test.cjs — isValidConfigKey never throws, returns
  boolean, accepts all VALID/RUNTIME_STATE_KEYS, rejects empty/null/unknown (8 tests)

### Stryker mutation config
- stryker.config.mjs — testRunner:'command', mutate bin/lib/**/*.cjs minus 13
  generated files, coverageAnalysis:'off', thresholds {high:80,low:60,break:50},
  incremental:true, reporters html+clear-text+progress

### CI workflow
- .github/workflows/mutation.yml — PR-gating job (pull_request + workflow_dispatch),
  runs stryker --incremental --since origin/next (changed files only), uploads
  HTML artifact; SINCE_REF via env not interpolation (injection-safe)

### Package config
- package.json: +test:mutation, +test:mutation:since scripts
- .gitignore: +.stryker-tmp/, +.stryker-incremental.json, +reports/mutation/
- package-lock.json: fast-check@4.8.0, @stryker-mutator/core@9.6.1

## Verified
node --test on all 5 property test files: 44 tests, 0 failures.
Stryker NOT run (slow; reserved for CI).

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(#454): pin upload-artifact to v7.0.1 SHA used across repo (bad SHA ea165f8d)

The SHA ea165f8d65b6e75b540449d3ec4f5dde0c5a4e1 (labeled v4.6.2) does not
resolve on GitHub Actions. All other workflows in this repo pin
043fb46d1a93c77aae656e7c1c64a875d1fc6a0a (v7.0.1) — align mutation.yml.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* fix(#454): mutation workflow — replace invalid --since flag with changed-core-files --mutate scoping

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: CI Rebase Check <ci@gsd-redux>
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Tom Boucher
2026-05-29 11:07:07 -04:00
committed by GitHub
parent 33afb4f6eb
commit 4b908e6bcd
11 changed files with 3060 additions and 9 deletions

94
.github/workflows/mutation.yml vendored Normal file
View File

@@ -0,0 +1,94 @@
name: Mutation Testing
# PR-GATING: runs on every pull_request targeting `next` or `main`.
# Computes changed core lib files via git diff and passes them to --mutate,
# so only mutants in CHANGED files are tested — keeps the job bounded.
# If no core lib files changed, the gate passes trivially (skip+exit 0).
# Full-repo mutation runs are reserved for local exploration (npm run test:mutation).
on:
pull_request:
branches:
- next
- main
paths:
# Only run when lib source or property tests change
- 'get-shit-done/bin/lib/**/*.cjs'
- 'tests/**/*.property.test.cjs'
- 'stryker.config.mjs'
workflow_dispatch:
concurrency:
group: mutation-${{ github.head_ref || github.run_id }}
cancel-in-progress: true
permissions:
contents: read
jobs:
mutation:
name: Stryker mutation score (changed files only)
runs-on: ubuntu-latest
timeout-minutes: 30
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
persist-credentials: true
- name: Set up Node
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0
with:
node-version-file: .nvmrc
cache: npm
- name: Install dependencies
run: npm ci
- name: Fetch base ref for diff
# Ensure the base branch tip is available for the git diff below.
# fetch-depth: 0 above gets all history, but the remote ref name must exist.
run: git fetch origin ${{ github.base_ref }} --depth=1
- name: Compute changed core lib files
id: changed
run: |
BASE_REF="origin/${{ github.base_ref }}"
# Find changed non-test, non-generated .cjs files in bin/lib
CHANGED=$(git diff --name-only "${BASE_REF}...HEAD" -- 'get-shit-done/bin/lib/**/*.cjs' \
| grep -v '\.test\.cjs$' \
| grep -v -E '/(configuration|command-aliases|commands|core|install-profiles|installer-migrations|phase|profile-output|state|verify|init|audit|gsd2-import)\.cjs$' \
|| true)
if [ -z "$CHANGED" ]; then
echo "changed=" >> "$GITHUB_OUTPUT"
else
# Join with commas for --mutate
MUTATE_LIST=$(echo "$CHANGED" | paste -sd, -)
echo "changed=${MUTATE_LIST}" >> "$GITHUB_OUTPUT"
fi
- name: Skip mutation gate (no core lib files changed)
if: steps.changed.outputs.changed == ''
run: echo "No core lib files changed; skipping mutation gate"
- name: Run Stryker (incremental, changed files only)
if: steps.changed.outputs.changed != ''
# --mutate scopes mutation to only the changed production files.
# --incremental reuses cached results for unchanged mutants.
# The break threshold (50) is read from stryker.config.mjs and causes
# Stryker to exit non-zero when mutation score < 50%, failing the PR check.
env:
NODE_OPTIONS: '--max-old-space-size=4096'
run: |
MUTATE_LIST="${{ steps.changed.outputs.changed }}"
echo "Running Stryker --mutate '${MUTATE_LIST}'"
npx stryker run --incremental --mutate "${MUTATE_LIST}"
- name: Upload mutation HTML report
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
if: always() # upload even on failure so the score is visible
with:
name: mutation-report-${{ github.run_number }}
path: reports/mutation/mutation.html
retention-days: 14

5
.gitignore vendored
View File

@@ -84,3 +84,8 @@ claude-test-command.md
# Observability audit trail (issue #177) — append-only, local only # Observability audit trail (issue #177) — append-only, local only
.planning/.gsd-trace.jsonl .planning/.gsd-trace.jsonl
# Mutation testing artifacts (issue #454)
.stryker-tmp/
.stryker-incremental.json
reports/mutation/

1829
package-lock.json generated

File diff suppressed because it is too large Load Diff

View File

@@ -49,10 +49,12 @@
}, },
"devDependencies": { "devDependencies": {
"@eslint/js": "^9.39.4", "@eslint/js": "^9.39.4",
"@stryker-mutator/core": "^8.7.1",
"c8": "^11.0.0", "c8": "^11.0.0",
"eslint": "^9.39.4", "eslint": "^9.39.4",
"eslint-plugin-n": "^17.24.0", "eslint-plugin-n": "^17.24.0",
"eslint-plugin-no-only-tests": "^3.4.0", "eslint-plugin-no-only-tests": "^3.4.0",
"fast-check": "^4.8.0",
"globals": "^16.5.0", "globals": "^16.5.0",
"js-yaml": "^4.1.1", "js-yaml": "^4.1.1",
"typescript": "^6.0.3", "typescript": "^6.0.3",
@@ -91,6 +93,8 @@
"test:affected": "node scripts/run-affected-tests.cjs", "test:affected": "node scripts/run-affected-tests.cjs",
"test:coverage": "c8 --check-coverage --lines 70 --reporter text --include 'get-shit-done/bin/lib/*.cjs' --exclude 'tests/**' --all node scripts/run-tests.cjs", "test:coverage": "c8 --check-coverage --lines 70 --reporter text --include 'get-shit-done/bin/lib/*.cjs' --exclude 'tests/**' --all node scripts/run-tests.cjs",
"test:coverage:unit": "c8 --check-coverage --lines 70 --reporter text --include 'get-shit-done/bin/lib/*.cjs' --exclude 'tests/**' --all node scripts/run-tests.cjs --suite unit", "test:coverage:unit": "c8 --check-coverage --lines 70 --reporter text --include 'get-shit-done/bin/lib/*.cjs' --exclude 'tests/**' --all node scripts/run-tests.cjs --suite unit",
"test:coverage:all": "npm run test:coverage" "test:coverage:all": "npm run test:coverage",
"test:mutation": "stryker run",
"test:mutation:since": "stryker run --incremental --since origin/next"
} }
} }

92
stryker.config.mjs Normal file
View File

@@ -0,0 +1,92 @@
/**
* stryker.config.mjs
*
* Mutation testing configuration for get-shit-done-redux.
*
* Test runner: 'command' (built into @stryker-mutator/core)
* Runs: node --test over the lib test files via the repo's run-tests invocation.
*
* Mutate scope: bin/lib/**\/*.cjs, excluding generated files and test files.
*
* coverageAnalysis: 'off' — command runner does not support per-mutant coverage
* thresholds: high=80, low=60, break=50
* incremental: true — caches results; PR-scoped runs pass --mutate <changed-files>
*
* Reports:
* - html: reports/mutation/mutation.html
* - clear-text (console)
* - progress (spinner)
*
* NOTE: This is incremental / changed-files-only in CI (--mutate <changed-files>)
* to stay bounded. Full runs are for local exploration only.
*/
// Generated files that must NEVER be mutated
const GENERATED_FILES = [
'!get-shit-done/bin/lib/configuration.cjs', // GENERATED — sdk/src/config/index.ts
'!get-shit-done/bin/lib/command-aliases.cjs', // GENERATED
'!get-shit-done/bin/lib/commands.cjs', // GENERATED
'!get-shit-done/bin/lib/core.cjs', // GENERATED
'!get-shit-done/bin/lib/install-profiles.cjs', // GENERATED
'!get-shit-done/bin/lib/installer-migrations.cjs', // GENERATED
'!get-shit-done/bin/lib/phase.cjs', // GENERATED
'!get-shit-done/bin/lib/profile-output.cjs', // GENERATED
'!get-shit-done/bin/lib/state.cjs', // GENERATED
'!get-shit-done/bin/lib/verify.cjs', // GENERATED
'!get-shit-done/bin/lib/init.cjs', // GENERATED
'!get-shit-done/bin/lib/audit.cjs', // GENERATED
'!get-shit-done/bin/lib/gsd2-import.cjs', // GENERATED
];
/** @type {import('@stryker-mutator/core').PartialStrykerOptions} */
export default {
// ── Test runner ──────────────────────────────────────────────────────────────
testRunner: 'command',
commandRunner: {
// Run property tests + unit tests over lib only.
// Deliberately avoids running the full integration suite (slow).
command: 'node --test tests/context-utilization.property.test.cjs tests/prompt-budget.property.test.cjs tests/frontmatter.property.test.cjs tests/adr-parser.property.test.cjs tests/config-schema.property.test.cjs tests/adr-parser.test.cjs tests/active-workstream-store.test.cjs',
},
// ── Files to mutate ──────────────────────────────────────────────────────────
mutate: [
'get-shit-done/bin/lib/**/*.cjs',
'!get-shit-done/bin/lib/**/*.test.cjs',
...GENERATED_FILES,
],
// ── Coverage ─────────────────────────────────────────────────────────────────
// 'off' is required for the command test runner — it cannot instrument per-mutant.
coverageAnalysis: 'off',
// ── Thresholds ───────────────────────────────────────────────────────────────
thresholds: {
high: 80,
low: 60,
break: 50,
},
// ── Incremental mode ─────────────────────────────────────────────────────────
// Cache mutation results; re-run only changed mutants on subsequent calls.
// In CI the workflow computes changed files and passes: stryker run --incremental --mutate <list>
incremental: true,
incrementalFile: '.stryker-incremental.json',
// ── Reporters ────────────────────────────────────────────────────────────────
reporters: ['html', 'clear-text', 'progress'],
htmlReporter: {
fileName: 'reports/mutation/mutation.html',
},
// ── Temp directory ───────────────────────────────────────────────────────────
tempDirName: '.stryker-tmp',
// ── Ignore patterns ──────────────────────────────────────────────────────────
ignorePatterns: [
'node_modules',
'reports',
'.stryker-tmp',
'coverage',
'hooks/dist',
],
};

View File

@@ -0,0 +1,212 @@
'use strict';
/**
* Property-based tests for adr-parser.cjs
*
* Module: get-shit-done/bin/lib/adr-parser.cjs
* Exported: parseAdrMarkdown(markdown, options), shouldRejectAdrStatus(status)
*
* Properties tested:
* (a) shouldRejectAdrStatus: boundary — rejects exactly the known 3 statuses
* (b) shouldRejectAdrStatus: never throws on any input
* (c) parseAdrMarkdown: never throws on any string (including binary/unicode)
* (d) parseAdrMarkdown: always returns the required typed shape
* (e) parseAdrMarkdown: title extracted from H1 heading when present
* (f) parseAdrMarkdown: status is always a string (never null/undefined)
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fc = require('./helpers/fast-check-setup.cjs');
const {
parseAdrMarkdown,
shouldRejectAdrStatus,
} = require('../get-shit-done/bin/lib/adr-parser.cjs');
// Required output keys
const REQUIRED_KEYS = [
'title', 'status', 'context', 'decisions', 'options_considered',
'consequences_positive', 'consequences_negative', 'out_of_scope',
'deferred', 'dependencies', 'updates', 'source_path', 'key_files',
'plan_sequence', 'format', 'unmapped_headers',
];
// Known reject-statuses
const REJECT_STATUSES = ['superseded', 'rejected', 'deprecated'];
const ACCEPT_STATUSES = ['accepted', 'proposed', 'active', 'draft', ''];
describe('adr-parser: shouldRejectAdrStatus properties', () => {
// (a) Boundary: exactly the 3 known reject statuses return true
test('property: reject statuses return true, others return false', () => {
for (const status of REJECT_STATUSES) {
assert.equal(shouldRejectAdrStatus(status), true, `${status} should be rejected`);
}
for (const status of ACCEPT_STATUSES) {
assert.equal(shouldRejectAdrStatus(status), false, `${status} should not be rejected`);
}
});
// (b) Never throws on any input
test('property: shouldRejectAdrStatus never throws on any input', () => {
fc.assert(
fc.property(
fc.oneof(
fc.constant(null),
fc.constant(undefined),
fc.constant(NaN),
fc.constant(0),
fc.constant(''),
fc.string({ unit: 'binary', maxLength: 50 }),
fc.string({ unit: 'grapheme-composite', maxLength: 50 }),
fc.string({ maxLength: 50 }),
fc.boolean(),
fc.constant([]),
fc.constant({})
),
(input) => {
assert.doesNotThrow(
() => shouldRejectAdrStatus(input),
`shouldRejectAdrStatus threw on: ${JSON.stringify(input)}`
);
}
)
);
});
test('property: shouldRejectAdrStatus always returns boolean', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 100 }),
(status) => {
const result = shouldRejectAdrStatus(status);
assert.ok(typeof result === 'boolean', `Expected boolean got ${typeof result} for ${JSON.stringify(status)}`);
}
)
);
});
// Additional boundary: case variations should not reject (function is case-sensitive)
test('property: uppercase variants of reject statuses are NOT rejected (case-sensitive)', () => {
for (const status of REJECT_STATUSES) {
const upper = status.toUpperCase();
// The parser normalizes status to lowercase internally, but shouldRejectAdrStatus
// is meant to receive an already-normalized status. If it receives uppercase,
// it should return false (or true — we just want it not to throw).
assert.doesNotThrow(() => shouldRejectAdrStatus(upper));
}
});
});
describe('adr-parser: parseAdrMarkdown properties', () => {
// (c) Never throws on any string input
test('property: parseAdrMarkdown never throws on any string input', () => {
fc.assert(
fc.property(
fc.oneof(
fc.string({ unit: 'binary', maxLength: 500 }),
fc.string({ unit: 'grapheme-composite', maxLength: 500 }),
fc.string({ maxLength: 500 }),
fc.constant(''),
fc.constant('# My ADR\n\n## Status\nAccepted\n\n## Decision\n- Do the thing.\n'),
fc.constant('---\n# Not a real ADR\n---'),
fc.constant('## No title\n\n## Status\nProposed')
),
(markdown) => {
assert.doesNotThrow(
() => parseAdrMarkdown(markdown),
`parseAdrMarkdown threw on: ${JSON.stringify(markdown.slice(0, 80))}`
);
}
)
);
});
// (d) Always returns the required typed shape
test('property: parseAdrMarkdown always returns required typed shape', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 500 }),
(markdown) => {
const result = parseAdrMarkdown(markdown);
assert.ok(typeof result === 'object' && result !== null, 'result must be object');
for (const key of REQUIRED_KEYS) {
assert.ok(
Object.prototype.hasOwnProperty.call(result, key),
`missing required key: ${key}`
);
}
assert.ok(typeof result.title === 'string', 'title must be string');
assert.ok(typeof result.status === 'string', 'status must be string');
assert.ok(typeof result.context === 'string', 'context must be string');
assert.ok(Array.isArray(result.decisions), 'decisions must be array');
assert.ok(Array.isArray(result.options_considered), 'options_considered must be array');
assert.ok(Array.isArray(result.consequences_positive), 'consequences_positive must be array');
assert.ok(Array.isArray(result.consequences_negative), 'consequences_negative must be array');
assert.ok(Array.isArray(result.updates), 'updates must be array');
assert.ok(Array.isArray(result.unmapped_headers), 'unmapped_headers must be array');
}
)
);
});
// (e) Title extracted from H1 heading
// The parser applies .trim() to the heading text, so the expected title must also be trimmed.
test('property: H1 heading at line start is extracted as trimmed title', () => {
fc.assert(
fc.property(
fc.stringMatching(/^[A-Za-z0-9][A-Za-z0-9 _-]{0,49}$/),
(titleText) => {
const markdown = `# ${titleText}\n\n## Status\nAccepted\n`;
const result = parseAdrMarkdown(markdown);
const expected = titleText.trim(); // parser trims heading text
assert.equal(
result.title,
expected,
`Expected title="${expected}" got "${result.title}"`
);
}
)
);
});
// (f) Status is always a non-null string
test('property: status is always a string (never null/undefined)', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 300 }),
(markdown) => {
const result = parseAdrMarkdown(markdown);
assert.ok(
typeof result.status === 'string',
`status must be string, got ${typeof result.status}`
);
}
)
);
});
// Robustness: sourcePath option with arbitrary strings
test('property: arbitrary sourcePath option never causes throws', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 200 }),
fc.oneof(
fc.string({ maxLength: 100 }),
fc.string({ unit: 'binary', maxLength: 50 }),
fc.constant(null),
fc.constant(undefined)
),
(markdown, sourcePath) => {
assert.doesNotThrow(
() => parseAdrMarkdown(markdown, { sourcePath }),
`parseAdrMarkdown threw with sourcePath=${JSON.stringify(sourcePath)}`
);
}
)
);
});
});

View File

@@ -0,0 +1,154 @@
'use strict';
/**
* Property-based tests for config-schema.cjs
*
* Module: get-shit-done/bin/lib/config-schema.cjs
* Exported: isValidConfigKey(keyPath) -> boolean
* VALID_CONFIG_KEYS: Set<string>
* RUNTIME_STATE_KEYS: Set<string>
* DYNAMIC_KEY_PATTERNS: Array<{ test(key): boolean, ... }>
*
* Properties tested:
* (a) isValidConfigKey never throws regardless of input type/content
* (b) isValidConfigKey(key) is true for every key in VALID_CONFIG_KEYS
* (c) isValidConfigKey(key) is true for every key in RUNTIME_STATE_KEYS
* (d) Robustness: null/undefined/NaN/control-chars/binary never throw
* (e) Arbitrary garbage strings return false (not throw) from isValidConfigKey
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fc = require('./helpers/fast-check-setup.cjs');
const {
isValidConfigKey,
VALID_CONFIG_KEYS,
RUNTIME_STATE_KEYS,
} = require('../get-shit-done/bin/lib/config-schema.cjs');
describe('config-schema: isValidConfigKey properties', () => {
// (a) Never throws on any input
test('property: isValidConfigKey never throws on hostile inputs', () => {
fc.assert(
fc.property(
fc.oneof(
fc.constant(null),
fc.constant(undefined),
fc.constant(NaN),
fc.constant(Infinity),
fc.constant(-Infinity),
fc.constant(0),
fc.constant(''),
fc.constant('\x00'),
fc.constant('\n\r\t'),
fc.string({ unit: 'binary', maxLength: 100 }),
fc.string({ unit: 'grapheme-composite', maxLength: 100 }),
fc.constant([]),
fc.constant({}),
fc.boolean(),
fc.string({ maxLength: 100 })
),
(input) => {
assert.doesNotThrow(
() => isValidConfigKey(input),
`isValidConfigKey threw on input: ${JSON.stringify(input)}`
);
}
)
);
});
// (b) Every key in VALID_CONFIG_KEYS returns true
test('all VALID_CONFIG_KEYS entries are recognized as valid', () => {
for (const key of VALID_CONFIG_KEYS) {
assert.equal(
isValidConfigKey(key),
true,
`Expected isValidConfigKey(${JSON.stringify(key)}) === true`
);
}
});
// (c) Every key in RUNTIME_STATE_KEYS returns true
test('all RUNTIME_STATE_KEYS entries are recognized as valid', () => {
for (const key of RUNTIME_STATE_KEYS) {
assert.equal(
isValidConfigKey(key),
true,
`Expected isValidConfigKey(${JSON.stringify(key)}) === true (runtime state key)`
);
}
});
// (d+e) Robustness: hostile strings return boolean (not throw)
test('property: isValidConfigKey always returns a boolean for any string', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 200 }),
(key) => {
const result = isValidConfigKey(key);
assert.ok(
typeof result === 'boolean',
`isValidConfigKey must return boolean, got ${typeof result} for ${JSON.stringify(key)}`
);
}
)
);
});
test('property: binary/control-char strings return false (not throw)', () => {
fc.assert(
fc.property(
fc.oneof(
fc.string({ unit: 'binary', maxLength: 100 }),
fc.string({ unit: 'grapheme-composite', maxLength: 100 })
),
(key) => {
// Either returns true (if it happens to match a valid key) or false
// It must NOT throw
let result;
assert.doesNotThrow(() => {
result = isValidConfigKey(key);
});
assert.ok(typeof result === 'boolean');
}
)
);
});
// Boundary: well-formed dotted paths that are NOT in the schema
test('property: plausible-but-invalid dotted paths return false', () => {
// Generate dot-separated alphanumeric paths that do not match known keys
const dotPath = fc.array(
fc.stringMatching(/^[a-z][a-z0-9_]{0,10}$/),
{ minLength: 3, maxLength: 5 }
).map((parts) => 'zz_unknown.' + parts.join('.'));
fc.assert(
fc.property(dotPath, (key) => {
// Must not throw
let result;
assert.doesNotThrow(() => {
result = isValidConfigKey(key);
});
// Result is a boolean
assert.ok(typeof result === 'boolean');
})
);
});
// Boundary: empty string is not a valid config key
test('empty string is not a valid config key', () => {
const result = isValidConfigKey('');
assert.equal(result, false, 'empty string must not be a valid config key');
});
// Boundary: null/undefined/number return false (not throw, not true)
test('null, undefined, number inputs return false', () => {
assert.equal(isValidConfigKey(null), false);
assert.equal(isValidConfigKey(undefined), false);
assert.equal(isValidConfigKey(42), false);
assert.equal(isValidConfigKey(NaN), false);
});
});

View File

@@ -0,0 +1,217 @@
'use strict';
/**
* Property-based tests for context-utilization.cjs
*
* Module: get-shit-done/bin/lib/context-utilization.cjs
* Exported: classifyContextUtilization(tokensUsed, contextWindow) -> { percent, state }
*
* Thresholds (from module source):
* ratio < 0.60 → healthy
* 0.60 <= ratio < 0.70 → warning
* ratio >= 0.70 → critical
*
* Properties tested:
* (a) Boundary: across the 60% and 70% thresholds the classify flips correctly
* (b) Robustness: hostile inputs (null/undefined/NaN/Infinity/negative/wrong-type)
* always throw TypeError (documented contract) and never throw non-TypeError
* (c) Return shape: all valid inputs return an object with typed { percent, state }
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fc = require('./helpers/fast-check-setup.cjs');
const { classifyContextUtilization, STATES } = require('../get-shit-done/bin/lib/context-utilization.cjs');
// ─── Boundary constants ───────────────────────────────────────────────────────
const WARNING_THRESHOLD = 0.60; // ratio < this → healthy
const CRITICAL_THRESHOLD = 0.70; // ratio < this → warning, else critical
const WINDOW = 10000; // A fixed contextWindow that gives us clean ratio math
const W = 50; // boundary exploration width: ±50 tokens around the threshold
describe('context-utilization property tests', () => {
// ─── (a) Boundary property: classify flips at exactly 60% and 70% ────────────
test('property: classify is healthy below 60%, warning in [60%,70%), critical at ≥70%', () => {
// Test across a range of context windows.
// The boundary is at the exact RATIO — not at Math.floor(window * ratio).
// e.g. with window=1001: floor(1001 * 0.60) = 600, but 600/1001 = 0.5994 < 0.60 → healthy.
// So we compute the exact first token count that meets or exceeds the threshold.
fc.assert(
fc.property(
// contextWindow: 1000..200000 to keep ratios well-defined
fc.integer({ min: 1000, max: 200_000 }),
(contextWindow) => {
// First integer where tokensUsed/contextWindow >= WARNING_THRESHOLD
const firstWarning = Math.ceil(contextWindow * WARNING_THRESHOLD);
// First integer where tokensUsed/contextWindow >= CRITICAL_THRESHOLD
const firstCritical = Math.ceil(contextWindow * CRITICAL_THRESHOLD);
// Just below warning threshold → healthy
if (firstWarning > 0) {
const below = firstWarning - 1;
const r = classifyContextUtilization(below, contextWindow);
assert.equal(
r.state,
STATES.HEALTHY,
`tokensUsed=${below} contextWindow=${contextWindow} ratio=${(below / contextWindow).toFixed(6)} expected healthy got ${r.state}`
);
}
// At exact warning boundary → warning (unless critical collapses to same point)
if (firstWarning < firstCritical && firstWarning <= contextWindow) {
const r2 = classifyContextUtilization(firstWarning, contextWindow);
assert.equal(
r2.state,
STATES.WARNING,
`tokensUsed=${firstWarning} ratio=${(firstWarning / contextWindow).toFixed(6)} expected warning got ${r2.state}`
);
}
// At or above critical threshold → critical
if (firstCritical <= contextWindow) {
const r3 = classifyContextUtilization(firstCritical, contextWindow);
assert.equal(
r3.state,
STATES.CRITICAL,
`tokensUsed=${firstCritical} ratio=${(firstCritical / contextWindow).toFixed(6)} expected critical got ${r3.state}`
);
}
}
)
);
});
test('property: near 60% boundary the state is always healthy (never warning/critical)', () => {
// Tokens strictly below ceil(WINDOW * 0.60) must classify as healthy.
// The boundary is the FIRST integer where ratio >= 0.60 (Math.ceil).
// We sample from [firstWarning - W, firstWarning - 1] to probe just below it.
const firstWarning = Math.ceil(WINDOW * WARNING_THRESHOLD); // = 6000
const rangeMin = Math.max(0, firstWarning - W); // = 5950
const rangeMax = firstWarning - 1; // = 5999
fc.assert(
fc.property(
fc.integer({ min: rangeMin, max: rangeMax }),
(tokensUsed) => {
const r = classifyContextUtilization(tokensUsed, WINDOW);
assert.equal(
r.state,
STATES.HEALTHY,
`tokensUsed=${tokensUsed}/${WINDOW}=${(tokensUsed / WINDOW * 100).toFixed(2)}% expected healthy got ${r.state}`
);
}
)
);
});
test('property: near 70% boundary tokens at/above critical threshold must be critical', () => {
const criticalFloor = Math.ceil(WINDOW * CRITICAL_THRESHOLD);
fc.assert(
fc.property(
fc.integer({ min: criticalFloor, max: WINDOW }),
(tokensUsed) => {
const r = classifyContextUtilization(tokensUsed, WINDOW);
assert.equal(
r.state,
STATES.CRITICAL,
`tokensUsed=${tokensUsed}/${WINDOW}=${(tokensUsed / WINDOW * 100).toFixed(2)}% expected critical got ${r.state}`
);
}
)
);
});
// ─── (b) Robustness: hostile inputs always throw TypeError ────────────────────
test('property: non-integer/negative tokensUsed always throws TypeError', () => {
const invalidTokensUsed = fc.oneof(
fc.constant(null),
fc.constant(undefined),
fc.constant(NaN),
fc.constant(Infinity),
fc.constant(-Infinity),
fc.constant(-1),
fc.integer({ min: -10000, max: -1 }),
fc.double({ min: 0.1, max: 0.9 }),
fc.string(),
fc.boolean(),
fc.constant([]),
fc.constant({})
);
fc.assert(
fc.property(invalidTokensUsed, (bad) => {
assert.throws(
() => classifyContextUtilization(bad, 10000),
(err) => {
assert.ok(err instanceof TypeError, `Expected TypeError but got ${err.constructor.name}: ${err.message}`);
return true;
}
);
})
);
});
test('property: non-integer/non-positive contextWindow always throws TypeError', () => {
const invalidWindows = fc.oneof(
fc.constant(null),
fc.constant(undefined),
fc.constant(NaN),
fc.constant(Infinity),
fc.constant(-Infinity),
fc.constant(0),
fc.constant(-1),
fc.integer({ min: -10000, max: 0 }),
fc.double({ min: 0.1, max: 0.9 }),
fc.string(),
fc.boolean()
);
fc.assert(
fc.property(invalidWindows, (bad) => {
assert.throws(
() => classifyContextUtilization(1000, bad),
(err) => {
assert.ok(err instanceof TypeError, `Expected TypeError for contextWindow=${bad} but got ${err.constructor.name}: ${err.message}`);
return true;
}
);
})
);
});
// ─── (c) Return shape: all valid inputs produce typed { percent, state } ──────
test('property: valid inputs always return { percent: number[0..100], state: string }', () => {
fc.assert(
fc.property(
fc.integer({ min: 1, max: 1_000_000 }), // contextWindow
(contextWindow) => {
// tokensUsed in [0, contextWindow]
const tokensUsed = Math.floor(Math.random() * (contextWindow + 1));
const r = classifyContextUtilization(tokensUsed, contextWindow);
assert.ok(typeof r === 'object' && r !== null, 'result must be object');
assert.ok(typeof r.percent === 'number', `percent must be number got ${typeof r.percent}`);
assert.ok(typeof r.state === 'string', `state must be string got ${typeof r.state}`);
assert.ok(r.percent >= 0 && r.percent <= 100, `percent ${r.percent} out of [0,100]`);
assert.ok(
[STATES.HEALTHY, STATES.WARNING, STATES.CRITICAL].includes(r.state),
`state must be one of the STATES enum, got ${r.state}`
);
}
)
);
});
test('property: tokensUsed exceeding contextWindow clamps to 100% critical', () => {
fc.assert(
fc.property(
fc.integer({ min: 1, max: 100_000 }),
fc.integer({ min: 1, max: 100_000 }),
(contextWindow, extra) => {
const tokensUsed = contextWindow + extra; // always exceeds window
const r = classifyContextUtilization(tokensUsed, contextWindow);
assert.equal(r.state, STATES.CRITICAL, `overflow ${tokensUsed}/${contextWindow} must be critical`);
assert.equal(r.percent, 100, `overflow percent must clamp to 100, got ${r.percent}`);
}
)
);
});
});

View File

@@ -0,0 +1,194 @@
'use strict';
/**
* Property-based tests for frontmatter.cjs
*
* Module: get-shit-done/bin/lib/frontmatter.cjs
* Exported (pure): extractFrontmatter, reconstructFrontmatter, spliceFrontmatter
*
* Properties tested:
* (a) extractFrontmatter never throws on ANY string input (including binary/unicode)
* (b) extractFrontmatter always returns a plain object (not null, not array)
* (c) round-trip: reconstructFrontmatter(extractFrontmatter(spliceFrontmatter(content, obj)))
* preserves key-value pairs for simple flat string values
* (d) spliceFrontmatter never throws on any string/object combination
* (e) extractFrontmatter returns {} for content without a leading ---...--- block
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fc = require('./helpers/fast-check-setup.cjs');
const {
extractFrontmatter,
reconstructFrontmatter,
spliceFrontmatter,
} = require('../get-shit-done/bin/lib/frontmatter.cjs');
// ─── Arbitraries ─────────────────────────────────────────────────────────────
// Simple YAML key: alphanumeric + underscore, at least 1 char
const yamlKey = fc.stringMatching(/^[a-z][a-z0-9_]{0,19}$/);
// Simple YAML scalar value: printable ASCII without : ' " # newlines
const yamlScalarValue = fc.stringMatching(/^[a-zA-Z0-9 ._/-]{1,40}$/);
// ─── Tests ────────────────────────────────────────────────────────────────────
describe('frontmatter: extractFrontmatter properties', () => {
// (a) Never throws on any string input
test('property: extractFrontmatter never throws on arbitrary binary/unicode input', () => {
fc.assert(
fc.property(
fc.oneof(
fc.string({ unit: 'binary', maxLength: 300 }),
fc.string({ unit: 'grapheme-composite', maxLength: 300 }),
fc.constant(''),
fc.constant('---\n---'),
fc.constant('---\nkey: value\n---\n# body'),
fc.string({ maxLength: 300 })
),
(input) => {
assert.doesNotThrow(
() => extractFrontmatter(input),
`extractFrontmatter threw on input: ${JSON.stringify(input.slice(0, 50))}`
);
}
)
);
});
// (b) Always returns a plain object
test('property: extractFrontmatter always returns a plain object', () => {
fc.assert(
fc.property(
fc.oneof(
fc.string({ unit: 'binary', maxLength: 200 }),
fc.string({ unit: 'grapheme-composite', maxLength: 200 }),
fc.string({ maxLength: 200 })
),
(input) => {
const result = extractFrontmatter(input);
assert.ok(
typeof result === 'object' && result !== null && !Array.isArray(result),
`extractFrontmatter must return plain object, got ${JSON.stringify(result)}`
);
}
)
);
});
// (e) Returns {} for content without leading --- block
test('property: content without leading --- block returns empty object', () => {
fc.assert(
fc.property(
fc.oneof(
fc.string({ minLength: 0, maxLength: 200 }).filter((s) => !s.startsWith('---')),
fc.constant('# Just a heading'),
fc.constant('plain text content'),
fc.constant('')
),
(input) => {
const result = extractFrontmatter(input);
assert.deepEqual(
result,
{},
`Expected {} for non-frontmatter input, got ${JSON.stringify(result)}`
);
}
)
);
});
});
describe('frontmatter: reconstructFrontmatter properties', () => {
test('property: reconstructFrontmatter never throws on plain objects with string values', () => {
fc.assert(
fc.property(
fc.dictionary(yamlKey, yamlScalarValue, { maxKeys: 10 }),
(obj) => {
assert.doesNotThrow(
() => reconstructFrontmatter(obj),
`reconstructFrontmatter threw on ${JSON.stringify(obj)}`
);
}
)
);
});
test('property: reconstructFrontmatter output is a string', () => {
fc.assert(
fc.property(
fc.dictionary(yamlKey, yamlScalarValue, { maxKeys: 8 }),
(obj) => {
const result = reconstructFrontmatter(obj);
assert.ok(typeof result === 'string', `Expected string got ${typeof result}`);
}
)
);
});
test('property: reconstructFrontmatter on {} returns empty string', () => {
assert.equal(reconstructFrontmatter({}), '');
});
});
describe('frontmatter: spliceFrontmatter properties', () => {
// (d) Never throws on any combination
test('property: spliceFrontmatter never throws on arbitrary content + object', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 300 }),
fc.dictionary(yamlKey, yamlScalarValue, { maxKeys: 8 }),
(content, obj) => {
assert.doesNotThrow(
() => spliceFrontmatter(content, obj),
`spliceFrontmatter threw on content=${JSON.stringify(content.slice(0, 30))}`
);
}
)
);
});
test('property: spliceFrontmatter always returns a string', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 200 }),
fc.dictionary(yamlKey, yamlScalarValue, { maxKeys: 5 }),
(content, obj) => {
const result = spliceFrontmatter(content, obj);
assert.ok(typeof result === 'string', `Expected string got ${typeof result}`);
}
)
);
});
// (c) Round-trip: splice then extract preserves flat string keys
test('property: splice then extract round-trip preserves flat string values', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 100 }), // existing document body
// Only keys + simple values without colons/hashes that would confuse the minimal parser
fc.dictionary(
fc.stringMatching(/^[a-z][a-z0-9]{0,14}$/),
fc.stringMatching(/^[a-zA-Z0-9]{1,30}$/),
{ minKeys: 1, maxKeys: 5 }
),
(body, obj) => {
const spliced = spliceFrontmatter(body, obj);
const extracted = extractFrontmatter(spliced);
for (const [key, value] of Object.entries(obj)) {
if (typeof value === 'string' && value.length > 0) {
assert.equal(
extracted[key],
value,
`Round-trip failed for key=${key}: expected ${value} got ${extracted[key]}`
);
}
}
}
)
);
});
});

View File

@@ -0,0 +1,24 @@
'use strict';
/**
* fast-check-setup.cjs
*
* Shared configuration for all property-based tests. Require this at the
* top of every *.property.test.cjs file before any fc.assert() call.
*
* Settings:
* numRuns: 200 — enough to catch boundary bugs without slow CI
* seed: 42 — deterministic across CI runs; set GSD_FC_SEED=<n> to
* override locally for exploration
*/
const fc = require('fast-check');
const seed = process.env.GSD_FC_SEED ? Number(process.env.GSD_FC_SEED) : 42;
fc.configureGlobal({
numRuns: 200,
seed,
});
module.exports = fc;

View File

@@ -0,0 +1,242 @@
'use strict';
/**
* Property-based tests for prompt-budget.cjs
*
* Module: get-shit-done/bin/lib/prompt-budget.cjs
* Exported: estimateTokens(text), applyBudget({ sections, budget, options })
*
* Key invariants:
* - estimateTokens: always >= 0, monotonically related to string length
* - applyBudget: when hardFailed=false, estimatedTokens <= effectiveBudget
* - applyBudget: instructions and roadmap are ALWAYS kept verbatim (never trimmed)
* - applyBudget: below the minSet the call returns hardFailed=true with prompt=''
* - Budget boundary: a budget just at the effective floor triggers hard-fail;
* a budget just above it passes
*/
const { describe, test } = require('node:test');
const assert = require('node:assert/strict');
const fc = require('./helpers/fast-check-setup.cjs');
const { estimateTokens, applyBudget } = require('../get-shit-done/bin/lib/prompt-budget.cjs');
// ─── Helpers ─────────────────────────────────────────────────────────────────
function minimalSections(overrides = {}) {
return {
instructions: 'Instructions text.',
roadmap: 'Roadmap text.',
plans: [{ file: 'plan.md', content: 'Plan content.' }],
projectMd: null,
context: null,
research: null,
requirements: null,
...overrides,
};
}
// ─── estimateTokens property tests ───────────────────────────────────────────
describe('prompt-budget: estimateTokens properties', () => {
test('property: estimateTokens is always >= 0', () => {
fc.assert(
fc.property(
fc.oneof(
fc.string(),
fc.constant(null),
fc.constant(undefined),
fc.constant(''),
fc.string({ unit: 'binary', maxLength: 200 }),
fc.string({ unit: 'grapheme-composite', maxLength: 200 })
),
(input) => {
const result = estimateTokens(input);
assert.ok(typeof result === 'number', `estimateTokens must return number, got ${typeof result}`);
assert.ok(result >= 0, `estimateTokens(${JSON.stringify(input)}) must be >= 0, got ${result}`);
assert.ok(Number.isInteger(result), `estimateTokens must return integer, got ${result}`);
}
)
);
});
test('property: estimateTokens is monotonically non-decreasing as text grows', () => {
fc.assert(
fc.property(
fc.string({ maxLength: 500 }),
fc.string({ minLength: 1, maxLength: 100 }),
(base, suffix) => {
const short = estimateTokens(base);
const long = estimateTokens(base + suffix);
assert.ok(long >= short, `tokens('${base}' + suffix)=${long} < tokens('${base}')=${short}`);
}
)
);
});
test('property: estimateTokens(null/undefined) returns 0', () => {
assert.equal(estimateTokens(null), 0);
assert.equal(estimateTokens(undefined), 0);
assert.equal(estimateTokens(''), 0);
});
test('property: estimateTokens approximation is ceil(len/4)', () => {
fc.assert(
fc.property(fc.string({ minLength: 1, maxLength: 1000 }), (text) => {
const expected = Math.ceil(text.length / 4);
assert.equal(estimateTokens(text), expected);
})
);
});
});
// ─── applyBudget property tests ───────────────────────────────────────────────
describe('prompt-budget: applyBudget properties', () => {
// (a) Boundary property: budget near the hardFail threshold
test('property: when minSet > effectiveBudget, applyBudget returns hardFailed=true and prompt=""', () => {
fc.assert(
fc.property(
// Use a very small budget to force hard-fail
fc.integer({ min: 1, max: 50 }),
(tinyBudget) => {
const sections = minimalSections({
instructions: 'A'.repeat(200), // ~50 tokens
roadmap: 'B'.repeat(200), // ~50 tokens
});
const result = applyBudget({ sections, budget: tinyBudget });
if (result.metadata.hardFailed) {
assert.equal(result.prompt, '', 'hardFailed must return empty prompt');
assert.equal(result.metadata.hardFailed, true);
}
// If not hard-failed, that is also valid — result is consistent
}
)
);
});
test('property: when budget is adequate, estimatedTokens never exceeds effectiveBudget', () => {
// Use a large budget: instructions ~10 tokens + roadmap ~10 tokens + plan ~10 tokens
// With margin 10%, effectiveBudget = floor(budget * 0.9)
fc.assert(
fc.property(
fc.integer({ min: 500, max: 10_000 }),
(budget) => {
const sections = minimalSections();
const result = applyBudget({ sections, budget });
if (!result.metadata.hardFailed) {
assert.ok(
result.metadata.estimatedTokens <= result.metadata.effectiveBudget,
`estimatedTokens ${result.metadata.estimatedTokens} > effectiveBudget ${result.metadata.effectiveBudget} at budget=${budget}`
);
assert.ok(result.prompt.length > 0, 'non-hardFailed must return non-empty prompt');
}
}
)
);
});
// (b) Robustness: hostile section inputs — applyBudget should either work or throw
// clearly — it must NEVER silently return a broken shape
test('property: applyBudget always returns typed { prompt, metadata } shape on valid budget', () => {
fc.assert(
fc.property(
fc.integer({ min: 100, max: 50_000 }),
fc.string({ maxLength: 200 }),
fc.string({ maxLength: 200 }),
(budget, instructions, roadmap) => {
const sections = minimalSections({ instructions, roadmap });
const result = applyBudget({ sections, budget });
assert.ok(typeof result === 'object' && result !== null);
assert.ok(typeof result.prompt === 'string', 'prompt must be string');
assert.ok(typeof result.metadata === 'object' && result.metadata !== null);
assert.ok(typeof result.metadata.hardFailed === 'boolean');
assert.ok(typeof result.metadata.budget === 'number');
assert.ok(typeof result.metadata.effectiveBudget === 'number');
assert.ok(Array.isArray(result.metadata.omitted));
}
)
);
});
test('property: instructions are always present verbatim in the output prompt', () => {
fc.assert(
fc.property(
fc.string({ minLength: 1, maxLength: 100 }),
(instructions) => {
const sections = minimalSections({ instructions });
const result = applyBudget({ sections, budget: 100_000 });
if (!result.metadata.hardFailed) {
assert.ok(
result.prompt.includes(instructions),
`Instructions not found verbatim in prompt. Instructions: "${instructions.slice(0, 50)}"`
);
}
}
)
);
});
test('property: roadmap is always present verbatim in the output prompt', () => {
fc.assert(
fc.property(
fc.string({ minLength: 1, maxLength: 100 }),
(roadmap) => {
const sections = minimalSections({ roadmap });
const result = applyBudget({ sections, budget: 100_000 });
if (!result.metadata.hardFailed) {
assert.ok(
result.prompt.includes(roadmap),
`Roadmap not found verbatim in prompt. Roadmap: "${roadmap.slice(0, 50)}"`
);
}
}
)
);
});
test('property: safetyMarginPct in [0,50] always produces effectiveBudget <= budget', () => {
fc.assert(
fc.property(
fc.integer({ min: 1000, max: 100_000 }),
fc.integer({ min: 0, max: 50 }),
(budget, safetyMarginPct) => {
const sections = minimalSections();
const result = applyBudget({ sections, budget, options: { safetyMarginPct } });
assert.ok(
result.metadata.effectiveBudget <= budget,
`effectiveBudget ${result.metadata.effectiveBudget} > budget ${budget} at margin ${safetyMarginPct}%`
);
}
)
);
});
test('property: context/research/requirements omission is tracked in metadata.omitted', () => {
// Build a sections object where extras push it over a tight budget
fc.assert(
fc.property(
fc.string({ minLength: 400, maxLength: 800 }), // ~100-200 tokens context
(contextText) => {
const sections = minimalSections({ context: contextText });
// Use a very tight budget that forces trimming
const baseTokens = estimateTokens('Instructions text.') +
estimateTokens('Roadmap text.') +
estimateTokens('Plan content.') + 20; // overhead
const tightBudget = Math.ceil(baseTokens / 0.9) + 1; // just barely fits without context
const result = applyBudget({ sections, budget: tightBudget });
if (!result.metadata.hardFailed && result.metadata.omitted.includes('context')) {
// The note must have been injected if context was dropped
assert.equal(result.metadata.noteInjected, true,
'noteInjected should be true when context was omitted');
}
// Whether or not context was dropped, omitted is always an array
assert.ok(Array.isArray(result.metadata.omitted));
}
)
);
});
});