* feat(#454): add antagonistic tier — fast-check property tests + Stryker mutation config Adds property-based testing (fast-check v4) and mutation testing scaffolding (Stryker v8) as the antagonistic validation tier for lib/*.cjs pure logic. ## Files added ### Shared setup - tests/helpers/fast-check-setup.cjs — configureGlobal({ numRuns:200, seed:42 }) for deterministic CI; override locally with GSD_FC_SEED ### Property test suites (node:test + fast-check) - tests/context-utilization.property.test.cjs — boundary at 60%/70% thresholds (exact Math.ceil boundary, not Math.floor), TypeError on all invalid inputs, overflow clamping to 100%/critical, shape invariants (7 tests, all pass) - tests/prompt-budget.property.test.cjs — estimateTokens monotonicity + ceil(len/4) exactness; applyBudget shape invariant, instructions/roadmap verbatim, budget envelope, omit tracking (11 tests, all pass) - tests/frontmatter.property.test.cjs — extractFrontmatter/reconstructFrontmatter/ spliceFrontmatter never-throw + type shape + splice→extract round-trip (9 tests) - tests/adr-parser.property.test.cjs — shouldRejectAdrStatus boundary (3 statuses only), parseAdrMarkdown shape + title trim invariant (discovered: parser trims trailing whitespace) (9 tests, all pass) - tests/config-schema.property.test.cjs — isValidConfigKey never throws, returns boolean, accepts all VALID/RUNTIME_STATE_KEYS, rejects empty/null/unknown (8 tests) ### Stryker mutation config - stryker.config.mjs — testRunner:'command', mutate bin/lib/**/*.cjs minus 13 generated files, coverageAnalysis:'off', thresholds {high:80,low:60,break:50}, incremental:true, reporters html+clear-text+progress ### CI workflow - .github/workflows/mutation.yml — PR-gating job (pull_request + workflow_dispatch), runs stryker --incremental --since origin/next (changed files only), uploads HTML artifact; SINCE_REF via env not interpolation (injection-safe) ### Package config - package.json: +test:mutation, +test:mutation:since scripts - .gitignore: +.stryker-tmp/, +.stryker-incremental.json, +reports/mutation/ - package-lock.json: fast-check@4.8.0, @stryker-mutator/core@9.6.1 ## Verified node --test on all 5 property test files: 44 tests, 0 failures. Stryker NOT run (slow; reserved for CI). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(#454): pin upload-artifact to v7.0.1 SHA used across repo (bad SHA ea165f8d) The SHA ea165f8d65b6e75b540449d3ec4f5dde0c5a4e1 (labeled v4.6.2) does not resolve on GitHub Actions. All other workflows in this repo pin 043fb46d1a93c77aae656e7c1c64a875d1fc6a0a (v7.0.1) — align mutation.yml. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(#454): mutation workflow — replace invalid --since flag with changed-core-files --mutate scoping Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> --------- Co-authored-by: CI Rebase Check <ci@gsd-redux> Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
218 lines
8.8 KiB
JavaScript
218 lines
8.8 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* Property-based tests for context-utilization.cjs
|
|
*
|
|
* Module: get-shit-done/bin/lib/context-utilization.cjs
|
|
* Exported: classifyContextUtilization(tokensUsed, contextWindow) -> { percent, state }
|
|
*
|
|
* Thresholds (from module source):
|
|
* ratio < 0.60 → healthy
|
|
* 0.60 <= ratio < 0.70 → warning
|
|
* ratio >= 0.70 → critical
|
|
*
|
|
* Properties tested:
|
|
* (a) Boundary: across the 60% and 70% thresholds the classify flips correctly
|
|
* (b) Robustness: hostile inputs (null/undefined/NaN/Infinity/negative/wrong-type)
|
|
* always throw TypeError (documented contract) and never throw non-TypeError
|
|
* (c) Return shape: all valid inputs return an object with typed { percent, state }
|
|
*/
|
|
|
|
const { describe, test } = require('node:test');
|
|
const assert = require('node:assert/strict');
|
|
const fc = require('./helpers/fast-check-setup.cjs');
|
|
|
|
const { classifyContextUtilization, STATES } = require('../get-shit-done/bin/lib/context-utilization.cjs');
|
|
|
|
// ─── Boundary constants ───────────────────────────────────────────────────────
|
|
const WARNING_THRESHOLD = 0.60; // ratio < this → healthy
|
|
const CRITICAL_THRESHOLD = 0.70; // ratio < this → warning, else critical
|
|
const WINDOW = 10000; // A fixed contextWindow that gives us clean ratio math
|
|
const W = 50; // boundary exploration width: ±50 tokens around the threshold
|
|
|
|
describe('context-utilization property tests', () => {
|
|
// ─── (a) Boundary property: classify flips at exactly 60% and 70% ────────────
|
|
test('property: classify is healthy below 60%, warning in [60%,70%), critical at ≥70%', () => {
|
|
// Test across a range of context windows.
|
|
// The boundary is at the exact RATIO — not at Math.floor(window * ratio).
|
|
// e.g. with window=1001: floor(1001 * 0.60) = 600, but 600/1001 = 0.5994 < 0.60 → healthy.
|
|
// So we compute the exact first token count that meets or exceeds the threshold.
|
|
fc.assert(
|
|
fc.property(
|
|
// contextWindow: 1000..200000 to keep ratios well-defined
|
|
fc.integer({ min: 1000, max: 200_000 }),
|
|
(contextWindow) => {
|
|
// First integer where tokensUsed/contextWindow >= WARNING_THRESHOLD
|
|
const firstWarning = Math.ceil(contextWindow * WARNING_THRESHOLD);
|
|
// First integer where tokensUsed/contextWindow >= CRITICAL_THRESHOLD
|
|
const firstCritical = Math.ceil(contextWindow * CRITICAL_THRESHOLD);
|
|
|
|
// Just below warning threshold → healthy
|
|
if (firstWarning > 0) {
|
|
const below = firstWarning - 1;
|
|
const r = classifyContextUtilization(below, contextWindow);
|
|
assert.equal(
|
|
r.state,
|
|
STATES.HEALTHY,
|
|
`tokensUsed=${below} contextWindow=${contextWindow} ratio=${(below / contextWindow).toFixed(6)} expected healthy got ${r.state}`
|
|
);
|
|
}
|
|
|
|
// At exact warning boundary → warning (unless critical collapses to same point)
|
|
if (firstWarning < firstCritical && firstWarning <= contextWindow) {
|
|
const r2 = classifyContextUtilization(firstWarning, contextWindow);
|
|
assert.equal(
|
|
r2.state,
|
|
STATES.WARNING,
|
|
`tokensUsed=${firstWarning} ratio=${(firstWarning / contextWindow).toFixed(6)} expected warning got ${r2.state}`
|
|
);
|
|
}
|
|
|
|
// At or above critical threshold → critical
|
|
if (firstCritical <= contextWindow) {
|
|
const r3 = classifyContextUtilization(firstCritical, contextWindow);
|
|
assert.equal(
|
|
r3.state,
|
|
STATES.CRITICAL,
|
|
`tokensUsed=${firstCritical} ratio=${(firstCritical / contextWindow).toFixed(6)} expected critical got ${r3.state}`
|
|
);
|
|
}
|
|
}
|
|
)
|
|
);
|
|
});
|
|
|
|
test('property: near 60% boundary the state is always healthy (never warning/critical)', () => {
|
|
// Tokens strictly below ceil(WINDOW * 0.60) must classify as healthy.
|
|
// The boundary is the FIRST integer where ratio >= 0.60 (Math.ceil).
|
|
// We sample from [firstWarning - W, firstWarning - 1] to probe just below it.
|
|
const firstWarning = Math.ceil(WINDOW * WARNING_THRESHOLD); // = 6000
|
|
const rangeMin = Math.max(0, firstWarning - W); // = 5950
|
|
const rangeMax = firstWarning - 1; // = 5999
|
|
fc.assert(
|
|
fc.property(
|
|
fc.integer({ min: rangeMin, max: rangeMax }),
|
|
(tokensUsed) => {
|
|
const r = classifyContextUtilization(tokensUsed, WINDOW);
|
|
assert.equal(
|
|
r.state,
|
|
STATES.HEALTHY,
|
|
`tokensUsed=${tokensUsed}/${WINDOW}=${(tokensUsed / WINDOW * 100).toFixed(2)}% expected healthy got ${r.state}`
|
|
);
|
|
}
|
|
)
|
|
);
|
|
});
|
|
|
|
test('property: near 70% boundary tokens at/above critical threshold must be critical', () => {
|
|
const criticalFloor = Math.ceil(WINDOW * CRITICAL_THRESHOLD);
|
|
fc.assert(
|
|
fc.property(
|
|
fc.integer({ min: criticalFloor, max: WINDOW }),
|
|
(tokensUsed) => {
|
|
const r = classifyContextUtilization(tokensUsed, WINDOW);
|
|
assert.equal(
|
|
r.state,
|
|
STATES.CRITICAL,
|
|
`tokensUsed=${tokensUsed}/${WINDOW}=${(tokensUsed / WINDOW * 100).toFixed(2)}% expected critical got ${r.state}`
|
|
);
|
|
}
|
|
)
|
|
);
|
|
});
|
|
|
|
// ─── (b) Robustness: hostile inputs always throw TypeError ────────────────────
|
|
test('property: non-integer/negative tokensUsed always throws TypeError', () => {
|
|
const invalidTokensUsed = fc.oneof(
|
|
fc.constant(null),
|
|
fc.constant(undefined),
|
|
fc.constant(NaN),
|
|
fc.constant(Infinity),
|
|
fc.constant(-Infinity),
|
|
fc.constant(-1),
|
|
fc.integer({ min: -10000, max: -1 }),
|
|
fc.double({ min: 0.1, max: 0.9 }),
|
|
fc.string(),
|
|
fc.boolean(),
|
|
fc.constant([]),
|
|
fc.constant({})
|
|
);
|
|
fc.assert(
|
|
fc.property(invalidTokensUsed, (bad) => {
|
|
assert.throws(
|
|
() => classifyContextUtilization(bad, 10000),
|
|
(err) => {
|
|
assert.ok(err instanceof TypeError, `Expected TypeError but got ${err.constructor.name}: ${err.message}`);
|
|
return true;
|
|
}
|
|
);
|
|
})
|
|
);
|
|
});
|
|
|
|
test('property: non-integer/non-positive contextWindow always throws TypeError', () => {
|
|
const invalidWindows = fc.oneof(
|
|
fc.constant(null),
|
|
fc.constant(undefined),
|
|
fc.constant(NaN),
|
|
fc.constant(Infinity),
|
|
fc.constant(-Infinity),
|
|
fc.constant(0),
|
|
fc.constant(-1),
|
|
fc.integer({ min: -10000, max: 0 }),
|
|
fc.double({ min: 0.1, max: 0.9 }),
|
|
fc.string(),
|
|
fc.boolean()
|
|
);
|
|
fc.assert(
|
|
fc.property(invalidWindows, (bad) => {
|
|
assert.throws(
|
|
() => classifyContextUtilization(1000, bad),
|
|
(err) => {
|
|
assert.ok(err instanceof TypeError, `Expected TypeError for contextWindow=${bad} but got ${err.constructor.name}: ${err.message}`);
|
|
return true;
|
|
}
|
|
);
|
|
})
|
|
);
|
|
});
|
|
|
|
// ─── (c) Return shape: all valid inputs produce typed { percent, state } ──────
|
|
test('property: valid inputs always return { percent: number[0..100], state: string }', () => {
|
|
fc.assert(
|
|
fc.property(
|
|
fc.integer({ min: 1, max: 1_000_000 }), // contextWindow
|
|
(contextWindow) => {
|
|
// tokensUsed in [0, contextWindow]
|
|
const tokensUsed = Math.floor(Math.random() * (contextWindow + 1));
|
|
const r = classifyContextUtilization(tokensUsed, contextWindow);
|
|
|
|
assert.ok(typeof r === 'object' && r !== null, 'result must be object');
|
|
assert.ok(typeof r.percent === 'number', `percent must be number got ${typeof r.percent}`);
|
|
assert.ok(typeof r.state === 'string', `state must be string got ${typeof r.state}`);
|
|
assert.ok(r.percent >= 0 && r.percent <= 100, `percent ${r.percent} out of [0,100]`);
|
|
assert.ok(
|
|
[STATES.HEALTHY, STATES.WARNING, STATES.CRITICAL].includes(r.state),
|
|
`state must be one of the STATES enum, got ${r.state}`
|
|
);
|
|
}
|
|
)
|
|
);
|
|
});
|
|
|
|
test('property: tokensUsed exceeding contextWindow clamps to 100% critical', () => {
|
|
fc.assert(
|
|
fc.property(
|
|
fc.integer({ min: 1, max: 100_000 }),
|
|
fc.integer({ min: 1, max: 100_000 }),
|
|
(contextWindow, extra) => {
|
|
const tokensUsed = contextWindow + extra; // always exceeds window
|
|
const r = classifyContextUtilization(tokensUsed, contextWindow);
|
|
assert.equal(r.state, STATES.CRITICAL, `overflow ${tokensUsed}/${contextWindow} must be critical`);
|
|
assert.equal(r.percent, 100, `overflow percent must clamp to 100, got ${r.percent}`);
|
|
}
|
|
)
|
|
);
|
|
});
|
|
});
|