* feat(#1579): deterministic gsd-tools query eval.score verb Split C of #1573 (pure code, lowest risk). Adds an eval.score query verb (coverage*0.6 + infra*0.4; bands 80/60/40) mirroring the verify.* chain; gsd-eval-auditor consumes it instead of doing weighted arithmetic in-prompt. Non-breaking — additive only. arXiv: 2601.15130 (Plausibility Trap/DPDM), 2507.10281 (Table Agent), 2508.15754 (TIR). * fix(#1579): address review — domain guard, property test, glossary, SKIP_ROOT, inventory/baseline - C3 input-domain: reject out-of-domain eval.score (require 0<=covered<=total; was emitting overall_score>100 / negatives) - C1 property test: add tests/eval.property.test.cjs (fast-check) — determinism, band monotonicity, [0,100] bounds, never-throws - C2 glossary: CONTEXT.md "Eval Scoring Module" entry (source-of-truth path + interface) - C4: add `eval` to SKIP_ROOT_RESOLUTION (pure arithmetic; no .planning/ access) - inventory: register generated eval.cjs/eval-command-router.cjs (INVENTORY-MANIFEST.json + INVENTORY.md rows) - size: regen agent-size baseline for gsd-eval-auditor (reused gsd_run shim + eval.score step) - eslint: ignore generated eval*.cjs (ADR-457 bin/lib migration coverage) * fix(#1579): register eval family in alias-drift gates Add EVAL_COMMAND_ALIASES/EVAL_SUBCOMMANDS to scripts/check-alias-drift.cjs families and to familyArrayKeys in the manifest-coverage test, so the eval family lands under the same drift guard as every sibling family (state/verify/init/phase/phases/validate/roadmap). Addresses trek-e review. check:alias-drift ok; feat-3251 coverage 9/9; eval suites 10/10. * docs(#1579): use half-open verdict band ranges in CLI-TOOLS overall_score is fractional and thresholds are >=80/>=60/>=40, so a score in [79,80) is correctly NEEDS WORK despite the old '60-79' label. Relabel bands as 60-<80 / 40-<60 / 0-<40 to match the code. Addresses trek-e nit. * fix(#1579): validate eval.score CLI inputs Reject unknown infra tokens and fractional counts, and pin the 80-point verdict boundary including rounding-before-banding behavior.
102 lines
4.3 KiB
JavaScript
102 lines
4.3 KiB
JavaScript
'use strict';
|
|
|
|
/**
|
|
* Property-based tests for the eval scoring module (#10 / #1579).
|
|
*
|
|
* Module: gsd-core/bin/lib/eval.cjs
|
|
* Exported: computeEvalScore(covered, total, infra), cmdEvalScore(cwd, args, raw)
|
|
*
|
|
* Properties tested:
|
|
* (a) determinism — computeEvalScore is pure: identical inputs deep-equal across calls
|
|
* (b) output shape — always { coverage_score, infra_score, overall_score, verdict };
|
|
* scores finite; verdict is exactly the band implied by overall_score
|
|
* (c) overall_score derivation — equals round(coverage*0.6 + infra*0.4) within rounding
|
|
* (d) band monotonicity — a higher overall_score never maps to a lower-quality verdict
|
|
* (e) valid-domain bounds — for 0<=covered<=total and infra in {ok,partial,missing},
|
|
* every score lands in [0,100]
|
|
* (f) never throws — tolerates arbitrary infra tokens/lengths and numeric inputs
|
|
*/
|
|
|
|
const { describe, test } = require('node:test');
|
|
const assert = require('node:assert/strict');
|
|
const fc = require('./helpers/fast-check-setup.cjs');
|
|
const { computeEvalScore } = require('../gsd-core/bin/lib/eval.cjs');
|
|
|
|
const INFRA_TOKENS = ['ok', 'partial', 'missing'];
|
|
const VERDICTS = ['PRODUCTION READY', 'NEEDS WORK', 'SIGNIFICANT GAPS', 'NOT IMPLEMENTED'];
|
|
const RANK = { 'NOT IMPLEMENTED': 0, 'SIGNIFICANT GAPS': 1, 'NEEDS WORK': 2, 'PRODUCTION READY': 3 };
|
|
const band = (o) =>
|
|
o >= 80 ? 'PRODUCTION READY' :
|
|
o >= 60 ? 'NEEDS WORK' :
|
|
o >= 40 ? 'SIGNIFICANT GAPS' : 'NOT IMPLEMENTED';
|
|
|
|
// Valid-domain generator: 0 <= covered <= total, exactly 5 infra tokens.
|
|
const validDomain = fc.record({
|
|
total: fc.nat({ max: 1000 }),
|
|
infra: fc.array(fc.constantFrom(...INFRA_TOKENS), { minLength: 5, maxLength: 5 }),
|
|
}).chain(({ total, infra }) =>
|
|
fc.nat({ max: total }).map((covered) => ({ covered, total, infra })));
|
|
|
|
describe('computeEvalScore — properties', () => {
|
|
test('(a) deterministic / pure', () => {
|
|
fc.assert(fc.property(validDomain, ({ covered, total, infra }) => {
|
|
assert.deepEqual(
|
|
computeEvalScore(covered, total, infra),
|
|
computeEvalScore(covered, total, infra),
|
|
);
|
|
}));
|
|
});
|
|
|
|
test('(b) output shape + verdict matches band', () => {
|
|
fc.assert(fc.property(validDomain, ({ covered, total, infra }) => {
|
|
const r = computeEvalScore(covered, total, infra);
|
|
for (const k of ['coverage_score', 'infra_score', 'overall_score']) {
|
|
assert.ok(Number.isFinite(r[k]), `${k} must be finite`);
|
|
}
|
|
assert.ok(VERDICTS.includes(r.verdict), `verdict must be one of the four bands`);
|
|
assert.equal(r.verdict, band(r.overall_score));
|
|
}));
|
|
});
|
|
|
|
test('(c) overall_score = round(coverage*0.6 + infra*0.4)', () => {
|
|
fc.assert(fc.property(validDomain, ({ covered, total, infra }) => {
|
|
const r = computeEvalScore(covered, total, infra);
|
|
const expected = Math.round((r.coverage_score * 0.6 + r.infra_score * 0.4) * 100) / 100;
|
|
// coverage_score/infra_score are pre-rounded to 2dp; allow compounded-rounding slack.
|
|
assert.ok(Math.abs(r.overall_score - expected) <= 0.05,
|
|
`overall_score ${r.overall_score} should equal ${expected} within rounding`);
|
|
}));
|
|
});
|
|
|
|
test('(d) verdict band monotonic in overall_score', () => {
|
|
fc.assert(fc.property(validDomain, validDomain, (a, b) => {
|
|
const ra = computeEvalScore(a.covered, a.total, a.infra);
|
|
const rb = computeEvalScore(b.covered, b.total, b.infra);
|
|
if (ra.overall_score <= rb.overall_score) {
|
|
assert.ok(RANK[ra.verdict] <= RANK[rb.verdict],
|
|
`score ${ra.overall_score}<=${rb.overall_score} but verdict rank ${ra.verdict}>${rb.verdict}`);
|
|
}
|
|
}));
|
|
});
|
|
|
|
test('(e) valid-domain scores stay within [0,100]', () => {
|
|
fc.assert(fc.property(validDomain, ({ covered, total, infra }) => {
|
|
const r = computeEvalScore(covered, total, infra);
|
|
for (const k of ['coverage_score', 'infra_score', 'overall_score']) {
|
|
assert.ok(r[k] >= 0 && r[k] <= 100, `${k}=${r[k]} must be in [0,100]`);
|
|
}
|
|
}));
|
|
});
|
|
|
|
test('(f) never throws on arbitrary infra tokens / lengths / numbers', () => {
|
|
fc.assert(fc.property(
|
|
fc.integer({ min: -1000, max: 1000 }),
|
|
fc.integer({ min: -1000, max: 1000 }),
|
|
fc.array(fc.string(), { maxLength: 12 }),
|
|
(covered, total, infra) => {
|
|
assert.doesNotThrow(() => computeEvalScore(covered, total, infra));
|
|
},
|
|
));
|
|
});
|
|
});
|