Files
msd-core/src/eval.cts
Alex V. 1a46109b97 enhance(#1579): deterministic gsd-tools query eval.score verb (#1583)
* feat(#1579): deterministic gsd-tools query eval.score verb

Split C of #1573 (pure code, lowest risk). Adds an eval.score query verb
(coverage*0.6 + infra*0.4; bands 80/60/40) mirroring the verify.* chain;
gsd-eval-auditor consumes it instead of doing weighted arithmetic in-prompt.
Non-breaking — additive only.

arXiv: 2601.15130 (Plausibility Trap/DPDM), 2507.10281 (Table Agent), 2508.15754 (TIR).

* fix(#1579): address review — domain guard, property test, glossary, SKIP_ROOT, inventory/baseline

- C3 input-domain: reject out-of-domain eval.score (require 0<=covered<=total; was emitting overall_score>100 / negatives)
- C1 property test: add tests/eval.property.test.cjs (fast-check) — determinism, band monotonicity, [0,100] bounds, never-throws
- C2 glossary: CONTEXT.md "Eval Scoring Module" entry (source-of-truth path + interface)
- C4: add `eval` to SKIP_ROOT_RESOLUTION (pure arithmetic; no .planning/ access)
- inventory: register generated eval.cjs/eval-command-router.cjs (INVENTORY-MANIFEST.json + INVENTORY.md rows)
- size: regen agent-size baseline for gsd-eval-auditor (reused gsd_run shim + eval.score step)
- eslint: ignore generated eval*.cjs (ADR-457 bin/lib migration coverage)

* fix(#1579): register eval family in alias-drift gates

Add EVAL_COMMAND_ALIASES/EVAL_SUBCOMMANDS to scripts/check-alias-drift.cjs
families and to familyArrayKeys in the manifest-coverage test, so the eval
family lands under the same drift guard as every sibling family
(state/verify/init/phase/phases/validate/roadmap). Addresses trek-e review.

check:alias-drift ok; feat-3251 coverage 9/9; eval suites 10/10.

* docs(#1579): use half-open verdict band ranges in CLI-TOOLS

overall_score is fractional and thresholds are >=80/>=60/>=40, so a score
in [79,80) is correctly NEEDS WORK despite the old '60-79' label. Relabel
bands as 60-<80 / 40-<60 / 0-<40 to match the code. Addresses trek-e nit.

* fix(#1579): validate eval.score CLI inputs

Reject unknown infra tokens and fractional counts, and pin the 80-point verdict boundary including rounding-before-banding behavior.
2026-06-24 17:16:13 -04:00

75 lines
3.2 KiB
TypeScript

/**
* Deterministic eval scoring verb (#10).
* Moves coverage/infra/overall arithmetic out of the gsd-eval-auditor prompt
* into code, per the framework's code-delegation discipline.
*/
interface EvalScoreResult {
coverage_score: number;
infra_score: number;
overall_score: number;
verdict: string;
}
function parseFlag(args: string[], flag: string): string | undefined {
const i = args.indexOf(flag);
return i >= 0 && i + 1 < args.length ? args[i + 1] : undefined;
}
const INFRA_VALUE: Record<string, number> = { ok: 1, partial: 0.5, missing: 0 };
const INFRA_TOKENS = new Set(Object.keys(INFRA_VALUE));
function computeEvalScore(covered: number, total: number, infra: string[]): EvalScoreResult {
const coverage = total > 0 ? (covered / total) * 100 : 0;
// unknown/typo tokens are treated as `missing` (score 0) by design — upstream agent only passes ok|partial|missing
const infraSum = infra.reduce((acc, s) => acc + (INFRA_VALUE[s.trim().toLowerCase()] ?? 0), 0);
const infraScore = (infraSum / 5) * 100;
const overall = coverage * 0.6 + infraScore * 0.4;
const round = (n: number) => Math.round(n * 100) / 100;
const o = round(overall);
const verdict =
o >= 80 ? 'PRODUCTION READY' :
o >= 60 ? 'NEEDS WORK' :
o >= 40 ? 'SIGNIFICANT GAPS' : 'NOT IMPLEMENTED';
return { coverage_score: round(coverage), infra_score: round(infraScore), overall_score: o, verdict };
}
function cmdEvalScore(_cwd: string, args: string[], raw: boolean): void {
const coveredRaw = parseFlag(args, '--covered');
const totalRaw = parseFlag(args, '--total');
const infraRaw = parseFlag(args, '--infra') || '';
const infra = infraRaw ? infraRaw.split(',').map((s) => s.trim().toLowerCase()) : [];
const covered = Number(coveredRaw);
const total = Number(totalRaw);
if (
coveredRaw === undefined || coveredRaw.trim() === '' ||
totalRaw === undefined || totalRaw.trim() === '' ||
!Number.isFinite(covered) || !Number.isFinite(total) ||
infra.length !== 5
) {
process.stderr.write('Usage: gsd-tools query eval.score --covered N --total N --infra a,b,c,d,e (each ok|partial|missing)\n');
process.exitCode = 1;
return;
}
// Domain validation: this is a public CLI verb, so reject out-of-domain inputs
// rather than emit nonsense (covered>total -> coverage_score>100; negatives ->
// negative scores). Counts must be non-negative integers and covered cannot
// exceed total; infra tokens must match the documented ok|partial|missing set.
if (!Number.isInteger(covered) || !Number.isInteger(total) || covered < 0 || total < 0 || covered > total) {
process.stderr.write('Invalid eval.score domain: require integer counts with 0 <= covered <= total.\n');
process.exitCode = 1;
return;
}
const invalidInfra = infra.find((s) => !INFRA_TOKENS.has(s));
if (invalidInfra !== undefined) {
process.stderr.write(`Invalid eval.score infra token: ${invalidInfra || '<empty>'}. Expected ok|partial|missing.\n`);
process.exitCode = 1;
return;
}
const result = computeEvalScore(covered, total, infra);
process.stdout.write(raw ? JSON.stringify(result) : JSON.stringify(result, null, 2));
process.stdout.write('\n');
}
export = { cmdEvalScore, computeEvalScore };