* feat(#1579): deterministic gsd-tools query eval.score verb Split C of #1573 (pure code, lowest risk). Adds an eval.score query verb (coverage*0.6 + infra*0.4; bands 80/60/40) mirroring the verify.* chain; gsd-eval-auditor consumes it instead of doing weighted arithmetic in-prompt. Non-breaking — additive only. arXiv: 2601.15130 (Plausibility Trap/DPDM), 2507.10281 (Table Agent), 2508.15754 (TIR). * fix(#1579): address review — domain guard, property test, glossary, SKIP_ROOT, inventory/baseline - C3 input-domain: reject out-of-domain eval.score (require 0<=covered<=total; was emitting overall_score>100 / negatives) - C1 property test: add tests/eval.property.test.cjs (fast-check) — determinism, band monotonicity, [0,100] bounds, never-throws - C2 glossary: CONTEXT.md "Eval Scoring Module" entry (source-of-truth path + interface) - C4: add `eval` to SKIP_ROOT_RESOLUTION (pure arithmetic; no .planning/ access) - inventory: register generated eval.cjs/eval-command-router.cjs (INVENTORY-MANIFEST.json + INVENTORY.md rows) - size: regen agent-size baseline for gsd-eval-auditor (reused gsd_run shim + eval.score step) - eslint: ignore generated eval*.cjs (ADR-457 bin/lib migration coverage) * fix(#1579): register eval family in alias-drift gates Add EVAL_COMMAND_ALIASES/EVAL_SUBCOMMANDS to scripts/check-alias-drift.cjs families and to familyArrayKeys in the manifest-coverage test, so the eval family lands under the same drift guard as every sibling family (state/verify/init/phase/phases/validate/roadmap). Addresses trek-e review. check:alias-drift ok; feat-3251 coverage 9/9; eval suites 10/10. * docs(#1579): use half-open verdict band ranges in CLI-TOOLS overall_score is fractional and thresholds are >=80/>=60/>=40, so a score in [79,80) is correctly NEEDS WORK despite the old '60-79' label. Relabel bands as 60-<80 / 40-<60 / 0-<40 to match the code. Addresses trek-e nit. * fix(#1579): validate eval.score CLI inputs Reject unknown infra tokens and fractional counts, and pin the 80-point verdict boundary including rounding-before-banding behavior.
75 lines
3.2 KiB
TypeScript
75 lines
3.2 KiB
TypeScript
/**
|
|
* Deterministic eval scoring verb (#10).
|
|
* Moves coverage/infra/overall arithmetic out of the gsd-eval-auditor prompt
|
|
* into code, per the framework's code-delegation discipline.
|
|
*/
|
|
|
|
interface EvalScoreResult {
|
|
coverage_score: number;
|
|
infra_score: number;
|
|
overall_score: number;
|
|
verdict: string;
|
|
}
|
|
|
|
function parseFlag(args: string[], flag: string): string | undefined {
|
|
const i = args.indexOf(flag);
|
|
return i >= 0 && i + 1 < args.length ? args[i + 1] : undefined;
|
|
}
|
|
|
|
const INFRA_VALUE: Record<string, number> = { ok: 1, partial: 0.5, missing: 0 };
|
|
const INFRA_TOKENS = new Set(Object.keys(INFRA_VALUE));
|
|
|
|
function computeEvalScore(covered: number, total: number, infra: string[]): EvalScoreResult {
|
|
const coverage = total > 0 ? (covered / total) * 100 : 0;
|
|
// unknown/typo tokens are treated as `missing` (score 0) by design — upstream agent only passes ok|partial|missing
|
|
const infraSum = infra.reduce((acc, s) => acc + (INFRA_VALUE[s.trim().toLowerCase()] ?? 0), 0);
|
|
const infraScore = (infraSum / 5) * 100;
|
|
const overall = coverage * 0.6 + infraScore * 0.4;
|
|
const round = (n: number) => Math.round(n * 100) / 100;
|
|
const o = round(overall);
|
|
const verdict =
|
|
o >= 80 ? 'PRODUCTION READY' :
|
|
o >= 60 ? 'NEEDS WORK' :
|
|
o >= 40 ? 'SIGNIFICANT GAPS' : 'NOT IMPLEMENTED';
|
|
return { coverage_score: round(coverage), infra_score: round(infraScore), overall_score: o, verdict };
|
|
}
|
|
|
|
function cmdEvalScore(_cwd: string, args: string[], raw: boolean): void {
|
|
const coveredRaw = parseFlag(args, '--covered');
|
|
const totalRaw = parseFlag(args, '--total');
|
|
const infraRaw = parseFlag(args, '--infra') || '';
|
|
const infra = infraRaw ? infraRaw.split(',').map((s) => s.trim().toLowerCase()) : [];
|
|
const covered = Number(coveredRaw);
|
|
const total = Number(totalRaw);
|
|
if (
|
|
coveredRaw === undefined || coveredRaw.trim() === '' ||
|
|
totalRaw === undefined || totalRaw.trim() === '' ||
|
|
!Number.isFinite(covered) || !Number.isFinite(total) ||
|
|
infra.length !== 5
|
|
) {
|
|
process.stderr.write('Usage: gsd-tools query eval.score --covered N --total N --infra a,b,c,d,e (each ok|partial|missing)\n');
|
|
process.exitCode = 1;
|
|
return;
|
|
}
|
|
// Domain validation: this is a public CLI verb, so reject out-of-domain inputs
|
|
// rather than emit nonsense (covered>total -> coverage_score>100; negatives ->
|
|
// negative scores). Counts must be non-negative integers and covered cannot
|
|
// exceed total; infra tokens must match the documented ok|partial|missing set.
|
|
if (!Number.isInteger(covered) || !Number.isInteger(total) || covered < 0 || total < 0 || covered > total) {
|
|
process.stderr.write('Invalid eval.score domain: require integer counts with 0 <= covered <= total.\n');
|
|
process.exitCode = 1;
|
|
return;
|
|
}
|
|
const invalidInfra = infra.find((s) => !INFRA_TOKENS.has(s));
|
|
if (invalidInfra !== undefined) {
|
|
process.stderr.write(`Invalid eval.score infra token: ${invalidInfra || '<empty>'}. Expected ok|partial|missing.\n`);
|
|
process.exitCode = 1;
|
|
return;
|
|
}
|
|
const result = computeEvalScore(covered, total, infra);
|
|
process.stdout.write(raw ? JSON.stringify(result) : JSON.stringify(result, null, 2));
|
|
process.stdout.write('\n');
|
|
}
|
|
|
|
export = { cmdEvalScore, computeEvalScore };
|