Removes kilo, kimi, kimi-code, copilot, windsurf, augment, trae, qwen, hermes, cline, codebuddy and pi end to end: capability descriptors, installer branches and converters (bin/install.js 14.9k -> 11.2k lines), TypeScript converters, hook surfaces and runtime homes, review lanes qwen/kimi-code, the two pi migrations, Kimi payload normalization in the hook guards, dead hostBehaviors vocabulary, launcher home probes, fixtures, runtime-specific tests and the prose that presented them as supported. Installer output for the six kept runtimes is byte-identical to before the prune. The Kimi tool-vocabulary tests in workflow-guard, read-guard and read-injection-scanner are left in place pending a decision.
725 lines
35 KiB
JavaScript
725 lines
35 KiB
JavaScript
'use strict';
|
|
process.env.MSD_TEST_MODE = '1';
|
|
|
|
/**
|
|
* reviewer-lane-declarations.test.cjs — behavioral tests for ADR-2782 Phase 5a
|
|
* (chore #2798): declaring the eleven existing reviewer lanes as capability-
|
|
* manifest data.
|
|
*
|
|
* Implements every row carrying a Test name in
|
|
* `.msd/phase/chore-2798-declare-reviewer-lanes/50-test-matrix.md` (sections
|
|
* A-E). See that phase's `40-design.md` for the behavior table the matrix
|
|
* derives from. Test names are copied verbatim from the matrix.
|
|
*
|
|
* AMENDED by ADR-2782 Phase 7 (chore #2801), which removes the
|
|
* `runtime.hostBehaviors.reviewerCli` derived legacy alias. Rows that asserted
|
|
* the alias still contributed a slug are inverted here rather than deleted, so
|
|
* the file keeps a guard against reintroduction. See
|
|
* `.msd/phase/chore-2801-remove-reviewercli-alias/50-test-matrix.md` rows
|
|
* B1 and C1-C10, P1.
|
|
*
|
|
* THE SINGLE MOST IMPORTANT PROPERTY (matrix "Red-before-green"): the roster is
|
|
* eleven slugs before this phase and eleven after, with IDENTICAL membership —
|
|
* C1 is the keystone, asserted against a LITERAL list, never against a value
|
|
* computed by the same machinery under test. E1 is the highest-value row: it
|
|
* asserts the manifest and `src/review-lane-descriptor.cts`'s `REVIEWER_LANES`
|
|
* describe the same eleven lanes field-for-field — the whole premise of
|
|
* ADR-2782 is that there is no translation layer between the two surfaces.
|
|
*
|
|
* Level choice, per the matrix's own "Units" note: the eleven
|
|
* `capabilities/*\/capability.json` files are validated through the existing
|
|
* registry generator (`loadAndValidate` / `validateCrossCapability` / the
|
|
* generated `capability-registry.cjs`), never by reading JSON text — a
|
|
* source-grep assertion would be rejected by `local/no-source-grep` and would
|
|
* prove nothing about validity. `src/review-reviewer-selection.cts`'s roster
|
|
* derivation (`deriveReviewerSlugs`) is exercised directly against synthetic
|
|
* registries for the C-section rows that need to isolate one input class at a
|
|
* time (Independence: each test builds its own fixture; no shared mutable
|
|
* state).
|
|
*
|
|
* `SHIPPED` below is the one deliberate exception to "each test builds its own
|
|
* fixture": it is the REAL, already-validated capability set (computed once,
|
|
* read-only — no test mutates `SHIPPED.capMap` or any capability object
|
|
* inside it), reused across the A/B/D/E rows that assert against the actual
|
|
* shipped repo rather than a synthetic input class. Recomputing it per test
|
|
* would re-scan and re-validate all of `capabilities/*` on every one of those
|
|
* rows for no behavioral benefit.
|
|
*/
|
|
|
|
const { test, describe } = require('node:test');
|
|
const assert = require('node:assert/strict');
|
|
const fs = require('node:fs');
|
|
const path = require('node:path');
|
|
const fc = require('fast-check');
|
|
|
|
const {
|
|
loadAndValidate,
|
|
validateCapability,
|
|
validateCrossCapability,
|
|
deriveProfileMembership,
|
|
deriveCapabilityClusters,
|
|
} = require('../scripts/gen-capability-registry.cjs');
|
|
|
|
const {
|
|
KEBAB_RE,
|
|
KNOWN_REVIEWER_FIELDS,
|
|
} = require('../msd-core/bin/lib/capability-validator.cjs');
|
|
|
|
const {
|
|
REVIEWER_LANES,
|
|
checkReviewerLaneParity,
|
|
} = require('../msd-core/bin/lib/review-lane-descriptor.cjs');
|
|
|
|
// Kept as a whole-module reference (rather than only destructuring) so C5 can
|
|
// assert on the module's OWN export surface, not just the names we happen to use.
|
|
const reviewerSelectionModule = require('../msd-core/bin/lib/review-reviewer-selection.cjs');
|
|
const {
|
|
KNOWN_REVIEWER_SLUGS,
|
|
deriveReviewerSlugs,
|
|
resolveReviewerSelection,
|
|
} = reviewerSelectionModule;
|
|
|
|
// Generated registry (ADR-894) — `.capabilities` is keyed by capability id;
|
|
// each value is the whole manifest object, exactly as loaded from disk.
|
|
const capabilityRegistry = require('../msd-core/bin/lib/capability-registry.cjs');
|
|
|
|
const ROOT = path.join(__dirname, '..');
|
|
|
|
/**
|
|
* The four net-new lane-only `role:"reviewer"` capabilities (ADR-2782 D3).
|
|
*
|
|
* Was five: `gemini` was retired by #4709 (Google sunset Gemini CLI 2026-06-18, the same sunset
|
|
* that removed the gemini RUNTIME in #1928/1.8.0), so `capabilities/gemini/` no longer exists.
|
|
*/
|
|
const NEW_LANE_ONLY_IDS = ['coderabbit', 'ollama', 'lm-studio', 'llama-cpp'];
|
|
|
|
/** The six pre-existing dual-purpose `role:"runtime"` capabilities. */
|
|
const RUNTIME_REVIEWER_IDS = ['antigravity', 'claude', 'codex', 'cursor', 'opencode'];
|
|
|
|
/**
|
|
* The shipped roster BEFORE this phase, as a literal (not computed) sorted
|
|
* list. C1 compares `KNOWN_REVIEWER_SLUGS` against THIS, never against a
|
|
* value produced by `deriveReviewerSlugs` itself — a self-consistent-but-wrong
|
|
* refactor would otherwise sail through.
|
|
*/
|
|
const LITERAL_ROSTER = [
|
|
'antigravity', 'claude', 'coderabbit', 'codex', 'cursor',
|
|
'llama_cpp', 'lm_studio', 'ollama', 'opencode',
|
|
];
|
|
|
|
/**
|
|
* The real, already-validated capability set. Read-only; see the file header
|
|
* for why this is shared instead of rebuilt per test. `new Set()` for central
|
|
* keys skips config-schema collision detection — irrelevant to this phase's
|
|
* rows and not part of what any of them assert (same choice the existing
|
|
* `shippedRegistryOutputIsUnchangedByHarvestWidening` test makes).
|
|
*/
|
|
const SHIPPED = loadAndValidate(new Set());
|
|
|
|
// ─── A. The five new lane-only capabilities ─────────────────────────────────
|
|
|
|
describe('A. The five new lane-only capabilities', () => {
|
|
test('newLaneOnlyCapabilitiesValidate', () => {
|
|
assert.deepEqual(
|
|
SHIPPED.errors, [],
|
|
`expected the shipped capability set to validate cleanly, got: ${JSON.stringify(SHIPPED.errors)}`,
|
|
);
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
assert.ok(SHIPPED.capMap.has(id), `expected capability "${id}" to be loaded from capabilities/${id}/`);
|
|
}
|
|
});
|
|
|
|
test('newLaneCapabilitiesUseTheReviewerRole', () => {
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.equal(cap.role, 'reviewer', `capability "${id}" must declare role:"reviewer"`);
|
|
}
|
|
});
|
|
|
|
test('newLaneCapabilitiesDeclareAReviewerBody', () => {
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.ok(
|
|
cap.reviewer && typeof cap.reviewer === 'object' && !Array.isArray(cap.reviewer),
|
|
`capability "${id}" must carry a "reviewer" body`,
|
|
);
|
|
assert.ok(
|
|
typeof cap.reviewer.slug === 'string' && cap.reviewer.slug.length > 0,
|
|
`capability "${id}" reviewer body must declare a non-empty slug`,
|
|
);
|
|
}
|
|
});
|
|
|
|
test('laneOnlyCapabilitiesHaveNoRuntimeBody', () => {
|
|
// A runtime body would be a validation ERROR for role:"reviewer" — these
|
|
// are not install targets (design 40-design.md "A3").
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.equal('runtime' in cap, false, `lane-only capability "${id}" must not carry a "runtime" body`);
|
|
}
|
|
});
|
|
|
|
test('laneOnlyCapabilitiesNeedNoRuntimeCompat', () => {
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.equal('runtimeCompat' in cap, false, `lane-only capability "${id}" must not declare runtimeCompat`);
|
|
// The absence must not cost it validity (D3: not required for role:"reviewer").
|
|
const errs = validateCapability(cap, id);
|
|
assert.deepEqual(
|
|
errs, [],
|
|
`capability "${id}" without runtimeCompat must still validate cleanly, got: ${JSON.stringify(errs)}`,
|
|
);
|
|
}
|
|
});
|
|
|
|
test('laneOnlyCapabilitiesContributeNoArtifacts', () => {
|
|
// No install surface, nothing to install: none of the fields buildRegistry
|
|
// harvests for a feature capability's install/loop-point surface.
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
for (const field of ['skills', 'agents', 'steps', 'gates', 'contributions']) {
|
|
const value = cap[field];
|
|
assert.ok(
|
|
value === undefined || (Array.isArray(value) && value.length === 0),
|
|
`lane-only capability "${id}" must contribute no ${field}, got: ${JSON.stringify(value)}`,
|
|
);
|
|
}
|
|
}
|
|
// Nothing routes through them at the registry level either.
|
|
const bySkillOwners = new Set(Object.values(capabilityRegistry.bySkill));
|
|
const byAgentOwners = new Set(Object.values(capabilityRegistry.byAgent));
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
assert.equal(bySkillOwners.has(id), false, `"${id}" must own no skill in the generated registry`);
|
|
assert.equal(byAgentOwners.has(id), false, `"${id}" must own no agent in the generated registry`);
|
|
}
|
|
});
|
|
|
|
test('laneCapabilityIdsAreKebabWhileSlugsMaySnake', () => {
|
|
// The build-breaking trap (ADR-2782's three-namespace trap): the epic body
|
|
// and #2798 both literally specified `capabilities/lm_studio/` and
|
|
// `capabilities/llama_cpp/`, which fail KEBAB_RE outright — the folder/id
|
|
// MUST be kebab while `reviewer.slug` keeps the shipped roster's snake form.
|
|
const kebabIdToSnakeSlug = { 'lm-studio': 'lm_studio', 'llama-cpp': 'llama_cpp' };
|
|
for (const [id, expectedSlug] of Object.entries(kebabIdToSnakeSlug)) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.ok(KEBAB_RE.test(id), `capability id "${id}" must satisfy KEBAB_RE (the folder-name grammar)`);
|
|
assert.equal(cap.reviewer.slug, expectedSlug, `capability "${id}" reviewer.slug must be "${expectedSlug}"`);
|
|
assert.equal(
|
|
KEBAB_RE.test(cap.reviewer.slug), false,
|
|
`reviewer.slug "${cap.reviewer.slug}" must NOT satisfy KEBAB_RE — it deliberately differs from the kebab id`,
|
|
);
|
|
// The rejected alternative the epic literally specified as a folder name.
|
|
assert.equal(
|
|
KEBAB_RE.test(expectedSlug), false,
|
|
`"${expectedSlug}" would fail KEBAB_RE as a folder/id — that is exactly the trap this row guards`,
|
|
);
|
|
}
|
|
// The other two new capabilities are single-word and unaffected: id === slug.
|
|
for (const id of ['coderabbit', 'ollama']) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.ok(KEBAB_RE.test(id), `capability id "${id}" must satisfy KEBAB_RE`);
|
|
assert.equal(cap.reviewer.slug, id, `single-word capability "${id}" must have a matching slug`);
|
|
}
|
|
});
|
|
|
|
test('laneOnlyCapabilitiesReceiveNoProfileMembership', () => {
|
|
// tier is required (source of truth for the requires-closure), but with no
|
|
// skills, deriveProfileMembership/deriveCapabilityClusters skip them —
|
|
// membership is computed and inert, not simply "not applicable".
|
|
const profileMembership = deriveProfileMembership(SHIPPED.capMap);
|
|
const capabilityClusters = deriveCapabilityClusters(SHIPPED.capMap);
|
|
for (const id of NEW_LANE_ONLY_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.ok(typeof cap.tier === 'string' && cap.tier.length > 0, `lane-only capability "${id}" must still declare a tier`);
|
|
assert.equal(id in profileMembership, false, `lane-only capability "${id}" must receive no profile membership`);
|
|
assert.equal(id in capabilityClusters, false, `lane-only capability "${id}" must own no cluster`);
|
|
}
|
|
});
|
|
});
|
|
|
|
// ─── B. The six existing runtime capabilities ───────────────────────────────
|
|
|
|
describe('B. The six existing runtime capabilities', () => {
|
|
test('dualPurposeRuntimesCarryAReviewerBody', () => {
|
|
for (const id of RUNTIME_REVIEWER_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
assert.equal(cap.role, 'runtime', `"${id}" must remain role:"runtime"`);
|
|
assert.ok(cap.runtime && typeof cap.runtime === 'object', `"${id}" must retain its runtime body`);
|
|
assert.ok(cap.reviewer && typeof cap.reviewer === 'object', `"${id}" must gain a reviewer body alongside it (D1)`);
|
|
const errs = validateCapability(cap, id);
|
|
assert.deepEqual(errs, [], `"${id}" carrying both bodies must validate cleanly, got: ${JSON.stringify(errs)}`);
|
|
}
|
|
});
|
|
|
|
/**
|
|
* Real per-capability snapshots of `cap.runtime`'s own KEY SET, captured at
|
|
* this phase's boundary (`git status` confirms these six files' only
|
|
* uncommitted change is the added `reviewer` key). Adding that sibling key
|
|
* must not add, remove, or rename anything inside `runtime` — a top-level
|
|
* key drift here means the edit that added `reviewer` also touched
|
|
* `runtime`, by accident or by a future careless merge of the two bodies.
|
|
*
|
|
* Deliberately a KEY-SET snapshot, not a full-content one: embedding all six
|
|
* ~15-70-field runtime bodies verbatim would duplicate six actively-edited
|
|
* install descriptors into the test as a second source of truth that drifts
|
|
* on every legitimate future runtime change (these are the most frequently
|
|
* touched capabilities in the repo). Value-level integrity is covered by
|
|
* `validateCapability` (schema-complete) below and by the no-cross-
|
|
* contamination check (no reviewer-only field name inside `runtime`, and
|
|
* vice versa) — together these catch "the sibling edit corrupted this
|
|
* body" without requiring line-for-line duplication.
|
|
*/
|
|
// #2871 Phase 2 added `triggerPrecedence` (required-with-default) to every
|
|
// runtime body — a real, deliberate key-set change, not drift. Reflected
|
|
// here in sorted position, same as every other key in these snapshots.
|
|
const EXPECTED_RUNTIME_KEYS = {
|
|
antigravity: ['artifactLayout', 'commandStyle', 'configFormat', 'configHome', 'extendedHookEvents', 'hookEvents', 'hooksSurface', 'hostBehaviors', 'hostIntegration', 'installSurface', 'localConfigDir', 'permissionWriter', 'sandboxTier', 'supportTier', 'triggerPrecedence', 'writesSharedSettings'],
|
|
claude: ['artifactLayout', 'commandStyle', 'configFormat', 'configHome', 'extendedHookEvents', 'harnessIsolationFlag', 'hookEvents', 'hooksSurface', 'hostBehaviors', 'hostIntegration', 'installSurface', 'localConfigDir', 'permissionWriter', 'sandboxTier', 'supportTier', 'triggerPrecedence', 'writesSharedSettings'],
|
|
codex: ['artifactLayout', 'commandStyle', 'configFormat', 'configHome', 'extendedHookEvents', 'hookEvents', 'hooksSurface', 'hostBehaviors', 'hostIntegration', 'installSurface', 'localConfigDir', 'orchestratorExec', 'permissionWriter', 'sandboxTier', 'supportTier', 'triggerPrecedence', 'writesSharedSettings'],
|
|
cursor: ['artifactLayout', 'commandStyle', 'configFormat', 'configHome', 'extendedHookEvents', 'harnessIsolationFlag', 'hookEvents', 'hooksSurface', 'hostBehaviors', 'hostIntegration', 'installSurface', 'localConfigDir', 'permissionWriter', 'sandboxTier', 'supportTier', 'triggerPrecedence', 'writesSharedSettings'],
|
|
opencode: ['artifactLayout', 'commandStyle', 'configFormat', 'configHome', 'extendedHookEvents', 'extensionEvents', 'hooksSurface', 'hostBehaviors', 'hostIntegration', 'installSurface', 'localConfigDir', 'orchestratorExec', 'permissionWriter', 'sandboxTier', 'supportTier', 'triggerPrecedence', 'writesSharedSettings'],
|
|
};
|
|
|
|
test('runtimeBodiesAreUnchangedByLaneDeclaration', () => {
|
|
for (const id of RUNTIME_REVIEWER_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
|
|
assert.deepEqual(
|
|
Object.keys(cap.runtime).sort(), EXPECTED_RUNTIME_KEYS[id],
|
|
`"${id}" runtime body's key set must be unperturbed by adding the sibling reviewer body`,
|
|
);
|
|
|
|
// No cross-contamination in either direction — a field from one body
|
|
// leaking into the other would be an install-behavior-changing defect
|
|
// that schema validation alone (which tolerates unknown fields via a
|
|
// warning, not an error) would not catch.
|
|
const runtimeKeys = new Set(Object.keys(cap.runtime));
|
|
for (const reviewerField of KNOWN_REVIEWER_FIELDS) {
|
|
assert.equal(
|
|
runtimeKeys.has(reviewerField), false,
|
|
`"${id}" runtime body must not contain reviewer-only field "${reviewerField}"`,
|
|
);
|
|
}
|
|
const reviewerKeys = Object.keys(cap.reviewer);
|
|
assert.ok(
|
|
reviewerKeys.every((k) => KNOWN_REVIEWER_FIELDS.has(k)),
|
|
`"${id}" reviewer body must contain only known reviewer fields, got: ${JSON.stringify(reviewerKeys)}`,
|
|
);
|
|
}
|
|
});
|
|
|
|
test('reviewerCliAliasIsRemovedFromEveryShippedManifest', () => {
|
|
// Phase 7 (#2801): the deprecation window opened in 1.9.0 (Phase 5a) and
|
|
// 1.9.1 + 1.10.0 have since shipped, so the alias goes. Each of these six
|
|
// already declares a `reviewer` body whose slug equals its capability id, so
|
|
// removing the key costs none of them a lane — C1 is the proof.
|
|
for (const id of RUNTIME_REVIEWER_IDS) {
|
|
const cap = SHIPPED.capMap.get(id);
|
|
const hb = cap.runtime.hostBehaviors || {};
|
|
assert.equal(
|
|
Object.prototype.hasOwnProperty.call(hb, 'reviewerCli'), false,
|
|
`"${id}" must no longer declare hostBehaviors.reviewerCli (removed in Phase 7 / #2801); declare a reviewer body instead`,
|
|
);
|
|
}
|
|
});
|
|
|
|
test('vestigialAliasKeyDoesNotAddASecondSlug', () => {
|
|
// Post-#2801 this is no longer a PRECEDENCE rule (body beats alias) — the
|
|
// key is simply never read. Kept, with slug !== capId so a stray
|
|
// contribution would be observable as a distinct roster entry rather than
|
|
// hidden by an accidental string match, because an out-of-tree manifest can
|
|
// still carry the vestigial key for years.
|
|
const registry = {
|
|
capabilities: {
|
|
'dual-purpose-cap': {
|
|
role: 'runtime',
|
|
runtime: { hostBehaviors: { reviewerCli: true } },
|
|
reviewer: { slug: 'dual-slug' },
|
|
},
|
|
},
|
|
};
|
|
const roster = deriveReviewerSlugs(registry);
|
|
assert.equal(roster.length, 1, `expected exactly one contribution, not two, got: ${JSON.stringify(roster)}`);
|
|
assert.deepEqual(roster, ['dual-slug']);
|
|
assert.equal(roster.includes('dual-purpose-cap'), false, 'the removed alias key must not contribute the capability id');
|
|
});
|
|
});
|
|
|
|
// ─── C. Roster derivation — src/review-reviewer-selection.cts ─────────────
|
|
|
|
describe('C. Roster derivation — src/review-reviewer-selection.cts', () => {
|
|
test('rosterMembershipIsUnchangedByAliasRemoval', () => {
|
|
// KEYSTONE. This row is GREEN before and after #2801 — it is the invariant
|
|
// the phase must not break, not a red row. The literal list is never
|
|
// computed by the machinery under test.
|
|
assert.equal(KNOWN_REVIEWER_SLUGS.length, 9, 'roster must be exactly 9');
|
|
assert.deepEqual(
|
|
[...KNOWN_REVIEWER_SLUGS].sort(), LITERAL_ROSTER,
|
|
`roster must be exactly the declared lane set, got: ${JSON.stringify(KNOWN_REVIEWER_SLUGS)}`,
|
|
);
|
|
});
|
|
|
|
test('declaredReviewerBodyContributesItsSlug', () => {
|
|
const registry = { capabilities: { 'my-cap': { role: 'reviewer', reviewer: { slug: 'my-lane' } } } };
|
|
assert.deepEqual(deriveReviewerSlugs(registry), ['my-lane']);
|
|
});
|
|
|
|
test('aliasOnlyCapabilityContributesNoSlug', () => {
|
|
// #2801 core inversion: no reviewer body, only the removed hostBehaviors
|
|
// flag. Before Phase 7 this contributed `legacy-cli`; now it contributes
|
|
// nothing, and `collectReviewerWarnings` says so (see section K of
|
|
// tests/reviewer-manifest-body.test.cjs).
|
|
const registry = {
|
|
capabilities: {
|
|
'legacy-cli': { role: 'runtime', runtime: { hostBehaviors: { reviewerCli: true } } },
|
|
},
|
|
};
|
|
assert.deepEqual(deriveReviewerSlugs(registry), []);
|
|
});
|
|
|
|
test('aliasWithAnyValueContributesNoSlug', () => {
|
|
// Value sweep: the old branch was a strict `=== true`, so `false`/`"true"`/`1`
|
|
// never contributed even before. Locking all of them at once means a partial
|
|
// revert that restores only the truthy branch is still caught.
|
|
for (const value of [true, false, 'true', 0, 1, null, {}, []]) {
|
|
const registry = {
|
|
capabilities: {
|
|
'legacy-cli': { role: 'runtime', runtime: { hostBehaviors: { reviewerCli: value } } },
|
|
},
|
|
};
|
|
assert.deepEqual(
|
|
deriveReviewerSlugs(registry), [],
|
|
`reviewerCli = ${JSON.stringify(value)} must contribute no slug`,
|
|
);
|
|
}
|
|
});
|
|
|
|
test('nonReviewerCapabilityContributesNoSlug', () => {
|
|
const registry = { capabilities: { 'plain-feature': { role: 'feature' } } };
|
|
assert.deepEqual(deriveReviewerSlugs(registry), []);
|
|
});
|
|
|
|
test('hardcodedNonRuntimeTailIsDeleted', () => {
|
|
assert.equal(
|
|
'NON_RUNTIME_REVIEWER_SLUGS' in reviewerSelectionModule, false,
|
|
'NON_RUNTIME_REVIEWER_SLUGS must no longer be exported from review-reviewer-selection',
|
|
);
|
|
assert.equal(reviewerSelectionModule.NON_RUNTIME_REVIEWER_SLUGS, undefined);
|
|
});
|
|
|
|
test('declaredBodyIsUnaffectedByAVestigialAliasKey', () => {
|
|
// capId (what the removed alias used to contribute) differs from
|
|
// reviewer.slug, so the result alone shows which surface was read.
|
|
const registry = {
|
|
capabilities: {
|
|
'conflicting-cap-id': {
|
|
role: 'runtime',
|
|
runtime: { hostBehaviors: { reviewerCli: true } },
|
|
reviewer: { slug: 'the-declared-slug' },
|
|
},
|
|
},
|
|
};
|
|
const roster = deriveReviewerSlugs(registry);
|
|
assert.deepEqual(
|
|
roster, ['the-declared-slug'],
|
|
`only the declared body's slug may contribute, got: ${JSON.stringify(roster)}`,
|
|
);
|
|
});
|
|
|
|
test('emptyRegistryYieldsEmptyRoster', () => {
|
|
assert.deepEqual(deriveReviewerSlugs({ capabilities: {} }), []);
|
|
assert.deepEqual(deriveReviewerSlugs({}), [], 'a registry object with no capabilities key at all must not throw');
|
|
});
|
|
|
|
test('rosterIsStableRegardlessOfRegistryOrder', () => {
|
|
// `gamma` was an alias-only capability before #2801. It is now body-declared
|
|
// so this row still proves a THREE-way sort; shrinking it to two entries
|
|
// would quietly weaken the ordering guarantee it exists to protect.
|
|
const capsForward = {
|
|
alpha: { role: 'reviewer', reviewer: { slug: 'zzz-lane' } },
|
|
beta: { role: 'reviewer', reviewer: { slug: 'aaa-lane' } },
|
|
gamma: { role: 'runtime', runtime: { hostBehaviors: {} }, reviewer: { slug: 'gamma' } },
|
|
};
|
|
const reversedCaps = {};
|
|
for (const key of Object.keys(capsForward).reverse()) reversedCaps[key] = capsForward[key];
|
|
|
|
const forwardRoster = deriveReviewerSlugs({ capabilities: capsForward });
|
|
const reversedRoster = deriveReviewerSlugs({ capabilities: reversedCaps });
|
|
assert.deepEqual(forwardRoster, reversedRoster, 'roster must not depend on registry key insertion order');
|
|
assert.deepEqual(
|
|
forwardRoster, ['aaa-lane', 'gamma', 'zzz-lane'],
|
|
'roster must be sorted, independent of declaration order',
|
|
);
|
|
});
|
|
|
|
test('derivedRosterNeverAdmitsAnAliasOnlyCapability', () => {
|
|
// P1 — property over arbitrary registries. Three invariants at once:
|
|
// (1) no capability that declares ONLY the removed alias ever reaches the
|
|
// roster, (2) the roster is sorted and duplicate-free, and (3) it does not
|
|
// depend on key insertion order. A partial revert of the alias branch is
|
|
// caught by (1) at any registry shape, not just the handful enumerated above.
|
|
fc.assert(
|
|
fc.property(
|
|
fc.array(
|
|
fc.record({
|
|
capId: fc.string({ minLength: 1, maxLength: 8 }).filter((s) => s.trim().length > 0),
|
|
declaresBody: fc.boolean(),
|
|
slug: fc.string({ minLength: 1, maxLength: 8 }).filter((s) => s.trim().length > 0),
|
|
aliasValue: fc.constantFrom(true, false, 'true', 1, 0, null, undefined),
|
|
}),
|
|
{ minLength: 0, maxLength: 12 },
|
|
),
|
|
(specs) => {
|
|
const capabilities = {};
|
|
const aliasOnlyIds = new Set();
|
|
const declaredSlugs = new Set();
|
|
for (const s of specs) {
|
|
if (s.capId === '__proto__' || s.capId === 'constructor' || s.capId === 'prototype') continue;
|
|
const cap = { role: 'runtime', runtime: { hostBehaviors: {} } };
|
|
if (s.aliasValue !== undefined) cap.runtime.hostBehaviors.reviewerCli = s.aliasValue;
|
|
if (s.declaresBody) {
|
|
cap.reviewer = { slug: s.slug };
|
|
declaredSlugs.add(s.slug.trim());
|
|
} else if (s.aliasValue !== undefined) {
|
|
aliasOnlyIds.add(s.capId);
|
|
}
|
|
capabilities[s.capId] = cap;
|
|
}
|
|
|
|
const roster = deriveReviewerSlugs({ capabilities });
|
|
|
|
// (1) An alias-only capability's id may appear ONLY if some other
|
|
// capability legitimately declared it as a body slug.
|
|
for (const id of aliasOnlyIds) {
|
|
if (declaredSlugs.has(id)) continue;
|
|
assert.equal(
|
|
roster.includes(id), false,
|
|
`alias-only capability "${id}" must not reach the roster; roster=${JSON.stringify(roster)}`,
|
|
);
|
|
}
|
|
|
|
// (2) sorted + unique
|
|
assert.deepEqual(roster, [...roster].sort(), `roster must be sorted, got: ${JSON.stringify(roster)}`);
|
|
assert.equal(new Set(roster).size, roster.length, `roster must be duplicate-free, got: ${JSON.stringify(roster)}`);
|
|
|
|
// (3) order-independent
|
|
const reversed = {};
|
|
for (const key of Object.keys(capabilities).reverse()) reversed[key] = capabilities[key];
|
|
assert.deepEqual(
|
|
deriveReviewerSlugs({ capabilities: reversed }), roster,
|
|
'roster must not depend on registry key insertion order',
|
|
);
|
|
return true;
|
|
},
|
|
),
|
|
{ numRuns: 200, seed: 2801 },
|
|
);
|
|
});
|
|
});
|
|
|
|
// ─── D. Cross-phase invariants that must not regress ────────────────────────
|
|
|
|
describe('D. Cross-phase invariants that must not regress', () => {
|
|
test('phase1ParityAssertionStillHolds', () => {
|
|
const workflowText = fs
|
|
.readFileSync(path.join(ROOT, 'msd-core', 'workflows', 'review.md'), 'utf-8')
|
|
.replace(/\r\n/g, '\n');
|
|
const registry = [...SHIPPED.capMap.values()]
|
|
.map((c) => c && c.reviewer && c.reviewer.slug)
|
|
.filter((x) => typeof x === 'string' && x)
|
|
.sort();
|
|
const result = checkReviewerLaneParity({
|
|
descriptor: REVIEWER_LANES,
|
|
roster: KNOWN_REVIEWER_SLUGS,
|
|
registry,
|
|
workflowText,
|
|
});
|
|
assert.deepEqual(
|
|
result.violations, [],
|
|
`descriptor <-> roster <-> registry parity must stay green across this migration, got: ${JSON.stringify(result.violations)}`,
|
|
);
|
|
assert.equal(result.ok, true);
|
|
});
|
|
|
|
test('allNineDeclaredLanesSatisfyUniqueness', () => {
|
|
const errs = validateCrossCapability(SHIPPED.capMap, new Set());
|
|
const laneErrs = errs.filter((e) => e.startsWith('reviewer '));
|
|
assert.deepEqual(
|
|
laneErrs, [],
|
|
`expected no reviewer-lane uniqueness violations (slug/flag/section) across the real nine, got: ${JSON.stringify(laneErrs)}`,
|
|
);
|
|
});
|
|
|
|
test('selectionPrecedenceIsUnchanged', () => {
|
|
// ADR-0011: explicit flags > --all > review.default_reviewers > all
|
|
// detected. Exercised with real roster members so a broken roster
|
|
// derivation would also surface here — normalizeReviewerInstances /
|
|
// resolveReviewerSelection gate config_default membership on
|
|
// KNOWN_REVIEWER_SLUGS.includes(...).
|
|
const detected = ['codex', 'claude', 'opencode'];
|
|
|
|
const explicit = resolveReviewerSelection({
|
|
detected, explicitFlags: ['codex'], allFlag: true, configuredDefaultReviewers: ['claude'],
|
|
});
|
|
assert.equal(explicit.source, 'explicit_flags');
|
|
assert.deepEqual(explicit.selected, ['codex']);
|
|
|
|
const allFlagResult = resolveReviewerSelection({
|
|
detected, explicitFlags: [], allFlag: true, configuredDefaultReviewers: ['claude'],
|
|
});
|
|
assert.equal(allFlagResult.source, 'all_flag');
|
|
assert.deepEqual(allFlagResult.selected, [...detected].sort());
|
|
|
|
const configDefault = resolveReviewerSelection({
|
|
detected, explicitFlags: [], allFlag: false, configuredDefaultReviewers: ['claude', 'opencode'],
|
|
});
|
|
assert.equal(configDefault.source, 'config_default');
|
|
assert.deepEqual(configDefault.selected, ['claude', 'opencode']);
|
|
|
|
const noConfig = resolveReviewerSelection({ detected, explicitFlags: [], allFlag: false });
|
|
assert.equal(noConfig.source, 'no_config_all_detected');
|
|
assert.deepEqual(noConfig.selected, [...detected].sort());
|
|
});
|
|
});
|
|
|
|
// ─── E. Lane fidelity — no translation layer ────────────────────────────────
|
|
|
|
describe('E. Lane fidelity — no translation layer', () => {
|
|
test('declaredManifestLanesMatchThePhase1Descriptor', () => {
|
|
// The epic's entire premise: the manifest and the core descriptor describe
|
|
// the SAME lane with no translation layer. Phase 2's review already caught
|
|
// one divergence (the slug grammar) invisible to every other test — assert
|
|
// the WHOLE table, per field, so any future divergence names itself: the
|
|
// lane AND the exact field (and, for nested fields, the sub-field) that
|
|
// diverged.
|
|
const bySlug = new Map();
|
|
for (const [capId, cap] of Object.entries(capabilityRegistry.capabilities)) {
|
|
if (cap && cap.reviewer && typeof cap.reviewer.slug === 'string') {
|
|
bySlug.set(cap.reviewer.slug, { capId, reviewer: cap.reviewer });
|
|
}
|
|
}
|
|
|
|
assert.equal(REVIEWER_LANES.length, 9, 'expected exactly 9 declared descriptor lanes');
|
|
assert.equal(bySlug.size, 9, `expected exactly 9 capabilities declaring a reviewer body, got: ${bySlug.size}`);
|
|
|
|
// Top-level scalar/array fields compared whole; the two fields that are
|
|
// themselves nested objects (probe, invoke) are compared sub-field-by-
|
|
// sub-field over the UNION of keys on both sides, so a field present on
|
|
// only one side is caught exactly as loudly as one with a differing value.
|
|
const TOP_FIELDS = [
|
|
'flags', 'transport', 'probe', 'invoke', 'timeoutFloorMs', 'emptyOutput',
|
|
'reviewsSection', 'evidenceClass', 'requiresBinaries', 'promptBudgetKey', 'handler',
|
|
];
|
|
const NESTED_OBJECT_FIELDS = new Set(['probe', 'invoke']);
|
|
|
|
for (const lane of REVIEWER_LANES) {
|
|
const declared = bySlug.get(lane.slug);
|
|
assert.ok(declared, `no capability declares a reviewer body for descriptor lane "${lane.slug}"`);
|
|
const { capId, reviewer } = declared;
|
|
|
|
for (const field of TOP_FIELDS) {
|
|
if (NESTED_OBJECT_FIELDS.has(field)) {
|
|
const laneSub = lane[field] || {};
|
|
const manifestSub = reviewer[field] || {};
|
|
const subKeys = new Set([...Object.keys(laneSub), ...Object.keys(manifestSub)]);
|
|
for (const subKey of subKeys) {
|
|
assert.deepEqual(
|
|
manifestSub[subKey], laneSub[subKey],
|
|
`lane "${lane.slug}" (capability "${capId}") field "${field}.${subKey}" diverges from Phase 1's descriptor: ` +
|
|
`expected ${JSON.stringify(laneSub[subKey])}, got ${JSON.stringify(manifestSub[subKey])}`,
|
|
);
|
|
}
|
|
} else {
|
|
assert.deepEqual(
|
|
reviewer[field], lane[field],
|
|
`lane "${lane.slug}" (capability "${capId}") field "${field}" diverges from Phase 1's descriptor: ` +
|
|
`expected ${JSON.stringify(lane[field])}, got ${JSON.stringify(reviewer[field])}`,
|
|
);
|
|
}
|
|
}
|
|
|
|
// Belt-and-suspenders whole-object comparison, in case a field exists on
|
|
// one side under a name the named-field loop above did not enumerate.
|
|
assert.deepEqual(
|
|
reviewer, lane,
|
|
`lane "${lane.slug}" (capability "${capId}") has a field-set divergence from Phase 1's descriptor`,
|
|
);
|
|
}
|
|
|
|
// Reverse direction: every capability-declared reviewer body maps back to
|
|
// a descriptor lane — no orphaned manifest lane the descriptor doesn't know.
|
|
for (const [slug, { capId }] of bySlug) {
|
|
assert.ok(
|
|
REVIEWER_LANES.some((l) => l.slug === slug),
|
|
`capability "${capId}" declares reviewer.slug "${slug}" with no matching Phase 1 descriptor lane`,
|
|
);
|
|
}
|
|
});
|
|
});
|
|
|
|
// ─── F. Isolated-security-review regressions (#2798) ─────────────────────────
|
|
//
|
|
// Both rows come from an independent adversarial review that reproduced them by
|
|
// execution. Neither is reachable through the checked-in registry — it is
|
|
// generated, JSON-sourced and code-reviewed — but `deriveReviewerSlugs` is
|
|
// EXPORTED for reuse and carries no other validation, so it must not depend on
|
|
// its caller's hygiene.
|
|
describe('F. Isolated-security-review regressions', () => {
|
|
test('whitespaceOnlySlugIsRejectedNotAdmittedToTheRoster', () => {
|
|
for (const blank of [' ', '\t', '\n', ' \t\n ']) {
|
|
const roster = deriveReviewerSlugs({ capabilities: { x: { reviewer: { slug: blank } } } });
|
|
assert.deepEqual(
|
|
roster, [],
|
|
`a whitespace-only slug can never match a real lane but would occupy a roster entry; got: ${JSON.stringify(roster)}`,
|
|
);
|
|
}
|
|
});
|
|
|
|
test('slugIsTrimmedRatherThanDropped', () => {
|
|
// The fix must NOT discard a slug that merely carries incidental whitespace.
|
|
assert.deepEqual(
|
|
deriveReviewerSlugs({ capabilities: { x: { reviewer: { slug: ' codex ' } } } }),
|
|
['codex'],
|
|
);
|
|
});
|
|
|
|
test('aBlankBodyContributesNothingAndNoAliasRescuesIt', () => {
|
|
// INVERTED by Phase 7 (#2801). While the alias existed, a blank slug fell
|
|
// through to it, so a malformed body could not silently remove a lane that
|
|
// worked before — that was the point of the original row. With the alias
|
|
// gone there is nothing to fall through TO: a blank body is not a
|
|
// declaration, and a declaration is now the only route onto the roster.
|
|
//
|
|
// The row is inverted rather than deleted because it is the combination the
|
|
// single-variable rows miss, and because it is the security-review
|
|
// provenance for the trim: `deriveReviewerSlugs` is exported and carries no
|
|
// other validation, so a whitespace slug must never occupy a roster entry
|
|
// it can never match.
|
|
const roster = deriveReviewerSlugs({
|
|
capabilities: {
|
|
claude: { reviewer: { slug: ' ' }, runtime: { hostBehaviors: { reviewerCli: true } } },
|
|
},
|
|
});
|
|
assert.deepEqual(
|
|
roster, [],
|
|
'a blank body is not a declaration, and the removed alias cannot rescue it',
|
|
);
|
|
});
|
|
|
|
test('moduleLoadSurvivesAHostileRegistryShape', () => {
|
|
// KNOWN_REVIEWER_SLUGS is computed at require() time, so an uncaught throw
|
|
// there breaks import for EVERY consumer rather than degrading selection.
|
|
// The module under test already imported successfully above; assert the
|
|
// derived roster is a usable array rather than a partially-initialised value.
|
|
assert.ok(Array.isArray([...KNOWN_REVIEWER_SLUGS]), 'roster must be iterable after module load');
|
|
assert.equal(KNOWN_REVIEWER_SLUGS.length, 9, 'the real registry still yields the nine lanes');
|
|
// And the derivation itself is total over the shapes JSON can express.
|
|
for (const hostile of [null, undefined, [], 0, 'x', { capabilities: null }, { capabilities: [] }]) {
|
|
assert.doesNotThrow(
|
|
() => deriveReviewerSlugs(hostile === undefined ? {} : (hostile || {})),
|
|
`deriveReviewerSlugs must tolerate ${JSON.stringify(hostile)}`,
|
|
);
|
|
}
|
|
});
|
|
});
|