diff --git a/.changeset/nimble-deer-purr.md b/.changeset/nimble-deer-purr.md new file mode 100644 index 000000000..efa6e26fe --- /dev/null +++ b/.changeset/nimble-deer-purr.md @@ -0,0 +1,5 @@ +--- +type: Fixed +pr: 2 +--- +**A passed UAT is no longer discarded or re-requested after gap closure** — a phase whose human UAT passed could loop forever between `/msd-execute-phase` and `/msd-verify-work`: a gap-closure plan (or a later phase editing a covered file) made the report `stale`, the re-run verifier could not see the UAT file, and execute-phase then rewrote `*-UAT.md` with every row back to `[pending]`. The verifier now reads recorded UAT results as evidence (`verification.uat-evidence`) and re-requests only rows whose tested code changed; the UAT file is merged, never overwritten (`verification.seed-uat`); `verify-work` and `progress` re-run the verifier themselves for a stale report instead of routing to `execute-phase`; and `progress` offers every stale-but-executed phase as one re-verify action. `verification.status` now reports `stale_reason`. A real change to covered code still makes the report stale. diff --git a/.gitignore b/.gitignore index d650dbfeb..b91217e54 100644 --- a/.gitignore +++ b/.gitignore @@ -312,6 +312,7 @@ build/ /msd-core/bin/lib/uat.cjs /msd-core/bin/lib/coverage.cjs /msd-core/bin/lib/uat-predicate.cjs +/msd-core/bin/lib/uat-evidence.cjs /msd-core/bin/lib/workstream.cjs /msd-core/bin/lib/roadmap.cjs /msd-core/bin/lib/audit.cjs diff --git a/agents/msd-verifier.md b/agents/msd-verifier.md index c71af6147..5c6ed8c91 100644 --- a/agents/msd-verifier.md +++ b/agents/msd-verifier.md @@ -537,6 +537,10 @@ Merge those harvested items into the same human verification list as your own an **Why human:** {Why can't verify programmatically} ``` +## Step 8b: Apply Recorded UAT Evidence + +If the phase directory holds a `*-UAT.md`, read and follow `~/.claude/msd-core/references/verifier-uat-evidence.md` before Step 9. A human item whose UAT row already passed on unchanged code is verified by that row and leaves the human verification list; a re-listed row that had passed carries a `retest_reason`. Never re-request a UAT the change did not touch. + ## Step 9: Determine Overall Status Classify status using this decision tree IN ORDER (most restrictive first): @@ -722,6 +726,7 @@ coincidental_reliance_items: # Only if a ✓ VERIFIED truth holds incidentally - truth: "Observable truth that holds incidentally" reason: undeclared-precondition | incidental-ordering | fixture-only harden: "Precondition/ordering to declare or enforce" +human_verified: # Only if Step 8b settled items from *-UAT.md — shape in verifier-uat-evidence.md human_verification: # Only if status: human_needed - test: "What to do" expected: "What should happen" diff --git a/docs/INVENTORY-MANIFEST.json b/docs/INVENTORY-MANIFEST.json index aa1f5744b..3f2323c7a 100644 --- a/docs/INVENTORY-MANIFEST.json +++ b/docs/INVENTORY-MANIFEST.json @@ -323,6 +323,7 @@ "sketch-variant-patterns.md", "specless-probe-fallback.md", "spidr-splitting.md", + "stale-reverification.md", "tdd.md", "thinking-models-debug.md", "thinking-models-execution.md", @@ -340,6 +341,7 @@ "verification-patterns.md", "verifier-evidence-gate.md", "verifier-phase-gates.md", + "verifier-uat-evidence.md", "verifier-wiring-patterns.md", "verify-command-path-resolvability.md", "verify-mvp-mode.md", @@ -565,6 +567,7 @@ "template.cjs", "text-lines.cjs", "token-scanner.cjs", + "uat-evidence.cjs", "uat-predicate.cjs", "uat.cjs", "ui-consideration-probe.cjs", @@ -652,6 +655,7 @@ "execute-phase/steps/executor-isolation-dispatch.md", "execute-phase/steps/executor-progress-policy.md", "execute-phase/steps/gap-closure-artifacts.md", + "execute-phase/steps/gap-closure-uat-record.md", "execute-phase/steps/partial-wave.md", "execute-phase/steps/per-plan-executor-routing.md", "execute-phase/steps/per-plan-worktree-gate.md", diff --git a/docs/INVENTORY.md b/docs/INVENTORY.md index e7b6a1a6b..4253c3c38 100644 --- a/docs/INVENTORY.md +++ b/docs/INVENTORY.md @@ -347,6 +347,8 @@ Full roster at `msd-core/references/*.md`. References are shared knowledge docum | `verification-overrides.md` | Per-artifact verification override rules. | | `verifier-phase-gates.md` | Verifier-time gates eagerly imported by `msd-verifier` (migrated from the retired `verify-phase` workflow, #1892): decision-coverage validation (#2492), test-quality audit, and infrastructure-phase human-verification scoping (#2504). | | `verifier-evidence-gate.md` | Re-verification convergence gate loaded by `msd-verifier` (#3304): a Step 7 anti-pattern blocker that is neither a carried-forward gap nor a regression needs deterministic evidence to stay blocking, else it downgrades to advisory. | +| `verifier-uat-evidence.md` | UAT-evidence gate loaded by `msd-verifier` at Step 8b when the phase has a `*-UAT.md`: a human item whose UAT row passed on unchanged code is verified by that row; a row whose tested code changed is re-listed with a `retest_reason`. | +| `stale-reverification.md` | The single procedure for refreshing a `stale` VERIFICATION.md on a fully summarized phase (verifier re-run, `verification.seed-uat` merge, fresh-status routing); shared by `execute-phase`, `verify-work` and `progress`. | | `planning-config.md` | Full config schema and behavior. | | `phase-id-convention.md` | Canonical bracket phase-ID grammar card and display/disk translation reference (ADR-612). | | `security-asvs-levels.md` | OWASP ASVS level definitions for MSD threat modeling — per-level planner disposition rigor and auditor verification depth (L1 opportunistic, L2 standard, L3 comprehensive). | @@ -709,6 +711,7 @@ Full listing: `msd-core/bin/lib/*.cjs`. | `normalize-test-command.cjs` | Normalizes a resolved test command to a one-shot form so a watch-mode runner (vitest/jest) cannot hang a verification gate (#1857); shared by all three live test-command gates (regression, post-merge, audit-fix) | | `uat.cjs` | UAT file parsing, verification debt tracking, audit-uat support | | `uat-predicate.cjs` | UAT-passed predicate — markdown-aware evaluation of HUMAN-UAT results; returns pass only when all required checks pass; ignores false-positive contexts (frontmatter, fenced code, blockquotes, HTML comments) | +| `uat-evidence.cjs` | UAT-evidence seam — `verification.uat-evidence` (recorded UAT rows plus covered files changed since), `verification.seed-uat` (merge the report's human items into `*-UAT.md` without overwriting results) and `verification.canonicalize-uat` (the single `human_needed` → `passed` flip, gated on the UAT-row predicate) | | `ui-consideration-probe.cjs` | Spec-completeness UI-consideration probe (compiled from `src/ui-consideration-probe.cts`, gitignored) — the third adapter of the `probe-core` resolution model (ADR-550 Decision 7): element-kind classification, applicable-category relevance filter, consideration proposal, `proposeElements`/`autoResolve` (propose-then-confirm + the `--auto` never-dismiss floor), and the `{explicit, backstop}` validators; delegates merge/rollup/CLI to `probe-core`; exports `classifyElement`, `applicableCategories`, `proposeConsiderations`, `proposeElements`, `autoResolve`, `analyzeCoverage`, `UI_TAXONOMY` (#1867) | | `ui-safety-gate.cjs` | Shell-free word-boundary UI token detector (#3706, #3718); reads phase-section text from stdin, exits 0 (UI found), 1 (no UI — real input, examined), NO_INPUT (empty/whitespace-only stdin) or UNAVAILABLE (stdin read failed) — the latter two are registry codes (ADR-3889 Phase 3, #3907); also deployed to `msd-core/bin/lib/` so the MSD installer ships it to `$RUNTIME_DIR` (#448) | | `ui-frontend-evidence.cjs` | Static frontend-evidence detector (compiled from `src/ui-frontend-evidence.cts`, gitignored) — plan-time structural corroboration for `computeUiPlanGate` (#3312): a `package.json` UI-framework dependency or a component-framework file (`*.tsx/*.jsx/*.vue/*.svelte`) in the tree, so a UI-token match on a hyphenated proper noun (e.g. repo `dashboard-financeiro`) cannot block planning in a repo with no frontend; mirrors the post-wave `computeUiSafetyGate` git-diff corroboration | diff --git a/eslint.config.mjs b/eslint.config.mjs index 95318cd50..2a517d7c0 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -333,6 +333,7 @@ export default tseslint.config( 'msd-core/bin/lib/uat.cjs', 'msd-core/bin/lib/coverage.cjs', 'msd-core/bin/lib/uat-predicate.cjs', + 'msd-core/bin/lib/uat-evidence.cjs', 'msd-core/bin/lib/workstream.cjs', 'msd-core/bin/lib/roadmap.cjs', 'msd-core/bin/lib/audit.cjs', diff --git a/msd-core/bin/msd-tools.cjs b/msd-core/bin/msd-tools.cjs index 7762050a3..b93c14804 100755 --- a/msd-core/bin/msd-tools.cjs +++ b/msd-core/bin/msd-tools.cjs @@ -326,6 +326,7 @@ const evalMod = require('./lib/eval.cjs'); const { routeVerificationCommand } = require('./lib/verification-command-router.cjs'); const { routePlanningCommand } = require('./lib/planning-command-router.cjs'); const verification = require('./lib/verification.cjs'); +const uatEvidence = require('./lib/uat-evidence.cjs'); const { routeInitCommand } = require('./lib/init-command-router.cjs'); // Stale-bake guard (#1688): warns once when model config changed since agents // were last baked on static-frontmatter runtimes (codex/opencode). Lazy-required @@ -1094,6 +1095,7 @@ function dispatchOverlayCapabilityCommand({ command, args, cwd, raw, error, load function routeVerification({ args, cwd, raw, error }) { routeVerificationCommand({ verification, + uatEvidence, args, cwd, raw, diff --git a/msd-core/references/stale-reverification.md b/msd-core/references/stale-reverification.md new file mode 100644 index 000000000..0297d41a3 --- /dev/null +++ b/msd-core/references/stale-reverification.md @@ -0,0 +1,88 @@ +# Stale Re-verification + +> The ONE procedure for refreshing a `stale` VERIFICATION.md when every plan in the phase +> already has a SUMMARY. Loaded by `execute-phase` (`steps/stale-reverification.md`), +> `verify-work` (its completion gate) and `progress` (Route V.stale), so the three cannot +> drift apart. Nothing here executes a plan, and nothing here discards a recorded UAT +> result. `msd_run` is the launcher shim of the loading workflow. + +`verification.status` reports `stale` with a `stale_reason`: + +| `stale_reason` | What happened | +|---|---| +| `uncovered_artifacts` | The phase gained a PLAN/SUMMARY the report never declared — every gap-closure cycle does this. | +| `covered_changed` | A covered file changed after the verifier ran — possibly edited by a LATER phase. | +| `fingerprint_malformed` / `summary_newer` | The fingerprint is unusable, or a legacy report predates a SUMMARY. | + +Every reason has the same remedy: the verifier looks again. What must NOT happen is the +human being asked to repeat a UAT the change did not touch. + +## 1. Re-run the verifier + +Dispatch `msd-verifier` with the same brief execute-phase's `verify_phase_goal` step uses. A +workflow that did not load execute-phase's init resolves the inputs itself: + +```bash +PHASE_INIT=$(msd_run query init.execute-phase "{phase}") # phase_dir, phase_req_ids, requirements_path, verifier_model +PHASE_GOAL=$(msd_run query roadmap.get-phase "{phase}" --pick goal) +VERIFIER_SKILLS=$(msd_run query agent-skills msd-verifier) +``` + +``` +Agent( + description="Re-verify phase {phase_number} (stale report)", + prompt="Verify phase {phase_number} goal achievement. +Phase directory: {phase_dir} +Phase goal: {PHASE_GOAL} +Phase requirement IDs: {phase_req_ids} +Check must_haves against actual codebase. The previous VERIFICATION.md is stale — regenerate it. +A recorded human UAT is evidence: apply Step 8b before deciding the status. + + +- {phase_dir}/*-PLAN.md +- {phase_dir}/*-SUMMARY.md +- {phase_dir}/*-UAT.md (if present) +- {requirements_path} + + +${VERIFIER_SKILLS}", + subagent_type="msd-verifier", + model="{verifier_model}" +) +``` + +The verifier reads the phase's `*-UAT.md` itself (`verification.uat-evidence`): a human item +whose UAT row passed on unchanged code comes back verified, and only a row whose tested code +changed comes back for re-test. + +Re-verifying several phases (progress's batch): dispatch one verifier per phase, in +parallel, and run steps 2–3 for each as it returns. + +## 2. Merge human items into the UAT file — never overwrite it + +```bash +SEED=$(msd_run query verification.seed-uat "$PHASE_DIR") +``` + +Run this only when the fresh status is `human_needed`. It keeps every existing row and +result, appends unseen items as `[pending]`, and resets a passing row only when the +verifier recorded a `retest_reason` (written into the row). When `.changed` is `true`, +commit `.uat_file`: + +```bash +msd_run query commit "test({phase_num}): merge human verification items into UAT" --files "$(printf '%s' "$SEED" | jq -r '.uat_file')" +``` + +## 3. Route on the fresh status + +```bash +STATUS=$(msd_run query verification.status "$PHASE_DIR" --pick status) +``` + +| Fresh status | Then | +|---|---| +| `passed` | Done — the phase is verified again. No human was asked anything. | +| `human_needed`, `SEED.pending` is `0` | Every UAT row may already pass — record it: `msd_run query verification.canonicalize-uat "$PHASE_DIR"`. `canonicalized: true` → `passed`, no test re-run. Otherwise present its `blockers` and route to `/msd:verify-work {phase}`. | +| `human_needed`, `SEED.pending` > `0` | Only the pending rows need a human — `/msd:verify-work {phase}` resumes at the first one. | +| `gaps_found` | A real regression: `/msd:plan-phase {phase} --gaps`. | +| `stale` again | Stop and present the report — never loop on it (#4623 covers what the digest hashes). | diff --git a/msd-core/references/verifier-uat-evidence.md b/msd-core/references/verifier-uat-evidence.md new file mode 100644 index 000000000..2336904e6 --- /dev/null +++ b/msd-core/references/verifier-uat-evidence.md @@ -0,0 +1,58 @@ +# Verifier UAT Evidence + +> Loaded by `agents/msd-verifier.md` at Step 8b, only when the phase directory holds a +> `*-UAT.md`. A human who already ran a check is evidence. Re-asking for it on every +> verifier re-run — after a gap-closure plan, or after a covered file drifted — is the +> loop this gate exists to end. `msd_run` is the launcher shim from the agent's Step 1. + +## 1. Read the evidence + +Pass the ROOT-relative implementation files you are about to declare in `covered_files`: + +```bash +msd_run query verification.uat-evidence "$PHASE_DIR" {impl file}... +``` + +The result carries `rows` (each UAT row: `test`, `name`, `result`, `passing`, `expected`), +`recorded_at` (when the UAT file last changed), and `changed_since_uat` (covered +implementation files modified, or gone, AFTER the UAT was recorded). An empty `uat_file` +means there is no UAT evidence — skip this gate. + +## 2. Settle each Step 8 human item against its row + +Match by meaning, not by wording: a row covers an item when a human performing the row's +check would have observed what the item asks for. + +| Row for the item | `changed_since_uat` | Outcome | +|---|---|---| +| `passing: true` | empty, or no changed file bears on what the row exercised | **Verified by human UAT.** Remove the item from the human verification list. Record it under `human_verified` (below) and mark the truth `✓ VERIFIED (human UAT, row N)`. | +| `passing: true` | a changed file bears on what the row exercised | **Re-test.** Keep the item in `human_verification` with `retest_reason` naming the file and `recorded_at`. Use the row's exact `name` as the item's `test`. | +| not passing (`pending`, `issue`, `blocked`, `skipped`, `missing`) | any | Ordinary human item. Use the row's exact `name` as `test`; no `retest_reason`. | +| no row | any | Ordinary new human item. | + +Rules: + +- **Unsure whether a changed file affects a row → re-test.** A changed file you cannot + rule out is a changed file that bears on it. Real change still costs a re-test. +- **A pass is only ever withdrawn with a reason.** `retest_reason` is what lets the UAT + writer reset that row; an item re-listed without one leaves the recorded pass in place. +- A `⚠️ PRESENT_BEHAVIOR_UNVERIFIED` truth whose row passed on unchanged code has directly + observed behavior: it is VERIFIED by that row and leaves `behavior_unverified_items`. +- Step 9 then runs on what is LEFT. If every human item was verified by UAT, the human + verification section is empty and the ordinary tree yields `passed` (or `gaps_found`). +- Never write or edit `*-UAT.md`. The orchestrator merges your `human_verification` items + into it with `verification.seed-uat`, which keeps every recorded result. + +## 3. Frontmatter + +```yaml +human_verified: # Only if UAT evidence settled at least one item + - test: "Exact UAT row name" + uat_row: 3 + recorded_at: "ISO time from uat-evidence" +human_verification: # Only if status: human_needed + - test: "Exact UAT row name" + expected: "What should happen" + why_human: "Why can't verify programmatically" + retest_reason: "src/auth/session.ts changed after the UAT pass (2026-10-05T…)" # ONLY for a row that had passed +``` diff --git a/msd-core/workflows/execute-phase.md b/msd-core/workflows/execute-phase.md index dd1a90f83..e391572ab 100644 --- a/msd-core/workflows/execute-phase.md +++ b/msd-core/workflows/execute-phase.md @@ -1189,6 +1189,8 @@ If `section_manifest` is `null` or `"regression-gate"` is in its `included` list Verify phase achieved its GOAL, not just completed tasks. +**Gap-closure runs:** if this run executed a `gap_closure: true` plan, first follow `execute-phase/steps/gap-closure-uat-record.md` — the human's checkpoint results go into `*-UAT.md` BEFORE the verifier is dispatched, so the same run can reach `passed`. + ```bash VERIFIER_SKILLS=$(msd_run query agent-skills msd-verifier) ``` @@ -1208,6 +1210,7 @@ Create VERIFICATION.md. Read these files before verification: - {phase_dir}/*-PLAN.md (All plans — understand intent, check must_haves) - {phase_dir}/*-SUMMARY.md (All summaries — cross-reference claimed vs actual) +- {phase_dir}/*-UAT.md (If present: recorded human UAT — evidence, verifier Step 8b) - {requirements_path} (Requirement traceability) ${CONTEXT_WINDOW >= 500000 ? `- {phase_dir}/*-CONTEXT.md (User decisions — verify they were honored) - {phase_dir}/*-RESEARCH.md (Known pitfalls — check for traps) @@ -1235,60 +1238,30 @@ Route on `$STATUS`: if `passed`, proceed to update_roadmap. Otherwise keep the p **If human_needed:** -**Step A: Persist human verification items as UAT file.** +**Step A: Merge human verification items into the UAT file — never overwrite it.** -Create `{phase_dir}/{phase_num}-UAT.md` using UAT template format: - -```markdown ---- -status: testing -phase: {phase_num}-{phase_name} -source: [{phase_num}-VERIFICATION.md] -started: [now ISO] -updated: [now ISO] ---- - -## Current Test - -number: 1 -name: {first human_verification item description} -expected: | - {expected behavior from VERIFICATION.md} -awaiting: user response - -## Tests - -{For each human_verification item from VERIFICATION.md:} - -### {N}. {item description} -expected: {expected behavior from VERIFICATION.md} -result: [pending] - -## Summary - -total: {count} -passed: 0 -issues: 0 -pending: {count} -skipped: 0 -blocked: 0 - -## Gaps -``` - -Commit the file: ```bash -msd_run query commit "test({phase_num}): persist human verification items as UAT" --files "{phase_dir}/{phase_num}-UAT.md" +SEED=$(msd_run query verification.seed-uat "$PHASE_DIR") +SEED_UAT=$(printf '%s' "$SEED" | jq -r '.uat_file // empty') +SEED_PENDING=$(printf '%s' "$SEED" | jq -r '.pending // 0') ``` +`verification.seed-uat` creates `{phase_num}-UAT.md` when none exists, and otherwise MERGES: every existing row and its recorded result is kept, only unseen items are appended as `[pending]`, and a passing row is reset only when the verifier recorded a `retest_reason` for it. Do NOT hand-write or regenerate the UAT file. When `.changed` is `true`, commit it: + +```bash +msd_run query commit "test({phase_num}): persist human verification items as UAT" --files "$SEED_UAT" +``` + +**If `SEED_PENDING` is `0`** — every UAT row already has a result, so no human test is owed. Do NOT present the items as tests to run. Say so, and hand off: `/msd:verify-work {X} ${MSD_WS}` records the pass against the report at once (`verification.canonicalize-uat`) and completes the phase without re-asking a single row. Skip Step B. + **Step B: Present to user**: ``` ## ◷ Phase {X}: {Name} — Human Verification Needed -All automated checks passed. {N} item(s) require human testing before this phase can be marked complete: +All automated checks passed. {SEED_PENDING} item(s) require human testing before this phase can be marked complete: -{From VERIFICATION.md human_verification section} +{The pending rows of {phase_num}-UAT.md — rows that already passed are NOT re-tested} Tests saved to `{phase_num}-UAT.md`. diff --git a/msd-core/workflows/execute-phase/steps/gap-closure-uat-record.md b/msd-core/workflows/execute-phase/steps/gap-closure-uat-record.md new file mode 100644 index 000000000..ebd37d583 --- /dev/null +++ b/msd-core/workflows/execute-phase/steps/gap-closure-uat-record.md @@ -0,0 +1,36 @@ +Apply response_language to all user-facing prose — narration between tool calls, status updates, progress notes, and findings included; preserve code, paths, and identifiers. + + +A gap-closure plan often ends in a human checkpoint that re-runs the UAT checks which failed. +The human answers inside this run. If those answers are not written down before the verifier +is dispatched, the verifier sees no evidence, reports `human_needed` again, and the user is +sent to repeat a UAT they finished minutes ago. + +**Skip if** no `gap_closure: true` plan executed in this run, or none of them had a +`checkpoint:human-verify` / `checkpoint:human-action` task whose response reported on UAT +checks. + +**1. Record each reported outcome in the phase's `*-UAT.md`.** For every UAT row the human +reported on at the checkpoint, edit ONLY that row: + +- passed → `result: pass`, plus `reported: "{verbatim response}"` and + `retested_by: {gap plan id}` +- failed → `result: issue`, `reported: "{verbatim response}"`, `severity: {inferred}` + +Leave every other row untouched. Do not invent results for rows the human did not mention. +Recompute the `## Summary` counters; set frontmatter `status: complete` when no row is +`[pending]`, and refresh `updated:`. + +**2. Resolve the gaps the plan closed.** For each `## Gaps` entry named in an executed plan's +`gap_ids` whose row now passes: `status: resolved`, with `resolved_by` / `resolved_at`. + +**3. Commit before verifying:** + +```bash +_MSD_SHIM_NAME="msd-tools.cjs"; _MSD_RUNTIME_ROOT="${RUNTIME_DIR:-$(git rev-parse --show-toplevel 2>/dev/null || pwd)}"; MSD_TOOLS="${_MSD_RUNTIME_ROOT}/msd-core/bin/${_MSD_SHIM_NAME}"; _msd_at() { for _p; do if [ -f "$_p" ]; then MSD_TOOLS="$_p"; return 0; fi; done; return 1; }; _msd_id_ok() { case "$("$1" runtime-identity --raw 2>/dev/null || true)" in '{"packageName":"@golem15/msd-core"'*'}') return 0;; *) return 1;; esac; }; _msd_homes() { _msd_at "${CLAUDE_CONFIG_DIR:-$HOME/.claude}/msd-core/bin/${_MSD_SHIM_NAME}" "${CURSOR_CONFIG_DIR:-$HOME/.cursor}/msd-core/bin/${_MSD_SHIM_NAME}" "${CODEX_HOME:-$HOME/.codex}/msd-core/bin/${_MSD_SHIM_NAME}" "${GEMINI_CONFIG_DIR:-$HOME/.gemini}/msd-core/bin/${_MSD_SHIM_NAME}" "${GROK_AGENTS_HOME:-$HOME/.agents}/msd-core/bin/${_MSD_SHIM_NAME}" "${ANTIGRAVITY_CONFIG_DIR:-$HOME/.gemini/antigravity}/msd-core/bin/${_MSD_SHIM_NAME}" "${OPENCODE_CONFIG_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/opencode}/msd-core/bin/${_MSD_SHIM_NAME}"; }; if _msd_at "${_MSD_RUNTIME_ROOT}/msd-core/bin/${_MSD_SHIM_NAME}" "${_MSD_RUNTIME_ROOT}/.claude/msd-core/bin/${_MSD_SHIM_NAME}" "${_MSD_RUNTIME_ROOT}/.codex/msd-core/bin/${_MSD_SHIM_NAME}"; then msd_run() { node "$MSD_TOOLS" "$@"; }; elif _msd_homes; then msd_run() { node "$MSD_TOOLS" "$@"; }; elif unset -f msd_run; _G="$(command -v msd_run)"; [ -n "$_G" ] && _msd_id_ok "$_G"; then MSD_TOOLS="$_G"; msd_run() { "$MSD_TOOLS" "$@"; }; else echo "ERROR: msd-tools.cjs not found at $MSD_TOOLS and no identity-proving msd_run is on PATH. Run: npx -y @golem15/msd-core@latest --claude --local" >&2; exit 1; fi; MSD_IDENTITY_STATUS=unverified; _msd_id_ok msd_run && MSD_IDENTITY_STATUS=ok; export MSD_IDENTITY_STATUS; [ "$MSD_IDENTITY_STATUS" = ok ] || echo "WARNING: \"$MSD_TOOLS\" did not prove it is @golem15/msd-core - it is either a different package or an @golem15/msd-core older than the runtime-identity verb. See docs/how-to/diagnose-a-foreign-msd-tools.md" >&2; if [ -n "${CLAUDE_ENV_FILE:-}" ] && [ -n "${MSD_TOOLS:-}" ]; then printf "export PATH='%s':\"\$PATH\"\n" "${MSD_TOOLS%/*}" >> "$CLAUDE_ENV_FILE" 2>/dev/null || true; fi +msd_run query commit "test({phase_num}): record gap-closure UAT results" --files "{phase_dir}/{phase_num}-UAT.md" +``` + +The verifier dispatched next reads these rows as evidence (`verification.uat-evidence`), so a +gap closure whose UAT passed ends this run at `passed` — no second command. + diff --git a/msd-core/workflows/execute-phase/steps/stale-reverification.md b/msd-core/workflows/execute-phase/steps/stale-reverification.md index c4254427a..bb23dc43c 100644 --- a/msd-core/workflows/execute-phase/steps/stale-reverification.md +++ b/msd-core/workflows/execute-phase/steps/stale-reverification.md @@ -1,9 +1,10 @@ Apply response_language to all user-facing prose — narration between tool calls, status updates, progress notes, and findings included; preserve code, paths, and identifiers. -Covered source files changed after the verifier last ran — the recorded digest no longer -matches, so the report cannot be trusted until the verifier re-runs (#4682). The plans are all -summarized: there is no wave work to do. +The report no longer matches the phase — a covered file changed, or the phase gained a +PLAN/SUMMARY the report never declared (`verification.status` names which in `stale_reason`) +— so it cannot be trusted until the verifier re-runs (#4682). The plans are all summarized: +there is no wave work to do, and no plan will execute. Report: ``` @@ -19,6 +20,10 @@ digest. This holds whether or not the phase was already marked complete — a st marked phase is refreshed the same way. `verification.status`'s `next_command` routes here for exactly this state. +Recorded UAT results survive this re-run: the verifier reads `*-UAT.md` as evidence, and the +`human_needed` branch merges into it (`verification.seed-uat`) instead of rewriting it — the +shared procedure is `msd-core/references/stale-reverification.md`. + Never silently proceed past a stale gate: if the re-run verifier still produces a stale report, stop and present it (#4623 covers what the digest hashes). diff --git a/msd-core/workflows/progress.md b/msd-core/workflows/progress.md index c228c647d..39fafa3b4 100644 --- a/msd-core/workflows/progress.md +++ b/msd-core/workflows/progress.md @@ -321,7 +321,7 @@ VERIFICATION_STATUS=$(printf '%s' "$VERIFICATION" | jq -r '.status' 2>/dev/null VERIFICATION_NEXT_ACTION=$(printf '%s' "$VERIFICATION" | jq -r '.next_action' 2>/dev/null || echo "") ``` -Track: `verification_status` — the `.status` field (`passed | stale | gaps_found | human_needed | missing | unknown`). The query/projection handles a missing VERIFICATION.md (`missing`), unexpected values, and stale verification (`stale`, when summaries are newer than verification). Only `passed` routes as phase complete (Step 3); every other status routes back to close verification debt (Step 2). +Track: `verification_status` — the `.status` field (`passed | stale | gaps_found | human_needed | missing | unknown`). The query/projection handles a missing VERIFICATION.md (`missing`), unexpected values, and stale verification (`stale` — the report no longer matches the phase; `stale_reason` says why). Only `passed` routes as phase complete (Step 3); every other status routes back to close verification debt (Step 2). **Step 2: Route based on counts** @@ -532,12 +532,22 @@ VERIFICATION.md has an unexpected status. The phase is implementation-complete, **Route V.stale: verification is stale** -VERIFICATION.md has `status: passed`, but one or more SUMMARY.md files are newer than the verification report. The phase is implementation-complete, not phase-complete. +Every plan is summarized; the only open item is verification bookkeeping. The report no longer matches the phase — a gap-closure plan it never saw, or a covered file edited since (often by a LATER phase). No plan needs to execute, so do NOT route to `/msd:execute-phase`: re-run the verifier here. + +```bash +REVERIFY=$(printf '%s' "$INIT" | jq -c '.reverify_phases // []') +``` + +`reverify_phases` lists EVERY phase in this state, not just the current one. Handle them as ONE action: ``` -`/msd:verify-work {phase} ${MSD_WS}` — re-run verification against the latest summaries +## Re-verifying {N} phase(s): {numbers} + +Their reports are stale ({stale_reason per phase}). Re-running the verifier — no plans will execute, and recorded UAT results are kept. ``` +Read and follow `msd-core/references/stale-reverification.md` for each listed phase (verifiers in parallel). Then re-run `init.progress` and route again from Step 2 — a phase the verifier returned as `passed` is complete; one with rows genuinely needing a re-test goes to **Route V.human**, naming only those rows. + --- **Route V.gaps: verification found gaps (gaps_found)** diff --git a/msd-core/workflows/verify-work.md b/msd-core/workflows/verify-work.md index 95f489464..13b5edd52 100644 --- a/msd-core/workflows/verify-work.md +++ b/msd-core/workflows/verify-work.md @@ -652,44 +652,45 @@ If an active secure-phase step hook exists AND `SECURITY_FILE` exists: check fro If no active secure-phase step hook exists OR (`SECURITY_FILE` exists AND `threats_open` is `0`): -If execution verification is waiting only on human UAT and this session recorded zero issues, canonicalize the report before the shared completion predicate. (#4663) Zero issues is NOT pass evidence on its own — blocked rows are not issues by this workflow's own rule, so a session that observed nothing (0 passed / 0 issues / N blocked) must NOT flip the report. The flip runs the SAME UAT-row predicate the phase-close uses, in its `--uat-only` form: it skips the verification-status blockers (the report still reads `human_needed` at this point — the full predicate could never pass here), and `passed` means at least one UAT check passed with no row pending/blocked/failed or skipped without a reason. The flagged transition-gate call below stays the final say on canonical verification: +Read the canonical verification status first: ```bash PHASE_DIR=$(printf '%s' "$INIT" | jq -r '.phase_dir // empty') -VERIFICATION_FILE=$(msd_run query verification.resolve-file "$PHASE_DIR" --raw 2>/dev/null) VERIFICATION_STATUS=$(msd_run query verification.status "$PHASE_DIR" 2>/dev/null) -VERIFICATION_STATUS_VALUE=$(printf '%s' "$VERIFICATION_STATUS" | jq -r '.status // empty' 2>/dev/null || echo "") -PHASE_VERIFICATION_STATUS="$VERIFICATION_STATUS_VALUE" -if [ "$VERIFICATION_STATUS_VALUE" = "human_needed" ]; then - UAT_PRECHECK=$(msd_run phase uat-passed "{phase}" --uat-only 2>/dev/null) - UAT_PRECHECK_PASSED=$(printf '%s' "$UAT_PRECHECK" | jq -r '.passed // false' 2>/dev/null || echo "false") - if [ "$UAT_PRECHECK_PASSED" = "true" ]; then - msd_run query frontmatter.set "$VERIFICATION_FILE" --field status --value passed - else - UAT_BLOCKERS=$(printf '%s' "$UAT_PRECHECK" | jq -r '.blockers | length' 2>/dev/null) - [ -n "$UAT_BLOCKERS" ] || UAT_BLOCKERS="?" - echo "NOT canonicalizing: ${UAT_BLOCKERS} UAT row(s) blocked or not passing; verification stays human_needed. Resolve or pass them, then re-run /msd:verify-work {phase}." >&2 - fi -fi +PHASE_VERIFICATION_STATUS=$(printf '%s' "$VERIFICATION_STATUS" | jq -r '.status // empty' 2>/dev/null || echo "") ``` -If `PHASE_VERIFICATION_STATUS` is `stale`, the covered source files changed after the verifier -last ran — re-run the VERIFIER, not this workflow (`/msd:verify-work` never rewrites -VERIFICATION.md; its only write is the human_needed canonicalization, #4663). Spawn the -verifier for this phase exactly as execute-phase's `verify_phase_goal` step does (subagent -`msd-verifier`; phase directory, goal, requirement IDs, and all SUMMARYs in -``), then re-read `verification.status` and continue at the fresh/passed -case below. (#4682) +If `PHASE_VERIFICATION_STATUS` is `stale`, the report no longer matches the phase — a gap-closure +plan added a PLAN/SUMMARY it never declared, or a covered file changed (`stale_reason`). Re-run the +VERIFIER here, in this session — do NOT send the user to `/msd:execute-phase` for a phase whose +plans are all summarized. Read and follow `msd-core/references/stale-reverification.md`: dispatch +`msd-verifier` exactly as execute-phase's `verify_phase_goal` step does, merge any human items +into the UAT file with `verification.seed-uat` (recorded results are kept), then re-read +`verification.status` into `PHASE_VERIFICATION_STATUS`. (#4682) + +- Fresh status `human_needed` with new `[pending]` rows → go to `resume_from_file`; only those + rows are tested. +- Fresh status still `stale`, or `gaps_found` → stop and present it: ``` -Verification is stale: covered source files changed after the verifier last ran. +Verification could not be refreshed: {fresh status} Blocking completion: -verification is stale +{verification.status next_action} +``` -- Re-run the verifier for phase {phase} (dispatch `msd-verifier` as in execute-phase's - verify_phase_goal step) to regenerate VERIFICATION.md with a fresh digest, then re-run - `/msd:verify-work {phase}` +- Otherwise continue below. + +If execution verification is waiting only on human UAT, canonicalize the report before the shared completion predicate. (#4663) Zero issues is NOT pass evidence on its own — blocked rows are not issues by this workflow's own rule, so a session that observed nothing (0 passed / 0 issues / N blocked) must NOT flip the report. `verification.canonicalize-uat` is the single seam for the flip: it acts only on a `human_needed` report (never a `stale` one) and runs the SAME UAT-row predicate the phase-close uses, in its uat-only form — at least one UAT check passed with no row pending/blocked/failed or skipped without a reason. The flagged transition-gate call below stays the final say on canonical verification: + +```bash +CANON=$(msd_run query verification.canonicalize-uat "$PHASE_DIR" 2>/dev/null) +CANON_REASON=$(printf '%s' "$CANON" | jq -r '.reason // empty' 2>/dev/null || echo "") +if [ "$CANON_REASON" = "uat_not_passed" ]; then + UAT_BLOCKERS=$(printf '%s' "$CANON" | jq -r '.blockers | length' 2>/dev/null) + [ -n "$UAT_BLOCKERS" ] || UAT_BLOCKERS="?" + echo "NOT canonicalizing: ${UAT_BLOCKERS} UAT row(s) blocked or not passing; verification stays human_needed. Resolve or pass them, then re-run /msd:verify-work {phase}." >&2 +fi ``` Otherwise, check the shared UAT-plus-verification completion predicate before transition: @@ -708,8 +709,8 @@ All UAT tests passed, but phase advancement is blocked until canonical verificat Blocking completion: {PHASE_COMPLETE_BLOCKERS} -- `/msd:execute-phase {phase}` — regenerate execution verification - `/msd:verify-work {phase}` — resume UAT if blockers remain +- `/msd:plan-phase {phase} --gaps` — if verification reports gaps ``` **Auto-transition: mark phase complete in ROADMAP.md and STATE.md** diff --git a/src/init.cts b/src/init.cts index 2a4ccc050..fe8042263 100644 --- a/src/init.cts +++ b/src/init.cts @@ -272,6 +272,12 @@ interface PhaseCompletionProjection { * complete) or ran to completion. */ verification_stale_check_indeterminate: boolean; + /** + * WHY the report reads `stale` (`readVerificationStatus`'s `stale_reason`), + * '' for every other status. Lets progress tell a gap-closure plan the + * report never saw from a covered file drifting under a finished phase. + */ + verification_stale_reason: string; } @@ -330,6 +336,7 @@ function buildPhaseCompletionProjection( // internal staleness check could not run to completion. verification_stale_check_indeterminate: 'staleCheckIndeterminate' in verificationStatus && verificationStatus.staleCheckIndeterminate === true, + verification_stale_reason: verificationStatus.stale_reason ?? '', }; } @@ -3872,6 +3879,17 @@ function cmdInitProgress(cwd: string, raw: boolean, options: Record p['status'] === 'complete').length, + // Every phase whose plans are all summarized and whose ONLY open item is a + // stale report — one verifier re-run each, no plan executes. Progress + // offers these as a single "re-verify" action instead of walking the user + // back through them one execute-phase route per session. + reverify_phases: phases + .filter((p) => p['implementation_complete'] === true && p['verification_status'] === 'stale') + .map((p) => ({ + number: p['number'], + directory: p['directory'], + stale_reason: p['verification_stale_reason'], + })), in_progress_count: phases.filter((p) => ['executed', 'in_progress'].includes(p['status'] as string), ).length, diff --git a/src/uat-evidence.cts b/src/uat-evidence.cts new file mode 100644 index 000000000..b2474251d --- /dev/null +++ b/src/uat-evidence.cts @@ -0,0 +1,735 @@ +/** + * UAT Evidence — the seam that lets a recorded human UAT survive a verifier + * re-run. + * + * A phase's `*-VERIFICATION.md` goes `stale` whenever the verifier has to look + * again: a gap-closure plan added a PLAN/SUMMARY it never declared, or a + * covered source file changed. Re-running the verifier is the right remedy — + * but before this module nothing carried the human's UAT results across that + * re-run. The verifier could not see `*-UAT.md`, so it re-emitted the same + * `human_needed` items, and execute-phase's `human_needed` branch then wrote a + * fresh `*-UAT.md` over the completed one, every row back to `[pending]`. The + * user was asked for the same UAT again, indefinitely. + * + * Two verbs close that loop, both under the `verification` family: + * + * - `verification.uat-evidence [files…]` (read-only) — the + * recorded UAT rows, plus which covered implementation files changed + * AFTER the UAT was recorded. The verifier is an LLM, not a clock: this + * does the deterministic part so the agent only decides which rows a + * changed file actually affects. + * - `verification.seed-uat ` — writes the report's + * `human_verification` items into `*-UAT.md` by MERGING: existing rows and + * their results are kept byte-for-byte, only unseen items are appended as + * `[pending]`, and a passing row is reset only when the verifier recorded + * a `retest_reason` for it (the reason is written into the row). + * + * - `verification.canonicalize-uat ` — the ONE place a + * `human_needed` report becomes `passed` on the strength of a passed UAT + * (#4663). verify-work, execute-phase and progress all call it, so the + * rule cannot be re-derived differently per workflow. + * + * The verifier remains the only emitter of a verdict it reached itself; the + * canonicalization records the human's half of a `human_needed` verdict, and + * refuses every other status — a `stale` report is never flipped. + * + * ADR-457 build-at-publish: compiled by tsc to msd-core/bin/lib/uat-evidence.cjs. + */ + +import fs from 'node:fs'; +import path from 'node:path'; +import { findProjectRoot } from './project-root.cjs'; +// eslint-disable-next-line @typescript-eslint/no-require-imports -- io.cjs is an export= CommonJS module +import io = require('./io.cjs'); +// eslint-disable-next-line @typescript-eslint/no-require-imports -- phase-id.cjs is an export= CommonJS module +import phaseId = require('./phase-id.cjs'); +// eslint-disable-next-line @typescript-eslint/no-require-imports -- frontmatter.cjs is an export= CommonJS module +import frontmatterMod = require('./frontmatter.cjs'); +// eslint-disable-next-line @typescript-eslint/no-require-imports -- core-utils.cjs is an export= CommonJS module +import coreUtilsMod = require('./core-utils.cjs'); +// eslint-disable-next-line @typescript-eslint/no-require-imports -- verification.cjs is an export= CommonJS module +import verification = require('./verification.cjs'); + +// eslint-disable-next-line @typescript-eslint/no-require-imports -- uat-predicate.cjs is an export= CommonJS module +import uatPredicate = require('./uat-predicate.cjs'); + +const { output, error } = io; +const { evaluateUatPassed } = uatPredicate; +const { extractPhaseToken } = phaseId; +const { extractFrontmatter, frontmatterListEntries, spliceFrontmatter } = frontmatterMod; +const { normalizeLineEndings } = coreUtilsMod; +const { + resolveUatFile, + resolveVerificationFile, + readVerificationStatus, + defaultPhaseCleanCommitTimesMs, + canonicalizeCoveredFiles, + parseFingerprintFileArgs, +} = verification; + +// ─── Types ──────────────────────────────────────────────────────────────────── + +interface UatRow { + test: number; + name: string; + /** Lowercased `result:` value without brackets; `missing` when the row has none. */ + result: string; + passing: boolean; + expected: string; + /** Present when a previous seed reset this row for re-test. */ + retest_reason: string; + /** Line indexes into the scanned document (internal to the merge). */ + headingLine: number; + resultLine: number; + endLine: number; +} + +interface HumanItem { + test: string; + expected: string; + why_human: string; + /** Set by the verifier ONLY when it re-lists an item whose UAT row already passed. */ + retest_reason: string; +} + +type PublicRow = Pick; + +interface UatEvidence { + uat_file: string; + uat_status: string; + /** ISO time the UAT file last changed (commit time when clean, else mtime); '' when no UAT file. */ + recorded_at: string; + rows: PublicRow[]; + /** Implementation files compared against `recorded_at` (`.planning/` paths are never compared). */ + checked_files: string[]; + changed_since_uat: Array<{ file: string; reason: 'modified' | 'missing' }>; + /** True only when a UAT file exists and no checked file changed after it. */ + code_unchanged_since_uat: boolean; +} + +interface SeedUatResult { + uat_file: string; + created: boolean; + changed: boolean; + kept: Array<{ test: number; name: string; result: string }>; + appended: Array<{ test: number; name: string }>; + retested: Array<{ test: number; name: string; previous_result: string; reason: string }>; + pending: number; + /** Every row passes — verify-work can canonicalize without asking the human anything. */ + all_passing: boolean; +} + +type CleanCommitTimesFn = (dir: string, files: string[]) => Map; + +// ─── Row scanning ───────────────────────────────────────────────────────────── + +const PASSING_RESULTS: ReadonlySet = new Set(['pass', 'passed']); +const HEADING_RE = /^###\s*(\d+)\.\s*(.+)$/; +const SECTION_RE = /^##\s+\S/; +const RESULT_RE = /^result:[ \t]*\[?([\w-]+)\]?/i; +const FENCE_RE = /^\s{0,3}(`{3,}|~{3,})/; + +/** Index of the first body line — the line after a byte-0 frontmatter block, or 0. */ +function bodyStart(lines: readonly string[]): number { + if (lines[0] !== '---') return 0; + for (let i = 1; i < lines.length; i++) { + if (/^---[ \t]*$/.test(lines[i])) return i + 1; + } + return 0; +} + +/** + * Scan a `*-UAT.md` body for `### N. Name` test rows, with the line positions + * the merge needs. Headings and `result:` lines inside fenced code or an HTML + * comment are not rows — the same contexts `uat-predicate.cts` discards, so a + * row this scan would rewrite is always one the completion predicate reads. + */ +function scanUatRows(lines: readonly string[]): UatRow[] { + const rows: UatRow[] = []; + let current: UatRow | null = null; + let fence: string | null = null; + let inComment = false; + + const close = (endLine: number): void => { + if (current) { + current.endLine = endLine; + rows.push(current); + current = null; + } + }; + + for (let i = bodyStart(lines); i < lines.length; i++) { + const line = lines[i]; + if (inComment) { + if (line.includes('-->')) inComment = false; + continue; + } + const fenceMatch = FENCE_RE.exec(line); + if (fence) { + if (fenceMatch && fenceMatch[1][0] === fence[0] && fenceMatch[1].length >= fence.length) fence = null; + continue; + } + if (fenceMatch) { + fence = fenceMatch[1]; + continue; + } + if (line.includes('', line.indexOf('