diff --git a/.claude-plugin/marketplace.json b/.claude-plugin/marketplace.json index 5dda4c795..28c3ced18 100644 --- a/.claude-plugin/marketplace.json +++ b/.claude-plugin/marketplace.json @@ -9,7 +9,7 @@ { "name": "gsd-core", "description": "GSD Core is a meta-prompting, context engineering, and spec-driven development system for AI coding agents.", - "version": "1.13.0", + "version": "1.14.0", "source": "./", "author": { "name": "open-gsd", diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index a6ecbf90e..2d2a04443 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "gsd-core", "displayName": "GSD Core", - "version": "1.13.0", + "version": "1.14.0", "description": "GSD Core is a meta-prompting, context engineering, and spec-driven development system for AI coding agents.", "author": { "name": "open-gsd", diff --git a/.github/workflows/dependabot-vendor-refresh.yml b/.github/workflows/dependabot-vendor-refresh.yml new file mode 100644 index 000000000..07036bd89 --- /dev/null +++ b/.github/workflows/dependabot-vendor-refresh.yml @@ -0,0 +1,120 @@ +name: Dependabot Vendor Refresh + +# #4573: scripts/lint-vendored-deps.cjs gates gsd-core/bin/lib/vendor/{js-yaml.cjs,re2js.cjs} +# for byte-freshness against node_modules, and requires package.json's +# devDependencies pin to literally match the installed version. Dependabot +# regularly opens lockfile-only PRs that bump these packages within the +# existing semver range — it never touches the vendor copy or the manifest +# pin, so the lint-vendored-deps check in test.yml correctly (but +# unhelpfully) flags every such PR as stale, and a human has to manually run +# the refresh command and push a fixup commit. +# +# This workflow does that refresh automatically: on a Dependabot PR that +# touches package.json/package-lock.json, it runs +# `node scripts/lint-vendored-deps.cjs --fix`, which mechanically re-copies +# the upstream build artifact over the vendored .cjs (and, for +# upstream-verbatim twins, their .d.cts files) and rewrites the +# package.json pin to match — see fixRow() in scripts/lint-vendored-deps.cjs +# for exactly what it does and does not touch (it never hand-edits a +# hand-authored twin like js-yaml.d.cts; a genuine upstream API break there +# is left for a human). +# +# Trust boundary: same-repo Dependabot PRs only. The `if:` guard below +# checks both github.actor and pull_request.user.login — defense-in-depth, +# mirroring dependabot-auto-merge.yml's own rationale (github.actor alone +# can't be forged to a different login, but pairing it with +# pull_request.user.login is GitHub's documented hardening pattern). This +# workflow never checks out or executes a fork's code: Dependabot PRs that +# only touch package.json/package-lock.json originate same-repo, and the +# checkout below pins `ref` to the PR head SHA of that same-repo branch. +# +# Why pushing here is not a bypass of the real check: the push (via +# GSD_BOT_PR_TOKEN, falling back to GITHUB_TOKEN — see +# auto-backmerge.yml's "Open or update PR" step for the same fallback +# pattern) lands a new commit on the PR branch, which re-triggers this +# workflow's own `synchronize` trigger AND the `pull_request: synchronize` +# trigger on test.yml. The real lint-vendored-deps check in test.yml then +# re-runs against the fixed commit and genuinely passes — this workflow +# fixes the actual drift the check complains about, it does not silence, +# skip, or override the check itself. + +on: + pull_request: + types: [opened, synchronize] + branches: [next] + paths: + - package.json + - package-lock.json + +concurrency: + group: dependabot-vendor-refresh-${{ github.event.pull_request.number }} + cancel-in-progress: true + +permissions: + contents: write + +jobs: + refresh-vendor: + # Defense-in-depth: check both the triggering actor and the PR author, + # same rationale as dependabot-auto-merge.yml. + if: | + github.actor == 'dependabot[bot]' && + github.event.pull_request.user.login == 'dependabot[bot]' + runs-on: ubuntu-latest + timeout-minutes: 5 + steps: + # persist-credentials: false — the write-capable token must not sit on + # disk while the --fix step below runs `npm ci` and requires the + # freshly-installed, Dependabot-proposed (unreviewed) vendor package. + # The token is only reintroduced, transiently, in the push step, after + # that require() has already run. + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + ref: ${{ github.event.pull_request.head.sha }} + token: ${{ secrets.GSD_BOT_PR_TOKEN || secrets.GITHUB_TOKEN }} + persist-credentials: false + + - uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0 + with: + node-version: 24 + + - name: Install dependencies + run: npm ci --ignore-scripts + + - name: Run lint-vendored-deps --fix + id: fix + run: | + set +e + node scripts/lint-vendored-deps.cjs --fix + echo "exit_code=$?" >> "$GITHUB_OUTPUT" + set -e + + - name: Check for changes + id: diff + run: | + if [ -n "$(git status --porcelain)" ]; then + echo "dirty=true" >> "$GITHUB_OUTPUT" + else + echo "dirty=false" >> "$GITHUB_OUTPUT" + fi + + - name: Commit and push the mechanical refresh + if: steps.fix.outputs.exit_code == '0' && steps.diff.outputs.dirty == 'true' + env: + GH_TOKEN: ${{ secrets.GSD_BOT_PR_TOKEN || secrets.GITHUB_TOKEN }} + run: | + set -euo pipefail + git config user.name "github-actions[bot]" + git config user.email "41898282+github-actions[bot]@users.noreply.github.com" + git add -A -- gsd-core/bin/lib/vendor package.json + git commit -m "chore: refresh vendored deps to match dependency bump" + git remote set-url origin "https://x-access-token:${GH_TOKEN}@github.com/${{ github.repository }}.git" + git push origin HEAD:${{ github.head_ref }} + + # Intentionally a no-op when the --fix step did not exit 0 (findings + # remain after --fix, e.g. a hand-authored twin's declared-export + # finding): a remaining finding means a real upstream API break that + # only a human can resolve. Auto-committing here would either mask + # that break behind a green re-run or ship something wrong. The real + # lint-vendored-deps check in test.yml still runs on the original + # commit and reports the failure to the PR exactly as it does today. diff --git a/.github/workflows/mutation.yml b/.github/workflows/mutation.yml index 7d1460132..ddebbca4d 100644 --- a/.github/workflows/mutation.yml +++ b/.github/workflows/mutation.yml @@ -127,7 +127,7 @@ jobs: run: echo "CI_JOB_START_EPOCH_MS=$(( $(date +%s) * 1000 ))" >> "$GITHUB_ENV" - name: Set up Node - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4.4.0 + uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0 with: node-version-file: .nvmrc cache: npm diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 9e0d442ac..4875d30a3 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -452,7 +452,7 @@ jobs: # exactly the layout c8 expects from a single run (test.yml's # coverage-gate job does the identical merge for the same reason). - name: Download every shard's raw coverage - uses: actions/download-artifact@018cc2cf5baa6db3ef3c5f8a56943fffe632ef53 # v6.0.0 + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: pattern: coverage-tmp-rc-shard-* path: coverage/tmp @@ -727,7 +727,7 @@ jobs: run: npm ci - name: Download every shard's raw coverage - uses: actions/download-artifact@018cc2cf5baa6db3ef3c5f8a56943fffe632ef53 # v6.0.0 + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: pattern: coverage-tmp-finalize-shard-* path: coverage/tmp diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 92731bc1f..0bf2bef1c 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -45,6 +45,72 @@ jobs: contents: read pull-requests: read + # #4422: three unrelated PRs merged on top of an already-broken `next` on + # 2026-09-06 before anyone noticed — nothing checked the base branch's OWN + # health before letting a PR land on it. This job queries the GitHub API for + # the base branch's last push-triggered Tests run and blocks the merge if it + # is red, with a `fix-next` label escape hatch for the PR that is itself the + # fix-forward. See scripts/ci-next-health.cjs's header for the full risk- + # asymmetry reasoning (it deliberately does NOT fail open on a definite red + # signal, unlike the mergeability preflight above). + # + # No `if:` guard, for the same reason `preflight` has none: a SKIPPED + # dependency skips its dependents exactly like a failed one, so guarding this + # job would skip `required-tests` on every push/workflow_dispatch run too. + # scripts/ci-next-health.cjs itself no-ops (zero API calls, exit 0) on any + # event other than pull_request/merge_group. + # + # `next-health` is likewise deliberately NOT gated behind `preflight`, same + # rationale as `changes` above: it is a ~2-minute, compute-free API read that + # runs in parallel with the preflight job, and serializing it behind + # `preflight` would add latency to every healthy PR for no saving. + next-health: + name: Base branch health + runs-on: ubuntu-latest + timeout-minutes: 2 + permissions: + contents: read + actions: read + steps: + - name: Check out the next-health script from the base branch + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + # Same rationale as pr-mergeable-preflight.yml's checkout: pinned to + # the BASE sha so the script that decides the gate comes from a + # trusted commit, never PR-supplied code. Empty on a non-pull_request + # event (e.g. merge_group), where actions/checkout falls back to the + # triggering ref — fine, since the script no-ops before reading + # anything on those events too. + ref: ${{ github.event.pull_request.base.sha }} + fetch-depth: 1 + sparse-checkout: | + scripts + persist-credentials: false + + - name: Check base branch health + id: check + env: + # Every value arrives through env. CONTRIBUTING.md forbids `${{ }}` + # inside a `run:` block. + GITHUB_TOKEN: ${{ github.token }} + PR_LABELS: ${{ join(github.event.pull_request.labels.*.name, ',') }} + MERGE_GROUP_BASE_REF: ${{ github.event.merge_group.base_ref }} + run: | + # Bootstrap arm, mirroring pr-mergeable-preflight.yml's: the checkout + # above is of the BASE sha, so this step runs the script as it exists + # on the base branch — which means it is absent on the PR that + # INTRODUCES it, and on any PR branched from a base predating it. + # Absent is not "red": it is one more thing we cannot determine, so + # it takes the same fail-open path as any other unresolved read. + # Self-healing — once the script is on the base branch this arm never + # fires again. + if [ ! -f scripts/ci-next-health.cjs ]; then + echo "::warning::scripts/ci-next-health.cjs is not present at the base sha; skipping the base-branch health gate (fail-open). Expected on the PR that introduces it, or on a branch whose base predates it." + echo "verdict=INDETERMINATE" >> "$GITHUB_OUTPUT" + exit 0 + fi + node scripts/ci-next-health.cjs + changes: name: Detect test scope runs-on: ubuntu-latest @@ -54,7 +120,6 @@ jobs: full_matrix: ${{ steps.scope.outputs.full_matrix }} product_changed: ${{ steps.scope.outputs.product_changed }} targeted_tests: ${{ steps.scope.outputs.targeted_tests }} - windows_tests: ${{ steps.scope.outputs.windows_tests }} steps: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 with: @@ -77,7 +142,6 @@ jobs: echo "product_changed=true" echo "full_matrix=true" echo "targeted_tests=" - echo "windows_tests=" } >> "$GITHUB_OUTPUT" { echo "## Test scope" @@ -98,7 +162,6 @@ jobs: console.log(`- code_changed: \`${result.code_changed}\``); console.log(`- full_matrix: \`${result.full_matrix}\``); console.log(`- targeted_tests: \`${result.targeted_tests.length}\``); - console.log(`- windows_tests: \`${result.windows_tests.length}\``); if (result.reasons.length > 0) { console.log(''); console.log('### Reasons'); @@ -196,25 +259,21 @@ jobs: # that failure mode, it is re-basing the SAME headroom policy on a number # that had gone stale. # - # The `scope: windows` lane is sharded three ways for the same reason - # (see #3057): on PR #3094 it reached 15m05s against a 15-minute cap and - # was CANCELLED, four shas in a row — a change to tests/helpers.cjs scoped - # in the install-heavy suites and pushed the single Windows lane over the - # top. Per #869, a timeout bump only moves that cliff; sharding removes - # it. That lane does not run any aux suite on its own shard 1 (the - # aux-suite `if:` conditions below are gated on `scope == 'full'` - # specifically), so it needs no reserve and is unaffected by this cap - # change beyond sharing the same job-level `timeout-minutes`. + # #4641: this job's `scope: windows` lane (three shards, #3057) is deleted + # — test-conformance is now the sole Windows selector. See + # docs/adr/4641-windows-selector-consolidation.md. # tests/ci-test-job-timeout-budget.test.cjs holds every lane here to a # headroom factor over its own measured cost. timeout-minutes: 32 env: GSD_PLUGIN_ROOT: .ci-gsd-plugin-root-disabled - # #2665: a live-config leak fails the run on Linux/macOS lanes. Windows - # stays report-only: the guard's first run found PRE-EXISTING USERPROFILE - # leaks there (~190 test sites sandbox HOME alone), a documented separate - # class — promote once that sweep lands (see live-config-guard.cjs SEVERITY). - GSD_STRICT_LIVE_CONFIG_GUARD: ${{ matrix.os != 'windows-latest' && '1' || '' }} + # #2665 / #4641: this job's matrix is ubuntu-only as of #4641 (the + # `scope: windows` rows moved to test-conformance, see the timeout + # comment above), so the live-config leak guard is strict unconditionally + # here. The Windows report-only carve-out (PRE-EXISTING USERPROFILE leaks, + # ~190 test sites sandbox HOME alone) now lives solely on jobs.test-conformance + # — promote it there once that sweep lands (see live-config-guard.cjs SEVERITY). + GSD_STRICT_LIVE_CONFIG_GUARD: '1' # #2854: pin the emitted gate's baseline to the SAME commit the tree was merged # with. "Rebase check" merges `pull_request.base.sha` (pinned by #2472 so all 12 # matrix jobs agree on one tree), but `resolveBase()` otherwise falls through to @@ -258,8 +317,8 @@ jobs: # runner grew until it blew a 15-minute cap and reddened `next`. Raising # the cap treated the symptom; sharding changes the shape. Shards are # partitioned by MEASURED per-file duration (tests/test-timings.json) - # using LPT in scripts/run-tests.cjs — the same cost-aware packer #2472 - # gave the `test-full` lane, which measures 0.0% spread across 3 bins on + # using LPT in scripts/run-tests.cjs — the #2472 cost-aware packer, + # which measures 0.0% spread across 3 bins on # the current table. The aux suites (integration/security/install/slow) # run on shard 1 only — sharding them too was evaluated and rejected for # #4070 (see tests/run-tests-harness.test.cjs's "selectShard @@ -271,12 +330,10 @@ jobs: # below reserves that fixed cost out of shard 1's LPT unit-test share, so # shard 1 gets fewer unit-test files rather than more aux-suite runs. # - # #3057: the `scope: windows` lane is now sharded three ways too, the - # same fix applied to the same cliff (#869's stated durable follow-up). - # It hit exactly 15m05s and was CANCELLED on PR #3094, four shas - # straight, after a change to tests/helpers.cjs scoped in the - # install-heavy suites. Its shards use the same `--shard i/n` - # flag on scripts/run-tests.cjs, applied AFTER scope selection. + # #4641: the `scope: windows` lane (three shards, #3057) that used to + # be listed here is deleted — test-conformance is now the sole + # Windows selector. See + # docs/adr/4641-windows-selector-consolidation.md. - os: ubuntu-latest node-version: 24 scope: targeted @@ -292,18 +349,6 @@ jobs: node-version: 24 scope: full shard: 3/3 - - os: windows-latest - node-version: 24 - scope: windows - shard: 1/3 - - os: windows-latest - node-version: 24 - scope: windows - shard: 2/3 - - os: windows-latest - node-version: 24 - scope: windows - shard: 3/3 steps: # Windows lane on checkout v5.0.1 (drops includeIf; no auth flake, uses Node 24 natively). @@ -381,7 +426,6 @@ jobs: env: TEST_SCOPE: ${{ matrix.scope }} TARGETED_TESTS: ${{ needs.changes.outputs.targeted_tests }} - WINDOWS_TESTS: ${{ needs.changes.outputs.windows_tests }} run: node scripts/ci-prepare-test-scope.cjs - name: Run scoped tests @@ -410,7 +454,7 @@ jobs: # invocations identically — each recomputes the whole 3-way partition # independently and must agree on it (see the `sig` cross-job # fingerprint diagnostic further down in run-tests.cjs). The - # `scope: windows` lane runs no aux suite on its own shard 1, so this + # `scope: targeted` lane runs no aux suite on its own, so this # must stay empty there. tests/ci-full-lane-sharding.test.cjs pins both # halves of this contract. Reserve-value derivation: # .gsd/bug/fix-4070-shard1-aux-suite-budget/10-diagnosis.md. @@ -524,81 +568,42 @@ jobs: env: TEST_SCOPE: targeted TARGETED_TESTS: ${{ needs.changes.outputs.targeted_tests }} - WINDOWS_TESTS: ${{ needs.changes.outputs.windows_tests }} run: node scripts/ci-prepare-test-scope.cjs - name: Run scoped tests run: node scripts/run-tests.cjs --files-from .ci-selected-tests.txt - test-full: - name: full test (${{ matrix.os }}, ${{ matrix.node-version }}, shard ${{ matrix.shard }}/3) + # #4591 (epic #4589 Phase 2): runs ONLY the platform-conformance-tier file + # list (scripts/lib/platform-conformance-tier.generated.cjs) on real + # Windows/macOS — the OS-agnostic bulk of the suite already ran once on + # ubuntu-latest in the `test` job above. Gated on product code changed AND + # full_matrix (#4591); this is the sole gating signal for windows/macos + # coverage (#4603 retired the parallel legacy full-matrix job). Windows is + # sharded 3 ways, for the same reason a prior full-suite windows lane + # needed 3-way sharding (#3057, since retired — #4603), after the first + # real CI run measured it: unsharded, windows-latest hit its 45-minute + # timeout and was CANCELLED (started 03:41:19Z, cancelled 04:26:25Z, run + # 34434252144) while macos-latest finished the identical file set in + # 26m58s — this conformance tier is dominated by subprocess-spawning tests + # (343/565 files). macos-latest stays unsharded; it has real headroom (27m + # against the 45m cap). + test-conformance: + name: conformance test (${{ matrix.os }}, ${{ matrix.node-version }}${{ matrix.shard && format(', shard {0}', matrix.shard) || '' }}) needs: [changes, preflight] if: needs.changes.outputs.code_changed == 'true' && needs.changes.outputs.full_matrix == 'true' runs-on: ${{ matrix.os }} defaults: run: shell: ${{ matrix.shell }} - # The unit suite is sharded across 3 parallel runners per OS/node leg - # (#1212, cost-weighted in #2472). Each shard runs a deterministic - # cost-balanced third of the sorted unit-file list via - # `run-tests.cjs --suite unit --shard i/3`, so per-job - # wall-clock scales as O(total/3) and stays well under the cap as the suite - # grows — replacing the #869 timeout bump (15→20m) which only deferred the - # cliff. The cap stays at 20m as a generous backstop; a healthy shard now - # finishes in roughly a third of the old single-lane wall-clock. - # - # The matrix is the cross-product of 2 OS/node legs × 3 shards = 6 jobs - # (windows-latest/24, macos-latest/24 — the Node floor is 24, so there is - # no separate node-22 leg to cross-product against), enumerated explicitly - # as `include:` rows. (A base `shard: [1,2,3]` - # dimension would NOT cross-product against `include` legs — include rows - # sharing no key with the base matrix are appended as standalone combos — - # and a NESTED `leg.os` key is not resolvable by the H1 shell-policy linter - # in scripts/workflow-policy.cjs, which reads `matrix.os`/`matrix.shell` - # directly. Explicit rows keep both the cross-product and the linter happy.) - # #2952: `full test (windows-latest, 22, shard 3/3)` reached 18m59s (94% of - # a 20-minute cap) on 05b170e44 and 18m14s (91%) on 81eeb8a53. The Windows - # shards are slow for platform reasons — process spawn and filesystem cost, - # not extra work — and this lane has already blown its cap twice before - # (#1051, #1212). - # #3787: fresh measurement on windows-latest/24 shard 3/3 hit 26m18s (run - # 32614439702), so the 18m59s figure above is stale and the 30-minute cap - # only had ~1.14x headroom, in violation of this repo's own 1.5x rule - # (tests/ci-test-job-timeout-budget.test.cjs). 1.5x of 27m requires 41m - # minimum; 45 is used instead of the bare minimum because shard - # composition is unstable — adding one test file reshuffled 115 of 268 - # unit files between shards — so the per-shard worst case moves run to - # run and a budget pinned to the exact minimum would be re-breached by - # the next file anyone adds. + # No LANE_COSTS entry exists yet for this brand-new job — see + # tests/ci-test-job-timeout-budget.test.cjs's own header: "no unit test + # can prove a lane fits its budget — only a real CI run measures that." + # Generous until a real measurement exists; the smaller file count + # should comfortably undercut this ceiling. timeout-minutes: 45 env: GSD_PLUGIN_ROOT: .ci-gsd-plugin-root-disabled - # #2665: strict on Linux/macOS, report-only on Windows (see the `test` job note). GSD_STRICT_LIVE_CONFIG_GUARD: ${{ matrix.os != 'windows-latest' && '1' || '' }} - # #2854: pin the emitted gate's baseline to the SAME commit the tree was merged - # with. "Rebase check" merges `pull_request.base.sha` (pinned by #2472 so all 12 - # matrix jobs agree on one tree), but `resolveBase()` otherwise falls through to - # `origin/next`, which `fetch-depth: 0` leaves at the LIVE tip. Whenever `next` - # advanced mid-flight the gate compared a tree built on base.sha against a - # baseline at a newer commit — so the correctly-keyed cache was rejected as - # "stale" and the run hard-failed on diffs that touched nothing related. - # This must stay equal to CI_REBASE_BASE_SHA; a test asserts that parity. GSD_EMITTED_BASE: ${{ github.event.pull_request.base.sha }} - # #4196: pin the npm-audit baseline the SAME way GSD_EMITTED_BASE pins - # its own baseline (see the comment above) -- origin/next is live under - # fetch-depth: 0 and can advance mid-run; base.sha is fixed for the life - # of the run. For a push event, github.event.before is git's own record - # of the ref's state immediately before this push landed -- correct - # even when a rebase-merged PR lands as multiple discrete commits in - # one push (HEAD~1 would be wrong there: it could already contain an - # earlier commit's newly-introduced vulnerable package, masking it). - # #4241: a merge_group event carries no pull_request/push context, so - # without this arm it silently fell through to the '' branch -- - # resolveBaselineRef()'s documented-unreachable origin/next live-tip - # fallback (npm-audit-baseline.cjs), reopening the exact race #4196 - # fixed, but only for merge-queue runs. github.event.merge_group.base_sha - # is "the SHA of the merge group's parent commit" (GitHub's merge_group - # webhook payload) -- the base tip the temporary merge-group commit was - # built against, pinned for the life of the run same as the other two arms. AUDIT_BASELINE_REF: ${{ github.event_name == 'pull_request' && github.event.pull_request.base.sha || (github.event_name == 'push' && github.event.before) || (github.event_name == 'merge_group' && github.event.merge_group.base_sha) || '' }} strategy: fail-fast: false @@ -607,27 +612,18 @@ jobs: - os: windows-latest node-version: 24 shell: pwsh - shard: 1 + shard: 1/3 - os: windows-latest node-version: 24 shell: pwsh - shard: 2 + shard: 2/3 - os: windows-latest node-version: 24 shell: pwsh - shard: 3 + shard: 3/3 - os: macos-latest node-version: 24 shell: 'zsh {0}' - shard: 1 - - os: macos-latest - node-version: 24 - shell: 'zsh {0}' - shard: 2 - - os: macos-latest - node-version: 24 - shell: 'zsh {0}' - shard: 3 steps: - uses: actions/checkout@93cb6efe18208431cddfb8368fd83d5badbf9bfd # v5.0.1 (Windows) @@ -654,12 +650,6 @@ jobs: if: github.event_name == 'pull_request' env: GITHUB_TOKEN: ${{ github.token }} - # Pin every job of this run to ONE base commit (#2472). Each job runs - # this step independently, minutes apart across a 12-job matrix, so - # merging the moving branch ref lets jobs see different trees when the - # base advances mid-run. The sharded lane needs all jobs to agree on a - # partition, and disagreement there drops a test file silently while - # CI stays green. base.sha is fixed for the life of the run. CI_REBASE_BASE_SHA: ${{ github.event.pull_request.base.sha }} run: node scripts/ci-rebase-check.cjs @@ -678,9 +668,6 @@ jobs: - name: Dependency integrity gate run: node scripts/check-npm-integrity.cjs - # #2724 (ADR-2719 §5): restore the differential attribution check's cached - # baseline, keyed on the PR's base sha. A miss degrades to an in-job build - # rather than failing (resolveBaseline()'s documented precedence). - name: Restore emitted-baseline cache if: github.event_name == 'pull_request' uses: actions/cache/restore@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 @@ -688,31 +675,33 @@ jobs: path: .gsd-cache/emitted-baseline.json key: emitted-baseline-${{ github.event.pull_request.base.sha }} - # The heavy unit suite is split across the 3 shards — each runs a - # deterministic cost-balanced third of the sorted unit-file list (#2472). The - # union of shards 1/3 + 2/3 + 3/3 is the full unit suite, so coverage is - # unchanged; only wall-clock per job drops to ~total/3. - - name: Run unit tests (shard ${{ matrix.shard }}/3) - run: node scripts/run-tests.cjs --suite unit --shard ${{ matrix.shard }}/3 + # #4591: the generated conformance-tier list is a .cjs module (so + # scripts/gen-platform-conformance-tier.cjs's own tests and other + # scripts can `require()` it directly) — run-tests.cjs --files-from + # expects a plain newline-delimited text file, so this step bridges + # the two, one path per line, matching the existing scoped-lane + # convention (.ci-selected-tests.txt). Windows keeps the general, + # Windows-inclusive list. + - name: Prepare conformance-tier test list (windows) + if: matrix.os == 'windows-latest' + run: node -e "require('./scripts/lib/platform-conformance-tier.generated.cjs').CONFORMANCE_TIER_FILES.forEach(f => console.log(f))" > .ci-conformance-tests.txt - # Integration and security suites are small; run them once per OS/node - # leg (on shard 1 only) instead of redundantly on all three shards. They - # still run on every leg (3 times total, once per platform), so each - # platform's integration/security coverage is unchanged — only the - # 3x-per-leg duplication is removed. - - name: Run integration tests - if: matrix.shard == 1 - run: npm run test:integration + # #4593: macOS reads a narrower, macOS-specific file list instead of the + # shared/Windows-oriented one above — see + # docs/adr/4593-macos-conformance-tier-architecture.md. Writes into the + # same conventional filename so the run step below is unchanged. + - name: Prepare conformance-tier test list (macos) + if: matrix.os == 'macos-latest' + run: node -e "require('./scripts/lib/macos-conformance-tier.generated.cjs').MACOS_CONFORMANCE_TIER_FILES.forEach(f => console.log(f))" > .ci-conformance-tests.txt - - name: Run security tests - if: matrix.shard == 1 - run: npm run test:security + - name: Run platform-conformance-tier tests + run: node scripts/run-tests.cjs --files-from .ci-conformance-tests.txt${{ matrix.shard && format(' --shard {0}', matrix.shard) || '' }} - name: Check job budget (near-cap advisory) if: always() continue-on-error: true env: - CI_JOB_LABEL: "full test (${{ matrix.os }}, ${{ matrix.node-version }}, shard ${{ matrix.shard }}/3)" + CI_JOB_LABEL: "conformance test (${{ matrix.os }}, ${{ matrix.node-version }})" CI_JOB_TIMEOUT_MINUTES: '45' run: node scripts/ci-check-job-near-cap.cjs @@ -747,7 +736,7 @@ jobs: # is exactly the layout c8 expects from a single run. The dumps are # per-process files with distinct names, so there is nothing to collide. - name: Download every shard's raw coverage - uses: actions/download-artifact@018cc2cf5baa6db3ef3c5f8a56943fffe632ef53 # v6.0.0 + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: pattern: coverage-tmp-shard-* path: coverage/tmp @@ -857,11 +846,12 @@ jobs: name: Required tests needs: - preflight + - next-health - changes - lint-tests - test - test-inert - - test-full + - test-conformance - coverage-gate - qa-loop-walk if: always() @@ -871,25 +861,27 @@ jobs: - name: Summarize required test gate env: PREFLIGHT_RESULT: ${{ needs.preflight.result }} + NEXT_HEALTH_RESULT: ${{ needs.next-health.result }} CODE_CHANGED: ${{ needs.changes.outputs.code_changed }} PRODUCT_CHANGED: ${{ needs.changes.outputs.product_changed }} CHANGES_RESULT: ${{ needs.changes.result }} LINT_RESULT: ${{ needs.lint-tests.result }} TEST_RESULT: ${{ needs.test.result }} + TEST_CONFORMANCE_RESULT: ${{ needs.test-conformance.result }} INERT_RESULT: ${{ needs.test-inert.result }} - FULL_TEST_RESULT: ${{ needs.test-full.result }} COVERAGE_GATE_RESULT: ${{ needs.coverage-gate.result }} QA_LOOP_WALK_RESULT: ${{ needs.qa-loop-walk.result }} run: | set -euo pipefail echo "preflight=$PREFLIGHT_RESULT" + echo "next-health=$NEXT_HEALTH_RESULT" echo "code_changed=$CODE_CHANGED" echo "product_changed=$PRODUCT_CHANGED" echo "changes=$CHANGES_RESULT" echo "lint-tests=$LINT_RESULT" echo "test=$TEST_RESULT" + echo "test-conformance=$TEST_CONFORMANCE_RESULT" echo "test-inert=$INERT_RESULT" - echo "test-full=$FULL_TEST_RESULT" echo "coverage-gate=$COVERAGE_GATE_RESULT" echo "qa-loop-walk=$QA_LOOP_WALK_RESULT" @@ -910,6 +902,14 @@ jobs: exit 1 fi + # #4422: applies UNCONDITIONALLY, not nested inside the + # PRODUCT_CHANGED/CODE_CHANGED branches below — a red base branch + # must block every PR, including doc-only ones. + if [ "$NEXT_HEALTH_RESULT" != "success" ]; then + echo "::error::the base branch's own last Tests run is red — see the 'Base branch health' job above for the failing run. Wait for a fix-forward merge, or if this PR IS the fix, ask a maintainer to apply the 'fix-next' label to override." + exit 1 + fi + if [ "$LINT_RESULT" != "success" ]; then echo "::error::lint-tests did not pass" exit 1 @@ -925,14 +925,17 @@ jobs: echo "::error::test matrix did not pass" exit 1 fi + # #4591 (epic #4589 Phase 2): test-conformance is the new gating + # signal for real Windows/macOS coverage — the conformance-tier + # file list, not the whole suite. + if [ "$TEST_CONFORMANCE_RESULT" != "success" ] && [ "$TEST_CONFORMANCE_RESULT" != "skipped" ]; then + echo "::error::platform-conformance-tier matrix did not pass" + exit 1 + fi # #2952: the coverage gates no longer run inside the `test` matrix — # sharding moved them to the `coverage-gate` job, which merges every # shard's dumps. TEST_RESULT therefore does NOT cover them any more; # COVERAGE_GATE_RESULT below is what gates coverage. - if [ "$FULL_TEST_RESULT" != "success" ] && [ "$FULL_TEST_RESULT" != "skipped" ]; then - echo "::error::full parity matrix did not pass" - exit 1 - fi # #2952: the coverage gate is skipped when product code did not change # (same condition as the test lane). Only a non-success, non-skipped @@ -959,7 +962,7 @@ jobs: # #2724 (ADR-2719 §5): publishes the differential attribution check's baseline # artifact after `next` advances, keyed on the merge sha. PR lanes restore it - # (see the `test` and `test-full` jobs' "Restore emitted-baseline cache" steps), + # (see the `test` and `test-conformance` jobs' "Restore emitted-baseline cache" steps), # keyed on `pull_request.base.sha` — "the next sha the PR was merged with", the # ADR's own phrasing. Not required by `required-tests`: a miss here degrades PR # lanes to an in-job build (resolveBaseline()'s documented precedence) rather diff --git a/.gitignore b/.gitignore index ea044ca1c..7b768feea 100644 --- a/.gitignore +++ b/.gitignore @@ -138,6 +138,8 @@ build/ /gsd-core/bin/lib/ui-consideration-probe.cjs /gsd-core/bin/lib/config-types.cjs /gsd-core/bin/lib/cli-exit.cjs +# #4145: emitted artifact of src/pristine-baseline.cts — never edited. +/gsd-core/bin/lib/pristine-baseline.cjs /gsd-core/bin/lib/code-review-flags.cjs /gsd-core/bin/lib/code-review-depth.cjs /gsd-core/bin/lib/context-utilization.cjs @@ -150,6 +152,7 @@ build/ /gsd-core/bin/lib/review-lane-descriptor.cjs /gsd-core/bin/lib/review-lane-invocation.cjs /gsd-core/bin/lib/review-lane-runner.cjs +/gsd-core/bin/lib/reviewer-step-dispatch.cjs /gsd-core/bin/lib/clusters.cjs /gsd-core/bin/lib/installer-migrations/001-legacy-orphan-files.cjs /gsd-core/bin/lib/observability/redaction.cjs @@ -311,6 +314,7 @@ build/ /gsd-core/bin/lib/git-base-branch.cjs /gsd-core/bin/lib/host-runtime-detection.cjs /gsd-core/bin/lib/task-content-resolution.cjs +/gsd-core/bin/lib/tdd-red-evidence.cjs __pycache__/ *.pyc .venv/ diff --git a/.out-of-scope/codex-native-supervisor-adapter.md b/.out-of-scope/codex-native-supervisor-adapter.md new file mode 100644 index 000000000..141de0e03 --- /dev/null +++ b/.out-of-scope/codex-native-supervisor-adapter.md @@ -0,0 +1,106 @@ +# Codex native supervisor adapter for orchestrator-worktree executors + +**Source:** [#4625](https://github.com/open-gsd/gsd-core/issues/4625) +**Decision:** wontfix — No-go as filed; the observability goal is redirected to EoS, behind [#4624](https://github.com/open-gsd/gsd-core/issues/4624) +**Date:** 2026-09-11 + +## Proposal summary + +Reporter proposed an opt-in, Codex-only supervision adapter layered on the existing negotiated +`orchestrator-worktree` dispatch backend. For each selected executor the root would create the +existing GSD-managed worktree and launch one **native Codex supervisor subagent**, which owns +exactly one external `codex exec --cd ` worker and surfaces evidence-backed progress in +Codex's native subagent UI as three strict states: + +- `executing` — the persisted worker process is alive after successful launch +- `verifying` — that process reached a terminal state and the supervisor is reconciling its + result, SUMMARY, plan-scoped commits, branch and manifest +- `completed` — reconciliation succeeded and the manifest-only merge/cleanup gauntlet finished + +The supervisor would run the worker in the foreground (explicitly not backgrounded with `&` +behind a cosmetic status), consume the persisted lifecycle protocol proposed in #4624, and be +disabled by default with no behavior change for other runtimes or the default direct adapter. + +## Why GSD does not own this + +- **The target surface cannot carry the states the proposal is built on.** Codex's native + subagent status is a fixed two-value enum — `CollabAgentToolCallStatus::{InProgress, + Completed}`, emitted by `wait_agent`. There is no free-text or custom status field. The + proposal's `executing` / `verifying` / `completed` triad cannot be rendered in that view **by + anyone, gsd-core included.** This is not a question of where the adapter lives, and it is the + decisive ground: the feature's central promise is not deliverable as specified. + +- **The native subagent list is populated only by Codex's own `spawn_agent` / `wait_agent` + machinery.** `notify` (a fire-and-forget subprocess spawn on `agent-turn-complete`) and the + `hooks.toml` command hooks observe Codex's *own* turn and tool events; neither is documented as + rendering into that list. A framework that did not itself drive `spawn_agent` could not reach it. + +- **That machinery is unreleased.** `multi_agents_v2` was found at source level on the `openai/codex` + `main` branch and could not be verified as shipped or GA; the `--experimental-json` alias beside it + points the same way. Committing a core-resident adapter to an unstable, unreleased third-party + interface would be building ahead of the vendor. + +- **A second core dispatch adapter is a permanent tax.** Every future change to the executor + lifecycle would have to be made twice, or proven to apply to both paths. The graph rates the + isolation resolver's blast radius MEDIUM, and `dispatch.isolation` is declared across multiple + runtime capability descriptors, so an adapter is a multi-descriptor change rather than a local one. + +Note what the proposal got **right**, none of which is a ground for rejection: the observability +gap it describes is real and was confirmed; its containment shape (opt-in, default-off, no change +to other runtimes or the default path) is the correct one for this class of change; its +self-imposed rule that a worker must never show `completed` merely because a process exited is +exactly the right invariant; and it correctly identified and filed its own correctness +prerequisite separately as #4624, which is now `confirmed-bug`. The filing was specific enough to +be checked against the runtime, which is why the mechanism problem surfaced at all. + +## What this does NOT cover + +This entry denies **gsd-core building a second, core-resident dispatch adapter around Codex's +native subagent UI.** Its keyword surface — supervisor, observability, worker state, lifecycle, +subagent, Codex, orchestrator-worktree — overlaps requests this decision deliberately does not +deny. Do not apply this entry to: + +- **Worker-state visibility for Codex executors as a goal.** It is legitimate and it is reachable. + `codex exec --json` emits structured JSONL `ThreadEvent`s — `thread.started` (carrying + `thread_id`), `turn.started` / `turn.completed` / `turn.failed`, `item.*`, `thread.error` — and a + session resumes via `codex exec --json resume `. That is a genuine terminal-state + signal for an external worker, and nothing here denies using it. +- **#4624, the persisted lifecycle record.** That is a confirmed defect fix and core work + regardless of this decision. Its diagnosis found the current `orchestrator-worktree` wait path is + a single prose sentence with no captured PID and no durable per-worker record. +- **An EoS capability that reads that record and reports state.** Once a durable per-worker record + exists on disk, a capability can register a `step` hook at `execute:wave:pre` / + `execute:wave:post` (ADR-857's loop extension points) and display each worker's state with no + core change of its own. **This is the sanctioned path for this request**, and it is open now. +- **Fixing the existing `orchestrator-worktree` adapter.** Defects in the shipped path are bug + reports, not this proposal. +- **Codex runtime support generally** (`capabilities/codex/capability.json`), which is unaffected. + +## Re-open criteria + +- **Codex ships a documented, non-experimental surface that accepts a custom status string from a + third party into its native subagent view.** This is the ground the decision actually rests on; + if it changes, the decision should be revisited rather than cited. A resubmission should name + the specific released Codex version and the surface. +- Separately, and only then: #4624 has shipped, and a resubmission **names a specific residual + visibility failure observed after the EoS read-and-report path was tried** — rather than arguing + for the native-UI adapter in the abstract. + +A core-resident adapter is accepted here only once the surface it targets can carry the states it +was specified to show. Until then, the wave-boundary read of a persisted record is strictly more +capable than the thing this entry declines, because it can express states the native enum cannot. + +## Related + +- [#4624](https://github.com/open-gsd/gsd-core/issues/4624) — the persisted-lifecycle defect this + was sequenced behind; `confirmed-bug` +- [`subagent-activity-watchdog.md`](./subagent-activity-watchdog.md) — denies a *generalized* + event-driven watchdog; explicitly preserves narrow per-spawn-site work, and so does **not** deny + the EoS path above +- [`codex-native-plugin-skips-preproposal.md`](./codex-native-plugin-skips-preproposal.md) — the + standing no-new-first-party-add-ons policy; targets distribution surfaces, and so was **not** the + ground for this decision +- [ADR-857](../docs/adr/857-capability-system.md) — the capability system and its 12 loop extension + points +- `gsd-core/references/loop-hook-dispatch.md` — the `contribution` / `step` / `gate` hook contract +- `capabilities/codex/capability.json` — existing Codex runtime support, unaffected diff --git a/.out-of-scope/executor-self-repair-worktree-base.md b/.out-of-scope/executor-self-repair-worktree-base.md new file mode 100644 index 000000000..fa85c91e3 --- /dev/null +++ b/.out-of-scope/executor-self-repair-worktree-base.md @@ -0,0 +1,86 @@ +# Executor self-repair of a worktree base mismatch via `git reset --hard` + +**Source:** [#4463](https://github.com/open-gsd/gsd-core/issues/4463) +**Decision:** wontfix — No-go as filed; conflicts with the shipped #48 design +**Date:** 2026-09-07 + +## Proposal summary + +Reporter ran a controlled probe on Claude Code and reported two findings, plus a proposal: + +1. **Measurement:** the harness worktree it was dispatched into forked from the + *orchestrator session's HEAD* (the main checkout's current branch tip), not from + `origin/HEAD` as the `#3659`/`#3779`/`#48` family assumed. The reporter is explicit + that this does not mean the `worktree.base-check` degrade is wrong in the case they + observed — the stated *reason* for the degrade may be inaccurate even where the + degrade itself is still warranted. +2. **Measurement:** because a linked worktree shares the common git object/ref store, + the phase branch is reachable as a local ref with no `git fetch`, and + `git reset --hard ` on the executor's own branch is cheap, does not + switch or detach HEAD, and does not conflict with "branch already checked out + elsewhere" (that check only fires on `git checkout`, not on moving one's own branch + pointer via `reset`). +3. **Proposal:** where `worktree-branch-check` currently halts with `exit 42` on a base + mismatch, let the executor **repair itself** — `git reset --hard ` + on its own branch — and proceed, converting the halt into a self-heal. + +## Why GSD does not own this + +- **This is the exact primitive #48 removed, for the exact failure mode #48 was filed + to fix.** #48 (closed, `approved-enhancement`, shipped) replaced sub-agent-side + `git reset --hard` recovery with a verify-only, fail-closed `exit 42` check, specifically + because (a) a permission deny-rule on `git reset --hard*` — common in safety-conscious + host configurations — can make the recovery command itself fail, and depending on shell + error handling the sub-agent may silently proceed on the wrong base or report success + without re-verifying; and (b) a sub-agent should not hold state-correction primitives + (`reset`, force-move, branch-switch) on a worktree it did not create — that + responsibility belongs to the lifecycle owner, the orchestrator. #4463's proposal + reintroduces precisely this: the executor mutating its own worktree state in response + to a detected mismatch, on the sub-agent side. +- **The shipped design is live in current source, not just historically decided.** + `gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md:7` states, verbatim: + *"`worktree_branch_check` is verify-only — an executor that hits a base/HEAD-namespace + mismatch prints `FATAL:` and exits **42** instead of self-recovering... The orchestrator — + the worktree lifecycle owner — performs any base correction... the sub-agent never does."* + This is an active architectural invariant, not stale rationale from a closed issue. +- **The "it costs almost nothing and nothing objects" framing is exactly what #48 warned + about.** #4463's own Finding 3 confirms `git reset --hard` ran with no hook and no + refusal in its probe — which is the *absence of a safety net* #48 is trying to compensate + for by moving the responsibility off the sub-agent entirely, not evidence the operation + is safe to grant back to it. + +## What this does NOT cover + +- **The fork-base measurement itself is not denied and is worth keeping.** If the + harness-worktree fork base is genuinely the orchestrator session's HEAD rather than + `origin/HEAD`, that is new information relevant to `#3659`'s and `#3779`'s closure + rationale and to how `worktree.base-check`'s degrade condition is described. A follow-up + that only re-verifies and documents this measurement (with the session cwd inside a + feature worktree, which the original probe did not test) is not this proposal and is + welcome. +- **Orchestrator-side repair is a different proposal.** #48's split explicitly assigns + base correction to the orchestrator (e.g., recreate the worktree on `{EXPECTED_BASE}`, + or fast-forward the branch from the orchestrator side before dispatch). A proposal that + moves the repair step to the orchestrator, before or around dispatch, rather than having + the executor self-repair after detecting a mismatch, is not denied here and would need + its own review. +- **Fixing `#4415`** (cleanup-wave blocking when Claude Code has already removed the + executor's worktree) is unrelated and not affected by this decision. + +## Re-open criteria + +- A proposal that keeps repair on the **orchestrator** side of the #48 split (the + lifecycle owner), not the sub-agent side — matching, not reversing, the shipped + architecture. +- Or: a demonstrated, host-enforced guarantee that the executor's `git reset --hard` call + cannot be silently denied or misreported by a permission policy — removing the specific + failure mode #48 was filed against. Absent that guarantee, granting the primitive back + to the sub-agent reintroduces the original risk regardless of how cheap the operation is + in the success case. + +## Related + +- [#48](https://github.com/open-gsd/gsd-core/issues/48) — the decision this proposal reverses +- `gsd-core/workflows/execute-phase/steps/worktree-recovery-policy.md` — the shipped fail-closed invariant +- [#3659](https://github.com/open-gsd/gsd-core/issues/3659), [#3779](https://github.com/open-gsd/gsd-core/issues/3779), [#683](https://github.com/open-gsd/gsd-core/issues/683) — the fork-base family #4463's measurement bears on +- [#4415](https://github.com/open-gsd/gsd-core/issues/4415) — adjacent cleanup-wave gap, unaffected by this decision diff --git a/CHANGELOG.md b/CHANGELOG.md index 6d0cf6fa5..7d1e1d099 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,118 @@ Format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/). ## [Unreleased] +## [1.14.0] - 2026-09-14 + +### Added + +- **The phase-directory membership seam threads the phase ID convention through the completion chain** — the #3511 seam (`isPhaseArtifact` / `scopeToPhase`) now takes the same optional convention every other read-path helper does, and the completion chain threads it: `state json` / `state sync`'s completed-phase counting, the planning snapshot, roadmap analysis, `state validate`'s drift scan, the verification-report resolver, and `phase complete`'s actual completion gate. A bracket directory therefore scopes its listing by its real phase token instead of the include-everything ambiguity fail-safe, so a cross-phase stray (`01-VERIFICATION.md` misfiled into phase 03's directory) can no longer supply the pass/fail verdict for a bracket phase — the same protection #3511 already gives legacy directories, **on the call sites this PR threads**. + + Call sites that do not yet resolve a convention keep the documented include-everything fail-safe on bracket directories, and this PR changes nothing for them: the aggregate scans (`uat`, `audit`, `init`'s projections, `gap-checker`, `phase-locator`); **`phase complete`'s advisory pre-scan** (`cmdPhaseComplete`, `src/phase.cts`), whose UAT and VERIFICATION warning sweeps still call the seam convention-lessly and can therefore surface a spurious warning for a cross-phase stray, although that scan cannot pass or block completion; and **the workstream inventory's per-phase completion projection** (`src/workstream-inventory.cts`), which calls the now-convention-aware `isPhaseComplete` without resolving a convention to pass it and can therefore still project a bracket phase complete or incomplete from a cross-phase stray. Threading those readers is follow-up-slice work alongside the epic's other convention-less readers. A project on any convention other than `"bracket"` is unaffected. (#4142) (#3644) +- **`workflow.compact_content` now actually does something: `plan-phase` is the first workflow split into a spine + detail file.** With the key off (default), nothing changes — the spine reads the deferred elaboration back in before continuing, so the instruction set is identical to today. With it on, that read is skipped and the orchestrator runs on the terser spine alone, which is complete enough to plan a phase correctly on its own. The check and the resolution rule live in one shared reference (`gsd-core/references/compact-content-gate.md`) that future splits reference instead of restating. (#4402) (#4471) +- **A new offline benchmark reports the token savings from compact-content splits** — `npm run benchmark:compact-content` measures, per registered `workflow.compact_content` spine/detail split, the token count with and without the split active using a pinned tokenizer, and prints the reduction against a committed baseline without ever failing CI. (#4404) (#4502) +- **Broken-windows ledger entries now record which milestone they belong to** — `windows append` stamps a new `milestone` field from the workstream's resolved milestone version. Phase numbers are unique only within one active phases directory, so two milestones routinely produced entries sharing the same phase number with nothing to distinguish them; `/gsd-ship`'s open-count gate could be silently blocked by another, already-shipped milestone's entries. Absence (an entry recorded before this field existed) reads as null — existing ledgers keep working with no migration. (#4583) +- **Five more workflow spines split into a terser form under `workflow.compact_content`** — `execute-phase`, `docs-update`, `new-project`, `verify-work`, and `complete-milestone` join `plan-phase` (#4402), bringing the total to six, each moving genuinely optional or rare content (interactive-mode flows, off-by-default features, gap-closure loops, cross-AI delegation, branch-merge mechanics) into a deferred `/detail/*.md` elaboration read only when the key is off; several pre-existing structural drift guards pin exact wording in specific spine steps (crash-resume detection, checkpoint auto-approval, learnings extraction, revision-conflict handling), so those sections keep their full text in the spine rather than deferring it. The refreshed benchmark reports a 15.66% aggregate token reduction across the six splits. The remaining eagerly-included workflows were reviewed and recorded as not worth splitting, with reasons, in `docs/PARTITION-RULES.md`. (#4405) (#4536) +- **Compact content mode is now discoverable, not just settable.** `/gsd-new-project` asks about it at init time and `/gsd-settings`/`/gsd-config` toggle it on an already-initialized project, closing out the #4139 compact-content epic. (#4408) (#4587) +- **`workflow.compact_content` now also covers lazily-read workflow fragments and planning-artifact templates.** With the key on, `help --full`'s reference doc and generated `SUMMARY.md`/`USER-SETUP.md` templates resolve to a terser `.compact.md` sibling at the point of their existing `Read` — two independent, complete files, picked per the same shared gate Phase 5 introduced (`gsd-core/references/compact-content-gate.md`). With the key off (default), nothing changes. (#4540) +- **`workflow.compact_content` splits now have a real CI guard.** Any workflow spine + `detail/*.md` split is enforced forever: completeness once at split time, disjointness and registration on every PR, and protected content (guardrails, output-format contracts, few-shot examples, security language, machine-parsed headings) that can never leave the spine, moved or not. Ordinary content moves between spine and detail need a `Boundary-Move-Declared` commit trailer naming the spine, mirroring ADR-3942's emitted-drift-ack trailers. The partition rule and the protected-content list live in one place, `docs/PARTITION-RULES.md`. (#4403) (#4497) +- **`/gsd:code-review` can now optionally corroborate its internal review with registered external reviewer lanes** — new roster-derived flags dispatch a bounded, read-only source review through each selected lane; findings are re-verified against real source and folded into the existing `REVIEW.md`. Bare `/gsd:code-review` (no flag) is unchanged. (#4323) +- **check decision-coverage-plan accepts --context ** — same convention as sibling check verbs. (#4130) (#4374) +- **`workflow.compact_content` is now a registered, validated, documented project config key.** It resolves to `false` when absent and is readable via `config-get`; no content branches on it yet. (#4401) (#4441) +- **Compact agent-persona payloads for non-Claude runtime dispatch, selected by `workflow.compact_content`.** When the key is on, the AGENTS-native persona fallback (kimi-code, opencode, kilo, and similar runtimes without named-subagent dispatch) now serves a token-minimized `.compact.md` variant of the agent's persona instead of the full file, chosen by the same CLI seam (`gsd_run query agent-skills`) that already resolves this content in code rather than prose. An agent with no compact variant registered falls back to the canonical persona and discloses the fallback in the payload itself, so nothing is ever served silently or left empty. (#4407) (#4553) + +### Changed + +- **Planning guidance now prefers the first sufficient implementation option** — existing project behavior, standard-library or native platform capability, installed dependencies, and only then minimum new implementation, without reducing required scope or verification. (#4118) +- **21 GSD skills now declare `Grep` in `allowed-tools`** — cleanup, complete-milestone, config, debug, graphify, health, mempalace-capture, mempalace-recall, new-milestone, new-project, next, pause-work, phase, pr-branch, resume-work, review-backlog, settings, stats, thread, workspace, and workstreams can now use the dedicated structured-search tool instead of shelling out through Bash grep. (#4397) +- **Context-monitor WARNING/CRITICAL fire-points are now readable from `.planning/config.json`** — `hooks.context_warning_threshold` (default 35) and `hooks.context_critical_threshold` (default 25) move the two rungs per project, so a tuned fire-point survives an update instead of being re-staged away with the managed hook file. Absent keys resolve to today's 35/25, so existing projects are unchanged. An unusable value falls back per key; both revert to their defaults only when the resolved pair violates `critical < warning`. The keys are root-project settings — the hook reads `/.planning/config.json` only, and they are read by that hook and nothing else, so on a runtime where it is not installed (Codex, per #2586) both keys are stored and validated but inert. `config-set` refuses the two endpoints that can never take effect — a warning of 0 and a critical of 100 — because `critical < warning` has no legal partner for either, and an absent key now reports the shipped default (35/25) instead of "Key not found". (#4285) (#4366) +- **The path-containment predicate is now a single exported seam** — `security.cjs` no longer exports `validatePath`. Containment is decided in exactly one place and resolved two ways: `assertWithinRoot` (throws) and `tryWithinRoot` (returns null) resolve symlinks, while `assertWithinRootLexical` and `tryWithinRootLexical` use string resolution alone and never touch the filesystem, for the few callers that must preserve a symlink rather than resolve it or that validate a destination before it exists. `requireSafePath` is preserved as an alias of the throwing form. All of them return a branded `ContainedPath` so a validated path cannot be silently swapped for an unvalidated one. The per-call-site `{ allowAbsolute: true }` flag is replaced by the named `PathAcceptance` policy, which states what it actually permits: an absolute path outside the root was always rejected and still is. The traversal rejection text `Path escapes allowed directory: is outside ` is preserved verbatim, and no command changes what it accepts or rejects. Three rejection MESSAGES are reworded, none of which now reveals a host path it previously hid: `state.cts`'s `