From 88f7683b991cacd7aeaf45fc90384b3f0d008acf Mon Sep 17 00:00:00 2001 From: Jakub Zych Date: Sun, 4 Oct 2026 00:53:53 +0200 Subject: [PATCH] test(14-06): check-phase14.sh gates the phase over both repositories, the corpus and the named flows - --self-test proves each detector fails on planted failing, skipped, zero-test, racy and non-JSON runs, a fourth pending route, a vendor SDK in the module graph and planted corpus secrets - --go runs vet and tests in summercms.go and, with -race, in fonoteka.go including sm-golem-plugin and sm-feedback-plugin - --parity requires 175 recorded, 172 ported and passing, 3 pending, every TestFonotekaNuxtFlows subtest, the sidecar replay, check_corpus secrets and the docs checks --- scripts/check-phase14.sh | 483 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 483 insertions(+) create mode 100755 scripts/check-phase14.sh diff --git a/scripts/check-phase14.sh b/scripts/check-phase14.sh new file mode 100755 index 0000000..318dda2 --- /dev/null +++ b/scripts/check-phase14.sh @@ -0,0 +1,483 @@ +#!/usr/bin/env bash +# Phase 14 fail-closed gate (domain jobs and external integrations: JOBS-02, +# JOBS-03, SRCH-02, INTG-01, INTG-02, API-08, CLI-05). +# +# Every stage exits non-zero on a failing command, a go test run that fails, +# skips, matches zero tests or prints "no tests to run", a named test that +# did not pass, a data race, a parity count other than the expected one, a +# coverage floor missed, a corpus secret or an evidence gap. --self-test +# proves each detector fails closed on planted inputs. +# +# Framework commands run in summercms.go; application commands run in the +# sibling repository named by PHASE14_APP (default ../fonoteka.go), whose +# workspace also holds the shared plugins sm-user-plugin, sm-golem-plugin +# and sm-feedback-plugin. No stage calls a live vendor (D-15): every +# Discogs, AI and G15Office exchange replays from a recorded sidecar. +set -euo pipefail + +# A colour-forcing shell variable changes the output some framework tests +# compare byte for byte; the gate runs without it. +unset FORCE_COLOR + +ROOT="${PHASE14_ROOT:-$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)}" +APP="${PHASE14_APP:-$(cd "$ROOT/../fonoteka.go" && pwd)}" +PHASE_DIR="${PHASE14_PHASE_DIR:-$ROOT/.planning/phases/14-domain-jobs-and-external-integrations}" +REVIEW="$PHASE_DIR/14-SECURITY-REVIEW.md" +VALIDATION="$PHASE_DIR/14-VALIDATION.md" +APP_PLUGINS=(./plugins/golem15/fonoteka/... ./plugins/golem15/user/... ./plugins/golem15/golem/... ./plugins/golem15/feedback/...) +EXPECTED_ROUTES=175 +EXPECTED_PORTED=172 +EXPECTED_PENDING=3 +COVERAGE_FLOOR=80 +FUZZ_CORPUS="plugins/golem15/fonoteka/testdata/fuzz" +# D-02: the AI adapters are hand-written JSON over fetchguard. A vendor SDK +# in the module graph means the decision was undone (T-14-SC). +VENDOR_SDK_RE='^(github\.com/anthropics/|github\.com/openai/|github\.com/sashabaranov/|github\.com/liushuangls/go-anthropic|github\.com/tmc/langchaingo)' + +usage() { + cat >&2 <<'EOF' +usage: + check-phase14.sh --self-test + check-phase14.sh --go + check-phase14.sh --parity + check-phase14.sh --all +EOF + exit 2 +} + +# phase14_detect reads go test -json. Exit 1 fail, 2 skip, 3 zero tests or +# "no tests to run", 4 non-JSON, 5 a required test did not pass, 6 a data +# race was reported. PHASE14_REQUIRE lists tests that must pass. +phase14_detect() { + python3 - "$1" <<'PY' +import json, os, sys +path = sys.argv[1] +require = set(os.environ.get("PHASE14_REQUIRE", "").split()) +passed = set() +failed_tests, failed_pkgs = {}, [] +build_failed = False +with open(path, encoding="utf-8", errors="replace") as fh: + for raw in fh: + line = raw.strip() + if not line.startswith("{"): + continue + try: + ev = json.loads(line) + except json.JSONDecodeError: + print("refuse: non-json test output", file=sys.stderr) + sys.exit(4) + action = ev.get("Action") + test = ev.get("Test") or "" + pkg = ev.get("Package") or "" + if action == "build-fail": + build_failed = True + if action == "output": + text = ev.get("Output") or "" + if "no tests to run" in text: + print(f"refuse: no tests to run in {pkg}", file=sys.stderr) + sys.exit(3) + if "WARNING: DATA RACE" in text: + print(f"refuse: data race in {pkg} {test}", file=sys.stderr) + sys.exit(6) + if action == "skip" and test: + print(f"refuse: skipped {pkg} {test}", file=sys.stderr) + sys.exit(2) + if action == "fail": + if ev.get("FailedBuild"): + build_failed = True + if test: + failed_tests.setdefault(pkg, []).append(test) + else: + failed_pkgs.append(pkg) + if action == "pass" and test: + passed.add(test) +if build_failed: + print("refuse: build failed", file=sys.stderr) + sys.exit(1) +for pkg, tests in failed_tests.items(): + for test in tests: + print(f"refuse: failed {pkg} {test}", file=sys.stderr) + sys.exit(1) +for pkg in failed_pkgs: + print(f"refuse: failed {pkg or 'unknown package'}", file=sys.stderr) + sys.exit(1) +missing = sorted(name for name in require if name not in passed) +if missing: + print("refuse: required tests did not pass: " + ", ".join(missing), file=sys.stderr) + sys.exit(5) +if not passed: + print("refuse: zero tests", file=sys.stderr) + sys.exit(3) +PY +} + +# phase14_go DIR ARGS... runs go test -json -count=1 ARGS through the +# detector. With PHASE14_KEEP set, the JSON log is copied there. +phase14_go() { + local dir="$1" + shift + local log err + log="$(mktemp)" + err="$(mktemp)" + set +e + (cd "$dir" && go test -json -count=1 "$@") >"$log" 2>"$err" + local rc=$? + set -e + local dc=0 verdict + verdict="$(phase14_detect "$log" 2>&1)" || dc=$? + if [[ "$dc" -ne 0 || "$rc" -ne 0 ]]; then + cat "$err" >&2 || true + tail -n 40 "$log" >&2 || true + # The full log stays for diagnosis; its path is printed. + local kept="${TMPDIR:-/tmp}/phase14-failed-$$.json" + mv "$log" "$kept" 2>/dev/null && echo "phase14: full go test log kept at $kept" >&2 + rm -f "$log" "$err" + # The detector's verdict comes last, after the log tail, so it is + # the line a reader sees. + [[ -n "$verdict" ]] && echo "$verdict" >&2 + echo "refuse: go test $* in $dir (test=$rc detect=$dc)" >&2 + exit 1 + fi + if [[ -n "${PHASE14_KEEP:-}" ]]; then + cp "$log" "$PHASE14_KEEP" + fi + rm -f "$log" "$err" +} + +# phase14_tests DIR PKG [-race] TEST... requires every named test to run and +# pass, each matched by its exact name. +phase14_tests() { + local dir="$1" pkg="$2" + shift 2 + local extra=() + if [[ "${1:-}" == "-race" ]]; then + extra=(-race) + shift + fi + local names="$*" + local regex="^($(tr ' ' '|' <<<"$names"))\$" + PHASE14_REQUIRE="$names" phase14_go "$dir" "$pkg" "${extra[@]}" -run "$regex" +} + +expect_detect() { + local name="$1" want="$2" payload="$3" + local log dc=0 + log="$(mktemp)" + printf '%s\n' "$payload" >"$log" + phase14_detect "$log" 2>/dev/null || dc=$? + rm -f "$log" + if [[ "$dc" -ne "$want" ]]; then + echo "refuse: self-test $name: detector exit $dc, want $want" >&2 + exit 1 + fi +} + +# module_hygiene MODLIST: no vendor AI SDK anywhere in the module graph +# (D-02, T-14-SC). +module_hygiene() { + local modlist="$1" hits + hits="$(grep -E "$VENDOR_SDK_RE" "$modlist" | sort -u || true)" + if [[ -n "$hits" ]]; then + echo "refuse: hygiene: a vendor SDK is in the module graph: $hits" >&2 + return 1 + fi + return 0 +} + +run_go() { + (cd "$ROOT" && go vet ./...) + phase14_go "$ROOT" ./... + (cd "$APP" && go vet ./... "${APP_PLUGINS[@]}") + phase14_go "$APP" -race ./... "${APP_PLUGINS[@]}" + local modlist + modlist="$(mktemp)" + (cd "$ROOT" && go list -m all) >"$modlist" + (cd "$APP" && go list -m all) >>"$modlist" + if ! module_hygiene "$modlist"; then + rm -f "$modlist" + exit 1 + fi + rm -f "$modlist" + echo "phase14 go passed" +} + +# corpus_scan DIR: the fuzz seed corpus holds synthetic values only: no +# 64-hex token, inv_ personal token, JWT, bearer header or vendor key. +corpus_scan() { + python3 - "$1" <<'PY' +import os, re, sys +root = sys.argv[1] +if not os.path.isdir(root): + print(f"refuse: fuzz corpus {root} is missing", file=sys.stderr) + sys.exit(1) +patterns = [ + ("64-hex value", re.compile(r"(?&2 + return 1 + fi + return 0 +} + +# corpus_coverage LOG: the replay's own coverage line in a go test -json +# log ("recorded R/T passing P failing F unrecorded U pending Q") must say +# EXPECTED_ROUTES recorded, EXPECTED_PORTED passing, 0 failing, 0 +# unrecorded and EXPECTED_PENDING pending. Pending routes are never counted +# as passing. +corpus_coverage() { + python3 - "$1" "$EXPECTED_ROUTES" "$EXPECTED_PORTED" "$EXPECTED_PENDING" <<'PY' +import json, re, sys +path, routes, ported, pending = sys.argv[1], int(sys.argv[2]), int(sys.argv[3]), int(sys.argv[4]) +rx = re.compile(r"recorded (\d+)/(\d+) passing (\d+) failing (\d+) unrecorded (\d+) pending (\d+)") +found = None +for raw in open(path, encoding="utf-8", errors="replace"): + raw = raw.strip() + if not raw.startswith("{"): + continue + try: + ev = json.loads(raw) + except json.JSONDecodeError: + continue + m = rx.search(ev.get("Output") or "") + if m: + found = [int(x) for x in m.groups()] +if found is None: + print("refuse: the corpus replay printed no coverage line", file=sys.stderr) + sys.exit(1) +recorded, total, passing, failing, unrecorded, pend = found +if total != routes or recorded != total or passing != ported or failing != 0 or unrecorded != 0 or pend != pending or passing + pend != total: + print(f"refuse: corpus coverage recorded {recorded}/{total} passing {passing} failing {failing} unrecorded {unrecorded} pending {pend}, want {routes}/{routes} recorded, {ported} passing, 0 failing, {pending} pending", file=sys.stderr) + sys.exit(1) +print(f"phase14 corpus: {recorded}/{total} recorded, {passing} ported and passing, 0 failing, {pend} pending") +PY +} + +# flow_subtests FILE TEST: the subtest names of TEST as declared in FILE +# (t.Run("name", ...)), each prefixed with TEST/. Every declared subtest is +# required, so a new flow cannot be added without the gate running it. +flow_subtests() { + local file="$1" test="$2" + grep -oE 't\.Run\("[^"]+"' "$file" | sed -E "s#t\\.Run\\(\"([^\"]+)\"#$test/\\1#" +} + +PARITY_REQUIRE_FIXED="TestParityCorpus TestParityCorpus/coverage TestBroadcastGoldens TestBroadcastGoldens/created TestBroadcastGoldens/updated TestBroadcastGoldens/deleted TestBroadcastGoldens/bulk TestFonotekaNuxtFlows TestUserAPINuxtFlows TestCheckCorpusPortedCaseStatus TestUpstreamSidecarsAreReplayed TestFeedbackJobSidecarMatchesPlugin" + +parity_require() { + local flows + flows="$(flow_subtests "$APP/parity/fonoteka_flows_test.go" TestFonotekaNuxtFlows | tr '\n' ' ')" + if [[ -z "${flows// /}" ]]; then + echo "refuse: TestFonotekaNuxtFlows declares no subtest" >&2 + exit 1 + fi + echo "$PARITY_REQUIRE_FIXED $flows" +} + +run_parity() { + manifest_check "$APP/parity/manifest.yaml" || exit 1 + local log require + require="$(parity_require)" + log="$(mktemp)" + PHASE14_KEEP="$log" PHASE14_REQUIRE="$require" \ + phase14_go "$APP" ./parity -run '^(TestParityCorpus|TestBroadcastGoldens|TestFonotekaNuxtFlows|TestUserAPINuxtFlows|TestCheckCorpusPortedCaseStatus|TestUpstreamSidecarsAreReplayed|TestFeedbackJobSidecarMatchesPlugin)$' + if ! corpus_coverage "$log"; then + rm -f "$log" + exit 1 + fi + rm -f "$log" + (cd "$APP" && go run ./parity/check_corpus.go --manifest parity/manifest.yaml --require-recorded --check-secrets) + corpus_scan "$APP/$FUZZ_CORPUS" + PHASE14_REQUIRE="TestDocsTree" phase14_go "$ROOT" ./cmd/summer -run '^TestDocsTree$' + (cd "$ROOT" && go run ./cmd/summer docs:build --check) + echo "phase14 parity passed ($EXPECTED_ROUTES recorded, $EXPECTED_PORTED ported and passing, 0 failing, $EXPECTED_PENDING pending)" +} + +run_self_test() { + bash -n "${BASH_SOURCE[0]}" + expect_detect pass 0 '{"Action":"pass","Package":"p","Test":"TestPhase14Threats"}' + expect_detect fail 1 '{"Action":"pass","Package":"p","Test":"TestA"} +{"Action":"fail","Package":"p","Test":"TestPhase14Threats/T-14-16"}' + expect_detect skip 2 '{"Action":"skip","Package":"p","Test":"TestPhase14Threats"}' + expect_detect zero 3 '{"Action":"pass","Package":"p"}' + expect_detect no-tests-to-run 3 '{"Action":"pass","Package":"p","Test":"TestA"} +{"Action":"output","Package":"q","Output":"testing: warning: no tests to run\n"}' + expect_detect nonjson 4 '{"Action":"pass",' + expect_detect build 1 '{"Action":"build-fail","ImportPath":"p"} +{"Action":"pass","Package":"q","Test":"TestA"}' + expect_detect build-flag 1 '{"Action":"pass","Package":"q","Test":"TestA"} +{"Action":"fail","Package":"p","FailedBuild":"p"}' + expect_detect package 1 '{"Action":"pass","Package":"p","Test":"TestA"} +{"Action":"fail","Package":"p"}' + expect_detect race 6 '{"Action":"output","Package":"p","Test":"TestA","Output":"WARNING: DATA RACE\n"} +{"Action":"pass","Package":"p","Test":"TestA"}' + PHASE14_REQUIRE="TestRouteTablePhase14 TestPhase14Threats" expect_detect required 5 \ + '{"Action":"pass","Package":"p","Test":"TestRouteTablePhase14"}' + local flag + for flag in $(grep -oE '^ check-phase14\.sh --[a-z-]+' "${BASH_SOURCE[0]}" | awk '{print $2}'); do + grep -q -- "^$flag)" "${BASH_SOURCE[0]}" || { + echo "refuse: usage names $flag but the script has no such mode" >&2 + exit 1 + } + done + if [[ -n "${FORCE_COLOR+x}" ]]; then + echo "refuse: self-test FORCE_COLOR is still set" >&2 + exit 1 + fi + + local scratch + scratch="$(mktemp -d)" + trap 'rm -rf "$scratch"' RETURN + + # Module hygiene refuses a vendor SDK and accepts the framework graph. + printf 'git.golem15.com/golem15/summercms v0.0.0\ngorm.io/gorm v1.31.2\n' >"$scratch/mods" + module_hygiene "$scratch/mods" 2>/dev/null || { + echo "refuse: self-test module_hygiene rejected a clean graph" >&2 + exit 1 + } + local plant + for plant in 'github.com/anthropics/anthropic-sdk-go v1.2.0' 'github.com/openai/openai-go v1.0.0' 'github.com/sashabaranov/go-openai v1.36.0'; do + printf 'gorm.io/gorm v1.31.2\n%s\n' "$plant" >"$scratch/mods" + if module_hygiene "$scratch/mods" 2>/dev/null; then + echo "refuse: self-test module_hygiene accepted $plant" >&2 + exit 1 + fi + done + + # The corpus scan refuses each planted secret shape and accepts + # synthetic values. + mkdir -p "$scratch/corpus" + printf 'go test fuzz v1\nstring("POST /x")\nstring("{\\"status\\":\\"imported\\",\\"pad\\":\\"QQQQ\\"}")\n' >"$scratch/corpus/seed" + corpus_scan "$scratch/corpus" 2>/dev/null || { + echo "refuse: self-test corpus_scan rejected a synthetic seed" >&2 + exit 1 + } + local secret + for secret in "$(printf 'a%.0s' $(seq 64))" "inv_ABCDEFGHIJKLMNOPQRST" "eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxIn0.sig" "Bearer abcdefghijklmnop" "sk-ant-api03-abcdefghijklmnopqrstuv"; do + printf 'string("%s")\n' "$secret" >"$scratch/corpus/planted" + if corpus_scan "$scratch/corpus" 2>/dev/null; then + echo "refuse: self-test corpus_scan accepted a planted secret ${secret:0:12}" >&2 + exit 1 + fi + done + rm -f "$scratch/corpus/planted" "$scratch/corpus/seed" + if corpus_scan "$scratch/corpus" 2>/dev/null; then + echo "refuse: self-test corpus_scan accepted an empty corpus" >&2 + exit 1 + fi + + # The manifest counter reads only status lines, and the manifest check + # refuses any count but the expected ones: a fourth pending route (one + # of the ported routes flipped back) fails --parity. + printf 'routes:\n - id: a\n status: ported\n - id: b\n status: pending\n - id: c\n status: ported\n# status: ported\n' >"$scratch/manifest.yaml" + if [[ "$(manifest_count "$scratch/manifest.yaml" ported)" -ne 2 || "$(manifest_count "$scratch/manifest.yaml" pending)" -ne 1 ]]; then + echo "refuse: self-test manifest_count miscounted" >&2 + exit 1 + fi + plant_manifest() { + local ported="$1" pending="$2" other="${3:-0}" i + { + echo "routes:" + for ((i = 0; i < ported; i++)); do printf ' - id: "r%d"\n status: ported\n' "$i"; done + for ((i = 0; i < pending; i++)); do printf ' - id: "p%d"\n status: pending\n' "$i"; done + for ((i = 0; i < other; i++)); do printf ' - id: "o%d"\n status: skipped\n' "$i"; done + } >"$scratch/manifest.yaml" + } + plant_manifest "$EXPECTED_PORTED" "$EXPECTED_PENDING" + manifest_check "$scratch/manifest.yaml" 2>/dev/null || { + echo "refuse: self-test manifest_check rejected the expected counts" >&2 + exit 1 + } + local counts + for counts in "$((EXPECTED_PORTED - 1)) $((EXPECTED_PENDING + 1)) 0" "$EXPECTED_PORTED $((EXPECTED_PENDING + 1)) 0" "$((EXPECTED_PORTED - 1)) $EXPECTED_PENDING 1" "$((EXPECTED_PORTED + 1)) $((EXPECTED_PENDING - 1)) 0"; do + # shellcheck disable=SC2086 + plant_manifest $counts + if manifest_check "$scratch/manifest.yaml" 2>/dev/null; then + echo "refuse: self-test manifest_check accepted ported/pending/other $counts" >&2 + exit 1 + fi + done + + # The corpus coverage line must show the expected counts. + local total=$EXPECTED_ROUTES + local line="recorded $total/$total passing $EXPECTED_PORTED failing 0 unrecorded 0 pending $EXPECTED_PENDING" + printf '{"Action":"output","Package":"p","Test":"TestParityCorpus/coverage","Output":"%s\\n"}\n' "$line" >"$scratch/cov.json" + corpus_coverage "$scratch/cov.json" >/dev/null 2>&1 || { + echo "refuse: self-test corpus_coverage rejected the expected counts" >&2 + exit 1 + } + local planted + for planted in \ + "recorded $total/$total passing $((EXPECTED_PORTED - 1)) failing 1 unrecorded 0 pending $EXPECTED_PENDING" \ + "recorded $total/$total passing $((EXPECTED_PORTED - 1)) failing 0 unrecorded 0 pending $((EXPECTED_PENDING + 1))" \ + "recorded $((total + 1))/$((total + 1)) passing $EXPECTED_PORTED failing 0 unrecorded 0 pending $((EXPECTED_PENDING + 1))" \ + "recorded $((total - 1))/$total passing $EXPECTED_PORTED failing 0 unrecorded 1 pending $((EXPECTED_PENDING - 1))" \ + "recorded $total/$total passing $((EXPECTED_PORTED + 1)) failing 0 unrecorded 0 pending $((EXPECTED_PENDING - 1))"; do + printf '{"Action":"output","Package":"p","Output":"%s\\n"}\n' "$planted" >"$scratch/cov.json" + if corpus_coverage "$scratch/cov.json" >/dev/null 2>&1; then + echo "refuse: self-test corpus_coverage accepted: $planted" >&2 + exit 1 + fi + done + printf '{"Action":"pass","Package":"p","Test":"TestParityCorpus"}\n' >"$scratch/cov.json" + if corpus_coverage "$scratch/cov.json" >/dev/null 2>&1; then + echo "refuse: self-test corpus_coverage accepted a log without a coverage line" >&2 + exit 1 + fi + + # Every declared flow subtest is required: a run that lacks one fails + # the required-test check. + printf 'func TestFlows(t *testing.T) {\n\tt.Run("nuxt-a", f)\n\tt.Run("mcp-b", f)\n}\n' >"$scratch/flows_test.go" + local subs + subs="$(flow_subtests "$scratch/flows_test.go" TestFlows | tr '\n' ' ')" + if [[ "$subs" != "TestFlows/nuxt-a TestFlows/mcp-b " ]]; then + echo "refuse: self-test flow_subtests read: $subs" >&2 + exit 1 + fi + PHASE14_REQUIRE="TestFlows $subs" expect_detect flow-missing 5 '{"Action":"pass","Package":"p","Test":"TestFlows"} +{"Action":"pass","Package":"p","Test":"TestFlows/nuxt-a"}' + echo "phase14 self-test passed" +} + +case "${1:-}" in +--self-test) run_self_test ;; +--go) run_go ;; +--parity) run_parity ;; +--all) + run_self_test + run_go + run_parity + echo "phase14 all passed" + ;; +*) usage ;; +esac