From e98b41aa15cddafc5e3e22610cc40de4f085e398 Mon Sep 17 00:00:00 2001 From: Tom Boucher Date: Fri, 20 Mar 2026 22:26:22 -0400 Subject: [PATCH] feat: add data-flow tracing, environment audit, and behavioral spot-checks Verification checked structure but not whether data actually flows end-to-end or whether external dependencies are available. Adds: - Step 4b (Data-Flow Trace): Level 4 verification traces upstream from wired artifacts to verify data sources produce real data, catching hollow components that render empty/hardcoded values - Step 7b (Behavioral Spot-Checks): lightweight smoke tests that verify runnable code produces expected output, not just that it exists - Step 2.6 (Environment Audit): researcher probes target machine for external tools/services/runtimes before planning, so plans include fallback strategies for missing dependencies Closes #1245 Co-Authored-By: Claude Opus 4.6 --- agents/gsd-phase-researcher.md | 77 ++++++++++++++++++++ agents/gsd-verifier.md | 117 ++++++++++++++++++++++++++++++- tests/agent-frontmatter.test.cjs | 76 ++++++++++++++++++++ 3 files changed, 269 insertions(+), 1 deletion(-) diff --git a/agents/gsd-phase-researcher.md b/agents/gsd-phase-researcher.md index eb9ffaae1..28a7a00f3 100644 --- a/agents/gsd-phase-researcher.md +++ b/agents/gsd-phase-researcher.md @@ -350,6 +350,20 @@ Verified patterns from official sources: - What's unclear: [the gap] - Recommendation: [how to handle] +## Environment Availability + +> Skip this section if the phase has no external dependencies (code/config-only changes). + +| Dependency | Required By | Available | Version | Fallback | +|------------|------------|-----------|---------|----------| +| [tool] | [feature/requirement] | ✓/✗ | [version or —] | [fallback or —] | + +**Missing dependencies with no fallback:** +- [items that block execution] + +**Missing dependencies with fallback:** +- [items with viable alternatives] + ## Validation Architecture > Skip this section entirely if workflow.nyquist_validation is explicitly set to false in .planning/config.json. If the key is absent, treat as enabled. @@ -469,6 +483,68 @@ For each item found: document (1) what needs changing, and (2) whether it requir If the answer for a category is "nothing" — say so explicitly. Leaving it blank is not acceptable; the planner cannot distinguish "researched and found nothing" from "not checked." +## Step 2.6: Environment Availability Audit + +**Trigger:** Any phase that depends on external tools, services, runtimes, or CLI utilities beyond the project's own code. + +Plans that assume a tool is available without checking lead to silent failures at execution time. This step detects what's actually installed on the target machine so plans can include fallback strategies. + +**How:** + +1. **Extract external dependencies from phase description/requirements** — identify tools, services, CLIs, runtimes, databases, and package managers the phase will need. + +2. **Probe availability** for each dependency: + +```bash +# CLI tools — check if command exists and get version +command -v $TOOL 2>/dev/null && $TOOL --version 2>/dev/null | head -1 + +# Runtimes — check version meets minimum +node --version 2>/dev/null +python3 --version 2>/dev/null +ruby --version 2>/dev/null + +# Package managers +npm --version 2>/dev/null +pip3 --version 2>/dev/null +cargo --version 2>/dev/null + +# Databases / services — check if process is running or port is open +pg_isready 2>/dev/null +redis-cli ping 2>/dev/null +curl -s http://localhost:27017 2>/dev/null + +# Docker +docker info 2>/dev/null | head -3 +``` + +3. **Document in RESEARCH.md** as `## Environment Availability`: + +```markdown +## Environment Availability + +| Dependency | Required By | Available | Version | Fallback | +|------------|------------|-----------|---------|----------| +| PostgreSQL | Data layer | ✓ | 15.4 | — | +| Redis | Caching | ✗ | — | Use in-memory cache | +| Docker | Containerization | ✓ | 24.0.7 | — | +| ffmpeg | Media processing | ✗ | — | Skip media features, flag for human | + +**Missing dependencies with no fallback:** +- {list items that block execution — planner must address these} + +**Missing dependencies with fallback:** +- {list items with viable alternatives — planner should use fallback} +``` + +4. **Classification:** + - **Available:** Tool found, version meets minimum → no action needed + - **Available, wrong version:** Tool found but version too old → document upgrade path + - **Missing with fallback:** Not found, but a viable alternative exists → planner uses fallback + - **Missing, blocking:** Not found, no fallback → planner must address (install step, or descope feature) + +**Skip condition:** If the phase is purely code/config changes with no external dependencies (e.g., refactoring, documentation), output: "Step 2.6: SKIPPED (no external dependencies identified)" and move on. + ## Step 3: Execute Research Protocol For each domain: Context7 first → Official docs → WebSearch → Cross-verify. Document findings with confidence levels as you go. @@ -603,6 +679,7 @@ Research is complete when: - [ ] Architecture patterns documented - [ ] Don't-hand-roll items listed - [ ] Common pitfalls catalogued +- [ ] Environment availability audited (or skipped with reason) - [ ] Code examples provided - [ ] Source hierarchy followed (Context7 → Official → WebSearch) - [ ] All findings have confidence levels diff --git a/agents/gsd-verifier.md b/agents/gsd-verifier.md index 63477f63e..55a494f2d 100644 --- a/agents/gsd-verifier.md +++ b/agents/gsd-verifier.md @@ -200,6 +200,63 @@ grep -r "$artifact_name" "${search_path:-src/}" --include="*.ts" --include="*.ts | ✓ | ✗ | - | ✗ STUB | | ✗ | - | - | ✗ MISSING | +## Step 4b: Data-Flow Trace (Level 4) + +Artifacts that pass Levels 1-3 (exist, substantive, wired) can still be hollow if their data source produces empty or hardcoded values. Level 4 traces upstream from the artifact to verify real data flows through the wiring. + +**When to run:** For each artifact that passes Level 3 (WIRED) and renders dynamic data (components, pages, dashboards — not utilities or configs). + +**How:** + +1. **Identify the data variable** — what state/prop does the artifact render? + +```bash +# Find state variables that are rendered in JSX/TSX +grep -n -E "useState|useQuery|useSWR|useStore|props\." "$artifact" 2>/dev/null +``` + +2. **Trace the data source** — where does that variable get populated? + +```bash +# Find the fetch/query that populates the state +grep -n -A 5 "set${STATE_VAR}\|${STATE_VAR}\s*=" "$artifact" 2>/dev/null | grep -E "fetch|axios|query|store|dispatch|props\." +``` + +3. **Verify the source produces real data** — does the API/store return actual data or static/empty values? + +```bash +# Check the API route or data source for real DB queries vs static returns +grep -n -E "prisma\.|db\.|query\(|findMany|findOne|select|FROM" "$source_file" 2>/dev/null +# Flag: static returns with no query +grep -n -E "return.*json\(\s*\[\]|return.*json\(\s*\{\}" "$source_file" 2>/dev/null +``` + +4. **Check for disconnected props** — props passed to child components that are hardcoded empty at the call site + +```bash +# Find where the component is used and check prop values +grep -r -A 3 "<${COMPONENT_NAME}" "${search_path:-src/}" --include="*.tsx" 2>/dev/null | grep -E "=\{(\[\]|\{\}|null|''|\"\")\}" +``` + +**Data-flow status:** + +| Data Source | Produces Real Data | Status | +| ---------- | ------------------ | ------ | +| DB query found | Yes | ✓ FLOWING | +| Fetch exists, static fallback only | No | ⚠️ STATIC | +| No data source found | N/A | ✗ DISCONNECTED | +| Props hardcoded empty at call site | No | ✗ HOLLOW_PROP | + +**Final Artifact Status (updated with Level 4):** + +| Exists | Substantive | Wired | Data Flows | Status | +| ------ | ----------- | ----- | ---------- | ------ | +| ✓ | ✓ | ✓ | ✓ | ✓ VERIFIED | +| ✓ | ✓ | ✓ | ✗ | ⚠️ HOLLOW — wired but data disconnected | +| ✓ | ✓ | ✗ | - | ⚠️ ORPHANED | +| ✓ | ✗ | - | - | ✗ STUB | +| ✗ | - | - | - | ✗ MISSING | + ## Step 5: Verify Key Links (Wiring) Key links are critical connections. If broken, the goal fails even with all artifacts present. @@ -321,6 +378,52 @@ grep -n -B 2 -A 2 "console\.log" "$file" 2>/dev/null | grep -E "^\s*(const|funct Categorize: 🛑 Blocker (prevents goal) | ⚠️ Warning (incomplete) | ℹ️ Info (notable) +## Step 7b: Behavioral Spot-Checks + +Anti-pattern scanning (Step 7) checks for code smells. Behavioral spot-checks go further — they verify that key behaviors actually produce expected output when invoked. + +**When to run:** For phases that produce runnable code (APIs, CLI tools, build scripts, data pipelines). Skip for documentation-only or config-only phases. + +**How:** + +1. **Identify checkable behaviors** from must-haves truths. Select 2-4 that can be tested with a single command: + +```bash +# API endpoint returns non-empty data +curl -s http://localhost:$PORT/api/$ENDPOINT 2>/dev/null | node -e "const d=JSON.parse(require('fs').readFileSync('/dev/stdin','utf8')); process.exit(Array.isArray(d) ? (d.length > 0 ? 0 : 1) : (Object.keys(d).length > 0 ? 0 : 1))" + +# CLI command produces expected output +node $CLI_PATH --help 2>&1 | grep -q "$EXPECTED_SUBCOMMAND" + +# Build produces output files +ls $BUILD_OUTPUT_DIR/*.{js,css} 2>/dev/null | wc -l + +# Module exports expected functions +node -e "const m = require('$MODULE_PATH'); console.log(typeof m.$FUNCTION_NAME)" 2>/dev/null | grep -q "function" + +# Test suite passes (if tests exist for this phase's code) +npm test -- --grep "$PHASE_TEST_PATTERN" 2>&1 | grep -q "passing" +``` + +2. **Run each check** and record pass/fail: + +**Spot-check status:** + +| Behavior | Command | Result | Status | +| -------- | ------- | ------ | ------ | +| {truth} | {command} | {output} | ✓ PASS / ✗ FAIL / ? SKIP | + +3. **Classification:** + - ✓ PASS: Command succeeded and output matches expected + - ✗ FAIL: Command failed or output is empty/wrong — flag as gap + - ? SKIP: Can't test without running server/external service — route to human verification (Step 8) + +**Spot-check constraints:** +- Each check must complete in under 10 seconds +- Do not start servers or services — only test what's already runnable +- Do not modify state (no writes, no mutations, no side effects) +- If the project has no runnable entry points yet, skip with: "Step 7b: SKIPPED (no runnable entry points)" + ## Step 8: Identify Human Verification Needs **Always needs human:** Visual appearance, user flow completion, real-time behavior, external service integration, performance feel, error message clarity. @@ -438,6 +541,16 @@ human_verification: # Only if status: human_needed | From | To | Via | Status | Details | | ---- | --- | --- | ------ | ------- | +### Data-Flow Trace (Level 4) + +| Artifact | Data Variable | Source | Produces Real Data | Status | +| -------- | ------------- | ------ | ------------------ | ------ | + +### Behavioral Spot-Checks + +| Behavior | Command | Result | Status | +| -------- | ------- | ------ | ------ | + ### Requirements Coverage | Requirement | Source Plan | Description | Status | Evidence | @@ -501,7 +614,7 @@ Automated checks passed. Awaiting human verification. **DO NOT trust SUMMARY claims.** Verify the component actually renders messages, not a placeholder. -**DO NOT assume existence = implementation.** Need level 2 (substantive) and level 3 (wired). +**DO NOT assume existence = implementation.** Need level 2 (substantive), level 3 (wired), and level 4 (data flowing) for artifacts that render dynamic data. **DO NOT skip key link verification.** 80% of stubs hide here — pieces exist but aren't connected. @@ -573,9 +686,11 @@ return
No messages
// Always shows "no messages" - [ ] If initial: must-haves established (from frontmatter or derived) - [ ] All truths verified with status and evidence - [ ] All artifacts checked at all three levels (exists, substantive, wired) +- [ ] Data-flow trace (Level 4) run on wired artifacts that render dynamic data - [ ] All key links verified - [ ] Requirements coverage assessed (if applicable) - [ ] Anti-patterns scanned and categorized +- [ ] Behavioral spot-checks run on runnable code (or skipped with reason) - [ ] Human verification items identified - [ ] Overall status determined - [ ] Gaps structured in YAML frontmatter (if gaps_found) diff --git a/tests/agent-frontmatter.test.cjs b/tests/agent-frontmatter.test.cjs index e1e5de596..f9977c338 100644 --- a/tests/agent-frontmatter.test.cjs +++ b/tests/agent-frontmatter.test.cjs @@ -233,6 +233,82 @@ describe('CLAUDEMD: CLAUDE.md compliance enforcement', () => { }); }); +// ─── Verification Data-Flow and Environment Audit (#1245) ──────────────────── + +describe('VERIFY: data-flow trace, environment audit, and behavioral spot-checks', () => { + test('gsd-verifier has Step 4b: Data-Flow Trace', () => { + const content = fs.readFileSync(path.join(AGENTS_DIR, 'gsd-verifier.md'), 'utf-8'); + assert.ok( + content.includes('Step 4b: Data-Flow Trace'), + 'gsd-verifier must have Step 4b for data-flow tracing' + ); + assert.ok( + content.includes('HOLLOW'), + 'gsd-verifier must define HOLLOW status for wired-but-disconnected artifacts' + ); + assert.ok( + content.includes('DISCONNECTED'), + 'gsd-verifier must define DISCONNECTED status for missing data sources' + ); + }); + + test('gsd-verifier has Step 7b: Behavioral Spot-Checks', () => { + const content = fs.readFileSync(path.join(AGENTS_DIR, 'gsd-verifier.md'), 'utf-8'); + assert.ok( + content.includes('Step 7b: Behavioral Spot-Checks'), + 'gsd-verifier must have Step 7b for behavioral spot-checks' + ); + assert.ok( + content.includes('SKIP'), + 'gsd-verifier spot-checks must support SKIP status for untestable items' + ); + }); + + test('gsd-verifier VERIFICATION.md template includes data-flow and spot-check sections', () => { + const content = fs.readFileSync(path.join(AGENTS_DIR, 'gsd-verifier.md'), 'utf-8'); + assert.ok( + content.includes('Data-Flow Trace (Level 4)'), + 'VERIFICATION.md template must include Data-Flow Trace section' + ); + assert.ok( + content.includes('Behavioral Spot-Checks'), + 'VERIFICATION.md template must include Behavioral Spot-Checks section' + ); + }); + + test('gsd-verifier success criteria include data-flow and spot-checks', () => { + const content = fs.readFileSync(path.join(AGENTS_DIR, 'gsd-verifier.md'), 'utf-8'); + assert.ok( + content.includes('Data-flow trace (Level 4)'), + 'success criteria must include data-flow trace step' + ); + assert.ok( + content.includes('Behavioral spot-checks run'), + 'success criteria must include behavioral spot-checks step' + ); + }); + + test('gsd-phase-researcher has Step 2.6: Environment Availability Audit', () => { + const content = fs.readFileSync(path.join(AGENTS_DIR, 'gsd-phase-researcher.md'), 'utf-8'); + assert.ok( + content.includes('Step 2.6: Environment Availability Audit'), + 'gsd-phase-researcher must have Step 2.6 for environment availability auditing' + ); + assert.ok( + content.includes('Environment Availability'), + 'gsd-phase-researcher must include Environment Availability section in RESEARCH.md template' + ); + }); + + test('gsd-phase-researcher success criteria include environment audit', () => { + const content = fs.readFileSync(path.join(AGENTS_DIR, 'gsd-phase-researcher.md'), 'utf-8'); + assert.ok( + content.includes('Environment availability audited'), + 'success criteria must include environment availability audit step' + ); + }); +}); + // ─── Discussion Log ────────────────────────────────────────────────────────── describe('DISCUSS: discussion log generation', () => {