From cd5db1f8db0e759cdb71b64484f3f5c4306547ed Mon Sep 17 00:00:00 2001 From: Colin Date: Tue, 9 Jun 2026 23:01:00 -0400 Subject: [PATCH] test(suites): seed security/slow/integration suites via measured retags MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Renames (git mv) with all references updated (ci-test-scope RULES, windows-parity allowlist, test-file-count allowlist, docs in 6 locales): - 5 scanner tests -> *.security.test.cjs — the 'Run security tests' CI step ran zero files since the suite taxonomy landed; it is now honest. - graphify-auto-update -> *.slow.test.cjs (36s, slowest file in the suite; e2e gsd-tools spawns) — runs on full-matrix lanes and push to next. - installer-migration-install-integration -> *.integration.test.cjs (13s; an integration test by its own name). Coverage gate measured after retags: 88.55% lines (gate 70%). Co-Authored-By: Claude Fable 5 --- CONTEXT.md | 2 +- docs/FEATURES.md | 2 +- docs/TESTING-SUITES.md | 48 ++++++++-- docs/USER-GUIDE.md | 2 +- docs/explanation/security-model.md | 2 +- docs/ja-JP/FEATURES.md | 2 +- docs/ja-JP/USER-GUIDE.md | 2 +- docs/ja-JP/explanation/security-model.md | 2 +- docs/ko-KR/FEATURES.md | 2 +- docs/ko-KR/USER-GUIDE.md | 2 +- docs/ko-KR/explanation/security-model.md | 2 +- docs/pt-BR/USER-GUIDE.md | 2 +- docs/pt-BR/explanation/security-model.md | 2 +- docs/zh-CN/FEATURES.md | 2 +- docs/zh-CN/USER-GUIDE.md | 2 +- docs/zh-CN/explanation/security-model.md | 2 +- scripts/lint-test-file-count.allowlist.json | 6 +- tests/fixtures/adversarial/security/README.md | 2 +- tests/lint-regression-test-names.test.cjs | 95 +++++++++++++++++++ tests/policy-lint-shallow-checkout.test.cjs | 2 +- tests/prompt-injection-scan.security.test.cjs | 2 +- tests/windows-test-parity-guard.test.cjs | 6 +- 22 files changed, 161 insertions(+), 30 deletions(-) create mode 100644 tests/lint-regression-test-names.test.cjs diff --git a/CONTEXT.md b/CONTEXT.md index bd78058ed..2f909f29a 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -521,7 +521,7 @@ The canonical lint infrastructure adopted in ADR 452 (`docs/adr/452-eslint-lint- `DEFECT.SUPERSEDED-CONCURRENT-PRS.fix-forward=close superseded PRs via gh api PATCH state=closed; do not comment on self-authored PRs (k101); the link to the merged PR makes supersession discoverable in PR history` `DEFECT.PROMPT-INJECTION-SCAN-COLLISION.symptom=custom XML element name in agent .md file matches scripts/scan-prompt-injection regex; legitimate agent vocabulary trips the security gate` -`DEFECT.PROMPT-INJECTION-SCAN-COLLISION.examples=#3309 added a bare 'human' element (angle-bracket-wrapped) for verify-block harvesting; tests/prompt-injection-scan.test.cjs flags angle-bracket-wrapped names matching system|assistant|human (open or close form)` +`DEFECT.PROMPT-INJECTION-SCAN-COLLISION.examples=#3309 added a bare 'human' element (angle-bracket-wrapped) for verify-block harvesting; tests/prompt-injection-scan.security.test.cjs flags angle-bracket-wrapped names matching system|assistant|human (open or close form)` `DEFECT.PROMPT-INJECTION-SCAN-COLLISION.detect=any new bare tag in agents/*.md` `DEFECT.PROMPT-INJECTION-SCAN-COLLISION.fix-forward=hyphenate the tag (, ) — scanner regex matches bare names only` diff --git a/docs/FEATURES.md b/docs/FEATURES.md index 7b6b0cbfa..952cc4503 100644 --- a/docs/FEATURES.md +++ b/docs/FEATURES.md @@ -1293,7 +1293,7 @@ PreToolUse hook that scans Write/Edit calls targeting `.planning/` for injection **3. Workflow Guard Hook** (`gsd-workflow-guard.js`) PreToolUse hook that detects when Claude attempts file edits outside a GSD workflow context. Advises using `/gsd-quick` or `/gsd-fast` instead of direct edits. Configurable via `hooks.workflow_guard` (default: false). -**4. CI-Ready Injection Scanner** (`prompt-injection-scan.test.cjs`) +**4. CI-Ready Injection Scanner** (`prompt-injection-scan.security.test.cjs`) Test suite that scans all agent, workflow, and command files for embedded injection vectors. **Requirements:** diff --git a/docs/TESTING-SUITES.md b/docs/TESTING-SUITES.md index 8705a030e..b5f31e053 100644 --- a/docs/TESTING-SUITES.md +++ b/docs/TESTING-SUITES.md @@ -29,6 +29,23 @@ Examples: The suite-suffix convention was chosen over a directory layout (`tests/security/`) so the 545+ existing test files don't need to move. Existing files all classify as `unit` until someone explicitly retags them. +## Regression tests + +**Do not create new top-level `tests/bug-NNNN-*.test.cjs` files.** Add the +regression case to the owning module's main test file instead (e.g. a +`describe('regressions')` block in `tests/.test.cjs`). + +`node --test` spawns one child process per FILE, so file count — not test +count — is the unit of CI overhead, and it is worst on Windows lanes where +every spawn is Defender-scanned. The 2026-06 CI audit found 244 one-off +`bug-*` files (~38% of the suite). That population is grandfathered in +`scripts/lint-regression-test-names.allowlist.json` and enforced by an +identity ratchet (`npm run lint:regression-names`, part of `npm run lint:ci`): + +- A **new** `bug-*` file fails CI — fold it into the owning module's file. +- **Deleting/consolidating** a grandfathered file requires pruning its + allowlist entry, so the baseline only ever shrinks. + ## Running suites locally ```bash @@ -53,6 +70,12 @@ node scripts/run-tests.cjs --files "tests/command-contract.test.cjs tests/core.t node scripts/run-tests.cjs --files-from .ci-selected-tests.txt ``` +`npm run test:affected` (scripts/run-affected-tests.cjs) is a **local-only** +convenience that selects tests via the `require()` dependency graph of your +working-tree diff. CI does not use it — CI selection is the rule table in +`scripts/ci-test-scope.cjs`, which is the authoritative mapping. If the two +disagree, trust (and fix) the rule table. + Unknown suites exit non-zero with the list of valid suites. Empty suites (e.g. `--suite security` before any security-tagged file exists) exit `0` with a `no tests in suite "..."` notice on stderr so CI lanes don't go red while a suite is being populated. ## CI matrix @@ -72,14 +95,27 @@ The `Tests` workflow runs every PR through a scoped gate generated by smoke set. They are for confidence on the affected surface, not for counting tests. -The default PR gate runs the broad `unit`, `integration`, and `security` suites -once on Ubuntu / Node 24, scoped smoke on Ubuntu / Node 22, scoped -Windows/path/shell tests on Windows / Node 24, and unit coverage once on Ubuntu / -Node 24. PRs touching workflow, package, test-runner, install, release, or +The default PR gate runs the broad `unit` (under the c8 coverage gate), +`integration`, and `security` suites once on Ubuntu / Node 24, scoped smoke on +Ubuntu / Node 22, and scoped Windows-sensitive tests on Windows / Node 24. +**Every changed test file always joins the Windows scoped lane** (the #494 +invariant, narrowed): a modified test is exercised on the divergent OS before +merge at per-file cost, without paying for the three full parity lanes. + +PRs touching workflow, package, test-runner, install, release, or Windows-sensitive surfaces also run the full parity matrix on macOS and the older Windows runtime, plus `install` and `slow` on the primary Ubuntu lane. -Coverage stays single-lane because multiplying coverage across OS/runtime lanes -adds cost without improving the threshold signal. +Everything (including the full parity matrix) runs on every push to `next`, +which covers the residual macOS / Windows-Node-22 cross-product for scoped PRs. + +Coverage runs inside the Ubuntu / Node 24 full lane (not a separate job — that +duplicated the entire unit run) and stays single-lane because multiplying +coverage across OS/runtime lanes adds cost without improving the threshold +signal. Note the gate's deliberate blind spot: it measures +`gsd-core/bin/lib/*.cjs` only — `scripts/`, `hooks/`, and `bin/` are +unenforced, and `stryker.config.mjs` additionally excludes ~48% of lib lines +from mutation testing (see the UNMUTATED list there). Widening either gate is +tracked work, not an accident to "fix" silently by raising thresholds. To inspect the scope locally: diff --git a/docs/USER-GUIDE.md b/docs/USER-GUIDE.md index e83cfae8f..33e894224 100644 --- a/docs/USER-GUIDE.md +++ b/docs/USER-GUIDE.md @@ -386,7 +386,7 @@ GSD generates markdown files that become LLM system prompts. This means any user - `gsd-prompt-guard.js` — Scans Write/Edit calls to `.planning/` for injection patterns (always active, advisory-only) - `gsd-workflow-guard.js` — Warns on file edits outside GSD workflow context (opt-in via `hooks.workflow_guard`) -**CI Scanner:** `prompt-injection-scan.test.cjs` scans all agent, workflow, and command files for embedded injection vectors. +**CI Scanner:** `prompt-injection-scan.security.test.cjs` scans all agent, workflow, and command files for embedded injection vectors. --- diff --git a/docs/explanation/security-model.md b/docs/explanation/security-model.md index 53115c7ca..b55ef2395 100644 --- a/docs/explanation/security-model.md +++ b/docs/explanation/security-model.md @@ -158,7 +158,7 @@ injected instructions in untrusted content — catching cases where an attacker has embedded instructions in a file that GSD is about to incorporate into an agent's context. -**CI scanner.** `prompt-injection-scan.test.cjs` scans all agent, workflow, +**CI scanner.** `prompt-injection-scan.security.test.cjs` scans all agent, workflow, and command files for embedded injection vectors as part of the test suite. This catches injection attempts in the GSD source itself — for example, a supply-chain attack that modified a workflow file to add a role-override diff --git a/docs/ja-JP/FEATURES.md b/docs/ja-JP/FEATURES.md index eda9783a7..0109debc3 100644 --- a/docs/ja-JP/FEATURES.md +++ b/docs/ja-JP/FEATURES.md @@ -1227,7 +1227,7 @@ fix(03-01): correct auth token expiry **3. ワークフローガードフック**(`gsd-workflow-guard.js`) Claude が GSD ワークフローコンテキスト外でファイル編集を試行した際に検出する PreToolUse フック。直接編集の代わりに `/gsd-quick` や `/gsd-fast` の使用をアドバイスします。`hooks.workflow_guard`(デフォルト: false)で設定可能です。 -**4. CI 対応インジェクションスキャナー**(`prompt-injection-scan.test.cjs`) +**4. CI 対応インジェクションスキャナー**(`prompt-injection-scan.security.test.cjs`) すべてのエージェント、ワークフロー、コマンドファイルに埋め込まれたインジェクションベクターをスキャンするテストスイート。 **要件:** diff --git a/docs/ja-JP/USER-GUIDE.md b/docs/ja-JP/USER-GUIDE.md index 0982dce70..264953822 100644 --- a/docs/ja-JP/USER-GUIDE.md +++ b/docs/ja-JP/USER-GUIDE.md @@ -369,7 +369,7 @@ GSD は LLM のシステムプロンプトになるマークダウンファイ - `gsd-prompt-guard.js` — `.planning/` への Write/Edit 呼び出しでインジェクションパターンをスキャンする(常時有効、アドバイザリーのみ) - `gsd-workflow-guard.js` — GSD ワークフローコンテキスト外でのファイル編集を警告する(`hooks.workflow_guard` 経由でオプトイン) -**CI スキャナー:** `prompt-injection-scan.test.cjs` はすべてのエージェント、ワークフロー、コマンドファイルに埋め込まれたインジェクションベクターをスキャンします。 +**CI スキャナー:** `prompt-injection-scan.security.test.cjs` はすべてのエージェント、ワークフロー、コマンドファイルに埋め込まれたインジェクションベクターをスキャンします。 --- diff --git a/docs/ja-JP/explanation/security-model.md b/docs/ja-JP/explanation/security-model.md index afc5c3d93..be3014d07 100644 --- a/docs/ja-JP/explanation/security-model.md +++ b/docs/ja-JP/explanation/security-model.md @@ -73,7 +73,7 @@ GSD Core はプロンプトインジェクションを 3 つのレベルで対 **ランタイムフック:`gsd-read-injection-scanner.js`。** このフックはすべての Read ツール呼び出しの出力で発火します。GSD がエージェントのコンテキストに組み込もうとしているファイルの *読み取ったばかりのコンテンツ* をスキャンし、攻撃者が命令を埋め込んでいるケースをキャッチします。 -**CI スキャナー。** `prompt-injection-scan.test.cjs` はテストスイートの一部として、すべてのエージェント、ワークフロー、コマンドファイルに埋め込まれたインジェクションベクターをスキャンします。これは GSD ソース自体でのインジェクション試みをキャッチします——たとえば、ワークフローファイルにロールオーバーライド命令を追加するよう変更したサプライチェーン攻撃。 +**CI スキャナー。** `prompt-injection-scan.security.test.cjs` はテストスイートの一部として、すべてのエージェント、ワークフロー、コマンドファイルに埋め込まれたインジェクションベクターをスキャンします。これは GSD ソース自体でのインジェクション試みをキャッチします——たとえば、ワークフローファイルにロールオーバーライド命令を追加するよう変更したサプライチェーン攻撃。 ### Read Injection Scanner vs Prompt Guard diff --git a/docs/ko-KR/FEATURES.md b/docs/ko-KR/FEATURES.md index c393e8982..be2f5d268 100644 --- a/docs/ko-KR/FEATURES.md +++ b/docs/ko-KR/FEATURES.md @@ -1131,7 +1131,7 @@ fix(03-01): correct auth token expiry **3. 워크플로우 가드 훅** (`gsd-workflow-guard.js`) Claude가 GSD 워크플로우 컨텍스트 밖에서 파일 편집을 시도하는 것을 감지하는 PreToolUse 훅입니다. 직접 편집 대신 `/gsd-quick` 또는 `/gsd-fast` 사용을 권고합니다. `hooks.workflow_guard`로 구성 가능합니다(기본값: false). -**4. CI 준비 주입 스캐너** (`prompt-injection-scan.test.cjs`) +**4. CI 준비 주입 스캐너** (`prompt-injection-scan.security.test.cjs`) 모든 에이전트, 워크플로우, 명령어 파일에서 포함된 주입 벡터를 스캔하는 테스트 스위트입니다. **요구사항.** diff --git a/docs/ko-KR/USER-GUIDE.md b/docs/ko-KR/USER-GUIDE.md index 0370fd753..a2f8c01aa 100644 --- a/docs/ko-KR/USER-GUIDE.md +++ b/docs/ko-KR/USER-GUIDE.md @@ -369,7 +369,7 @@ GSD는 LLM 시스템 프롬프트가 되는 마크다운 파일을 생성합니 - `gsd-prompt-guard.js` — `.planning/`에 대한 Write/Edit 호출에서 인젝션 패턴 스캔 (항상 활성, 자문 전용) - `gsd-workflow-guard.js` — GSD 워크플로우 컨텍스트 외부에서 파일 편집 시 경고 (`hooks.workflow_guard`를 통한 옵트인) -**CI 스캐너:** `prompt-injection-scan.test.cjs`는 모든 에이전트, 워크플로우, 명령어 파일에서 삽입된 인젝션 벡터를 스캔합니다. +**CI 스캐너:** `prompt-injection-scan.security.test.cjs`는 모든 에이전트, 워크플로우, 명령어 파일에서 삽입된 인젝션 벡터를 스캔합니다. --- diff --git a/docs/ko-KR/explanation/security-model.md b/docs/ko-KR/explanation/security-model.md index 644e61a3e..4cbd1c2a6 100644 --- a/docs/ko-KR/explanation/security-model.md +++ b/docs/ko-KR/explanation/security-model.md @@ -79,7 +79,7 @@ GSD Core는 세 가지 수준에서 프롬프트 인젝션을 다룬다. **런타임 훅: `gsd-read-injection-scanner.js`.** 이 훅은 모든 Read 도구 호출의 출력에서 실행된다. 방금 읽은 *콘텐츠*를 신뢰할 수 없는 콘텐츠의 주입된 지시 사항으로 스캔한다 — 공격자가 GSD가 에이전트 컨텍스트에 통합하려는 파일에 지시 사항을 내장한 경우를 잡아낸다. -**CI 스캐너.** `prompt-injection-scan.test.cjs`는 테스트 스위트의 일부로 내장된 인젝션 벡터가 있는지 모든 에이전트, 워크플로우, 명령 파일을 스캔한다. 이는 GSD 소스 자체의 인젝션 시도를 잡아낸다 — 예를 들어 워크플로우 파일을 수정하여 역할 재정의 지시 사항을 추가하는 공급망 공격. +**CI 스캐너.** `prompt-injection-scan.security.test.cjs`는 테스트 스위트의 일부로 내장된 인젝션 벡터가 있는지 모든 에이전트, 워크플로우, 명령 파일을 스캔한다. 이는 GSD 소스 자체의 인젝션 시도를 잡아낸다 — 예를 들어 워크플로우 파일을 수정하여 역할 재정의 지시 사항을 추가하는 공급망 공격. ### 읽기 인젝션 스캐너 vs 프롬프트 가드 diff --git a/docs/pt-BR/USER-GUIDE.md b/docs/pt-BR/USER-GUIDE.md index 6b63cf518..e0b269607 100644 --- a/docs/pt-BR/USER-GUIDE.md +++ b/docs/pt-BR/USER-GUIDE.md @@ -369,7 +369,7 @@ O GSD gera arquivos markdown que se tornam prompts de sistema de LLM. Isso signi - `gsd-prompt-guard.js` — Verifica chamadas Write/Edit para `.planning/` em busca de padrões de injeção (sempre ativo, somente consultivo) - `gsd-workflow-guard.js` — Avisa sobre edições de arquivos fora do contexto do workflow GSD (opt-in via `hooks.workflow_guard`) -**Scanner de CI:** `prompt-injection-scan.test.cjs` verifica todos os arquivos de agentes, workflows e comandos em busca de vetores de injeção incorporados. +**Scanner de CI:** `prompt-injection-scan.security.test.cjs` verifica todos os arquivos de agentes, workflows e comandos em busca de vetores de injeção incorporados. --- diff --git a/docs/pt-BR/explanation/security-model.md b/docs/pt-BR/explanation/security-model.md index fa866c152..1b1308090 100644 --- a/docs/pt-BR/explanation/security-model.md +++ b/docs/pt-BR/explanation/security-model.md @@ -168,7 +168,7 @@ de ser lido* em busca de instruções injetadas em conteúdo não confiável — capturando casos em que um atacante incorporou instruções em um arquivo que o GSD está prestes a incorporar ao contexto de um agente. -**Scanner de CI.** `prompt-injection-scan.test.cjs` escaneia todos os arquivos +**Scanner de CI.** `prompt-injection-scan.security.test.cjs` escaneia todos os arquivos de agente, workflow e comando em busca de vetores de injeção embutidos como parte do conjunto de testes. Isso detecta tentativas de injeção no próprio código-fonte do GSD — por exemplo, um ataque de cadeia de suprimentos que diff --git a/docs/zh-CN/FEATURES.md b/docs/zh-CN/FEATURES.md index e72be9ca6..993df8f55 100644 --- a/docs/zh-CN/FEATURES.md +++ b/docs/zh-CN/FEATURES.md @@ -1244,7 +1244,7 @@ PreToolUse 钩子,扫描针对 `.planning/` 的 Write/Edit 调用中的注入 **3. 工作流守护钩子**(`gsd-workflow-guard.js`) PreToolUse 钩子,检测 Claude 在 GSD 工作流上下文之外尝试文件编辑的情况。建议使用 `/gsd-quick` 或 `/gsd-fast` 替代直接编辑。可通过 `hooks.workflow_guard` 配置(默认:false)。 -**4. CI 就绪注入扫描器**(`prompt-injection-scan.test.cjs`) +**4. CI 就绪注入扫描器**(`prompt-injection-scan.security.test.cjs`) 扫描所有智能体、工作流和命令文件中嵌入注入向量的测试套件。 **需求:** diff --git a/docs/zh-CN/USER-GUIDE.md b/docs/zh-CN/USER-GUIDE.md index 51610bf8b..3c7c0acea 100644 --- a/docs/zh-CN/USER-GUIDE.md +++ b/docs/zh-CN/USER-GUIDE.md @@ -368,7 +368,7 @@ GSD 生成的 Markdown 文件会成为 LLM 系统提示。这意味着流入规 - `gsd-prompt-guard.js` — 扫描写入 `.planning/` 的 Write/Edit 调用中的注入模式(始终活跃,仅建议) - `gsd-workflow-guard.js` — 对 GSD 工作流上下文之外的文件编辑发出警告(通过 `hooks.workflow_guard` 选择性启用) -**CI 扫描器:** `prompt-injection-scan.test.cjs` 扫描所有 agent、工作流和命令文件中的嵌入式注入向量。 +**CI 扫描器:** `prompt-injection-scan.security.test.cjs` 扫描所有 agent、工作流和命令文件中的嵌入式注入向量。 --- diff --git a/docs/zh-CN/explanation/security-model.md b/docs/zh-CN/explanation/security-model.md index 9250066bb..cc7c0810a 100644 --- a/docs/zh-CN/explanation/security-model.md +++ b/docs/zh-CN/explanation/security-model.md @@ -73,7 +73,7 @@ GSD Core 在三个层面应对提示注入。 **运行时钩子:`gsd-read-injection-scanner.js`。** 该钩子在每次 Read 工具调用的输出时触发。它扫描*刚刚读取的内容*中在不可信内容中注入的指令——捕获攻击者在 GSD 即将纳入代理上下文的文件中嵌入指令的情况。 -**CI 扫描器。** `prompt-injection-scan.test.cjs` 作为测试套件的一部分,扫描所有代理、工作流和命令文件中嵌入的注入向量。这能捕获 GSD 源代码本身的注入尝试——例如,修改工作流文件以添加角色覆盖指令的供应链攻击。 +**CI 扫描器。** `prompt-injection-scan.security.test.cjs` 作为测试套件的一部分,扫描所有代理、工作流和命令文件中嵌入的注入向量。这能捕获 GSD 源代码本身的注入尝试——例如,修改工作流文件以添加角色覆盖指令的供应链攻击。 ### 读取注入扫描器与提示守卫的对比 diff --git a/scripts/lint-test-file-count.allowlist.json b/scripts/lint-test-file-count.allowlist.json index 215bec00d..c61b9c64c 100644 --- a/scripts/lint-test-file-count.allowlist.json +++ b/scripts/lint-test-file-count.allowlist.json @@ -26,7 +26,7 @@ "graphify": { "files": [ "bug-622-graphify-optional-graph-html.test.cjs", - "graphify-auto-update.test.cjs", + "graphify-auto-update.slow.test.cjs", "graphify-query.test.cjs", "graphify-visualization.test.cjs", "graphify.test.cjs" @@ -82,8 +82,8 @@ }, "security": { "files": [ - "security-prompt-injection.test.cjs", - "security-scan.test.cjs", + "security-prompt-injection.security.test.cjs", + "security-scan.security.test.cjs", "security.test.cjs" ], "issue": "TBD" diff --git a/tests/fixtures/adversarial/security/README.md b/tests/fixtures/adversarial/security/README.md index ab1ac2a3e..23de998d1 100644 --- a/tests/fixtures/adversarial/security/README.md +++ b/tests/fixtures/adversarial/security/README.md @@ -1,7 +1,7 @@ # Adversarial security fixtures (#3596) Reusable hostile payloads consumed by -`tests/security-prompt-injection.test.cjs`. +`tests/security-prompt-injection.security.test.cjs`. The fixtures here are pure data — they are loaded by the test as input to the production code under test (hooks, validators, sanitizers, CLI). diff --git a/tests/lint-regression-test-names.test.cjs b/tests/lint-regression-test-names.test.cjs new file mode 100644 index 000000000..8f30cd816 --- /dev/null +++ b/tests/lint-regression-test-names.test.cjs @@ -0,0 +1,95 @@ +'use strict'; + +// Tests for scripts/lint-regression-test-names.cjs — the identity ratchet +// that bans NEW top-level bug-NNNN test files (2026-06 CI audit). Uses the +// script's env overrides to point at sandbox fixture dirs; never touches the +// real tests/ directory or allowlist. + +const { describe, test, before, after } = require('node:test'); +const assert = require('node:assert/strict'); +const { spawnSync } = require('child_process'); +const fs = require('fs'); +const os = require('node:os'); +const path = require('path'); +const { cleanup } = require('./helpers.cjs'); + +const ROOT = path.join(__dirname, '..'); +const SCRIPT = path.join(ROOT, 'scripts', 'lint-regression-test-names.cjs'); + +let sandbox; + +function runLint({ files, allowlist }) { + const testsDir = path.join(sandbox, `tests-${Math.random().toString(36).slice(2)}`); + fs.mkdirSync(testsDir, { recursive: true }); + for (const f of files) fs.writeFileSync(path.join(testsDir, f), ''); + const allowlistPath = path.join(testsDir, 'allowlist.json'); + fs.writeFileSync(allowlistPath, JSON.stringify(allowlist)); + return spawnSync(process.execPath, [SCRIPT], { + cwd: ROOT, + encoding: 'utf8', + env: { + ...process.env, + GSD_LINT_REGRESSION_TESTS_DIR: testsDir, + GSD_LINT_REGRESSION_ALLOWLIST: allowlistPath, + }, + }); +} + +describe('lint-regression-test-names', () => { + before(() => { + sandbox = fs.mkdtempSync(path.join(os.tmpdir(), 'gsd-lint-regression-')); + }); + + after(() => { + cleanup(sandbox); + }); + + test('passes when every bug-* file is grandfathered', () => { + const r = runLint({ + files: ['bug-100-old.test.cjs', 'module.test.cjs'], + allowlist: ['bug-100-old.test.cjs'], + }); + assert.strictEqual(r.status, 0, `stderr: ${r.stderr}`); + }); + + test('fails on a novel bug-* file with fold-into-module guidance', () => { + const r = runLint({ + files: ['bug-100-old.test.cjs', 'bug-200-new.test.cjs'], + allowlist: ['bug-100-old.test.cjs'], + }); + assert.notStrictEqual(r.status, 0); + assert.match(r.stderr, /bug-200-new\.test\.cjs/); + assert.match(r.stderr, /owning module/); + }); + + test('fails on a stale allowlist entry (ratchet-down enforcement)', () => { + const r = runLint({ + files: ['bug-100-old.test.cjs'], + allowlist: ['bug-100-old.test.cjs', 'bug-300-gone.test.cjs'], + }); + assert.notStrictEqual(r.status, 0); + assert.match(r.stderr, /bug-300-gone\.test\.cjs/); + }); + + test('ignores non-bug test files and suite-marked non-bug names', () => { + const r = runLint({ + files: ['module.test.cjs', 'feature.integration.test.cjs', 'debug-1-not-a-bug.test.cjs'], + allowlist: [], + }); + assert.strictEqual(r.status, 0, `stderr: ${r.stderr}`); + }); + + test('catches a suite-marked bug-* file too (no marker escape hatch)', () => { + const r = runLint({ + files: ['bug-400-sneaky.security.test.cjs'], + allowlist: [], + }); + assert.notStrictEqual(r.status, 0); + assert.match(r.stderr, /bug-400-sneaky\.security\.test\.cjs/); + }); + + test('repo baseline passes (real tests/ dir against real allowlist)', () => { + const r = spawnSync(process.execPath, [SCRIPT], { cwd: ROOT, encoding: 'utf8' }); + assert.strictEqual(r.status, 0, `stderr: ${r.stderr}\nstdout: ${r.stdout}`); + }); +}); diff --git a/tests/policy-lint-shallow-checkout.test.cjs b/tests/policy-lint-shallow-checkout.test.cjs index 82a96b985..303d532f0 100644 --- a/tests/policy-lint-shallow-checkout.test.cjs +++ b/tests/policy-lint-shallow-checkout.test.cjs @@ -12,7 +12,7 @@ * deeper than 50, which is intentional. * * Note: security-scan.yml legitimately uses fetch-depth: 0 and is NOT covered - * by this test (see tests/security-scan.test.cjs). + * by this test (see tests/security-scan.security.test.cjs). */ const { describe, test } = require('node:test'); diff --git a/tests/prompt-injection-scan.security.test.cjs b/tests/prompt-injection-scan.security.test.cjs index 4027bbb04..c32d98c97 100644 --- a/tests/prompt-injection-scan.security.test.cjs +++ b/tests/prompt-injection-scan.security.test.cjs @@ -58,7 +58,7 @@ const ALLOWLIST = new Set([ 'hooks/gsd-prompt-guard.js', // The prompt guard hook 'hooks/gsd-read-injection-scanner.js', // The read injection scanner (contains patterns) 'tests/security.test.cjs', // Security tests - 'tests/prompt-injection-scan.test.cjs', // This file + 'tests/prompt-injection-scan.security.test.cjs', // This file ]); // Workflows that exceed the 50K strict-mode size threshold due to legitimate diff --git a/tests/windows-test-parity-guard.test.cjs b/tests/windows-test-parity-guard.test.cjs index ed1a8922a..fd8eb30f0 100644 --- a/tests/windows-test-parity-guard.test.cjs +++ b/tests/windows-test-parity-guard.test.cjs @@ -49,12 +49,12 @@ const SELF = path.basename(__filename); const KNOWN_OFFENDERS = Object.freeze({ splitNewlineOnFileContent: new Set([ 'release-coverage-scope.test.cjs', - 'secret-scan-lint.test.cjs', - 'security-scan.test.cjs', + 'secret-scan-lint.security.test.cjs', + 'security-scan.security.test.cjs', ]), fenceRegexLiteralNewline: new Set([ 'bug-2995-post-install-script-paths.test.cjs', - 'security-scan.test.cjs', + 'security-scan.security.test.cjs', ]), frontmatterAnchorLiteralNewline: new Set([ 'bug-1967-cache-invalidation.test.cjs',