diff --git a/.changeset/sharp-quails-climb.md b/.changeset/sharp-quails-climb.md new file mode 100644 index 000000000..a2d7a689e --- /dev/null +++ b/.changeset/sharp-quails-climb.md @@ -0,0 +1,5 @@ +--- +type: Fixed +pr: 2698 +--- +**The host-integration capability matrix now documents the `effortSurface` axis for every runtime** — the axis shipped in #2481 with real values in 19 runtime descriptors, but the matrix that ADR-1239 designates its cited source of truth had no legend entry and not one per-runtime row, so every committed value was undocumented in the one place meant to explain it. (#2615) diff --git a/CONTEXT.md b/CONTEXT.md index 4c0145fed..320adc462 100644 --- a/CONTEXT.md +++ b/CONTEXT.md @@ -218,7 +218,7 @@ Runtime seam (`gsd-core/bin/lib/capability-loader.cjs`, ADR-1244 D2) that compos Human-facing discoverability catalog (`docs/registries/capability-registry.md`, generated from `docs/registries/capabilities.json`; issue #2182) listing third-party Feature Capabilities registered by a docs PR so a solo developer can find one before installing it. Distinct from **Capability Registry** (the generated runtime manifest compiled from first-party `capability.json` declarations, ADR-894) and **Capability Registry Overlay** (the runtime seam that merges an installed third-party manifest into that generated registry at load time, ADR-1244 D2): this registry is a static document rendered by `scripts/gen-registry.cjs`, not a runtime data structure or loader. Each entry enumerates the capability's Loop Extension Points and hook kinds so a reader can judge blast radius before running `gsd capability install`, and declares its `engines.gsd` range. Inclusion is an explicit non-endorsement — a maintainer merged a link, nothing more — per `docs/registries/README.md`. ### EoS Registry -Human-facing discoverability catalog (`docs/registries/eos-registry.md`, generated from `docs/registries/eos.json`; issue #2182) listing third-party Embeddable Orchestration System (EoS) host integrations — projects that embed GSD as an orchestration engine behind the ADR-1239 six-interface-point Host-Integration Interface. Entries are registered by the same docs-PR process, schema conventions, and non-endorsement stance as the **Community Capability Registry**, but enumerate the six interface points, the eight negotiated axes, and `protocolVersion` in place of Loop Extension Points and hook kinds. It has no generated-manifest or Capability Registry Overlay counterpart: an ADR-1239 host integration runs inside the third-party host, not inside GSD's own capability loader, so there is nothing for a runtime registry to merge. See `docs/registries/README.md` for the full entry schema. +Human-facing discoverability catalog (`docs/registries/eos-registry.md`, generated from `docs/registries/eos.json`; issue #2182) listing third-party Embeddable Orchestration System (EoS) host integrations — projects that embed GSD as an orchestration engine behind the ADR-1239 six-interface-point Host-Integration Interface. Entries are registered by the same docs-PR process, schema conventions, and non-endorsement stance as the **Community Capability Registry**, but enumerate the six interface points, the nine negotiated axes, and `protocolVersion` in place of Loop Extension Points and hook kinds. It has no generated-manifest or Capability Registry Overlay counterpart: an ADR-1239 host integration runs inside the third-party host, not inside GSD's own capability loader, so there is nothing for a runtime registry to merge. See `docs/registries/README.md` for the full entry schema. ### Capability Validator Shared conformance validator (`gsd-core/bin/lib/capability-validator.cjs`, ADR-1244 D2) extracted from `scripts/gen-capability-registry.cjs` so the build-time generator and the runtime overlay loader share one validator implementation. Exports the same `validateCapability(manifest)` surface consumed by both the generator (build-time) and `capability-loader.cjs` (runtime). Generative-parity is CI-guarded: a drift between the generator's validation logic and the extracted module is a hard failure. Callers that previously inlined validation against the generator's internal helpers are migrated to import this module directly. Source of truth: `gsd-core/bin/lib/capability-validator.cjs`. diff --git a/docs/how-to/add-or-update-a-host-integration.md b/docs/how-to/add-or-update-a-host-integration.md index 5acf1e3e1..dd51e4617 100644 --- a/docs/how-to/add-or-update-a-host-integration.md +++ b/docs/how-to/add-or-update-a-host-integration.md @@ -2,7 +2,7 @@ This guide is for GSD maintainers adding a new host CLI, or updating an existing host's host-integration axes (ADR-1239 Phase A). It covers the **documentation-sourcing rule**, the -eight `runtime.hostIntegration` axes, the `undocumented` sentinel, and how to validate. +nine `runtime.hostIntegration` axes, the `undocumented` sentinel, and how to validate. The governing rule for this whole process: **every axis value must come from the host's own authoritative documentation. Never infer, guess, or assume.** Where the docs do not state an axis, @@ -22,7 +22,7 @@ In order of preference: Capture the exact source (Context7 library id + query, or the doc URL) and a short verbatim quote for each value you determine. You will paste these into the matrix in step 4. -## 2. Determine each of the eight axes from the docs +## 2. Determine each of the nine axes from the docs Read the docs and map them to the closed vocabulary. Do not pick a value unless a source states it. @@ -36,6 +36,7 @@ Read the docs and map them to the closed vocabulary. Do not pick a value unless | `stateIO` | `filesystem`, `sandboxed-storage` (web IDE, no arbitrary FS), or `session-log-append`. | | `transport` | `mcp` (native MCP support) vs. `native-extension` (MCP needs a community extension). | | `runtime` | The plugin/extension runtime: `node`, `bun`, `sandboxed-web`, `python`, `go`, `rust`, `electron`, `other`. | +| `effortSurface` | How reasoning effort reaches the host: `argv` (a flag on the host's own invocation) or `none` (no reasoning-effort mechanism). Added by #2481. There is deliberately **no** config-file member — do not invent one; use `undocumented` when the host's docs state no reasoning setting. | ## 3. Write the `runtime.hostIntegration` block diff --git a/docs/how-to/author-a-host-plugin.md b/docs/how-to/author-a-host-plugin.md index 8f488bd3e..195de8667 100644 --- a/docs/how-to/author-a-host-plugin.md +++ b/docs/how-to/author-a-host-plugin.md @@ -27,8 +27,8 @@ from your axes): ## 2. Declare your host's integration axes -Declare the eight negotiated axes (`embeddingMode`, `commandSurface`, `dispatch`, -`modelMode`, `hookBus`, `stateIO`, `transport`, `runtime`) from your host's +Declare the nine negotiated axes (`embeddingMode`, `commandSurface`, `dispatch`, +`modelMode`, `hookBus`, `stateIO`, `transport`, `runtime`, `effortSurface`) from your host's **authoritative documentation** — never infer. Where the docs are silent, use the `undocumented` sentinel (the SDK degrades it fail-closed). See `docs/reference/host-integration-interface.md` for the closed vocabulary. diff --git a/docs/reference/host-integration-capability-matrix.md b/docs/reference/host-integration-capability-matrix.md index 2a3d96201..50c7c8688 100644 --- a/docs/reference/host-integration-capability-matrix.md +++ b/docs/reference/host-integration-capability-matrix.md @@ -24,6 +24,7 @@ consumed verbatim by `gen:capability-registry` and validated by `capability-vali | `stateIO` | Filesystem access model: `filesystem` (full local FS), `sandboxed-storage`, `session-log-append`. | | `transport` | Integration transport: `mcp` (Model Context Protocol), `native-extension`. | | `runtime` | Plugin/extension execution runtime: `node`, `bun`, `python`, `go`, `rust`, `electron`, `sandboxed-web`, `other`. | +| `effortSurface` | How reasoning effort reaches this host: `argv` (deliverable as an argument on the host's own invocation), `none` (the host exposes no reasoning-effort mechanism), or `undocumented`. Added by #2481; there is deliberately **no** config-file member — the only host that ever had one (Gemini CLI's `thinkingConfig`) was removed as a sunset runtime, and naming a member with no host would be a guess. | ### dispatch sub-axes @@ -60,6 +61,7 @@ consumed verbatim by `gen:capability-registry` and validated by `capability-vali | stateIO | filesystem | https://code.claude.com/docs/en/sandboxing | "The sandboxed Bash tool restricts file system access, granting read and write access to the current working directory and session temp direc" | | transport | mcp | https://code.claude.com/docs/en/mcp | "Project-Scoped MCP Server Configuration in .mcp.json ... This JSON structure illustrates the format for a project-scoped MCP server configur" | | runtime | node | https://code.claude.com/docs/en/agent-sdk/typescript | "import { query } from \"@anthropic-ai/claude-agent-sdk\"; ... pathToClaudeCodeExecutable (string) - Specifies the path to the Claude Code CLI" | +| effortSurface | argv | https://code.claude.com/docs/en/cli-reference ; `claude --help` | `--effort ` is a documented flag on the host's own invocation, so GSD renders the resolved universal effort straight onto the argv it spawns (#2481). | | dispatch.namedDispatch | true | https://code.claude.com/docs/en/agent-sdk/subagents | "agents: { 'code-reviewer': AgentDefinition({ description: 'Expert code reviewer.', ... }) } ... subagent_type: block.inp" | | dispatch.nested | true | https://code.claude.com/docs/en/sub-agents | "As of Claude Code v2.1.172, a subagent can spawn its own subagents, allowing delegated tasks to split into parallel subt" | | dispatch.maxDepth | 5 | https://code.claude.com/docs/en/sub-agents | "foreground subagents can spawn at any depth, blocking their parent until completion. Background subagents are limited to" | @@ -95,6 +97,7 @@ Sources consulted: | stateIO | filesystem | https://developers.openai.com/codex/concepts/sandboxing | "workspace-write: The default mode allowing Codex to read files, edit within the workspace, and run routine local commands inside that bounda" | | transport | mcp | https://github.com/openai/codex/blob/main/codex-rs/config/src/config_toml.rs | "pub mcp_servers: HashMap ... Definition for MCP servers that Codex can reach out to for tool calls." | | runtime | node | https://github.com/openai/codex/blob/main/codex-cli/package.json | "\"engines\": {\"node\": \">=16\"} ... The npm-distributed CLI wrapper is a Node.js script (#!/usr/bin/env node)" | +| effortSurface | argv | https://github.com/openai/codex/blob/main/codex-rs/exec/src/cli.rs ; https://developers.openai.com/codex/config-reference | `model_reasoning_effort` is a `config.toml` key and **not** a dedicated CLI flag, so the generic `-c =` override is the only argv route — which is still argv, hence `argv` rather than a config-file member (#2481). | | dispatch.namedDispatch | true | https://github.com/openai/codex/blob/main/codex-rs/core/src/tools/handlers/multi_agents_spec.rs | "\"agent_type\".to_string(), JsonSchema::string(Some(agent_type_description.to_string())) ... apply_role_to_config(&mut con" | | dispatch.nested | true | https://developers.openai.com/codex/multi-agent | "agents.max_depth defaults to 1, which allows a direct child agent to spawn but prevents deeper nesting." | | dispatch.maxDepth | 1 | https://developers.openai.com/codex/config-reference | "agents.max_depth: Maximum nesting depth allowed for spawned agent threads (root sessions start at depth 0; default: 1)" | @@ -139,6 +142,7 @@ Documentation gaps: | stateIO | filesystem | https://opencode.ai/docs/plugins | "Plugin context includes `directory` (working directory path), `worktree` (git worktree path), and `$` (\"Bun's shell API\")" | | transport | mcp | https://opencode.ai/docs/mcp-servers | "\"OpenCode supports both local and remote servers.\" and \"Once added, MCP tools are automatically available to the LLM\"" | | runtime | bun | https://opencode.ai/docs/plugins | "\"$\": Bun's shell API for executing commands\" (plugin context property); \"OpenCode runs `bun install` at startup\"" | +| effortSurface | argv | https://opencode.ai/docs/cli ; `opencode run --help` | `--variant` is accepted on `opencode run`, so the resolved effort is deliverable as an invocation argument (#2481). | | dispatch.namedDispatch | true | https://opencode.ai/docs/agents | "\"Subagents can be invoked: Automatically by primary agents for specialized tasks based on their descriptions. Manually b" | | dispatch.nested | undocumented | no authoritative doc — searched: https://opencode.ai/docs/agents | — | | dispatch.maxDepth | undocumented | no authoritative doc — searched: https://opencode.ai/docs/agents | — | @@ -173,6 +177,7 @@ Documentation gaps: | stateIO | filesystem | https://cursor.com/docs/reference/sandbox | "Local agents run with sandbox options disabled by default." | | transport | mcp | https://cursor.com/docs/mcp | "The Model Context Protocol (MCP) allows Cursor to connect to external tools and data sources." | | runtime | node | https://cursor.com/docs/sdk/typescript | "The SDK runs on Node.js. It requires Node.js 22.13 or later and is described as a Node-first package." | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://cursor.com/docs/subagents | "Invoke specific subagents using slash commands in your prompt. This allows for direct control over which agent performs" | | dispatch.nested | true | https://cursor.com/docs/sdk/typescript | "The top-level agent and its direct subagents can launch subagents, but a subagent launched by another subagent can't lau" | | dispatch.maxDepth | 2 | https://cursor.com/docs/sdk/typescript | "The top-level agent and its direct subagents can launch subagents, but a subagent launched by another subagent can't lau" | @@ -209,6 +214,7 @@ Sources consulted: | stateIO | filesystem | /cline/cline (Context7) — https://github.com/cline/cline/blob/main/cline/sdk/packages/shared/src/storage/paths.ts | "resolveClineDir() returns ~/.cline; resolveDocumentsExtensionPath('Workflows') returns ~/Documents/Cline/Workflows." | | transport | mcp | /cline/cline (Context7) — https://github.com/cline/cline/blob/main/docs/mcp/mcp-overview.mdx | "MCP (Model Context Protocol) enables Cline to interact with external tools and data sources" | | runtime | node | /cline/cline (Context7) — https://github.com/cline/cline/blob/main/sdk/examples/plugins/typescript-lsp/README.md | "Installs a portable subagent plugin ... cp examples/plugins/agents-squad/index.ts ~/.cline/plugins/portable-subagents.ts." | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | /cline/cline (Context7) — https://github.com/cline/cline/blob/main/sdk/examples/plugins/agents-squad/README.md | "parent → start_subagent(preset: \"phantom\", task: \"Map the auth module\") → phantom: save_handoff(...)" | | dispatch.nested | false | /cline/cline (Context7) — https://github.com/cline/cline/blob/main/docs/features/subagents.mdx | "subagents are restricted from editing files, using the browser, accessing MCP servers, or creating nested subagents." | | dispatch.maxDepth | 1 | /cline/cline (Context7) — https://github.com/cline/cline/blob/main/docs/features/subagents.mdx | "They are explicitly prohibited from ... spawning other subagents." | @@ -246,6 +252,7 @@ Sources consulted: | stateIO | filesystem | https://hermes-agent.nousresearch.com/docs/user-guide/configuration | "The agent has the same filesystem access as your user account." | | transport | mcp | https://hermes-agent.nousresearch.com/docs/user-guide/features/mcp | "MCP support ships with the standard install — no extra step needed." | | runtime | python | Context7 /nousresearch/hermes-agent | "The plugin and agent runtime is Python (confirmed by register(ctx) in __init__.py, importlib.import_module, run_agent.py, tools/registry.py)" | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | false | https://hermes-agent.nousresearch.com/docs/user-guide/features/delegation | "The documentation contains no mention of named agents. Subagents are identified only by role ('leaf' or 'orchestrator')" | | dispatch.nested | true | /nousresearch/hermes-agent (Context7) — configuration.md | "max_spawn_depth: 1 — Delegation tree depth cap (1-3, clamped). 1 = flat (default): parent spawns leaves that cannot dele" | | dispatch.maxDepth | 1 | /nousresearch/hermes-agent (Context7) — configuration.md | "max_spawn_depth: 1 # Delegation tree depth cap (1-3, clamped). 1 = flat (default): parent spawns leaves that cannot dele" | @@ -283,6 +290,7 @@ Documentation gaps: | stateIO | filesystem | https://www.explainx.ai/blog/antigravity-cli-features-sandbox-plugins-subagents-2026 | "Plugin staging at ~/.gemini/antigravity-cli/plugins//; skills at ~/.gemini/antigravity-cli/skills/" | | transport | mcp | https://dev.to/arindam_1729/antigravity-cli-a-hands-on-guide-to-googles-terminal-coding-agent-5bc7 | "Both local (stdio) and remote (HTTP) Model Context Protocol servers are supported" | | runtime | go | https://developers.googleblog.com/an-important-update-transitioning-gemini-cli-to-antigravity-cli/ | "Built in Go, Antigravity CLI is snappier and more responsive." | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | undocumented | no authoritative doc — searched: https://www.aibuilderclub.com/blog/antigravity-cli-guide, https://antigravity.google/docs/agents | — | | dispatch.nested | undocumented | no authoritative doc — searched: https://antigravity.google/docs/agents | — | | dispatch.maxDepth | undocumented | no authoritative doc — searched: https://antigravity.google/docs/agents | — | @@ -321,6 +329,7 @@ Documentation gaps: | stateIO | filesystem | https://github.com/augmentcode/auggie | "Node.js 22+ required. Hook configurations use `${AUGMENT_PLUGIN_ROOT}`" | | transport | mcp | https://docs.augmentcode.com/cli/plugins | "Auggie supports a plugin system that allows you to extend its functionality with... MCP server integrations." | | runtime | node | https://github.com/augmentcode/auggie | "Node.js 22+ required" | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://docs.augmentcode.com/cli/subagents | "| **name** | Yes | Name of the agent | ... you can trigger it by sending a message that references the agent name." | | dispatch.nested | undocumented | no authoritative doc — searched: https://docs.augmentcode.com/cli/subagents | — | | dispatch.maxDepth | undocumented | no authoritative doc — searched: https://docs.augmentcode.com/cli/subagents | — | @@ -384,6 +393,7 @@ upgrade coverage is in `tests/augment-upgrades.test.cjs`. | stateIO | filesystem | https://qwenlm.github.io/qwen-code-docs/en/developers/channel-plugins | "Runtime Environment: Node.js only. The architecture uses standard Node.js APIs: import, async/await, file I/O (writeFileSync), OS utilities" | | transport | mcp | https://qwenlm.github.io/qwen-code-docs/en/developers/tools/mcp-server | "Qwen Code integrates with MCP servers through a sophisticated discovery and execution system" | | runtime | node | https://qwenlm.github.io/qwen-code-docs/en/developers/channel-plugins | "Language: Node.js (TypeScript/JavaScript). Execution model: In-process — plugins load at startup as extensions." | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://qwenlm.github.io/qwen-code-docs/en/users/features/sub-agents/ | "Named subagents are invoked when the AI identifies tasks matching their specialization... Users can also explicitly requ" | | dispatch.nested | false | https://qwenlm.github.io/qwen-code-docs/en/users/features/sub-agents/ | "Fork children cannot create further forks. This is enforced at runtime — if a fork attempts to spawn another fork, it re" | | dispatch.maxDepth | 1 | https://qwenlm.github.io/qwen-code-docs/en/users/features/sub-agents/ | "Fork children cannot create further forks. This is enforced at runtime" | @@ -420,6 +430,7 @@ Documentation gaps: | stateIO | filesystem | https://www.codebuddy.ai/docs/cli/settings | "Storage operates in non-sandboxed mode by default ... Default: Full filesystem access governed by permission rules" | | transport | mcp | https://www.codebuddy.ai/docs/cli/cli-reference | "MCP (Model Context Protocol) is built-in as a core feature ... codebuddy mcp command to 'Configure Model Context Protocol (MCP) servers'" | | runtime | node | https://www.codebuddy.ai/docs/cli/sdk | "TypeScript/JavaScript: Node.js >= 18.20 ... npm install @tencent-ai/agent-sdk" | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://www.codebuddy.ai/docs/cli/sub-agents | "Sub-agents can be invoked explicitly by name: 'Request a specific sub-agent by mentioning it in your command'" | | dispatch.nested | false | https://www.codebuddy.ai/docs/cli/sub-agents | "This prevents infinite nesting of agents (sub-agents cannot spawn other sub-agents)" | | dispatch.maxDepth | 1 | https://www.codebuddy.ai/docs/cli/sub-agents | "The architecture enforces exactly one level of nesting — only the main CodeBuddy Code instance can invoke sub-agents." | @@ -452,6 +463,7 @@ Sources consulted: | stateIO | filesystem | https://docs.github.com/en/copilot/how-tos/copilot-cli/customize-copilot/add-mcp-servers | "Configuration file Location: ~/.copilot/mcp-config.json. Hook config files stored in .github/hooks/*.json" | | transport | mcp | https://docs.github.com/en/copilot/how-tos/copilot-cli/customize-copilot/add-mcp-servers | "Copilot CLI comes with the GitHub MCP server already configured. STDIO is the standard transport." | | runtime | undocumented | no authoritative doc — searched: https://github.com/github/copilot-cli/blob/main/README.md, https://github.com/github/copilot-sdk/blob/main/nodejs/README.md | — | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://github.com/github/copilot-sdk/blob/main/docs/features/custom-agents.md | "A custom agent is a named agent configuration that includes its own prompt and tool set. A sub-agent is a custom agent i" | | dispatch.nested | false | https://awesome-copilot.github.com/learning-hub/agents-and-subagents/ | "By default, subagents do not keep spawning additional subagents." | | dispatch.maxDepth | 1 | https://awesome-copilot.github.com/learning-hub/agents-and-subagents/ | "Depth counts how many agents are nested within one another. When the depth limit is reached, the innermost agent cannot" | @@ -487,6 +499,7 @@ Documentation gaps: | stateIO | filesystem | https://kilo.ai/docs/contributing/architecture | "Local execution and hosted execution are separate boundaries. Local runtime instances are Directory-keyed runtime context" | | transport | mcp | https://kilo.ai/docs/automate/mcp/what-is-mcp | "Kilo Code implements the Model Context Protocol to connect to both local and remote MCP servers" | | runtime | bun | https://kilo.ai/docs/automate/extending/plugins | "npm plugins are installed automatically at startup using Bun. Plugin context includes $ (Bun shell). Plugins are TypeScript or JavaScript mo" | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://kilo.ai/docs/customize/custom-subagents | "Configured subagents can be invoked automatically by primary agents (like the Orchestrator) using the Task tool" | | dispatch.nested | true | https://github.com/Kilo-Org/kilocode/issues/7055 | "A subagent can still call the task tool if its merged permissions contain an explicit task rule, which enables nested su" | | dispatch.maxDepth | -1 | https://github.com/Kilo-Org/kilocode/issues/8637 | "there is no maximum nesting depth and the system relies entirely on permission gating" | @@ -524,6 +537,7 @@ Documentation gaps: | stateIO | filesystem | https://docs.devin.ai/desktop/cascade/cascade | "Cascade can create and modify codebases directly … File access can be restricted through .codeiumignore files" | | transport | mcp | https://docs.devin.ai/desktop/cascade/mcp | "Cascade now natively integrates with MCP, allowing you to bring your own selection of MCP servers for Cascade to use." | | runtime | undocumented | no authoritative doc — searched: https://docs.devin.ai/windsurf/plugins/getting-started.md, /llmstxt/windsurf_llms-full_txt (Context7) | — | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | undocumented | no authoritative doc — searched: https://docs.devin.ai/cli/subagents.md, https://docs.devin.ai/desktop/agent-command-center.md | — | | dispatch.nested | undocumented | no authoritative doc — searched: https://docs.devin.ai/cli/subagents.md | — | | dispatch.maxDepth | undocumented | no authoritative doc — searched: https://docs.devin.ai/cli/subagents.md | — | @@ -565,6 +579,7 @@ Documentation gaps: | stateIO | filesystem | https://traeide.com/news/6 | "Rules at '.trae/project_rules.md', skills at '.trae/skills/', MCP config at '.trae/mcp.json'; 'codebase files always remain on your local de" | | transport | mcp | https://docs.trae.ai/ide/model-context-protocol | "Page title from official docs: 'In TRAE IDE, MCP servers support three transport types' — MCP is built-in" | | runtime | node | https://news.ycombinator.com/item?id=44703164 | "Trae is a VSCode fork built on Electron; 'Electron is designed to create desktop applications… a backend using the Node.js runtime'" | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://docs.trae.ai/ide/agent | "Agents in Trae 'can be called individually, or automatically called by SOLO Agent at the corresponding stage'" | | dispatch.nested | undocumented | no authoritative doc — searched: https://docs.trae.ai/ide/solo-mode, https://docs.trae.ai/ide/agent | — | | dispatch.maxDepth | undocumented | no authoritative doc — searched: https://docs.trae.ai/ide/solo-mode | — | @@ -606,6 +621,7 @@ Documentation gaps: | stateIO | filesystem | https://github.com/MoonshotAI/kimi-cli | "Kimi Code CLI is an AI agent that runs in the terminal ... capable of reading and editing code, executing shell commands, searching files" | | transport | mcp | https://github.com/moonshotai/kimi-cli/blob/main/docs/en/reference/kimi-mcp.md | "kimi mcp add ... --transport stdio|http ... Manage MCP Servers: Use the kimi mcp sub-command group to add, list, remove, or authorize MCP se" | | runtime | python | https://context7.com/moonshotai/kimi-cli/llms.txt | "from kimi_cli.app import KimiCLI ... from kosong.tooling import CallableTool2 — CLI core is Python" | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://moonshotai.github.io/kimi-cli/en/customization/agents.html | "subagents:\n coder:\n path: ./coder-sub.yaml\n description: \"Handle coding tasks\"\n reviewer:\n path: ./reviewer-sub.yaml" | | dispatch.nested | false | https://moonshotai.github.io/kimi-cli/en/customization/agents.html | "All subagent types are prohibited from nesting the `Agent` tool (subagents cannot create their own subagents). Only root" | | dispatch.maxDepth | 1 | https://moonshotai.github.io/kimi-cli/en/customization/agents.html | "All subagent types are prohibited from nesting the `Agent` tool (subagents cannot create their own subagents). Only root" | @@ -643,6 +659,7 @@ Documentation gaps: | stateIO | filesystem | https://github.com/MoonshotAI/kimi-code | "Kimi Code CLI is an AI coding agent that runs in your terminal — it can read and edit code, run shell commands, search files, fetch web pages, and choose the next step based on the feedback it receives." Full local filesystem; no sandboxed-storage tier is documented. | | transport | mcp | https://github.com/MoonshotAI/kimi-code ; /docs/en/customization/plugins.md | "AI-native MCP configuration. Add, edit, and authenticate Model Context Protocol servers conversationally with `/mcp-config`, without hand-editing JSON." Plugin manifests additionally carry an `mcpServers` field. | | runtime | node | https://github.com/MoonshotAI/kimi-code | "Requirements: Node.js ≥ 24.15.0, pnpm 10.33.0." The CLI is a TypeScript pnpm monorepo (`apps/kimi-code`, `packages/agent-core-v2`); hook commands are shell, and MCP servers are external processes. | +| effortSurface | *(not declared)* | https://github.com/moonshotai/kimi-code/blob/main/apps/kimi-code/src/tui/commands/registry.ts | Kimi Code documents `/effort` (alias `/thinking`) to change reasoning effort, but only as an **interactive slash command** — there is no `--effort`-style flag on its invocation, and `-m, --model` is the only model/effort-adjacent argv. The axis vocabulary is `argv` | `none`, and neither is accurate: `none` would deny a mechanism the host does have, and `argv` would claim one it does not expose. The descriptor therefore declares no value, which negotiation degrades closed identically to the sentinel. See Documentation gaps. | | dispatch.namedDispatch | false | https://github.com/moonshotai/kimi-code/blob/main/docs/en/customization/agents.md ; `capabilities/kimi-code/capability.json` (`artifactLayout` — skills only) | "The system includes three built-in sub-agents: 'coder' for general software engineering tasks like file modification, 'explore' for read-only codebase navigation and summarization, and 'plan' for architecture design without file or shell access." GSD's kimi-code artifact layout installs Agent **Skills** only (no `agents` kind), so no named GSD subagent is ever registered with the host and every GSD role resolves to one of the three built-ins (`resolveDispatchType`, `src/host-integration.cts`). **This is a GSD-integration-scoped `false`, not a claim that the host lacks named agents — see Documentation gaps.** | | dispatch.nested | true | https://github.com/moonshotai/kimi-code/blob/main/docs/en/customization/agents.md | The `coder` sub-agent "can dispatch its own nested sub-agents when a task decomposes naturally." (Agent files also expose a `subagents` delegation allowlist, where `subagents: []` is what *prevents* further delegation — nesting is the default.) | | dispatch.maxDepth | undocumented | searched: https://github.com/moonshotai/kimi-code/blob/main/docs/en/customization/agents.md | Nesting is documented, but no maximum nesting depth is stated anywhere in the agents or tools reference. | @@ -667,6 +684,7 @@ Sources consulted: Documentation gaps: - **dispatch.namedDispatch — host capability vs. GSD surface.** Kimi Code *does* support user-authored named agents: "Beyond the three built-in sub-agents, you can define your own agents as Markdown files", discovered across five scopes (`--agent-file` > `.kimi-code/agents/`, `.agents/agents/` > extra dirs > `$KIMI_CODE_HOME/agents/` > built-in). The axis is `false` because GSD installs **no** agent files for this host, so its reachable dispatch surface is the three built-ins — flipping the axis without also shipping agent artifacts would reintroduce the dispatch failure recorded in `docs/migration/kimi-to-kimi-code.md` ("Every workflow that called a named GSD subagent … **failed at dispatch**"). Shipping GSD agent files to kimi-code is an unbuilt capability upgrade, not a gap in the host's documentation. +- **effortSurface — a real mechanism the vocabulary cannot name.** Kimi Code documents `/effort` (alias `/thinking`) for changing reasoning effort, but only as an interactive slash command; `-m, --model` is the only model-adjacent argv on its invocation. The axis vocabulary is `argv` | `none`, and neither is accurate here — `none` denies a mechanism the host has, `argv` claims one it does not expose. Kimi Code's descriptor was created a day after #2481 added the axis and has carried no value since. Resolving this needs a vocabulary decision (an interactive-only member, mirroring the deliberate absence of a config-file member), which is a negotiation change rather than a documentation one; until then the absent value degrades closed exactly as the sentinel does. - **dispatch.maxDepth** — nesting is documented but no depth bound is published, so the axis carries the `undocumented` sentinel rather than a guessed integer. - **runtime** — the published artifact is a single bundled binary; `Node.js ≥ 24.15.0` is stated as a *development* requirement. The sources are a TypeScript/Node monorepo, so `node` is the best-supported classification, but the docs do not name a canonical plugin-execution runtime (the same ambiguity noted for `codex` above). @@ -685,6 +703,7 @@ Documentation gaps: | stateIO | filesystem | https://zcode.z.ai/en/docs/skill | "User-level skills for ZCode Agent: `~/.zcode/skills//SKILL.md`" — full local filesystem (desktop app). | | transport | mcp | https://zcode.z.ai/en/docs/mcp-services | "MCP (Model Context Protocol) connects external capabilities ... type as `stdio` (SSE and HTTP remote servers are also supported)" — native MCP. | | runtime | electron | https://zcode.z.ai/en/docs/install (download path `cdn-zcode.z.ai/zcode/electron/releases/3.2.5/ZCode-3.2.5-mac-arm64.dmg`) | ZCode is shipped as an Electron desktop application; the release artifact lives under the `electron/releases` path. | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://zcode.z.ai/en/docs/subagents | "you can let the Agent pick the subagent automatically, or reference it with `@` in the chat box" — subagents are invoked by name via the Agent tool. | | dispatch.nested | undocumented | searched: https://zcode.z.ai/en/docs/subagents | The docs do not state whether a subagent can itself spawn further subagents. | | dispatch.maxDepth | undocumented | searched: https://zcode.z.ai/en/docs/subagents | No maximum nesting depth is documented. | @@ -726,6 +745,7 @@ EoS migration status (#2101, ADR-1239): ZCode's install is fully dogfooded throu | stateIO | session-log-append | https://pi.dev/docs/latest/session-format | pi persists conversation/tool-call state as an append-only session log/transcript format rather than exposing unrestricted local filesystem access to extensions. | | transport | native-extension | https://pi.dev/docs/latest/extensions | Integration is a single loaded extension file (`~/.pi/agent/extensions/.cjs`), not an MCP server process — the peer mechanism to OpenCode's native `plugins/*.js` adapter. | | runtime | bun | https://pi.dev | pi is distributed and executed as a bun-runtime CLI (its extensions are loaded via jiti under bun, not Node.js or Python). | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | undocumented | no authoritative doc — searched: https://pi.dev/docs/latest/extensions | The `ExtensionAPI` documents `registerCommand`/`registerTool`/`registerProvider`/`pi.on`; it does not document a named-subagent-invocation primitive. | | dispatch.nested | undocumented | no authoritative doc — searched: https://pi.dev/docs/latest/extensions | No documented subagent-of-subagent nesting capability. | | dispatch.maxDepth | 0 | no authoritative doc — searched: https://pi.dev/docs/latest/extensions | No named-dispatch primitive is documented at all (see `dispatch.namedDispatch`), so there is no nesting depth to bound; `0` records "no dispatch levels beyond the root extension," not a measured limit. | @@ -773,6 +793,7 @@ EoS migration status (#2102 Stage 2, ADR-1239): Stage 1's "in-process `gsd-core` | stateIO | sandboxed-storage | https://code.visualstudio.com/api/references/vscode-api#Memento | `context.globalState`/`context.workspaceState` (both `Memento`) are the extension's persistent storage surface — sandboxed key/value storage scoped to the extension, not unrestricted local filesystem access. | | transport | mcp | https://code.visualstudio.com/api/extension-guides/ai/mcp | VS Code 1.99 added native MCP client support; on the Web (webworker) entry, full GSD command dispatch is available through VS Code's native MCP client connecting to the GSD companion MCP server (`gsd-mcp-server`), not an in-process Node dispatch (which the web entry cannot run at all). | | runtime | sandboxed-web | https://code.visualstudio.com/api/extension-guides/web-extensions | The `browser` entry point (`vscode/browser.js`) runs in a webworker context with no Node core modules — the Web Extension execution model VS Code documents for extensions that must run in vscode.dev/github.dev. | +| effortSurface | undocumented | no authoritative doc — per-host reasoning-effort survey, #2481 (`09b535ac0`) | This host's documentation states no reasoning-effort setting, so the axis carries the fail-closed sentinel rather than inheriting a profile baseline. No host-specific URL is cited because the finding is an ABSENCE: #2481 surveyed all hosts for a reasoning-effort mechanism and found one only for claude/opencode/codex. | | dispatch.namedDispatch | true | https://code.visualstudio.com/docs/copilot/chat/chat-agent-mode#_agent-mode-tools | Registered `languageModelTools` (and the chat participant) are addressable by name — the primary agent references a tool/participant by its declared name/`toolReferenceName`, not only positionally. | | dispatch.nested | true | https://code.visualstudio.com/docs/copilot/copilot-chat-agents (subagents) | VS Code's chat subagent model (`#runSubagent`) explicitly supports a subagent invoking further subagents, gated by `chat.subagents.allowInvocationsFromSubagents`. | | dispatch.maxDepth | 5 | https://code.visualstudio.com/docs/copilot/copilot-chat-agents (subagents) | Documented as VS Code's maximum nesting depth for `#runSubagent` chains — also matches this repo's existing `PROFILE_BASELINES.ide.dispatch.maxDepth` baseline. | diff --git a/docs/reference/host-integration-interface.md b/docs/reference/host-integration-interface.md index 55b628108..db2d8d024 100644 --- a/docs/reference/host-integration-interface.md +++ b/docs/reference/host-integration-interface.md @@ -18,7 +18,7 @@ assumes a capability. negotiated capability set; see the [versioning policy](../explanation/interface-versioning-policy.md) for what a bump means. -## The eight negotiated axes +## The nine negotiated axes `HOST_INTEGRATION_AXES` is the closed vocabulary. Each axis takes a documented value (or the `undocumented` sentinel): @@ -33,6 +33,7 @@ value (or the `undocumented` sentinel): | `stateIO` | `filesystem` \| `sandboxed-storage` \| `session-log-append` | | `transport` | `mcp` \| `native-extension` | | `runtime` | `node` \| `bun` \| `sandboxed-web` \| `python` \| `go` \| `rust` \| `electron` \| `other` | +| `effortSurface` | `argv` \| `none` | ## Classification + negotiation diff --git a/docs/registries/README.md b/docs/registries/README.md index 58fb070ca..e0a327bd0 100644 --- a/docs/registries/README.md +++ b/docs/registries/README.md @@ -118,7 +118,7 @@ Example: |---|---|---| | `interfacePoints` | yes, non-empty | Subset of the six ADR-1239 interface points it binds: `command`, `dispatch`, `model`, `hooks`, `state`, `artifact`. | | `profile` | yes | One of the three host-capability profiles: `programmatic-cli`, `declarative-cli`, `ide`. | -| `axes` | yes | Object with **exactly** the eight ADR-1239 negotiated axes keys: `embeddingMode`, `commandSurface`, `dispatch`, `modelMode`, `hookBus`, `stateIO`, `transport`, `runtime`. | +| `axes` | yes | Object with **exactly** the nine ADR-1239 negotiated axes keys: `embeddingMode`, `commandSurface`, `dispatch`, `modelMode`, `hookBus`, `stateIO`, `transport`, `runtime`, `effortSurface`. | `axes` value vocabulary: diff --git a/tests/fix-2615-effortsurface-matrix-parity.test.cjs b/tests/fix-2615-effortsurface-matrix-parity.test.cjs new file mode 100644 index 000000000..0fe485a6f --- /dev/null +++ b/tests/fix-2615-effortsurface-matrix-parity.test.cjs @@ -0,0 +1,99 @@ +// allow-test-rule: runtime-contract-is-the-product #2615 — the host-integration matrix +// IS the cited source of truth for every descriptor axis (ADR-1239); asserting that a +// shipped axis value appears there, and matches, is a contract assertion. + +/** + * Regression test for #2615 — `effortSurface` was a shipped `hostIntegration` axis + * with NO presence in the matrix that is supposed to be its cited source of truth. + * + * #2481 added the axis and wrote docs-sourced values into 18 descriptors + * (`claude`/`codex`/`opencode` -> `argv`, 15 others -> `undocumented`) but never + * touched `docs/reference/host-integration-capability-matrix.md`: the axes legend + * omitted it and not one per-runtime table carried a row. `src/host-integration.cts` + * states "every value is documented or explicitly 'undocumented'" — for this axis + * that was false for every runtime. + * + * This test is deliberately GENERIC rather than a hardcoded list: it derives the + * runtimes from the registry, so a runtime added later fails here until its matrix + * row exists. That is the ratchet the original gap needed — #2481 added an axis and + * nothing caught the missing documentation. + */ + +'use strict'; + +process.env.GSD_TEST_MODE = '1'; + +const { describe, test } = require('node:test'); +const assert = require('node:assert/strict'); +const fs = require('node:fs'); +const path = require('node:path'); + +const ROOT = path.join(__dirname, '..'); +const MATRIX = path.join(ROOT, 'docs', 'reference', 'host-integration-capability-matrix.md'); +const registry = require(path.join(ROOT, 'gsd-core', 'bin', 'lib', 'capability-registry.cjs')); + +// Normalize CRLF so the row regexes hold on a Windows autocrlf checkout. +const MATRIX_TEXT = fs.readFileSync(MATRIX, 'utf-8').replace(/\r\n/g, '\n'); + +/** Extract a `## ` section body, stopping at the next top-level host heading. */ +function section(host) { + const start = MATRIX_TEXT.indexOf(`\n## ${host}\n`); + if (start === -1) return null; + const rest = MATRIX_TEXT.slice(start + 1); + const end = rest.indexOf('\n## '); + return end === -1 ? rest : rest.slice(0, end); +} + +/** Read the value cell of a `| | | …` row. */ +function axisValue(body, axis) { + const row = body.split(/\r?\n/).find((l) => l.startsWith(`| ${axis} |`)); + return row ? row.split('|')[2].trim() : null; +} + +const RUNTIMES = Object.keys(registry.runtimes).filter( + (id) => registry.runtimes[id]?.runtime?.hostIntegration, +); + +describe('#2615: the matrix documents the effortSurface axis', () => { + test('the axes legend defines effortSurface and its vocabulary', () => { + const legendRow = MATRIX_TEXT.split(/\r?\n/).find((l) => l.startsWith('| `effortSurface` |')); + assert.ok(legendRow, 'the axes legend must define effortSurface (#2615)'); + for (const member of ['`argv`', '`none`', '`undocumented`']) { + assert.ok(legendRow.includes(member), + `the legend must document the ${member} vocabulary member (#2615)`); + } + }); + + test('there is at least one runtime to check', () => { + // Guards the loops below against silently asserting nothing. + assert.ok(RUNTIMES.length >= 18, `expected the full runtime corpus, got ${RUNTIMES.length}`); + }); + + for (const id of RUNTIMES) { + describe(`runtime: ${id}`, () => { + test('has a matrix section', () => { + assert.ok(section(id), `${id}: every installed runtime needs a matrix section (ADR-1239)`); + }); + + test('documents effortSurface, and the value matches the descriptor', () => { + const body = section(id); + assert.ok(body, `${id}: missing matrix section`); + + const documented = axisValue(body, 'effortSurface'); + assert.ok(documented, `${id}: the matrix must carry an effortSurface row (#2615)`); + + const declared = registry.runtimes[id].runtime.hostIntegration.effortSurface; + if (declared === undefined) { + // kimi-code declares no value: its mechanism (`/effort`) is interactive-only + // and neither `argv` nor `none` describes it. The matrix must say so rather + // than invent a value. + assert.match(documented, /not declared/i, + `${id}: an absent descriptor value must be documented as absent, not guessed (#2615)`); + } else { + assert.equal(documented, declared, + `${id}: the matrix effortSurface value must match the shipped descriptor`); + } + }); + }); + } +});