* fix(#2310): guard Codex agent model_overrides so Anthropic aliases never leak into .toml generateCodexAgentToml embedded a per-agent `model_overrides` value verbatim as the Codex `.toml` `model`, leaking GSD/Claude tier aliases (opus/sonnet/haiku/fable) and `claude-*` ids. Codex/ChatGPT rejects those (400 "The 'sonnet' model is not supported when using Codex with a ChatGPT account"), and since spawn_agent has no inline model param, the model is baked into the .toml at install time — so the orchestrator could not recover and fell back to the non-equivalent generic-agent workaround. Translate a GSD tier alias through the Codex tier map (sonnet -> gpt-5.6-terra); drop with a deduped warning any Anthropic-flavored value with no Codex mapping (fable) or a `claude-*` id, so emission falls through to the runtime-aware resolver or Codex's default. A final safety gate blocks an Anthropic-flavored model from the runtime- resolver path too (runtime/target mismatch). Mirrors the Claude-side override guard (#2041). Real Codex/OpenAI model ids in model_overrides still pass through verbatim (#2256 preserved); runtime:"codex" tier resolution unchanged (#2517). Adds regression + fast-check property tests in tests/codex-config.test.cjs. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> * chore(#2310): backfill changeset PR number to #2312 * fix(#2310): Codex passive-model posture — omit Anthropic-flavored model (all namespacings) Adopt ADR-1239's passive/session-only posture for Codex model handling: a Codex agent .toml `model` is embedded ONLY for an explicit real-Codex model_overrides pin; any Anthropic-flavored value is omitted so the agent inherits the always- available session model (never a 400). - model_overrides tier alias (opus/sonnet/haiku/fable) or a Claude model id → omit (was: translate to gpt-*); an explicit real-Codex model id → embed verbatim (#2256 preserved). - Detect ALL Anthropic namespacings, not just `claude-*`: single-source the canonical CLAUDE_AGENT_ALIASES from model-resolver.cts and treat any id whose value contains "claude" (case-insensitive) as Anthropic-flavored — catching `anthropic/claude-*` and `us.anthropic.claude-*` (the forms the catalog assigns to opencode/hermes/kilo), which reach a Codex .toml via the runtime-resolver path on a mixed-runtime + Codex install. - The final safety gate applies to the runtime-resolver path too. The full passive posture (removing #2517's runtime-resolver per-tier embedding + a correctness health-check + a Codex TOML sync path) is tracked as the ADR-2310 epic #2313. Regression + fast-check property tests in tests/codex-config.test.cjs. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> --------- Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
5
.changeset/tidy-goats-hop.md
Normal file
5
.changeset/tidy-goats-hop.md
Normal file
@@ -0,0 +1,5 @@
|
||||
---
|
||||
type: Fixed
|
||||
pr: 2312
|
||||
---
|
||||
**Codex agents no longer fail to launch with an unsupported-model error** — GSD was writing an Anthropic tier name (`opus`/`sonnet`/`haiku`/`fable`) or a `claude-*` id into each Codex agent's `.toml` `model` field, which Codex rejects — fatally on a ChatGPT account (`The 'sonnet' model is not supported when using Codex with a ChatGPT account`). GSD now never writes an Anthropic-flavored model to a Codex agent: an explicit real-Codex model pin is kept, anything else is omitted so the agent inherits the working session model. (#2310)
|
||||
@@ -375,6 +375,7 @@ const {
|
||||
} = require(path.join(_gsdLibDir, 'model-catalog.cjs'));
|
||||
const {
|
||||
resolveTierEntry: gsdResolveTierEntry,
|
||||
CLAUDE_AGENT_ALIASES,
|
||||
} = require(path.join(_gsdLibDir, 'model-resolver.cjs'));
|
||||
|
||||
// #2071 — install-time effort resolution (readGsdEffectiveEffortConfig /
|
||||
@@ -4005,6 +4006,36 @@ purpose: ${toSingleLine(description)}
|
||||
return `${cleanFrontmatter}\n\n${roleHeader}\n${body}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* #2310 — True if `model` is an Anthropic-flavored value that must never appear as a
|
||||
* Codex agent `.toml` `model`. Two forms: (a) a bare Claude Agent-tool tier alias
|
||||
* (opus/sonnet/haiku/fable — the canonical CLAUDE_AGENT_ALIASES, imported from
|
||||
* src/model-resolver.cts so it can't diverge); (b) any Claude model id in any provider
|
||||
* namespacing — `claude-*`, `anthropic/claude-*`, `us.anthropic.claude-*` (the forms the
|
||||
* catalog assigns to opencode/hermes/kilo, reachable on a Codex .toml via the runtime-
|
||||
* resolver path). No OpenAI/Codex model id contains "claude", so a case-insensitive
|
||||
* substring test is a safe, exhaustive guard for (b). Codex/ChatGPT rejects all of these.
|
||||
*/
|
||||
function _isAnthropicFlavoredModel(model) {
|
||||
return typeof model === 'string' && (CLAUDE_AGENT_ALIASES.has(model) || model.toLowerCase().includes('claude'));
|
||||
}
|
||||
|
||||
// #2310 — dedupe stderr warnings so repeated agent emits don't spam (mirrors the
|
||||
// #2041/#1133 model-resolver warn-dedupe). Value is length-capped so an oversized
|
||||
// or secret-shaped override cannot leak in full to logs.
|
||||
const _codexModelOverrideDroppedWarned = new Set();
|
||||
function _warnCodexModelOverrideDropped(agentName, value) {
|
||||
const key = `${agentName}::${value}`;
|
||||
if (_codexModelOverrideDroppedWarned.has(key)) return;
|
||||
_codexModelOverrideDroppedWarned.add(key);
|
||||
const safe = String(value).length > 64 ? `${String(value).slice(0, 64)}…` : String(value);
|
||||
process.stderr.write(
|
||||
`gsd: warning — Codex agent "${agentName}" model "${safe}" is not a valid Codex model ` +
|
||||
`(Anthropic alias/id); dropping it so Codex uses a valid default. ` +
|
||||
`Set runtime:"codex" or pin a gpt-* model to route it.\n`,
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a per-agent .toml config file for Codex.
|
||||
* Sets required agent metadata, sandbox_mode, and developer_instructions
|
||||
@@ -4038,21 +4069,43 @@ function generateCodexAgentToml(agentName, agentContent, modelOverrides = null,
|
||||
// model_overrides is respected on Codex (which uses static TOML, not inline
|
||||
// Task() model parameters). See #2256.
|
||||
// Precedence: per-agent model_overrides > runtime-aware tier resolution (#2517).
|
||||
const modelOverride = modelOverrides?.[resolvedName] || modelOverrides?.[agentName];
|
||||
let hasPinnedModel = false;
|
||||
if (modelOverride) {
|
||||
lines.push(`model = ${JSON.stringify(modelOverride)}`);
|
||||
hasPinnedModel = true;
|
||||
} else if (runtimeResolver) {
|
||||
// #2310 — a Codex .toml `model` MUST be a real Codex/OpenAI model id. Codex is a
|
||||
// passive/session-only model host (ADR-1239): GSD cannot reliably route per-agent
|
||||
// tiers, and a bare GSD/Claude tier alias (opus/sonnet/haiku/fable) or a claude-*
|
||||
// id 400s on a ChatGPT-account Codex ("The 'sonnet' model is not supported when
|
||||
// using Codex with a ChatGPT account"). So: embed ONLY an explicit real-Codex
|
||||
// model pin from model_overrides; omit anything Anthropic-flavored so the agent
|
||||
// inherits the always-available session model. (Removing the runtime-resolver
|
||||
// per-tier embedding below is the ADR-2310 passive-posture epic.)
|
||||
const rawModelOverride = modelOverrides?.[resolvedName] || modelOverrides?.[agentName];
|
||||
let pinnedModel = null;
|
||||
if (rawModelOverride) {
|
||||
if (typeof rawModelOverride === 'string' && rawModelOverride && !_isAnthropicFlavoredModel(rawModelOverride)) {
|
||||
pinnedModel = rawModelOverride; // explicit real-Codex model pin → embed verbatim (#2256)
|
||||
} else {
|
||||
_warnCodexModelOverrideDropped(resolvedName, rawModelOverride); // alias/claude-* → omit
|
||||
}
|
||||
}
|
||||
if (!pinnedModel && runtimeResolver) {
|
||||
// #2517 — runtime-aware tier resolution. Embeds Codex-native model + reasoning_effort
|
||||
// from RUNTIME_PROFILE_MAP / model_profile_overrides for the configured tier.
|
||||
// (Superseded on the default path by the ADR-2310 passive-posture epic.)
|
||||
const entry = runtimeResolver.resolve(resolvedName) || runtimeResolver.resolve(agentName);
|
||||
if (entry?.model) {
|
||||
lines.push(`model = ${JSON.stringify(entry.model)}`);
|
||||
hasPinnedModel = true;
|
||||
// model is resolved here; reasoning_effort from catalog tier is REPLACED by the
|
||||
// unified effort resolver below (#443). Do NOT emit entry.reasoning_effort here.
|
||||
}
|
||||
if (entry?.model) pinnedModel = entry.model;
|
||||
}
|
||||
// #2310 — final safety gate: never emit an Anthropic-flavored model into a Codex
|
||||
// .toml, even from the runtime-resolver path (e.g. a defaults.json runtime that
|
||||
// does not match the codex install target).
|
||||
if (pinnedModel && _isAnthropicFlavoredModel(pinnedModel)) {
|
||||
_warnCodexModelOverrideDropped(resolvedName, pinnedModel);
|
||||
pinnedModel = null;
|
||||
}
|
||||
let hasPinnedModel = false;
|
||||
if (pinnedModel) {
|
||||
lines.push(`model = ${JSON.stringify(pinnedModel)}`);
|
||||
hasPinnedModel = true;
|
||||
// model is resolved here; reasoning_effort from catalog tier is REPLACED by the
|
||||
// unified effort resolver below (#443). Do NOT emit entry.reasoning_effort here.
|
||||
}
|
||||
|
||||
// #443 — Unified effort for Codex .toml. Uses the same config-driven precedence chain
|
||||
|
||||
@@ -573,6 +573,7 @@ function resolveEffortForTier(cwd: string, agentType: string, attempt?: number):
|
||||
|
||||
export = {
|
||||
resolveTierEntry,
|
||||
CLAUDE_AGENT_ALIASES,
|
||||
resolveModelPolicy,
|
||||
resolveModelInternal,
|
||||
_resetModelPolicyWarningCacheForTests,
|
||||
|
||||
@@ -20,6 +20,8 @@ const path = require('path');
|
||||
const os = require('os');
|
||||
const { execFileSync } = require('child_process');
|
||||
const { cleanup } = require('./helpers.cjs');
|
||||
const fc = require('fast-check');
|
||||
const { CLAUDE_AGENT_ALIASES } = require('../gsd-core/bin/lib/model-resolver.cjs');
|
||||
|
||||
// #2153 follow-up: ensure hooks/dist/ exists before any install integration
|
||||
// test runs. The Codex install path copies hook files from hooks/dist/, which
|
||||
@@ -523,6 +525,97 @@ tools: Read, Grep, Glob
|
||||
assert.ok(modelIdx < instrIdx, 'model field must appear before developer_instructions');
|
||||
});
|
||||
|
||||
// ─── #2310: never leak an Anthropic-flavored model into the Codex .toml ─────
|
||||
|
||||
test('omits a bare GSD tier alias in model_overrides (Codex passive/session-only) (#2310)', () => {
|
||||
// ADR-1239: Codex is a passive/session-only model host. A tier alias cannot be
|
||||
// honored per-agent, so it is dropped and the agent inherits the session model (no 400).
|
||||
for (const alias of ['opus', 'sonnet', 'haiku', 'fable']) {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': alias });
|
||||
assert.ok(!/^model = /m.test(result), `alias "${alias}" must be omitted (no model pinned)`);
|
||||
}
|
||||
});
|
||||
|
||||
test('never emits a bare Anthropic tier alias as the Codex model (#2310)', () => {
|
||||
for (const alias of ['opus', 'sonnet', 'haiku', 'fable']) {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': alias });
|
||||
assert.ok(!/^model = "(opus|sonnet|haiku|fable)"$/m.test(result), `must not emit model = "${alias}"`);
|
||||
}
|
||||
});
|
||||
|
||||
test('drops a claude-* model_overrides id instead of leaking it into the Codex .toml (#2310)', () => {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': 'claude-sonnet-5' });
|
||||
assert.ok(!result.includes('claude-'), 'a claude-* id must never appear as the Codex model');
|
||||
// No runtime resolver → nothing to fall through to → no model line at all.
|
||||
assert.ok(!result.includes('model ='), 'unmappable Anthropic override falls through to Codex default (no model pinned)');
|
||||
});
|
||||
|
||||
test('a dropped claude-* override falls through to the runtime resolver (#2310)', () => {
|
||||
const runtimeResolver = { runtime: 'codex', resolve: () => ({ model: 'gpt-5.6-terra' }) };
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': 'claude-opus-4-8' }, runtimeResolver);
|
||||
assert.ok(result.includes('model = "gpt-5.6-terra"'), 'falls through to the runtime-resolved Codex model');
|
||||
assert.ok(!result.includes('claude-'), 'claude id must not leak even with a resolver present');
|
||||
});
|
||||
|
||||
test('final gate blocks an Anthropic model from the runtime-resolver path too (#2310)', () => {
|
||||
// Simulate a defaults.json runtime that does not match the codex install target:
|
||||
// the resolver hands back a Claude id, which must still never reach the Codex .toml.
|
||||
const runtimeResolver = { runtime: 'claude', resolve: () => ({ model: 'claude-sonnet-5' }) };
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, null, runtimeResolver);
|
||||
assert.ok(!result.includes('claude-'), 'runtime-resolver Claude id must be gated out of the Codex .toml');
|
||||
assert.ok(!result.includes('model ='), 'no valid Codex model available → none pinned');
|
||||
});
|
||||
|
||||
test('still emits a real Codex/OpenAI model_overrides id verbatim (#2310 preserves #2256)', () => {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': 'gpt-5.6-sol' });
|
||||
assert.ok(result.includes('model = "gpt-5.6-sol"'), 'a real gpt-* override must still pass through unchanged');
|
||||
});
|
||||
|
||||
test('gates a provider-namespaced anthropic/claude-* model from the runtime-resolver path (#2310 review)', () => {
|
||||
// Catalog assigns anthropic/claude-* to opencode/hermes/kilo. A mixed-runtime config
|
||||
// (runtime: opencode) + Codex install resolves those; they must NOT reach the .toml.
|
||||
const runtimeResolver = { runtime: 'opencode', resolve: () => ({ model: 'anthropic/claude-opus-4-8' }) };
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, null, runtimeResolver);
|
||||
assert.ok(!/claude/i.test(result.split('\n').find((l) => /^model = /.test(l)) || ''), 'no claude-bearing model may be emitted');
|
||||
assert.ok(!/^model = /m.test(result), 'anthropic/claude-* is gated out → no model pinned');
|
||||
});
|
||||
|
||||
test('omits a provider-namespaced anthropic/claude-* model_overrides pin (#2310 review)', () => {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': 'anthropic/claude-sonnet-5' });
|
||||
assert.ok(!result.includes('claude'), 'anthropic/claude-* must be omitted, never emitted');
|
||||
assert.ok(!/^model = /m.test(result), 'no model pinned');
|
||||
});
|
||||
|
||||
test('every canonical Claude tier alias is omitted (single-source guard, #2310 review)', () => {
|
||||
// Iterates the CANONICAL set so a future alias is covered automatically (no divergence).
|
||||
for (const alias of CLAUDE_AGENT_ALIASES) {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': alias });
|
||||
assert.ok(!/^model = /m.test(result), `canonical alias "${alias}" must be omitted`);
|
||||
}
|
||||
});
|
||||
|
||||
test('drops the fable alias (Claude Agent alias with no Codex mapping) (#2310)', () => {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': 'fable' });
|
||||
assert.ok(!result.includes('model = "fable"'), 'fable must never be emitted as the Codex model');
|
||||
assert.ok(!result.includes('model ='), 'fable has no Codex mapping → dropped, no model pinned');
|
||||
});
|
||||
|
||||
test('property: no model_overrides value ever yields an Anthropic-flavored Codex model (#2310)', () => {
|
||||
const anthropicish = fc.oneof(
|
||||
fc.constantFrom('opus', 'sonnet', 'haiku', 'fable'),
|
||||
fc.string().map((s) => `claude-${s}`),
|
||||
fc.string().map((s) => `anthropic/claude-${s}`),
|
||||
fc.string().map((s) => `us.anthropic.claude-${s}`),
|
||||
fc.string(),
|
||||
);
|
||||
fc.assert(fc.property(anthropicish, (v) => {
|
||||
const result = generateCodexAgentToml('gsd-executor', sampleAgent, { 'gsd-executor': v });
|
||||
const m = result.split('\n').find((l) => /^model = /.test(l));
|
||||
if (!m) return true; // no model pinned is always safe
|
||||
return !/claude/i.test(m) && !/^model = "(opus|sonnet|haiku|fable)"$/.test(m);
|
||||
}), { numRuns: 400 });
|
||||
});
|
||||
|
||||
// ─── #774: service_tier / model_verbosity for light-tier agents ───────────────
|
||||
|
||||
test('emits service_tier="flex" and model_verbosity="low" for light-tier agents (#774)', () => {
|
||||
|
||||
Reference in New Issue
Block a user