diff --git a/CHANGELOG.md b/CHANGELOG.md index c71135b..b521482 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,37 @@ appear in a patch rather than inflating the version toward 1.0 on a crate still shape. **Where that happens the entry says so at the top**, because a version number that under-signals is only acceptable if the changelog over-signals to compensate. +## 0.5.1 + +Re-verified against the installed CLIs: claude 2.1.267, codex-cli 0.154.0, GitHub Copilot +CLI 1.0.88 and grok 1.0.40. Every flag was checked against each CLI's `--help`, and the live +suite passed against all four. + +### Fixed + +- **Codex: resuming with extra directories no longer fails.** `codex exec resume` refuses + `--add-dir`; a resumed run now carries its roots as + `-c sandbox_workspace_write.writable_roots=[...]`. A fresh `exec` still takes the flag. +- **Codex: app-server errors carry their message.** The notification is + `{ error: { message }, willRetry }`; the crate read a top-level `message` and always got + nothing. A notification Codex will retry no longer ends the turn as an error. +- **Claude: a `[1m]` run reports its window even when the Haiku helper ran.** + `--output-format json` has no `init` record, so with the helper listed beside the run's + model there was no name to choose by and the window came back unknown. The run's own + `modelUsage` entry is now found by its counts, which equal the top-level `usage`; anything + ambiguous is still unknown rather than guessed. +- **Copilot: effort is passed as `--reasoning-effort`**, the documented spelling in 1.0.88. + `--effort` still parses but is no longer in `--help`. +- **Claude's login hint is `claude auth login`**, the subcommand 2.1.267 documents. + +### Changed + +- **Codex catalogue** follows what 0.154.0 reports: `gpt-6-astra` (now the default), + `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-daybreak-blue-latest` (account + dependent) and `gpt-5.5` (retiring 2026-10-14). `gpt-5.4` and `gpt-5.4-mini` are gone. +- **Grok catalogue** follows `grok models` on 1.0.40: `grok-4.7` (now the default), + `grok-4.7-build-fast`, `grok-4.6`, `grok-4.5`. + ## 0.5.0 The crate now runs on nagoya instead of tokio, and every entry point that spawns a CLI diff --git a/Cargo.toml b/Cargo.toml index 5a4417c..62db939 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "agent-abstraction" -version = "0.5.0" +version = "0.5.1" edition = "2024" # The floor edition 2024 requires, and where the strictest dependencies (uuid, # getrandom) sit. Derived from the dependency graph rather than compile-tested. diff --git a/README.md b/README.md index bb170f6..7df04e6 100644 --- a/README.md +++ b/README.md @@ -45,8 +45,8 @@ Every entry point that spawns a CLI (`run`, `stream`, `interrupt`, `Probe::run`, ## What each agent can actually do -Verified live, against `claude 2.1.205`, `codex-cli 0.146.0` and `GitHub Copilot CLI 1.0.78` -and not inferred from documentation. +Verified live, against `claude 2.1.267`, `codex-cli 0.154.0`, `GitHub Copilot CLI 1.0.88` and +`grok 1.0.40`, and not inferred from documentation. | | session id | fork | events | system prompt | resume flag | |---|---|---|---|---|---| diff --git a/src/agent.rs b/src/agent.rs index 8ded80b..983b366 100644 --- a/src/agent.rs +++ b/src/agent.rs @@ -373,13 +373,13 @@ impl Agent { /// The command that resolves a missing login for this agent. /// /// Verified against each CLI's own help: Codex and Copilot expose a `login` - /// subcommand, while Claude authenticates interactively or through a - /// long-lived token. + /// subcommand, and so does Claude (`claude auth login`, verified against + /// 2.1.267), which also offers `setup-token` for a long-lived token. #[must_use] pub fn login_hint(self) -> &'static str { match self { Agent::Claude => { - "run `claude` and use /login, or `claude setup-token` for a \ + "run `claude auth login`, or `claude setup-token` for a \ long-lived token" } Agent::Codex => "run `codex login`", @@ -397,14 +397,14 @@ impl Agent { #[must_use] pub fn verified_version(self) -> crate::Version { let (major, minor, patch) = match self { - // `claude --version` -> "2.1.220 (Claude Code)" - Agent::Claude => (2, 1, 220), - // `codex --version` -> "codex-cli 0.147.0" - Agent::Codex => (0, 147, 0), - // `copilot --version` -> "GitHub Copilot CLI 1.0.78." - Agent::Copilot => (1, 0, 78), - // `grok --version` -> "grok 1.0.30 (04b7ffed98c6) [stable]" - Agent::Grok => (1, 0, 30), + // `claude --version` -> "2.1.267 (Claude Code)" + Agent::Claude => (2, 1, 267), + // `codex --version` -> "codex-cli 0.154.0" + Agent::Codex => (0, 154, 0), + // `copilot --version` -> "GitHub Copilot CLI 1.0.88." + Agent::Copilot => (1, 0, 88), + // `grok --version` -> "grok 1.0.40 (eb1a2256660d)" + Agent::Grok => (1, 0, 40), }; crate::Version { major, @@ -540,7 +540,11 @@ impl Agent { }, // `codex exec --json` emits `thread_id`; interactive turns use the // app-server protocol, which exposes steering and approvals. - // Continuation remains linear (`codex fork` is TUI-only). + // Continuation remains linear here. codex-cli 0.154.0 does have a + // headless `codex exec fork ` (and app-server + // `thread/fork`), but this crate does not drive it yet, so `fork` + // stays false and a fork request is `Error::Unsupported` rather + // than a silent linear resume. Agent::Codex => Caps { session: SessionSupport::Printed, fork: false, @@ -939,7 +943,8 @@ fn argv_claude(plan: &Plan) -> Vec { } if plan.format == Format::Stream { // Claude refuses `-p --output-format stream-json` without it: - // "--print with --output-format=stream-json requires --verbose". + // "Error: When using --print, --output-format=stream-json requires + // --verbose" (claude 2.1.267). a.bare("--verbose"); // Without this Claude emits only *completed* messages, so text arrives // a paragraph at a time. With it, `stream_event` records carry the @@ -1017,11 +1022,32 @@ fn argv_codex(plan: &Plan) -> Vec { if let Some(effort) = plan.effort.as_ref() { a.pair("-c", format!("model_reasoning_effort={effort}")); } - // Verified against codex-cli 0.146.0. Options remain options after the - // positional prompt, but keeping roots before it makes the command's - // security posture readable and matches the CLI's help shape. - for dir in &plan.extra_dirs { - a.pair("--add-dir", dir); + // Verified against codex-cli 0.154.0. `codex exec` takes `--add-dir`, but + // `codex exec resume` refuses it ("unexpected argument '--add-dir'"), the + // same way it refuses `--sandbox`. A resumed run carries the roots through + // the config key `--add-dir` sets, so resuming with extra directories no + // longer fails before the agent starts. JSON string encoding is a valid + // TOML basic string, so each path is quoted and escaped the same way. + // Keeping roots before the prompt makes the security posture readable. + if resuming { + if !plan.extra_dirs.is_empty() { + let roots: Vec = plan + .extra_dirs + .iter() + .map(|dir| serde_json::Value::from(dir.as_str()).to_string()) + .collect(); + a.pair( + "-c", + format!( + "sandbox_workspace_write.writable_roots=[{}]", + roots.join(",") + ), + ); + } + } else { + for dir in &plan.extra_dirs { + a.pair("--add-dir", dir); + } } // Codex reads the schema from a file, which the runner writes before the // spawn. `Request::argv` has no file to name, so it shows a placeholder: @@ -1082,11 +1108,13 @@ fn argv_copilot(plan: &Plan) -> Vec { }; a.opt("--model", plan.model.as_ref()); - // Verified against Copilot CLI 1.0.78: `--effort` is the documented spelling - // and `--reasoning-effort` its alias (none, minimal, low, medium, high, - // xhigh, max). A wider set than Claude's, which is why the level is passed - // through rather than mapped to a shared enum. - a.opt("--effort", plan.effort.as_ref()); + // Verified against Copilot CLI 1.0.88: `--reasoning-effort` is now the + // documented spelling (none, minimal, low, medium, high, xhigh, max). + // `--effort`, documented in 1.0.78, still parses but is gone from `--help`, + // so the documented flag is the one passed. A wider set than Claude's, + // which is why the level is passed through rather than mapped to a shared + // enum. + a.opt("--reasoning-effort", plan.effort.as_ref()); // One flag serves both directions: it sets the UUID for a new session and // resumes an existing one by id. match &plan.cont { @@ -1362,7 +1390,10 @@ mod tests { assert_eq!(claude[pos(&claude, "--effort").unwrap() + 1], "xhigh"); let copilot = argv(Agent::Copilot, &p); - assert_eq!(copilot[pos(&copilot, "--effort").unwrap() + 1], "xhigh"); + assert_eq!( + copilot[pos(&copilot, "--reasoning-effort").unwrap() + 1], + "xhigh" + ); let codex = argv(Agent::Codex, &p); assert!( @@ -1384,6 +1415,7 @@ mod tests { for agent in [Agent::Claude, Agent::Codex, Agent::Copilot] { let a = argv(agent, &p); assert!(pos(&a, "--effort").is_none(), "{agent}: {a:?}"); + assert!(pos(&a, "--reasoning-effort").is_none(), "{agent}: {a:?}"); assert!( !a.iter() .any(|arg| arg.starts_with("model_reasoning_effort")), @@ -1495,6 +1527,31 @@ mod tests { assert_eq!(a.last().unwrap(), "hi"); } + /// codex-cli 0.154.0 refuses `--add-dir` on `exec resume`, so a resumed + /// run carries its roots in the config key instead, TOML-quoted. + #[test] + fn codex_resume_carries_extra_roots_as_config_not_add_dir() { + let mut p = plan("codex"); + p.cont = Continue::Resume("thread-9".into()); + p.extra_dirs = vec!["/repo".into(), "/with \"quote\"".into()]; + let a = argv(Agent::Codex, &p); + assert!(!a.contains(&"--add-dir".to_string()), "{a:?}"); + let at = a + .iter() + .position(|arg| arg.starts_with("sandbox_workspace_write.writable_roots=")) + .expect("the roots ride a config override"); + assert_eq!(a[at - 1], "-c"); + assert_eq!( + a[at], + r#"sandbox_workspace_write.writable_roots=["/repo","/with \"quote\""]"# + ); + + // A fresh exec still takes the flag. + p.cont = Continue::New; + let fresh = argv(Agent::Codex, &p); + assert!(fresh.contains(&"--add-dir".to_string()), "{fresh:?}"); + } + /// `Minimal` exists to withhold secrets, so nothing it passes through may /// be a credential carrier. Proxy URLs in particular routinely embed /// `user:pass`, which is why they are offered separately instead. diff --git a/src/auth.rs b/src/auth.rs index aecb143..1969224 100644 --- a/src/auth.rs +++ b/src/auth.rs @@ -291,7 +291,11 @@ mod tests { let status = AuthStatus::read(Agent::Claude, r#"{"loggedIn": false}"#, true); assert_eq!(status.state, AuthState::LoggedOut); assert!(status.needs_login()); - assert!(status.summary().contains("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/login"), "{}", status.summary()); + assert!( + status.summary().contains("claude auth login"), + "{}", + status.summary() + ); } /// Verbatim from `codex login status`. diff --git a/src/codex_app_server.rs b/src/codex_app_server.rs index 8beaf2a..f24284d 100644 --- a/src/codex_app_server.rs +++ b/src/codex_app_server.rs @@ -356,10 +356,20 @@ impl Protocol { .map(str::to_string); self.finished = true; } + // Verified against the codex-cli 0.154.0 app-server schema: + // `ErrorNotification { error: TurnError { message, .. }, threadId, + // turnId, willRetry }`. The message is under `error`, so reading + // `params.message` always came back empty. A notification codex + // says it will retry is not the end of the turn: marking it an + // error failed runs that went on to succeed. "error" => { + if params.get("willRetry").and_then(Value::as_bool) == Some(true) { + return step; + } self.terminal.stop = Stop::Error; self.terminal.error_message = params - .get("message") + .pointer("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/error/message") + .or_else(|| params.get("message")) .and_then(Value::as_str) .map(str::to_string); } @@ -811,6 +821,43 @@ mod tests { protocol } + /// codex-cli 0.154.0 sends `{ error: { message }, willRetry }`. A retried + /// error is not the end of the turn; a final one carries its message. + #[test] + fn an_error_notification_reads_its_nested_message_and_skips_retries() { + let mut protocol = running_protocol(); + protocol.push(&json!({ + "method": "error", + "params": { + "error": {"message": "overloaded"}, + "threadId": "thread-7", + "turnId": "turn-9", + "willRetry": true + } + })); + assert_ne!( + protocol.terminal.stop, + Stop::Error, + "a retry is not a failure" + ); + assert_eq!(protocol.terminal.error_message, None); + + protocol.push(&json!({ + "method": "error", + "params": { + "error": {"message": "quota exhausted"}, + "threadId": "thread-7", + "turnId": "turn-9", + "willRetry": false + } + })); + assert_eq!(protocol.terminal.stop, Stop::Error); + assert_eq!( + protocol.terminal.error_message.as_deref(), + Some("quota exhausted") + ); + } + #[test] fn interruption_targets_the_active_codex_turn() { let mut protocol = running_protocol(); diff --git a/src/event.rs b/src/event.rs index 2a042a6..648ee78 100644 --- a/src/event.rs +++ b/src/event.rs @@ -1292,6 +1292,15 @@ fn claude_usage(v: &Value, model: Option<&str>) -> Usage { // The key is the resolved model name the `init` record announced, verified // against claude 2.1.212: asking for `sonnet[1m]`, init says // `claude-sonnet-5[1m]` and that exact string keys `modelUsage`. + // + // `--output-format json`, which a plain run uses, has no `init` record, so + // there is often no name at all. When the helper also ran, that left two + // entries and nothing to choose by, and a `[1m]` run reported no window + // (claude 2.1.267, intermittently, depending on whether the helper ran). + // The top-level `usage` is the run's own model: its four counts equal that + // model's `modelUsage` entry and not the helper's. So an entry is chosen + // by those counts only when exactly one matches; anything else is still + // reported as unknown rather than guessed. let per_model = v .get("modelUsage") .and_then(Value::as_object) @@ -1300,9 +1309,32 @@ fn claude_usage(v: &Value, model: Option<&str>) -> Usage { (Some(entry), _) => Some(entry), // One entry and no name to match: it can only be the run's model. (None, 1) => models.values().next(), - // Several entries and no match. Guessing here is how the bug - // happened, so the window is reported as unknown instead. - (None, _) => None, + // Several entries and no name: the one whose counts are the + // run's own, if exactly one is. + (None, _) => { + let top = v.get("usage"); + let count = |value: Option<&Value>, key: &str| { + value.and_then(|u| u.get(key)).and_then(Value::as_u64) + }; + let same = |entry: &Value| { + [ + ("input_tokens", "inputTokens"), + ("output_tokens", "outputTokens"), + ("cache_read_input_tokens", "cacheReadInputTokens"), + ("cache_creation_input_tokens", "cacheCreationInputTokens"), + ] + .iter() + .all(|(ours, theirs)| { + count(top, ours).is_some() + && count(top, ours) == count(Some(entry), theirs) + }) + }; + let mut matching = models.values().filter(|entry| same(entry)); + match (matching.next(), matching.next()) { + (Some(entry), None) => Some(entry), + _ => None, + } + } }, ); let of_model = |key: &str| per_model.and_then(|m| m.get(key)).and_then(Value::as_u64); @@ -1678,6 +1710,21 @@ mod tests { assert_eq!(single.usage.context_window, Some(200_000)); } + /// `--output-format json` has no init record, and claude 2.1.267 lists the + /// Haiku helper beside the run's model when the helper ran. The run's own + /// entry is the one whose counts equal the top-level `usage`. + #[test] + fn a_json_result_finds_its_model_by_its_own_counts() { + let (_, term) = run( + Agent::Claude, + &[ + r#"{"type":"result","subtype":"success","is_error":false,"result":"ok","session_id":"s","usage":{"input_tokens":2,"output_tokens":4,"cache_read_input_tokens":27128,"cache_creation_input_tokens":9825},"modelUsage":{"claude-haiku-4-5-20251001":{"inputTokens":521,"outputTokens":12,"cacheReadInputTokens":0,"cacheCreationInputTokens":0,"contextWindow":200000},"claude-sonnet-5[1m]":{"inputTokens":2,"outputTokens":4,"cacheReadInputTokens":27128,"cacheCreationInputTokens":9825,"contextWindow":1000000,"maxOutputTokens":64000}}}"#, + ], + ); + assert_eq!(term.usage.context_window, Some(1_000_000)); + assert_eq!(term.usage.max_output_tokens, Some(64_000)); + } + #[test] fn claude_token_deltas_stream_without_duplicating_the_finished_message() { let (events, _) = run( diff --git a/src/model.rs b/src/model.rs index 8ee175a..837c15e 100644 --- a/src/model.rs +++ b/src/model.rs @@ -167,8 +167,8 @@ impl Agent { }, Agent::Codex => Verified { source: Source::Cli, - checked: "2026-08-07", - against: "codex-cli 0.146.0", + checked: "2026-09-23", + against: "codex-cli 0.154.0", }, // Read from the `/model` picker. Copilot has no headless list; see // `discover_models`. @@ -179,8 +179,8 @@ impl Agent { }, Agent::Grok => Verified { source: Source::Cli, - checked: "2026-09-13", - against: "grok 1.0.30", + checked: "2026-09-23", + against: "grok 1.0.40", }, } } @@ -392,26 +392,38 @@ fn claude_pinned() -> Vec { /// Codex, in the priority order the CLI itself reports. /// -/// Verified by running `codex debug models` against codex-cli 0.146.0 on -/// 2026-08-07. `codex-auto-review` is reported with `visibility: "hide"` and is -/// left out for that reason; [`discover_codex`] applies the same filter. +/// Verified by running `codex debug models` against codex-cli 0.154.0 on +/// 2026-09-23. `gpt-reserve` and `codex-auto-review` are reported with +/// `visibility: "hide"` and are left out for that reason; [`discover_codex`] +/// applies the same filter. `gpt-daybreak-blue-latest` is listed only in the +/// server-refreshed catalogue (the bundled one hides it), so it is +/// account-dependent. `gpt-5.5` carries an upgrade notice retiring it on +/// 2026-10-14 in favour of `gpt-5.6-sol`. The descriptions are Codex's own. fn codex_models() -> Vec { const FULL: &[&str] = &["low", "medium", "high", "xhigh", "max", "ultra"]; const TO_MAX: &[&str] = &["low", "medium", "high", "xhigh", "max"]; const TO_XHIGH: &[&str] = &["low", "medium", "high", "xhigh"]; vec![ + Model::new( + "gpt-6-astra", + "GPT-6-Astra", + "Frontier intelligence for the most demanding work.", + Kind::Pinned, + FULL, + true, + ), Model::new( "gpt-5.6-sol", "GPT-5.6-Sol", - "Latest frontier agentic coding model.", + "Older coding model for complex work.", Kind::Pinned, FULL, - true, + false, ), Model::new( "gpt-5.6-terra", "GPT-5.6-Terra", - "Balanced agentic coding model for everyday work.", + "Older balanced model for straightforward work.", Kind::Pinned, FULL, false, @@ -419,31 +431,23 @@ fn codex_models() -> Vec { Model::new( "gpt-5.6-luna", "GPT-5.6-Luna", - "Fast and affordable agentic coding model.", + "Older fast and efficient model.", Kind::Pinned, TO_MAX, false, ), Model::new( - "gpt-5.5", - "GPT-5.5", - "Frontier model for complex coding, research, and real-world tasks.", + "gpt-daybreak-blue-latest", + "Daybreak Blue", + "Latest frontier agentic coding model for broad defensive cybersecurity work.", Kind::Pinned, - TO_XHIGH, - false, - ), - Model::new( - "gpt-5.4", - "GPT-5.4", - "Strong model for everyday coding.", - Kind::Pinned, - TO_XHIGH, + FULL, false, ), Model::new( - "gpt-5.4-mini", - "GPT-5.4-Mini", - "Small, fast, and cost-efficient model for simpler coding tasks.", + "gpt-5.5", + "GPT-5.5", + "Legacy coding model.", Kind::Pinned, TO_XHIGH, false, @@ -518,22 +522,41 @@ fn pinned(id: &'static str, name: &'static str) -> Model { Model::new(id, name, "", Kind::Pinned, COPILOT_EFFORTS, false) } -/// Grok, from `grok models` on 1.0.30 (2026-09-13). +/// Grok, from `grok models` on 1.0.40 (2026-09-23). /// /// Effort tokens from grok `--help` (`--reasoning-effort` / `--effort`) and -/// the session config option `reasoning_effort`. +/// the session config option `reasoning_effort`. The 1.0.40 binary's validator +/// also names `none` and `max`, but that is a string in the binary, not a +/// verified run, so they are not claimed here. const GROK_EFFORTS: &[&str] = &["minimal", "low", "medium", "high", "xhigh"]; fn grok_models() -> Vec { vec![ Model::new( - "grok-4.6", - "Grok 4.6", + "grok-4.7", + "Grok 4.7", "Default Grok Build model", Kind::Pinned, GROK_EFFORTS, true, ), + // `grok models` prints ids only; the display name follows the others. + Model::new( + "grok-4.7-build-fast", + "Grok 4.7 Build Fast", + "", + Kind::Pinned, + GROK_EFFORTS, + false, + ), + Model::new( + "grok-4.6", + "Grok 4.6", + "", + Kind::Pinned, + GROK_EFFORTS, + false, + ), Model::new( "grok-4.5", "Grok 4.5", diff --git a/src/run.rs b/src/run.rs index 48d1caa..b2cf052 100644 --- a/src/run.rs +++ b/src/run.rs @@ -2473,7 +2473,7 @@ mod tests { panic!("expected NotAuthenticated, got {err:?}") }; assert_eq!(*agent, Agent::Claude); - assert!(hint.contains("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/login"), "{hint}"); + assert!(hint.contains("claude auth login"), "{hint}"); assert!(err.is_auth_failure()); }