mirror of
https://github.com/pchuan98/codex.git
synced 2026-07-01 00:31:56 +08:00
[codex] Add Ultra reasoning effort (#29899)
## Why Ultra should be one user-facing reasoning selection for work that benefits from both maximum reasoning and proactive multi-agent delegation. Without it, clients must coordinate maximum reasoning with the experimental `multiAgentMode` setting, even though the inference backend still expects its existing `max` effort value. This change makes reasoning effort the source of truth: clients select `ultra`, core derives proactive multi-agent behavior when the turn is eligible for multi-agent V2, and inference requests continue to use the backend-compatible `max` value. ## What changed - Add `ultra` as a first-class reasoning effort and preserve model-catalog ordering when exposing it to clients. - Convert `ultra` to `max` at the inference request boundary, including Responses HTTP/WebSocket requests, startup prewarm, compaction, and memory summarization. - Derive effective multi-agent mode per turn from effective reasoning effort: - eligible multi-agent V2 + `ultra` → `proactive` - eligible multi-agent V2 + any other effort → `explicitRequestOnly` - V1 or otherwise ineligible sessions → no multi-agent mode instruction - Keep the derived effective mode in turn context history so successive turns can emit a developer-message update only when the effective mode changes. - Remove selected multi-agent mode from core session configuration, turn construction, thread settings, resume/fork restoration, and subagent spawn plumbing. Subagents inherit reasoning effort and derive their own effective mode. - Retain the experimental app-server `multiAgentMode` fields for wire compatibility while marking them deprecated. Request values are accepted but ignored; compatibility response fields report `explicitRequestOnly`. - Display Ultra in the TUI using the order supplied by `model/list`. ## Validation - `just test -p codex-core ultra_reasoning_uses_max_for_requests` - `just test -p codex-tui model_reasoning_selection_popup`
This commit is contained in:
@@ -663,7 +663,6 @@ impl TestCodexBuilder {
|
||||
thread_source: None,
|
||||
dynamic_tools: Vec::new(),
|
||||
metrics_service_name: None,
|
||||
multi_agent_mode: None,
|
||||
parent_trace: None,
|
||||
environments,
|
||||
thread_extension_init: Default::default(),
|
||||
|
||||
@@ -442,7 +442,6 @@ async fn loads_user_instructions_without_a_primary_environment() -> Result<()> {
|
||||
thread_source: None,
|
||||
dynamic_tools: Vec::new(),
|
||||
metrics_service_name: None,
|
||||
multi_agent_mode: None,
|
||||
parent_trace: None,
|
||||
environments: Vec::new(),
|
||||
thread_extension_init: Default::default(),
|
||||
@@ -648,7 +647,6 @@ async fn multi_environment_thread_loads_every_project_and_keeps_creation_snapsho
|
||||
thread_source: None,
|
||||
dynamic_tools: Vec::new(),
|
||||
metrics_service_name: None,
|
||||
multi_agent_mode: None,
|
||||
parent_trace: None,
|
||||
environments: vec![
|
||||
TurnEnvironmentSelection {
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
use anyhow::Result;
|
||||
use codex_core::config::Config;
|
||||
use codex_features::Feature;
|
||||
use codex_protocol::config_types::MultiAgentMode;
|
||||
use codex_protocol::openai_models::ModelInfo;
|
||||
use codex_protocol::openai_models::ReasoningEffort;
|
||||
use codex_protocol::openai_models::ReasoningEffortPreset;
|
||||
use codex_protocol::protocol::EventMsg;
|
||||
use codex_protocol::protocol::MULTI_AGENT_MODE_OPEN_TAG;
|
||||
use codex_protocol::protocol::Op;
|
||||
@@ -19,9 +22,30 @@ use pretty_assertions::assert_eq;
|
||||
use serde_json::Value;
|
||||
|
||||
const NO_SPAWN_TEXT: &str = "Do not spawn sub-agents unless the user explicitly asks for sub-agents, delegation, or parallel agent work.";
|
||||
const NO_MODE_TEXT: &str = "Multi-agent delegation mode instructions are inactive.";
|
||||
const PROACTIVE_TEXT: &str = "Proactive multi-agent delegation is active.";
|
||||
|
||||
fn add_ultra_reasoning(model_info: &mut ModelInfo) {
|
||||
model_info.supports_reasoning_summaries = true;
|
||||
model_info
|
||||
.supported_reasoning_levels
|
||||
.push(ReasoningEffortPreset {
|
||||
effort: ReasoningEffort::Ultra,
|
||||
description: "Ultra".to_string(),
|
||||
});
|
||||
}
|
||||
|
||||
fn configure_multi_agent_v2(config: &mut Config) {
|
||||
config
|
||||
.features
|
||||
.enable(Feature::MultiAgentV2)
|
||||
.expect("test config should allow feature update");
|
||||
}
|
||||
|
||||
fn configure_ultra(config: &mut Config) {
|
||||
configure_multi_agent_v2(config);
|
||||
config.model_reasoning_effort = Some(ReasoningEffort::Ultra);
|
||||
}
|
||||
|
||||
fn developer_texts(input: &[Value]) -> Vec<&str> {
|
||||
input
|
||||
.iter()
|
||||
@@ -39,7 +63,7 @@ fn count_containing(texts: &[&str], target: &str) -> usize {
|
||||
async fn submit_turn(
|
||||
codex: &codex_core::CodexThread,
|
||||
prompt: &str,
|
||||
mode: Option<MultiAgentMode>,
|
||||
effort: Option<ReasoningEffort>,
|
||||
) -> Result<()> {
|
||||
codex
|
||||
.submit(Op::UserInput {
|
||||
@@ -51,7 +75,7 @@ async fn submit_turn(
|
||||
responsesapi_client_metadata: None,
|
||||
additional_context: Default::default(),
|
||||
thread_settings: ThreadSettingsOverrides {
|
||||
multi_agent_mode: mode,
|
||||
effort: effort.map(Some),
|
||||
..Default::default()
|
||||
},
|
||||
})
|
||||
@@ -61,214 +85,43 @@ async fn submit_turn(
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn multi_agent_mode_is_sticky_and_emits_only_on_change() -> Result<()> {
|
||||
async fn ultra_reasoning_uses_max_and_proactive_mode() -> Result<()> {
|
||||
skip_if_no_network!(Ok(()));
|
||||
|
||||
let server = start_mock_server().await;
|
||||
let responses = mount_sse_sequence(
|
||||
&server,
|
||||
(1..=5)
|
||||
.map(|index| {
|
||||
sse(vec![
|
||||
ev_response_created(&format!("resp-{index}")),
|
||||
ev_completed(&format!("resp-{index}")),
|
||||
])
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.await;
|
||||
let test = test_codex()
|
||||
.with_config(|config| {
|
||||
config
|
||||
.features
|
||||
.enable(Feature::MultiAgentV2)
|
||||
.expect("test config should allow feature update");
|
||||
})
|
||||
.build(&server)
|
||||
.await?;
|
||||
|
||||
submit_turn(&test.codex, "turn one", /*mode*/ None).await?;
|
||||
assert_eq!(
|
||||
test.codex.config_snapshot().await.multi_agent_mode,
|
||||
MultiAgentMode::ExplicitRequestOnly
|
||||
);
|
||||
submit_turn(&test.codex, "turn two", Some(MultiAgentMode::Proactive)).await?;
|
||||
submit_turn(&test.codex, "turn three", /*mode*/ None).await?;
|
||||
submit_turn(&test.codex, "turn four", Some(MultiAgentMode::None)).await?;
|
||||
submit_turn(&test.codex, "turn five", /*mode*/ None).await?;
|
||||
|
||||
assert_eq!(
|
||||
test.codex.config_snapshot().await.multi_agent_mode,
|
||||
MultiAgentMode::None
|
||||
);
|
||||
|
||||
let requests = responses.requests();
|
||||
let inputs = requests
|
||||
.iter()
|
||||
.map(core_test_support::responses::ResponsesRequest::input)
|
||||
.collect::<Vec<_>>();
|
||||
let first = developer_texts(&inputs[0]);
|
||||
let second = developer_texts(&inputs[1]);
|
||||
let third = developer_texts(&inputs[2]);
|
||||
let fourth = developer_texts(&inputs[3]);
|
||||
let fifth = developer_texts(&inputs[4]);
|
||||
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&first, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&first, NO_SPAWN_TEXT),
|
||||
count_containing(&first, PROACTIVE_TEXT),
|
||||
),
|
||||
(1, 1, 0)
|
||||
);
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&second, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&second, NO_SPAWN_TEXT),
|
||||
count_containing(&second, PROACTIVE_TEXT),
|
||||
),
|
||||
(2, 1, 1)
|
||||
);
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&third, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&third, NO_SPAWN_TEXT),
|
||||
count_containing(&third, PROACTIVE_TEXT),
|
||||
),
|
||||
(2, 1, 1)
|
||||
);
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&fourth, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&fourth, NO_SPAWN_TEXT),
|
||||
count_containing(&fourth, PROACTIVE_TEXT),
|
||||
count_containing(&fourth, NO_MODE_TEXT),
|
||||
),
|
||||
(3, 1, 1, 1)
|
||||
);
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&fifth, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&fifth, NO_SPAWN_TEXT),
|
||||
count_containing(&fifth, PROACTIVE_TEXT),
|
||||
count_containing(&fifth, NO_MODE_TEXT),
|
||||
),
|
||||
(3, 1, 1, 1)
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn multi_agent_mode_none_omits_instructions_and_survives_resume() -> Result<()> {
|
||||
skip_if_no_network!(Ok(()));
|
||||
|
||||
let server = start_mock_server().await;
|
||||
let responses = mount_sse_sequence(
|
||||
&server,
|
||||
(1..=2)
|
||||
.map(|index| {
|
||||
sse(vec![
|
||||
ev_response_created(&format!("resp-{index}")),
|
||||
ev_completed(&format!("resp-{index}")),
|
||||
])
|
||||
})
|
||||
.collect(),
|
||||
)
|
||||
.await;
|
||||
let initial = test_codex()
|
||||
.with_config(|config| {
|
||||
config
|
||||
.features
|
||||
.enable(Feature::MultiAgentV2)
|
||||
.expect("test config should allow feature update");
|
||||
})
|
||||
.build(&server)
|
||||
.await?;
|
||||
let home = initial.home.clone();
|
||||
let rollout_path = initial
|
||||
.session_configured
|
||||
.rollout_path
|
||||
.clone()
|
||||
.expect("rollout path");
|
||||
|
||||
submit_turn(&initial.codex, "before resume", Some(MultiAgentMode::None)).await?;
|
||||
assert_eq!(
|
||||
initial.codex.config_snapshot().await.multi_agent_mode,
|
||||
MultiAgentMode::None
|
||||
);
|
||||
drop(initial);
|
||||
|
||||
let mut resume_builder = test_codex().with_config(|config| {
|
||||
config
|
||||
.features
|
||||
.enable(Feature::MultiAgentV2)
|
||||
.expect("test config should allow feature update");
|
||||
});
|
||||
let resumed = resume_builder.resume(&server, home, rollout_path).await?;
|
||||
submit_turn(&resumed.codex, "after resume", /*mode*/ None).await?;
|
||||
|
||||
assert_eq!(
|
||||
resumed.codex.config_snapshot().await.multi_agent_mode,
|
||||
MultiAgentMode::None
|
||||
);
|
||||
let requests = responses.requests();
|
||||
assert_eq!(requests.len(), 2);
|
||||
for request in requests {
|
||||
let input = request.input();
|
||||
let texts = developer_texts(&input);
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&texts, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&texts, NO_SPAWN_TEXT),
|
||||
count_containing(&texts, PROACTIVE_TEXT),
|
||||
count_containing(&texts, NO_MODE_TEXT),
|
||||
),
|
||||
(0, 0, 0, 0)
|
||||
);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn multi_agent_mode_applies_without_usage_hint_text() -> Result<()> {
|
||||
skip_if_no_network!(Ok(()));
|
||||
|
||||
let server = start_mock_server().await;
|
||||
let responses = mount_sse_once(
|
||||
let response = mount_sse_once(
|
||||
&server,
|
||||
sse(vec![ev_response_created("resp-1"), ev_completed("resp-1")]),
|
||||
)
|
||||
.await;
|
||||
let test = test_codex()
|
||||
.with_config(|config| {
|
||||
config
|
||||
.features
|
||||
.enable(Feature::MultiAgentV2)
|
||||
.expect("test config should allow feature update");
|
||||
config.multi_agent_v2.root_agent_usage_hint_text = None;
|
||||
})
|
||||
.with_model_info_override("gpt-5.4", add_ultra_reasoning)
|
||||
.with_config(configure_ultra)
|
||||
.build(&server)
|
||||
.await?;
|
||||
|
||||
submit_turn(&test.codex, "hello", Some(MultiAgentMode::Proactive)).await?;
|
||||
submit_turn(&test.codex, "hello", /*effort*/ None).await?;
|
||||
|
||||
let input = responses.single_request().input();
|
||||
let request = response.single_request();
|
||||
assert_eq!(
|
||||
request.body_json()["reasoning"]["effort"].as_str(),
|
||||
Some("max")
|
||||
);
|
||||
let input = request.input();
|
||||
let texts = developer_texts(&input);
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&texts, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&texts, NO_SPAWN_TEXT),
|
||||
count_containing(&texts, PROACTIVE_TEXT),
|
||||
),
|
||||
(1, 1)
|
||||
(0, 1)
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn resume_compares_against_previous_effective_multi_agent_mode() -> Result<()> {
|
||||
async fn leaving_ultra_after_cold_resume_emits_explicit_mode() -> Result<()> {
|
||||
skip_if_no_network!(Ok(()));
|
||||
|
||||
let server = start_mock_server().await;
|
||||
@@ -285,12 +138,8 @@ async fn resume_compares_against_previous_effective_multi_agent_mode() -> Result
|
||||
)
|
||||
.await;
|
||||
let initial = test_codex()
|
||||
.with_config(|config| {
|
||||
config
|
||||
.features
|
||||
.enable(Feature::MultiAgentV2)
|
||||
.expect("test config should allow feature update");
|
||||
})
|
||||
.with_model_info_override("gpt-5.4", add_ultra_reasoning)
|
||||
.with_config(configure_ultra)
|
||||
.build(&server)
|
||||
.await?;
|
||||
let home = initial.home.clone();
|
||||
@@ -300,29 +149,27 @@ async fn resume_compares_against_previous_effective_multi_agent_mode() -> Result
|
||||
.clone()
|
||||
.expect("rollout path");
|
||||
|
||||
submit_turn(
|
||||
&initial.codex,
|
||||
"before resume",
|
||||
Some(MultiAgentMode::Proactive),
|
||||
)
|
||||
.await?;
|
||||
submit_turn(&initial.codex, "before resume", /*effort*/ None).await?;
|
||||
drop(initial);
|
||||
|
||||
let mut resume_builder = test_codex().with_config(|config| {
|
||||
config
|
||||
.features
|
||||
.enable(Feature::MultiAgentV2)
|
||||
.expect("test config should allow feature update");
|
||||
});
|
||||
let mut resume_builder = test_codex()
|
||||
.with_model_info_override("gpt-5.4", add_ultra_reasoning)
|
||||
.with_config(configure_ultra);
|
||||
let resumed = resume_builder.resume(&server, home, rollout_path).await?;
|
||||
submit_turn(&resumed.codex, "after resume", /*mode*/ None).await?;
|
||||
|
||||
assert_eq!(
|
||||
resumed.codex.config_snapshot().await.multi_agent_mode,
|
||||
MultiAgentMode::Proactive
|
||||
);
|
||||
submit_turn(&resumed.codex, "after resume", Some(ReasoningEffort::High)).await?;
|
||||
|
||||
let requests = responses.requests();
|
||||
assert_eq!(
|
||||
(
|
||||
requests[0].body_json()["reasoning"]["effort"]
|
||||
.as_str()
|
||||
.map(str::to_string),
|
||||
requests[1].body_json()["reasoning"]["effort"]
|
||||
.as_str()
|
||||
.map(str::to_string),
|
||||
),
|
||||
(Some("max".to_string()), Some("high".to_string()))
|
||||
);
|
||||
let resumed_input = requests[1].input();
|
||||
let texts = developer_texts(&resumed_input);
|
||||
assert_eq!(
|
||||
@@ -331,39 +178,40 @@ async fn resume_compares_against_previous_effective_multi_agent_mode() -> Result
|
||||
count_containing(&texts, NO_SPAWN_TEXT),
|
||||
count_containing(&texts, PROACTIVE_TEXT),
|
||||
),
|
||||
(1, 0, 1)
|
||||
(2, 1, 1)
|
||||
);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn multi_agent_mode_is_retained_without_multi_agent_v2() -> Result<()> {
|
||||
async fn ultra_on_multi_agent_v1_uses_max_without_mode_instructions() -> Result<()> {
|
||||
skip_if_no_network!(Ok(()));
|
||||
|
||||
let server = start_mock_server().await;
|
||||
let responses = mount_sse_once(
|
||||
let response = mount_sse_once(
|
||||
&server,
|
||||
sse(vec![ev_response_created("resp-1"), ev_completed("resp-1")]),
|
||||
)
|
||||
.await;
|
||||
let test = test_codex().build(&server).await?;
|
||||
let test = test_codex()
|
||||
.with_model_info_override("gpt-5.4", add_ultra_reasoning)
|
||||
.with_config(|config| {
|
||||
config.model_reasoning_effort = Some(ReasoningEffort::Ultra);
|
||||
})
|
||||
.build(&server)
|
||||
.await?;
|
||||
|
||||
submit_turn(&test.codex, "hello", Some(MultiAgentMode::Proactive)).await?;
|
||||
submit_turn(&test.codex, "hello", /*effort*/ None).await?;
|
||||
|
||||
let request = response.single_request();
|
||||
assert_eq!(
|
||||
test.codex.config_snapshot().await.multi_agent_mode,
|
||||
MultiAgentMode::Proactive
|
||||
request.body_json()["reasoning"]["effort"].as_str(),
|
||||
Some("max")
|
||||
);
|
||||
let input = responses.single_request().input();
|
||||
let input = request.input();
|
||||
let texts = developer_texts(&input);
|
||||
assert_eq!(
|
||||
(
|
||||
count_containing(&texts, MULTI_AGENT_MODE_OPEN_TAG),
|
||||
count_containing(&texts, PROACTIVE_TEXT),
|
||||
),
|
||||
(0, 0)
|
||||
);
|
||||
assert_eq!(count_containing(&texts, MULTI_AGENT_MODE_OPEN_TAG), 0);
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -766,7 +766,6 @@ async fn subagent_stop_replaces_stop_and_skips_internal_subagents() -> Result<()
|
||||
thread_source: None,
|
||||
dynamic_tools: Vec::new(),
|
||||
metrics_service_name: None,
|
||||
multi_agent_mode: None,
|
||||
parent_trace: None,
|
||||
environments: Vec::new(),
|
||||
thread_extension_init: Default::default(),
|
||||
|
||||
Reference in New Issue
Block a user