diff --git a/crates/op-host-desktop/src/chat_session_launch_design.rs b/crates/op-host-desktop/src/chat_session_launch_design.rs index e0d0dbb2d..2b1792996 100644 --- a/crates/op-host-desktop/src/chat_session_launch_design.rs +++ b/crates/op-host-desktop/src/chat_session_launch_design.rs @@ -127,7 +127,7 @@ fn resolve_design_thinking(model: Option<&str>, chat_default: ThinkingMode) -> T /// falls through to the orchestrator path). /// /// Mirrors `launch_if_pending`'s builtin chat branch but uses the -/// design toolset and a 8192-token budget. +/// design toolset and a 16384-token per-turn budget. pub(super) fn launch_design_loop_turn( host: &mut WidgetHostNative, user_text: String, @@ -161,7 +161,14 @@ pub(super) fn launch_design_loop_turn( system_prompt: op_ai_skills::design_agent_system_prompt().to_string(), user_message: user_text, history, - max_output_tokens: 8192, + // Per-TURN output cap. The real budget-burner was hidden reasoning, not + // the DSL: a glm-5.2 loop run streamed ~94k thinking chars vs ~4k of + // actual batch_design, blowing the cap so a turn truncated mid-JSON, its + // tool call failed to parse, and the loop stopped early with an + // unfinished design. The root fix is at the wire (the loop now sends + // thinking:{type:disabled} for glm/minimax); 16384 (up from 8192) is + // headroom so a genuinely large section-batch still completes per turn. + max_output_tokens: 16384, thinking, effort, attachments, diff --git a/crates/op-host-services/src/chat_agent_loop.rs b/crates/op-host-services/src/chat_agent_loop.rs index 8e6128a0f..4c0672c8c 100644 --- a/crates/op-host-services/src/chat_agent_loop.rs +++ b/crates/op-host-services/src/chat_agent_loop.rs @@ -53,6 +53,14 @@ pub struct AgentLoopConfig { /// shared execution path for BOTH regular builtin chat and the design /// loop, so the finalize side-effect MUST be opt-in per provider. pub finalize_on_exit: bool, + /// When true AND the model is a MiniMax / GLM reasoning model, send + /// `thinking:{type:"disabled"}` in the OpenAI-compat body. Without it a + /// reasoning model (glm-5.2) streams tens of thousands of hidden reasoning + /// tokens per turn — measured 94k thinking chars vs 4k of actual + /// `batch_design` DSL — which burns the per-turn `max_output_tokens` and + /// truncates the design mid-build. The single-shot builtin chat path + /// already gates this on the same flag; the loop body was missing it. + pub disable_thinking: bool, } impl AgentLoopConfig { @@ -658,13 +666,26 @@ pub async fn run_openai_agent_loop( messages.push(json!({ "role": "user", "content": cfg.user_prompt })); for _turn in 0..cfg.max_turns.max(1) { - let body = json!({ + let mut body = json!({ "model": cfg.model, "stream": true, "max_tokens": cfg.max_output_tokens, "messages": messages, "tools": tools_json, }); + // Turn OFF hidden reasoning for MiniMax / GLM. Without this a glm-5.2 + // design turn spends its whole `max_tokens` on `reasoning_content` and + // truncates the `batch_design` mid-JSON — the single-shot builtin body + // gates the same field on the same flag (`chat_builtin_http`), but the + // loop body was missing it, so every loop turn leaked thinking. + if cfg.disable_thinking + && (crate::chat_builtin_http::is_minimax_model(&cfg.model) + || crate::chat_builtin_http::is_glm_model(&cfg.model)) + { + if let Some(obj) = body.as_object_mut() { + obj.insert("thinking".into(), json!({ "type": "disabled" })); + } + } let resp = reqwest::Client::new() .post(&cfg.url) .bearer_auth(&cfg.api_key) diff --git a/crates/op-host-services/src/chat_agent_loop_tests.rs b/crates/op-host-services/src/chat_agent_loop_tests.rs index b70e75abd..063bd511f 100644 --- a/crates/op-host-services/src/chat_agent_loop_tests.rs +++ b/crates/op-host-services/src/chat_agent_loop_tests.rs @@ -242,6 +242,7 @@ fn anthropic_loop_executes_tool_and_continues_with_tool_result() { executor: executor.clone(), max_turns: 5, finalize_on_exit: true, + disable_thinking: false, }; let (outcome, deltas) = run_loop_collect(cfg, true); assert_eq!(outcome, Ok(true)); @@ -318,6 +319,7 @@ fn loop_skips_finalize_when_disabled_for_plain_chat() { executor: executor.clone(), max_turns: 5, finalize_on_exit: false, + disable_thinking: false, }; let (outcome, _deltas) = run_loop_collect(cfg, true); assert_eq!(outcome, Ok(true)); @@ -347,6 +349,7 @@ fn anthropic_loop_stops_at_turn_cap_with_max_tokens_reason() { executor: executor.clone(), max_turns: 2, finalize_on_exit: true, + disable_thinking: false, }; let (outcome, deltas) = run_loop_collect(cfg, true); assert_eq!(outcome, Ok(true)); @@ -406,6 +409,7 @@ fn openai_loop_executes_tool_and_continues_with_role_tool_message() { executor: executor.clone(), max_turns: 5, finalize_on_exit: true, + disable_thinking: false, }; let (outcome, deltas) = run_loop_collect(cfg, false); assert_eq!(outcome, Ok(true)); @@ -518,6 +522,7 @@ fn anthropic_loop_replays_screenshot_result_as_image_content_block() { executor: executor.clone(), max_turns: 5, finalize_on_exit: true, + disable_thinking: false, }; let (outcome, _deltas) = run_loop_collect(cfg, true); assert_eq!(outcome, Ok(true)); @@ -615,6 +620,7 @@ fn openai_loop_replays_screenshot_result_as_image_url_part() { executor: executor.clone(), max_turns: 5, finalize_on_exit: true, + disable_thinking: false, }; let (outcome, _deltas) = run_loop_collect(cfg, false); assert_eq!(outcome, Ok(true)); diff --git a/crates/op-host-services/src/chat_builtin_http.rs b/crates/op-host-services/src/chat_builtin_http.rs index bfc912bd4..001a4844f 100644 --- a/crates/op-host-services/src/chat_builtin_http.rs +++ b/crates/op-host-services/src/chat_builtin_http.rs @@ -180,6 +180,7 @@ impl ChatProvider for ConfiguredBuiltinProvider { executor, max_turns: MAX_TOOL_TURNS, finalize_on_exit: provider.finalize_on_exit, + disable_thinking, }; match provider.kind { BuiltinAgentKind::Anthropic => run_anthropic_agent_loop(cfg, &tx).await, @@ -237,7 +238,7 @@ impl ChatProvider for ConfiguredBuiltinProvider { /// MiniMax M 系("MiniMax-M*"、旧 "abab*")是推理模型,其思考由 MiniMax 专属的 /// `thinking` body 字段控制。据模型名判定,以便只对它发关思考字段。 -fn is_minimax_model(model: &str) -> bool { +pub(crate) fn is_minimax_model(model: &str) -> bool { let m = model.to_ascii_lowercase(); m.starts_with("minimax") || m.starts_with("abab") } @@ -247,7 +248,7 @@ fn is_minimax_model(model: &str) -> bool { /// 一个设计子任务 thinking_len≈3 万、text_len=0,整段 parse 失败)。它接受和 /// MiniMax 同样的 `thinking:{type:"disabled"}`(curl 对 ark glm-5.2 验证:关思考后 /// reasoning_tokens=0、content 为干净 JSON)。按名判定,只对 GLM 下发。 -fn is_glm_model(model: &str) -> bool { +pub(crate) fn is_glm_model(model: &str) -> bool { model.to_ascii_lowercase().contains("glm") }