fix(agent): disable reasoning + raise budget in the design tool-loop

The OpenAI-compat tool-loop body never sent the reasoning-off field, so a GLM/
MiniMax reasoning model spent its whole per-turn budget on hidden reasoning and
truncated the design mid-op. Send thinking:{type:disabled} for those models in
the loop body (matching the single-shot path) and raise the per-turn output
budget for headroom.
This commit is contained in:
Fini 2026-07-02 21:21:45 +08:00
parent 32192b9348
commit d03c2d7dd3
4 changed files with 40 additions and 5 deletions

View file

@ -127,7 +127,7 @@ fn resolve_design_thinking(model: Option<&str>, chat_default: ThinkingMode) -> T
/// falls through to the orchestrator path).
///
/// Mirrors `launch_if_pending`'s builtin chat branch but uses the
/// design toolset and a 8192-token budget.
/// design toolset and a 16384-token per-turn budget.
pub(super) fn launch_design_loop_turn(
host: &mut WidgetHostNative,
user_text: String,
@ -161,7 +161,14 @@ pub(super) fn launch_design_loop_turn(
system_prompt: op_ai_skills::design_agent_system_prompt().to_string(),
user_message: user_text,
history,
max_output_tokens: 8192,
// Per-TURN output cap. The real budget-burner was hidden reasoning, not
// the DSL: a glm-5.2 loop run streamed ~94k thinking chars vs ~4k of
// actual batch_design, blowing the cap so a turn truncated mid-JSON, its
// tool call failed to parse, and the loop stopped early with an
// unfinished design. The root fix is at the wire (the loop now sends
// thinking:{type:disabled} for glm/minimax); 16384 (up from 8192) is
// headroom so a genuinely large section-batch still completes per turn.
max_output_tokens: 16384,
thinking,
effort,
attachments,

View file

@ -53,6 +53,14 @@ pub struct AgentLoopConfig {
/// shared execution path for BOTH regular builtin chat and the design
/// loop, so the finalize side-effect MUST be opt-in per provider.
pub finalize_on_exit: bool,
/// When true AND the model is a MiniMax / GLM reasoning model, send
/// `thinking:{type:"disabled"}` in the OpenAI-compat body. Without it a
/// reasoning model (glm-5.2) streams tens of thousands of hidden reasoning
/// tokens per turn — measured 94k thinking chars vs 4k of actual
/// `batch_design` DSL — which burns the per-turn `max_output_tokens` and
/// truncates the design mid-build. The single-shot builtin chat path
/// already gates this on the same flag; the loop body was missing it.
pub disable_thinking: bool,
}
impl AgentLoopConfig {
@ -658,13 +666,26 @@ pub async fn run_openai_agent_loop(
messages.push(json!({ "role": "user", "content": cfg.user_prompt }));
for _turn in 0..cfg.max_turns.max(1) {
let body = json!({
let mut body = json!({
"model": cfg.model,
"stream": true,
"max_tokens": cfg.max_output_tokens,
"messages": messages,
"tools": tools_json,
});
// Turn OFF hidden reasoning for MiniMax / GLM. Without this a glm-5.2
// design turn spends its whole `max_tokens` on `reasoning_content` and
// truncates the `batch_design` mid-JSON — the single-shot builtin body
// gates the same field on the same flag (`chat_builtin_http`), but the
// loop body was missing it, so every loop turn leaked thinking.
if cfg.disable_thinking
&& (crate::chat_builtin_http::is_minimax_model(&cfg.model)
|| crate::chat_builtin_http::is_glm_model(&cfg.model))
{
if let Some(obj) = body.as_object_mut() {
obj.insert("thinking".into(), json!({ "type": "disabled" }));
}
}
let resp = reqwest::Client::new()
.post(&cfg.url)
.bearer_auth(&cfg.api_key)

View file

@ -242,6 +242,7 @@ fn anthropic_loop_executes_tool_and_continues_with_tool_result() {
executor: executor.clone(),
max_turns: 5,
finalize_on_exit: true,
disable_thinking: false,
};
let (outcome, deltas) = run_loop_collect(cfg, true);
assert_eq!(outcome, Ok(true));
@ -318,6 +319,7 @@ fn loop_skips_finalize_when_disabled_for_plain_chat() {
executor: executor.clone(),
max_turns: 5,
finalize_on_exit: false,
disable_thinking: false,
};
let (outcome, _deltas) = run_loop_collect(cfg, true);
assert_eq!(outcome, Ok(true));
@ -347,6 +349,7 @@ fn anthropic_loop_stops_at_turn_cap_with_max_tokens_reason() {
executor: executor.clone(),
max_turns: 2,
finalize_on_exit: true,
disable_thinking: false,
};
let (outcome, deltas) = run_loop_collect(cfg, true);
assert_eq!(outcome, Ok(true));
@ -406,6 +409,7 @@ fn openai_loop_executes_tool_and_continues_with_role_tool_message() {
executor: executor.clone(),
max_turns: 5,
finalize_on_exit: true,
disable_thinking: false,
};
let (outcome, deltas) = run_loop_collect(cfg, false);
assert_eq!(outcome, Ok(true));
@ -518,6 +522,7 @@ fn anthropic_loop_replays_screenshot_result_as_image_content_block() {
executor: executor.clone(),
max_turns: 5,
finalize_on_exit: true,
disable_thinking: false,
};
let (outcome, _deltas) = run_loop_collect(cfg, true);
assert_eq!(outcome, Ok(true));
@ -615,6 +620,7 @@ fn openai_loop_replays_screenshot_result_as_image_url_part() {
executor: executor.clone(),
max_turns: 5,
finalize_on_exit: true,
disable_thinking: false,
};
let (outcome, _deltas) = run_loop_collect(cfg, false);
assert_eq!(outcome, Ok(true));

View file

@ -180,6 +180,7 @@ impl ChatProvider for ConfiguredBuiltinProvider {
executor,
max_turns: MAX_TOOL_TURNS,
finalize_on_exit: provider.finalize_on_exit,
disable_thinking,
};
match provider.kind {
BuiltinAgentKind::Anthropic => run_anthropic_agent_loop(cfg, &tx).await,
@ -237,7 +238,7 @@ impl ChatProvider for ConfiguredBuiltinProvider {
/// MiniMax M 系("MiniMax-M*"、旧 "abab*")是推理模型,其思考由 MiniMax 专属的
/// `thinking` body 字段控制。据模型名判定,以便只对它发关思考字段。
fn is_minimax_model(model: &str) -> bool {
pub(crate) fn is_minimax_model(model: &str) -> bool {
let m = model.to_ascii_lowercase();
m.starts_with("minimax") || m.starts_with("abab")
}
@ -247,7 +248,7 @@ fn is_minimax_model(model: &str) -> bool {
/// 一个设计子任务 thinking_len≈3 万、text_len=0,整段 parse 失败)。它接受和
/// MiniMax 同样的 `thinking:{type:"disabled"}`(curl 对 ark glm-5.2 验证:关思考后
/// reasoning_tokens=0、content 为干净 JSON)。按名判定,只对 GLM 下发。
fn is_glm_model(model: &str) -> bool {
pub(crate) fn is_glm_model(model: &str) -> bool {
model.to_ascii_lowercase().contains("glm")
}