fix(agent): disable reasoning + raise budget in the design tool-loop
The OpenAI-compat tool-loop body never sent the reasoning-off field, so a GLM/
MiniMax reasoning model spent its whole per-turn budget on hidden reasoning and
truncated the design mid-op. Send thinking:{type:disabled} for those models in
the loop body (matching the single-shot path) and raise the per-turn output
budget for headroom.
This commit is contained in:
parent
32192b9348
commit
d03c2d7dd3
|
|
@ -127,7 +127,7 @@ fn resolve_design_thinking(model: Option<&str>, chat_default: ThinkingMode) -> T
|
|||
/// falls through to the orchestrator path).
|
||||
///
|
||||
/// Mirrors `launch_if_pending`'s builtin chat branch but uses the
|
||||
/// design toolset and a 8192-token budget.
|
||||
/// design toolset and a 16384-token per-turn budget.
|
||||
pub(super) fn launch_design_loop_turn(
|
||||
host: &mut WidgetHostNative,
|
||||
user_text: String,
|
||||
|
|
@ -161,7 +161,14 @@ pub(super) fn launch_design_loop_turn(
|
|||
system_prompt: op_ai_skills::design_agent_system_prompt().to_string(),
|
||||
user_message: user_text,
|
||||
history,
|
||||
max_output_tokens: 8192,
|
||||
// Per-TURN output cap. The real budget-burner was hidden reasoning, not
|
||||
// the DSL: a glm-5.2 loop run streamed ~94k thinking chars vs ~4k of
|
||||
// actual batch_design, blowing the cap so a turn truncated mid-JSON, its
|
||||
// tool call failed to parse, and the loop stopped early with an
|
||||
// unfinished design. The root fix is at the wire (the loop now sends
|
||||
// thinking:{type:disabled} for glm/minimax); 16384 (up from 8192) is
|
||||
// headroom so a genuinely large section-batch still completes per turn.
|
||||
max_output_tokens: 16384,
|
||||
thinking,
|
||||
effort,
|
||||
attachments,
|
||||
|
|
|
|||
|
|
@ -53,6 +53,14 @@ pub struct AgentLoopConfig {
|
|||
/// shared execution path for BOTH regular builtin chat and the design
|
||||
/// loop, so the finalize side-effect MUST be opt-in per provider.
|
||||
pub finalize_on_exit: bool,
|
||||
/// When true AND the model is a MiniMax / GLM reasoning model, send
|
||||
/// `thinking:{type:"disabled"}` in the OpenAI-compat body. Without it a
|
||||
/// reasoning model (glm-5.2) streams tens of thousands of hidden reasoning
|
||||
/// tokens per turn — measured 94k thinking chars vs 4k of actual
|
||||
/// `batch_design` DSL — which burns the per-turn `max_output_tokens` and
|
||||
/// truncates the design mid-build. The single-shot builtin chat path
|
||||
/// already gates this on the same flag; the loop body was missing it.
|
||||
pub disable_thinking: bool,
|
||||
}
|
||||
|
||||
impl AgentLoopConfig {
|
||||
|
|
@ -658,13 +666,26 @@ pub async fn run_openai_agent_loop(
|
|||
messages.push(json!({ "role": "user", "content": cfg.user_prompt }));
|
||||
|
||||
for _turn in 0..cfg.max_turns.max(1) {
|
||||
let body = json!({
|
||||
let mut body = json!({
|
||||
"model": cfg.model,
|
||||
"stream": true,
|
||||
"max_tokens": cfg.max_output_tokens,
|
||||
"messages": messages,
|
||||
"tools": tools_json,
|
||||
});
|
||||
// Turn OFF hidden reasoning for MiniMax / GLM. Without this a glm-5.2
|
||||
// design turn spends its whole `max_tokens` on `reasoning_content` and
|
||||
// truncates the `batch_design` mid-JSON — the single-shot builtin body
|
||||
// gates the same field on the same flag (`chat_builtin_http`), but the
|
||||
// loop body was missing it, so every loop turn leaked thinking.
|
||||
if cfg.disable_thinking
|
||||
&& (crate::chat_builtin_http::is_minimax_model(&cfg.model)
|
||||
|| crate::chat_builtin_http::is_glm_model(&cfg.model))
|
||||
{
|
||||
if let Some(obj) = body.as_object_mut() {
|
||||
obj.insert("thinking".into(), json!({ "type": "disabled" }));
|
||||
}
|
||||
}
|
||||
let resp = reqwest::Client::new()
|
||||
.post(&cfg.url)
|
||||
.bearer_auth(&cfg.api_key)
|
||||
|
|
|
|||
|
|
@ -242,6 +242,7 @@ fn anthropic_loop_executes_tool_and_continues_with_tool_result() {
|
|||
executor: executor.clone(),
|
||||
max_turns: 5,
|
||||
finalize_on_exit: true,
|
||||
disable_thinking: false,
|
||||
};
|
||||
let (outcome, deltas) = run_loop_collect(cfg, true);
|
||||
assert_eq!(outcome, Ok(true));
|
||||
|
|
@ -318,6 +319,7 @@ fn loop_skips_finalize_when_disabled_for_plain_chat() {
|
|||
executor: executor.clone(),
|
||||
max_turns: 5,
|
||||
finalize_on_exit: false,
|
||||
disable_thinking: false,
|
||||
};
|
||||
let (outcome, _deltas) = run_loop_collect(cfg, true);
|
||||
assert_eq!(outcome, Ok(true));
|
||||
|
|
@ -347,6 +349,7 @@ fn anthropic_loop_stops_at_turn_cap_with_max_tokens_reason() {
|
|||
executor: executor.clone(),
|
||||
max_turns: 2,
|
||||
finalize_on_exit: true,
|
||||
disable_thinking: false,
|
||||
};
|
||||
let (outcome, deltas) = run_loop_collect(cfg, true);
|
||||
assert_eq!(outcome, Ok(true));
|
||||
|
|
@ -406,6 +409,7 @@ fn openai_loop_executes_tool_and_continues_with_role_tool_message() {
|
|||
executor: executor.clone(),
|
||||
max_turns: 5,
|
||||
finalize_on_exit: true,
|
||||
disable_thinking: false,
|
||||
};
|
||||
let (outcome, deltas) = run_loop_collect(cfg, false);
|
||||
assert_eq!(outcome, Ok(true));
|
||||
|
|
@ -518,6 +522,7 @@ fn anthropic_loop_replays_screenshot_result_as_image_content_block() {
|
|||
executor: executor.clone(),
|
||||
max_turns: 5,
|
||||
finalize_on_exit: true,
|
||||
disable_thinking: false,
|
||||
};
|
||||
let (outcome, _deltas) = run_loop_collect(cfg, true);
|
||||
assert_eq!(outcome, Ok(true));
|
||||
|
|
@ -615,6 +620,7 @@ fn openai_loop_replays_screenshot_result_as_image_url_part() {
|
|||
executor: executor.clone(),
|
||||
max_turns: 5,
|
||||
finalize_on_exit: true,
|
||||
disable_thinking: false,
|
||||
};
|
||||
let (outcome, _deltas) = run_loop_collect(cfg, false);
|
||||
assert_eq!(outcome, Ok(true));
|
||||
|
|
|
|||
|
|
@ -180,6 +180,7 @@ impl ChatProvider for ConfiguredBuiltinProvider {
|
|||
executor,
|
||||
max_turns: MAX_TOOL_TURNS,
|
||||
finalize_on_exit: provider.finalize_on_exit,
|
||||
disable_thinking,
|
||||
};
|
||||
match provider.kind {
|
||||
BuiltinAgentKind::Anthropic => run_anthropic_agent_loop(cfg, &tx).await,
|
||||
|
|
@ -237,7 +238,7 @@ impl ChatProvider for ConfiguredBuiltinProvider {
|
|||
|
||||
/// MiniMax M 系("MiniMax-M*"、旧 "abab*")是推理模型,其思考由 MiniMax 专属的
|
||||
/// `thinking` body 字段控制。据模型名判定,以便只对它发关思考字段。
|
||||
fn is_minimax_model(model: &str) -> bool {
|
||||
pub(crate) fn is_minimax_model(model: &str) -> bool {
|
||||
let m = model.to_ascii_lowercase();
|
||||
m.starts_with("minimax") || m.starts_with("abab")
|
||||
}
|
||||
|
|
@ -247,7 +248,7 @@ fn is_minimax_model(model: &str) -> bool {
|
|||
/// 一个设计子任务 thinking_len≈3 万、text_len=0,整段 parse 失败)。它接受和
|
||||
/// MiniMax 同样的 `thinking:{type:"disabled"}`(curl 对 ark glm-5.2 验证:关思考后
|
||||
/// reasoning_tokens=0、content 为干净 JSON)。按名判定,只对 GLM 下发。
|
||||
fn is_glm_model(model: &str) -> bool {
|
||||
pub(crate) fn is_glm_model(model: &str) -> bool {
|
||||
model.to_ascii_lowercase().contains("glm")
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Reference in a new issue