fix(orchestrator): stop planning truncation from shipping skeleton designs

The desktop LLM adapter capped design turns at 8192 output tokens; a
rich plan truncated mid-JSON, the parse failed, and the heuristic
fallback shipped a skeleton (hero + three cards) with no visible error.
Raise the budget to the headless-harness value that ran a 52-prompt
corpus with zero truncations (16384; 24576 for M3 which spends budget
on reasoning), and give planning a second attempt before the fallback —
a truncated stream or transient blip usually parses fine on retry.
This commit is contained in:
Fini 2026-07-03 03:02:42 +08:00
parent 0d326ec9a5
commit 3bce0ea5d8
3 changed files with 83 additions and 30 deletions

View file

@ -111,7 +111,12 @@ impl LlmClient for ChatProviderLlmClient {
// The orchestrator's prompts can run long (planner system
// is ~12 KB, sub-agents emit dense JSON). Give them room —
// and M3's think stream burns budget before the answer.
max_output_tokens: if m3_keeps_thinking { 16384 } else { 8192 },
// 16384 matches the headless harness value that ran a
// 52-prompt corpus with ZERO truncated responses; the old
// 8192 truncated a rich plan mid-JSON on the desktop
// ("planning parse failure; using fallback plan" → a
// skeleton design with a hero and three cards).
max_output_tokens: if m3_keeps_thinking { 24576 } else { 16384 },
thinking: if m3_keeps_thinking {
ThinkingMode::Adaptive
} else {
@ -213,11 +218,11 @@ mod tests {
fn minimax_m3_keeps_thinking_others_disable_it() {
let m3 = call_with_model(Some("MiniMax-M3"));
assert_eq!(m3.thinking, ThinkingMode::Adaptive);
assert_eq!(m3.max_output_tokens, 16384);
assert_eq!(m3.max_output_tokens, 24576);
let m27 = call_with_model(Some("MiniMax-M2.7"));
assert_eq!(m27.thinking, ThinkingMode::Disabled);
assert_eq!(m27.max_output_tokens, 8192);
assert_eq!(m27.max_output_tokens, 16384);
let unknown = call_with_model(None);
assert_eq!(unknown.thinking, ThinkingMode::Disabled);

View file

@ -607,39 +607,52 @@ async fn planning_loop(
llm: &dyn LlmClient,
abort: &AbortFlag,
) -> Result<(OrchestratorPlan, NormInfo), OrchestratorError> {
let pp = build_orchestrator_prompt(request, PlanningMode::Rich, abort.clone());
let forced_style_guide_name = pp.forced_style_guide_name.clone();
// TWO attempts before the heuristic fallback: a truncated stream or a
// transient provider blip fails the parse once and usually succeeds
// immediately after (measured on the desktop: a rich plan cut mid-JSON →
// "planning parse failure" → a skeleton fallback design, while the very
// same prompt parsed fine on retry). The fallback plan stays as the
// final safety net, not the first response to a hiccup.
for attempt in 1..=2u8 {
let pp = build_orchestrator_prompt(request, PlanningMode::Rich, abort.clone());
let forced_style_guide_name = pp.forced_style_guide_name.clone();
match collect_text(llm.call(pp.call_request)).await {
Ok(raw) => {
// abort 在流结束后被置位(两次检查对齐 TS)
if abort.is_set() {
match collect_text(llm.call(pp.call_request)).await {
Ok(raw) => {
// abort 在流结束后被置位(两次检查对齐 TS)
if abort.is_set() {
return Err(OrchestratorError::Aborted);
}
if let Some((mut plan, _repaired)) = parse_orchestrator_response(&raw, request) {
// 回填 forced_style_guide_name(若 plan 未携带)
if plan.style_guide_name.is_none() {
if let Some(forced) = forced_style_guide_name {
plan.style_guide_name = Some(forced);
}
}
let norm = normalize(&mut plan, request);
return Ok((plan, norm));
}
let preview = raw.trim().chars().take(150).collect::<String>();
tracing::warn!(
attempt,
preview = %preview,
"planning parse failure"
);
}
Err(true) => {
// abort 在流中发生 → 立即返回
return Err(OrchestratorError::Aborted);
}
if let Some((mut plan, _repaired)) = parse_orchestrator_response(&raw, request) {
// 回填 forced_style_guide_name(若 plan 未携带)
if plan.style_guide_name.is_none() {
if let Some(forced) = forced_style_guide_name {
plan.style_guide_name = Some(forced);
}
}
let norm = normalize(&mut plan, request);
return Ok((plan, norm));
Err(false) => {
tracing::warn!(attempt, "planning stream error");
}
let preview = raw.trim().chars().take(150).collect::<String>();
tracing::warn!(
preview = %preview,
"planning parse failure; using fallback plan"
);
}
Err(true) => {
// abort 在流中发生 → 立即返回
if abort.is_set() {
return Err(OrchestratorError::Aborted);
}
Err(false) => {
tracing::warn!("planning stream error; using fallback plan");
}
}
tracing::warn!("planning failed twice; using fallback plan");
// 规划失败 → fallback plan(规划不可出错)
let mut fallback = build_fallback_plan(request);

View file

@ -266,6 +266,7 @@ fn run_zero_node_subtask_preserves_failure_context() {
fn run_planning_failure_uses_fallback_plan() {
// 规划吐垃圾 → fallback plan;subtask 正常 → 成功。
let llm = ScriptedLlm::new(vec![
ScriptResponse::Text("no json here".into()),
ScriptResponse::Text("no json here".into()),
ScriptResponse::Text(node_json("section-1")),
]);
@ -291,7 +292,8 @@ fn run_planning_failure_uses_fallback_plan() {
#[test]
fn planning_parse_failure_uses_fallback_plan() {
let llm = ScriptedLlm::new(vec![
// planning call → bad JSON
// planning attempts 1 + 2 → bad JSON (the retry consumes one more)
ScriptResponse::Text("not valid json at all".into()),
ScriptResponse::Text("not valid json at all".into()),
// fallback plan's single subtask
ScriptResponse::Text(node_json("section-1")),
@ -315,7 +317,12 @@ fn planning_parse_failure_uses_fallback_plan() {
fn planning_stream_error_uses_fallback_plan() {
use crate::types::LlmError;
let llm = ScriptedLlm::new(vec![
// planning call → stream error (non-abort)
// planning attempts 1 + 2 → stream error (non-abort); the retry
// consumes the second before the heuristic fallback engages.
ScriptResponse::Fail(LlmError {
message: "HTTP 500 upstream".into(),
aborted: false,
}),
ScriptResponse::Fail(LlmError {
message: "HTTP 500 upstream".into(),
aborted: false,
@ -698,3 +705,31 @@ fn run_dashboard_shell_keeps_sidebar_fill_height_end_to_end() {
.collect::<String>()
);
}
#[test]
fn planning_retries_once_before_the_fallback_plan() {
// A truncated planning response fails the parse; the SECOND attempt
// returns a valid plan and must be used (no skeleton fallback).
let llm = ScriptedLlm::new(vec![
ScriptResponse::Text(r##"{"palette":{"background":"#0B0C0E","surface":"#1A1B"##.into()),
ScriptResponse::Text(PLAN_JSON.into()),
ScriptResponse::Text(node_json("hero")),
ScriptResponse::Text(node_json("feat")),
]);
let mut sink = VecDocSink::new();
let mut events: Vec<Progress> = Vec::new();
let mut on_progress = |p: Progress| events.push(p);
let summary = futures::executor::block_on(Orchestrator::new().run(
req(),
&mut sink,
&llm,
&mut on_progress,
&AbortFlag::new(),
&stub_providers(),
))
.expect("run ok after planning retry");
// The REAL plan (2 subtasks) landed — not the single-subtask fallback.
assert_eq!(summary.subtasks.len(), 2, "retried plan used, not fallback");
}