fix(orchestrator): stop planning truncation from shipping skeleton designs
The desktop LLM adapter capped design turns at 8192 output tokens; a rich plan truncated mid-JSON, the parse failed, and the heuristic fallback shipped a skeleton (hero + three cards) with no visible error. Raise the budget to the headless-harness value that ran a 52-prompt corpus with zero truncations (16384; 24576 for M3 which spends budget on reasoning), and give planning a second attempt before the fallback — a truncated stream or transient blip usually parses fine on retry.
This commit is contained in:
parent
0d326ec9a5
commit
3bce0ea5d8
|
|
@ -111,7 +111,12 @@ impl LlmClient for ChatProviderLlmClient {
|
|||
// The orchestrator's prompts can run long (planner system
|
||||
// is ~12 KB, sub-agents emit dense JSON). Give them room —
|
||||
// and M3's think stream burns budget before the answer.
|
||||
max_output_tokens: if m3_keeps_thinking { 16384 } else { 8192 },
|
||||
// 16384 matches the headless harness value that ran a
|
||||
// 52-prompt corpus with ZERO truncated responses; the old
|
||||
// 8192 truncated a rich plan mid-JSON on the desktop
|
||||
// ("planning parse failure; using fallback plan" → a
|
||||
// skeleton design with a hero and three cards).
|
||||
max_output_tokens: if m3_keeps_thinking { 24576 } else { 16384 },
|
||||
thinking: if m3_keeps_thinking {
|
||||
ThinkingMode::Adaptive
|
||||
} else {
|
||||
|
|
@ -213,11 +218,11 @@ mod tests {
|
|||
fn minimax_m3_keeps_thinking_others_disable_it() {
|
||||
let m3 = call_with_model(Some("MiniMax-M3"));
|
||||
assert_eq!(m3.thinking, ThinkingMode::Adaptive);
|
||||
assert_eq!(m3.max_output_tokens, 16384);
|
||||
assert_eq!(m3.max_output_tokens, 24576);
|
||||
|
||||
let m27 = call_with_model(Some("MiniMax-M2.7"));
|
||||
assert_eq!(m27.thinking, ThinkingMode::Disabled);
|
||||
assert_eq!(m27.max_output_tokens, 8192);
|
||||
assert_eq!(m27.max_output_tokens, 16384);
|
||||
|
||||
let unknown = call_with_model(None);
|
||||
assert_eq!(unknown.thinking, ThinkingMode::Disabled);
|
||||
|
|
|
|||
|
|
@ -607,39 +607,52 @@ async fn planning_loop(
|
|||
llm: &dyn LlmClient,
|
||||
abort: &AbortFlag,
|
||||
) -> Result<(OrchestratorPlan, NormInfo), OrchestratorError> {
|
||||
let pp = build_orchestrator_prompt(request, PlanningMode::Rich, abort.clone());
|
||||
let forced_style_guide_name = pp.forced_style_guide_name.clone();
|
||||
// TWO attempts before the heuristic fallback: a truncated stream or a
|
||||
// transient provider blip fails the parse once and usually succeeds
|
||||
// immediately after (measured on the desktop: a rich plan cut mid-JSON →
|
||||
// "planning parse failure" → a skeleton fallback design, while the very
|
||||
// same prompt parsed fine on retry). The fallback plan stays as the
|
||||
// final safety net, not the first response to a hiccup.
|
||||
for attempt in 1..=2u8 {
|
||||
let pp = build_orchestrator_prompt(request, PlanningMode::Rich, abort.clone());
|
||||
let forced_style_guide_name = pp.forced_style_guide_name.clone();
|
||||
|
||||
match collect_text(llm.call(pp.call_request)).await {
|
||||
Ok(raw) => {
|
||||
// abort 在流结束后被置位(两次检查对齐 TS)
|
||||
if abort.is_set() {
|
||||
match collect_text(llm.call(pp.call_request)).await {
|
||||
Ok(raw) => {
|
||||
// abort 在流结束后被置位(两次检查对齐 TS)
|
||||
if abort.is_set() {
|
||||
return Err(OrchestratorError::Aborted);
|
||||
}
|
||||
if let Some((mut plan, _repaired)) = parse_orchestrator_response(&raw, request) {
|
||||
// 回填 forced_style_guide_name(若 plan 未携带)
|
||||
if plan.style_guide_name.is_none() {
|
||||
if let Some(forced) = forced_style_guide_name {
|
||||
plan.style_guide_name = Some(forced);
|
||||
}
|
||||
}
|
||||
let norm = normalize(&mut plan, request);
|
||||
return Ok((plan, norm));
|
||||
}
|
||||
let preview = raw.trim().chars().take(150).collect::<String>();
|
||||
tracing::warn!(
|
||||
attempt,
|
||||
preview = %preview,
|
||||
"planning parse failure"
|
||||
);
|
||||
}
|
||||
Err(true) => {
|
||||
// abort 在流中发生 → 立即返回
|
||||
return Err(OrchestratorError::Aborted);
|
||||
}
|
||||
if let Some((mut plan, _repaired)) = parse_orchestrator_response(&raw, request) {
|
||||
// 回填 forced_style_guide_name(若 plan 未携带)
|
||||
if plan.style_guide_name.is_none() {
|
||||
if let Some(forced) = forced_style_guide_name {
|
||||
plan.style_guide_name = Some(forced);
|
||||
}
|
||||
}
|
||||
let norm = normalize(&mut plan, request);
|
||||
return Ok((plan, norm));
|
||||
Err(false) => {
|
||||
tracing::warn!(attempt, "planning stream error");
|
||||
}
|
||||
let preview = raw.trim().chars().take(150).collect::<String>();
|
||||
tracing::warn!(
|
||||
preview = %preview,
|
||||
"planning parse failure; using fallback plan"
|
||||
);
|
||||
}
|
||||
Err(true) => {
|
||||
// abort 在流中发生 → 立即返回
|
||||
if abort.is_set() {
|
||||
return Err(OrchestratorError::Aborted);
|
||||
}
|
||||
Err(false) => {
|
||||
tracing::warn!("planning stream error; using fallback plan");
|
||||
}
|
||||
}
|
||||
tracing::warn!("planning failed twice; using fallback plan");
|
||||
|
||||
// 规划失败 → fallback plan(规划不可出错)
|
||||
let mut fallback = build_fallback_plan(request);
|
||||
|
|
|
|||
|
|
@ -266,6 +266,7 @@ fn run_zero_node_subtask_preserves_failure_context() {
|
|||
fn run_planning_failure_uses_fallback_plan() {
|
||||
// 规划吐垃圾 → fallback plan;subtask 正常 → 成功。
|
||||
let llm = ScriptedLlm::new(vec![
|
||||
ScriptResponse::Text("no json here".into()),
|
||||
ScriptResponse::Text("no json here".into()),
|
||||
ScriptResponse::Text(node_json("section-1")),
|
||||
]);
|
||||
|
|
@ -291,7 +292,8 @@ fn run_planning_failure_uses_fallback_plan() {
|
|||
#[test]
|
||||
fn planning_parse_failure_uses_fallback_plan() {
|
||||
let llm = ScriptedLlm::new(vec![
|
||||
// planning call → bad JSON
|
||||
// planning attempts 1 + 2 → bad JSON (the retry consumes one more)
|
||||
ScriptResponse::Text("not valid json at all".into()),
|
||||
ScriptResponse::Text("not valid json at all".into()),
|
||||
// fallback plan's single subtask
|
||||
ScriptResponse::Text(node_json("section-1")),
|
||||
|
|
@ -315,7 +317,12 @@ fn planning_parse_failure_uses_fallback_plan() {
|
|||
fn planning_stream_error_uses_fallback_plan() {
|
||||
use crate::types::LlmError;
|
||||
let llm = ScriptedLlm::new(vec![
|
||||
// planning call → stream error (non-abort)
|
||||
// planning attempts 1 + 2 → stream error (non-abort); the retry
|
||||
// consumes the second before the heuristic fallback engages.
|
||||
ScriptResponse::Fail(LlmError {
|
||||
message: "HTTP 500 upstream".into(),
|
||||
aborted: false,
|
||||
}),
|
||||
ScriptResponse::Fail(LlmError {
|
||||
message: "HTTP 500 upstream".into(),
|
||||
aborted: false,
|
||||
|
|
@ -698,3 +705,31 @@ fn run_dashboard_shell_keeps_sidebar_fill_height_end_to_end() {
|
|||
.collect::<String>()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn planning_retries_once_before_the_fallback_plan() {
|
||||
// A truncated planning response fails the parse; the SECOND attempt
|
||||
// returns a valid plan and must be used (no skeleton fallback).
|
||||
let llm = ScriptedLlm::new(vec![
|
||||
ScriptResponse::Text(r##"{"palette":{"background":"#0B0C0E","surface":"#1A1B"##.into()),
|
||||
ScriptResponse::Text(PLAN_JSON.into()),
|
||||
ScriptResponse::Text(node_json("hero")),
|
||||
ScriptResponse::Text(node_json("feat")),
|
||||
]);
|
||||
let mut sink = VecDocSink::new();
|
||||
let mut events: Vec<Progress> = Vec::new();
|
||||
let mut on_progress = |p: Progress| events.push(p);
|
||||
|
||||
let summary = futures::executor::block_on(Orchestrator::new().run(
|
||||
req(),
|
||||
&mut sink,
|
||||
&llm,
|
||||
&mut on_progress,
|
||||
&AbortFlag::new(),
|
||||
&stub_providers(),
|
||||
))
|
||||
.expect("run ok after planning retry");
|
||||
|
||||
// The REAL plan (2 subtasks) landed — not the single-subtask fallback.
|
||||
assert_eq!(summary.subtasks.len(), 2, "retried plan used, not fallback");
|
||||
}
|
||||
|
|
|
|||
Loading…
Reference in a new issue