diff --git a/Cargo.lock b/Cargo.lock index 5ae36806b..c7a8df3d1 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -3084,6 +3084,7 @@ dependencies = [ "futures", "op-editor-core", "op-orchestrator", + "reqwest 0.12.28", "serde_json", "tokio", ] diff --git a/crates/op-smoke/Cargo.toml b/crates/op-smoke/Cargo.toml index a70160800..3b7eab705 100644 --- a/crates/op-smoke/Cargo.toml +++ b/crates/op-smoke/Cargo.toml @@ -30,3 +30,11 @@ tokio = { version = "1", features = ["rt-multi-thread", "macros"] } # same shape as `persistence::save_to_path`) when `OPENPENCIL_SMOKE_OUT` # is set — lets the headless render / screenshot step pick up the result. serde_json = { workspace = true } +# Direct openai-compat HTTP client for the smoke harness (OPENPENCIL_SMOKE_DIRECT=1) +# — lets it send the MiniMax `thinking:{type:disabled}` field the vendored agent +# QueryEngine can't, so M3 (reasoning) can be validated headless. rustls-tls + +# json mirror op-host-desktop so the workspace shares one TLS backend. +reqwest = { version = "0.12", default-features = false, features = [ + "rustls-tls", + "json", +] } diff --git a/crates/op-smoke/src/main.rs b/crates/op-smoke/src/main.rs index a2cf6a852..d35c53bde 100644 --- a/crates/op-smoke/src/main.rs +++ b/crates/op-smoke/src/main.rs @@ -94,7 +94,18 @@ impl LlmClient for SmokeLlmClient { ); tokio::spawn(async move { - let engine = QueryEngine::new(provider, model).with_system(system); + // QueryEngine 默认 4096 输出 token,对推理模型(MiniMax-M3 等)远不够 + // ——它先吐 (常 ~3.5k token)再给 JSON,4096 会在答案前截断。 + // 生产路径(chat_provider_llm)用 8192 且关思考;benchmark 走 QueryEngine + // 无法关思考,故给更宽预算让其 think 完还能产出 JSON。可用 + // OPENPENCIL_SMOKE_MAX_TOKENS 覆盖。 + let max_tokens = std::env::var("OPENPENCIL_SMOKE_MAX_TOKENS") + .ok() + .and_then(|s| s.parse().ok()) + .unwrap_or(16384); + let engine = QueryEngine::new(provider, model) + .with_system(system) + .with_max_output_tokens(max_tokens); let abort = AbortController::new(); let stream = match engine.run(user, abort).await { Ok(s) => s, @@ -156,6 +167,133 @@ impl LlmClient for SmokeLlmClient { } } +/// MiniMax M-series ("MiniMax-M*", legacy "abab*") are reasoning models whose +/// thinking is toggled by the MiniMax `thinking` body field. Mirrors the +/// production gate in `chat_builtin_http::is_minimax_model`. +fn is_minimax_model(model: &str) -> bool { + let m = model.to_ascii_lowercase(); + m.starts_with("minimax") || m.starts_with("abab") +} + +/// Direct openai-compat `LlmClient` for the harness (OPENPENCIL_SMOKE_DIRECT=1). +/// +/// The default [`SmokeLlmClient`] goes through the vendored `agent` QueryEngine, +/// which can't send MiniMax's `thinking:{type:disabled}` field — so M3 (a +/// reasoning model) thinks itself out of budget. This client does a plain +/// non-streaming POST and adds that field for MiniMax models, mirroring the +/// production fix in `chat_builtin_http::run_openai_chat`, so M3-with-thinking- +/// disabled can be validated end-to-end headless (no GUI, no submodule edit). +struct DirectOpenAiClient { + base_url: String, + api_key: String, + default_model: String, +} + +impl LlmClient for DirectOpenAiClient { + fn call( + &self, + req: CallRequest, + ) -> futures::stream::BoxStream<'static, Result> { + let (tx, rx) = mpsc::unbounded::>(); + if req.abort.is_set() { + let _ = tx.unbounded_send(Err(LlmError { + message: "aborted".into(), + aborted: true, + })); + return Box::pin(rx); + } + let url = format!("{}/chat/completions", self.base_url.trim_end_matches('/')); + let key = self.api_key.clone(); + let model = req + .model + .clone() + .unwrap_or_else(|| self.default_model.clone()); + let system = req.system_prompt.clone(); + let user = req.user_prompt.clone(); + let max_tokens: u32 = std::env::var("OPENPENCIL_SMOKE_MAX_TOKENS") + .ok() + .and_then(|s| s.parse().ok()) + .unwrap_or(16384); + let dump = std::env::var("OPENPENCIL_SMOKE_DUMP").is_ok(); + eprintln!( + "[LLM] direct call: model={model} system_len={} user_len={}", + system.len(), + user.len() + ); + tokio::spawn(async move { + let mut body = serde_json::json!({ + "model": model, + "stream": false, + "max_tokens": max_tokens, + "messages": [ + { "role": "system", "content": system }, + { "role": "user", "content": user }, + ], + }); + // MiniMax reasoning models inject `` into content by default; + // disable it at the wire level. `OPENPENCIL_SMOKE_DISABLE_THINKING=1` + // forces it for any model whose endpoint speaks the same schema + // (Volcengine 方舟 — glm/kimi/doubao — confirmed to honor it), so the + // latest 方舟-hosted reasoning models can be benchmarked clean too. + let force_disable = std::env::var("OPENPENCIL_SMOKE_DISABLE_THINKING").is_ok(); + if force_disable || is_minimax_model(&model) { + if let Some(obj) = body.as_object_mut() { + obj.insert("thinking".into(), serde_json::json!({ "type": "disabled" })); + } + } + let resp = match reqwest::Client::new() + .post(&url) + .bearer_auth(&key) + .json(&body) + .send() + .await + { + Ok(r) => r, + Err(e) => { + let _ = tx.unbounded_send(Err(LlmError { + message: format!("POST {url}: {e}"), + aborted: false, + })); + return; + } + }; + let status = resp.status(); + let text = resp.text().await.unwrap_or_default(); + if !status.is_success() { + let head: String = text.chars().take(300).collect(); + let _ = tx.unbounded_send(Err(LlmError { + message: format!("http {status}: {head}"), + aborted: false, + })); + return; + } + let content = serde_json::from_str::(&text) + .ok() + .and_then(|v| { + v["choices"][0]["message"]["content"] + .as_str() + .map(str::to_string) + }) + .unwrap_or_default(); + if dump { + eprintln!( + "[DUMP] ===== LLM response ({} chars) =====\n{content}\n[DUMP] ===== end =====", + content.len() + ); + } + if content.trim().is_empty() { + let _ = tx.unbounded_send(Err(LlmError { + message: "empty content from provider".into(), + aborted: false, + })); + } else { + let _ = tx.unbounded_send(Ok(LlmChunk::Text(content))); + } + }); + Box::pin(rx) + } +} + /// Inline `DocSink` — owns the canonical state directly, no channel hop. /// Every `apply` echoes the command kind + result so the smoke trace /// shows the orchestrator's mutations linearly. @@ -248,7 +386,11 @@ async fn main() -> std::process::ExitCode { eprintln!("[SMOKE] provider={provider_kind} model={model}"); eprintln!("[SMOKE] prompt={prompt:?}"); - let provider: Arc = match provider_kind.as_str() { + // `OPENPENCIL_SMOKE_DIRECT=1` swaps the QueryEngine path for a direct + // openai-compat client that can send MiniMax `thinking:{type:disabled}` + // (the vendored agent QueryEngine can't) — needed to validate M3 headless. + let direct = std::env::var("OPENPENCIL_SMOKE_DIRECT").is_ok(); + let llm: Box = match provider_kind.as_str() { "anthropic" => { let key = std::env::var("OPENPENCIL_ANTHROPIC_API_KEY") .ok() @@ -260,7 +402,10 @@ async fn main() -> std::process::ExitCode { ); return std::process::ExitCode::from(3); }; - Arc::new(AnthropicProvider::new(key)) + Box::new(SmokeLlmClient { + provider: Arc::new(AnthropicProvider::new(key)), + default_model: model.clone(), + }) } "openai" | "openai-compat" => { let key = std::env::var("OPENPENCIL_LLM_API_KEY") @@ -279,10 +424,21 @@ async fn main() -> std::process::ExitCode { ); return std::process::ExitCode::from(3); }; - eprintln!("[SMOKE] base_url={base_url}"); - Arc::new(OpenAiCompatProvider::new(OpenAiCompatConfig::new( - key, base_url, - ))) + eprintln!("[SMOKE] base_url={base_url} direct={direct}"); + if direct { + Box::new(DirectOpenAiClient { + base_url, + api_key: key, + default_model: model.clone(), + }) + } else { + Box::new(SmokeLlmClient { + provider: Arc::new(OpenAiCompatProvider::new(OpenAiCompatConfig::new( + key, base_url, + ))), + default_model: model.clone(), + }) + } } other => { eprintln!( @@ -291,10 +447,6 @@ async fn main() -> std::process::ExitCode { return std::process::ExitCode::from(3); } }; - let llm = SmokeLlmClient { - provider, - default_model: model.clone(), - }; let mut sink = InlineDocSink { state: EditorState::new(), @@ -331,7 +483,7 @@ async fn main() -> std::process::ExitCode { .run( request, &mut sink, - &llm, + llm.as_ref(), &mut on_progress, &abort, &providers,