test(ai): add direct-client headless benchmark harness to op-smoke

op-smoke's QueryEngine path can't send MiniMax's `thinking:{type:
"disabled"}` field, so reasoning models (M3 etc.) thought themselves out
of the token budget and couldn't be benchmarked end-to-end. Add a
non-streaming DirectClient (OPENPENCIL_SMOKE_DIRECT=1) that posts
openai-compat directly and disables thinking at the wire for MiniMax —
or any endpoint via OPENPENCIL_SMOKE_DISABLE_THINKING=1 (Volcengine
honors the same param) — so the production thinking-disable path can be
exercised headless without a GUI.
This commit is contained in:
Fini 2026-06-06 15:45:37 +08:00
parent e8e768fefe
commit cd8837128b
3 changed files with 173 additions and 12 deletions

1
Cargo.lock generated
View file

@ -3084,6 +3084,7 @@ dependencies = [
"futures",
"op-editor-core",
"op-orchestrator",
"reqwest 0.12.28",
"serde_json",
"tokio",
]

View file

@ -30,3 +30,11 @@ tokio = { version = "1", features = ["rt-multi-thread", "macros"] }
# same shape as `persistence::save_to_path`) when `OPENPENCIL_SMOKE_OUT`
# is set — lets the headless render / screenshot step pick up the result.
serde_json = { workspace = true }
# Direct openai-compat HTTP client for the smoke harness (OPENPENCIL_SMOKE_DIRECT=1)
# — lets it send the MiniMax `thinking:{type:disabled}` field the vendored agent
# QueryEngine can't, so M3 (reasoning) can be validated headless. rustls-tls +
# json mirror op-host-desktop so the workspace shares one TLS backend.
reqwest = { version = "0.12", default-features = false, features = [
"rustls-tls",
"json",
] }

View file

@ -94,7 +94,18 @@ impl LlmClient for SmokeLlmClient {
);
tokio::spawn(async move {
let engine = QueryEngine::new(provider, model).with_system(system);
// QueryEngine 默认 4096 输出 token,对推理模型(MiniMax-M3 等)远不够
// ——它先吐 <think>(常 ~3.5k token)再给 JSON,4096 会在答案前截断。
// 生产路径(chat_provider_llm)用 8192 且关思考;benchmark 走 QueryEngine
// 无法关思考,故给更宽预算让其 think 完还能产出 JSON。可用
// OPENPENCIL_SMOKE_MAX_TOKENS 覆盖。
let max_tokens = std::env::var("OPENPENCIL_SMOKE_MAX_TOKENS")
.ok()
.and_then(|s| s.parse().ok())
.unwrap_or(16384);
let engine = QueryEngine::new(provider, model)
.with_system(system)
.with_max_output_tokens(max_tokens);
let abort = AbortController::new();
let stream = match engine.run(user, abort).await {
Ok(s) => s,
@ -156,6 +167,133 @@ impl LlmClient for SmokeLlmClient {
}
}
/// MiniMax M-series ("MiniMax-M*", legacy "abab*") are reasoning models whose
/// thinking is toggled by the MiniMax `thinking` body field. Mirrors the
/// production gate in `chat_builtin_http::is_minimax_model`.
fn is_minimax_model(model: &str) -> bool {
let m = model.to_ascii_lowercase();
m.starts_with("minimax") || m.starts_with("abab")
}
/// Direct openai-compat `LlmClient` for the harness (OPENPENCIL_SMOKE_DIRECT=1).
///
/// The default [`SmokeLlmClient`] goes through the vendored `agent` QueryEngine,
/// which can't send MiniMax's `thinking:{type:disabled}` field — so M3 (a
/// reasoning model) thinks itself out of budget. This client does a plain
/// non-streaming POST and adds that field for MiniMax models, mirroring the
/// production fix in `chat_builtin_http::run_openai_chat`, so M3-with-thinking-
/// disabled can be validated end-to-end headless (no GUI, no submodule edit).
struct DirectOpenAiClient {
base_url: String,
api_key: String,
default_model: String,
}
impl LlmClient for DirectOpenAiClient {
fn call(
&self,
req: CallRequest,
) -> futures::stream::BoxStream<'static, Result<LlmChunk, LlmError>> {
let (tx, rx) = mpsc::unbounded::<Result<LlmChunk, LlmError>>();
if req.abort.is_set() {
let _ = tx.unbounded_send(Err(LlmError {
message: "aborted".into(),
aborted: true,
}));
return Box::pin(rx);
}
let url = format!("{}/chat/completions", self.base_url.trim_end_matches('/'));
let key = self.api_key.clone();
let model = req
.model
.clone()
.unwrap_or_else(|| self.default_model.clone());
let system = req.system_prompt.clone();
let user = req.user_prompt.clone();
let max_tokens: u32 = std::env::var("OPENPENCIL_SMOKE_MAX_TOKENS")
.ok()
.and_then(|s| s.parse().ok())
.unwrap_or(16384);
let dump = std::env::var("OPENPENCIL_SMOKE_DUMP").is_ok();
eprintln!(
"[LLM] direct call: model={model} system_len={} user_len={}",
system.len(),
user.len()
);
tokio::spawn(async move {
let mut body = serde_json::json!({
"model": model,
"stream": false,
"max_tokens": max_tokens,
"messages": [
{ "role": "system", "content": system },
{ "role": "user", "content": user },
],
});
// MiniMax reasoning models inject `<think>` into content by default;
// disable it at the wire level. `OPENPENCIL_SMOKE_DISABLE_THINKING=1`
// forces it for any model whose endpoint speaks the same schema
// (Volcengine 方舟 — glm/kimi/doubao — confirmed to honor it), so the
// latest 方舟-hosted reasoning models can be benchmarked clean too.
let force_disable = std::env::var("OPENPENCIL_SMOKE_DISABLE_THINKING").is_ok();
if force_disable || is_minimax_model(&model) {
if let Some(obj) = body.as_object_mut() {
obj.insert("thinking".into(), serde_json::json!({ "type": "disabled" }));
}
}
let resp = match reqwest::Client::new()
.post(&url)
.bearer_auth(&key)
.json(&body)
.send()
.await
{
Ok(r) => r,
Err(e) => {
let _ = tx.unbounded_send(Err(LlmError {
message: format!("POST {url}: {e}"),
aborted: false,
}));
return;
}
};
let status = resp.status();
let text = resp.text().await.unwrap_or_default();
if !status.is_success() {
let head: String = text.chars().take(300).collect();
let _ = tx.unbounded_send(Err(LlmError {
message: format!("http {status}: {head}"),
aborted: false,
}));
return;
}
let content = serde_json::from_str::<serde_json::Value>(&text)
.ok()
.and_then(|v| {
v["choices"][0]["message"]["content"]
.as_str()
.map(str::to_string)
})
.unwrap_or_default();
if dump {
eprintln!(
"[DUMP] ===== LLM response ({} chars) =====\n{content}\n[DUMP] ===== end =====",
content.len()
);
}
if content.trim().is_empty() {
let _ = tx.unbounded_send(Err(LlmError {
message: "empty content from provider".into(),
aborted: false,
}));
} else {
let _ = tx.unbounded_send(Ok(LlmChunk::Text(content)));
}
});
Box::pin(rx)
}
}
/// Inline `DocSink` — owns the canonical state directly, no channel hop.
/// Every `apply` echoes the command kind + result so the smoke trace
/// shows the orchestrator's mutations linearly.
@ -248,7 +386,11 @@ async fn main() -> std::process::ExitCode {
eprintln!("[SMOKE] provider={provider_kind} model={model}");
eprintln!("[SMOKE] prompt={prompt:?}");
let provider: Arc<dyn Provider> = match provider_kind.as_str() {
// `OPENPENCIL_SMOKE_DIRECT=1` swaps the QueryEngine path for a direct
// openai-compat client that can send MiniMax `thinking:{type:disabled}`
// (the vendored agent QueryEngine can't) — needed to validate M3 headless.
let direct = std::env::var("OPENPENCIL_SMOKE_DIRECT").is_ok();
let llm: Box<dyn LlmClient> = match provider_kind.as_str() {
"anthropic" => {
let key = std::env::var("OPENPENCIL_ANTHROPIC_API_KEY")
.ok()
@ -260,7 +402,10 @@ async fn main() -> std::process::ExitCode {
);
return std::process::ExitCode::from(3);
};
Arc::new(AnthropicProvider::new(key))
Box::new(SmokeLlmClient {
provider: Arc::new(AnthropicProvider::new(key)),
default_model: model.clone(),
})
}
"openai" | "openai-compat" => {
let key = std::env::var("OPENPENCIL_LLM_API_KEY")
@ -279,10 +424,21 @@ async fn main() -> std::process::ExitCode {
);
return std::process::ExitCode::from(3);
};
eprintln!("[SMOKE] base_url={base_url}");
Arc::new(OpenAiCompatProvider::new(OpenAiCompatConfig::new(
key, base_url,
)))
eprintln!("[SMOKE] base_url={base_url} direct={direct}");
if direct {
Box::new(DirectOpenAiClient {
base_url,
api_key: key,
default_model: model.clone(),
})
} else {
Box::new(SmokeLlmClient {
provider: Arc::new(OpenAiCompatProvider::new(OpenAiCompatConfig::new(
key, base_url,
))),
default_model: model.clone(),
})
}
}
other => {
eprintln!(
@ -291,10 +447,6 @@ async fn main() -> std::process::ExitCode {
return std::process::ExitCode::from(3);
}
};
let llm = SmokeLlmClient {
provider,
default_model: model.clone(),
};
let mut sink = InlineDocSink {
state: EditorState::new(),
@ -331,7 +483,7 @@ async fn main() -> std::process::ExitCode {
.run(
request,
&mut sink,
&llm,
llm.as_ref(),
&mut on_progress,
&abort,
&providers,