test(ai): add direct-client headless benchmark harness to op-smoke
op-smoke's QueryEngine path can't send MiniMax's `thinking:{type:
"disabled"}` field, so reasoning models (M3 etc.) thought themselves out
of the token budget and couldn't be benchmarked end-to-end. Add a
non-streaming DirectClient (OPENPENCIL_SMOKE_DIRECT=1) that posts
openai-compat directly and disables thinking at the wire for MiniMax —
or any endpoint via OPENPENCIL_SMOKE_DISABLE_THINKING=1 (Volcengine
honors the same param) — so the production thinking-disable path can be
exercised headless without a GUI.
This commit is contained in:
parent
e8e768fefe
commit
cd8837128b
1
Cargo.lock
generated
1
Cargo.lock
generated
|
|
@ -3084,6 +3084,7 @@ dependencies = [
|
|||
"futures",
|
||||
"op-editor-core",
|
||||
"op-orchestrator",
|
||||
"reqwest 0.12.28",
|
||||
"serde_json",
|
||||
"tokio",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -30,3 +30,11 @@ tokio = { version = "1", features = ["rt-multi-thread", "macros"] }
|
|||
# same shape as `persistence::save_to_path`) when `OPENPENCIL_SMOKE_OUT`
|
||||
# is set — lets the headless render / screenshot step pick up the result.
|
||||
serde_json = { workspace = true }
|
||||
# Direct openai-compat HTTP client for the smoke harness (OPENPENCIL_SMOKE_DIRECT=1)
|
||||
# — lets it send the MiniMax `thinking:{type:disabled}` field the vendored agent
|
||||
# QueryEngine can't, so M3 (reasoning) can be validated headless. rustls-tls +
|
||||
# json mirror op-host-desktop so the workspace shares one TLS backend.
|
||||
reqwest = { version = "0.12", default-features = false, features = [
|
||||
"rustls-tls",
|
||||
"json",
|
||||
] }
|
||||
|
|
|
|||
|
|
@ -94,7 +94,18 @@ impl LlmClient for SmokeLlmClient {
|
|||
);
|
||||
|
||||
tokio::spawn(async move {
|
||||
let engine = QueryEngine::new(provider, model).with_system(system);
|
||||
// QueryEngine 默认 4096 输出 token,对推理模型(MiniMax-M3 等)远不够
|
||||
// ——它先吐 <think>(常 ~3.5k token)再给 JSON,4096 会在答案前截断。
|
||||
// 生产路径(chat_provider_llm)用 8192 且关思考;benchmark 走 QueryEngine
|
||||
// 无法关思考,故给更宽预算让其 think 完还能产出 JSON。可用
|
||||
// OPENPENCIL_SMOKE_MAX_TOKENS 覆盖。
|
||||
let max_tokens = std::env::var("OPENPENCIL_SMOKE_MAX_TOKENS")
|
||||
.ok()
|
||||
.and_then(|s| s.parse().ok())
|
||||
.unwrap_or(16384);
|
||||
let engine = QueryEngine::new(provider, model)
|
||||
.with_system(system)
|
||||
.with_max_output_tokens(max_tokens);
|
||||
let abort = AbortController::new();
|
||||
let stream = match engine.run(user, abort).await {
|
||||
Ok(s) => s,
|
||||
|
|
@ -156,6 +167,133 @@ impl LlmClient for SmokeLlmClient {
|
|||
}
|
||||
}
|
||||
|
||||
/// MiniMax M-series ("MiniMax-M*", legacy "abab*") are reasoning models whose
|
||||
/// thinking is toggled by the MiniMax `thinking` body field. Mirrors the
|
||||
/// production gate in `chat_builtin_http::is_minimax_model`.
|
||||
fn is_minimax_model(model: &str) -> bool {
|
||||
let m = model.to_ascii_lowercase();
|
||||
m.starts_with("minimax") || m.starts_with("abab")
|
||||
}
|
||||
|
||||
/// Direct openai-compat `LlmClient` for the harness (OPENPENCIL_SMOKE_DIRECT=1).
|
||||
///
|
||||
/// The default [`SmokeLlmClient`] goes through the vendored `agent` QueryEngine,
|
||||
/// which can't send MiniMax's `thinking:{type:disabled}` field — so M3 (a
|
||||
/// reasoning model) thinks itself out of budget. This client does a plain
|
||||
/// non-streaming POST and adds that field for MiniMax models, mirroring the
|
||||
/// production fix in `chat_builtin_http::run_openai_chat`, so M3-with-thinking-
|
||||
/// disabled can be validated end-to-end headless (no GUI, no submodule edit).
|
||||
struct DirectOpenAiClient {
|
||||
base_url: String,
|
||||
api_key: String,
|
||||
default_model: String,
|
||||
}
|
||||
|
||||
impl LlmClient for DirectOpenAiClient {
|
||||
fn call(
|
||||
&self,
|
||||
req: CallRequest,
|
||||
) -> futures::stream::BoxStream<'static, Result<LlmChunk, LlmError>> {
|
||||
let (tx, rx) = mpsc::unbounded::<Result<LlmChunk, LlmError>>();
|
||||
if req.abort.is_set() {
|
||||
let _ = tx.unbounded_send(Err(LlmError {
|
||||
message: "aborted".into(),
|
||||
aborted: true,
|
||||
}));
|
||||
return Box::pin(rx);
|
||||
}
|
||||
let url = format!("{}/chat/completions", self.base_url.trim_end_matches('/'));
|
||||
let key = self.api_key.clone();
|
||||
let model = req
|
||||
.model
|
||||
.clone()
|
||||
.unwrap_or_else(|| self.default_model.clone());
|
||||
let system = req.system_prompt.clone();
|
||||
let user = req.user_prompt.clone();
|
||||
let max_tokens: u32 = std::env::var("OPENPENCIL_SMOKE_MAX_TOKENS")
|
||||
.ok()
|
||||
.and_then(|s| s.parse().ok())
|
||||
.unwrap_or(16384);
|
||||
let dump = std::env::var("OPENPENCIL_SMOKE_DUMP").is_ok();
|
||||
eprintln!(
|
||||
"[LLM] direct call: model={model} system_len={} user_len={}",
|
||||
system.len(),
|
||||
user.len()
|
||||
);
|
||||
tokio::spawn(async move {
|
||||
let mut body = serde_json::json!({
|
||||
"model": model,
|
||||
"stream": false,
|
||||
"max_tokens": max_tokens,
|
||||
"messages": [
|
||||
{ "role": "system", "content": system },
|
||||
{ "role": "user", "content": user },
|
||||
],
|
||||
});
|
||||
// MiniMax reasoning models inject `<think>` into content by default;
|
||||
// disable it at the wire level. `OPENPENCIL_SMOKE_DISABLE_THINKING=1`
|
||||
// forces it for any model whose endpoint speaks the same schema
|
||||
// (Volcengine 方舟 — glm/kimi/doubao — confirmed to honor it), so the
|
||||
// latest 方舟-hosted reasoning models can be benchmarked clean too.
|
||||
let force_disable = std::env::var("OPENPENCIL_SMOKE_DISABLE_THINKING").is_ok();
|
||||
if force_disable || is_minimax_model(&model) {
|
||||
if let Some(obj) = body.as_object_mut() {
|
||||
obj.insert("thinking".into(), serde_json::json!({ "type": "disabled" }));
|
||||
}
|
||||
}
|
||||
let resp = match reqwest::Client::new()
|
||||
.post(&url)
|
||||
.bearer_auth(&key)
|
||||
.json(&body)
|
||||
.send()
|
||||
.await
|
||||
{
|
||||
Ok(r) => r,
|
||||
Err(e) => {
|
||||
let _ = tx.unbounded_send(Err(LlmError {
|
||||
message: format!("POST {url}: {e}"),
|
||||
aborted: false,
|
||||
}));
|
||||
return;
|
||||
}
|
||||
};
|
||||
let status = resp.status();
|
||||
let text = resp.text().await.unwrap_or_default();
|
||||
if !status.is_success() {
|
||||
let head: String = text.chars().take(300).collect();
|
||||
let _ = tx.unbounded_send(Err(LlmError {
|
||||
message: format!("http {status}: {head}"),
|
||||
aborted: false,
|
||||
}));
|
||||
return;
|
||||
}
|
||||
let content = serde_json::from_str::<serde_json::Value>(&text)
|
||||
.ok()
|
||||
.and_then(|v| {
|
||||
v["choices"][0]["message"]["content"]
|
||||
.as_str()
|
||||
.map(str::to_string)
|
||||
})
|
||||
.unwrap_or_default();
|
||||
if dump {
|
||||
eprintln!(
|
||||
"[DUMP] ===== LLM response ({} chars) =====\n{content}\n[DUMP] ===== end =====",
|
||||
content.len()
|
||||
);
|
||||
}
|
||||
if content.trim().is_empty() {
|
||||
let _ = tx.unbounded_send(Err(LlmError {
|
||||
message: "empty content from provider".into(),
|
||||
aborted: false,
|
||||
}));
|
||||
} else {
|
||||
let _ = tx.unbounded_send(Ok(LlmChunk::Text(content)));
|
||||
}
|
||||
});
|
||||
Box::pin(rx)
|
||||
}
|
||||
}
|
||||
|
||||
/// Inline `DocSink` — owns the canonical state directly, no channel hop.
|
||||
/// Every `apply` echoes the command kind + result so the smoke trace
|
||||
/// shows the orchestrator's mutations linearly.
|
||||
|
|
@ -248,7 +386,11 @@ async fn main() -> std::process::ExitCode {
|
|||
eprintln!("[SMOKE] provider={provider_kind} model={model}");
|
||||
eprintln!("[SMOKE] prompt={prompt:?}");
|
||||
|
||||
let provider: Arc<dyn Provider> = match provider_kind.as_str() {
|
||||
// `OPENPENCIL_SMOKE_DIRECT=1` swaps the QueryEngine path for a direct
|
||||
// openai-compat client that can send MiniMax `thinking:{type:disabled}`
|
||||
// (the vendored agent QueryEngine can't) — needed to validate M3 headless.
|
||||
let direct = std::env::var("OPENPENCIL_SMOKE_DIRECT").is_ok();
|
||||
let llm: Box<dyn LlmClient> = match provider_kind.as_str() {
|
||||
"anthropic" => {
|
||||
let key = std::env::var("OPENPENCIL_ANTHROPIC_API_KEY")
|
||||
.ok()
|
||||
|
|
@ -260,7 +402,10 @@ async fn main() -> std::process::ExitCode {
|
|||
);
|
||||
return std::process::ExitCode::from(3);
|
||||
};
|
||||
Arc::new(AnthropicProvider::new(key))
|
||||
Box::new(SmokeLlmClient {
|
||||
provider: Arc::new(AnthropicProvider::new(key)),
|
||||
default_model: model.clone(),
|
||||
})
|
||||
}
|
||||
"openai" | "openai-compat" => {
|
||||
let key = std::env::var("OPENPENCIL_LLM_API_KEY")
|
||||
|
|
@ -279,10 +424,21 @@ async fn main() -> std::process::ExitCode {
|
|||
);
|
||||
return std::process::ExitCode::from(3);
|
||||
};
|
||||
eprintln!("[SMOKE] base_url={base_url}");
|
||||
Arc::new(OpenAiCompatProvider::new(OpenAiCompatConfig::new(
|
||||
key, base_url,
|
||||
)))
|
||||
eprintln!("[SMOKE] base_url={base_url} direct={direct}");
|
||||
if direct {
|
||||
Box::new(DirectOpenAiClient {
|
||||
base_url,
|
||||
api_key: key,
|
||||
default_model: model.clone(),
|
||||
})
|
||||
} else {
|
||||
Box::new(SmokeLlmClient {
|
||||
provider: Arc::new(OpenAiCompatProvider::new(OpenAiCompatConfig::new(
|
||||
key, base_url,
|
||||
))),
|
||||
default_model: model.clone(),
|
||||
})
|
||||
}
|
||||
}
|
||||
other => {
|
||||
eprintln!(
|
||||
|
|
@ -291,10 +447,6 @@ async fn main() -> std::process::ExitCode {
|
|||
return std::process::ExitCode::from(3);
|
||||
}
|
||||
};
|
||||
let llm = SmokeLlmClient {
|
||||
provider,
|
||||
default_model: model.clone(),
|
||||
};
|
||||
|
||||
let mut sink = InlineDocSink {
|
||||
state: EditorState::new(),
|
||||
|
|
@ -331,7 +483,7 @@ async fn main() -> std::process::ExitCode {
|
|||
.run(
|
||||
request,
|
||||
&mut sink,
|
||||
&llm,
|
||||
llm.as_ref(),
|
||||
&mut on_progress,
|
||||
&abort,
|
||||
&providers,
|
||||
|
|
|
|||
Loading…
Reference in a new issue