feat(ai): route element manifest by model family (M4)

Replace the global OPENPENCIL_MANIFEST env gate with per-model routing:
manifest defaults ON for the families that cleared the ab-v9.2 KPI gate
(minimax 92% / ark-code 92% / glm 98% / deepseek 83%, all >=70%) and
stays OFF for models without benchmark data — the catalog is a floor
for weak models, not a ceiling for strong ones. The env var remains a
both-ways override (1/true/on forces on, 0/false/off forces off) so
op-smoke benchmarks and rollback keep their one-knob workflow.

Also plumb the selected built-in agent's model id into DesignRequest:
the production chat path always sent model:None, which routed every
model to Full-tier prompts and dead-armed both this gate and the M3
thinking policy outside op-smoke. Live-verified: glm-5.1 with no env
set engages the manifest protocol purely via routing.
This commit is contained in:
Fini 2026-06-13 00:52:26 +08:00
parent a9b0e729fd
commit 05e1e4e2d8
4 changed files with 167 additions and 36 deletions

View file

@ -1,6 +1,27 @@
use op_editor_core::EditorState;
use op_orchestrator::{AppendContext, DesignRequest};
/// Resolve the selected chat model's id for the orchestrator. Only
/// built-in (API-key) agents expose a concrete model id; CLI/ACP agents
/// pick their own model internally and yield `None` (the CLI-side
/// selection rides `ChatProviderLlmClient::with_model` instead). The id
/// feeds model-aware orchestrator policy — tier-gated skill filtering,
/// the element-manifest routing gate, and the M3 thinking policy — and
/// matches the configuration the ab-v9 benchmarks ran with (op-smoke
/// has always passed `OPENPENCIL_ORCHESTRATOR_MODEL` through).
fn selected_builtin_model(state: &EditorState) -> Option<String> {
let entry = state.chat.selected_model_entry()?;
let id = entry.builtin_provider_id.as_deref()?;
state
.editor_ui
.agent_settings
.builtin_agents
.iter()
.find(|agent| agent.id == id)
.map(|agent| agent.model.trim().to_string())
.filter(|model| !model.is_empty())
}
pub(crate) fn build_design_request(
prompt: String,
state: &EditorState,
@ -8,10 +29,7 @@ pub(crate) fn build_design_request(
) -> DesignRequest {
DesignRequest {
prompt,
// The chosen chat agent decides its own model; the orchestrator
// only passes through `req.model` when it explicitly overrides per
// sub-call, which it does not today.
model: None,
model: selected_builtin_model(state),
provider: None,
design_md: state.doc.design_md.clone(),
// Detected by `chat_intent::detect_append_intent` when the
@ -28,7 +46,10 @@ pub(crate) fn build_design_request(
#[cfg(test)]
mod tests {
use super::*;
use op_editor_core::EditorState;
use op_editor_core::{
AgentProvider, BuiltinAgentConfig, BuiltinAgentKind, BuiltinAgentPresetKey, EditorState,
ModelEntry,
};
#[test]
fn built_in_design_requests_enable_validation() {
@ -40,6 +61,7 @@ mod tests {
assert!(req.validation_enabled);
assert!(!req.visual_ref_enabled);
assert_eq!(req.concurrency, 4);
// No selected model entry → no model id (CLI agents pick their own).
assert_eq!(req.model, None);
assert!(req.append_context.is_none());
}
@ -60,4 +82,33 @@ mod tests {
assert_eq!(ctx.target_parent_id, "content-root");
assert_eq!(ctx.existing_section_labels, vec!["Hero".to_string()]);
}
#[test]
fn selected_builtin_agent_model_reaches_the_orchestrator() {
let mut state = EditorState::new();
state
.editor_ui
.agent_settings
.builtin_agents
.push(BuiltinAgentConfig {
id: "builtin-1".into(),
preset: BuiltinAgentPresetKey::Custom,
display_name: "MiniMax".into(),
kind: BuiltinAgentKind::OpenAiCompat,
api_key: "sk-test".into(),
model: "MiniMax-M3".into(),
base_url: "http://localhost:9".into(),
enabled: true,
});
let mut entry = ModelEntry::new(AgentProvider::ClaudeCode, "MiniMax-M3", "MiniMax M3");
entry.builtin_provider_id = Some("builtin-1".into());
state.chat.available_models = vec![entry];
state.chat.selected_model = 0;
let req = build_design_request("draw a dashboard".into(), &state, None);
// Drives tier-gated prompts, the manifest routing gate, and the
// M3 thinking policy — must match the agent the session will call.
assert_eq!(req.model.as_deref(), Some("MiniMax-M3"));
}
}

View file

@ -40,15 +40,32 @@ const MAX_SECTION_DEPTH: usize = 2;
static NEXT_SECTION_ID: AtomicU64 = AtomicU64::new(1);
/// M2 开关:`OPENPENCIL_MANIFEST=1`(或 `true`/`on`)把内置 agent 的
/// 子代理输出协议切到元素清单。走环境变量是为了贴合 op-smoke 基准的
/// per-process 切换习惯(参考 provider 切换),M4 过 ab-v9 门后翻默认值,
/// 届时本变量保留为回滚开关。
pub fn manifest_enabled() -> bool {
matches!(
std::env::var("OPENPENCIL_MANIFEST").as_deref(),
Ok("1") | Ok("true") | Ok("on")
)
/// M4 路由表:ab-v9.2 全矩阵(2026-06-12)过 70% KPI 门的弱模型家族
/// —— minimax-m3 92% / ark-code 92% / glm-5.1 98% / deepseek 83%。
/// 强模型(Claude / GPT / Gemini 等)无基准数据不翻:目录是弱模型的
/// 地板,不做强模型的天花板。kimi 同样无数据,待跑分后再进表。
const MANIFEST_DEFAULT_FAMILIES: &[&str] = &["minimax", "glm", "deepseek", "ark-code"];
/// M2 起的清单协议开关,M4 升级为按模型路由:`OPENPENCIL_MANIFEST`
/// 保留为双向 override —— `1/true/on` 强制开(op-smoke 基准的
/// per-process 切换习惯)、`0/false/off` 强制关(回滚开关);未设置时
/// 按 model id 家族路由。id 归一化与 `resolve_model_profile` 一致:
/// strip `provider/` 前缀 → 小写 → 子串命中。空 id(chat 路径没有
/// 模型信息时)不命中任何家族,走裸 JSONL。
pub fn manifest_enabled_for_model(model_id: &str) -> bool {
match std::env::var("OPENPENCIL_MANIFEST").as_deref() {
Ok("1") | Ok("true") | Ok("on") => return true,
Ok("0") | Ok("false") | Ok("off") => return false,
_ => {}
}
let normalized = match model_id.find('/') {
Some(i) => &model_id[i + 1..],
None => model_id,
};
let lower = normalized.to_lowercase();
MANIFEST_DEFAULT_FAMILIES
.iter()
.any(|family| lower.contains(family))
}
/// 从 LLM 文本解析元素清单。文本里没有任何 `"el"` 行时返回 `None`,
@ -725,4 +742,62 @@ Here is the design:
assert_eq!(outcome.nodes.len(), 1);
assert_eq!(frame_children(&outcome.nodes[0]).len(), 1);
}
// ── M4 按模型路由 ──────────────────────────────────────────────────
// 所有用例都持 MANIFEST_ENV_LOCK:函数读进程全局 env,与 subagent
// 的 set/remove 测试并行会互相拆台。
#[test]
fn manifest_routes_on_for_benchmarked_families_without_env() {
let _guard = crate::test_support::MANIFEST_ENV_LOCK
.lock()
.unwrap_or_else(|e| e.into_inner());
std::env::remove_var("OPENPENCIL_MANIFEST");
for id in [
"MiniMax-M3",
"MiniMax-M2.7",
"glm-5.1",
"deepseek-v4-pro",
"ark-code-latest",
"opencode/glm-5.1", // provider 前缀被剥掉
] {
assert!(manifest_enabled_for_model(id), "{id} should route ON");
}
}
#[test]
fn manifest_routes_off_for_unbenchmarked_models_without_env() {
let _guard = crate::test_support::MANIFEST_ENV_LOCK
.lock()
.unwrap_or_else(|e| e.into_inner());
std::env::remove_var("OPENPENCIL_MANIFEST");
for id in [
"claude-sonnet-4-6",
"claude-haiku",
"gpt-4o",
"gemini-2.5-pro",
"kimi-k2.6", // 无基准数据,刻意不进表
"", // chat 路径没有模型信息 → 裸 JSONL
] {
assert!(!manifest_enabled_for_model(id), "{id:?} should route OFF");
}
}
#[test]
fn manifest_env_overrides_routing_both_ways() {
let _guard = crate::test_support::MANIFEST_ENV_LOCK
.lock()
.unwrap_or_else(|e| e.into_inner());
std::env::set_var("OPENPENCIL_MANIFEST", "1");
assert!(
manifest_enabled_for_model("claude-sonnet-4-6"),
"force-on must beat the family table"
);
std::env::set_var("OPENPENCIL_MANIFEST", "0");
assert!(
!manifest_enabled_for_model("MiniMax-M3"),
"force-off is the rollback switch"
);
std::env::remove_var("OPENPENCIL_MANIFEST");
}
}

View file

@ -297,7 +297,10 @@ pub fn build_subagent_prompt(
// the full first attempt — the retry ladder (reduced/minimal) falls back
// to the smaller raw-JSONL prompt, and `parse_manifest` returning `None`
// on such output routes parsing back through `parse_nodes`.
let manifest_on = crate::manifest::manifest_enabled() && !reduced_complexity && !minimal_skills;
let manifest_on =
crate::manifest::manifest_enabled_for_model(req.model.as_deref().unwrap_or(""))
&& !reduced_complexity
&& !minimal_skills;
build_subagent_prompt_with_manifest(
subtask,
plan,

View file

@ -68,31 +68,33 @@ pub async fn run_subtask(
}
}
// 解析成 PenNode 树。manifest 模式(`OPENPENCIL_MANIFEST=1`)先按元素
// 清单解析;文本里没有清单行(如重试梯度回落到裸 JSONL prompt 后的
// 输出)时回落到既有裸 PenNode 路径,两条路汇入同一套后处理。
let mut nodes = if crate::manifest::manifest_enabled() {
match crate::manifest::parse_manifest(&text) {
Some(outcome) => {
for warning in &outcome.warnings {
eprintln!("[manifest] {warning}");
// 解析成 PenNode 树。manifest 模式(按模型路由 + `OPENPENCIL_MANIFEST`
// override)先按元素清单解析;文本里没有清单行(如重试梯度回落到裸
// JSONL prompt 后的输出)时回落到既有裸 PenNode 路径,两条路汇入同
// 一套后处理。
let mut nodes =
if crate::manifest::manifest_enabled_for_model(req.model.as_deref().unwrap_or("")) {
match crate::manifest::parse_manifest(&text) {
Some(outcome) => {
for warning in &outcome.warnings {
eprintln!("[manifest] {warning}");
}
if outcome.nodes.is_empty() {
return fail("manifest parsed but produced no nodes".into());
}
outcome.nodes
}
if outcome.nodes.is_empty() {
return fail("manifest parsed but produced no nodes".into());
}
outcome.nodes
None => match parse_nodes(&text) {
Ok(n) => n,
Err(e) => return fail(e.to_string()),
},
}
None => match parse_nodes(&text) {
} else {
match parse_nodes(&text) {
Ok(n) => n,
Err(e) => return fail(e.to_string()),
},
}
} else {
match parse_nodes(&text) {
Ok(n) => n,
Err(e) => return fail(e.to_string()),
}
};
}
};
if is_blank_container_forest(&nodes) {
return fail("blank container root produced no content nodes".into());
}