test(ai): ab-v9 manifest matrix harness

scripts/ab-v9/run_matrix.py runs the ab-v3 corpus (52 prompts) through
the full Rust orchestrator per provider (op-smoke headless,
OPENPENCIL_MANIFEST=1) and scores M3 expected-shape (required roles in
the saved .op tree) + M5 element selection, appending rows per cell for
crash-safe resume. Keys come from env only (MM_KEY/ARK_KEY/DS_KEY).

op-smoke grows OPENPENCIL_SMOKE_KEEP_THINKING=1 to keep MiniMax
reasoning ON: ab-v9 showed M3-nothink emits lazy minimal manifests
(17%, ~10s answers) while M3-with-thinking lands 60% with composite
tied-best at ~110s — the MiniMax production routing target.
This commit is contained in:
Fini 2026-06-10 21:04:07 +08:00
parent 4bbf6b0883
commit 2cd197a7ec
2 changed files with 379 additions and 1 deletions

View file

@ -236,7 +236,12 @@ impl LlmClient for DirectOpenAiClient {
// (Volcengine 方舟 — glm/kimi/doubao — confirmed to honor it), so the
// latest 方舟-hosted reasoning models can be benchmarked clean too.
let force_disable = std::env::var("OPENPENCIL_SMOKE_DISABLE_THINKING").is_ok();
if force_disable || is_minimax_model(&model) {
// `OPENPENCIL_SMOKE_KEEP_THINKING=1` keeps reasoning ON even for
// MiniMax — ab-v9 showed M3-nothink emits lazy minimal manifests
// (17% M3, ~10s answers); this lets the M3-with-think arm be
// benchmarked (strip_reasoning handles the <think> blocks).
let keep_thinking = std::env::var("OPENPENCIL_SMOKE_KEEP_THINKING").is_ok();
if !keep_thinking && (force_disable || is_minimax_model(&model)) {
if let Some(obj) = body.as_object_mut() {
obj.insert("thinking".into(), serde_json::json!({ "type": "disabled" }));
}

373
scripts/ab-v9/run_matrix.py Normal file
View file

@ -0,0 +1,373 @@
#!/usr/bin/env python3
"""ab-v9 — element-manifest arm matrix over the ab-v3 corpus (Rust pipeline).
Spec: openpencil-docs/superpowers/specs/2026-06-10-element-manifest-v2.md §2.
Runs the FULL Rust orchestrator (op-smoke headless, `OPENPENCIL_MANIFEST=1`)
for every corpus prompt × provider, then scores M3 (expected-shape: required
roles present in the saved .op tree) + M5 (semantic element selection,
version-suffix-stripped) like the TS harness's score-run.ts.
Keys come from env ONLY (never hardcoded — Codex review of the 06-04
benchmark scripts): MM_KEY / ARK_KEY / DS_KEY. Rows append to scores.jsonl
as each cell finishes (crash recovery — lesson from ab-corpus 2cd7a6b2).
Usage:
MM_KEY=... ARK_KEY=... DS_KEY=... python3 scripts/ab-v9/run_matrix.py \
[--out /tmp/ab-v9] [--workers 6] [--models minimax,ark,deepseek] \
[--only prompt-id,...]
python3 scripts/ab-v9/run_matrix.py --score-only --out /tmp/ab-v9
"""
import argparse
import json
import os
import re
import subprocess
import sys
import threading
import time
from collections import Counter
from concurrent.futures import ThreadPoolExecutor
REPO = os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
CORPUS = os.path.join(REPO, "packages/pen-ai-skills/corpus/ab-v3")
SMOKE = os.path.join(REPO, "target/debug/op-smoke")
ARK_BASE = "https://ark.cn-beijing.volces.com/api/coding/v3"
PROVIDERS = {
# 用户指定:MiniMax 用最新 M3(关思考由 DirectClient 的 is_minimax_model
# 自动下发);DS 官方端点没有 v4.1,v4-pro 即最新 pro;方舟挑主流最新
# (该账号无 GLM5.1):GLM-4.7 + Kimi-K2-thinking,另保留 ark-code-latest
# 作为方舟基线。
"minimax-m3": {
"base": "https://api.minimaxi.com/v1",
"model": "MiniMax-M3",
"key_env": "MM_KEY",
},
"minimax-m2.7": {
"base": "https://api.minimaxi.com/v1",
"model": "MiniMax-M2.7",
"key_env": "MM_KEY",
},
# M3 开思考臂:nothink 摆烂(ab-v9 17%),开思考是 M3 线唯一候选。
"minimax-m3-think": {
"base": "https://api.minimaxi.com/v1",
"model": "MiniMax-M3",
"key_env": "MM_KEY",
"extra_env": {"OPENPENCIL_SMOKE_KEEP_THINKING": "1"},
},
"deepseek-v4-pro": {
"base": "https://api.deepseek.com/v1",
"model": "deepseek-v4-pro",
"key_env": "DS_KEY",
},
"ark-code": {
"base": ARK_BASE,
"model": "ark-code-latest",
"key_env": "ARK_KEY",
},
# glm-5.1 是 coding-plan 隐藏别名(不在 /models 列表,实测可调,与 TS
# ab-corpus 2026-04-22 起的路由一致)。kimi-k2.6 同为别名但 2026-06-10
# 起按用户要求移除——方舟主流模型一次只允许启用一个,账号当前启用的
# 是 GLM-5.1。
"ark-glm-5.1": {
"base": ARK_BASE,
"model": "glm-5.1",
"key_env": "ARK_KEY",
},
}
CELL_TIMEOUT_S = 1200
def parse_corpus_yaml(path):
"""Constrained parser for the corpus schema (no pyyaml on this host).
Handles: top-level scalars, `prompt: |` block, `expected:` with
`must_contain_roles` list + `min_roles` map.
"""
entry = {"must_contain_roles": [], "min_roles": {}}
lines = open(path, encoding="utf-8").read().splitlines()
i, n = 0, len(lines)
section = None
while i < n:
line = lines[i]
if line.startswith("prompt: |"):
block = []
i += 1
while i < n and (lines[i].startswith(" ") or lines[i] == ""):
block.append(lines[i][2:])
i += 1
entry["prompt"] = "\n".join(block).strip()
continue
if line.startswith("expected:"):
section = "expected"
i += 1
continue
if section == "expected":
stripped = line.strip()
if line.startswith(" must_contain_roles:"):
sub = "roles"
elif line.startswith(" min_roles:"):
sub = "min"
elif line.startswith(" - "):
entry["must_contain_roles"].append(stripped[2:].strip())
elif line.startswith(" ") and ":" in stripped:
k, v = stripped.split(":", 1)
try:
entry["min_roles"][k.strip()] = int(v.strip())
except ValueError:
pass
elif line and not line.startswith(" "):
section = None
continue
i += 1
continue
m = re.match(r"^([a-z_]+):\s*(.*)$", line)
if m:
entry[m.group(1)] = m.group(2).strip()
i += 1
return entry if entry.get("id") and entry.get("prompt") else None
def load_corpus(only=None):
prompts = []
for name in sorted(os.listdir(CORPUS)):
if not name.endswith(".yaml"):
continue
entry = parse_corpus_yaml(os.path.join(CORPUS, name))
if entry and (not only or entry["id"] in only):
prompts.append(entry)
return prompts
def walk_roles(node, counter):
if isinstance(node, dict):
role = node.get("role")
if isinstance(role, str) and role:
counter[role] += 1
for key in ("children", "pages"):
for child in node.get(key) or []:
walk_roles(child, counter)
elif isinstance(node, list):
for item in node:
walk_roles(item, counter)
def count_nodes(node):
if isinstance(node, dict):
n = 1 if node.get("type") else 0
for key in ("children", "pages"):
for child in node.get(key) or []:
n += count_nodes(child)
return n
if isinstance(node, list):
return sum(count_nodes(item) for item in node)
return 0
def base_kind(tool):
base = tool.strip().lower().replace("-", "_")
base = re.sub(r"^add_", "", base)
return re.sub(r"_v\d+$", "", base)
def score_cell(entry, op_path, log_path, rc, duration):
row = {
"prompt_id": entry["id"],
"category": entry.get("category", ""),
"difficulty": entry.get("difficulty", "obvious"),
"duration_s": round(duration, 1),
"rc": rc,
}
log_text = ""
if os.path.exists(log_path):
log_text = open(log_path, encoding="utf-8", errors="replace").read()
# Reasoning drafts mention element kinds the final answer may not use —
# strip <think> blocks before counting (mirrors parse.rs strip_reasoning).
log_text = re.sub(r"<think(?:ing)?>.*?</think(?:ing)?>", "", log_text, flags=re.S)
row["el_lines"] = len(re.findall(r'\{"el"\s*:', log_text))
row["warnings"] = dict(Counter(re.findall(r"\[manifest\] (W-[A-Z-]+)", log_text)))
sys_lens = [int(x) for x in re.findall(r"system_len=(\d+)", log_text)]
row["max_system_chars"] = max(sys_lens) if sys_lens else 0
# M5 — semantic element selection (suffix-stripped), only for prompts
# that name an expected tool.
expected_tool = entry.get("expected_tool_if_any", "")
if expected_tool:
used = {base_kind(k) for k in re.findall(r'\{"el"\s*:\s*"([a-z_0-9-]+)"', log_text)}
row["m5_expected"] = base_kind(expected_tool)
row["m5_success"] = row["m5_expected"] in used
if rc != 0 or not os.path.exists(op_path):
row["m3_success"] = False
row["m3_failure_reason"] = f"generation failed (rc={rc})"
row["garbage"] = True
return row
try:
doc = json.load(open(op_path, encoding="utf-8"))
except (json.JSONDecodeError, OSError) as err:
row["m3_success"] = False
row["m3_failure_reason"] = f"unreadable .op: {err}"
row["garbage"] = True
return row
row["node_count"] = count_nodes(doc)
row["garbage"] = row["node_count"] <= 1
roles = Counter()
walk_roles(doc, roles)
missing = [r for r in entry["must_contain_roles"] if roles.get(r, 0) == 0]
below = [
f"{r} ({roles.get(r, 0)}/{want})"
for r, want in entry["min_roles"].items()
if roles.get(r, 0) < want
]
if missing:
row["m3_success"] = False
row["m3_failure_reason"] = "missing required role(s): " + ", ".join(missing)
elif below:
row["m3_success"] = False
row["m3_failure_reason"] = "role counts below minimum: " + ", ".join(below)
else:
row["m3_success"] = True
row["m3_failure_reason"] = ""
return row
def run_cell(entry, provider_id, cfg, out_dir, write_row):
cell = f"{provider_id}/{entry['id']}"
op_path = os.path.join(out_dir, "op", provider_id, f"{entry['id']}.op")
log_path = os.path.join(out_dir, "logs", provider_id, f"{entry['id']}.log")
os.makedirs(os.path.dirname(op_path), exist_ok=True)
os.makedirs(os.path.dirname(log_path), exist_ok=True)
env = dict(
os.environ,
OPENPENCIL_MANIFEST="1",
OPENPENCIL_SMOKE_DIRECT="1",
OPENPENCIL_SMOKE_DUMP="1",
OPENPENCIL_LLM_PROVIDER="openai-compat",
OPENPENCIL_LLM_API_KEY=os.environ[cfg["key_env"]],
OPENPENCIL_LLM_BASE_URL=cfg["base"],
OPENPENCIL_ORCHESTRATOR_MODEL=cfg["model"],
OPENPENCIL_SMOKE_OUT=op_path,
**cfg.get("extra_env", {}),
)
start = time.time()
try:
with open(log_path, "w", encoding="utf-8") as log:
rc = subprocess.run(
[SMOKE, entry["prompt"]],
env=env,
stdout=log,
stderr=subprocess.STDOUT,
timeout=CELL_TIMEOUT_S,
).returncode
except subprocess.TimeoutExpired:
rc = -9
duration = time.time() - start
row = score_cell(entry, op_path, log_path, rc, duration)
row["model"] = provider_id
write_row(row)
status = "ok" if row["m3_success"] else f"M3-FAIL ({row['m3_failure_reason'][:60]})"
print(f"[{cell}] {duration:.0f}s {status}", flush=True)
def aggregate(rows):
out = ["# ab-v9 — manifest arm × ab-v3 corpus\n"]
out.append(
"| model | n | M3 | M3 rate | composite M3 | garbage | M5 | avg s |"
)
out.append("|---|---|---|---|---|---|---|---|")
for model in sorted({r["model"] for r in rows}):
sub = [r for r in rows if r["model"] == model]
m3 = sum(1 for r in sub if r["m3_success"])
comp = [r for r in sub if r["difficulty"] == "composite"]
comp_m3 = sum(1 for r in comp if r["m3_success"])
garbage = sum(1 for r in sub if r.get("garbage"))
m5_rows = [r for r in sub if "m5_success" in r]
m5 = sum(1 for r in m5_rows if r["m5_success"])
avg = sum(r["duration_s"] for r in sub) / max(len(sub), 1)
out.append(
f"| {model} | {len(sub)} | {m3}/{len(sub)} | {m3 / max(len(sub), 1):.0%} "
f"| {comp_m3}/{len(comp)} | {garbage} | {m5}/{len(m5_rows)} | {avg:.0f} |"
)
out.append("\n## 失败归因 top\n")
reasons = Counter(
r["m3_failure_reason"].split(":")[0] for r in rows if not r["m3_success"]
)
for reason, count in reasons.most_common(8):
out.append(f"- {count} × {reason}")
out.append("\n## warning 直方图\n")
warnings = Counter()
for r in rows:
warnings.update(r.get("warnings", {}))
for w, count in warnings.most_common():
out.append(f"- {count} × {w}")
return "\n".join(out) + "\n"
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--out", default="/tmp/ab-v9")
ap.add_argument("--workers", type=int, default=6)
ap.add_argument("--models", default=",".join(PROVIDERS))
ap.add_argument("--only", default="")
ap.add_argument("--score-only", action="store_true")
args = ap.parse_args()
os.makedirs(args.out, exist_ok=True)
scores_path = os.path.join(args.out, "scores.jsonl")
only = set(filter(None, args.only.split(",")))
prompts = load_corpus(only or None)
models = [m for m in args.models.split(",") if m in PROVIDERS]
if args.score_only:
rows = []
for entry in prompts:
for model in models:
op_path = os.path.join(args.out, "op", model, f"{entry['id']}.op")
log_path = os.path.join(args.out, "logs", model, f"{entry['id']}.log")
row = score_cell(entry, op_path, log_path, 0, 0.0)
row["model"] = model
rows.append(row)
open(scores_path, "w", encoding="utf-8").write(
"\n".join(json.dumps(r, ensure_ascii=False) for r in rows) + "\n"
)
else:
for model in models:
if not os.environ.get(PROVIDERS[model]["key_env"]):
sys.exit(f"error: {PROVIDERS[model]['key_env']} not set for {model}")
lock = threading.Lock()
done = set()
if os.path.exists(scores_path): # resume: skip finished cells
for line in open(scores_path, encoding="utf-8"):
try:
row = json.loads(line)
done.add((row["model"], row["prompt_id"]))
except json.JSONDecodeError:
pass
def write_row(row):
with lock:
with open(scores_path, "a", encoding="utf-8") as f:
f.write(json.dumps(row, ensure_ascii=False) + "\n")
cells = [
(entry, model)
for entry in prompts
for model in models
if (model, entry["id"]) not in done
]
print(f"[ab-v9] {len(cells)} cells ({len(prompts)} prompts × {models})", flush=True)
with ThreadPoolExecutor(max_workers=args.workers) as pool:
for entry, model in cells:
pool.submit(run_cell, entry, model, PROVIDERS[model], args.out, write_row)
rows = [json.loads(line) for line in open(scores_path, encoding="utf-8")]
report = aggregate(rows)
open(os.path.join(args.out, "report.md"), "w", encoding="utf-8").write(report)
print(report)
if __name__ == "__main__":
main()