From 6842d0fccab8b2fa8b26e9217efd997fbf161492 Mon Sep 17 00:00:00 2001 From: Fini Date: Thu, 2 Jul 2026 21:46:35 +0800 Subject: [PATCH] chore(smoke): audit mode, reasoning gate for glm, self-loop harness MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - OPENPENCIL_SMOKE_AUDIT=: load the doc, run the real-layout geometry diagnostics, print a JSON report, exit non-zero on issues — the machine-checkable leg of the generate→render→audit loop. - extend the harness thinking-disable gate to glm models (reasoning burned the whole token budget and returned empty content; an orchestrator sidebar subtask failed 3x and shipped missing). - scripts/self-loop.sh: prompts file → generate → render (the real canvas pipeline) → audit → scorecard.json, fully unattended. --- crates/op-smoke/src/main.rs | 62 +++++++++++++++++++- scripts/self-loop.sh | 113 ++++++++++++++++++++++++++++++++++++ 2 files changed, 174 insertions(+), 1 deletion(-) create mode 100755 scripts/self-loop.sh diff --git a/crates/op-smoke/src/main.rs b/crates/op-smoke/src/main.rs index 49c493dbf..856b5f343 100644 --- a/crates/op-smoke/src/main.rs +++ b/crates/op-smoke/src/main.rs @@ -178,6 +178,14 @@ fn is_minimax_model(model: &str) -> bool { m.starts_with("minimax") || m.starts_with("abab") } +/// GLM reasoning models burn the whole `max_tokens` budget on reasoning and +/// return EMPTY content unless thinking is disabled (measured: an orchestrator +/// sidebar subtask failed 3× "empty content from provider" and shipped +/// missing). Mirrors the production gate in `chat_builtin_http::is_glm_model`. +fn is_glm_model(model: &str) -> bool { + model.to_ascii_lowercase().starts_with("glm") +} + /// Direct openai-compat `LlmClient` for the harness (OPENPENCIL_SMOKE_DIRECT=1). /// /// The default [`SmokeLlmClient`] goes through the vendored `agent` QueryEngine, @@ -244,7 +252,8 @@ impl LlmClient for DirectOpenAiClient { // (17% M3, ~10s answers); this lets the M3-with-think arm be // benchmarked (strip_reasoning handles the blocks). let keep_thinking = std::env::var("OPENPENCIL_SMOKE_KEEP_THINKING").is_ok(); - if !keep_thinking && (force_disable || is_minimax_model(&model)) { + if !keep_thinking && (force_disable || is_minimax_model(&model) || is_glm_model(&model)) + { if let Some(obj) = body.as_object_mut() { obj.insert("thinking".into(), serde_json::json!({ "type": "disabled" })); } @@ -659,6 +668,57 @@ async fn main() -> std::process::ExitCode { return code; } + // `OPENPENCIL_SMOKE_AUDIT=` — self-loop quality gate: load the + // doc, run the REAL-layout geometry diagnostics (the same detector family + // the per-batch feedback uses), print a JSON report, exit. Zero LLM calls. + // Exit code 0 = structurally clean, 1 = issues found. This is the + // machine-checkable leg of the develop→generate→render→audit loop. + if let Ok(audit_path) = std::env::var("OPENPENCIL_SMOKE_AUDIT") { + let text = match std::fs::read_to_string(&audit_path) { + Ok(s) => s, + Err(e) => { + eprintln!("[AUDIT] read {audit_path}: {e}"); + return std::process::ExitCode::from(3); + } + }; + let doc: jian_ops_schema::PenDocument = match serde_json::from_str(&text) { + Ok(d) => d, + Err(e) => { + eprintln!("[AUDIT] parse {audit_path}: {e}"); + return std::process::ExitCode::from(3); + } + }; + let state = op_editor_core::EditorState::from_document(doc); + let issues = op_orchestrator::geometry_validation::geometry_diagnostics(&state); + let roots: Vec = state + .active_children() + .iter() + .map(|n| { + use op_editor_core::PenNodeExt; + format!( + "{} ({})", + n.base().name.as_deref().unwrap_or("?"), + n.id_str() + ) + }) + .collect(); + let report = serde_json::json!({ + "file": audit_path, + "roots": roots, + "issueCount": issues.len(), + "issues": issues, + }); + println!( + "{}", + serde_json::to_string_pretty(&report).unwrap_or_default() + ); + return if issues.is_empty() { + std::process::ExitCode::SUCCESS + } else { + std::process::ExitCode::from(1) + }; + } + // `OPENPENCIL_SMOKE_PROGRAM=` runs a Pencil-style batch_design DSL // PROGRAM (a `binding=I(parent,{...})` tree-builder) against the doc and // saves it, bypassing the orchestrator entirely. This is the experiment diff --git a/scripts/self-loop.sh b/scripts/self-loop.sh new file mode 100755 index 000000000..6bc69ac0b --- /dev/null +++ b/scripts/self-loop.sh @@ -0,0 +1,113 @@ +#!/usr/bin/env bash +# Self-contained develop→generate→render→audit loop, zero human in the loop. +# +# scripts/self-loop.sh [out-dir] +# +# prompts.txt: one prompt per line (# comments / blank lines skipped). +# For each prompt N: +# 1. GENERATE op-smoke (orchestrator DIRECT by default; set +# OPENPENCIL_SMOKE_LOOP=1 for the agentic loop) +# → //pNN.op +# 2. RENDER openpencil-desktop --render-shots — the REAL jian-core +# (taffy) layout + jian-skia paint, the exact desktop canvas +# pipeline → //pNN_shots/*.png +# 3. AUDIT op-smoke OPENPENCIL_SMOKE_AUDIT — real-layout geometry +# diagnostics (collapse / table overflow / text & frame +# overflow / sibling jam), ALSO computed by jian, so +# generation feedback, audit and final pixels share ONE +# layout engine → //pNN.audit.json +# Finally assembles //scorecard.json. +# +# Vision scoring is the agent's job afterwards: Read each shot PNG and score +# it against openpencil-docs/self-loop/rubric.md (same rubric scores the +# Pencil reference anchors, giving the quantified gap). +# +# Model env (required): OPENPENCIL_LLM_PROVIDER / OPENPENCIL_LLM_BASE_URL / +# OPENPENCIL_LLM_API_KEY / OPENPENCIL_ORCHESTRATOR_MODEL. +set -uo pipefail +REPO="$(cd "$(dirname "$0")/.." && pwd)" +RUN="${1:?run name required}" +PROMPTS="${2:?prompts file required}" +OUT="${3:-/tmp/self-loop}/$RUN" +mkdir -p "$OUT" +SMOKE="$REPO/target/release/op-smoke" +DESKTOP="$REPO/target/release/openpencil-desktop" +[ -x "$SMOKE" ] && [ -x "$DESKTOP" ] || { + echo "build first: cargo build --release -p op-smoke -p op-host-desktop" >&2 + exit 2 +} + +i=0 +while IFS= read -r prompt; do + case "$prompt" in ''|'#'*) continue ;; esac + i=$((i + 1)) + tag=$(printf 'p%02d' "$i") + op="$OUT/$tag.op" + printf '%s' "$prompt" >"$OUT/$tag.prompt" + echo "== [$tag] ${prompt:0:80}..." >&2 + + if [ -n "${OPENPENCIL_SMOKE_LOOP:-}" ]; then + OPENPENCIL_SMOKE_OUT="$op" "$SMOKE" "$prompt" \ + >"$OUT/$tag.stdout" 2>"$OUT/$tag.log" + else + OPENPENCIL_SMOKE_DIRECT=1 OPENPENCIL_SMOKE_OUT="$op" "$SMOKE" "$prompt" \ + >"$OUT/$tag.stdout" 2>"$OUT/$tag.log" + fi + gen_ok=false + [ -s "$op" ] && gen_ok=true + + render_ok=false + shots="$OUT/${tag}_shots" + if [ "$gen_ok" = true ]; then + mkdir -p "$shots" + "$DESKTOP" --render-shots "$op" "$shots" 2 >/dev/null 2>&1 \ + && [ -n "$(ls -A "$shots" 2>/dev/null)" ] && render_ok=true + fi + + if [ "$gen_ok" = true ]; then + OPENPENCIL_SMOKE_AUDIT="$op" "$SMOKE" audit >"$OUT/$tag.audit.json" 2>/dev/null || true + fi + + GEN_OK="$gen_ok" RENDER_OK="$render_ok" TAG="$tag" OUT_DIR="$OUT" python3 - <<'PY' +import json, os +out, tag = os.environ["OUT_DIR"], os.environ["TAG"] +row = { + "tag": tag, + "prompt": open(f"{out}/{tag}.prompt").read()[:120], + "genOk": os.environ["GEN_OK"] == "true", + "renderOk": os.environ["RENDER_OK"] == "true", + "nodes": 0, + "auditIssues": -1, +} +try: + d = json.load(open(f"{out}/{tag}.op")) + def cnt(n): + return 1 + sum(cnt(c) for c in n.get("children", [])) + roots = d.get("children") or d.get("pages", [{}])[0].get("children", []) + row["nodes"] = sum(cnt(x) for x in roots) +except Exception: + pass +try: + row["auditIssues"] = json.load(open(f"{out}/{tag}.audit.json"))["issueCount"] +except Exception: + pass +json.dump(row, open(f"{out}/{tag}.row.json", "w"), ensure_ascii=False) +print(f" gen={row['genOk']} render={row['renderOk']} nodes={row['nodes']} audit_issues={row['auditIssues']}") +PY +done <"$PROMPTS" + +RUN_NAME="$RUN" OUT_DIR="$OUT" python3 - <<'PY' +import glob, json, os +out = os.environ["OUT_DIR"] +rows = [json.load(open(p)) for p in sorted(glob.glob(f"{out}/p*.row.json"))] +clean = sum(1 for r in rows if r["renderOk"] and r["auditIssues"] == 0) +card = { + "run": os.environ["RUN_NAME"], + "prompts": len(rows), + "structurallyClean": clean, + "rows": rows, +} +json.dump(card, open(f"{out}/scorecard.json", "w"), indent=1, ensure_ascii=False) +print(json.dumps(card, indent=1, ensure_ascii=False)) +PY +echo "scorecard → $OUT/scorecard.json" >&2