Harness at scripts/ab-corpus/ wires the pen-ai-skills corpus evaluator to
real model endpoints and pen-mcp handlers:
run.ts — CLI entry (--dry-run / --live / --models A,B,C / --only ID)
apply.ts — ApplyFn impl dispatching tool_call → element handler
and batch_design DSL → handleBatchDesign, against a
fresh tmp .op per run (isolated, auto-cleanup)
build-prompt.ts — B variant strips elements.md + appends batch_design
<op_tool> format instruction; T keeps elements + adds
element-tool PRIMARY / batch_design FALLBACK
instruction. Uniform <op_tool> wrapper in both arms
isolates "tool set width" as the only A/B variable.
stub-model.ts — fixture-based offline model for --dry-run
real-model.ts — router by model id (minimax* / gpt-*/o* / glm-5.1 /
glm-* / kimi-*)
clients/
openai-compat.ts — generic chat/completions POST
minimax.ts — api.minimax.io/v1, MINIMAX_API_KEY
codex-cli.ts — spawns `codex exec` (GPT-5.4 via Codex Pro sub)
bailian.ts — coding.dashscope.aliyuncs.com/v1 CP,
DASHSCOPE_BAILIAN_CODING_KEY (hosts glm-4.7, kimi-k2.5)
glm.ts — open.bigmodel.cn/api/coding/paas/v4 official CP,
GLM_OFFICIAL_CODING_KEY
write-report.ts — Report → report.md + report.json in out dir;
4-way routing breakdown table per model
Kept entirely outside packages/ — scripts are a local dev tool, not part
of the published SDK. API keys never hit disk or git.
v1 run results logged separately in openpencil-docs
superpowers/notes/2026-04-20-ab-v1-results.md (5 models × 24 prompts).
123 lines
5 KiB
TypeScript
123 lines
5 KiB
TypeScript
/**
|
|
* Offline model stub for --dry-run mode. Returns hardcoded outputs that
|
|
* cover all four kinds the scorer can see (tool_call, batch_design,
|
|
* garbage, plus a "mostly right" batch_design that passes shape checks
|
|
* too). Used to verify the full pipeline without burning API credits.
|
|
*
|
|
* Keyed by (promptId, variant) so individual prompts can be told to
|
|
* return specific shapes. Fallback response when a prompt isn't
|
|
* explicitly fixtured: treatment variant returns a generic
|
|
* batch_design (simulates "weak model didn't route to element tool"),
|
|
* baseline returns a minimal valid design.
|
|
*
|
|
* When the real model adapter lands, swap this out with a thin
|
|
* wrapper around apps/web/src/services/ai/ai-service.ts::streamChat
|
|
* (or a provider-native fetch). The return-shape contract — a single
|
|
* raw string the scorer parses — stays the same.
|
|
*/
|
|
|
|
import type { CorpusPrompt } from '@zseven-w/pen-ai-skills';
|
|
|
|
export interface ModelCall {
|
|
model: string;
|
|
prompt: CorpusPrompt;
|
|
variant: 'B' | 'T';
|
|
/** Full resolved system+user prompt text sent to the model. Ignored
|
|
* by the stub, kept so the real adapter has the same signature. */
|
|
systemPrompt: string;
|
|
userPrompt: string;
|
|
}
|
|
|
|
export async function stubModelCall(call: ModelCall): Promise<string> {
|
|
const key = `${call.prompt.id}|${call.variant}`;
|
|
const fixture = FIXTURES[key];
|
|
if (fixture) return fixture;
|
|
// Fallback: treatment-arm on unfixtured prompts simulates a weak
|
|
// model that knew about the tools but picked batch_design anyway
|
|
// (interesting M5 signal); baseline always emits batch_design DSL.
|
|
return DEFAULT_BATCH_DESIGN;
|
|
}
|
|
|
|
// batch_design's DSL parser expects one operation per line. Keep this
|
|
// on a single physical line or it fails at apply time — the point of
|
|
// a fallback stub is to produce a *legal* design so M1 passes.
|
|
const DEFAULT_BATCH_DESIGN =
|
|
'root=I(null, {"type":"frame","name":"Root","role":"section","width":375,"height":812,"layout":"vertical","children":[]})';
|
|
|
|
/**
|
|
* A few hand-crafted fixtures so the dry-run report has variety in it
|
|
* (some M1 passes, some fails, some tool_calls, some batch_design). The
|
|
* shapes here aren't meant to be realistic model outputs — they're just
|
|
* enough to exercise the pipeline's branches.
|
|
*/
|
|
const FIXTURES: Record<string, string> = {
|
|
// mobile-filter-chips (obvious) in treatment: correct tool_call routing
|
|
'mobile-filter-chips|T': [
|
|
'<op_tool>',
|
|
'{"name": "add_nav_chip_row_v0", "arguments": {',
|
|
' "items": [',
|
|
' {"label": "All", "active": true},',
|
|
' {"label": "Videos", "icon": "video"},',
|
|
' {"label": "Photos", "icon": "image"},',
|
|
' {"label": "Music", "icon": "music"},',
|
|
' {"label": "Articles", "icon": "file-text"},',
|
|
' {"label": "Podcasts", "icon": "headphones"},',
|
|
' {"label": "Events", "icon": "calendar"},',
|
|
' {"label": "Books", "icon": "book"}',
|
|
' ]',
|
|
'}}',
|
|
'</op_tool>',
|
|
].join('\n'),
|
|
// mobile-empty-inbox (obvious) in treatment: correct tool routing
|
|
'mobile-empty-inbox|T': [
|
|
'<op_tool>',
|
|
'{"name": "add_empty_state_v0", "arguments": {',
|
|
' "title": "No messages yet",',
|
|
' "subtitle": "New conversations will appear here",',
|
|
' "icon": "inbox",',
|
|
' "cta_label": "Start a conversation"',
|
|
'}}',
|
|
'</op_tool>',
|
|
].join('\n'),
|
|
// dashboard-alert-banner (obvious) in treatment: correct routing
|
|
'dashboard-alert-banner|T': [
|
|
'<op_tool>',
|
|
'{"name": "add_alert_v0", "arguments": {',
|
|
' "message": "Your subscription expires in 7 days.",',
|
|
' "icon": "triangle-alert",',
|
|
' "dismissible": true',
|
|
'}}',
|
|
'</op_tool>',
|
|
].join('\n'),
|
|
// landing-testimonial (obvious) in treatment: correct routing
|
|
'landing-testimonial|T': [
|
|
'<op_tool>',
|
|
'{"name": "add_quote_block_v0", "arguments": {',
|
|
' "quote": "OpenPencil replaced two tools in our stack.",',
|
|
' "author": "Alex Chen"',
|
|
'}}',
|
|
'</op_tool>',
|
|
].join('\n'),
|
|
// landing-product-ratings (obvious) in treatment: WRONG tool routing.
|
|
// Prompt asks for star rating (expected_tool_if_any=add_rating_stars_v0)
|
|
// but stub simulates a weak model routing to add_badge_v0 — the
|
|
// badge IS a schema-valid tool call, just wrong intent. Exercises
|
|
// the wrong-tool routing classification.
|
|
'landing-product-ratings|T': [
|
|
'<op_tool>',
|
|
'{"name": "add_badge_v0", "arguments": {',
|
|
' "label": "4.8"',
|
|
'}}',
|
|
'</op_tool>',
|
|
].join('\n'),
|
|
// Baseline garbage response (simulates a weak model producing prose)
|
|
'mobile-tabs-demo|B':
|
|
'I can help you design that! Let me create a tabs demo for you with three underline tabs.',
|
|
// Treatment garbage on an OBVIOUS prompt — exercises the Garbage
|
|
// column in the routing breakdown. Codex stop-hook 2026-04-20 fix:
|
|
// if garbage isn't counted in the M5 denominator, right-tool rates
|
|
// look rosy even when most outputs fail to parse.
|
|
'dashboard-wizard-stepper|T':
|
|
'Sure! Here is a stepper wizard header: <thinking about design> ... actually let me just describe it.',
|
|
};
|