openpencil/scripts/ab-corpus/stub-model.ts
Fini 51878c3894 feat(scripts): ab-corpus harness with multi-provider model adapters
Harness at scripts/ab-corpus/ wires the pen-ai-skills corpus evaluator to
real model endpoints and pen-mcp handlers:

run.ts            — CLI entry (--dry-run / --live / --models A,B,C / --only ID)
apply.ts          — ApplyFn impl dispatching tool_call → element handler
                    and batch_design DSL → handleBatchDesign, against a
                    fresh tmp .op per run (isolated, auto-cleanup)
build-prompt.ts   — B variant strips elements.md + appends batch_design
                    <op_tool> format instruction; T keeps elements + adds
                    element-tool PRIMARY / batch_design FALLBACK
                    instruction. Uniform <op_tool> wrapper in both arms
                    isolates "tool set width" as the only A/B variable.
stub-model.ts     — fixture-based offline model for --dry-run
real-model.ts     — router by model id (minimax* / gpt-*/o* / glm-5.1 /
                    glm-* / kimi-*)
clients/
  openai-compat.ts — generic chat/completions POST
  minimax.ts       — api.minimax.io/v1, MINIMAX_API_KEY
  codex-cli.ts     — spawns `codex exec` (GPT-5.4 via Codex Pro sub)
  bailian.ts       — coding.dashscope.aliyuncs.com/v1 CP,
                     DASHSCOPE_BAILIAN_CODING_KEY (hosts glm-4.7, kimi-k2.5)
  glm.ts           — open.bigmodel.cn/api/coding/paas/v4 official CP,
                     GLM_OFFICIAL_CODING_KEY
write-report.ts   — Report → report.md + report.json in out dir;
                    4-way routing breakdown table per model

Kept entirely outside packages/ — scripts are a local dev tool, not part
of the published SDK. API keys never hit disk or git.

v1 run results logged separately in openpencil-docs
superpowers/notes/2026-04-20-ab-v1-results.md (5 models × 24 prompts).
2026-04-20 23:53:23 +08:00

123 lines
5 KiB
TypeScript

/**
* Offline model stub for --dry-run mode. Returns hardcoded outputs that
* cover all four kinds the scorer can see (tool_call, batch_design,
* garbage, plus a "mostly right" batch_design that passes shape checks
* too). Used to verify the full pipeline without burning API credits.
*
* Keyed by (promptId, variant) so individual prompts can be told to
* return specific shapes. Fallback response when a prompt isn't
* explicitly fixtured: treatment variant returns a generic
* batch_design (simulates "weak model didn't route to element tool"),
* baseline returns a minimal valid design.
*
* When the real model adapter lands, swap this out with a thin
* wrapper around apps/web/src/services/ai/ai-service.ts::streamChat
* (or a provider-native fetch). The return-shape contract — a single
* raw string the scorer parses — stays the same.
*/
import type { CorpusPrompt } from '@zseven-w/pen-ai-skills';
export interface ModelCall {
model: string;
prompt: CorpusPrompt;
variant: 'B' | 'T';
/** Full resolved system+user prompt text sent to the model. Ignored
* by the stub, kept so the real adapter has the same signature. */
systemPrompt: string;
userPrompt: string;
}
export async function stubModelCall(call: ModelCall): Promise<string> {
const key = `${call.prompt.id}|${call.variant}`;
const fixture = FIXTURES[key];
if (fixture) return fixture;
// Fallback: treatment-arm on unfixtured prompts simulates a weak
// model that knew about the tools but picked batch_design anyway
// (interesting M5 signal); baseline always emits batch_design DSL.
return DEFAULT_BATCH_DESIGN;
}
// batch_design's DSL parser expects one operation per line. Keep this
// on a single physical line or it fails at apply time — the point of
// a fallback stub is to produce a *legal* design so M1 passes.
const DEFAULT_BATCH_DESIGN =
'root=I(null, {"type":"frame","name":"Root","role":"section","width":375,"height":812,"layout":"vertical","children":[]})';
/**
* A few hand-crafted fixtures so the dry-run report has variety in it
* (some M1 passes, some fails, some tool_calls, some batch_design). The
* shapes here aren't meant to be realistic model outputs — they're just
* enough to exercise the pipeline's branches.
*/
const FIXTURES: Record<string, string> = {
// mobile-filter-chips (obvious) in treatment: correct tool_call routing
'mobile-filter-chips|T': [
'<op_tool>',
'{"name": "add_nav_chip_row_v0", "arguments": {',
' "items": [',
' {"label": "All", "active": true},',
' {"label": "Videos", "icon": "video"},',
' {"label": "Photos", "icon": "image"},',
' {"label": "Music", "icon": "music"},',
' {"label": "Articles", "icon": "file-text"},',
' {"label": "Podcasts", "icon": "headphones"},',
' {"label": "Events", "icon": "calendar"},',
' {"label": "Books", "icon": "book"}',
' ]',
'}}',
'</op_tool>',
].join('\n'),
// mobile-empty-inbox (obvious) in treatment: correct tool routing
'mobile-empty-inbox|T': [
'<op_tool>',
'{"name": "add_empty_state_v0", "arguments": {',
' "title": "No messages yet",',
' "subtitle": "New conversations will appear here",',
' "icon": "inbox",',
' "cta_label": "Start a conversation"',
'}}',
'</op_tool>',
].join('\n'),
// dashboard-alert-banner (obvious) in treatment: correct routing
'dashboard-alert-banner|T': [
'<op_tool>',
'{"name": "add_alert_v0", "arguments": {',
' "message": "Your subscription expires in 7 days.",',
' "icon": "triangle-alert",',
' "dismissible": true',
'}}',
'</op_tool>',
].join('\n'),
// landing-testimonial (obvious) in treatment: correct routing
'landing-testimonial|T': [
'<op_tool>',
'{"name": "add_quote_block_v0", "arguments": {',
' "quote": "OpenPencil replaced two tools in our stack.",',
' "author": "Alex Chen"',
'}}',
'</op_tool>',
].join('\n'),
// landing-product-ratings (obvious) in treatment: WRONG tool routing.
// Prompt asks for star rating (expected_tool_if_any=add_rating_stars_v0)
// but stub simulates a weak model routing to add_badge_v0 — the
// badge IS a schema-valid tool call, just wrong intent. Exercises
// the wrong-tool routing classification.
'landing-product-ratings|T': [
'<op_tool>',
'{"name": "add_badge_v0", "arguments": {',
' "label": "4.8"',
'}}',
'</op_tool>',
].join('\n'),
// Baseline garbage response (simulates a weak model producing prose)
'mobile-tabs-demo|B':
'I can help you design that! Let me create a tabs demo for you with three underline tabs.',
// Treatment garbage on an OBVIOUS prompt — exercises the Garbage
// column in the routing breakdown. Codex stop-hook 2026-04-20 fix:
// if garbage isn't counted in the M5 denominator, right-tool rates
// look rosy even when most outputs fail to parse.
'dashboard-wizard-stepper|T':
'Sure! Here is a stepper wizard header: <thinking about design> ... actually let me just describe it.',
};