openpencil/scripts/ab-corpus/real-model.ts
Fini 51878c3894 feat(scripts): ab-corpus harness with multi-provider model adapters
Harness at scripts/ab-corpus/ wires the pen-ai-skills corpus evaluator to
real model endpoints and pen-mcp handlers:

run.ts            — CLI entry (--dry-run / --live / --models A,B,C / --only ID)
apply.ts          — ApplyFn impl dispatching tool_call → element handler
                    and batch_design DSL → handleBatchDesign, against a
                    fresh tmp .op per run (isolated, auto-cleanup)
build-prompt.ts   — B variant strips elements.md + appends batch_design
                    <op_tool> format instruction; T keeps elements + adds
                    element-tool PRIMARY / batch_design FALLBACK
                    instruction. Uniform <op_tool> wrapper in both arms
                    isolates "tool set width" as the only A/B variable.
stub-model.ts     — fixture-based offline model for --dry-run
real-model.ts     — router by model id (minimax* / gpt-*/o* / glm-5.1 /
                    glm-* / kimi-*)
clients/
  openai-compat.ts — generic chat/completions POST
  minimax.ts       — api.minimax.io/v1, MINIMAX_API_KEY
  codex-cli.ts     — spawns `codex exec` (GPT-5.4 via Codex Pro sub)
  bailian.ts       — coding.dashscope.aliyuncs.com/v1 CP,
                     DASHSCOPE_BAILIAN_CODING_KEY (hosts glm-4.7, kimi-k2.5)
  glm.ts           — open.bigmodel.cn/api/coding/paas/v4 official CP,
                     GLM_OFFICIAL_CODING_KEY
write-report.ts   — Report → report.md + report.json in out dir;
                    4-way routing breakdown table per model

Kept entirely outside packages/ — scripts are a local dev tool, not part
of the published SDK. API keys never hit disk or git.

v1 run results logged separately in openpencil-docs
superpowers/notes/2026-04-20-ab-v1-results.md (5 models × 24 prompts).
2026-04-20 23:53:23 +08:00

117 lines
4.2 KiB
TypeScript

/**
* Live model dispatcher. Picks the right client per model id and
* invokes it with the full system + user prompt. Exposed as
* `realModelCall` with the same signature as `stubModelCall` so the
* harness swaps them behind the `--live` flag without conditional
* logic in the main loop.
*
* Router rules:
* - `minimax*` → clients/minimax.ts (needs MINIMAX_API_KEY in env)
* - `gpt-*` / `o1` / `o3` / `o4` → clients/codex-cli.ts (uses
* Codex CLI subscription)
* - anything else → throw with the supported list
*/
import { callMinimax } from './clients/minimax';
import { callCodex } from './clients/codex-cli';
import { callBailian } from './clients/bailian';
import { callGlm } from './clients/glm';
import { buildSystemPrompt } from './build-prompt';
import type { ModelCall } from './stub-model';
/**
* Router rules (checked in priority order, first match wins):
*
* - `minimax*` → clients/minimax.ts (MINIMAX_API_KEY)
* - `gpt-*` / o* → clients/codex-cli.ts (Codex CLI subscription)
* - `glm-5.1` → clients/glm.ts (GLM_OFFICIAL_CODING_KEY,
* GLM-official coding-plan endpoint)
* - `glm-*` → clients/bailian.ts (DASHSCOPE_BAILIAN_CODING_KEY,
* earlier GLM versions hosted on
* Bailian's DashScope aggregator)
* - `kimi-*` → clients/bailian.ts (DASHSCOPE_BAILIAN_CODING_KEY,
* Kimi K-series also on Bailian)
*
* Anything else → throw with the supported list.
*/
export async function realModelCall(call: ModelCall): Promise<string> {
const built = buildSystemPrompt(call.variant);
const user = call.prompt.prompt;
const model = call.model;
if (/^minimax/i.test(model)) {
return callMinimax({
model: mapMinimaxId(model),
system: built.system,
user,
});
}
if (/^(gpt-|o1|o3|o4)/i.test(model)) {
return callCodex({ model, system: built.system, user });
}
if (/^glm-5\.1/i.test(model)) {
return callGlm({
model: mapGlmOfficialId(model),
system: built.system,
user,
});
}
if (/^glm-/i.test(model)) {
return callBailian({
model: mapGlmBailianId(model),
system: built.system,
user,
});
}
if (/^kimi/i.test(model)) {
return callBailian({
model: mapKimiBailianId(model),
system: built.system,
user,
});
}
throw new Error(
`No live adapter for model "${model}". Supported: minimax-* (MINIMAX_API_KEY), gpt-*/o1/o3/o4 (Codex CLI), glm-5.1 (GLM_OFFICIAL_CODING_KEY), glm-5 / kimi-* (DASHSCOPE_BAILIAN_CODING_KEY).`,
);
}
function mapMinimaxId(id: string): string {
const lower = id.toLowerCase();
if (lower.includes('m2.7') || lower === 'minimax-m2' || lower === 'minimax-m2.7') {
return 'MiniMax-M2.7';
}
return id;
}
function mapGlmOfficialId(id: string): string {
// GLM official CP ships "glm-4.7" as the current coding model
// (per builtin-provider-presets.ts modelPlaceholder). User's
// preferred short form "glm-5.1" maps to that id here.
const lower = id.toLowerCase();
if (lower === 'glm-5.1' || lower === 'glm-5.1-coding') return 'glm-4.7';
return id;
}
function mapGlmBailianId(id: string): string {
return mapGlmBailianIdInternal(id);
}
function mapKimiBailianId(id: string): string {
// Bailian CP exposes Moonshot Kimi as `kimi-k2.5` — the `k` prefix
// matters; plain `kimi-2.5` gets HTTP 400 "model not supported".
// Source: Alibaba Cloud Model Studio Coding Plan FAQ (Apr 2026).
const lower = id.toLowerCase();
if (lower === 'kimi-2.5' || lower === 'kimi-k2.5' || lower === 'kimi') return 'kimi-k2.5';
return id;
}
function mapGlmBailianIdInternal(id: string): string {
// Bailian CP exposes Zhipu GLM as `glm-4.7` (current coding-plan
// preset id). Harness id `glm-5` is accepted as an alias by the
// server and produced responses in smoke tests, but standardize on
// the documented id to avoid silent deprecation.
const lower = id.toLowerCase();
if (lower === 'glm-5' || lower === 'glm-4.7' || lower === 'glm-coding') return 'glm-4.7';
return id;
}