/** * Generic OpenAI-compatible chat client. Most Chinese providers (MiniMax, * Bailian/DashScope, Zhipu/GLM, Moonshot/Kimi) expose an OpenAI-compatible * `/chat/completions` endpoint — we route every one of them through this * single helper so the per-provider wrappers stay 3-5 lines each and the * harness doesn't hardcode a different fetch shape for every vendor. * * Treats API errors as fatal for the single call (throws); the harness's * loop catches and records them as apply-phase failures so one bad call * never aborts a corpus sweep. */ import type { TokenUsage } from '@zseven-w/pen-ai-skills'; interface ChatMessage { role: 'system' | 'user'; content: string; } interface OpenAICompatResponse { choices?: Array<{ message?: { content?: string }; finish_reason?: string; }>; usage?: { prompt_tokens?: number; completion_tokens?: number; total_tokens?: number; }; error?: { message?: string; code?: string }; } /** * Standardized provider response carrying the assistant message plus * usage stats. All ab-corpus clients (openai-compat, codex-cli, stub) * normalize to this shape so the harness can plumb token counts into * ScoreRow without per-client special cases. usage = 0/0 means * "provider didn't return usage", not "actually used 0 tokens" — * aggregate.avgUsage skips zero rows so codex-cli columns don't * appear free. */ export interface ChatCallResult { content: string; usage: TokenUsage; } export interface CallOpenAICompatArgs { /** Full endpoint base URL, e.g. `https://api.minimaxi.com/v1`. No trailing `/chat/completions`. */ baseURL: string; /** Bearer token. Passed verbatim as `Authorization: Bearer `. */ apiKey: string; /** Provider-specific model id, e.g. `MiniMax-M2.7`, `glm-5`, `kimi-k2.5`. */ model: string; system: string; user: string; temperature?: number; maxTokens?: number; /** Label for error messages so "