fix(ab-corpus): pin codex reasoning effort to medium

gpt-5.5 在 codex CLI 默认走 xhigh,单调用 1-2 分钟,把 ab-v5 全
量 sweep 从 ~15 分钟拉到预测 13 小时。pin 到 medium 跟 ab-v4
gpt-5.4 历史值同档,保留 AB_CORPUS_CODEX_REASONING env override
留给以后想测全火力上限。
This commit is contained in:
Fini 2026-04-29 09:49:59 +08:00
parent 2902bdc88e
commit be04c8beea

View file

@ -61,6 +61,13 @@ export async function callCodex(args: CallCodexArgs): Promise<ChatCallResult> {
const prompt = `SYSTEM CONTEXT (treat as authoritative system prompt):\n<<<\n${args.system}\n>>>\n\nUSER REQUEST:\n${args.user}`;
try {
const content = await new Promise<string>((resolve, reject) => {
// Pin reasoning effort to `medium` so ab-corpus stays comparable
// across model bumps. Codex CLI's per-model default isn't fixed:
// gpt-5.4 historically ran ~medium; gpt-5.5 defaults to `xhigh`,
// which sent ab-v5 wall time from ~15 min (ab-v4) to a projected
// ~13 h. Override via `AB_CORPUS_CODEX_REASONING` if you ever
// need to measure full-effort ceiling.
const reasoningEffort = process.env.AB_CORPUS_CODEX_REASONING || 'medium';
const proc = spawn(
'codex',
[
@ -69,6 +76,8 @@ export async function callCodex(args: CallCodexArgs): Promise<ChatCallResult> {
'--ephemeral',
'--sandbox',
'read-only',
'-c',
`model_reasoning_effort=${reasoningEffort}`,
'-m',
args.model,
'--output-last-message',