From bacafae52a574883ba72279dd4c7582f2dccc69b Mon Sep 17 00:00:00 2001 From: Fini Date: Wed, 29 Apr 2026 09:49:48 +0800 Subject: [PATCH] feat(ab-corpus): bump minimax max_tokens to 8192 (defensive) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ab-v4 raw output capture on dashboard-search-filters-composite shows minimax-m2.7 emitting ... + 4 op_tool tags that fit inside the 4096 default — its measured completion-token average for this run was 697, well under the cap. So thinking-budget truncation is NOT the actual root cause of minimax's lower multi-tool hit rate (25% vs gpt+deepseek 50%); the real issues are instruction-following (mixed Strategy A + B despite the explicit forbidance, invented "canvas" parent_id placeholder). Still doubling the cap defensively: composite multi-tool outputs can chain 12-13 op_tool tags + thinking, and "fit easy" today doesn't mean "fits headroom-free on a longer brief tomorrow." The bump is free on the happy path (provider stops generating when done, doesn't bill unused headroom) and only ever helps when the model would otherwise hit a real ceiling. Real follow-up for minimax: instruction compliance — the no-mix rule needs to land harder than a single trailing sentence. Probably wants the rule moved to top-of-prompt + a few-shot bad-example contrast. Out of scope here; tracked under Phase 2 prompt design. --- scripts/ab-corpus/clients/minimax.ts | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/scripts/ab-corpus/clients/minimax.ts b/scripts/ab-corpus/clients/minimax.ts index d40587c4f..e6303e736 100644 --- a/scripts/ab-corpus/clients/minimax.ts +++ b/scripts/ab-corpus/clients/minimax.ts @@ -33,7 +33,16 @@ export async function callMinimax(args: CallMinimaxArgs): Promise blocks in output-parser.ts); the openai-compat + // default 4096 is fine for obvious prompts but cuts close on + // composite multi-tool outputs (12-tag responses + thinking can + // approach 3-4k easy). Double the cap defensively — we already pay + // for thinking either way, and most replies still come in well + // under 1k completion tokens (ab-v4 avg 697), so the bigger cap + // costs nothing on the happy path and only matters when the model + // would otherwise truncate mid-output. + maxTokens: args.maxTokens ?? 8192, label: 'minimax', // ab-v3 saw 4 minimax timeouts on 104 runs — all wall-clock aborts, // none were content-quality issues. One retry picks up the