Codex stop-hook caught: ab-v3 introduced composite-difficulty prompts
that *expect* multi-tool emit (e.g. 5× member_row + 1× invite_row
for a team page), but `ParsedOutput.tool_call` was a single
{name, arguments} so the parser silently dropped every call after
the first. apply.ts only invoked one tool, M3 min_roles couldn't
pass on legitimately-routed multi-tool runs, and byTool stats
under-counted. The composite routing 'multi-tool' bucket was
correctly assigned in classifyRouting, but downstream the pipeline
behaved as if the model emitted a single call.
This commit replaces `kind: 'tool_call'` with
`kind: 'tool_calls'` (NON-EMPTY list) across every consumer:
- types.ts: ParsedOutput tagged union; new ParsedOpToolCall.
ScoreRow.toolName → toolNames: string[].
- output-parser.ts: collects ALL element-tool tags in emit order;
unknown-tool path also surfaces as single-element tool_calls so
routing keeps the same wrong-tool semantics.
- score-run.ts: classifyRouting uses Array.includes for obvious
prompts (right-tool when ANY emitted call matches expected_tool —
over-production isn't a routing miss). Composite stays multi-tool
on any non-empty list.
- aggregate.ts byTool: tallies EVERY name in toolNames, so a
composite row that emits 6× add_activity_log_v0 + 1×
add_section_header_v0 contributes 6+1 = 7 invocations across two
tools (with row-level m1_legal applied to both buckets — apply is
all-or-nothing).
- apply.ts: loops over parsed.calls and invokes
handleElementToolCall in emit order. Any single call failing
aborts the row (M1=false); we don't partial-apply.
- mock-llm.ts mockLlmParsed: collects all `<op_tool>` tags into the
list (composite-prompt mocks can carry multi-call raw strings).
- apps/web design-parser.tryParseElementToolOutput: maps tool_calls
→ its single-shape DesignOutputShape contract using the FIRST
call (the multi-tag path `tryParseAllElementToolOutputs` was
already correct).
Tests: 3746 → 3750 vitest. New cases:
- output-parser: surfaces ALL element-tool tags in emit order with
intermixed batch_design scaffolds dropped (3 element calls from
5 tags).
- score-run: right-tool when expected appears alongside extras;
composite multi-call captures every name in toolNames.
- aggregate: 6× activity_log + 1× section_header → byTool reports
6 and 1 invocations respectively.
dry-run on ab-v3 produces a 208-row report; tsc + format clean.
89 lines
3.3 KiB
TypeScript
89 lines
3.3 KiB
TypeScript
/**
|
|
* Concrete ApplyFn implementation that routes element-tool calls and
|
|
* batch_design DSL through the real pen-mcp handlers. Lives in
|
|
* scripts/ (not a package) to keep the pen-ai-skills ↔ pen-mcp
|
|
* dependency unidirectional — the scorer in packages/pen-ai-skills
|
|
* declares the ApplyFn contract, this file implements it.
|
|
*/
|
|
|
|
import { mkdtempSync, writeFileSync, readFileSync, rmSync } from 'node:fs';
|
|
import { tmpdir } from 'node:os';
|
|
import { join } from 'node:path';
|
|
import type { PenDocument } from '@zseven-w/pen-types';
|
|
import type { ApplyFn, ApplyResult, ParsedOutput } from '@zseven-w/pen-ai-skills';
|
|
import {
|
|
ELEMENT_TOOL_NAMES,
|
|
handleBatchDesign,
|
|
handleElementToolCall,
|
|
invalidateCache,
|
|
} from '@zseven-w/pen-mcp';
|
|
|
|
const EMPTY = JSON.stringify({ version: '1.0.0', children: [] });
|
|
|
|
/**
|
|
* Apply a parsed output to a fresh, disposable .op file. The scorer
|
|
* reads the resulting PenDocument back and does not need the file to
|
|
* persist — we clean up per call.
|
|
*
|
|
* On success: `ok: true` + `doc: PenDocument`.
|
|
* On failure (handler throws OR writes no root): `ok: false` +
|
|
* `error: <message>` + `doc: null`. Throws are caught so a single
|
|
* bad run never halts the whole corpus sweep.
|
|
*/
|
|
export const applyToFreshDoc: ApplyFn = async (parsed: ParsedOutput): Promise<ApplyResult> => {
|
|
if (parsed.kind === 'garbage') {
|
|
return { ok: false, error: `unparseable: ${parsed.reason}`, doc: null };
|
|
}
|
|
const tmpDir = mkdtempSync(join(tmpdir(), 'ab-corpus-apply-'));
|
|
const fp = join(tmpDir, 'doc.op');
|
|
writeFileSync(fp, EMPTY, 'utf-8');
|
|
try {
|
|
if (parsed.kind === 'tool_calls') {
|
|
// Apply every call into the same fresh doc, in emit order.
|
|
// Composite prompts route here with N≥2 calls; obvious prompts
|
|
// typically with N=1. Any single call failing aborts the row
|
|
// (M1=false) — we don't partially apply.
|
|
for (let i = 0; i < parsed.calls.length; i += 1) {
|
|
const call = parsed.calls[i];
|
|
if (!ELEMENT_TOOL_NAMES.has(call.name)) {
|
|
return {
|
|
ok: false,
|
|
error: `unknown element tool "${call.name}" (call ${i + 1}/${parsed.calls.length})`,
|
|
doc: null,
|
|
};
|
|
}
|
|
await handleElementToolCall(call.name, { ...call.arguments, filePath: fp });
|
|
}
|
|
} else {
|
|
const result = await handleBatchDesign({
|
|
operations: parsed.dsl,
|
|
filePath: fp,
|
|
postProcess: false,
|
|
});
|
|
if (result.errors && result.errors.length > 0) {
|
|
const summary = result.errors.map((e) => `${e.line.slice(0, 60)}: ${e.error}`).join('; ');
|
|
return { ok: false, error: `batch_design errors: ${summary}`, doc: null };
|
|
}
|
|
}
|
|
const doc = JSON.parse(readFileSync(fp, 'utf-8')) as PenDocument;
|
|
const topLevel = (doc.children ?? doc.pages?.[0]?.children ?? []) as unknown[];
|
|
if (topLevel.length === 0) {
|
|
return { ok: false, error: 'apply produced no root nodes', doc: null };
|
|
}
|
|
return { ok: true, error: '', doc };
|
|
} catch (err) {
|
|
return {
|
|
ok: false,
|
|
error: err instanceof Error ? err.message : String(err),
|
|
doc: null,
|
|
};
|
|
} finally {
|
|
invalidateCache(fp);
|
|
try {
|
|
rmSync(tmpDir, { recursive: true, force: true });
|
|
} catch {
|
|
// Best-effort cleanup; leaving a tmp file behind is harmless.
|
|
}
|
|
}
|
|
};
|