Built-in design generation now runs as an agentic MCP tool-loop (reusing the agent-rs BuiltInProvider), gated behind OPENPENCIL_DESIGN_AGENT_LOOP / the Settings experimental toggle; the orchestrator stays the default. - design-agent system prompt + in-process design toolset (parity-locked with the MCP surface) + flag-gated Intent::Design routing - spawn_agents execution as sequential sub-loops + live creation-mode badges (per-agent glow + 'N/M designing...' header) - new MCP tools: get_guidelines, ToolSearch, get_screenshot, get_editor_state, export_nodes, spawn_agents; style-guide local audit - #27 AI panel restyle: rounded tool cards + green check-rings, gray user bubbles, model-pill bottom toolbar, header, empty-state pills, the PARALLEL AGENTS (agent_team_size) 1x-6x chip dropdown - multi-chat tabs: ChatSessions model (Deref-to-active) + tab row UI (switch / close / + / Cmd+T) with each run bound to its tab Large checkpoint commit spanning the working tree (Rust shell crates).
60 lines
1.8 KiB
TypeScript
60 lines
1.8 KiB
TypeScript
/**
|
|
* Print every issue of a specific category across an entire run, with the
|
|
* row id, node id, and reason. Generic version of inspect-contrast-hits
|
|
* — used for spot-checking any single detector's true-positive vs
|
|
* false-positive rate against a real corpus.
|
|
*/
|
|
import { readFileSync } from 'node:fs';
|
|
import { join } from 'node:path';
|
|
import {
|
|
detectAllIssues,
|
|
parseModelOutput,
|
|
type Issue,
|
|
type IssueCategory,
|
|
} from '@zseven-w/pen-ai-skills';
|
|
import { applyToFreshDoc } from './apply';
|
|
|
|
interface JsonlRow {
|
|
promptId: string;
|
|
category: string;
|
|
difficulty: string;
|
|
variant: string;
|
|
rawOutput: string;
|
|
}
|
|
|
|
async function main(): Promise<void> {
|
|
const runId = process.argv[2];
|
|
const wantCat = process.argv[3] as IssueCategory | undefined;
|
|
if (!runId || !wantCat) {
|
|
console.error('usage: bun run scripts/ab-corpus/inspect-issue-category.ts <run-id> <category>');
|
|
process.exit(1);
|
|
}
|
|
const path = join(import.meta.dir, 'runs', runId, 'scores.jsonl');
|
|
const rows: JsonlRow[] = readFileSync(path, 'utf-8')
|
|
.split('\n')
|
|
.filter(Boolean)
|
|
.map((l) => JSON.parse(l));
|
|
|
|
for (const r of rows) {
|
|
const parsed = parseModelOutput(r.rawOutput);
|
|
if (parsed.kind === 'garbage') continue;
|
|
const result = await applyToFreshDoc(parsed);
|
|
if (!result.ok || !result.doc) continue;
|
|
const root = result.doc.children?.[0] ?? null;
|
|
if (!root) continue;
|
|
const issues: Issue[] = detectAllIssues(root, result.doc).filter((i) => i.category === wantCat);
|
|
if (issues.length === 0) continue;
|
|
console.log(
|
|
`\n=== ${r.promptId} [${r.category}/${r.difficulty}/${r.variant}] (${issues.length} hits) ===`,
|
|
);
|
|
for (const issue of issues) {
|
|
console.log(` ${issue.nodeId.padEnd(50)} ${issue.reason}`);
|
|
}
|
|
}
|
|
}
|
|
|
|
main().catch((e) => {
|
|
console.error(e);
|
|
process.exit(1);
|
|
});
|