openpencil/scripts/ab-corpus/inspect-issue-category.ts
Kayshen-X a7d73ebb62 feat(ai): pencil-style agentic design tool-loop, multi-chat tabs, #27 panel restyle
Built-in design generation now runs as an agentic MCP tool-loop (reusing the
agent-rs BuiltInProvider), gated behind OPENPENCIL_DESIGN_AGENT_LOOP / the
Settings experimental toggle; the orchestrator stays the default.

- design-agent system prompt + in-process design toolset (parity-locked with
  the MCP surface) + flag-gated Intent::Design routing
- spawn_agents execution as sequential sub-loops + live creation-mode badges
  (per-agent glow + 'N/M designing...' header)
- new MCP tools: get_guidelines, ToolSearch, get_screenshot, get_editor_state,
  export_nodes, spawn_agents; style-guide local audit
- #27 AI panel restyle: rounded tool cards + green check-rings, gray user
  bubbles, model-pill bottom toolbar, header, empty-state pills, the
  PARALLEL AGENTS (agent_team_size) 1x-6x chip dropdown
- multi-chat tabs: ChatSessions model (Deref-to-active) + tab row UI
  (switch / close / + / Cmd+T) with each run bound to its tab

Large checkpoint commit spanning the working tree (Rust shell crates).
2026-07-02 21:21:06 +08:00

60 lines
1.8 KiB
TypeScript

/**
* Print every issue of a specific category across an entire run, with the
* row id, node id, and reason. Generic version of inspect-contrast-hits
* — used for spot-checking any single detector's true-positive vs
* false-positive rate against a real corpus.
*/
import { readFileSync } from 'node:fs';
import { join } from 'node:path';
import {
detectAllIssues,
parseModelOutput,
type Issue,
type IssueCategory,
} from '@zseven-w/pen-ai-skills';
import { applyToFreshDoc } from './apply';
interface JsonlRow {
promptId: string;
category: string;
difficulty: string;
variant: string;
rawOutput: string;
}
async function main(): Promise<void> {
const runId = process.argv[2];
const wantCat = process.argv[3] as IssueCategory | undefined;
if (!runId || !wantCat) {
console.error('usage: bun run scripts/ab-corpus/inspect-issue-category.ts <run-id> <category>');
process.exit(1);
}
const path = join(import.meta.dir, 'runs', runId, 'scores.jsonl');
const rows: JsonlRow[] = readFileSync(path, 'utf-8')
.split('\n')
.filter(Boolean)
.map((l) => JSON.parse(l));
for (const r of rows) {
const parsed = parseModelOutput(r.rawOutput);
if (parsed.kind === 'garbage') continue;
const result = await applyToFreshDoc(parsed);
if (!result.ok || !result.doc) continue;
const root = result.doc.children?.[0] ?? null;
if (!root) continue;
const issues: Issue[] = detectAllIssues(root, result.doc).filter((i) => i.category === wantCat);
if (issues.length === 0) continue;
console.log(
`\n=== ${r.promptId} [${r.category}/${r.difficulty}/${r.variant}] (${issues.length} hits) ===`,
);
for (const issue of issues) {
console.log(` ${issue.nodeId.padEnd(50)} ${issue.reason}`);
}
}
}
main().catch((e) => {
console.error(e);
process.exit(1);
});