diff --git a/apps/web/src/services/ai/__tests__/element-tools-dispatcher.test.ts b/apps/web/src/services/ai/__tests__/element-tools-dispatcher.test.ts index 3ea91f9c7..6c0ec5448 100644 --- a/apps/web/src/services/ai/__tests__/element-tools-dispatcher.test.ts +++ b/apps/web/src/services/ai/__tests__/element-tools-dispatcher.test.ts @@ -1,6 +1,10 @@ import { describe, it, expect, beforeEach, vi } from 'vitest'; import { ELEMENT_TOOL_NAMES } from '@zseven-w/pen-mcp'; -import { dispatchElementToolCall, dispatchElementToolCalls } from '../element-tools-dispatcher'; +import { + dispatchElementToolCall, + dispatchElementToolCalls, + looksLikeJsonl, +} from '../element-tools-dispatcher'; import { SUPPORTED_EMBEDDED_ELEMENT_TOOLS } from '../element-tool-shims'; import { useHistoryStore } from '@/stores/history-store'; import { useDocumentStore } from '@/stores/document-store'; @@ -693,3 +697,64 @@ describe('dispatchElementToolCall — M2 shim + M3 HTTP fallback', () => { // afterEach is imported implicitly via vitest globals; re-import for clarity import { afterEach } from 'vitest'; + +describe('looksLikeJsonl', () => { + // Each weak/mid-tier model puts the JSONL in a slightly different + // wrapper. The detector has to recognize ALL of them so the + // dispatcher can re-route to applyBatchDesignAsJsonl instead of + // failing the whole subtask via the DSL parser. + + it('matches pure JSONL (one node per line, starts with {)', () => { + const jsonl = + '{"_parent":null,"id":"r","type":"frame","children":[]}\n' + + '{"_parent":"r","id":"t","type":"text","content":"hi"}'; + expect(looksLikeJsonl(jsonl)).toBe(true); + }); + + it('matches JSON-array-of-nodes (M2.7 food-app shape)', () => { + // Real failure from the M2.7 food-app run: model emitted the full + // subtask design wrapped in a single JSON array literal. The + // previous gate only checked `startsWith('{')` so this fell + // through to the DSL parser, every line of `[`/`{`/`}` failed, + // and the subtask returned empty → orchestrator retry → minimal- + // skills fallback → still empty → user got a single-frame + // placeholder with one section instead of the full design. + const jsonArray = `[ + { + "id": "filterChips-root", + "type": "frame", + "name": "Filter Chips Section", + "role": "section", + "_parent": null, + "children": [] + } +]`; + expect(looksLikeJsonl(jsonArray)).toBe(true); + }); + + it('matches array shape with leading whitespace (newlines / indent)', () => { + const padded = '\n\n [{"_parent":null,"id":"x","type":"frame"}]'; + expect(looksLikeJsonl(padded)).toBe(true); + }); + + it('rejects DSL-style assignment lines (foo=I("parent",{...}))', () => { + // Real DSL: dispatch should NOT re-route to JSONL apply. + const dsl = 'root=I(null,{"type":"frame"})\nlabel=I(root,{"type":"text"})'; + expect(looksLikeJsonl(dsl)).toBe(false); + }); + + it('rejects empty / non-bracketed strings', () => { + expect(looksLikeJsonl('')).toBe(false); + expect(looksLikeJsonl(' ')).toBe(false); + expect(looksLikeJsonl('hello world')).toBe(false); + }); + + it('rejects bracketed-but-non-JSONL content (e.g. JSON array of strings)', () => { + // A `[ ... ]` literal that doesn't carry JSONL shape keys (no + // `_parent`, no PenNode `type`) should NOT match — otherwise + // we'd reroute legit non-design array operations into the + // JSONL apply path. + expect(looksLikeJsonl('["foo", "bar"]')).toBe(false); + expect(looksLikeJsonl('[{"foo": "bar"}]')).toBe(false); + }); +}); diff --git a/apps/web/src/services/ai/element-tools-dispatcher.ts b/apps/web/src/services/ai/element-tools-dispatcher.ts index 5ad7a2877..6da4262c8 100644 --- a/apps/web/src/services/ai/element-tools-dispatcher.ts +++ b/apps/web/src/services/ai/element-tools-dispatcher.ts @@ -448,19 +448,30 @@ async function applyElementTool( /** * Detect whether `operations` looks like flat JSONL nodes rather than the * `foo=I("parent",{...})` assignment-based DSL. Weak / mid-tier models - * (GPT-5.5 standard tier observed) emit their full design as raw JSONL - * but stuff it inside `{name:"batch_design",arguments:{operations}}` - * because the ELEMENT_TOOL_OUTPUT_FORMAT prompt teaches ``-only - * output but doesn't show DSL syntax. The DSL parser then rejects every - * line. Detect this case at dispatch time and route through the JSONL - * apply path instead of failing. + * (GPT-5.5 standard tier, MiniMax M2.7 basic tier observed) emit their + * full design as JSONL but stuff it inside + * `{name:"batch_design",arguments:{operations}}` because the + * ELEMENT_TOOL_OUTPUT_FORMAT prompt teaches ``-only output + * but doesn't show DSL syntax. The DSL parser then rejects every line. + * Detect this case at dispatch time and route through the JSONL apply + * path instead of failing. * - * Signature: starts with `{` AND contains JSONL-shape keys (`_parent` - * or a `type:"frame|text|…"` field) within the first chunk. + * Two shapes count: + * - Pure JSONL: starts with `{`, one node per line. + * - JSON array: starts with `[`, all nodes wrapped in a single + * array literal. M2.7 reflexively does this because its training + * data includes a lot of "here's a JSON array of things" patterns; + * the food-app run shipped a 1300-byte single-line `[{…}, {…}]` + * that DSL parsed to "[" + "{" + "}" + "]" each on its own line, + * all rejected, the whole subtask returned empty. + * + * Either shape carries the same JSONL-style keys (`_parent` / + * `type:"frame|text|…"`) — that's the actual signature, the + * leading-character check just disambiguates from real DSL. */ -function looksLikeJsonl(operations: string): boolean { +export function looksLikeJsonl(operations: string): boolean { const trimmed = operations.trim(); - if (!trimmed.startsWith('{')) return false; + if (!trimmed.startsWith('{') && !trimmed.startsWith('[')) return false; return /"_parent"\s*:|"type"\s*:\s*"(frame|text|rectangle|ellipse|icon_font|image|path|line|polygon|group)"/.test( trimmed.slice(0, 800), );