* feat(ai): enhance Windows CLI binary resolution and connection handling - Introduced functions to handle both .cmd and .ps1 wrappers for Windows installations, improving compatibility for CLI tools like Codex and Copilot. - Updated connection logic in `connect-agent.ts` to utilize environment variables for Codex home directory, enhancing flexibility in locating configuration files. - Added warning messages in connection results to inform users when no models are found, guiding them to run the CLI tools for model population. - Modified the agent settings dialog to accommodate the new warning field in connection responses, improving user feedback during the connection process. This update significantly enhances the user experience for Windows users by ensuring better handling of CLI binaries and providing clearer connection status information. * feat(ai): improve Windows binary resolution for CLI tools - Added `resolveWinExtension` function to handle extensionless binaries returned by the `where` command on Windows, ensuring compatibility with `.cmd` and `.ps1` wrappers. - Updated `connect-agent.ts` and `copilot-client.ts` to utilize the new resolution function, enhancing the reliability of binary path lookups for Codex and Copilot. - Enhanced logging to provide clearer information on resolved paths and their existence status. This update significantly improves the handling of CLI binaries on Windows, ensuring users have a smoother experience when connecting to AI tools. * feat(ai): enhance binary path handling in connect-agent - Added logic to ensure the directory of the resolved OpenCode binary is included in the PATH environment variable, improving the execution of CLI tools. - Enhanced logging to provide feedback when the binary directory is prepended to the PATH, aiding in troubleshooting and user awareness. This update improves the reliability of connecting to AI tools by ensuring the necessary binaries are accessible during execution. * feat(patching): add patch for @opencode-ai/sdk and update package configurations - Introduced a new patch for the @opencode-ai/sdk version 1.2.6 to address specific issues. - Updated package.json and bun.lock to include the new patchedDependencies section, ensuring the patch is applied during installation. - Removed redundant binary path handling code in connect-agent.ts to streamline the connection process. This update enhances the SDK's functionality while simplifying the connection logic for improved performance. * fix(ai): add orchestrator fallback and three-way intent routing Orchestrator now falls back to a heuristic plan when the model returns non-JSON instead of throwing an error. Intent classification upgraded from binary (DESIGN/CHAT) to three-way (DESIGN_NEW/DESIGN_MODIFY/CHAT). Modification requests without an explicit selection auto-target the last top-level frame on the active page, preventing them from being misrouted to the orchestrator. * chore: bump version to 0.4.4 and remove deprecated patch for @opencode-ai/sdk - Updated package version in package.json from 0.4.3 to 0.4.4. - Removed the patchedDependencies section for @opencode-ai/sdk as the patch is no longer needed. - Added new utility function `buildSpawnClaudeCodeProcess` to enhance agent SDK functionality across multiple files. * chore: bump version to 0.4.4 in package.json * chore(electron): optimize application icon assets Reduce icon file sizes for faster builds and smaller distribution. * chore: remove deprecated patchedDependencies for @opencode-ai/sdk in bun.lock - Eliminated the patchedDependencies section for @opencode-ai/sdk as the patch is no longer necessary, streamlining the dependency management in the project. * fix(ai): improve Codex CLI error extraction for auth and structured log errors Parse Codex's structured log format (<timestamp> ERROR <module>: <message>) to surface real errors like expired auth tokens instead of the unhelpful "Warning: no last agent message" fallback. * fix(ai): update error handling and remove deprecated reasoning field - Updated error checks in chat and generate handlers to ensure proper validation of required fields. - Removed the deprecated reasoning field from the OpenCode SDK integration, streamlining the prompt parameters. - Enhanced logging for system prompt injection errors to improve debugging capabilities. * fix(ai): enhance Windows compatibility for Codex CLI execution - Updated the handling of prompts on Windows to avoid parsing errors caused by shell escaping in PowerShell and cmd.exe. - Introduced a temporary PowerShell script to manage prompt input and command execution, ensuring special characters are processed correctly. - Refactored the argument passing mechanism to accommodate the new script-based approach. * fix(electron): improve process cleanup and Windows compatibility - Simplified the application quit logic to always call app.quit() on window close. - Enhanced the before-quit event to ensure proper cleanup of the Nitro process and port file. - Updated the Nitro process termination logic to use SIGKILL for reliable cleanup on non-Windows platforms. - Refactored argument handling in the Codex CLI execution to utilize PowerShell array splatting for safer argument passing. * feat(ai): add fallback for model parsing from Codex's latest-model.md - Implemented a new function to parse model IDs from the latest-model.md file when models_cache.json is unavailable, enhancing compatibility for fresh installations. - Updated logging to reflect the loading of models from the fallback method and improved error handling for model loading scenarios. * feat(ai): update prompt structure for design generation assistant - Revised the prompt format to enhance clarity and organization for the design generation assistant. - Introduced clear section headers for system instructions and user tasks, improving user experience and guidance. * refactor(ai): streamline prompt handling for Codex CLI execution - Simplified the prompt input mechanism for all platforms by utilizing stdin mode, eliminating the need for temporary scripts on Windows. - Improved argument passing to avoid shell escaping issues and command-line length limits, enhancing compatibility and reliability across environments. - Updated the execution logic to handle prompts more efficiently, ensuring a smoother user experience. --------- Co-authored-by: Fini <fini.yang@gmail.com>
243 lines
7.9 KiB
TypeScript
243 lines
7.9 KiB
TypeScript
import { defineEventHandler, readBody, setResponseHeaders } from 'h3'
|
|
import { resolveClaudeCli } from '../../utils/resolve-claude-cli'
|
|
import {
|
|
buildClaudeAgentEnv,
|
|
buildSpawnClaudeCodeProcess,
|
|
getClaudeAgentDebugFilePath,
|
|
} from '../../utils/resolve-claude-agent-env'
|
|
import { writeFile, mkdtemp, rm } from 'node:fs/promises'
|
|
import { tmpdir } from 'node:os'
|
|
import { join } from 'node:path'
|
|
import { runCodexExec } from '../../utils/codex-client'
|
|
|
|
interface ValidateBody {
|
|
system: string
|
|
message: string
|
|
imageBase64: string
|
|
model?: string
|
|
provider?: 'anthropic' | 'openai' | 'opencode'
|
|
}
|
|
|
|
/**
|
|
* Vision-based validation endpoint.
|
|
* Accepts a base64 PNG screenshot and a text prompt, sends multimodal
|
|
* content blocks for analysis via Agent SDK.
|
|
*
|
|
* Saves screenshot to temp file, asks Claude Code to read it via its
|
|
* built-in Read tool.
|
|
*/
|
|
export default defineEventHandler(async (event) => {
|
|
const body = await readBody<ValidateBody>(event)
|
|
|
|
if (!body?.system || !body?.message || !body?.imageBase64) {
|
|
setResponseHeaders(event, { 'Content-Type': 'application/json' })
|
|
return { error: 'Missing required fields: system, message, imageBase64' }
|
|
}
|
|
|
|
if (!body.model?.trim()) {
|
|
setResponseHeaders(event, { 'Content-Type': 'application/json' })
|
|
return { error: 'Missing model. Model fallback is disabled.' }
|
|
}
|
|
|
|
try {
|
|
if (body.provider === 'anthropic') {
|
|
return await validateViaAgentSDK(body, body.model)
|
|
}
|
|
if (body.provider === 'openai') {
|
|
return await validateViaCodex(body, body.model)
|
|
}
|
|
if (body.provider === 'opencode') {
|
|
return await validateViaOpenCode(body, body.model)
|
|
}
|
|
return { error: 'Missing or unsupported provider. Provider fallback is disabled.' }
|
|
} catch (error) {
|
|
const message = error instanceof Error ? error.message : 'Unknown error'
|
|
return { error: message }
|
|
}
|
|
})
|
|
|
|
function toImageBase64(data: string): string {
|
|
const dataUrlPrefix = 'data:image/png;base64,'
|
|
return data.startsWith(dataUrlPrefix) ? data.slice(dataUrlPrefix.length) : data
|
|
}
|
|
|
|
async function withTempImageFile<T>(
|
|
imageBase64: string,
|
|
run: (tempPath: string) => Promise<T>,
|
|
insideProject = false,
|
|
): Promise<T> {
|
|
let tempDir: string
|
|
if (insideProject) {
|
|
// Save inside the project directory so Claude Code Agent SDK (plan mode)
|
|
// can read the file — it restricts reads to the project directory.
|
|
const { mkdirSync, chmodSync } = await import('node:fs')
|
|
const baseDir = join(process.cwd(), '.openpencil-tmp')
|
|
mkdirSync(baseDir, { recursive: true })
|
|
chmodSync(baseDir, 0o700)
|
|
tempDir = await mkdtemp(join(baseDir, 'validate-'))
|
|
} else {
|
|
tempDir = await mkdtemp(join(tmpdir(), 'openpencil-validate-'))
|
|
}
|
|
const tempPath = join(tempDir, 'screenshot.png')
|
|
try {
|
|
await writeFile(tempPath, Buffer.from(toImageBase64(imageBase64), 'base64'))
|
|
return await run(tempPath)
|
|
} finally {
|
|
await rm(tempDir, { recursive: true, force: true }).catch(() => {})
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Agent SDK: save screenshot to a temp PNG file inside the project directory,
|
|
* then ask Claude Code to read it (Claude Code's Read tool supports images
|
|
* natively). Must use insideProject=true because plan mode restricts reads
|
|
* to the project directory.
|
|
*/
|
|
async function validateViaAgentSDK(
|
|
body: ValidateBody,
|
|
model?: string,
|
|
): Promise<{ text: string; skipped?: boolean; error?: string }> {
|
|
return await withTempImageFile(body.imageBase64, async (tempPath) => {
|
|
const { query } = await import('@anthropic-ai/claude-agent-sdk')
|
|
|
|
const env = buildClaudeAgentEnv()
|
|
const debugFile = getClaudeAgentDebugFilePath()
|
|
const claudePath = resolveClaudeCli()
|
|
|
|
const prompt = `IMPORTANT: First, use the Read tool to read the image file at "${tempPath}". This is a PNG screenshot of a UI design.
|
|
|
|
After viewing the image, analyze it according to these instructions:
|
|
|
|
${body.system}
|
|
|
|
${body.message}
|
|
|
|
CRITICAL: Your ENTIRE response must be a single JSON object. No markdown, no explanation, no tool calls after reading the image. Just the JSON.`
|
|
|
|
const q = query({
|
|
prompt,
|
|
options: {
|
|
...(model ? { model } : {}),
|
|
maxTurns: 3,
|
|
tools: [],
|
|
plugins: [],
|
|
permissionMode: 'plan',
|
|
persistSession: false,
|
|
env,
|
|
...(debugFile ? { debugFile } : {}),
|
|
...(claudePath ? { pathToClaudeCodeExecutable: claudePath } : {}),
|
|
...(buildSpawnClaudeCodeProcess() ? { spawnClaudeCodeProcess: buildSpawnClaudeCodeProcess() } : {}),
|
|
},
|
|
})
|
|
|
|
try {
|
|
for await (const message of q) {
|
|
if (message.type === 'result') {
|
|
const isErrorResult = 'is_error' in message && Boolean((message as { is_error?: boolean }).is_error)
|
|
if (message.subtype === 'success' && !isErrorResult) {
|
|
return { text: message.result }
|
|
}
|
|
const errors = 'errors' in message ? (message.errors as string[]) : []
|
|
const resultText = 'result' in message ? String(message.result ?? '') : ''
|
|
return { error: errors.join('; ') || resultText || `Query ended with: ${message.subtype}`, text: '' }
|
|
}
|
|
}
|
|
} finally {
|
|
q.close()
|
|
}
|
|
|
|
return { text: '', skipped: true }
|
|
}, true)
|
|
}
|
|
|
|
async function validateViaCodex(
|
|
body: ValidateBody,
|
|
model?: string,
|
|
): Promise<{ text: string; skipped?: boolean; error?: string }> {
|
|
return await withTempImageFile(body.imageBase64, async (tempPath) => {
|
|
const result = await runCodexExec(
|
|
`${body.message}\n\nOutput ONLY the JSON object, no markdown fences, no explanation.`,
|
|
{
|
|
model,
|
|
systemPrompt: body.system,
|
|
imageFiles: [tempPath],
|
|
},
|
|
)
|
|
if (result.error) {
|
|
return { text: '', error: result.error }
|
|
}
|
|
return { text: result.text ?? '' }
|
|
})
|
|
}
|
|
|
|
function parseOpenCodeModel(model?: string): { providerID: string; modelID: string } | undefined {
|
|
if (!model || !model.includes('/')) return undefined
|
|
const idx = model.indexOf('/')
|
|
return { providerID: model.slice(0, idx), modelID: model.slice(idx + 1) }
|
|
}
|
|
|
|
async function validateViaOpenCode(
|
|
body: ValidateBody,
|
|
model?: string,
|
|
): Promise<{ text: string; skipped?: boolean; error?: string }> {
|
|
let ocServer: { close(): void } | undefined
|
|
try {
|
|
const { getOpencodeClient } = await import('../../utils/opencode-client')
|
|
const oc = await getOpencodeClient()
|
|
const ocClient: any = oc.client
|
|
ocServer = oc.server
|
|
|
|
const { data: session, error: sessionError } = await ocClient.session.create({
|
|
title: 'OpenPencil Validate',
|
|
})
|
|
if (sessionError || !session) {
|
|
return { text: '', error: 'Failed to create OpenCode session' }
|
|
}
|
|
|
|
await ocClient.session.prompt({
|
|
sessionID: session.id,
|
|
noReply: true,
|
|
parts: [{ type: 'text', text: body.system }],
|
|
})
|
|
|
|
const parsed = parseOpenCodeModel(model)
|
|
if (!parsed) {
|
|
return { text: '', error: 'Invalid OpenCode model format. Expected "provider/model".' }
|
|
}
|
|
|
|
const base64 = toImageBase64(body.imageBase64)
|
|
const promptPayload = {
|
|
sessionID: session.id,
|
|
model: parsed,
|
|
parts: [
|
|
{ type: 'image', url: `data:image/png;base64,${base64}` },
|
|
{
|
|
type: 'text',
|
|
text: `${body.message}\n\nOutput ONLY the JSON object, no markdown fences, no explanation.`,
|
|
},
|
|
],
|
|
}
|
|
|
|
const { data: result, error: promptError } = await ocClient.session.prompt(promptPayload)
|
|
if (promptError) {
|
|
return { text: '', error: 'OpenCode validation failed' }
|
|
}
|
|
|
|
const texts: string[] = []
|
|
if (result?.parts) {
|
|
for (const part of result.parts) {
|
|
if (part.type === 'text' && part.text) {
|
|
texts.push(part.text)
|
|
}
|
|
}
|
|
}
|
|
return { text: texts.join('') }
|
|
} catch (error) {
|
|
const message = error instanceof Error ? error.message : 'Unknown error'
|
|
return { text: '', error: message }
|
|
} finally {
|
|
const { releaseOpencodeServer } = await import('../../utils/opencode-client')
|
|
releaseOpencodeServer(ocServer)
|
|
}
|
|
}
|