diff --git a/src/services/ai/ai-runtime-config.ts b/src/services/ai/ai-runtime-config.ts new file mode 100644 index 000000000..9b79a4139 --- /dev/null +++ b/src/services/ai/ai-runtime-config.ts @@ -0,0 +1,113 @@ +export type ThinkingMode = 'adaptive' | 'disabled' | 'enabled' +export type ThinkingEffort = 'low' | 'medium' | 'high' | 'max' + +export const DEFAULT_THINKING_MODE: ThinkingMode = 'adaptive' +export const DEFAULT_THINKING_EFFORT: ThinkingEffort = 'low' + +export const DEFAULT_THINKING_CONFIG = { + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, +} as const + +export const CHAT_STREAM_THINKING_CONFIG = { + ...DEFAULT_THINKING_CONFIG, +} as const + +export const STREAM_TIMEOUT_MIN_MS = 10_000 +export const DEFAULT_STREAM_HARD_TIMEOUT_MS = 600_000 +export const DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS = 300_000 +export const DEFAULT_GENERATE_TIMEOUT_MS = 180_000 + +export const PROMPT_OPTIMIZER_LIMITS = { + longPromptCharThreshold: 2200, + maxPromptCharsForOrchestrator: 2400, + maxPromptCharsForSubAgent: 3600, + maxFeatureLines: 12, + maxSectionLines: 14, + maxFallbackSections: 8, +} as const + +export const SUB_AGENT_TIMEOUT_BASE = { + hardTimeoutMs: 420_000, + noTextTimeoutMs: 210_000, + thinkingResetsTimeout: true, +} as const + +export const PROMPT_TIMEOUT_BUCKETS = { + mediumPromptMaxChars: 4200, +} as const + +export const SUB_AGENT_TIMEOUT_PROFILES = { + short: { + ...SUB_AGENT_TIMEOUT_BASE, + pingResetsTimeout: false, + firstTextTimeoutMs: 420_000, + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, + }, + medium: { + ...SUB_AGENT_TIMEOUT_BASE, + hardTimeoutMs: 600_000, + noTextTimeoutMs: 300_000, + pingResetsTimeout: false, + firstTextTimeoutMs: 600_000, + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, + }, + long: { + ...SUB_AGENT_TIMEOUT_BASE, + hardTimeoutMs: 900_000, + noTextTimeoutMs: 480_000, + pingResetsTimeout: false, + firstTextTimeoutMs: 900_000, + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, + }, +} as const + +export const ORCHESTRATOR_TIMEOUT_PROFILES = { + short: { + hardTimeoutMs: 300_000, + noTextTimeoutMs: 150_000, + thinkingResetsTimeout: true, + pingResetsTimeout: false, + firstTextTimeoutMs: 300_000, + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, + }, + medium: { + hardTimeoutMs: 420_000, + noTextTimeoutMs: 210_000, + thinkingResetsTimeout: true, + pingResetsTimeout: false, + firstTextTimeoutMs: 420_000, + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, + }, + long: { + hardTimeoutMs: 600_000, + noTextTimeoutMs: 300_000, + thinkingResetsTimeout: true, + pingResetsTimeout: false, + firstTextTimeoutMs: 600_000, + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, + }, +} as const + +export const DESIGN_STREAM_TIMEOUTS = { + hardTimeoutMs: 900_000, + noTextTimeoutMs: 480_000, + thinkingResetsTimeout: true, + pingResetsTimeout: false, + firstTextTimeoutMs: 900_000, + thinkingMode: DEFAULT_THINKING_MODE, + effort: DEFAULT_THINKING_EFFORT, +} as const + +export const RETRY_TIMEOUT_CONFIG = { + multiplier: 2, + hardTimeoutMaxMs: 1_200_000, + noTextTimeoutMaxMs: 480_000, + firstTextTimeoutMaxMs: 1_200_000, +} as const diff --git a/src/services/ai/ai-service.ts b/src/services/ai/ai-service.ts index e64e0dfe6..22140b462 100644 --- a/src/services/ai/ai-service.ts +++ b/src/services/ai/ai-service.ts @@ -1,8 +1,11 @@ import type { AIStreamChunk } from './ai-types' import type { AIModelInfo } from '@/stores/ai-store' - -const DEFAULT_STREAM_HARD_TIMEOUT_MS = 180_000 -const DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS = 75_000 +import { + DEFAULT_GENERATE_TIMEOUT_MS, + DEFAULT_STREAM_HARD_TIMEOUT_MS, + DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS, + STREAM_TIMEOUT_MIN_MS, +} from './ai-runtime-config' interface StreamChatOptions { hardTimeoutMs?: number @@ -13,6 +16,28 @@ interface StreamChatOptions { * where thinking should NOT prevent the no-text timeout from firing. */ thinkingResetsTimeout?: boolean + /** + * Whether keep-alive ping events reset the no-text timeout. + * Default: true (backward compatible). Set to false to avoid endless + * waiting when the server only emits pings. + */ + pingResetsTimeout?: boolean + /** + * Max time to wait for the first non-empty text token. + * This timeout is independent from keep-alive pings/thinking chunks. + */ + firstTextTimeoutMs?: number + /** + * Controls provider thinking mode. + * - adaptive: model decides thinking depth + * - disabled: disable extended thinking for faster first text + * - enabled: explicitly enable extended thinking + */ + thinkingMode?: 'adaptive' | 'disabled' | 'enabled' + /** Thinking budget (used when thinkingMode === 'enabled'). */ + thinkingBudgetTokens?: number + /** Model effort level (low is usually faster). */ + effort?: 'low' | 'medium' | 'high' | 'max' } /** @@ -26,13 +51,19 @@ export async function* streamChat( options?: StreamChatOptions, provider?: string, ): AsyncGenerator { - const hardTimeoutMs = Math.max(10_000, options?.hardTimeoutMs ?? DEFAULT_STREAM_HARD_TIMEOUT_MS) - const noTextTimeoutMs = Math.max(10_000, options?.noTextTimeoutMs ?? DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS) + const hardTimeoutMs = Math.max(STREAM_TIMEOUT_MIN_MS, options?.hardTimeoutMs ?? DEFAULT_STREAM_HARD_TIMEOUT_MS) + const noTextTimeoutMs = Math.max(STREAM_TIMEOUT_MIN_MS, options?.noTextTimeoutMs ?? DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS) const thinkingResetsTimeout = options?.thinkingResetsTimeout ?? true + const pingResetsTimeout = options?.pingResetsTimeout ?? true + const firstTextTimeoutMs = options?.firstTextTimeoutMs + ? Math.max(STREAM_TIMEOUT_MIN_MS, options.firstTextTimeoutMs) + : null const controller = new AbortController() - let abortReason: 'hard_timeout' | 'no_text_timeout' | null = null + let abortReason: 'hard_timeout' | 'no_text_timeout' | 'first_text_timeout' | null = null let noTextTimeout: ReturnType | null = null + let firstTextTimeout: ReturnType | null = null + let sawText = false const clearNoTextTimeout = () => { if (noTextTimeout) { @@ -41,6 +72,13 @@ export async function* streamChat( } } + const clearFirstTextTimeout = () => { + if (firstTextTimeout) { + clearTimeout(firstTextTimeout) + firstTextTimeout = null + } + } + const resetActivityTimeout = () => { clearNoTextTimeout() noTextTimeout = setTimeout(() => { @@ -54,13 +92,29 @@ export async function* streamChat( controller.abort() }, hardTimeoutMs) + if (firstTextTimeoutMs) { + firstTextTimeout = setTimeout(() => { + if (sawText) return + abortReason = 'first_text_timeout' + controller.abort() + }, firstTextTimeoutMs) + } + resetActivityTimeout() try { const response = await fetch('/api/ai/chat', { method: 'POST', headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ system: systemPrompt, messages, model, provider }), + body: JSON.stringify({ + system: systemPrompt, + messages, + model, + provider, + thinkingMode: options?.thinkingMode, + thinkingBudgetTokens: options?.thinkingBudgetTokens, + effort: options?.effort, + }), signal: controller.signal, }) @@ -69,6 +123,7 @@ export async function* streamChat( yield { type: 'error', content: `Server error: ${response.status} ${errBody}` } clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() return } @@ -77,6 +132,7 @@ export async function* streamChat( yield { type: 'error', content: 'No response stream available' } clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() return } @@ -102,6 +158,7 @@ export async function* streamChat( if (chunk.type === 'done') { clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() try { await reader.cancel() } catch { @@ -112,7 +169,9 @@ export async function* streamChat( // Keep-alive pings from server — reset activity timeout but don't yield if (chunk.type === 'ping') { - resetActivityTimeout() + if (pingResetsTimeout) { + resetActivityTimeout() + } continue } @@ -123,6 +182,8 @@ export async function* streamChat( // Any non-empty text counts as activity; thinking only resets // the timeout when thinkingResetsTimeout is true (default). if (chunk.type === 'text' && chunk.content.trim().length > 0) { + sawText = true + clearFirstTextTimeout() resetActivityTimeout() } else if (chunk.type === 'thinking' && chunk.content.trim().length > 0 && thinkingResetsTimeout) { resetActivityTimeout() @@ -132,6 +193,7 @@ export async function* streamChat( if (chunk.type === 'error') { clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() try { await reader.cancel() } catch { @@ -155,15 +217,22 @@ export async function* streamChat( if (chunk.type === 'done') { clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() return } if (chunk.type === 'thinking' && !chunk.content) { clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() return } + if (chunk.type === 'text' && chunk.content.trim().length > 0) { + sawText = true + clearFirstTextTimeout() + } clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() yield chunk if (chunk.type === 'error') { return @@ -185,6 +254,11 @@ export async function* streamChat( type: 'error', content: 'AI request timed out. Please retry.', } + } else if (abortReason === 'first_text_timeout') { + yield { + type: 'error', + content: 'AI spent too long thinking without producing output. Request stopped, please retry.', + } } else { yield { type: 'error', @@ -193,6 +267,7 @@ export async function* streamChat( } clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() return } @@ -202,6 +277,7 @@ export async function* streamChat( } finally { clearTimeout(hardTimeout) clearNoTextTimeout() + clearFirstTextTimeout() } } @@ -209,8 +285,6 @@ export async function* streamChat( * Non-streaming completion for design/code generation. * Calls the server-side endpoint which reads ANTHROPIC_API_KEY from env. */ -const DEFAULT_GENERATE_TIMEOUT_MS = 180_000 - export async function generateCompletion( systemPrompt: string, userMessage: string, diff --git a/src/services/ai/ai-types.ts b/src/services/ai/ai-types.ts index 9aa0ac6e6..1e799c84f 100644 --- a/src/services/ai/ai-types.ts +++ b/src/services/ai/ai-types.ts @@ -46,6 +46,24 @@ export interface SubTask { parentFrameId: string | null } +/** Style guide produced by the orchestrator for visual consistency */ +export interface StyleGuide { + palette: { + background: string + surface: string + text: string + secondary: string + accent: string + accent2: string + border: string + } + fonts: { + heading: string + body: string + } + aesthetic: string +} + /** Plan produced by the orchestrator (lightweight — only structure) */ export interface OrchestratorPlan { rootFrame: { @@ -57,6 +75,7 @@ export interface OrchestratorPlan { gap?: number fill?: Array<{ type: string; color: string }> } + styleGuide?: StyleGuide subtasks: SubTask[] }