feat(ai): enhance streaming service with runtime config and thinking modes
- Extract timeout constants to ai-runtime-config.ts - Add thinking mode control (adaptive/disabled/enabled) to stream options - Add StyleGuide interface for orchestrator visual consistency - Add ping timeout and first-text timeout options
This commit is contained in:
parent
531874da10
commit
7ebc4db95b
113
src/services/ai/ai-runtime-config.ts
Normal file
113
src/services/ai/ai-runtime-config.ts
Normal file
|
|
@ -0,0 +1,113 @@
|
|||
export type ThinkingMode = 'adaptive' | 'disabled' | 'enabled'
|
||||
export type ThinkingEffort = 'low' | 'medium' | 'high' | 'max'
|
||||
|
||||
export const DEFAULT_THINKING_MODE: ThinkingMode = 'adaptive'
|
||||
export const DEFAULT_THINKING_EFFORT: ThinkingEffort = 'low'
|
||||
|
||||
export const DEFAULT_THINKING_CONFIG = {
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
} as const
|
||||
|
||||
export const CHAT_STREAM_THINKING_CONFIG = {
|
||||
...DEFAULT_THINKING_CONFIG,
|
||||
} as const
|
||||
|
||||
export const STREAM_TIMEOUT_MIN_MS = 10_000
|
||||
export const DEFAULT_STREAM_HARD_TIMEOUT_MS = 600_000
|
||||
export const DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS = 300_000
|
||||
export const DEFAULT_GENERATE_TIMEOUT_MS = 180_000
|
||||
|
||||
export const PROMPT_OPTIMIZER_LIMITS = {
|
||||
longPromptCharThreshold: 2200,
|
||||
maxPromptCharsForOrchestrator: 2400,
|
||||
maxPromptCharsForSubAgent: 3600,
|
||||
maxFeatureLines: 12,
|
||||
maxSectionLines: 14,
|
||||
maxFallbackSections: 8,
|
||||
} as const
|
||||
|
||||
export const SUB_AGENT_TIMEOUT_BASE = {
|
||||
hardTimeoutMs: 420_000,
|
||||
noTextTimeoutMs: 210_000,
|
||||
thinkingResetsTimeout: true,
|
||||
} as const
|
||||
|
||||
export const PROMPT_TIMEOUT_BUCKETS = {
|
||||
mediumPromptMaxChars: 4200,
|
||||
} as const
|
||||
|
||||
export const SUB_AGENT_TIMEOUT_PROFILES = {
|
||||
short: {
|
||||
...SUB_AGENT_TIMEOUT_BASE,
|
||||
pingResetsTimeout: false,
|
||||
firstTextTimeoutMs: 420_000,
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
},
|
||||
medium: {
|
||||
...SUB_AGENT_TIMEOUT_BASE,
|
||||
hardTimeoutMs: 600_000,
|
||||
noTextTimeoutMs: 300_000,
|
||||
pingResetsTimeout: false,
|
||||
firstTextTimeoutMs: 600_000,
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
},
|
||||
long: {
|
||||
...SUB_AGENT_TIMEOUT_BASE,
|
||||
hardTimeoutMs: 900_000,
|
||||
noTextTimeoutMs: 480_000,
|
||||
pingResetsTimeout: false,
|
||||
firstTextTimeoutMs: 900_000,
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
},
|
||||
} as const
|
||||
|
||||
export const ORCHESTRATOR_TIMEOUT_PROFILES = {
|
||||
short: {
|
||||
hardTimeoutMs: 300_000,
|
||||
noTextTimeoutMs: 150_000,
|
||||
thinkingResetsTimeout: true,
|
||||
pingResetsTimeout: false,
|
||||
firstTextTimeoutMs: 300_000,
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
},
|
||||
medium: {
|
||||
hardTimeoutMs: 420_000,
|
||||
noTextTimeoutMs: 210_000,
|
||||
thinkingResetsTimeout: true,
|
||||
pingResetsTimeout: false,
|
||||
firstTextTimeoutMs: 420_000,
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
},
|
||||
long: {
|
||||
hardTimeoutMs: 600_000,
|
||||
noTextTimeoutMs: 300_000,
|
||||
thinkingResetsTimeout: true,
|
||||
pingResetsTimeout: false,
|
||||
firstTextTimeoutMs: 600_000,
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
},
|
||||
} as const
|
||||
|
||||
export const DESIGN_STREAM_TIMEOUTS = {
|
||||
hardTimeoutMs: 900_000,
|
||||
noTextTimeoutMs: 480_000,
|
||||
thinkingResetsTimeout: true,
|
||||
pingResetsTimeout: false,
|
||||
firstTextTimeoutMs: 900_000,
|
||||
thinkingMode: DEFAULT_THINKING_MODE,
|
||||
effort: DEFAULT_THINKING_EFFORT,
|
||||
} as const
|
||||
|
||||
export const RETRY_TIMEOUT_CONFIG = {
|
||||
multiplier: 2,
|
||||
hardTimeoutMaxMs: 1_200_000,
|
||||
noTextTimeoutMaxMs: 480_000,
|
||||
firstTextTimeoutMaxMs: 1_200_000,
|
||||
} as const
|
||||
|
|
@ -1,8 +1,11 @@
|
|||
import type { AIStreamChunk } from './ai-types'
|
||||
import type { AIModelInfo } from '@/stores/ai-store'
|
||||
|
||||
const DEFAULT_STREAM_HARD_TIMEOUT_MS = 180_000
|
||||
const DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS = 75_000
|
||||
import {
|
||||
DEFAULT_GENERATE_TIMEOUT_MS,
|
||||
DEFAULT_STREAM_HARD_TIMEOUT_MS,
|
||||
DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS,
|
||||
STREAM_TIMEOUT_MIN_MS,
|
||||
} from './ai-runtime-config'
|
||||
|
||||
interface StreamChatOptions {
|
||||
hardTimeoutMs?: number
|
||||
|
|
@ -13,6 +16,28 @@ interface StreamChatOptions {
|
|||
* where thinking should NOT prevent the no-text timeout from firing.
|
||||
*/
|
||||
thinkingResetsTimeout?: boolean
|
||||
/**
|
||||
* Whether keep-alive ping events reset the no-text timeout.
|
||||
* Default: true (backward compatible). Set to false to avoid endless
|
||||
* waiting when the server only emits pings.
|
||||
*/
|
||||
pingResetsTimeout?: boolean
|
||||
/**
|
||||
* Max time to wait for the first non-empty text token.
|
||||
* This timeout is independent from keep-alive pings/thinking chunks.
|
||||
*/
|
||||
firstTextTimeoutMs?: number
|
||||
/**
|
||||
* Controls provider thinking mode.
|
||||
* - adaptive: model decides thinking depth
|
||||
* - disabled: disable extended thinking for faster first text
|
||||
* - enabled: explicitly enable extended thinking
|
||||
*/
|
||||
thinkingMode?: 'adaptive' | 'disabled' | 'enabled'
|
||||
/** Thinking budget (used when thinkingMode === 'enabled'). */
|
||||
thinkingBudgetTokens?: number
|
||||
/** Model effort level (low is usually faster). */
|
||||
effort?: 'low' | 'medium' | 'high' | 'max'
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -26,13 +51,19 @@ export async function* streamChat(
|
|||
options?: StreamChatOptions,
|
||||
provider?: string,
|
||||
): AsyncGenerator<AIStreamChunk> {
|
||||
const hardTimeoutMs = Math.max(10_000, options?.hardTimeoutMs ?? DEFAULT_STREAM_HARD_TIMEOUT_MS)
|
||||
const noTextTimeoutMs = Math.max(10_000, options?.noTextTimeoutMs ?? DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS)
|
||||
const hardTimeoutMs = Math.max(STREAM_TIMEOUT_MIN_MS, options?.hardTimeoutMs ?? DEFAULT_STREAM_HARD_TIMEOUT_MS)
|
||||
const noTextTimeoutMs = Math.max(STREAM_TIMEOUT_MIN_MS, options?.noTextTimeoutMs ?? DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS)
|
||||
const thinkingResetsTimeout = options?.thinkingResetsTimeout ?? true
|
||||
const pingResetsTimeout = options?.pingResetsTimeout ?? true
|
||||
const firstTextTimeoutMs = options?.firstTextTimeoutMs
|
||||
? Math.max(STREAM_TIMEOUT_MIN_MS, options.firstTextTimeoutMs)
|
||||
: null
|
||||
|
||||
const controller = new AbortController()
|
||||
let abortReason: 'hard_timeout' | 'no_text_timeout' | null = null
|
||||
let abortReason: 'hard_timeout' | 'no_text_timeout' | 'first_text_timeout' | null = null
|
||||
let noTextTimeout: ReturnType<typeof setTimeout> | null = null
|
||||
let firstTextTimeout: ReturnType<typeof setTimeout> | null = null
|
||||
let sawText = false
|
||||
|
||||
const clearNoTextTimeout = () => {
|
||||
if (noTextTimeout) {
|
||||
|
|
@ -41,6 +72,13 @@ export async function* streamChat(
|
|||
}
|
||||
}
|
||||
|
||||
const clearFirstTextTimeout = () => {
|
||||
if (firstTextTimeout) {
|
||||
clearTimeout(firstTextTimeout)
|
||||
firstTextTimeout = null
|
||||
}
|
||||
}
|
||||
|
||||
const resetActivityTimeout = () => {
|
||||
clearNoTextTimeout()
|
||||
noTextTimeout = setTimeout(() => {
|
||||
|
|
@ -54,13 +92,29 @@ export async function* streamChat(
|
|||
controller.abort()
|
||||
}, hardTimeoutMs)
|
||||
|
||||
if (firstTextTimeoutMs) {
|
||||
firstTextTimeout = setTimeout(() => {
|
||||
if (sawText) return
|
||||
abortReason = 'first_text_timeout'
|
||||
controller.abort()
|
||||
}, firstTextTimeoutMs)
|
||||
}
|
||||
|
||||
resetActivityTimeout()
|
||||
|
||||
try {
|
||||
const response = await fetch('/api/ai/chat', {
|
||||
method: 'POST',
|
||||
headers: { 'Content-Type': 'application/json' },
|
||||
body: JSON.stringify({ system: systemPrompt, messages, model, provider }),
|
||||
body: JSON.stringify({
|
||||
system: systemPrompt,
|
||||
messages,
|
||||
model,
|
||||
provider,
|
||||
thinkingMode: options?.thinkingMode,
|
||||
thinkingBudgetTokens: options?.thinkingBudgetTokens,
|
||||
effort: options?.effort,
|
||||
}),
|
||||
signal: controller.signal,
|
||||
})
|
||||
|
||||
|
|
@ -69,6 +123,7 @@ export async function* streamChat(
|
|||
yield { type: 'error', content: `Server error: ${response.status} ${errBody}` }
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
return
|
||||
}
|
||||
|
||||
|
|
@ -77,6 +132,7 @@ export async function* streamChat(
|
|||
yield { type: 'error', content: 'No response stream available' }
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
return
|
||||
}
|
||||
|
||||
|
|
@ -102,6 +158,7 @@ export async function* streamChat(
|
|||
if (chunk.type === 'done') {
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
try {
|
||||
await reader.cancel()
|
||||
} catch {
|
||||
|
|
@ -112,7 +169,9 @@ export async function* streamChat(
|
|||
|
||||
// Keep-alive pings from server — reset activity timeout but don't yield
|
||||
if (chunk.type === 'ping') {
|
||||
resetActivityTimeout()
|
||||
if (pingResetsTimeout) {
|
||||
resetActivityTimeout()
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
|
|
@ -123,6 +182,8 @@ export async function* streamChat(
|
|||
// Any non-empty text counts as activity; thinking only resets
|
||||
// the timeout when thinkingResetsTimeout is true (default).
|
||||
if (chunk.type === 'text' && chunk.content.trim().length > 0) {
|
||||
sawText = true
|
||||
clearFirstTextTimeout()
|
||||
resetActivityTimeout()
|
||||
} else if (chunk.type === 'thinking' && chunk.content.trim().length > 0 && thinkingResetsTimeout) {
|
||||
resetActivityTimeout()
|
||||
|
|
@ -132,6 +193,7 @@ export async function* streamChat(
|
|||
if (chunk.type === 'error') {
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
try {
|
||||
await reader.cancel()
|
||||
} catch {
|
||||
|
|
@ -155,15 +217,22 @@ export async function* streamChat(
|
|||
if (chunk.type === 'done') {
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
return
|
||||
}
|
||||
if (chunk.type === 'thinking' && !chunk.content) {
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
return
|
||||
}
|
||||
if (chunk.type === 'text' && chunk.content.trim().length > 0) {
|
||||
sawText = true
|
||||
clearFirstTextTimeout()
|
||||
}
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
yield chunk
|
||||
if (chunk.type === 'error') {
|
||||
return
|
||||
|
|
@ -185,6 +254,11 @@ export async function* streamChat(
|
|||
type: 'error',
|
||||
content: 'AI request timed out. Please retry.',
|
||||
}
|
||||
} else if (abortReason === 'first_text_timeout') {
|
||||
yield {
|
||||
type: 'error',
|
||||
content: 'AI spent too long thinking without producing output. Request stopped, please retry.',
|
||||
}
|
||||
} else {
|
||||
yield {
|
||||
type: 'error',
|
||||
|
|
@ -193,6 +267,7 @@ export async function* streamChat(
|
|||
}
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
return
|
||||
}
|
||||
|
||||
|
|
@ -202,6 +277,7 @@ export async function* streamChat(
|
|||
} finally {
|
||||
clearTimeout(hardTimeout)
|
||||
clearNoTextTimeout()
|
||||
clearFirstTextTimeout()
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -209,8 +285,6 @@ export async function* streamChat(
|
|||
* Non-streaming completion for design/code generation.
|
||||
* Calls the server-side endpoint which reads ANTHROPIC_API_KEY from env.
|
||||
*/
|
||||
const DEFAULT_GENERATE_TIMEOUT_MS = 180_000
|
||||
|
||||
export async function generateCompletion(
|
||||
systemPrompt: string,
|
||||
userMessage: string,
|
||||
|
|
|
|||
|
|
@ -46,6 +46,24 @@ export interface SubTask {
|
|||
parentFrameId: string | null
|
||||
}
|
||||
|
||||
/** Style guide produced by the orchestrator for visual consistency */
|
||||
export interface StyleGuide {
|
||||
palette: {
|
||||
background: string
|
||||
surface: string
|
||||
text: string
|
||||
secondary: string
|
||||
accent: string
|
||||
accent2: string
|
||||
border: string
|
||||
}
|
||||
fonts: {
|
||||
heading: string
|
||||
body: string
|
||||
}
|
||||
aesthetic: string
|
||||
}
|
||||
|
||||
/** Plan produced by the orchestrator (lightweight — only structure) */
|
||||
export interface OrchestratorPlan {
|
||||
rootFrame: {
|
||||
|
|
@ -57,6 +75,7 @@ export interface OrchestratorPlan {
|
|||
gap?: number
|
||||
fill?: Array<{ type: string; color: string }>
|
||||
}
|
||||
styleGuide?: StyleGuide
|
||||
subtasks: SubTask[]
|
||||
}
|
||||
|
||||
|
|
|
|||
Loading…
Reference in a new issue