feat(ai): enhance streaming service with runtime config and thinking modes

- Extract timeout constants to ai-runtime-config.ts
- Add thinking mode control (adaptive/disabled/enabled) to stream options
- Add StyleGuide interface for orchestrator visual consistency
- Add ping timeout and first-text timeout options
This commit is contained in:
Fini 2026-02-22 08:19:05 +08:00
parent 531874da10
commit 7ebc4db95b
3 changed files with 216 additions and 10 deletions

View file

@ -0,0 +1,113 @@
export type ThinkingMode = 'adaptive' | 'disabled' | 'enabled'
export type ThinkingEffort = 'low' | 'medium' | 'high' | 'max'
export const DEFAULT_THINKING_MODE: ThinkingMode = 'adaptive'
export const DEFAULT_THINKING_EFFORT: ThinkingEffort = 'low'
export const DEFAULT_THINKING_CONFIG = {
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
} as const
export const CHAT_STREAM_THINKING_CONFIG = {
...DEFAULT_THINKING_CONFIG,
} as const
export const STREAM_TIMEOUT_MIN_MS = 10_000
export const DEFAULT_STREAM_HARD_TIMEOUT_MS = 600_000
export const DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS = 300_000
export const DEFAULT_GENERATE_TIMEOUT_MS = 180_000
export const PROMPT_OPTIMIZER_LIMITS = {
longPromptCharThreshold: 2200,
maxPromptCharsForOrchestrator: 2400,
maxPromptCharsForSubAgent: 3600,
maxFeatureLines: 12,
maxSectionLines: 14,
maxFallbackSections: 8,
} as const
export const SUB_AGENT_TIMEOUT_BASE = {
hardTimeoutMs: 420_000,
noTextTimeoutMs: 210_000,
thinkingResetsTimeout: true,
} as const
export const PROMPT_TIMEOUT_BUCKETS = {
mediumPromptMaxChars: 4200,
} as const
export const SUB_AGENT_TIMEOUT_PROFILES = {
short: {
...SUB_AGENT_TIMEOUT_BASE,
pingResetsTimeout: false,
firstTextTimeoutMs: 420_000,
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
},
medium: {
...SUB_AGENT_TIMEOUT_BASE,
hardTimeoutMs: 600_000,
noTextTimeoutMs: 300_000,
pingResetsTimeout: false,
firstTextTimeoutMs: 600_000,
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
},
long: {
...SUB_AGENT_TIMEOUT_BASE,
hardTimeoutMs: 900_000,
noTextTimeoutMs: 480_000,
pingResetsTimeout: false,
firstTextTimeoutMs: 900_000,
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
},
} as const
export const ORCHESTRATOR_TIMEOUT_PROFILES = {
short: {
hardTimeoutMs: 300_000,
noTextTimeoutMs: 150_000,
thinkingResetsTimeout: true,
pingResetsTimeout: false,
firstTextTimeoutMs: 300_000,
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
},
medium: {
hardTimeoutMs: 420_000,
noTextTimeoutMs: 210_000,
thinkingResetsTimeout: true,
pingResetsTimeout: false,
firstTextTimeoutMs: 420_000,
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
},
long: {
hardTimeoutMs: 600_000,
noTextTimeoutMs: 300_000,
thinkingResetsTimeout: true,
pingResetsTimeout: false,
firstTextTimeoutMs: 600_000,
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
},
} as const
export const DESIGN_STREAM_TIMEOUTS = {
hardTimeoutMs: 900_000,
noTextTimeoutMs: 480_000,
thinkingResetsTimeout: true,
pingResetsTimeout: false,
firstTextTimeoutMs: 900_000,
thinkingMode: DEFAULT_THINKING_MODE,
effort: DEFAULT_THINKING_EFFORT,
} as const
export const RETRY_TIMEOUT_CONFIG = {
multiplier: 2,
hardTimeoutMaxMs: 1_200_000,
noTextTimeoutMaxMs: 480_000,
firstTextTimeoutMaxMs: 1_200_000,
} as const

View file

@ -1,8 +1,11 @@
import type { AIStreamChunk } from './ai-types'
import type { AIModelInfo } from '@/stores/ai-store'
const DEFAULT_STREAM_HARD_TIMEOUT_MS = 180_000
const DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS = 75_000
import {
DEFAULT_GENERATE_TIMEOUT_MS,
DEFAULT_STREAM_HARD_TIMEOUT_MS,
DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS,
STREAM_TIMEOUT_MIN_MS,
} from './ai-runtime-config'
interface StreamChatOptions {
hardTimeoutMs?: number
@ -13,6 +16,28 @@ interface StreamChatOptions {
* where thinking should NOT prevent the no-text timeout from firing.
*/
thinkingResetsTimeout?: boolean
/**
* Whether keep-alive ping events reset the no-text timeout.
* Default: true (backward compatible). Set to false to avoid endless
* waiting when the server only emits pings.
*/
pingResetsTimeout?: boolean
/**
* Max time to wait for the first non-empty text token.
* This timeout is independent from keep-alive pings/thinking chunks.
*/
firstTextTimeoutMs?: number
/**
* Controls provider thinking mode.
* - adaptive: model decides thinking depth
* - disabled: disable extended thinking for faster first text
* - enabled: explicitly enable extended thinking
*/
thinkingMode?: 'adaptive' | 'disabled' | 'enabled'
/** Thinking budget (used when thinkingMode === 'enabled'). */
thinkingBudgetTokens?: number
/** Model effort level (low is usually faster). */
effort?: 'low' | 'medium' | 'high' | 'max'
}
/**
@ -26,13 +51,19 @@ export async function* streamChat(
options?: StreamChatOptions,
provider?: string,
): AsyncGenerator<AIStreamChunk> {
const hardTimeoutMs = Math.max(10_000, options?.hardTimeoutMs ?? DEFAULT_STREAM_HARD_TIMEOUT_MS)
const noTextTimeoutMs = Math.max(10_000, options?.noTextTimeoutMs ?? DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS)
const hardTimeoutMs = Math.max(STREAM_TIMEOUT_MIN_MS, options?.hardTimeoutMs ?? DEFAULT_STREAM_HARD_TIMEOUT_MS)
const noTextTimeoutMs = Math.max(STREAM_TIMEOUT_MIN_MS, options?.noTextTimeoutMs ?? DEFAULT_STREAM_NO_TEXT_TIMEOUT_MS)
const thinkingResetsTimeout = options?.thinkingResetsTimeout ?? true
const pingResetsTimeout = options?.pingResetsTimeout ?? true
const firstTextTimeoutMs = options?.firstTextTimeoutMs
? Math.max(STREAM_TIMEOUT_MIN_MS, options.firstTextTimeoutMs)
: null
const controller = new AbortController()
let abortReason: 'hard_timeout' | 'no_text_timeout' | null = null
let abortReason: 'hard_timeout' | 'no_text_timeout' | 'first_text_timeout' | null = null
let noTextTimeout: ReturnType<typeof setTimeout> | null = null
let firstTextTimeout: ReturnType<typeof setTimeout> | null = null
let sawText = false
const clearNoTextTimeout = () => {
if (noTextTimeout) {
@ -41,6 +72,13 @@ export async function* streamChat(
}
}
const clearFirstTextTimeout = () => {
if (firstTextTimeout) {
clearTimeout(firstTextTimeout)
firstTextTimeout = null
}
}
const resetActivityTimeout = () => {
clearNoTextTimeout()
noTextTimeout = setTimeout(() => {
@ -54,13 +92,29 @@ export async function* streamChat(
controller.abort()
}, hardTimeoutMs)
if (firstTextTimeoutMs) {
firstTextTimeout = setTimeout(() => {
if (sawText) return
abortReason = 'first_text_timeout'
controller.abort()
}, firstTextTimeoutMs)
}
resetActivityTimeout()
try {
const response = await fetch('/api/ai/chat', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ system: systemPrompt, messages, model, provider }),
body: JSON.stringify({
system: systemPrompt,
messages,
model,
provider,
thinkingMode: options?.thinkingMode,
thinkingBudgetTokens: options?.thinkingBudgetTokens,
effort: options?.effort,
}),
signal: controller.signal,
})
@ -69,6 +123,7 @@ export async function* streamChat(
yield { type: 'error', content: `Server error: ${response.status} ${errBody}` }
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
return
}
@ -77,6 +132,7 @@ export async function* streamChat(
yield { type: 'error', content: 'No response stream available' }
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
return
}
@ -102,6 +158,7 @@ export async function* streamChat(
if (chunk.type === 'done') {
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
try {
await reader.cancel()
} catch {
@ -112,7 +169,9 @@ export async function* streamChat(
// Keep-alive pings from server — reset activity timeout but don't yield
if (chunk.type === 'ping') {
resetActivityTimeout()
if (pingResetsTimeout) {
resetActivityTimeout()
}
continue
}
@ -123,6 +182,8 @@ export async function* streamChat(
// Any non-empty text counts as activity; thinking only resets
// the timeout when thinkingResetsTimeout is true (default).
if (chunk.type === 'text' && chunk.content.trim().length > 0) {
sawText = true
clearFirstTextTimeout()
resetActivityTimeout()
} else if (chunk.type === 'thinking' && chunk.content.trim().length > 0 && thinkingResetsTimeout) {
resetActivityTimeout()
@ -132,6 +193,7 @@ export async function* streamChat(
if (chunk.type === 'error') {
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
try {
await reader.cancel()
} catch {
@ -155,15 +217,22 @@ export async function* streamChat(
if (chunk.type === 'done') {
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
return
}
if (chunk.type === 'thinking' && !chunk.content) {
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
return
}
if (chunk.type === 'text' && chunk.content.trim().length > 0) {
sawText = true
clearFirstTextTimeout()
}
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
yield chunk
if (chunk.type === 'error') {
return
@ -185,6 +254,11 @@ export async function* streamChat(
type: 'error',
content: 'AI request timed out. Please retry.',
}
} else if (abortReason === 'first_text_timeout') {
yield {
type: 'error',
content: 'AI spent too long thinking without producing output. Request stopped, please retry.',
}
} else {
yield {
type: 'error',
@ -193,6 +267,7 @@ export async function* streamChat(
}
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
return
}
@ -202,6 +277,7 @@ export async function* streamChat(
} finally {
clearTimeout(hardTimeout)
clearNoTextTimeout()
clearFirstTextTimeout()
}
}
@ -209,8 +285,6 @@ export async function* streamChat(
* Non-streaming completion for design/code generation.
* Calls the server-side endpoint which reads ANTHROPIC_API_KEY from env.
*/
const DEFAULT_GENERATE_TIMEOUT_MS = 180_000
export async function generateCompletion(
systemPrompt: string,
userMessage: string,

View file

@ -46,6 +46,24 @@ export interface SubTask {
parentFrameId: string | null
}
/** Style guide produced by the orchestrator for visual consistency */
export interface StyleGuide {
palette: {
background: string
surface: string
text: string
secondary: string
accent: string
accent2: string
border: string
}
fonts: {
heading: string
body: string
}
aesthetic: string
}
/** Plan produced by the orchestrator (lightweight — only structure) */
export interface OrchestratorPlan {
rootFrame: {
@ -57,6 +75,7 @@ export interface OrchestratorPlan {
gap?: number
fill?: Array<{ type: string; color: string }>
}
styleGuide?: StyleGuide
subtasks: SubTask[]
}