diff --git a/app/api/realtime/ephemeral-token/route.ts b/app/api/realtime/ephemeral-token/route.ts deleted file mode 100644 index 7f60058..0000000 --- a/app/api/realtime/ephemeral-token/route.ts +++ /dev/null @@ -1,56 +0,0 @@ -import { NextRequest, NextResponse } from 'next/server'; - -const REALTIME_MODEL = process.env.OPENAI_REALTIME_MODEL || 'gpt-4o-realtime-preview-2024-12-17'; - -export const runtime = 'nodejs'; -export const dynamic = 'force-dynamic'; - -export async function POST(request: NextRequest) { - try { - const authHeader = request.headers.get('authorization'); - if (!authHeader) { - return NextResponse.json({ error: 'Missing authorization header' }, { status: 401 }); - } - - const apiKey = - process.env.RAH_REALTIME_OPENAI_API_KEY || - process.env.RAH_DELEGATE_OPENAI_API_KEY || - process.env.OPENAI_API_KEY; - - if (!apiKey) { - return NextResponse.json({ error: 'Realtime OpenAI API key is not configured' }, { status: 500 }); - } - - const response = await fetch('https://api.openai.com/v1/realtime/sessions', { - method: 'POST', - headers: { - 'Content-Type': 'application/json', - Authorization: `Bearer ${apiKey}`, - }, - body: JSON.stringify({ - model: REALTIME_MODEL, - modalities: ['text'], - instructions: 'Provide high-accuracy streaming transcription only. Never speak responses.', - }), - }); - - if (!response.ok) { - const errorPayload = await response.json().catch(() => null); - const message = errorPayload?.error?.message || response.statusText || 'Failed to create realtime session'; - return NextResponse.json({ error: message }, { status: response.status }); - } - - const data = await response.json(); - return NextResponse.json({ - client_secret: data.client_secret, - expires_at: data.expires_at, - model: data.model ?? REALTIME_MODEL, - voice: data.voice, - id: data.id, - }); - } catch (error) { - const message = error instanceof Error ? error.message : String(error); - console.error('[realtime] Failed to mint ephemeral token:', message); - return NextResponse.json({ error: message }, { status: 500 }); - } -} diff --git a/app/api/voice/tts/route.ts b/app/api/voice/tts/route.ts deleted file mode 100644 index 6687158..0000000 --- a/app/api/voice/tts/route.ts +++ /dev/null @@ -1,101 +0,0 @@ -import { NextRequest, NextResponse } from 'next/server'; -import { randomUUID } from 'crypto'; -import { recordVoiceUsage } from '@/services/voice/usageLogger'; - -const OPENAI_TTS_MODEL = process.env.RAH_TTS_MODEL || 'gpt-4o-mini-tts'; -const DEFAULT_TTS_VOICE = process.env.RAH_TTS_VOICE || 'ash'; -const DEFAULT_TTS_COST_PER_1K_CHAR_USD = 0.015; -const TTS_COST_PER_1K_CHAR_USD = (() => { - const raw = process.env.RAH_TTS_COST_PER_1K_CHAR_USD; - if (!raw) return DEFAULT_TTS_COST_PER_1K_CHAR_USD; - const parsed = Number(raw); - return Number.isFinite(parsed) && parsed > 0 ? parsed : DEFAULT_TTS_COST_PER_1K_CHAR_USD; -})(); - -function estimateTtsCost(charCount: number) { - const cost = (charCount / 1000) * TTS_COST_PER_1K_CHAR_USD; - return Number.isFinite(cost) ? parseFloat(cost.toFixed(6)) : 0; -} - -export const runtime = 'nodejs'; -export const dynamic = 'force-dynamic'; - -export async function POST(request: NextRequest) { - try { - const apiKey = process.env.RAH_VOICE_OPENAI_API_KEY || process.env.RAH_DELEGATE_OPENAI_API_KEY || process.env.OPENAI_API_KEY; - if (!apiKey) { - return NextResponse.json({ error: 'OpenAI API key is not configured' }, { status: 500 }); - } - - const body = await request.json().catch(() => null); - const text = typeof body?.text === 'string' ? body.text.trim() : ''; - const voice = typeof body?.voice === 'string' && body.voice.trim().length > 0 ? body.voice.trim() : DEFAULT_TTS_VOICE; - const helperName = typeof body?.helper === 'string' && body.helper.trim().length > 0 ? body.helper.trim() : null; - const sessionId = typeof body?.sessionId === 'string' && body.sessionId.trim().length > 0 ? body.sessionId.trim() : null; - const messageId = typeof body?.messageId === 'string' && body.messageId.trim().length > 0 ? body.messageId.trim() : null; - const providedRequestId = typeof body?.requestId === 'string' && body.requestId.trim().length > 0 ? body.requestId.trim() : null; - const voiceRequestId = providedRequestId || randomUUID(); - - if (!text) { - return NextResponse.json({ error: 'Text is required for TTS' }, { status: 400 }); - } - - const requestStartedAt = Date.now(); - const response = await fetch('https://api.openai.com/v1/audio/speech', { - method: 'POST', - headers: { - 'Content-Type': 'application/json', - Authorization: `Bearer ${apiKey}`, - }, - body: JSON.stringify({ - model: OPENAI_TTS_MODEL, - voice, - input: text, - format: 'mp3', - }), - }); - - if (!response.ok || !response.body) { - const errorPayload = await response.json().catch(() => null); - const message = errorPayload?.error?.message || response.statusText || 'Failed to synthesize audio'; - return NextResponse.json({ error: message }, { status: response.status || 500 }); - } - - const durationMs = Date.now() - requestStartedAt; - const charCount = [...text].length; - const estimatedCostUsd = estimateTtsCost(charCount); - const textPreview = - text.length > 240 ? `${text.slice(0, 237)}...` : text; - - try { - recordVoiceUsage({ - sessionId, - helperName, - requestId: voiceRequestId, - messageId, - voice, - model: OPENAI_TTS_MODEL, - charCount, - costUsd: estimatedCostUsd, - durationMs, - textPreview, - }); - } catch (loggingError) { - console.error('[voice/tts] failed to record usage', loggingError); - } - - const headers = new Headers(); - headers.set('Content-Type', response.headers.get('Content-Type') || 'audio/mpeg'); - headers.set('Cache-Control', 'no-cache'); - headers.set('X-Voice-Request-Id', voiceRequestId); - - return new Response(response.body, { - status: 200, - headers, - }); - } catch (error) { - const message = error instanceof Error ? error.message : String(error); - console.error('[voice/tts] failed to synthesize:', message); - return NextResponse.json({ error: message }, { status: 500 }); - } -} diff --git a/ralph/prd.json b/ralph/prd.json index 69d6690..f6d901e 100644 --- a/ralph/prd.json +++ b/ralph/prd.json @@ -13,8 +13,8 @@ "No remaining imports of 'delegation' from services/agents", "npm run type-check passes" ], - "passes": false, - "notes": "" + "passes": true, + "notes": "Replaced AgentDelegationService with direct execution in workflowExecutor and quickAdd. Added stub AgentDelegation types and voice hook stubs to UI components for compatibility (to be removed in later stories)." }, { "id": "2", @@ -31,8 +31,8 @@ "RAHChat.tsx no longer imports or uses voice hooks", "npm run type-check passes" ], - "passes": false, - "notes": "" + "passes": true, + "notes": "Deleted all 4 voice hooks, voice API routes (/api/voice/, /api/realtime/), voice service directory. Removed voice stub hooks and voice-related code from RAHChat.tsx and TerminalInput.tsx." }, { "id": "3", diff --git a/ralph/progress.txt b/ralph/progress.txt index 1affe33..11c7411 100644 --- a/ralph/progress.txt +++ b/ralph/progress.txt @@ -8,4 +8,49 @@ Goal: Strip ra-h_os to lightweight 2-panel UI + MCP server - Target: remove chat agents, voice, delegations, simplify to 2-panel - Keep: nodes/edges/dimensions, MCP server, workflows, extraction +## Story 1: Remove agent delegation system +- Completed: Thu 29 Jan 2026 +- Files deleted: + - src/services/agents/delegation.ts + - app/api/rah/delegations/route.ts + - app/api/rah/delegations/stream/route.ts + - app/api/rah/delegations/[sessionId]/route.ts + - app/api/rah/delegations/[sessionId]/summary/route.ts + - src/components/agents/DelegationIndicator.tsx +- Files modified: + - src/services/agents/workflowExecutor.ts (removed delegation streaming) + - src/services/agents/quickAdd.ts (direct execution instead of delegation) + - src/tools/orchestration/executeWorkflow.ts (direct execution) + - src/tools/orchestration/delegateToWiseRAH.ts (direct execution) + - app/api/workflows/execute/route.ts (direct execution) + - src/components/layout/ThreePanelLayout.tsx (stub type, removed API calls) + - src/components/agents/RAHChat.tsx (stub type, stub voice hooks) + - src/components/agents/AgentsPanel.tsx (stub type) + - src/components/agents/MiniRAHPanel.tsx (stub type) + - src/components/agents/WiseRAHPanel.tsx (stub type) + - src/components/agents/QuickAddStatus.tsx (stub type) + - src/components/panes/types.ts (stub type) + - src/components/panes/WorkflowsPane.tsx (import from types.ts) +- Learnings: + - Delegation system was deeply integrated with workflows; had to refactor to direct execution + - UI components that will be deleted later needed stub types to compile + - Voice hooks also needed stubs in RAHChat.tsx (will be cleaned up in Story 2) + +## Story 2: Remove voice features +- Completed: Thu 29 Jan 2026 +- Files deleted: + - src/components/agents/hooks/useVoiceSession.ts + - src/components/agents/hooks/useRealtimeVoiceClient.ts + - src/components/agents/hooks/useAssistantTTS.ts + - src/components/agents/hooks/useVoiceInterruption.ts + - app/api/voice/tts/route.ts + - app/api/realtime/ephemeral-token/route.ts + - src/services/voice/usageLogger.ts +- Files modified: + - src/components/agents/RAHChat.tsx (removed all voice-related code and stub hooks) + - src/components/agents/TerminalInput.tsx (removed voice props and voice UI) +- Learnings: + - Voice hooks were stubbed in Story 1, so removal was straightforward + - TerminalInput had voice UI (amplitude bars, mic icons) that needed cleanup + - No imports to these files from other parts of the codebase diff --git a/src/components/agents/RAHChat.tsx b/src/components/agents/RAHChat.tsx index f7c173e..c6e7456 100644 --- a/src/components/agents/RAHChat.tsx +++ b/src/components/agents/RAHChat.tsx @@ -8,7 +8,6 @@ import TerminalInput from './TerminalInput'; import { Zap, Flame } from 'lucide-react'; import { useSSEChat, ChatMessage, MessageRole } from './hooks/useSSEChat'; import { useQuotaHandler } from '@/hooks/useQuotaHandler'; -import { apiKeyService } from '@/services/storage/apiKeys'; // Stub type for delegation (delegation system removed in rah-light) type AgentDelegation = { @@ -27,45 +26,7 @@ function DelegationIndicator({ delegations }: { delegations: AgentDelegation[] } return null; } -// Stub voice hooks (voice system will be removed in story 2) -function useVoiceSession() { - return { - isActive: false, - amplitude: 0, - startSession: () => {}, - stopSession: () => {}, - resetTranscript: () => {}, - setStatus: (_status: string) => {}, - setAmplitude: (_amp: number) => {}, - setInterimTranscript: (_text: string) => {}, - appendFinalTranscript: (_text: string) => {}, - }; -} - -function useAssistantTTS(_options: { onSpeechStart?: () => void; onSpeechComplete?: () => void; onError?: (e: Error) => void }) { - return { - speak: (_text: string, _options?: { flush?: boolean; metadata?: Record }) => {}, - stop: () => {}, - status: 'idle' as const, - }; -} - -function useVoiceInterruption(_options: { amplitude: number; isVoiceActive: boolean; ttsStatus: string; onInterruption: () => void }) {} - -function useRealtimeVoiceClient(_handlers: { onStatusChange?: (status: string) => void; onInterimTranscript?: (text: string) => void; onFinalTranscript?: (text: string) => void; onAmplitude?: (amp: number) => void; onError?: (e: Error) => void }, _options: { getAuthToken: () => string | null }) { - return { - connect: () => {}, - disconnect: () => {}, - start: () => {}, - stop: () => {}, - }; -} - const createSessionId = () => `session_${Date.now()}_${Math.random().toString(36).slice(2, 8)}`; -const createVoiceRequestId = () => - typeof crypto !== 'undefined' && typeof crypto.randomUUID === 'function' - ? crypto.randomUUID() - : `voice_${Date.now()}_${Math.random().toString(36).slice(2, 8)}`; interface HighlightedPassage { selectedText: string; @@ -110,7 +71,7 @@ export default function RAHChat({ const [internalMessages, internalSetMessages] = useState([]); const messages = externalMessages !== undefined ? externalMessages : internalMessages; const setMessages = externalSetMessages || internalSetMessages; - + const [sessionId, setSessionId] = useState(() => createSessionId()); const messagesEndRef = useRef(null); const chatMode = mode === 'hard' ? 'hard' : 'easy'; @@ -127,53 +88,6 @@ export default function RAHChat({ await refetchUsage(); }); const setMessagesRef = useRef(setMessages); - const voice = useVoiceSession(); - const { - isActive: isVoiceActive, - amplitude: voiceAmplitude, - startSession: startVoice, - stopSession: stopVoice, - resetTranscript: resetVoiceTranscript, - setStatus: setVoiceStatus, - setAmplitude: setVoiceAmplitude, - setInterimTranscript, - appendFinalTranscript, - } = voice; - const pendingVoiceQueueRef = useRef<{ text: string; queuedAt: number }[]>([]); - const assistantSpeechMapRef = useRef>(new Map()); - const [voiceError, setVoiceError] = useState(null); - const voiceErrorHandledRef = useRef(false); - const voiceStartTimestampRef = useRef(null); - - const handleVoiceError = useCallback((error: Error) => { - console.error('[RAHChat] Voice error:', error); - setVoiceError(error.message); - if (isVoiceActive) { - stopVoice(); - } - resetVoiceTranscript(); - setVoiceAmplitude(0); - setVoiceStatus('idle'); - pendingVoiceQueueRef.current = []; - assistantSpeechMapRef.current.clear(); - }, [isVoiceActive, resetVoiceTranscript, setVoiceAmplitude, setVoiceStatus, stopVoice]); - - const { speak: speakAssistantResponse, stop: stopAssistantTTS, status: ttsStatus } = useAssistantTTS({ - onSpeechStart: () => { - if (isVoiceActive) { - setVoiceStatus('speaking'); - } - }, - onSpeechComplete: () => { - setVoiceStatus(isVoiceActive ? 'listening' : 'idle'); - }, - onError: handleVoiceError, - }); - - const handleVoiceInterruption = useCallback(() => { - stopAssistantTTS(); - setVoiceStatus('listening'); - }, [setVoiceStatus, stopAssistantTTS]); const sse = useSSEChat('/api/rah/chat', setMessages, { getAuthToken: () => null, @@ -187,14 +101,6 @@ export default function RAHChat({ }, }); - useVoiceInterruption({ - amplitude: voiceAmplitude, - isVoiceActive, - ttsStatus, - onInterruption: handleVoiceInterruption, - }); - - const sendMessage = useCallback(async (text: string) => { if (delegationMode) return; // Delegation chats are read-only if (isQuotaExceeded) { @@ -211,71 +117,9 @@ export default function RAHChat({ }); }, [activeTabId, chatMode, checkQuotaBeforeRequest, delegationMode, isQuotaExceeded, messages, openTabsData, sse, sessionId]); - const handleVoiceFinalTranscript = useCallback( - (raw: string) => { - const normalized = raw.trim(); - console.info('[RAHVoice] Final transcript received:', normalized || '(empty)'); - setInterimTranscript(''); - if (!normalized) { - console.info('[RAHVoice] Ignoring empty transcript'); - return; - } - appendFinalTranscript(normalized); - if (sse.isLoading) { - pendingVoiceQueueRef.current.push({ text: normalized, queuedAt: Date.now() }); - console.info('[RAHVoice] SSE busy, queueing transcript', { - queuedCount: pendingVoiceQueueRef.current.length, - }); - setVoiceStatus('thinking'); - return; - } - setVoiceStatus('thinking'); - console.info('[RAHVoice] Dispatching transcript to /api/rah/chat'); - void sendMessage(normalized); - }, - [appendFinalTranscript, sendMessage, setInterimTranscript, setVoiceStatus, sse.isLoading] - ); - - const voiceRealtime = useRealtimeVoiceClient( - { - onStatusChange: (status) => { - if (!isVoiceActive && status !== 'idle') return; - if (status === 'listening' && (sse.isLoading || pendingVoiceQueueRef.current.length > 0)) { - setVoiceStatus('thinking'); - return; - } - setVoiceStatus(status); - }, - onInterimTranscript: setInterimTranscript, - onFinalTranscript: handleVoiceFinalTranscript, - onAmplitude: setVoiceAmplitude, - onError: handleVoiceError, - }, - { - getAuthToken: () => null, - } - ); - const handleStreamComplete = useCallback(async () => { - if (pendingVoiceQueueRef.current.length > 0) { - console.info('[RAHVoice] SSE stream complete, draining queued transcripts', { - queued: pendingVoiceQueueRef.current.length, - }); - } - while (pendingVoiceQueueRef.current.length > 0) { - const nextQueued = pendingVoiceQueueRef.current.shift(); - if (!nextQueued) { - break; - } - const queueLatency = Date.now() - nextQueued.queuedAt; - setVoiceStatus('thinking'); - console.info('[RAHVoice] Dispatching queued transcript to /api/rah/chat', { - queuedMs: queueLatency, - }); - await sendMessage(nextQueued.text); - } await refetchUsage(); - }, [sendMessage, setVoiceStatus, refetchUsage]); + }, [refetchUsage]); useEffect(() => { streamCompleteHandlerRef.current = handleStreamComplete; @@ -285,17 +129,6 @@ export default function RAHChat({ setMessagesRef.current = setMessages; }, [setMessages]); - useEffect(() => { - if (!voiceError) { - voiceErrorHandledRef.current = false; - return; - } - if (voiceErrorHandledRef.current) return; - voiceErrorHandledRef.current = true; - voiceRealtime.stop(); - stopAssistantTTS(); - }, [voiceError, voiceRealtime, stopAssistantTTS]); - const focusSummary = useMemo(() => { if (!openTabsData.length) return null; const titles = openTabsData.map((node) => node?.title || 'Untitled'); @@ -315,41 +148,11 @@ export default function RAHChat({ messagesEndRef.current?.scrollIntoView({ behavior: 'smooth' }); }, [messages]); - useEffect(() => { - if (!isVoiceActive) { - assistantSpeechMapRef.current.clear(); - stopAssistantTTS(); - return; - } - if (sse.isLoading) return; - const assistantMessages = messages.filter((m) => m.role === MessageRole.ASSISTANT); - if (!assistantMessages.length) return; - const latest = assistantMessages[assistantMessages.length - 1]; - const spokenContent = assistantSpeechMapRef.current.get(latest.id); - if (!latest.content.trim() || spokenContent === latest.content) return; - assistantSpeechMapRef.current.set(latest.id, latest.content); - const voiceRequestId = createVoiceRequestId(); - speakAssistantResponse(latest.content, { - flush: true, - metadata: { - sessionId, - helper: helperKey, - requestId: voiceRequestId, - messageId: latest.id, - }, - }); - }, [helperKey, isVoiceActive, messages, sessionId, sse.isLoading, speakAssistantResponse, stopAssistantTTS]); - - const handleNewChat = () => { if (delegationMode) return; sse.abort(); setMessages((_prev) => []); setSessionId(createSessionId()); - if (isVoiceActive) { - stopVoice(); - resetVoiceTranscript(); - } }; // Subscribe to delegation stream if in delegation mode @@ -361,7 +164,7 @@ export default function RAHChat({ eventSource.onmessage = (event) => { try { const data = JSON.parse(event.data); - + if (data.type === 'text-delta' && data.delta) { setMessagesRef.current((prev) => { const lastMsg = prev[prev.length - 1]; @@ -429,52 +232,6 @@ export default function RAHChat({ }; }, [delegationMode, delegationSessionId]); - useEffect(() => { - if (delegationMode && isVoiceActive) { - voiceRealtime.stop(); - stopVoice(); - resetVoiceTranscript(); - stopAssistantTTS(); - } - }, [delegationMode, isVoiceActive, resetVoiceTranscript, stopAssistantTTS, stopVoice, voiceRealtime]); - - const handleVoiceToggle = useCallback(async () => { - if (isVoiceActive) { - voiceRealtime.stop(); - stopVoice(); - resetVoiceTranscript(); - setVoiceAmplitude(0); - setVoiceStatus('idle'); - assistantSpeechMapRef.current.clear(); - pendingVoiceQueueRef.current = []; - stopAssistantTTS(); - voiceStartTimestampRef.current = null; - return; - } - setVoiceError(null); - try { - voiceStartTimestampRef.current = performance.now(); - console.info('[RAHVoice] Voice session starting'); - await voiceRealtime.start(); - startVoice(); - setVoiceStatus('listening'); - } catch (error) { - voiceStartTimestampRef.current = null; - handleVoiceError(error instanceof Error ? error : new Error(String(error))); - } - }, [ - handleVoiceError, - isVoiceActive, - resetVoiceTranscript, - setVoiceAmplitude, - setVoiceStatus, - setVoiceError, - startVoice, - stopAssistantTTS, - stopVoice, - voiceRealtime, - ]); - return (
)} - {/* Voice transcript preview removed for streamlined UI */}
@@ -569,14 +325,8 @@ export default function RAHChat({
void; @@ -10,10 +9,6 @@ interface TerminalInputProps { helperId?: number; disabledExternally?: boolean; disabledMessage?: string; - onVoiceToggle?: () => void; - isVoiceActive?: boolean; - voiceAmplitude?: number; - voiceError?: string | null; } export default function TerminalInput({ @@ -23,10 +18,6 @@ export default function TerminalInput({ helperId, disabledExternally = false, disabledMessage, - onVoiceToggle, - isVoiceActive = false, - voiceAmplitude = 0, - voiceError, }: TerminalInputProps) { const [input, setInput] = useState(''); const [rows, setRows] = useState(1); @@ -54,26 +45,26 @@ export default function TerminalInput({ useEffect(() => { const textarea = textareaRef.current; if (!textarea) return; - + const lineHeight = 24; // line-height * font-size approximately const minHeight = 32; const maxHeight = 120; - + // Check if content overflows current height (need to grow) const needsGrow = textarea.scrollHeight > textarea.clientHeight; - + // Check if we cleared content significantly (need to shrink) const lineCount = (input.match(/\n/g) || []).length + 1; const estimatedHeight = Math.max(lineCount * lineHeight, minHeight); const needsShrink = !input.trim() || (textarea.clientHeight > estimatedHeight + lineHeight); - + if (needsGrow || needsShrink) { // Only recalculate when necessary textarea.style.height = 'auto'; const scrollHeight = textarea.scrollHeight; const newHeight = Math.min(Math.max(scrollHeight, minHeight), maxHeight); textarea.style.height = `${newHeight}px`; - + const newRows = Math.min(Math.max(1, Math.floor(newHeight / lineHeight)), 5); setRows(newRows); } @@ -135,22 +126,7 @@ export default function TerminalInput({ }, [input]); const trimmedInput = input.trim(); - const showVoiceStart = !isVoiceActive && !trimmedInput && Boolean(onVoiceToggle); - const showVoiceStop = Boolean(onVoiceToggle) && isVoiceActive; - const buttonIsDisabled = - showVoiceStart || showVoiceStop - ? false - : (!trimmedInput || isProcessing || disabledExternally); - - const handlePrimaryAction = () => { - if (showVoiceStart || showVoiceStop) { - onVoiceToggle?.(); - return; - } - handleSubmit(); - }; - - const amplitudeBars = Array.from({ length: 8 }); + const buttonIsDisabled = !trimmedInput || isProcessing || disabledExternally; // Handle node drag over chat input const handleDragOver = (e: DragEvent) => { @@ -253,7 +229,6 @@ export default function TerminalInput({ gap: '8px', padding: '8px 16px 12px', background: 'transparent', - // Remove separator/border between chat area and input borderTop: 'none', fontFamily: 'inherit' }}> @@ -276,49 +251,6 @@ export default function TerminalInput({ flexDirection: 'column', gap: '4px' }}> - {isVoiceActive && ( -
-
- - RA-H is listening - -
-
- {amplitudeBars.map((_, index) => { - const level = (index + 1) / amplitudeBars.length; - const active = voiceAmplitude >= level - 0.0001; - return ( -
- ); - })} -
- {voiceError && ( - {voiceError} - )} -
- )} {/* Input Row with Textarea and Button */}
{}} onBlur={() => {}} /> @@ -382,30 +313,18 @@ export default function TerminalInput({ ))}
)} - - {/* Submit Button (minimal icon) */} + + {/* Submit Button */}
diff --git a/src/components/agents/hooks/useAssistantTTS.ts b/src/components/agents/hooks/useAssistantTTS.ts deleted file mode 100644 index f2fd95e..0000000 --- a/src/components/agents/hooks/useAssistantTTS.ts +++ /dev/null @@ -1,300 +0,0 @@ -import { useCallback, useEffect, useRef, useState } from 'react'; - -export type TTSStatus = 'idle' | 'loading' | 'speaking'; - -interface UseAssistantTTSOptions { - voice?: string; - onSpeechStart?: () => void; - onSpeechComplete?: () => void; - onError?: (error: Error) => void; -} - -interface SpeakRequestMetadata { - sessionId?: string | null; - helper?: string | null; - requestId?: string; - messageId?: string | null; -} - -interface SpeakOptions { - flush?: boolean; - metadata?: SpeakRequestMetadata; -} - -type SpeakQueueItem = { - text: string; - metadata?: SpeakRequestMetadata; -}; - -export function useAssistantTTS(options: UseAssistantTTSOptions = {}) { - const [status, setStatus] = useState('idle'); - const queueRef = useRef([]); - const isProcessingRef = useRef(false); - const abortRef = useRef(null); - const audioRef = useRef(null); - const objectUrlRef = useRef(null); - const mediaSourceRef = useRef(null); - const sourceBufferRef = useRef(null); - const chunkQueueRef = useRef([]); - const updateEndHandlerRef = useRef<(() => void) | null>(null); - const readerDoneRef = useRef(false); - const processQueueRef = useRef<(() => void) | null>(null); - - const onSpeechStartRef = useRef(options.onSpeechStart); - const onSpeechCompleteRef = useRef(options.onSpeechComplete); - const onErrorRef = useRef(options.onError); - const voiceRef = useRef(options.voice); - - useEffect(() => { - onSpeechStartRef.current = options.onSpeechStart; - }, [options.onSpeechStart]); - - useEffect(() => { - onSpeechCompleteRef.current = options.onSpeechComplete; - }, [options.onSpeechComplete]); - - useEffect(() => { - onErrorRef.current = options.onError; - }, [options.onError]); - - useEffect(() => { - voiceRef.current = options.voice; - }, [options.voice]); - - const setStatusSafe = useCallback((next: TTSStatus) => { - setStatus(next); - }, []); - - const cleanupMedia = useCallback(() => { - if (sourceBufferRef.current && updateEndHandlerRef.current) { - try { - sourceBufferRef.current.removeEventListener('updateend', updateEndHandlerRef.current); - } catch (err) { - console.warn('[TTS] Failed to detach source buffer listener', err); - } - } - updateEndHandlerRef.current = null; - sourceBufferRef.current = null; - mediaSourceRef.current = null; - if (objectUrlRef.current) { - URL.revokeObjectURL(objectUrlRef.current); - objectUrlRef.current = null; - } - chunkQueueRef.current = []; - readerDoneRef.current = false; - }, []); - - const handleError = useCallback((error: Error) => { - console.error('[TTS] Playback error', error); - onErrorRef.current?.(error); - }, []); - - const stopCurrentPlayback = useCallback(() => { - abortRef.current?.abort(); - abortRef.current = null; - const audio = audioRef.current; - if (audio) { - audio.pause(); - audio.removeAttribute('src'); - try { - audio.load(); - } catch (err) { - console.warn('[TTS] Failed to reset audio element', err); - } - } - cleanupMedia(); - isProcessingRef.current = false; - }, [cleanupMedia]); - - const ensureAudioElement = useCallback(() => { - if (audioRef.current) { - return audioRef.current; - } - const audio = new Audio(); - audio.autoplay = true; - audio.preload = 'auto'; - audio.addEventListener('playing', () => { - setStatusSafe('speaking'); - onSpeechStartRef.current?.(); - }); - audio.addEventListener('ended', () => { - cleanupMedia(); - isProcessingRef.current = false; - setStatusSafe('idle'); - onSpeechCompleteRef.current?.(); - processQueueRef.current?.(); - }); - audio.addEventListener('error', () => { - handleError(new Error('Audio playback failed')); - stopCurrentPlayback(); - processQueueRef.current?.(); - }); - audioRef.current = audio; - return audio; - }, [cleanupMedia, handleError, setStatusSafe, stopCurrentPlayback]); - - const flushChunks = useCallback(() => { - const sourceBuffer = sourceBufferRef.current; - const mediaSource = mediaSourceRef.current; - if (!sourceBuffer || !mediaSource || sourceBuffer.updating) { - return; - } - const nextChunk = chunkQueueRef.current.shift(); - if (nextChunk) { - try { - sourceBuffer.appendBuffer(nextChunk); - } catch (error) { - handleError(error instanceof Error ? error : new Error(String(error))); - } - return; - } - if (readerDoneRef.current && !sourceBuffer.updating) { - try { - mediaSource.endOfStream(); - } catch (error) { - console.warn('[TTS] Failed to end media source stream', error); - } - } - }, [handleError]); - - const processQueue = useCallback(async () => { - if (isProcessingRef.current) return; - const next = queueRef.current.shift(); - if (!next) { - setStatusSafe('idle'); - return; - } - isProcessingRef.current = true; - setStatusSafe('loading'); - try { - const controller = new AbortController(); - abortRef.current = controller; - const payload: Record = { - text: next.text, - voice: voiceRef.current, - }; - - if (next.metadata?.sessionId) { - payload.sessionId = next.metadata.sessionId; - } - if (next.metadata?.helper) { - payload.helper = next.metadata.helper; - } - if (next.metadata?.requestId) { - payload.requestId = next.metadata.requestId; - } - if (next.metadata?.messageId) { - payload.messageId = next.metadata.messageId; - } - - const response = await fetch('/api/voice/tts', { - method: 'POST', - headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify(payload), - signal: controller.signal, - }); - if (!response.ok || !response.body) { - throw new Error((await response.text().catch(() => '')) || 'Failed to synthesize audio'); - } - const reader = response.body.getReader(); - const audio = ensureAudioElement(); - cleanupMedia(); - const mimeType = response.headers.get('Content-Type') || 'audio/mpeg'; - const mediaSource = new MediaSource(); - mediaSourceRef.current = mediaSource; - const objectUrl = URL.createObjectURL(mediaSource); - objectUrlRef.current = objectUrl; - audio.src = objectUrl; - audio.load(); - - const onSourceOpen = () => { - mediaSource.removeEventListener('sourceopen', onSourceOpen); - let sourceBuffer: SourceBuffer; - try { - sourceBuffer = mediaSource.addSourceBuffer(mimeType); - } catch { - throw new Error(`Unsupported audio format: ${mimeType}`); - } - sourceBufferRef.current = sourceBuffer; - const handleUpdateEnd = () => flushChunks(); - updateEndHandlerRef.current = handleUpdateEnd; - sourceBuffer.addEventListener('updateend', handleUpdateEnd); - flushChunks(); - }; - - mediaSource.addEventListener('sourceopen', onSourceOpen); - - const pump = async () => { - while (true) { - const { value, done } = await reader.read(); - if (done) { - readerDoneRef.current = true; - flushChunks(); - break; - } - if (value) { - const buffer = value.buffer.slice(value.byteOffset, value.byteOffset + value.byteLength); - chunkQueueRef.current.push(buffer); - flushChunks(); - } - } - }; - - pump().catch((error) => { - const err = error instanceof Error ? error : new Error(String(error)); - if ((err as DOMException).name !== 'AbortError') { - handleError(err); - } - stopCurrentPlayback(); - processQueueRef.current?.(); - }); - - audio.play().catch((error) => { - handleError(error instanceof Error ? error : new Error(String(error))); - }); - } catch (error) { - if (error instanceof DOMException && error.name === 'AbortError') { - return; - } - isProcessingRef.current = false; - setStatusSafe('idle'); - handleError(error instanceof Error ? error : new Error(String(error))); - processQueueRef.current?.(); - } - }, [cleanupMedia, ensureAudioElement, flushChunks, handleError, setStatusSafe, stopCurrentPlayback]); - - processQueueRef.current = processQueue; - - const speak = useCallback( - (text: string, options?: SpeakOptions) => { - const trimmed = text.trim(); - if (!trimmed) return; - if (options?.flush) { - queueRef.current = [{ text: trimmed, metadata: options.metadata }]; - stopCurrentPlayback(); - } else { - queueRef.current.push({ text: trimmed, metadata: options?.metadata }); - } - processQueue(); - }, - [processQueue, stopCurrentPlayback] - ); - - const stop = useCallback(() => { - queueRef.current = []; - stopCurrentPlayback(); - setStatusSafe('idle'); - }, [setStatusSafe, stopCurrentPlayback]); - - useEffect(() => { - return () => { - stop(); - }; - }, [stop]); - - return { - status, - speak, - stop, - } as const; -} diff --git a/src/components/agents/hooks/useRealtimeVoiceClient.ts b/src/components/agents/hooks/useRealtimeVoiceClient.ts deleted file mode 100644 index 351055e..0000000 --- a/src/components/agents/hooks/useRealtimeVoiceClient.ts +++ /dev/null @@ -1,570 +0,0 @@ -"use client"; - -import { useCallback, useEffect, useRef } from 'react'; - -export type RealtimeConnectionState = 'idle' | 'connecting' | 'ready' | 'capturing'; - -interface VoiceRealtimeCallbacks { - onStatusChange?: (status: 'idle' | 'listening' | 'thinking' | 'speaking') => void; - onInterimTranscript?: (text: string) => void; - onFinalTranscript?: (text: string) => void; - onAmplitude?: (value: number) => void; - onError?: (error: Error) => void; -} - -interface UseRealtimeVoiceClientOptions { - getAuthToken?: () => string | null | undefined; - fetchEphemeralToken?: ( - authToken: string | null - ) => Promise<{ client_secret: { value: string }; model: string; voice: string }>; - silenceThresholdMs?: number; - silenceAmplitudeCutoff?: number; -} - -const DEFAULT_SILENCE_THRESHOLD_MS = 800; -const DEFAULT_SILENCE_AMPLITUDE = 0.0015; - -type Nullable = T | null; - -function calculateRms(buffer: Float32Array) { - if (!buffer.length) return 0; - let sumSquares = 0; - for (let i = 0; i < buffer.length; i += 1) { - const value = buffer[i]; - sumSquares += value * value; - } - return Math.sqrt(sumSquares / buffer.length); -} - - -export function useRealtimeVoiceClient( - callbacks: VoiceRealtimeCallbacks, - options: UseRealtimeVoiceClientOptions = {} -) { - const { onStatusChange, onInterimTranscript, onFinalTranscript, onAmplitude, onError } = callbacks; - - const { getAuthToken, fetchEphemeralToken, silenceThresholdMs, silenceAmplitudeCutoff } = options; - - const connectionStateRef = useRef('idle'); - const peerConnectionRef = useRef>(null); - const dataChannelRef = useRef>(null); - const audioContextRef = useRef>(null); - const processorNodeRef = useRef>(null); - const mediaStreamRef = useRef>(null); - const mediaSourceRef = useRef>(null); - const awaitingTranscriptRef = useRef(false); - const hasUncommittedAudioRef = useRef(false); - const lastSpeechAtRef = useRef(null); - const destroyedRef = useRef(false); - const channelReadyRef = useRef(false); - - const silenceWindowMs = silenceThresholdMs ?? DEFAULT_SILENCE_THRESHOLD_MS; - const amplitudeGate = silenceAmplitudeCutoff ?? DEFAULT_SILENCE_AMPLITUDE; - - const onStatusChangeRef = useRef(onStatusChange); - const onInterimTranscriptRef = useRef(onInterimTranscript); - const onFinalTranscriptRef = useRef(onFinalTranscript); - const onAmplitudeRef = useRef(onAmplitude); - const onErrorRef = useRef(onError); - - useEffect(() => { - onStatusChangeRef.current = onStatusChange; - }, [onStatusChange]); - - useEffect(() => { - onInterimTranscriptRef.current = onInterimTranscript; - }, [onInterimTranscript]); - - useEffect(() => { - onFinalTranscriptRef.current = onFinalTranscript; - }, [onFinalTranscript]); - - useEffect(() => { - onAmplitudeRef.current = onAmplitude; - }, [onAmplitude]); - - useEffect(() => { - onErrorRef.current = onError; - }, [onError]); - - const setConnectionState = useCallback((next: RealtimeConnectionState) => { - connectionStateRef.current = next; - }, []); - - const teardownInputNodes = useCallback(() => { - processorNodeRef.current?.disconnect(); - mediaSourceRef.current?.disconnect(); - processorNodeRef.current?.removeEventListener('audioprocess', () => undefined); - processorNodeRef.current = null; - mediaSourceRef.current = null; - if (mediaStreamRef.current) { - mediaStreamRef.current.getTracks().forEach((track) => track.stop()); - mediaStreamRef.current = null; - } - }, []); - - const closePeerConnection = useCallback(() => { - if (dataChannelRef.current) { - try { - dataChannelRef.current.close(); - } catch (err) { - console.warn('[VoiceRealtime] Failed to close data channel:', err); - } - } - dataChannelRef.current = null; - - if (peerConnectionRef.current) { - try { - peerConnectionRef.current.ontrack = null; - peerConnectionRef.current.onconnectionstatechange = null; - peerConnectionRef.current.close(); - } catch (err) { - console.warn('[VoiceRealtime] Failed to close peer connection:', err); - } - } - peerConnectionRef.current = null; - - channelReadyPromiseRef.current = null; - channelReadyResolveRef.current = null; - }, []); - - const resetState = useCallback(() => { - awaitingTranscriptRef.current = false; - hasUncommittedAudioRef.current = false; - lastSpeechAtRef.current = null; - channelReadyPromiseRef.current = null; - channelReadyResolveRef.current = null; - }, []); - - const notifyError = useCallback((message: string | Error) => { - const error = message instanceof Error ? message : new Error(message); - onErrorRef.current?.(error); - }, []); - - const channelReadyPromiseRef = useRef | null>(null); - const channelReadyResolveRef = useRef<(() => void) | null>(null); - - const ensureChannelReady = useCallback(async () => { - const channel = dataChannelRef.current; - if (channel?.readyState === 'open') return; - if (!channelReadyPromiseRef.current) { - channelReadyPromiseRef.current = new Promise((resolve) => { - channelReadyResolveRef.current = resolve; - }); - } - await channelReadyPromiseRef.current; - }, []); - - const sendEvent = useCallback( - async (event: Record) => { - await ensureChannelReady(); - const channel = dataChannelRef.current; - if (!channel || channel.readyState !== 'open') { - throw new Error('Realtime data channel is not open'); - } - channel.send(JSON.stringify(event)); - }, - [ensureChannelReady] - ); - - const initialiseMicrophone = useCallback(async () => { - if (typeof window === 'undefined') { - throw new Error('Voice not supported in this environment'); - } - - const stream = await navigator.mediaDevices.getUserMedia({ - audio: { - channelCount: 1, - echoCancellation: false, - autoGainControl: false, - noiseSuppression: false, - }, - }); - - if (destroyedRef.current) { - stream.getTracks().forEach((track) => { - try { - track.stop(); - } catch (err) { - console.warn('[VoiceRealtime] Failed to stop track after destroy', err); - } - }); - return stream; - } - - const audioContext = new AudioContext(); - const source = audioContext.createMediaStreamSource(stream); - const processor = audioContext.createScriptProcessor(2048, 1, 1); - - processor.onaudioprocess = (event) => { - if (destroyedRef.current) return; - const inputBuffer = event.inputBuffer.getChannelData(0); - const amplitude = calculateRms(inputBuffer); - onAmplitudeRef.current?.(Math.min(1, amplitude * 8)); - - const now = Date.now(); - const channelReady = channelReadyRef.current; - if (amplitude > amplitudeGate && channelReady) { - hasUncommittedAudioRef.current = true; - lastSpeechAtRef.current = now; - if (!awaitingTranscriptRef.current) { - onStatusChangeRef.current?.('listening'); - } - } else if ( - channelReady && - hasUncommittedAudioRef.current && - !awaitingTranscriptRef.current && - lastSpeechAtRef.current && - now - lastSpeechAtRef.current > silenceWindowMs - ) { - awaitingTranscriptRef.current = true; - hasUncommittedAudioRef.current = false; - lastSpeechAtRef.current = null; - onStatusChangeRef.current?.('thinking'); - } - }; - - source.connect(processor); - processor.connect(audioContext.destination); - - audioContextRef.current = audioContext; - processorNodeRef.current = processor; - mediaSourceRef.current = source; - mediaStreamRef.current = stream; - return stream; - }, [amplitudeGate, silenceWindowMs]); - - const disconnect = useCallback(() => { - destroyedRef.current = true; - teardownInputNodes(); - closePeerConnection(); - audioContextRef.current?.close().catch(() => undefined); - audioContextRef.current = null; - resetState(); - channelReadyRef.current = false; - setConnectionState('idle'); - onStatusChangeRef.current?.('idle'); - onAmplitudeRef.current?.(0); - }, [closePeerConnection, resetState, setConnectionState, teardownInputNodes]); - - const extractTextDelta = useCallback((payload: unknown): string => { - if (!payload) return ''; - if (typeof payload === 'string') return payload; - if (typeof payload !== 'object') return ''; - - const record = payload as Record; - if (typeof record.delta === 'string') return record.delta; - if (typeof record.text === 'string') return record.text; - - const outputText = record.output_text as Record | undefined; - if (typeof outputText?.text === 'string') return outputText.text; - - if (Array.isArray(record.output)) { - return record.output - .flatMap((item) => - Array.isArray((item as Record)?.content) - ? ((item as Record).content as unknown[]) - : [] - ) - .map((content) => { - if (!content || typeof content !== 'object') return ''; - const entry = content as Record; - const nestedText = entry.text as Record | string | undefined; - if (typeof nestedText === 'string') return nestedText; - if (typeof nestedText === 'object' && nestedText !== null && typeof nestedText.value === 'string') { - return nestedText.value; - } - return ''; - }) - .filter(Boolean) - .join(''); - } - - if (Array.isArray(record.content)) { - return record.content - .map((content) => { - if (!content || typeof content !== 'object') return ''; - const entry = content as Record; - const nestedText = entry.text as Record | string | undefined; - if (typeof nestedText === 'string') return nestedText; - if (typeof nestedText === 'object' && nestedText !== null && typeof nestedText.value === 'string') { - return nestedText.value; - } - return ''; - }) - .filter(Boolean) - .join(''); - } - - return ''; - }, []); - - const handleDataMessage = useCallback( - (raw: string) => { - try { - const data = JSON.parse(raw); - if (process.env.NODE_ENV !== 'production') { - console.debug('[VoiceRealtime] Event', data.type, data); - } - - if (data.type === 'conversation.item.input_audio_transcription.completed') { - const transcriptSource = - typeof data.transcript === 'string' - ? data.transcript - : extractTextDelta(data.item ?? data.content ?? data); - const transcript = transcriptSource?.trim(); - awaitingTranscriptRef.current = false; - hasUncommittedAudioRef.current = false; - lastSpeechAtRef.current = null; - if (transcript) { - console.info('[VoiceRealtime] Transcript completed', transcript); - onInterimTranscriptRef.current?.(''); - onFinalTranscriptRef.current?.(transcript); - } else { - console.warn('[VoiceRealtime] Received empty transcription event'); - onInterimTranscriptRef.current?.(''); - } - onStatusChangeRef.current?.('listening'); - return; - } - - if (data.type === 'error' && data.error) { - console.error('[VoiceRealtime] Server error', data.error); - } - - if (data.type === 'response.error' && data.error) { - console.error('[VoiceRealtime] Response error', data.error); - } - } catch (err) { - notifyError(err instanceof Error ? err : new Error(String(err))); - } - }, - [extractTextDelta, notifyError] - ); - - const waitForIceGatheringComplete = useCallback((pc: RTCPeerConnection, timeoutMs = 2000) => { - if (pc.iceGatheringState === 'complete') { - return Promise.resolve(); - } - - return new Promise((resolve) => { - let resolved = false; - let timeoutHandle: ReturnType | null = null; - - const finish = () => { - if (resolved) return; - resolved = true; - pc.removeEventListener('icegatheringstatechange', handleStateChange); - if (timeoutHandle) { - clearTimeout(timeoutHandle); - timeoutHandle = null; - } - resolve(); - }; - - const handleStateChange = () => { - if (pc.iceGatheringState === 'complete') { - finish(); - } - }; - - if (timeoutMs) { - timeoutHandle = setTimeout(() => { - console.warn('[VoiceRealtime] ICE gathering timeout reached, proceeding with partial candidates'); - finish(); - }, timeoutMs); - } - - pc.addEventListener('icegatheringstatechange', handleStateChange); - }); - }, []); - - const start = useCallback(async () => { - try { - destroyedRef.current = false; - resetState(); - setConnectionState('connecting'); - - const authToken = getAuthToken?.() ?? null; - const fetchToken = fetchEphemeralToken - ? fetchEphemeralToken - : async (token: string | null) => { - const response = await fetch('/api/realtime/ephemeral-token', { - method: 'POST', - headers: token ? { Authorization: `Bearer ${token}` } : undefined, - }); - if (!response.ok) { - const payload = await response.json().catch(() => ({})); - throw new Error(payload.error || `Failed to mint ephemeral token (${response.status})`); - } - return response.json(); - }; - - const sessionPromise = fetchToken(authToken); - const microphonePromise = initialiseMicrophone(); - - const session = await sessionPromise; - const clientSecret = session?.client_secret?.value || session?.client_secret; - if (!clientSecret || typeof clientSecret !== 'string') { - throw new Error('Realtime session did not include a client_secret'); - } - - const pc = new RTCPeerConnection({ - iceServers: [{ urls: ['stun:stun.l.google.com:19302'] }], - }); - peerConnectionRef.current = pc; - - const channel = pc.createDataChannel('oai-events'); - dataChannelRef.current = channel; - channelReadyPromiseRef.current = new Promise((resolve) => { - channelReadyResolveRef.current = resolve; - }); - channelReadyRef.current = false; - - channel.onmessage = (event) => { - handleDataMessage(event.data); - }; - - channel.onopen = () => { - console.info('[VoiceRealtime] Data channel open'); - setConnectionState('ready'); - channelReadyResolveRef.current?.(); - channelReadyResolveRef.current = null; - channelReadyRef.current = true; - - void sendEvent({ - type: 'session.update', - session: { - modalities: ['text'], - input_audio_transcription: { - model: 'whisper-1', - }, - turn_detection: { - type: 'server_vad', - threshold: 0.5, - silence_duration_ms: 1800, - prefix_padding_ms: 300, - create_response: false, - }, - instructions: 'You are the RA-H voice transport. Only transcribe user speech accurately, never speak responses.', - }, - }).catch((err) => { - console.error('[VoiceRealtime] Failed to send session update:', err); - }); - }; - - channel.onerror = (event) => { - console.error('[VoiceRealtime] Data channel error:', event); - notifyError(new Error('Realtime connection error')); - channelReadyRef.current = false; - disconnect(); - }; - - channel.onclose = () => { - console.warn('[VoiceRealtime] Data channel closed'); - channelReadyPromiseRef.current = null; - channelReadyResolveRef.current = null; - channelReadyRef.current = false; - if (!destroyedRef.current) { - notifyError(new Error('Realtime data channel closed unexpectedly')); - disconnect(); - } - }; - - pc.onconnectionstatechange = () => { - const state = pc.connectionState; - console.info('[VoiceRealtime] Peer connection state:', state); - if (state === 'connected') { - onStatusChangeRef.current?.('listening'); - setConnectionState('capturing'); - } - if (state === 'failed' || state === 'closed' || state === 'disconnected') { - if (!destroyedRef.current) { - notifyError(new Error(`Realtime connection ${state}`)); - } - disconnect(); - } - }; - - pc.oniceconnectionstatechange = () => { - console.info('[VoiceRealtime] ICE connection state:', pc.iceConnectionState); - }; - - pc.ontrack = () => undefined; - - const mediaStream = await microphonePromise; - mediaStream?.getAudioTracks().forEach((track) => { - if (peerConnectionRef.current?.signalingState !== 'closed') { - peerConnectionRef.current?.addTrack(track, mediaStream); - } - }); - - const offer = await pc.createOffer(); - await pc.setLocalDescription(offer); - await waitForIceGatheringComplete(pc, 2000); - - const localDescription = pc.localDescription; - if (!localDescription) { - throw new Error('Failed to create local description for realtime call'); - } - - const response = await fetch('https://api.openai.com/v1/realtime/calls', { - method: 'POST', - headers: { - Authorization: `Bearer ${clientSecret}`, - 'Content-Type': 'application/sdp', - 'OpenAI-Beta': 'realtime=v1', - }, - body: localDescription.sdp, - }); - - if (!response.ok) { - let errorDetail = ''; - try { - const text = await response.text(); - errorDetail = text; - const maybeJson = text ? JSON.parse(text) : null; - if (maybeJson?.error?.message) { - errorDetail = maybeJson.error.message; - } - } catch { - // ignore JSON parse errors and fall back to raw text - } - const friendly = errorDetail || response.statusText || 'Unknown realtime error'; - console.error('[VoiceRealtime] Failed to start realtime call:', friendly, { status: response.status }); - throw new Error(`Failed to establish realtime call (${response.status}): ${friendly}`); - } - - const answerSdp = await response.text(); - await pc.setRemoteDescription({ type: 'answer', sdp: answerSdp }); - - onStatusChangeRef.current?.('listening'); - setConnectionState('capturing'); - } catch (err) { - disconnect(); - notifyError(err instanceof Error ? err : new Error(String(err))); - throw err; - } - }, [disconnect, fetchEphemeralToken, getAuthToken, handleDataMessage, initialiseMicrophone, notifyError, resetState, sendEvent, setConnectionState, waitForIceGatheringComplete]); - - const stop = useCallback(() => { - disconnect(); - }, [disconnect]); - - const latestStopRef = useRef<() => void>(() => {}); - - useEffect(() => { - latestStopRef.current = stop; - }, [stop]); - - useEffect(() => { - return () => { - latestStopRef.current(); - }; - }, []); - - return { - start, - stop, - } as const; -} diff --git a/src/components/agents/hooks/useVoiceInterruption.ts b/src/components/agents/hooks/useVoiceInterruption.ts deleted file mode 100644 index 7b21bb9..0000000 --- a/src/components/agents/hooks/useVoiceInterruption.ts +++ /dev/null @@ -1,83 +0,0 @@ -import { useEffect, useRef, useState } from 'react'; -import type { TTSStatus } from './useAssistantTTS'; - -interface UseVoiceInterruptionOptions { - amplitude: number; - isVoiceActive: boolean; - ttsStatus: TTSStatus; - threshold?: number; - holdDurationMs?: number; - cooldownMs?: number; - onInterruption: () => void; -} - -const DEFAULT_THRESHOLD = 0.18; -const DEFAULT_HOLD_MS = 120; -const DEFAULT_COOLDOWN_MS = 800; - -export function useVoiceInterruption(options: UseVoiceInterruptionOptions) { - const { - amplitude, - isVoiceActive, - ttsStatus, - onInterruption, - threshold = DEFAULT_THRESHOLD, - holdDurationMs = DEFAULT_HOLD_MS, - cooldownMs = DEFAULT_COOLDOWN_MS, - } = options; - - const [isInterrupting, setIsInterrupting] = useState(false); - const detectionTimeoutRef = useRef | null>(null); - const lastInterruptionAtRef = useRef(0); - - useEffect(() => { - return () => { - if (detectionTimeoutRef.current) { - clearTimeout(detectionTimeoutRef.current); - detectionTimeoutRef.current = null; - } - }; - }, []); - - useEffect(() => { - const clearDetection = () => { - if (detectionTimeoutRef.current) { - clearTimeout(detectionTimeoutRef.current); - detectionTimeoutRef.current = null; - } - }; - - if (!isVoiceActive || ttsStatus === 'idle') { - clearDetection(); - if (isInterrupting) { - setIsInterrupting(false); - } - return; - } - - if (amplitude < threshold) { - clearDetection(); - if (isInterrupting) { - setIsInterrupting(false); - } - return; - } - - const now = Date.now(); - if (now - lastInterruptionAtRef.current < cooldownMs) { - return; - } - - if (detectionTimeoutRef.current) { - return; - } - - detectionTimeoutRef.current = setTimeout(() => { - lastInterruptionAtRef.current = Date.now(); - setIsInterrupting(true); - onInterruption(); - }, holdDurationMs); - }, [amplitude, cooldownMs, holdDurationMs, isInterrupting, isVoiceActive, onInterruption, threshold, ttsStatus]); - - return { isInterrupting } as const; -} diff --git a/src/components/agents/hooks/useVoiceSession.ts b/src/components/agents/hooks/useVoiceSession.ts deleted file mode 100644 index 03f091a..0000000 --- a/src/components/agents/hooks/useVoiceSession.ts +++ /dev/null @@ -1,150 +0,0 @@ -import { useCallback, useReducer } from 'react'; - -export type VoiceSessionStatus = 'idle' | 'listening' | 'thinking' | 'speaking'; - -export interface VoiceTranscriptSegment { - id: string; - text: string; - createdAt: number; -} - -interface VoiceSessionState { - isActive: boolean; - status: VoiceSessionStatus; - interimTranscript: string; - segments: VoiceTranscriptSegment[]; - amplitude: number; - startedAt: number | null; -} - -type VoiceSessionAction = - | { type: 'start' } - | { type: 'stop' } - | { type: 'set-status'; status: VoiceSessionStatus } - | { type: 'set-amplitude'; amplitude: number } - | { type: 'set-interim'; transcript: string } - | { type: 'append-segment'; text: string } - | { type: 'replace-segments'; segments: VoiceTranscriptSegment[] } - | { type: 'reset-transcript' }; - -const initialState: VoiceSessionState = { - isActive: false, - status: 'idle', - interimTranscript: '', - segments: [], - amplitude: 0, - startedAt: null, -}; - -function reducer(state: VoiceSessionState, action: VoiceSessionAction): VoiceSessionState { - switch (action.type) { - case 'start': - return { - ...state, - isActive: true, - status: 'listening', - interimTranscript: '', - segments: [], - amplitude: 0, - startedAt: Date.now(), - }; - case 'stop': - return { - ...state, - isActive: false, - status: 'idle', - amplitude: 0, - interimTranscript: '', - startedAt: null, - }; - case 'set-status': - return { - ...state, - status: action.status, - }; - case 'set-amplitude': - return { - ...state, - amplitude: Math.max(0, Math.min(1, action.amplitude)), - }; - case 'set-interim': - return { - ...state, - interimTranscript: action.transcript, - }; - case 'append-segment': { - const trimmed = action.text.trim(); - if (!trimmed) return state; - const segment: VoiceTranscriptSegment = { - id: `voice-${Date.now()}-${Math.random().toString(36).slice(2, 8)}`, - text: trimmed, - createdAt: Date.now(), - }; - return { - ...state, - segments: [...state.segments, segment], - }; - } - case 'replace-segments': - return { - ...state, - segments: action.segments, - }; - case 'reset-transcript': - return { - ...state, - interimTranscript: '', - segments: [], - }; - default: - return state; - } -} - -export function useVoiceSession() { - const [state, dispatch] = useReducer(reducer, initialState); - - const startSession = useCallback(() => { - dispatch({ type: 'start' }); - }, []); - - const stopSession = useCallback(() => { - dispatch({ type: 'stop' }); - }, []); - - const setStatus = useCallback((status: VoiceSessionStatus) => { - dispatch({ type: 'set-status', status }); - }, []); - - const setAmplitude = useCallback((amplitude: number) => { - dispatch({ type: 'set-amplitude', amplitude }); - }, []); - - const setInterimTranscript = useCallback((transcript: string) => { - dispatch({ type: 'set-interim', transcript }); - }, []); - - const appendFinalTranscript = useCallback((text: string) => { - dispatch({ type: 'append-segment', text }); - }, []); - - const resetTranscript = useCallback(() => { - dispatch({ type: 'reset-transcript' }); - }, []); - - const replaceSegments = useCallback((segments: VoiceTranscriptSegment[]) => { - dispatch({ type: 'replace-segments', segments }); - }, []); - - return { - ...state, - startSession, - stopSession, - setStatus, - setAmplitude, - setInterimTranscript, - appendFinalTranscript, - resetTranscript, - replaceSegments, - } as const; -} diff --git a/src/services/voice/usageLogger.ts b/src/services/voice/usageLogger.ts deleted file mode 100644 index 7f9e9d5..0000000 --- a/src/services/voice/usageLogger.ts +++ /dev/null @@ -1,153 +0,0 @@ -import { getSQLiteClient } from '@/services/database/sqlite-client'; - -export interface VoiceUsageLogEntry { - sessionId?: string | null; - helperName?: string | null; - requestId: string; - messageId?: string | null; - voice: string; - model: string; - charCount: number; - costUsd: number; - durationMs?: number | null; - textPreview?: string | null; -} - -type ChatRow = { - id: number; - metadata: string | null; -}; - -function parseMetadata(raw: unknown): Record { - if (!raw) return {}; - if (typeof raw === 'string') { - try { - return JSON.parse(raw); - } catch (error) { - console.warn('[VoiceUsage] Failed to parse chat metadata JSON', error); - return {}; - } - } - if (typeof raw === 'object') { - return (raw as Record) || {}; - } - return {}; -} - -function applyVoiceUsageMetadata( - metadata: Record, - entry: VoiceUsageLogEntry, - loggedAt: string -): Record { - const usageList = Array.isArray(metadata.voice_usage) ? metadata.voice_usage : []; - - const usageEntry = { - request_id: entry.requestId, - message_id: entry.messageId ?? null, - chars: entry.charCount, - cost_usd: entry.costUsd, - voice: entry.voice, - model: entry.model, - duration_ms: entry.durationMs ?? null, - logged_at: loggedAt, - }; - - usageList.push(usageEntry); - const MAX_USAGE_HISTORY = 20; - metadata.voice_usage = usageList.slice(-MAX_USAGE_HISTORY); - - const currentCharsTotal = Number(metadata.voice_tts_chars_total) || 0; - const currentCostTotal = Number(metadata.voice_tts_cost_usd_total) || Number(metadata.voice_tts_cost_total_usd) || 0; - const currentRequestCount = Number(metadata.voice_tts_request_count) || 0; - - metadata.voice_tts_chars_total = currentCharsTotal + entry.charCount; - metadata.voice_tts_cost_usd_total = parseFloat((currentCostTotal + entry.costUsd).toFixed(6)); - metadata.voice_tts_request_count = currentRequestCount + 1; - - metadata.voice_tts_chars = entry.charCount; - metadata.voice_tts_cost_usd = entry.costUsd; - metadata.voice_request_id = entry.requestId; - metadata.voice_tts_voice = entry.voice; - metadata.voice_tts_model = entry.model; - metadata.voice_tts_duration_ms = entry.durationMs ?? null; - metadata.voice_tts_last_logged_at = loggedAt; - metadata.voice_message_id = entry.messageId ?? null; - - return metadata; -} - -export function recordVoiceUsage(entry: VoiceUsageLogEntry): void { - try { - const sqlite = getSQLiteClient(); - const loggedAt = new Date().toISOString(); - let chatId: number | null = null; - - if (entry.sessionId) { - try { - const row = sqlite - .prepare( - ` - SELECT id, metadata - FROM chats - WHERE json_extract(metadata, '$.session_id') = ? - ORDER BY id DESC - LIMIT 1 - ` - ) - .get(entry.sessionId) as ChatRow | undefined; - - if (row) { - chatId = row.id; - const parsedMetadata = applyVoiceUsageMetadata(parseMetadata(row.metadata), entry, loggedAt); - sqlite - .prepare(`UPDATE chats SET metadata = ? WHERE id = ?`) - .run(JSON.stringify(parsedMetadata), row.id); - } else { - console.warn(`[VoiceUsage] No chat row found for session ${entry.sessionId}, logging standalone entry.`); - } - } catch (error) { - console.error('[VoiceUsage] Failed to attach usage to chat metadata', error); - } - } - - try { - sqlite - .prepare( - ` - INSERT INTO voice_usage ( - chat_id, - session_id, - helper_name, - request_id, - message_id, - voice, - model, - chars, - cost_usd, - duration_ms, - text_preview, - created_at - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - ` - ) - .run( - chatId, - entry.sessionId ?? null, - entry.helperName ?? null, - entry.requestId, - entry.messageId ?? null, - entry.voice, - entry.model, - entry.charCount, - entry.costUsd, - entry.durationMs ?? null, - entry.textPreview ?? null, - loggedAt - ); - } catch (error) { - console.error('[VoiceUsage] Failed to insert voice usage row', error); - } - } catch (outerError) { - console.error('[VoiceUsage] Unexpected error while recording usage', outerError); - } -}