osborn 0.9.180 → 0.9.182

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,6 +1,7 @@
1
1
  // Load environment variables FIRST before any other imports
2
2
  import 'dotenv/config';
3
3
  import { voice, initializeLogger } from '@livekit/agents';
4
+ import { CloudTurnDetector } from './turn-detector-shim.js';
4
5
  import { Room, RoomEvent, } from '@livekit/rtc-node';
5
6
  import { AccessToken } from 'livekit-server-sdk';
6
7
  // Initialize logger before anything else
@@ -28,31 +29,23 @@ import { createGunzip } from 'node:zlib';
28
29
  const __filename = fileURLToPath(import.meta.url);
29
30
  const __dirname = dirname(__filename);
30
31
  import { createPatch } from 'diff';
31
- import { loadConfig, getMcpServers, getEnabledMcpServerNames, getVoiceMode, getRealtimeConfig, getDirectConfig, listAllClaudeSessions, getMostRecentSessionId, sessionExists, getSessionSummary, getConversationHistory, ensureSessionWorkspace, getSessionWorkspace, getMcpServerStatusList, buildMcpServersForKeys, listWorkspaceArtifacts } from './config.js';
32
- import { createSTT, createTTS, createRealtimeModelFromConfig, DIRECT_MODE_STT, DIRECT_MODE_TTS } from './voice-io.js';
32
+ import { loadConfig, getMcpServers, getEnabledMcpServerNames, getVoiceMode, getDirectConfig, listAllClaudeSessions, invalidateSessionListCache, getMostRecentSessionId, sessionExists, getSessionSummary, getConversationHistory, ensureSessionWorkspace, getMcpServerStatusList, buildMcpServersForKeys, listWorkspaceArtifacts } from './config.js';
33
+ import { createSTT, createTTS, DIRECT_MODE_STT, DIRECT_MODE_TTS } from './voice-io.js';
33
34
  import { createClaudeLLM, NAMED_AGENTS, applyTurbo } from './claude-llm.js';
34
35
  import { clearPipelineFastBrainSession, prewarmBM25Index } from './pipeline-fastbrain.js';
36
+ import { getIndexPath, buildSummaryIndex } from './summary-index.js';
35
37
  import { ensureClaudeAuth } from './claude-auth.js';
36
38
  import { createSmitheryProxy, destroySmitheryProxy, parseSmitheryUrl, isSmitheryUrl, SmitheryAuthorizationError } from './smithery-proxy.js';
37
- import { askHaiku, askFastBrain, updateSpecFromJSONL, processResearchCompletion, handleResearchBatch, prepareBriefingScript, prepareRecoveryScript, writeQuestionToSpec, checkOutputAgainstQuestions, generateProactivePrompt, clearFastBrainSession } from './fast-brain.js';
38
- import { DIRECT_MODE_PROMPT, getRealtimeInstructions, getScriptInjection, getProactiveInjection, getNotificationInjection } from './prompts.js';
39
+ import { DIRECT_MODE_PROMPT } from './prompts.js';
39
40
  import { MCP_CATALOG } from './config.js';
40
41
  import { getRecallClient } from './recall-client.js';
41
42
  import { MeetingTranscriptPoller } from './meeting-transcript-poller.js';
42
43
  import { llm } from '@livekit/agents';
43
- import { z } from 'zod';
44
44
  // ============================================================
45
- // DUAL MODE VOICE ARCHITECTURE
46
- // ============================================================
47
- // DIRECT MODE (default): STT → Claude Agent SDK → TTS
45
+ // PIPELINE MODE: STT (Deepgram) → Claude Agent SDK → TTS
48
46
  // - Full coding capabilities via Claude Agent SDK
47
+ // - Parallel fast brain for session memory recall
49
48
  // - Permission system flows to frontend
50
- // - Best for actual coding tasks
51
- //
52
- // REALTIME MODE: OpenAI/Gemini native speech-to-speech
53
- // - Faster response, lower latency
54
- // - Voice LLM with tool calling (ask_agent, respond_permission)
55
- // - Routes tasks to Claude agents for execution
56
49
  // ============================================================
57
50
  // Build an enriched tool-use event for the frontend Logs drawer so it can
58
51
  // render Claude-style review cards (Read/Edited/Ran with file names, +/- line
@@ -1449,6 +1442,7 @@ function startApiServer(workingDir, port) {
1449
1442
  return;
1450
1443
  }
1451
1444
  const { filesWritten, remapped, skillsWritten } = await mergeExtractedClaudeDir(tmpExtractDir, targetWorkDir);
1445
+ invalidateSessionListCache(); // imported sessions now on disk
1452
1446
  res.writeHead(200, { 'Content-Type': 'application/json' });
1453
1447
  res.end(JSON.stringify({ ok: true, filesWritten, remapped, skillsWritten }));
1454
1448
  }
@@ -1763,36 +1757,47 @@ function startApiServer(workingDir, port) {
1763
1757
  * OpenAI handles full history (30 exchanges, 2000 char content).
1764
1758
  */
1765
1759
  /**
1766
- * Load full session conversation history into the realtime model's ChatContext.
1767
- * This gives the model persistent memory of what was discussed/researched,
1768
- * enabling deeper follow-up conversations without re-delegating to ask_agent.
1760
+ * Pass the session index path to the voice agent's ChatContext.
1761
+ *
1762
+ * For active/previously-run sessions the index already exists — startIndexWatcher
1763
+ * keeps it current throughout the conversation. We just hand the agent the file
1764
+ * path so it can reference it if the user asks about prior work. No file read at
1765
+ * resume time.
1769
1766
  *
1770
- * NOTE: Gemini's Live API doesn't support updateChatCtx (crashes with code 1008).
1771
- * For Gemini, the session resume context is already injected via generateReply({ userInput })
1772
- * which becomes part of the conversation history as model turns.
1767
+ * Only exception: first-ever resume of a session that was never indexed (e.g. a
1768
+ * session started before the index watcher was added). In that case we build once.
1773
1769
  */
1774
- function loadSessionHistoryIntoChatCtx(agent, history, provider) {
1775
- if (!agent || history.length === 0)
1770
+ function injectSessionIndexIntoChatCtx(agent, sessionId, workingDir) {
1771
+ if (!agent || !sessionId)
1776
1772
  return;
1777
- // Skip for Gemini — updateChatCtx triggers unsupported operations on Gemini Live API
1778
- if (provider === 'gemini') {
1779
- console.log(`🧠 Skipping ChatCtx load for Gemini (${history.length} exchanges) — context injected via generateReply`);
1780
- return;
1781
- }
1782
1773
  try {
1783
- const chatCtx = agent.chatCtx.copy();
1784
- // Inject each conversation exchange as a proper chat message
1785
- for (const exchange of history) {
1786
- chatCtx.addMessage({
1787
- role: exchange.role === 'user' ? 'user' : 'assistant',
1788
- content: exchange.content,
1789
- });
1774
+ let indexPath = getIndexPath(sessionId, workingDir);
1775
+ if (!indexPath) {
1776
+ // First-ever resume with no index — build once, then point at it
1777
+ console.log(`🗂️ No index for ${sessionId.slice(0, 8)} — building (one-time)...`);
1778
+ try {
1779
+ const state = buildSummaryIndex(sessionId, workingDir);
1780
+ indexPath = (state.indexPath && existsSync(state.indexPath) && statSync(state.indexPath).size > 0)
1781
+ ? state.indexPath : null;
1782
+ }
1783
+ catch {
1784
+ return;
1785
+ }
1790
1786
  }
1787
+ if (!indexPath)
1788
+ return;
1789
+ // Pass the path — don't read the file. Native --resume already loaded the session;
1790
+ // this is just a reference the agent can use to search history if asked.
1791
+ const chatCtx = agent.chatCtx.copy();
1792
+ chatCtx.addMessage({
1793
+ role: 'user',
1794
+ content: `[Session resume] Previous conversation history is indexed at: ${indexPath}\nFormat per line: lineNum|timestamp|source|msgType|summary`,
1795
+ });
1791
1796
  agent.updateChatCtx(chatCtx);
1792
- console.log(`🧠 Loaded ${history.length} conversation exchanges into ChatCtx (${history.reduce((sum, e) => sum + e.content.length, 0)} chars)`);
1797
+ console.log(`🗂️ Session index path injected: ${indexPath}`);
1793
1798
  }
1794
1799
  catch (err) {
1795
- console.log('⚠️ Failed to load session history into ChatCtx:', err);
1800
+ console.log('⚠️ Failed to inject session index path:', err);
1796
1801
  }
1797
1802
  }
1798
1803
  // Main function
@@ -1858,18 +1863,10 @@ async function main() {
1858
1863
  console.log(`📂 Session base directory: ${sessionBaseDir}`);
1859
1864
  console.log(` (cwd from ${cwdSource})`);
1860
1865
  console.log(`🔬 Mode: RESEARCH`);
1861
- // Determine voice mode
1866
+ // Pipeline mode
1862
1867
  const voiceMode = getVoiceMode(config);
1863
- const realtimeConfig = getRealtimeConfig(config);
1864
1868
  const directConfig = getDirectConfig(config);
1865
- if (voiceMode === 'realtime') {
1866
- console.log(`🎙️ REALTIME MODE: ${realtimeConfig.provider} native speech-to-speech`);
1867
- console.log(` Voice: ${realtimeConfig.provider === 'openai' ? realtimeConfig.openaiVoice : realtimeConfig.geminiVoice}`);
1868
- }
1869
- else {
1870
- console.log(`🎯 DIRECT MODE: ${directConfig.stt.provider} STT → Claude Agent SDK → ${directConfig.tts.provider} TTS`);
1871
- console.log(' 🔥 Full coding capabilities!');
1872
- }
1869
+ console.log(`🎯 PIPELINE MODE: ${directConfig.stt.provider} STT → Claude Agent SDK + fast brain → ${directConfig.tts.provider} TTS`);
1873
1870
  // Determine room code. STABLE PER MACHINE, derived from identity we
1874
1871
  // already have: the Fly app name (one app per user). No new storage, no
1875
1872
  // rotation — the same user always lands in the same room, and the room
@@ -2228,7 +2225,6 @@ async function main() {
2228
2225
  let sessionAlwaysAllowPaths = new Set();
2229
2226
  let userState = 'listening'; // Track user speech state for queue safety
2230
2227
  let currentVoiceMode = voiceMode; // Track active voice mode for data handlers
2231
- let currentProvider = realtimeConfig.provider; // Track active realtime provider
2232
2228
  // Authenticated Supabase userId from participant metadata. Used to scope
2233
2229
  // workspace artifact uploads to the owner's prefix in Supabase Storage.
2234
2230
  // Empty string = anonymous / unauthenticated; uploads fall back to a
@@ -2407,7 +2403,7 @@ async function main() {
2407
2403
  console.log('🏁 Meeting over with no user connected — releasing LLM + arming idle-exit');
2408
2404
  killCurrentLLM(`meeting_ended(${reason})_no_user`);
2409
2405
  currentLLM = null;
2410
- clearFastBrainSession();
2406
+ clearPipelineFastBrainSession();
2411
2407
  clearPipelineFastBrainSession();
2412
2408
  armIdleExitTimer(`meeting ended (${reason}), no user`);
2413
2409
  }
@@ -2962,29 +2958,11 @@ async function main() {
2962
2958
  }
2963
2959
  }, 15000);
2964
2960
  try {
2965
- // Skip interrupt for Gemini — disrupts Gemini's state machine, causing it to
2966
- // never transition back to 'listening' (hangs in speaking state indefinitely)
2967
- if (currentProvider !== 'gemini') {
2968
- currentSession.interrupt();
2969
- }
2970
- if (currentProvider === 'gemini') {
2971
- // LiveKit SDK v1.0.51: generateReply({ instructions }) sends a system turn +
2972
- // synthetic "." user turn. After Gemini processes a tool call in this flow,
2973
- // autoToolReplyGeneration does NOT trigger continuation (system-only limitation).
2974
- // Using userInput instead makes it a "user-initiated" request where auto-continuation
2975
- // works. The ask_fast_brain injection bypass handles [SCRIPT]/[PROACTIVE]/[NOTIFICATION]
2976
- // prefixes and returns the content directly as a tool response.
2977
- currentSession.generateReply({
2978
- userInput: batchedInstruction,
2979
- });
2980
- }
2981
- else {
2982
- // OpenAI respects toolChoice:'none' — speaks instructions directly
2983
- currentSession.generateReply({
2984
- instructions: batchedInstruction,
2985
- toolChoice: 'none',
2986
- });
2987
- }
2961
+ currentSession.interrupt();
2962
+ currentSession.generateReply({
2963
+ instructions: batchedInstruction,
2964
+ toolChoice: 'none',
2965
+ });
2988
2966
  // Model transitions to thinking/speaking after this call.
2989
2967
  // When it returns to 'listening', agent_state_changed triggers processVoiceQueue() again.
2990
2968
  // Also inject into chatCtx as persistent context so the model remembers across turns
@@ -3003,9 +2981,6 @@ async function main() {
3003
2981
  function injectIntoChatCtx(content) {
3004
2982
  if (!currentAgent)
3005
2983
  return;
3006
- // Skip for Gemini — updateChatCtx triggers unsupported operations on Gemini Live API
3007
- if (currentVoiceMode === 'realtime' && currentProvider === 'gemini')
3008
- return;
3009
2984
  try {
3010
2985
  const chatCtx = currentAgent.chatCtx.copy();
3011
2986
  chatCtx.addMessage({
@@ -3064,63 +3039,11 @@ async function main() {
3064
3039
  isStreaming: true,
3065
3040
  agentRole: 'research-progress',
3066
3041
  });
3067
- // Route through fast brain — it decides whether to speak (usually silent)
3068
- if (activeResearch.voiceUpdateCount < 2) {
3069
- const voiceSid = currentLLM?.sessionId;
3070
- if (voiceSid) {
3071
- const chatHistory = getChatHistory(10);
3072
- handleResearchBatch(workingDir, voiceSid, lastTaskRequest || '', updates, activeResearch.researchLog, chatHistory, workingDir)
3073
- .then(script => {
3074
- if (script && activeResearch) {
3075
- activeResearch.voiceUpdateCount++;
3076
- queueVoiceInjection(getScriptInjection(script));
3077
- }
3078
- })
3079
- .catch(() => { }); // Silent fail — updates are optional
3080
- }
3081
- }
3042
+ // Research batch updates are logged; pipeline uses fast brain for recall
3082
3043
  }, 8000); // 8s debounce: reduces voice queue flooding during research
3083
3044
  }
3084
- // Proactive conversational loop — keeps conversation alive during research
3085
- let proactiveTimer = null;
3086
- let proactivePromptHistory = [];
3087
- const PROACTIVE_INTERVAL = 15000; // 15 seconds (offset from 8s batch timer)
3088
- const MAX_PROACTIVE_PROMPTS = 2; // Cap per research task (reduced from 4 to minimize realtime LLM tokens)
3089
- function startProactiveLoop(task, sessionId) {
3090
- stopProactiveLoop();
3091
- proactivePromptHistory = [];
3092
- let proactiveCount = 0;
3093
- proactiveTimer = setInterval(async () => {
3094
- if (!activeResearch) {
3095
- stopProactiveLoop();
3096
- return;
3097
- }
3098
- if (proactiveCount >= MAX_PROACTIVE_PROMPTS)
3099
- return;
3100
- if (agentState !== 'listening' || userState === 'speaking')
3101
- return;
3102
- if (researchBatchTimer)
3103
- return; // Don't collide with batch updates
3104
- if (isProcessingQueue)
3105
- return; // Don't collide with voice queue
3106
- try {
3107
- const prompt = await generateProactivePrompt(workingDir, sessionId, task, activeResearch.researchLog, proactivePromptHistory, sessionBaseDir);
3108
- if (prompt && prompt !== 'NOTHING') {
3109
- proactivePromptHistory.push(prompt);
3110
- proactiveCount++;
3111
- queueVoiceInjection(getProactiveInjection(prompt));
3112
- }
3113
- }
3114
- catch { } // Silent fail — proactive prompts are optional
3115
- }, PROACTIVE_INTERVAL);
3116
- }
3117
- function stopProactiveLoop() {
3118
- if (proactiveTimer) {
3119
- clearInterval(proactiveTimer);
3120
- proactiveTimer = null;
3121
- }
3122
- proactivePromptHistory = [];
3123
- }
3045
+ function startProactiveLoop(_task, _sessionId) { }
3046
+ function stopProactiveLoop() { }
3124
3047
  // Helper to send data to frontend (with size limit handling)
3125
3048
  //
3126
3049
  // WebRTC SCTP data channel max message size is ~256KB. Sending larger
@@ -3182,20 +3105,14 @@ async function main() {
3182
3105
  console.error('❌ sendToFrontend error:', err);
3183
3106
  }
3184
3107
  }
3185
- // Helper: announce via voice - uses voice queue for realtime, say() for direct
3186
3108
  async function announceViaVoice(text) {
3187
3109
  if (!currentSession)
3188
3110
  return;
3189
- if (currentVoiceMode === 'realtime') {
3190
- queueVoiceInjection(getNotificationInjection(text));
3111
+ try {
3112
+ await currentSession.say(text);
3191
3113
  }
3192
- else {
3193
- try {
3194
- await currentSession.say(text);
3195
- }
3196
- catch (err) {
3197
- console.log('⚠️ Voice announcement failed:', err);
3198
- }
3114
+ catch (err) {
3115
+ console.log('⚠️ Voice announcement failed:', err);
3199
3116
  }
3200
3117
  }
3201
3118
  // Compaction event → frontend bridge. Forwards the raw event (consumed by the
@@ -3297,6 +3214,7 @@ async function main() {
3297
3214
  directLLM.events.once('session_id', ({ sessionId }) => {
3298
3215
  const workspace = ensureSessionWorkspace(workingDir, sessionId);
3299
3216
  console.log(`📁 Session workspace created: ${workspace}`);
3217
+ invalidateSessionListCache(); // new session now on disk — bust stale cache
3300
3218
  // Pipeline mode: pre-warm BM25 index so first fast brain query is fast
3301
3219
  if (currentVoiceMode === 'pipeline') {
3302
3220
  prewarmBM25Index(sessionId, workingDir).catch(() => { });
@@ -3685,8 +3603,9 @@ async function main() {
3685
3603
  // discardAudioIfUninterruptible: true, ttsReadIdleTimeout: 10000,
3686
3604
  // maxUnrecoverableErrors: 3) are what was silently running via caret-resolved
3687
3605
  // 1.4.5 throughout the user's working month. Restoring them.
3606
+ const turnDetector = process.env.LIVEKIT_REMOTE_EOT_URL ? new CloudTurnDetector() : undefined;
3688
3607
  const session = new voice.AgentSession({
3689
- turnDetection: 'stt',
3608
+ turnDetection: (turnDetector ?? 'stt'),
3690
3609
  preemptiveGeneration: false, // Only fire LLM on final committed transcript, not partial preemptives
3691
3610
  // Commented out — kept for reference. These were added across 0.9.60/0.9.61
3692
3611
  // to try to harden interrupt + TTS handling, but evidence (osbornojure
@@ -3761,528 +3680,6 @@ async function main() {
3761
3680
  return { session, agent };
3762
3681
  }
3763
3682
  // ============================================================
3764
- // REALTIME MODE - OpenAI/Gemini native speech-to-speech
3765
- // ============================================================
3766
- // Claude handler for realtime mode tool execution
3767
- let realtimeClaudeHandler = null;
3768
- // Create REALTIME session (OpenAI/Gemini native speech-to-speech)
3769
- async function createRealtimeSession(sessionRealtimeConfig, resumeSessionId) {
3770
- const rtConfig = sessionRealtimeConfig || realtimeConfig;
3771
- console.log(`🎯 Creating realtime session (${rtConfig.provider})...`);
3772
- // Create Claude LLM for tool execution (research tasks)
3773
- realtimeClaudeHandler = createClaudeLLM({
3774
- workingDirectory: workingDir,
3775
- sessionBaseDir,
3776
- mcpServers,
3777
- resumeSessionId,
3778
- onCompactionEvent: buildOnCompactionEvent(),
3779
- });
3780
- currentLLM = realtimeClaudeHandler;
3781
- // For resumed sessions, eagerly create workspace (we know the real ID)
3782
- if (resumeSessionId) {
3783
- const workspace = ensureSessionWorkspace(workingDir, resumeSessionId);
3784
- console.log(`📁 Session workspace (resumed): ${workspace}`);
3785
- }
3786
- // For new sessions, create workspace when SDK assigns real session ID
3787
- realtimeClaudeHandler.events.once('session_id', ({ sessionId }) => {
3788
- const workspace = ensureSessionWorkspace(workingDir, sessionId);
3789
- console.log(`📁 Session workspace created: ${workspace}`);
3790
- });
3791
- // Wire up MCP server changes to frontend
3792
- realtimeClaudeHandler.events.on('mcp_servers_changed', (data) => {
3793
- console.log(`🔌 MCP servers changed: ${data.enabledKeys.join(', ') || 'none'}`);
3794
- sendToFrontend({
3795
- type: 'mcp_servers_changed',
3796
- enabledKeys: data.enabledKeys,
3797
- mcpServers: getMcpServerStatusList(config),
3798
- });
3799
- });
3800
- // Wire up Claude events to frontend
3801
- realtimeClaudeHandler.events.on('tool_use', (data) => {
3802
- console.log(`🔧 Claude: ${data.name}`);
3803
- sendToFrontend(buildToolLogEvent(data.name, data.input, 'running', data.agentRole || 'main'));
3804
- });
3805
- realtimeClaudeHandler.events.on('tool_result', (data) => {
3806
- console.log(`✅ Done: ${data.name}`);
3807
- sendToFrontend(buildToolLogEvent(data.name, data.input, 'completed', data.agentRole || 'main'));
3808
- // Detect research artifact writes (session workspace or legacy research dir)
3809
- if ((data.name === 'Write' || data.name === 'Edit') && data.input?.file_path) {
3810
- const fp = data.input.file_path;
3811
- if (fp.includes('/osb/') || fp.includes('.osborn/sessions/') || fp.includes('.osborn/research/')) {
3812
- sendToFrontend({
3813
- type: 'research_artifact_updated',
3814
- filePath: fp,
3815
- fileName: fp.split('/').pop(),
3816
- });
3817
- }
3818
- }
3819
- });
3820
- realtimeClaudeHandler.events.on('assistant_result', (data) => {
3821
- console.log(`📋 Claude result (${data.text?.length || 0} chars): ${data.text || ''}`);
3822
- sendToFrontend({
3823
- type: 'claude_output',
3824
- text: data.text,
3825
- isStreaming: false,
3826
- isFinal: true,
3827
- agentRole: 'realtime',
3828
- });
3829
- });
3830
- // Stream Claude's research text to frontend as progress updates
3831
- // Skips during active research to avoid duplication with per-task onText handler
3832
- realtimeClaudeHandler.events.on('assistant_text', (data) => {
3833
- if (data.text && data.text.trim()) {
3834
- if (activeResearch)
3835
- return;
3836
- sendToFrontend({
3837
- type: 'claude_output',
3838
- text: data.text,
3839
- isStreaming: true,
3840
- agentRole: 'realtime-agent',
3841
- });
3842
- }
3843
- });
3844
- realtimeClaudeHandler.events.on('permission_request', (data) => {
3845
- console.log(`⚠️ Permission needed: ${data.toolName}`);
3846
- const toolName = data.toolName;
3847
- const input = data.input || {};
3848
- // Build descriptive message based on tool type
3849
- let description = `I need permission to use ${toolName}.`;
3850
- if (toolName === 'Bash' && input.command) {
3851
- const cmd = String(input.command).substring(0, 60);
3852
- description = `I want to run the command: ${cmd}${String(input.command).length > 60 ? '...' : ''}`;
3853
- }
3854
- else if (toolName === 'Write' && input.file_path) {
3855
- description = `I want to create or overwrite the file: ${input.file_path}`;
3856
- }
3857
- else if (toolName === 'Edit' && input.file_path) {
3858
- description = `I want to edit the file: ${input.file_path}`;
3859
- }
3860
- else if (toolName === 'WebFetch' && input.url) {
3861
- description = `I want to fetch content from: ${input.url}`;
3862
- }
3863
- sendToFrontend({
3864
- type: 'permission_request',
3865
- toolName: data.toolName,
3866
- input: data.input,
3867
- description,
3868
- agentRole: 'realtime',
3869
- });
3870
- });
3871
- // Wire up session resume failure for realtime mode
3872
- realtimeClaudeHandler.events.on('session_resume_failed', (data) => {
3873
- console.error(`❌ Session resume failed: ${data.requestedSessionId} → ${data.actualSessionId}`);
3874
- sendToFrontend({
3875
- type: 'session_resume_failed',
3876
- requestedSessionId: data.requestedSessionId,
3877
- actualSessionId: data.actualSessionId,
3878
- });
3879
- });
3880
- // Wire up file checkpoint capture for realtime mode
3881
- realtimeClaudeHandler.events.on('checkpoint_captured', (data) => {
3882
- console.log(`📍 Checkpoint: ${data.checkpointId.substring(0, 8)}...`);
3883
- sendToFrontend({
3884
- type: 'checkpoint_captured',
3885
- checkpointId: data.checkpointId,
3886
- });
3887
- });
3888
- // Extracted research execution — called by ask_agent, SDK handles queuing internally
3889
- function executeResearch(task) {
3890
- sendToFrontend({ type: 'system', text: `Executing: ${task}` });
3891
- // Fire-and-forget: write user question to spec.md BEFORE agent starts
3892
- const questionSid = currentLLM?.sessionId || resumeSessionId;
3893
- if (questionSid) {
3894
- writeQuestionToSpec(workingDir, questionSid, task).catch(err => console.error('❌ writeQuestionToSpec failed:', err));
3895
- }
3896
- // Clean up previous research UI tracking — but let the SDK query complete in background.
3897
- // The SDK has an internal queue: new query() calls enqueue behind running ones.
3898
- // Old research results land in JSONL and fast brain can access them later.
3899
- if (activeResearch) {
3900
- activeResearch.cleanup(); // Remove event listeners so UI tracks new task
3901
- if (researchBatchTimer) {
3902
- clearTimeout(researchBatchTimer);
3903
- researchBatchTimer = null;
3904
- }
3905
- // NOTE: NOT aborting — old SDK process continues writing to JSONL
3906
- }
3907
- // Set up research log batching — events push to queue for state-driven injection
3908
- const researchLog = [];
3909
- const pendingUpdates = [];
3910
- const onToolUse = (data) => {
3911
- const input = data.input || {};
3912
- let entry;
3913
- if (data.name === 'Read' && input.file_path) {
3914
- const fileName = input.file_path.split('/').pop() || input.file_path;
3915
- entry = `Reading ${fileName}`;
3916
- }
3917
- else if (data.name === 'Bash' && input.command) {
3918
- const cmd = input.command.substring(0, 80);
3919
- entry = `Running: ${cmd}`;
3920
- }
3921
- else if (data.name === 'Glob' && input.pattern) {
3922
- entry = `Searching for files matching ${input.pattern}`;
3923
- }
3924
- else if (data.name === 'Grep' && input.pattern) {
3925
- entry = `Searching for "${input.pattern}" in files`;
3926
- }
3927
- else if (data.name === 'WebSearch' && input.query) {
3928
- entry = `Searching the web for "${input.query}"`;
3929
- }
3930
- else if (data.name === 'WebFetch' && input.url) {
3931
- const hostname = input.url.replace(/https?:\/\//, '').split('/')[0];
3932
- entry = `Fetching content from ${hostname}`;
3933
- }
3934
- else if (data.name === 'Write' && input.file_path) {
3935
- const fileName = input.file_path.split('/').pop() || input.file_path;
3936
- entry = `Writing ${fileName}`;
3937
- }
3938
- else if (data.name === 'Edit' && input.file_path) {
3939
- const fileName = input.file_path.split('/').pop() || input.file_path;
3940
- entry = `Editing ${fileName}`;
3941
- }
3942
- else if (data.name.startsWith('mcp__')) {
3943
- const parts = data.name.split('__');
3944
- const serverName = parts[1] || 'external';
3945
- const toolAction = parts.slice(2).join(' ') || 'tool';
3946
- entry = `Using ${serverName}: ${toolAction}`;
3947
- }
3948
- else {
3949
- entry = `Using ${data.name}`;
3950
- }
3951
- researchLog.push(entry);
3952
- pendingUpdates.push(entry);
3953
- scheduleResearchBatch();
3954
- };
3955
- const ANSWER_CHECK_THRESHOLD = 300; // chars — only check substantial outputs
3956
- const onToolResult = (data) => {
3957
- // Only log to researchLog for the final summary — don't push to pendingUpdates
3958
- // This prevents redundant "Reading config.ts. Read done." voice updates
3959
- researchLog.push(`${data.name} completed`);
3960
- // Fire-and-forget: check if substantial tool results answer any spec questions
3961
- // Note: PostToolUse emits { name, input, response } — use data.response (not data.result)
3962
- const resultText = typeof data.response === 'string' ? data.response : JSON.stringify(data.response || '');
3963
- if (resultText.length > ANSWER_CHECK_THRESHOLD) {
3964
- const sid = currentLLM?.sessionId || resumeSessionId;
3965
- if (sid)
3966
- checkOutputAgainstQuestions(workingDir, sid, resultText, 'tool_result').catch(() => { });
3967
- }
3968
- // When AskUserQuestion completes, the user's answer is a decision — track it in spec
3969
- if (data.name === 'AskUserQuestion' && data.response) {
3970
- const sid = currentLLM?.sessionId || resumeSessionId;
3971
- if (sid) {
3972
- const questionText = JSON.stringify(data.input?.questions || data.input || {});
3973
- const answerText = typeof data.response === 'string' ? data.response : JSON.stringify(data.response);
3974
- const specUpdate = `User answered a clarifying question during research.\nQuestion: ${questionText}\nAnswer: ${answerText}\nRecord this as a user decision in spec.md.`;
3975
- askHaiku(workingDir, sid, specUpdate, undefined, undefined, undefined, workingDir).catch(err => console.error('❌ Failed to record AskUserQuestion answer in spec:', err));
3976
- console.log(`📝 AskUserQuestion answer forwarded to fast brain for spec tracking`);
3977
- }
3978
- }
3979
- };
3980
- const onText = (data) => {
3981
- if (data.text?.trim()) {
3982
- const text = data.text.trim();
3983
- const preview = text.substring(0, 150);
3984
- const firstSentence = preview.match(/^[^.!?\n]+[.!?]/)?.[0] || preview;
3985
- researchLog.push(firstSentence);
3986
- pendingUpdates.push(firstSentence);
3987
- scheduleResearchBatch();
3988
- // Fire-and-forget: check if substantial agent reasoning answers any spec questions
3989
- if (text.length > ANSWER_CHECK_THRESHOLD) {
3990
- const sid = currentLLM?.sessionId || resumeSessionId;
3991
- if (sid)
3992
- checkOutputAgainstQuestions(workingDir, sid, text, 'assistant_text').catch(() => { });
3993
- }
3994
- }
3995
- };
3996
- // Capture the SDK's requestId for this query — identifies this research task
3997
- // in the JSONL file for targeted retrieval by fast brain
3998
- let sdkRequestId = null;
3999
- const onQueryRequestId = (data) => {
4000
- if (!sdkRequestId && data.requestId) {
4001
- sdkRequestId = data.requestId;
4002
- console.log(`📋 [research] SDK requestId: ${sdkRequestId}`);
4003
- }
4004
- };
4005
- realtimeClaudeHandler.events.on('tool_use', onToolUse);
4006
- realtimeClaudeHandler.events.on('tool_result', onToolResult);
4007
- realtimeClaudeHandler.events.on('assistant_text', onText);
4008
- realtimeClaudeHandler.events.on('query_request_id', onQueryRequestId);
4009
- const cleanupListeners = () => {
4010
- realtimeClaudeHandler?.events.off('tool_use', onToolUse);
4011
- realtimeClaudeHandler?.events.off('tool_result', onToolResult);
4012
- realtimeClaudeHandler?.events.off('assistant_text', onText);
4013
- realtimeClaudeHandler?.events.off('query_request_id', onQueryRequestId);
4014
- };
4015
- // Create AbortController for this research task — abort on disconnect/cleanup
4016
- const researchAbortController = new AbortController();
4017
- // Track active research — updates drain when model enters 'listening' state
4018
- const thisResearch = {
4019
- researchLog,
4020
- pendingUpdates,
4021
- cleanup: cleanupListeners,
4022
- voiceUpdateCount: 0,
4023
- abortController: researchAbortController,
4024
- };
4025
- activeResearch = thisResearch;
4026
- // Start proactive conversational loop
4027
- const proactiveSid = currentLLM?.sessionId || resumeSessionId;
4028
- if (proactiveSid) {
4029
- startProactiveLoop(task, proactiveSid);
4030
- }
4031
- // Run research in the background (non-blocking)
4032
- // Pass AbortController so research can be stopped on disconnect
4033
- const researchPromise = (async () => {
4034
- const stream = realtimeClaudeHandler.chat({
4035
- chatCtx: {
4036
- items: [{ type: 'message', role: 'user', content: [task] }],
4037
- },
4038
- abortController: researchAbortController,
4039
- });
4040
- let result = '';
4041
- for await (const chunk of stream) {
4042
- if (chunk.delta?.content) {
4043
- result += chunk.delta.content;
4044
- }
4045
- }
4046
- return result;
4047
- })();
4048
- // Handle completion asynchronously
4049
- researchPromise.then(async (result) => {
4050
- // Check if aborted — empty result means clean abort, skip pipeline
4051
- if (researchAbortController.signal.aborted || !result.trim()) {
4052
- console.log(`🛑 [realtime] Research aborted or empty: ${task.substring(0, 60)}`);
4053
- cleanupListeners();
4054
- if (activeResearch === thisResearch) {
4055
- activeResearch = null;
4056
- }
4057
- return;
4058
- }
4059
- const isStillCurrent = activeResearch === thisResearch;
4060
- console.log(`✅ [realtime] Research complete (${result.length} chars${isStillCurrent ? '' : ', superseded by newer task'})`);
4061
- // Clean up
4062
- cleanupListeners();
4063
- // Send raw result to frontend as a log entry (not assistant_response — that's reserved
4064
- // for the voice model's spoken response, avoiding duplication in chat)
4065
- await sendToFrontend({ type: 'claude_output', text: result, isStreaming: false, agentRole: 'research-result' });
4066
- const resultPreview = result.length > 150
4067
- ? result.substring(0, 150) + '...'
4068
- : result;
4069
- await sendToFrontend({ type: 'task_completed', task, resultPreview });
4070
- // Only modify global state if we're still the current research task.
4071
- // If a newer task replaced us, don't clobber its timers/state.
4072
- if (isStillCurrent) {
4073
- if (researchBatchTimer) {
4074
- clearTimeout(researchBatchTimer);
4075
- researchBatchTimer = null;
4076
- }
4077
- stopProactiveLoop();
4078
- }
4079
- // Preserve research context for follow-up questions
4080
- lastCompletedResearch = {
4081
- task,
4082
- researchLog: [...researchLog],
4083
- completedAt: Date.now(),
4084
- };
4085
- // Only clear activeResearch if we're still the current task
4086
- if (isStillCurrent) {
4087
- activeResearch = null;
4088
- }
4089
- // Send research_task_complete to frontend for inline chat tracking
4090
- await sendToFrontend({
4091
- type: 'research_task_complete',
4092
- task,
4093
- summary: result.substring(0, 500),
4094
- });
4095
- // Route through fast brain to generate a teleprompter script from the findings
4096
- // Fast brain reads full JSONL and writes a spoken monologue
4097
- const voiceSid = currentLLM?.sessionId || resumeSessionId;
4098
- const chatHistory = getChatHistory(10);
4099
- console.log(`📡 [realtime] Generating teleprompter script via fast brain (result: ${result.length} chars, agentState: ${agentState})`);
4100
- // Create sendToChat for research completion to send structured data to frontend
4101
- const completionSendToChat = (text) => {
4102
- sendToFrontend({ type: 'assistant_response', text });
4103
- };
4104
- if (voiceSid) {
4105
- processResearchCompletion(workingDir, voiceSid, task, result, chatHistory, completionSendToChat, workingDir)
4106
- .then(script => {
4107
- queueVoiceInjection(getScriptInjection(script));
4108
- })
4109
- .catch(() => {
4110
- // Fallback: use truncated result directly if fast brain fails
4111
- queueVoiceInjection(getScriptInjection(result.substring(0, 500)));
4112
- });
4113
- }
4114
- else {
4115
- queueVoiceInjection(getScriptInjection(result.substring(0, 500)));
4116
- }
4117
- // Fire-and-forget JSONL-based refinement pass via fast brain
4118
- // Reads FULL untruncated data from JSONL — no content buffer, no truncation
4119
- const postResearchSessionId = currentLLM?.sessionId || resumeSessionId;
4120
- if (postResearchSessionId) {
4121
- updateSpecFromJSONL(workingDir, postResearchSessionId, task, researchLog, workingDir)
4122
- .then(updateResult => {
4123
- if (!updateResult)
4124
- return;
4125
- // Notify frontend about spec.md update
4126
- if (updateResult.spec) {
4127
- const specPath = join(getSessionWorkspace(workingDir, postResearchSessionId), 'spec.md');
4128
- sendToFrontend({
4129
- type: 'research_artifact_updated',
4130
- filePath: specPath,
4131
- fileName: 'spec.md',
4132
- });
4133
- }
4134
- });
4135
- }
4136
- }).catch(async (err) => {
4137
- // Clean up
4138
- cleanupListeners();
4139
- const isStillCurrent = activeResearch === thisResearch;
4140
- if (isStillCurrent) {
4141
- if (researchBatchTimer) {
4142
- clearTimeout(researchBatchTimer);
4143
- researchBatchTimer = null;
4144
- }
4145
- stopProactiveLoop();
4146
- activeResearch = null;
4147
- }
4148
- // If aborted (user disconnected), log quietly
4149
- if (researchAbortController.signal.aborted) {
4150
- console.log(`🛑 [realtime] Research aborted: ${task.substring(0, 60)}`);
4151
- return;
4152
- }
4153
- console.error(`❌ [realtime] Research failed:`, err);
4154
- // Queue error notification — will be spoken when model is available
4155
- queueVoiceInjection(getNotificationInjection(`Research encountered an error: ${err.message}. You could try asking again.`));
4156
- });
4157
- // Return immediately to unblock the voice model
4158
- return 'Research started. I\'ll relay findings as they come in — you can keep talking to the user while I work.';
4159
- }
4160
- // Create tools for the realtime voice LLM
4161
- // The realtime model is a thin teleprompter — only 2 tools:
4162
- // 1. ask_fast_brain: ALL user questions route here (the fast brain decides everything)
4163
- // 2. respond_permission: voice permission flow for Claude SDK blocked operations
4164
- const askFastBrainTool = llm.tool({
4165
- description: `Ask your brain. Call this for EVERY user message — greetings, questions, decisions, requests, everything. No exceptions. Returns what you should say.`,
4166
- parameters: z.object({
4167
- question: z.string().describe('The user\'s question or statement'),
4168
- }),
4169
- execute: async ({ question }) => {
4170
- // INJECTION BYPASS: When Gemini receives a system injection via generateReply(),
4171
- // it calls ask_fast_brain with the injection content (Gemini always calls tools).
4172
- // For Gemini: this is the INTENDED path — we deliberately don't set toolChoice:'none'
4173
- // so the tool call goes through and we return the content as a tool response.
4174
- // For OpenAI: this is a fallback guard — OpenAI normally speaks instructions directly
4175
- // with toolChoice:'none', but if it somehow calls the tool, we handle it here.
4176
- const injectionMatch = question.match(/\[(SCRIPT|PROACTIVE|NOTIFICATION)\]\s*([\s\S]*)/);
4177
- if (injectionMatch) {
4178
- const content = injectionMatch[2].trim();
4179
- console.log(`⚡ [fast brain] BYPASS: injection [${injectionMatch[1]}] → returning content directly (${content.length} chars)`);
4180
- return content || question;
4181
- }
4182
- // Use pending sessionId for fresh sessions where SDK hasn't assigned one yet
4183
- const sessionId = currentLLM?.sessionId || currentResumeSessionId || resumeSessionId || 'pending';
4184
- console.log(`🧠 [fast brain] Question: "${question.substring(0, 80)}..."`);
4185
- // Track in-flight state
4186
- haikuInFlight = { question, time: Date.now() };
4187
- // Build research context — from active research or last completed research
4188
- let researchContext;
4189
- if (activeResearch && activeResearch.researchLog.length > 0) {
4190
- const recentLog = activeResearch.researchLog.slice(-15);
4191
- researchContext = `Research topic: "${lastTaskRequest || 'unknown'}"\nSteps completed (${activeResearch.researchLog.length} total, showing last ${recentLog.length}):\n${recentLog.join('\n')}`;
4192
- }
4193
- else if (lastCompletedResearch && (Date.now() - lastCompletedResearch.completedAt) < 600000) {
4194
- // Include context from last completed research (within 10 minutes)
4195
- const recentLog = lastCompletedResearch.researchLog.slice(-15);
4196
- researchContext = `[COMPLETED RESEARCH] Topic: "${lastCompletedResearch.task}"\nSteps completed (${lastCompletedResearch.researchLog.length} total, showing last ${recentLog.length}):\n${recentLog.join('\n')}\n\n(Research completed — results are in JSONL and spec.md. Answer from those, do NOT trigger new research on this topic.)`;
4197
- }
4198
- const callbacks = {
4199
- triggerResearch: (task) => {
4200
- // Deduplication guard
4201
- const now = Date.now();
4202
- if (task === lastTaskRequest && (now - lastTaskTime) < 10000) {
4203
- console.log('⏭️ Skipping duplicate research task (within 10s window)');
4204
- return;
4205
- }
4206
- lastTaskRequest = task;
4207
- lastTaskTime = now;
4208
- executeResearch(task);
4209
- },
4210
- queueVoice: (script) => {
4211
- queueVoiceInjection(getScriptInjection(script));
4212
- },
4213
- sendToFrontend: (data) => {
4214
- sendToFrontend(data);
4215
- },
4216
- };
4217
- try {
4218
- const chatHistory = getChatHistory(20);
4219
- const result = await askFastBrain(workingDir, sessionId, question, {
4220
- chatHistory,
4221
- researchContext,
4222
- callbacks,
4223
- });
4224
- haikuInFlight = null;
4225
- // Voice queue items may have been held while fast brain was in flight — retry now
4226
- if (voiceQueue.length > 0) {
4227
- setTimeout(() => processVoiceQueue(), 500);
4228
- }
4229
- console.log(`🧠 [fast brain] Response type: ${result.type}, script: ${result.script.length} chars`);
4230
- // If this was a user direction during active research,
4231
- // pass it to the agent SDK so it picks up the context
4232
- if (activeResearch && result.type === 'recorded' && (question.toLowerCase().includes('decided') ||
4233
- question.toLowerCase().includes('prefers') ||
4234
- question.toLowerCase().includes('focus on') ||
4235
- question.toLowerCase().includes('redirect'))) {
4236
- console.log(`📨 [fast brain] Passing user direction to agent SDK queue`);
4237
- executeResearch(`[USER DIRECTION during active research] ${question}. The user's spec.md has been updated. Acknowledge briefly and incorporate.`);
4238
- }
4239
- return result.script;
4240
- }
4241
- catch (err) {
4242
- haikuInFlight = null;
4243
- // Voice queue items may have been held while fast brain was in flight — retry now
4244
- if (voiceQueue.length > 0) {
4245
- setTimeout(() => processVoiceQueue(), 500);
4246
- }
4247
- console.error('❌ Fast brain failed:', err);
4248
- return 'I\'m having trouble processing that. Could you try again?';
4249
- }
4250
- },
4251
- });
4252
- const respondPermissionTool = llm.tool({
4253
- description: `Respond to a permission request. Call after hearing user's response.`,
4254
- parameters: z.object({
4255
- response: z.enum(['allow', 'deny', 'always_allow']),
4256
- }),
4257
- execute: async ({ response }) => {
4258
- if (!realtimeClaudeHandler?.hasPendingPermission()) {
4259
- return 'No pending permission.';
4260
- }
4261
- const pending = realtimeClaudeHandler.getPendingPermission();
4262
- const allow = response === 'allow' || response === 'always_allow';
4263
- realtimeClaudeHandler.respondToPermission(allow);
4264
- await sendToFrontend({ type: 'permission_response', response, toolName: pending?.toolName });
4265
- return `Permission ${response} for ${pending?.toolName || 'tool'}.`;
4266
- },
4267
- });
4268
- // Instructions for realtime voice LLM
4269
- const realtimeInstructions = getRealtimeInstructions(workingDir);
4270
- // Create realtime model
4271
- const realtimeModel = createRealtimeModelFromConfig(rtConfig, realtimeInstructions);
4272
- // Create the Agent with MINIMAL tools — fast brain handles all routing
4273
- const agent = new voice.Agent({
4274
- instructions: realtimeInstructions,
4275
- llm: realtimeModel,
4276
- tools: {
4277
- ask_fast_brain: askFastBrainTool,
4278
- respond_permission: respondPermissionTool,
4279
- },
4280
- });
4281
- // Create the session
4282
- const session = new voice.AgentSession({});
4283
- return { session, agent };
4284
- }
4285
- // ============================================================
4286
3683
  // Room Event Handlers (0.9.83: registered per-session via wireRoomHandlers)
4287
3684
  // ============================================================
4288
3685
  //
@@ -4382,7 +3779,7 @@ async function main() {
4382
3779
  // subprocess BEFORE dropping the reference. See killCurrentLLM() for full context.
4383
3780
  killCurrentLLM('disconnected_cleanup');
4384
3781
  currentLLM = null;
4385
- clearFastBrainSession();
3782
+ clearPipelineFastBrainSession();
4386
3783
  clearPipelineFastBrainSession();
4387
3784
  // ── Voluntary-leave guard ──
4388
3785
  // If we left the room ON PURPOSE (user clicked leave → /leave-room, or the
@@ -4513,7 +3910,7 @@ async function main() {
4513
3910
  researchBatchTimer = null;
4514
3911
  }
4515
3912
  stopProactiveLoop();
4516
- clearFastBrainSession();
3913
+ clearPipelineFastBrainSession();
4517
3914
  clearPipelineFastBrainSession();
4518
3915
  if (activeResearch) {
4519
3916
  activeResearch.abortController.abort();
@@ -4540,8 +3937,7 @@ async function main() {
4540
3937
  }
4541
3938
  // Extract voice architecture, provider, and sessionId from participant metadata (sent by frontend)
4542
3939
  // This overrides the config file setting for per-session flexibility
4543
- let sessionVoiceMode = voiceMode; // Default to config
4544
- let sessionRealtimeProvider = realtimeConfig.provider; // Default to config
3940
+ const sessionVoiceMode = 'pipeline';
4545
3941
  let preSelectedSessionId = null;
4546
3942
  try {
4547
3943
  const metadata = JSON.parse(participant.metadata || '{}');
@@ -4555,18 +3951,6 @@ async function main() {
4555
3951
  else {
4556
3952
  currentUserId = '';
4557
3953
  }
4558
- if (metadata.voiceArch === 'realtime' || metadata.voiceArch === 'direct' || metadata.voiceArch === 'pipeline') {
4559
- sessionVoiceMode = metadata.voiceArch;
4560
- console.log(`🎙️ Using voice mode from frontend: ${sessionVoiceMode}`);
4561
- }
4562
- else if (metadata.voiceArch) {
4563
- console.log(`⚠️ Unknown voiceArch "${metadata.voiceArch}", using config: ${voiceMode}`);
4564
- }
4565
- // Read provider selection from frontend (openai or gemini)
4566
- if (metadata.provider === 'openai' || metadata.provider === 'gemini') {
4567
- sessionRealtimeProvider = metadata.provider;
4568
- console.log(`🎙️ Using provider from frontend: ${sessionRealtimeProvider}`);
4569
- }
4570
3954
  // Read pre-selected session ID from frontend (session browser selection)
4571
3955
  if (metadata.sessionId && typeof metadata.sessionId === 'string' && metadata.sessionId.length > 0) {
4572
3956
  preSelectedSessionId = metadata.sessionId;
@@ -4598,7 +3982,6 @@ async function main() {
4598
3982
  }
4599
3983
  // Sync to outer scope so DataReceived handler can use it
4600
3984
  currentVoiceMode = sessionVoiceMode;
4601
- currentProvider = sessionRealtimeProvider;
4602
3985
  // Resume session ID — only set when resuming an existing session
4603
3986
  const resumeSessionId = preSelectedSessionId || undefined;
4604
3987
  currentResumeSessionId = resumeSessionId;
@@ -4631,16 +4014,8 @@ async function main() {
4631
4014
  // Create session based on voice mode (from frontend or config)
4632
4015
  let session;
4633
4016
  let agent;
4634
- if (sessionVoiceMode === 'realtime') {
4635
- // Override the config provider with the frontend's selection
4636
- const sessionRealtimeConfig = { ...realtimeConfig, provider: sessionRealtimeProvider };
4637
- console.log(`🎙️ REALTIME MODE: ${sessionRealtimeConfig.provider} native speech-to-speech`);
4638
- const result = await createRealtimeSession(sessionRealtimeConfig, resumeSessionId);
4639
- session = result.session;
4640
- agent = result.agent;
4641
- }
4642
- else if (sessionVoiceMode === 'pipeline') {
4643
- console.log(`🎯 PIPELINE MODE: Claude SDK + parallel Gemini fast brain observer`);
4017
+ if (true) {
4018
+ console.log(`🎯 PIPELINE MODE: Claude SDK + parallel fast brain`);
4644
4019
  // Pipeline mode = direct mode underneath + parallel fast brain
4645
4020
  // Fast brain runs in PipelineDirectLLM.chat() — fires Gemini alongside Claude
4646
4021
  const { createPipelineDirectLLM } = await import('./pipeline-direct-llm.js');
@@ -4686,12 +4061,6 @@ async function main() {
4686
4061
  session = result.session;
4687
4062
  agent = result.agent;
4688
4063
  }
4689
- else {
4690
- console.log(`🎯 DIRECT MODE: Claude Agent SDK with full coding capabilities`);
4691
- const result = await createDirectSession(resumeSessionId);
4692
- session = result.session;
4693
- agent = result.agent;
4694
- }
4695
4064
  currentSession = session;
4696
4065
  currentAgent = agent; // Store for updateChatCtx() context injection
4697
4066
  // ============================================================
@@ -4779,7 +4148,7 @@ async function main() {
4779
4148
  const prev = userState;
4780
4149
  userState = ev.newState;
4781
4150
  console.log(`👤 User state: ${prev} → ${ev.newState} (agent: ${agentState})`);
4782
- if (ev.newState === 'speaking' && agentState === 'speaking' && sessionVoiceMode !== 'realtime') {
4151
+ if (ev.newState === 'speaking' && agentState === 'speaking') {
4783
4152
  // 0.9.67: action commented out, condition + debug kept.
4784
4153
  //
4785
4154
  // Why removed: in @livekit/agents 1.4.x SpeechHandle.interrupt() calls
@@ -5026,7 +4395,7 @@ async function main() {
5026
4395
  console.log('⚠️ TTS abort from user interruption — recovering session (SDK killed it internally)');
5027
4396
  }
5028
4397
  // Auto-recover from crashes in direct/pipeline mode (includes TTS abort)
5029
- if ((ev.reason === 'error' || ev.reason === 'disconnected') && (sessionVoiceMode === 'direct' || sessionVoiceMode === 'pipeline')) {
4398
+ if ((ev.reason === 'error' || ev.reason === 'disconnected')) {
5030
4399
  const now = Date.now();
5031
4400
  if (now - lastRecoveryTime < MIN_RECOVERY_INTERVAL) {
5032
4401
  console.log(`⚠️ Recovery too frequent — scheduling retry in ${MIN_RECOVERY_INTERVAL}ms`);
@@ -5071,7 +4440,7 @@ async function main() {
5071
4440
  currentLLM.stopIndexWatcher();
5072
4441
  }
5073
4442
  let result;
5074
- if (sessionVoiceMode === 'pipeline') {
4443
+ {
5075
4444
  // Pipeline mode: recreate PipelineDirectLLM wrapper with fast brain
5076
4445
  console.log('🔄 Rebuilding pipeline mode (PipelineDirectLLM + fast brain)...');
5077
4446
  const { createPipelineDirectLLM } = await import('./pipeline-direct-llm.js');
@@ -5103,9 +4472,6 @@ async function main() {
5103
4472
  });
5104
4473
  result = await createDirectSession(recoverySessionId, pipelineLLM);
5105
4474
  }
5106
- else {
5107
- result = await createDirectSession(recoverySessionId);
5108
- }
5109
4475
  const newSession = result.session;
5110
4476
  const newAgent = result.agent;
5111
4477
  currentSession = newSession;
@@ -5127,9 +4493,7 @@ async function main() {
5127
4493
  if (recoveredId) {
5128
4494
  const conversationHistory = await getConversationHistory(recoveredId, workingDir, 10);
5129
4495
  const historyForScript = conversationHistory.map(e => ({ role: e.role, text: e.content }));
5130
- const script = await prepareRecoveryScript(historyForScript);
5131
- // Direct mode: use session.say() for recovery notification
5132
- newSession.say(script, { allowInterruptions: true });
4496
+ newSession.say("I'm back after a brief interruption. What were we working on?", { allowInterruptions: true });
5133
4497
  }
5134
4498
  else {
5135
4499
  newSession.say('Voice session was briefly interrupted but I\'m back. What were we working on?', { allowInterruptions: true });
@@ -5149,81 +4513,6 @@ async function main() {
5149
4513
  }
5150
4514
  return;
5151
4515
  }
5152
- // Auto-recover from crashes in realtime mode
5153
- if (ev.reason === 'error' && sessionVoiceMode === 'realtime') {
5154
- const now = Date.now();
5155
- if (now - lastRecoveryTime < MIN_RECOVERY_INTERVAL) {
5156
- console.log('⚠️ Recovery too frequent — skipping to prevent loop');
5157
- sendToFrontend({ type: 'agent_state', state: 'error' });
5158
- return;
5159
- }
5160
- lastRecoveryTime = now;
5161
- console.log('🔄 Auto-recovering from session crash...');
5162
- // Clean up dead session
5163
- try {
5164
- sess.removeAllListeners();
5165
- }
5166
- catch { }
5167
- currentSession = null;
5168
- currentAgent = null;
5169
- // Clear voice queue — stale injections from the crashed session
5170
- voiceQueue.length = 0;
5171
- isProcessingQueue = false;
5172
- if (researchBatchTimer) {
5173
- clearTimeout(researchBatchTimer);
5174
- researchBatchTimer = null;
5175
- }
5176
- stopProactiveLoop();
5177
- if (activeResearch) {
5178
- activeResearch.abortController.abort();
5179
- activeResearch.cleanup();
5180
- activeResearch = null;
5181
- }
5182
- try {
5183
- const recoveryConfig = { ...realtimeConfig, provider: currentProvider };
5184
- // Reuse existing session ID for workspace continuity during recovery
5185
- // Prefer real SDK session ID, fall back to original resume ID
5186
- const recoverySessionId = currentLLM?.sessionId || resumeSessionId;
5187
- const result = await createRealtimeSession(recoveryConfig, recoverySessionId);
5188
- const newSession = result.session;
5189
- const newAgent = result.agent;
5190
- currentSession = newSession;
5191
- currentAgent = newAgent;
5192
- // Re-wire event listeners on the new session
5193
- wireSessionEvents(newSession, newAgent);
5194
- await newSession.start({ agent: newAgent, room: activeRoom });
5195
- // Sync state
5196
- agentState = 'listening';
5197
- sendToFrontend({ type: 'agent_state', state: 'listening' });
5198
- // Resume Claude session if one was active
5199
- if (currentLLM?.sessionId) {
5200
- currentLLM.setContinueSession(true);
5201
- }
5202
- // Generate recovery script via fast brain
5203
- const recoveredSessionId = currentLLM?.sessionId || recoverySessionId;
5204
- if (recoveredSessionId) {
5205
- try {
5206
- const conversationHistory = await getConversationHistory(recoveredSessionId, workingDir, 10);
5207
- const historyForScript = conversationHistory.map(e => ({ role: e.role, text: e.content }));
5208
- const script = await prepareRecoveryScript(historyForScript);
5209
- queueVoiceInjection(getScriptInjection(script));
5210
- console.log('📋 Injected recovery script into recovered session');
5211
- }
5212
- catch (err) {
5213
- console.log('⚠️ Failed to generate recovery script:', err);
5214
- queueVoiceInjection(getNotificationInjection('Voice session was briefly interrupted but I\'m back. What were we working on?'));
5215
- }
5216
- }
5217
- else {
5218
- queueVoiceInjection(getNotificationInjection('Voice session was briefly interrupted but I\'m back. What were we working on?'));
5219
- }
5220
- console.log('✅ Auto-recovery complete');
5221
- }
5222
- catch (err) {
5223
- console.error('❌ Auto-recovery failed:', err);
5224
- sendToFrontend({ type: 'agent_state', state: 'error' });
5225
- }
5226
- }
5227
4516
  });
5228
4517
  }
5229
4518
  // Wire events on the initial session
@@ -5238,7 +4527,7 @@ async function main() {
5238
4527
  // Send ready signal with persistent retry
5239
4528
  console.log('💓 Sending agent_ready signal...');
5240
4529
  let readySent = false;
5241
- const provider = sessionVoiceMode === 'realtime' ? realtimeConfig.provider : 'claude';
4530
+ // (realtime provider removed)
5242
4531
  // Fetch full session list for startup session browser (all Claude projects)
5243
4532
  const allSessions = await listAllClaudeSessions();
5244
4533
  const recentSessionId = allSessions.length > 0 ? allSessions[0].sessionId : null;
@@ -5259,7 +4548,7 @@ async function main() {
5259
4548
  return;
5260
4549
  await sendToFrontend({
5261
4550
  type: 'agent_ready',
5262
- provider,
4551
+ provider: 'claude',
5263
4552
  voiceMode: sessionVoiceMode,
5264
4553
  hasRecentSession,
5265
4554
  recentSessionId,
@@ -5289,13 +4578,10 @@ async function main() {
5289
4578
  // For realtime mode: use generateReply() since there's no standalone TTS
5290
4579
  // For direct mode: use say() which goes through the configured TTS
5291
4580
  const greetViaVoice = async (text) => {
5292
- if (sessionVoiceMode === 'realtime') {
5293
- // Use instructions (not userInput) to avoid system text appearing as user transcript
5294
- await session.generateReply({ instructions: getScriptInjection(text) });
5295
- }
5296
- else {
4581
+ try {
5297
4582
  await session.say(text);
5298
4583
  }
4584
+ catch { }
5299
4585
  };
5300
4586
  if (preSelectedSessionId && sessionExists(preSelectedSessionId, workingDir)) {
5301
4587
  // User pre-selected a session from the session browser — auto-resume immediately
@@ -5305,7 +4591,6 @@ async function main() {
5305
4591
  console.log(`🔄 Session resume configured: ${preSelectedSessionId}`);
5306
4592
  // Fetch context and greet with it
5307
4593
  const summary = await getSessionSummary(preSelectedSessionId, workingDir);
5308
- const conversationHistory = await getConversationHistory(preSelectedSessionId, workingDir, 30);
5309
4594
  await sendToFrontend({
5310
4595
  type: 'session_resume_set',
5311
4596
  sessionId: preSelectedSessionId,
@@ -5326,22 +4611,15 @@ async function main() {
5326
4611
  }))
5327
4612
  });
5328
4613
  }
5329
- // Generate briefing script via fast brain
4614
+ // Greet first, then point agent to session index (non-blocking)
5330
4615
  if (summary) {
5331
- loadSessionHistoryIntoChatCtx(currentAgent, conversationHistory, currentProvider);
5332
4616
  try {
5333
- if (sessionVoiceMode === 'realtime') {
5334
- const historyForScript = conversationHistory.map(e => ({ role: e.role, text: e.content }));
5335
- const script = await prepareBriefingScript(workingDir, preSelectedSessionId, historyForScript);
5336
- await session.generateReply({ instructions: getScriptInjection(script) });
5337
- }
5338
- else {
5339
- await session.say("Welcome back! Ready to continue our previous conversation.");
5340
- }
4617
+ await session.say("Welcome back! Ready to continue our previous conversation.");
5341
4618
  }
5342
4619
  catch (err) {
5343
4620
  console.log('⚠️ Pre-selected session greeting failed:', err);
5344
4621
  }
4622
+ injectSessionIndexIntoChatCtx(currentAgent, preSelectedSessionId, workingDir);
5345
4623
  }
5346
4624
  }
5347
4625
  }
@@ -5421,7 +4699,7 @@ async function main() {
5421
4699
  // persistent session keeps running tools and pushing TTS into a dead session.
5422
4700
  killCurrentLLM('participant_disconnected');
5423
4701
  currentLLM = null;
5424
- clearFastBrainSession();
4702
+ clearPipelineFastBrainSession();
5425
4703
  clearPipelineFastBrainSession();
5426
4704
  // Auto-leave path for a NON-meeting session. 0.9.83: a real session just
5427
4705
  // ended → use the FAST leave (~20s), not the 3-min alone grace. Fires even
@@ -5482,7 +4760,7 @@ async function main() {
5482
4760
  console.log(`📝 Text (${fullContent.length} chars): "${fullContent}"`);
5483
4761
  }
5484
4762
  // Skip interrupt for Gemini — disrupts state machine (hangs in speaking state)
5485
- if (currentProvider !== 'gemini') {
4763
+ if (true) { // always non-gemini in pipeline mode
5486
4764
  currentSession.interrupt();
5487
4765
  }
5488
4766
  await currentSession.generateReply({ userInput: fullContent });
@@ -5618,7 +4896,6 @@ async function main() {
5618
4896
  currentResumeSessionId = recentId;
5619
4897
  console.log(`🔄 Continuing most recent session: ${recentId}`);
5620
4898
  const summary = await getSessionSummary(recentId, workingDir);
5621
- const conversationHistory = await getConversationHistory(recentId, workingDir, 30);
5622
4899
  await sendToFrontend({
5623
4900
  type: 'session_resume_set',
5624
4901
  sessionId: recentId,
@@ -5640,21 +4917,13 @@ async function main() {
5640
4917
  });
5641
4918
  }
5642
4919
  if (currentSession && summary) {
5643
- loadSessionHistoryIntoChatCtx(currentAgent, conversationHistory, currentProvider);
5644
- console.log('📋 Injecting session context into voice agent...');
5645
4920
  try {
5646
- if (currentVoiceMode === 'realtime') {
5647
- const historyForScript = conversationHistory.map(e => ({ role: e.role, text: e.content }));
5648
- const script = await prepareBriefingScript(workingDir, recentId, historyForScript);
5649
- await currentSession.generateReply({ instructions: getScriptInjection(script) });
5650
- }
5651
- else {
5652
- await currentSession.say("Continuing where we left off.");
5653
- }
4921
+ await currentSession.say("Continuing where we left off.");
5654
4922
  }
5655
4923
  catch (err) {
5656
4924
  console.log('⚠️ Context injection failed:', err);
5657
4925
  }
4926
+ injectSessionIndexIntoChatCtx(currentAgent, recentId, workingDir);
5658
4927
  }
5659
4928
  }
5660
4929
  else {
@@ -5678,7 +4947,7 @@ async function main() {
5678
4947
  currentLLM.resetForSessionSwitch();
5679
4948
  currentLLM.setResumeSessionId(sessionId);
5680
4949
  currentResumeSessionId = sessionId;
5681
- clearFastBrainSession();
4950
+ clearPipelineFastBrainSession();
5682
4951
  clearPipelineFastBrainSession();
5683
4952
  console.log(`🔄 Switched to session: ${sessionId}`);
5684
4953
  // Step 3: Send full context to frontend (including conversation history)
@@ -5704,25 +4973,18 @@ async function main() {
5704
4973
  }))
5705
4974
  });
5706
4975
  }
5707
- // Step 4: Voice agent acknowledges context via fast brain
4976
+ // Step 4: Voice agent acknowledges context — greet first, point to index after
5708
4977
  if (currentSession && summary) {
5709
- loadSessionHistoryIntoChatCtx(currentAgent, conversationHistory, currentProvider);
5710
4978
  try {
5711
- if (currentVoiceMode === 'realtime') {
5712
- const historyForScript = conversationHistory.map(e => ({ role: e.role, text: e.content }));
5713
- const briefingScript = await prepareBriefingScript(workingDir, sessionId, historyForScript, 'switch');
5714
- queueVoiceInjection(getScriptInjection(briefingScript));
5715
- }
5716
- else {
5717
- const acknowledgment = summary.lastMessages.length > 0
5718
- ? `I've switched to your previous session. You were working on: ${summary.lastMessages[summary.lastMessages.length - 1]?.substring(0, 100)}`
5719
- : `Switched to previous session with ${summary.messageCount} messages. What would you like to continue with?`;
5720
- await currentSession.say(acknowledgment);
5721
- }
4979
+ const acknowledgment = summary.lastMessages.length > 0
4980
+ ? `I've switched to your previous session. You were working on: ${summary.lastMessages[summary.lastMessages.length - 1]?.substring(0, 100)}`
4981
+ : `Switched to previous session with ${summary.messageCount} messages. What would you like to continue with?`;
4982
+ await currentSession.say(acknowledgment);
5722
4983
  }
5723
4984
  catch (err) {
5724
4985
  console.log('⚠️ Switch acknowledgment failed:', err);
5725
4986
  }
4987
+ injectSessionIndexIntoChatCtx(currentAgent, sessionId, workingDir);
5726
4988
  }
5727
4989
  }
5728
4990
  else {
@@ -6240,7 +5502,6 @@ async function main() {
6240
5502
  console.log(`🔄 Resuming session: ${sessionId}`);
6241
5503
  // Fetch context and greet with it
6242
5504
  const summary = await getSessionSummary(sessionId, workingDir);
6243
- const conversationHistory = await getConversationHistory(sessionId, workingDir, 30);
6244
5505
  await sendToFrontend({
6245
5506
  type: 'session_resume_set',
6246
5507
  sessionId,
@@ -6261,14 +5522,10 @@ async function main() {
6261
5522
  }))
6262
5523
  });
6263
5524
  }
6264
- // RESUME meeting-context fix (2026-08-05): a resumed session that
6265
- // previously ran a meeting has NO live meeting — the in-memory bot +
6266
- // poller reset on process/session start. But the LLM would infer
6267
- // "still in a meeting" from the replayed [MEETING —] lines + notes and
6268
- // refuse to leave (user hit exactly this: "why does it think we're in
6269
- // the meeting?"). Detect meeting history in THIS session and tell the
6270
- // LLM the meeting has ended so it behaves as a normal voice assistant.
6271
- const hadMeeting = conversationHistory.some(e => /\[MEETING\b|now in a meeting|Recall bot ID/i.test(e.content || ''));
5525
+ // RESUME meeting-context fix: lightweight check (5 exchanges max) to detect
5526
+ // prior meeting history and tell the LLM the meeting has already ended.
5527
+ const meetingCheckHistory = await getConversationHistory(sessionId, workingDir, 5);
5528
+ const hadMeeting = meetingCheckHistory.some(e => /\[MEETING\b|now in a meeting|Recall bot ID/i.test(e.content || ''));
6272
5529
  if (hadMeeting && currentLLM) {
6273
5530
  try {
6274
5531
  const endedCtx = new llm.ChatContext();
@@ -6280,22 +5537,15 @@ async function main() {
6280
5537
  console.warn('⚠️ meeting-ended injection failed:', e.message);
6281
5538
  }
6282
5539
  }
6283
- // Load full session history and greet with context via fast brain
5540
+ // Greet immediately, then point agent to session index (no raw JSONL loading)
6284
5541
  if (currentSession && summary) {
6285
- loadSessionHistoryIntoChatCtx(currentAgent, conversationHistory, currentProvider);
6286
5542
  try {
6287
- if (currentVoiceMode === 'realtime') {
6288
- const historyForScript = conversationHistory.map(e => ({ role: e.role, text: e.content }));
6289
- const briefingScript = await prepareBriefingScript(workingDir, sessionId, historyForScript, 'resume');
6290
- queueVoiceInjection(getScriptInjection(briefingScript));
6291
- }
6292
- else {
6293
- await currentSession.say("Welcome back! Ready to continue our previous conversation.");
6294
- }
5543
+ await currentSession.say("Welcome back! Ready to continue our previous conversation.");
6295
5544
  }
6296
5545
  catch (err) {
6297
5546
  console.log('⚠️ Session gate greeting failed:', err);
6298
5547
  }
5548
+ injectSessionIndexIntoChatCtx(currentAgent, sessionId, workingDir);
6299
5549
  }
6300
5550
  }
6301
5551
  else {
@@ -6304,12 +5554,7 @@ async function main() {
6304
5554
  console.log('🆕 Starting fresh session');
6305
5555
  if (currentSession) {
6306
5556
  try {
6307
- if (currentVoiceMode === 'realtime') {
6308
- queueVoiceInjection(getScriptInjection("Hey! I'm Osborn, your AI research assistant. What are you working on today?"));
6309
- }
6310
- else {
6311
- await currentSession.say("Hey! I'm Osborn. What are you working on?");
6312
- }
5557
+ await currentSession.say("Hey! I'm Osborn. What are you working on?");
6313
5558
  }
6314
5559
  catch (err) {
6315
5560
  console.log('⚠️ Fresh session greeting failed:', err);