osborn 0.9.146 → 0.9.148

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -210,6 +210,12 @@ export declare class ClaudeLLM extends llm.LLM {
210
210
  setActiveQuery(q: any): void;
211
211
  /** Remove an active query (called from ClaudeLLMStream when query completes) */
212
212
  removeActiveQuery(q: any): void;
213
+ /**
214
+ * Stop a single dispatch flow by agent_id.
215
+ * Returns true if the agent was found and stopped; false if not found.
216
+ * Does NOT affect other active queries (use abortQuery() to kill all).
217
+ */
218
+ stopAgent(agentId: string): boolean;
213
219
  /** Whether a persistent session is alive and consuming messages */
214
220
  hasSession(): boolean;
215
221
  /**
@@ -552,10 +552,17 @@ export class ClaudeLLM extends llm.LLM {
552
552
  // No JSONL replay after the first cold start.
553
553
  #persistentQuery = null;
554
554
  #messageChannel = null;
555
+ // Read-along tracking — per-turn message ID and chunk counter for TTS highlighting
556
+ #currentTurnMessageId = null;
557
+ #currentTurnChunkIndex = 0;
558
+ #currentTurnChunks = [];
555
559
  #backgroundConsumerRunning = false;
556
560
  // Active queries — multiple can be running (SDK queues them internally).
557
561
  // We keep ALL references so interrupt() can stop whatever is currently executing.
558
562
  #activeQueries = new Set();
563
+ // Per-agent-id query map — allows targeted stop of a single dispatch flow.
564
+ // Strictly additive; abortQuery/interruptQuery still iterate #activeQueries (kill-all).
565
+ #activeQueriesById = new Map();
559
566
  // Dedup guard — prevents double-firing reviewer/gate if SubagentStop fires
560
567
  // more than once for the same agent_id (e.g. retry edge cases).
561
568
  #dispatchedFor = new Set();
@@ -889,6 +896,7 @@ export class ClaudeLLM extends llm.LLM {
889
896
  catch { }
890
897
  }
891
898
  this.#activeQueries.clear();
899
+ this.#activeQueriesById.clear();
892
900
  console.log('🛑 All queries aborted (Ctrl+C equivalent)');
893
901
  }
894
902
  /**
@@ -943,6 +951,23 @@ export class ClaudeLLM extends llm.LLM {
943
951
  removeActiveQuery(q) {
944
952
  this.#activeQueries.delete(q);
945
953
  }
954
+ /**
955
+ * Stop a single dispatch flow by agent_id.
956
+ * Returns true if the agent was found and stopped; false if not found.
957
+ * Does NOT affect other active queries (use abortQuery() to kill all).
958
+ */
959
+ stopAgent(agentId) {
960
+ const q = this.#activeQueriesById.get(agentId);
961
+ if (!q)
962
+ return false;
963
+ try {
964
+ q.return?.();
965
+ }
966
+ catch { }
967
+ this.#activeQueriesById.delete(agentId);
968
+ this.#activeQueries.delete(q);
969
+ return true;
970
+ }
946
971
  // ============================================================
947
972
  // PERSISTENT SESSION — V1 query() with AsyncIterable<SDKUserMessage>
948
973
  // Single subprocess per voice session. First chat() does JSONL cold
@@ -1085,22 +1110,41 @@ export class ClaudeLLM extends llm.LLM {
1085
1110
  }
1086
1111
  // Stream assistant text → tts_say events
1087
1112
  if (msg.type === 'assistant' && msg.message?.content) {
1113
+ // Assign a stable messageId for this turn (first block sets it, rest reuse)
1114
+ if (!this.#currentTurnMessageId) {
1115
+ this.#currentTurnMessageId = crypto.randomUUID();
1116
+ this.#currentTurnChunkIndex = 0;
1117
+ this.#currentTurnChunks = [];
1118
+ }
1119
+ const turnMessageId = this.#currentTurnMessageId;
1088
1120
  for (const block of msg.message.content) {
1089
1121
  if (block.type === 'text' && block.text) {
1090
- callbacks.eventEmitter.emit('assistant_text', { text: block.text });
1122
+ const chunkIndex = this.#currentTurnChunkIndex;
1123
+ callbacks.eventEmitter.emit('assistant_text', { text: block.text, messageId: turnMessageId, chunkIndex });
1091
1124
  const ttsChunk = stripMarkdownForTTS(block.text);
1092
1125
  if (ttsChunk.trim()) {
1126
+ this.#currentTurnChunks.push(ttsChunk);
1127
+ this.#currentTurnChunkIndex++;
1093
1128
  console.log(`🔊 TTS say (${ttsChunk.length} chars): "${ttsChunk}"`);
1094
- callbacks.eventEmitter.emit('tts_say', { text: ttsChunk });
1129
+ callbacks.eventEmitter.emit('tts_say', { text: ttsChunk, messageId: turnMessageId, chunkIndex });
1095
1130
  }
1096
1131
  }
1097
1132
  }
1098
1133
  }
1099
1134
  // Result — marks end of a turn (but we keep consuming for next turn)
1100
1135
  if (msg.type === 'result') {
1136
+ const turnMessageId = this.#currentTurnMessageId;
1137
+ const turnChunks = [...this.#currentTurnChunks];
1101
1138
  if (msg.result) {
1102
- callbacks.eventEmitter.emit('assistant_result', { text: msg.result });
1139
+ callbacks.eventEmitter.emit('assistant_result', { text: msg.result, messageId: turnMessageId });
1140
+ }
1141
+ if (turnMessageId && turnChunks.length > 0) {
1142
+ callbacks.eventEmitter.emit('tts_chunks', { messageId: turnMessageId, chunks: turnChunks });
1103
1143
  }
1144
+ // Reset per-turn state for next turn
1145
+ this.#currentTurnMessageId = null;
1146
+ this.#currentTurnChunkIndex = 0;
1147
+ this.#currentTurnChunks = [];
1104
1148
  console.log('✅ Claude turn complete (persistent session stays alive)');
1105
1149
  }
1106
1150
  }
@@ -1200,6 +1244,7 @@ export class ClaudeLLM extends llm.LLM {
1200
1244
  console.log(`[DISPATCH] spawning reviewer for agentId=${agentId.slice(0, 8)}`);
1201
1245
  const reviewerQuery = query({ prompt, options: reviewerOptions });
1202
1246
  this.#activeQueries.add(reviewerQuery);
1247
+ this.#activeQueriesById.set(agentId, reviewerQuery);
1203
1248
  let reviewerText = '';
1204
1249
  try {
1205
1250
  for await (const msg of reviewerQuery) {
@@ -1211,6 +1256,7 @@ export class ClaudeLLM extends llm.LLM {
1211
1256
  }
1212
1257
  finally {
1213
1258
  this.#activeQueries.delete(reviewerQuery);
1259
+ this.#activeQueriesById.delete(agentId);
1214
1260
  }
1215
1261
  const verdictMatch = reviewerText.match(/VERDICT:\s*(ACCEPT|REJECT)/i);
1216
1262
  const verdict = verdictMatch ? verdictMatch[1].toUpperCase() : null;
@@ -1284,6 +1330,7 @@ export class ClaudeLLM extends llm.LLM {
1284
1330
  console.log(`[DISPATCH] spawning research-gate for agentId=${agentId.slice(0, 8)}`);
1285
1331
  const gateQuery = query({ prompt, options: gateOptions });
1286
1332
  this.#activeQueries.add(gateQuery);
1333
+ this.#activeQueriesById.set(agentId, gateQuery);
1287
1334
  let review = '';
1288
1335
  try {
1289
1336
  for await (const msg of gateQuery) {
@@ -1295,6 +1342,7 @@ export class ClaudeLLM extends llm.LLM {
1295
1342
  }
1296
1343
  finally {
1297
1344
  this.#activeQueries.delete(gateQuery);
1345
+ this.#activeQueriesById.delete(agentId);
1298
1346
  }
1299
1347
  const gateMatch = review.match(/GATE:\s*(PASS|NEEDS-MORE)/i);
1300
1348
  const gateVerdict = gateMatch ? gateMatch[1].toUpperCase() : null;
@@ -1781,6 +1829,7 @@ class ClaudeLLMStream extends llm.LLMStream {
1781
1829
  matcher: '.*',
1782
1830
  hooks: [async (input) => {
1783
1831
  console.log('[LIFECYCLE-PROBE] SubagentStart', JSON.stringify(input));
1832
+ this.#eventEmitter.emit('agent_started', { agent_type: input?.agent_type, agent_id: input?.agent_id });
1784
1833
  return {};
1785
1834
  }]
1786
1835
  }],
@@ -1792,6 +1841,7 @@ class ClaudeLLMStream extends llm.LLMStream {
1792
1841
  const msg = String(input?.last_assistant_message ?? '');
1793
1842
  const aid = input?.agent_id ?? ('sa-' + Date.now());
1794
1843
  statusManager.upsertDispatch(aid, { subagentType: at, dispatchState: 'completed', artifact: msg });
1844
+ this.#eventEmitter.emit('task_completed', { agent_type: at, agent_id: aid, last_assistant_message: String(msg).slice(0, 400) });
1795
1845
  // Infinite-loop guard — never re-dispatch the reviewer or reasoner.
1796
1846
  if (at === 'reviewer' || at === 'reasoner')
1797
1847
  return {};
@@ -1827,6 +1877,10 @@ class ClaudeLLMStream extends llm.LLMStream {
1827
1877
  // Run Claude Agent SDK query() and stream results
1828
1878
  let hasOutput = false;
1829
1879
  let fullResponse = ''; // Collect full response for frontend
1880
+ // Per-turn read-along tracking (non-skipTTSQueue path)
1881
+ let streamTurnMessageId = null;
1882
+ let streamTurnChunkIndex = 0;
1883
+ let streamTurnChunks = [];
1830
1884
  // DIRECT MODE OPTIMIZATION: When skipTTSQueue is true, we run the Claude query
1831
1885
  // in the background and return from run() immediately. This is critical because:
1832
1886
  //
@@ -1913,20 +1967,29 @@ class ClaudeLLMStream extends llm.LLMStream {
1913
1967
  if (sdkRequestId) {
1914
1968
  this.#eventEmitter.emit('query_request_id', { requestId: sdkRequestId });
1915
1969
  }
1970
+ // Assign a stable messageId for this turn (first block sets it, rest reuse)
1971
+ if (!streamTurnMessageId) {
1972
+ streamTurnMessageId = crypto.randomUUID();
1973
+ streamTurnChunkIndex = 0;
1974
+ streamTurnChunks = [];
1975
+ }
1916
1976
  for (const block of message.message.content) {
1917
1977
  if (block.type === 'text' && block.text) {
1918
1978
  hasOutput = true;
1919
1979
  const rawText = block.text;
1980
+ const chunkIndex = streamTurnChunkIndex;
1920
1981
  // Emit RAW text to frontend (for chat bubbles with full formatting)
1921
- this.#eventEmitter.emit('assistant_text', { text: rawText });
1982
+ this.#eventEmitter.emit('assistant_text', { text: rawText, messageId: streamTurnMessageId, chunkIndex });
1922
1983
  // Strip markdown for clean speech
1923
1984
  const ttsChunk = stripMarkdownForTTS(rawText);
1924
1985
  if (ttsChunk.trim()) {
1986
+ streamTurnChunks.push(ttsChunk);
1987
+ streamTurnChunkIndex++;
1925
1988
  if (this.#opts.skipTTSQueue) {
1926
1989
  // Direct mode: emit event for session.say() — bypasses LiveKit's
1927
1990
  // BufferedTokenStream which causes stuck/delayed/out-of-order audio
1928
1991
  console.log(`🔊 TTS say (${ttsChunk.length} chars): "${ttsChunk}"`);
1929
- this.#eventEmitter.emit('tts_say', { text: ttsChunk });
1992
+ this.#eventEmitter.emit('tts_say', { text: ttsChunk, messageId: streamTurnMessageId, chunkIndex });
1930
1993
  }
1931
1994
  else {
1932
1995
  // Realtime mode: use LLM stream queue (framework handles TTS)
@@ -1944,7 +2007,15 @@ class ClaudeLLMStream extends llm.LLMStream {
1944
2007
  if (message.type === 'result' && message.result) {
1945
2008
  const rawResult = message.result;
1946
2009
  // Emit RAW result to frontend
1947
- this.#eventEmitter.emit('assistant_result', { text: rawResult });
2010
+ this.#eventEmitter.emit('assistant_result', { text: rawResult, messageId: streamTurnMessageId });
2011
+ // Emit ordered chunk list for frontend read-along
2012
+ if (streamTurnMessageId && streamTurnChunks.length > 0) {
2013
+ this.#eventEmitter.emit('tts_chunks', { messageId: streamTurnMessageId, chunks: streamTurnChunks });
2014
+ }
2015
+ // Reset per-turn state
2016
+ streamTurnMessageId = null;
2017
+ streamTurnChunkIndex = 0;
2018
+ streamTurnChunks = [];
1948
2019
  if (!hasOutput) {
1949
2020
  hasOutput = true;
1950
2021
  const ttsText = stripMarkdownForTTS(rawResult);
package/dist/index.js CHANGED
@@ -2814,6 +2814,8 @@ async function main() {
2814
2814
  text: data.text,
2815
2815
  isStreaming: true,
2816
2816
  agentRole: 'direct',
2817
+ messageId: data.messageId,
2818
+ chunkIndex: data.chunkIndex,
2817
2819
  });
2818
2820
  });
2819
2821
  // Wire up Claude final result - RAW result goes to frontend
@@ -2825,6 +2827,15 @@ async function main() {
2825
2827
  isStreaming: false,
2826
2828
  isFinal: true,
2827
2829
  agentRole: 'direct',
2830
+ messageId: data.messageId,
2831
+ });
2832
+ });
2833
+ // Wire up ordered TTS chunk list — emitted once per turn for read-along
2834
+ directLLM.events.on('tts_chunks', (data) => {
2835
+ sendToFrontend({
2836
+ type: 'tts_chunks',
2837
+ messageId: data.messageId,
2838
+ chunks: data.chunks,
2828
2839
  });
2829
2840
  });
2830
2841
  // Wire up permission requests - sends to frontend for user approval
@@ -3001,6 +3012,15 @@ async function main() {
3001
3012
  playbackStartedAt = Date.now();
3002
3013
  console.log(`🔊 [${sayId}] audio first frame out (playbackStarted)`);
3003
3014
  audioOutputRef.off('playbackStarted', onPlaybackStarted);
3015
+ // Notify frontend that this TTS chunk is now audibly playing
3016
+ if (data.messageId != null && data.chunkIndex != null) {
3017
+ sendToFrontend({
3018
+ type: 'tts_chunk_playing',
3019
+ messageId: data.messageId,
3020
+ chunkIndex: data.chunkIndex,
3021
+ text: data.text,
3022
+ });
3023
+ }
3004
3024
  };
3005
3025
  audioOutputRef.on('playbackStarted', onPlaybackStarted);
3006
3026
  }
@@ -3096,6 +3116,19 @@ async function main() {
3096
3116
  review: data.review,
3097
3117
  });
3098
3118
  });
3119
+ // Dispatcher v1 — sub-agent completed → frontend (Feature A)
3120
+ directLLM.events.on('task_completed', (d) => sendToFrontend({
3121
+ type: 'task_completed',
3122
+ agent_type: d.agent_type,
3123
+ agent_id: d.agent_id,
3124
+ last_assistant_message: d.last_assistant_message,
3125
+ }));
3126
+ // Dispatcher v1 — sub-agent started → frontend stop-key signal (Feature A2)
3127
+ directLLM.events.on('agent_started', (d) => sendToFrontend({
3128
+ type: 'agent_started',
3129
+ agent_type: d.agent_type,
3130
+ agent_id: d.agent_id,
3131
+ }));
3099
3132
  // Create the Agent with instructions, STT, LLM, TTS
3100
3133
  // VAD (Silero ONNX) removed — caused 2-5s inference lag on CPU, making interruption detection worse
3101
3134
  // Turn detection is server-side (Deepgram endpointing), interruptions handled by STT
@@ -5515,6 +5548,10 @@ async function main() {
5515
5548
  }
5516
5549
  }
5517
5550
  }
5551
+ // Feature B — per-flow stop (does NOT affect other running flows)
5552
+ else if (data.type === 'stop_dispatch' && currentLLM) {
5553
+ currentLLM.stopAgent?.(String(data.agentId));
5554
+ }
5518
5555
  else if (data.type === 'join_meeting') {
5519
5556
  const meetingUrl = data.url;
5520
5557
  if (meetingUrl) {
@@ -67,6 +67,7 @@ export declare class PipelineDirectLLM extends llm.LLM {
67
67
  abortAgent(): void;
68
68
  rewindAgent(checkpointId?: string): Promise<boolean>;
69
69
  hasActiveAgent(): boolean;
70
+ stopAgent(agentId: string): boolean;
70
71
  /** Send a new prompt to Claude via direct chat() — event listeners stay attached */
71
72
  sendPrompt(prompt: string): void;
72
73
  enableMcpServer(k: string, c: any): void;
@@ -120,6 +120,7 @@ export class PipelineDirectLLM extends llm.LLM {
120
120
  abortAgent() { this.#claudeLLM.abortQuery(); }
121
121
  async rewindAgent(checkpointId) { return this.#claudeLLM.rewindToCheckpoint(checkpointId); }
122
122
  hasActiveAgent() { return this.#claudeLLM.hasActiveQuery(); }
123
+ stopAgent(agentId) { return this.#claudeLLM.stopAgent(agentId); }
123
124
  /** Send a new prompt to Claude via direct chat() — event listeners stay attached */
124
125
  sendPrompt(prompt) {
125
126
  console.log(`📋 [pipeline] Sending prompt to Claude (${prompt.length} chars)`);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.146",
3
+ "version": "0.9.148",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {