osborn 0.9.105 → 0.9.107

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -91,16 +91,19 @@ When a chunk says **`YOU WERE ADDRESSED`**, people in the meeting are WAITING
91
91
  for your voice. Measured failure (2026-08-01): notes-writing inside the reply
92
92
  turn pushed speech→reply past 30 seconds ("very very delayed").
93
93
 
94
- **Mandatory turn shape when addressed:**
95
- 1. **FIRST tool call = the `/canvas say` POST.** One short spoken reply (1–2
96
- sentences, conversational). Nothing runs before it — no Read, no Edit, no
97
- transcript pull, no sub-agent.
94
+ **Mandatory turn shape when addressed (0.9.106+):**
95
+ 1. **Just ANSWER in plain text — your words are spoken into the meeting
96
+ automatically.** The regular voice pipeline is redirected to the meeting
97
+ for addressed turns; do NOT use Bash/curl to speak (that adds a slow tool
98
+ roundtrip and can eat your tool budget). 1–2 short conversational
99
+ sentences, text first, before any tool call.
98
100
  2. **THEN** delegate note-taking / research to the writer/researcher
99
101
  sub-agents in the background as usual.
100
- 3. If you genuinely need a fact before answering, say a holding line FIRST
101
- ("Good question — one second while I check"), then look it up, then follow
102
- up with a second `/canvas say`. Never leave the room in silence while you
103
- work.
102
+ 3. If you genuinely need a fact before answering, SAY a holding line first
103
+ ("Good question — one second while I check"), look it up, then continue in
104
+ plain text. Never leave the room in silence while you work.
105
+ (The `/canvas say` curl remains for NON-addressed cases: when the voice-native
106
+ user says "tell the meeting X", or proactive announcements.)
104
107
 
105
108
  ## Browse requests while CASTING a live stream — drive the ENGINE, not the canvas
106
109
 
package/dist/index.js CHANGED
@@ -570,11 +570,42 @@ function startApiServer(workingDir, port) {
570
570
  // speechSynthesis is NOT captured by Recall, a media element IS.
571
571
  if (req.method === 'GET' && url.pathname === '/tts') {
572
572
  const text = (url.searchParams.get('text') || '').slice(0, 4000);
573
+ if (!text) {
574
+ res.writeHead(400, { 'Content-Type': 'application/json' });
575
+ res.end(JSON.stringify({ error: 'no text' }));
576
+ return;
577
+ }
578
+ const t0 = Date.now();
579
+ // Deepgram FIRST (2026-08-01 latency fix): same TTS family the regular
580
+ // voice pipeline uses — ~3-6x faster than the OpenAI full-file synth this
581
+ // endpoint used before (measured 2-4s of the meeting reply lag). OpenAI
582
+ // stays as fallback.
583
+ const dgKey = process.env.DEEPGRAM_API_KEY;
584
+ if (dgKey) {
585
+ try {
586
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3', {
587
+ method: 'POST',
588
+ headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
589
+ body: JSON.stringify({ text }),
590
+ });
591
+ if (dg.ok) {
592
+ const buf = Buffer.from(await dg.arrayBuffer());
593
+ console.log(`🗣️ /tts deepgram ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
594
+ res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
595
+ res.end(buf);
596
+ return;
597
+ }
598
+ console.warn(`⚠️ /tts deepgram ${dg.status} — falling back to OpenAI`);
599
+ }
600
+ catch (e) {
601
+ console.warn(`⚠️ /tts deepgram error: ${e.message} — falling back to OpenAI`);
602
+ }
603
+ }
573
604
  const voice = url.searchParams.get('voice') || 'alloy';
574
605
  const key = process.env.OPENAI_API_KEY;
575
- if (!text || !key) {
606
+ if (!key) {
576
607
  res.writeHead(400, { 'Content-Type': 'application/json' });
577
- res.end(JSON.stringify({ error: !key ? 'no OPENAI_API_KEY' : 'no text' }));
608
+ res.end(JSON.stringify({ error: 'no TTS provider keys' }));
578
609
  return;
579
610
  }
580
611
  try {
@@ -590,6 +621,7 @@ function startApiServer(workingDir, port) {
590
621
  return;
591
622
  }
592
623
  const buf = Buffer.from(await tts.arrayBuffer());
624
+ console.log(`🗣️ /tts openai ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
593
625
  res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
594
626
  res.end(buf);
595
627
  }
@@ -1617,6 +1649,10 @@ async function main() {
1617
1649
  // gone forever, matching the skill-catalog pattern.
1618
1650
  let userRemovedAgents = [];
1619
1651
  let activeMeetingBotId = null; // Recall.ai bot ID if in a meeting
1652
+ // While set (Date.now() < value), the agent's streaming tts_say output is
1653
+ // REDIRECTED into the meeting canvas instead of suppressed — the regular
1654
+ // voice pipeline reused with the meeting as the sink (latency fix).
1655
+ let meetingAddressedUntil = 0;
1620
1656
  let activeMeetingPoller = null; // Transcript poller bound to that bot
1621
1657
  // LIVE meeting transcript → LLM (buffered webhook finals). See recall.on('transcript').
1622
1658
  const meetingTranscriptBuffer = [];
@@ -1628,8 +1664,17 @@ async function main() {
1628
1664
  if (!meetingTranscriptBuffer.length || !currentLLM)
1629
1665
  return;
1630
1666
  const turns = meetingTranscriptBuffer.splice(0); // drain
1667
+ if (addressed) {
1668
+ // REUSE the regular voice pipeline (2026-08-01 latency fix): during an
1669
+ // addressed turn the agent's normal streaming tts_say output is REDIRECTED
1670
+ // to the canvas (see the tts_say handler) — same speak path as regular
1671
+ // mode, sink = meeting. The old prompt made the agent compose a Bash curl
1672
+ // (extra LLM tool roundtrip, ate the 3-call budget, sometimes lost the
1673
+ // reply entirely) — measured as most of the 5-8s reply lag.
1674
+ meetingAddressedUntil = Date.now() + 90_000;
1675
+ }
1631
1676
  const header = addressed
1632
- ? `[MEETING — ${botId}] — YOU WERE ADDRESSED. Respond OUT LOUD into the meeting now: POST http://localhost:${apiPort}/canvas {"kind":"say","text":"..."} with a short, direct reply. Then note it.`
1677
+ ? `[MEETING — ${botId}] — YOU WERE ADDRESSED. Reply now in PLAIN TEXT — your words are spoken into the meeting automatically (do NOT use Bash/curl to speak). One or two short conversational sentences first; then, if needed, delegate notes/research in the background.`
1633
1678
  : `[MEETING — ${botId}]:`;
1634
1679
  // Prepend + consume any interruption context (bot was cut off mid-sentence).
1635
1680
  const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
@@ -2494,6 +2539,21 @@ async function main() {
2494
2539
  // re-captures in the same room → feedback). Set by PipelineDirectLLM.chat()
2495
2540
  // when the turn is a [MEETING —] chunk. Normal user turns are unaffected.
2496
2541
  if (directLLM.suppressMeetingTTS) {
2542
+ // ADDRESSED turn → REDIRECT the normal streaming reply into the meeting
2543
+ // (reuse of the regular speak path; no Bash roundtrip). Silent-observer
2544
+ // turns stay suppressed as before.
2545
+ if (activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2546
+ // Sentence-split so the FIRST sentence synthesizes + plays while the
2547
+ // rest queue behind it (canvas plays says sequentially + prefetches)
2548
+ // — first-audio latency = one short sentence's synth, not the whole
2549
+ // reply's.
2550
+ const sentences = data.text.match(/[^.!?]+[.!?]+["']?|[^.!?]+$/g)?.map((s) => s.trim()).filter(Boolean) || [data.text];
2551
+ console.log(`🔊➡️📽️ tts_say → canvas (${sentences.length} sentence(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2552
+ for (const s of sentences)
2553
+ pushCanvas({ kind: 'say', text: s });
2554
+ markMeetingSpeaking(data.text);
2555
+ return;
2556
+ }
2497
2557
  console.log(`🔇 tts_say suppressed (meeting turn — response goes to /canvas, not browser): "${data.text.slice(0, 60)}"`);
2498
2558
  return;
2499
2559
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.105",
3
+ "version": "0.9.107",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {