osborn 0.9.106 → 0.9.108

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +51 -5
  2. package/package.json +1 -1
package/dist/index.js CHANGED
@@ -570,11 +570,45 @@ function startApiServer(workingDir, port) {
570
570
  // speechSynthesis is NOT captured by Recall, a media element IS.
571
571
  if (req.method === 'GET' && url.pathname === '/tts') {
572
572
  const text = (url.searchParams.get('text') || '').slice(0, 4000);
573
+ if (!text) {
574
+ res.writeHead(400, { 'Content-Type': 'application/json' });
575
+ res.end(JSON.stringify({ error: 'no text' }));
576
+ return;
577
+ }
578
+ const t0 = Date.now();
579
+ // Deepgram FIRST (2026-08-01 latency fix): same TTS family the regular
580
+ // voice pipeline uses — ~3-6x faster than the OpenAI full-file synth this
581
+ // endpoint used before (measured 2-4s of the meeting reply lag). OpenAI
582
+ // stays as fallback.
583
+ const dgKey = process.env.DEEPGRAM_API_KEY;
584
+ if (dgKey) {
585
+ try {
586
+ // WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
587
+ // trailing silence + boundary click) — back-to-back sentence clips
588
+ // were heard as "cracking". WAV is gapless-safe and cheaper to decode.
589
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=linear16&sample_rate=24000&container=wav', {
590
+ method: 'POST',
591
+ headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
592
+ body: JSON.stringify({ text }),
593
+ });
594
+ if (dg.ok) {
595
+ const buf = Buffer.from(await dg.arrayBuffer());
596
+ console.log(`🗣️ /tts deepgram wav ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
597
+ res.writeHead(200, { 'Content-Type': 'audio/wav', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
598
+ res.end(buf);
599
+ return;
600
+ }
601
+ console.warn(`⚠️ /tts deepgram ${dg.status} — falling back to OpenAI`);
602
+ }
603
+ catch (e) {
604
+ console.warn(`⚠️ /tts deepgram error: ${e.message} — falling back to OpenAI`);
605
+ }
606
+ }
573
607
  const voice = url.searchParams.get('voice') || 'alloy';
574
608
  const key = process.env.OPENAI_API_KEY;
575
- if (!text || !key) {
609
+ if (!key) {
576
610
  res.writeHead(400, { 'Content-Type': 'application/json' });
577
- res.end(JSON.stringify({ error: !key ? 'no OPENAI_API_KEY' : 'no text' }));
611
+ res.end(JSON.stringify({ error: 'no TTS provider keys' }));
578
612
  return;
579
613
  }
580
614
  try {
@@ -590,6 +624,7 @@ function startApiServer(workingDir, port) {
590
624
  return;
591
625
  }
592
626
  const buf = Buffer.from(await tts.arrayBuffer());
627
+ console.log(`🗣️ /tts openai ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
593
628
  res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
594
629
  res.end(buf);
595
630
  }
@@ -1641,8 +1676,12 @@ async function main() {
1641
1676
  // reply entirely) — measured as most of the 5-8s reply lag.
1642
1677
  meetingAddressedUntil = Date.now() + 90_000;
1643
1678
  }
1679
+ // PROMPT PARITY (user directive 2026-08-01): addressed turns carry ONLY a
1680
+ // minimal tag — no behavioral re-instruction. The agent replies exactly as
1681
+ // it would to a regular voice turn; the tts_say→canvas redirect handles
1682
+ // where the words go. Same agent, same behavior, both fronts.
1644
1683
  const header = addressed
1645
- ? `[MEETING — ${botId}] — YOU WERE ADDRESSED. Reply now in PLAIN TEXT — your words are spoken into the meeting automatically (do NOT use Bash/curl to speak). One or two short conversational sentences first; then, if needed, delegate notes/research in the background.`
1684
+ ? `[MEETING — ${botId}] (addressed — reply is spoken into the meeting):`
1646
1685
  : `[MEETING — ${botId}]:`;
1647
1686
  // Prepend + consume any interruption context (bot was cut off mid-sentence).
1648
1687
  const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
@@ -2511,8 +2550,15 @@ async function main() {
2511
2550
  // (reuse of the regular speak path; no Bash roundtrip). Silent-observer
2512
2551
  // turns stay suppressed as before.
2513
2552
  if (activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2514
- console.log(`🔊➡️📽️ tts_say → canvas (addressed meeting turn): "${data.text.slice(0, 60)}"`);
2515
- pushCanvas({ kind: 'say', text: data.text });
2553
+ // Split into AT MOST 2 chunks: first sentence (fast first-audio) +
2554
+ // the remainder as ONE chunk. Per-sentence files created an audible
2555
+ // seam at every boundary (mp3 padding + scheduling jitter — heard as
2556
+ // "cracking"); two chunks keeps the latency win with one seam max.
2557
+ const m = data.text.match(/^([^.!?]+[.!?]+["']?)\s*([\s\S]*)$/);
2558
+ const chunks = m && m[2]?.trim() ? [m[1].trim(), m[2].trim()] : [data.text.trim()];
2559
+ console.log(`🔊➡️📽️ tts_say → canvas (${chunks.length} chunk(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2560
+ for (const s of chunks)
2561
+ pushCanvas({ kind: 'say', text: s });
2516
2562
  markMeetingSpeaking(data.text);
2517
2563
  return;
2518
2564
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.106",
3
+ "version": "0.9.108",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {