osborn 0.9.106 → 0.9.107

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +42 -4
  2. package/package.json +1 -1
package/dist/index.js CHANGED
@@ -570,11 +570,42 @@ function startApiServer(workingDir, port) {
570
570
  // speechSynthesis is NOT captured by Recall, a media element IS.
571
571
  if (req.method === 'GET' && url.pathname === '/tts') {
572
572
  const text = (url.searchParams.get('text') || '').slice(0, 4000);
573
+ if (!text) {
574
+ res.writeHead(400, { 'Content-Type': 'application/json' });
575
+ res.end(JSON.stringify({ error: 'no text' }));
576
+ return;
577
+ }
578
+ const t0 = Date.now();
579
+ // Deepgram FIRST (2026-08-01 latency fix): same TTS family the regular
580
+ // voice pipeline uses — ~3-6x faster than the OpenAI full-file synth this
581
+ // endpoint used before (measured 2-4s of the meeting reply lag). OpenAI
582
+ // stays as fallback.
583
+ const dgKey = process.env.DEEPGRAM_API_KEY;
584
+ if (dgKey) {
585
+ try {
586
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3', {
587
+ method: 'POST',
588
+ headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
589
+ body: JSON.stringify({ text }),
590
+ });
591
+ if (dg.ok) {
592
+ const buf = Buffer.from(await dg.arrayBuffer());
593
+ console.log(`🗣️ /tts deepgram ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
594
+ res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
595
+ res.end(buf);
596
+ return;
597
+ }
598
+ console.warn(`⚠️ /tts deepgram ${dg.status} — falling back to OpenAI`);
599
+ }
600
+ catch (e) {
601
+ console.warn(`⚠️ /tts deepgram error: ${e.message} — falling back to OpenAI`);
602
+ }
603
+ }
573
604
  const voice = url.searchParams.get('voice') || 'alloy';
574
605
  const key = process.env.OPENAI_API_KEY;
575
- if (!text || !key) {
606
+ if (!key) {
576
607
  res.writeHead(400, { 'Content-Type': 'application/json' });
577
- res.end(JSON.stringify({ error: !key ? 'no OPENAI_API_KEY' : 'no text' }));
608
+ res.end(JSON.stringify({ error: 'no TTS provider keys' }));
578
609
  return;
579
610
  }
580
611
  try {
@@ -590,6 +621,7 @@ function startApiServer(workingDir, port) {
590
621
  return;
591
622
  }
592
623
  const buf = Buffer.from(await tts.arrayBuffer());
624
+ console.log(`🗣️ /tts openai ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
593
625
  res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
594
626
  res.end(buf);
595
627
  }
@@ -2511,8 +2543,14 @@ async function main() {
2511
2543
  // (reuse of the regular speak path; no Bash roundtrip). Silent-observer
2512
2544
  // turns stay suppressed as before.
2513
2545
  if (activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2514
- console.log(`🔊➡️📽️ tts_say → canvas (addressed meeting turn): "${data.text.slice(0, 60)}"`);
2515
- pushCanvas({ kind: 'say', text: data.text });
2546
+ // Sentence-split so the FIRST sentence synthesizes + plays while the
2547
+ // rest queue behind it (canvas plays says sequentially + prefetches)
2548
+ // — first-audio latency = one short sentence's synth, not the whole
2549
+ // reply's.
2550
+ const sentences = data.text.match(/[^.!?]+[.!?]+["']?|[^.!?]+$/g)?.map((s) => s.trim()).filter(Boolean) || [data.text];
2551
+ console.log(`🔊➡️📽️ tts_say → canvas (${sentences.length} sentence(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2552
+ for (const s of sentences)
2553
+ pushCanvas({ kind: 'say', text: s });
2516
2554
  markMeetingSpeaking(data.text);
2517
2555
  return;
2518
2556
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.106",
3
+ "version": "0.9.107",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {