osborn 0.9.107 β†’ 0.9.108

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +19 -11
  2. package/package.json +1 -1
package/dist/index.js CHANGED
@@ -583,15 +583,18 @@ function startApiServer(workingDir, port) {
583
583
  const dgKey = process.env.DEEPGRAM_API_KEY;
584
584
  if (dgKey) {
585
585
  try {
586
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3', {
586
+ // WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
587
+ // trailing silence + boundary click) β€” back-to-back sentence clips
588
+ // were heard as "cracking". WAV is gapless-safe and cheaper to decode.
589
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=linear16&sample_rate=24000&container=wav', {
587
590
  method: 'POST',
588
591
  headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
589
592
  body: JSON.stringify({ text }),
590
593
  });
591
594
  if (dg.ok) {
592
595
  const buf = Buffer.from(await dg.arrayBuffer());
593
- console.log(`πŸ—£οΈ /tts deepgram ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
594
- res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
596
+ console.log(`πŸ—£οΈ /tts deepgram wav ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
597
+ res.writeHead(200, { 'Content-Type': 'audio/wav', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
595
598
  res.end(buf);
596
599
  return;
597
600
  }
@@ -1673,8 +1676,12 @@ async function main() {
1673
1676
  // reply entirely) β€” measured as most of the 5-8s reply lag.
1674
1677
  meetingAddressedUntil = Date.now() + 90_000;
1675
1678
  }
1679
+ // PROMPT PARITY (user directive 2026-08-01): addressed turns carry ONLY a
1680
+ // minimal tag β€” no behavioral re-instruction. The agent replies exactly as
1681
+ // it would to a regular voice turn; the tts_say→canvas redirect handles
1682
+ // where the words go. Same agent, same behavior, both fronts.
1676
1683
  const header = addressed
1677
- ? `[MEETING β€” ${botId}] β€” YOU WERE ADDRESSED. Reply now in PLAIN TEXT β€” your words are spoken into the meeting automatically (do NOT use Bash/curl to speak). One or two short conversational sentences first; then, if needed, delegate notes/research in the background.`
1684
+ ? `[MEETING β€” ${botId}] (addressed β€” reply is spoken into the meeting):`
1678
1685
  : `[MEETING β€” ${botId}]:`;
1679
1686
  // Prepend + consume any interruption context (bot was cut off mid-sentence).
1680
1687
  const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
@@ -2543,13 +2550,14 @@ async function main() {
2543
2550
  // (reuse of the regular speak path; no Bash roundtrip). Silent-observer
2544
2551
  // turns stay suppressed as before.
2545
2552
  if (activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2546
- // Sentence-split so the FIRST sentence synthesizes + plays while the
2547
- // rest queue behind it (canvas plays says sequentially + prefetches)
2548
- // β€” first-audio latency = one short sentence's synth, not the whole
2549
- // reply's.
2550
- const sentences = data.text.match(/[^.!?]+[.!?]+["']?|[^.!?]+$/g)?.map((s) => s.trim()).filter(Boolean) || [data.text];
2551
- console.log(`πŸ”Šβž‘οΈπŸ“½οΈ tts_say β†’ canvas (${sentences.length} sentence(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2552
- for (const s of sentences)
2553
+ // Split into AT MOST 2 chunks: first sentence (fast first-audio) +
2554
+ // the remainder as ONE chunk. Per-sentence files created an audible
2555
+ // seam at every boundary (mp3 padding + scheduling jitter β€” heard as
2556
+ // "cracking"); two chunks keeps the latency win with one seam max.
2557
+ const m = data.text.match(/^([^.!?]+[.!?]+["']?)\s*([\s\S]*)$/);
2558
+ const chunks = m && m[2]?.trim() ? [m[1].trim(), m[2].trim()] : [data.text.trim()];
2559
+ console.log(`πŸ”Šβž‘οΈπŸ“½οΈ tts_say β†’ canvas (${chunks.length} chunk(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2560
+ for (const s of chunks)
2553
2561
  pushCanvas({ kind: 'say', text: s });
2554
2562
  markMeetingSpeaking(data.text);
2555
2563
  return;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.107",
3
+ "version": "0.9.108",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {