osborn 0.9.107 β†’ 0.9.109

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +28 -12
  2. package/package.json +1 -1
package/dist/index.js CHANGED
@@ -583,15 +583,18 @@ function startApiServer(workingDir, port) {
583
583
  const dgKey = process.env.DEEPGRAM_API_KEY;
584
584
  if (dgKey) {
585
585
  try {
586
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=mp3', {
586
+ // WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
587
+ // trailing silence + boundary click) β€” back-to-back sentence clips
588
+ // were heard as "cracking". WAV is gapless-safe and cheaper to decode.
589
+ const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-thalia-en&encoding=linear16&sample_rate=24000&container=wav', {
587
590
  method: 'POST',
588
591
  headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
589
592
  body: JSON.stringify({ text }),
590
593
  });
591
594
  if (dg.ok) {
592
595
  const buf = Buffer.from(await dg.arrayBuffer());
593
- console.log(`πŸ—£οΈ /tts deepgram ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
594
- res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
596
+ console.log(`πŸ—£οΈ /tts deepgram wav ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
597
+ res.writeHead(200, { 'Content-Type': 'audio/wav', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
595
598
  res.end(buf);
596
599
  return;
597
600
  }
@@ -1671,10 +1674,21 @@ async function main() {
1671
1674
  // mode, sink = meeting. The old prompt made the agent compose a Bash curl
1672
1675
  // (extra LLM tool roundtrip, ate the 3-call budget, sometimes lost the
1673
1676
  // reply entirely) β€” measured as most of the 5-8s reply lag.
1674
- meetingAddressedUntil = Date.now() + 90_000;
1677
+ // ONCE ADDRESSED, STAY CONVERSATIONAL (2026-08-01 fix): the old 90s
1678
+ // window silently ATE every reply generated after it expired β€” the agent
1679
+ // kept "answering" into suppression while the room heard nothing
1680
+ // (user transcript full of unheard replies; direct /canvas say test WAS
1681
+ // heard). A participant who's been spoken to stays in the conversation;
1682
+ // observer restraint comes from the skill not generating chatter, not
1683
+ // from muting the audio path. Reset by endMeeting.
1684
+ meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
1675
1685
  }
1686
+ // PROMPT PARITY (user directive 2026-08-01): addressed turns carry ONLY a
1687
+ // minimal tag β€” no behavioral re-instruction. The agent replies exactly as
1688
+ // it would to a regular voice turn; the tts_say→canvas redirect handles
1689
+ // where the words go. Same agent, same behavior, both fronts.
1676
1690
  const header = addressed
1677
- ? `[MEETING β€” ${botId}] β€” YOU WERE ADDRESSED. Reply now in PLAIN TEXT β€” your words are spoken into the meeting automatically (do NOT use Bash/curl to speak). One or two short conversational sentences first; then, if needed, delegate notes/research in the background.`
1691
+ ? `[MEETING β€” ${botId}] (addressed β€” reply is spoken into the meeting):`
1678
1692
  : `[MEETING β€” ${botId}]:`;
1679
1693
  // Prepend + consume any interruption context (bot was cut off mid-sentence).
1680
1694
  const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
@@ -1732,6 +1746,7 @@ async function main() {
1732
1746
  meetingLeaveGraceTimer = null;
1733
1747
  }
1734
1748
  activeMeetingBotId = null;
1749
+ meetingAddressedUntil = 0;
1735
1750
  // leaveBot=false when Recall itself reported the meeting over (bot already gone).
1736
1751
  if (opts.leaveBot !== false) {
1737
1752
  const recall = getRecallClient();
@@ -2543,13 +2558,14 @@ async function main() {
2543
2558
  // (reuse of the regular speak path; no Bash roundtrip). Silent-observer
2544
2559
  // turns stay suppressed as before.
2545
2560
  if (activeMeetingBotId && Date.now() < meetingAddressedUntil) {
2546
- // Sentence-split so the FIRST sentence synthesizes + plays while the
2547
- // rest queue behind it (canvas plays says sequentially + prefetches)
2548
- // β€” first-audio latency = one short sentence's synth, not the whole
2549
- // reply's.
2550
- const sentences = data.text.match(/[^.!?]+[.!?]+["']?|[^.!?]+$/g)?.map((s) => s.trim()).filter(Boolean) || [data.text];
2551
- console.log(`πŸ”Šβž‘οΈπŸ“½οΈ tts_say β†’ canvas (${sentences.length} sentence(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2552
- for (const s of sentences)
2561
+ // Split into AT MOST 2 chunks: first sentence (fast first-audio) +
2562
+ // the remainder as ONE chunk. Per-sentence files created an audible
2563
+ // seam at every boundary (mp3 padding + scheduling jitter β€” heard as
2564
+ // "cracking"); two chunks keeps the latency win with one seam max.
2565
+ const m = data.text.match(/^([^.!?]+[.!?]+["']?)\s*([\s\S]*)$/);
2566
+ const chunks = m && m[2]?.trim() ? [m[1].trim(), m[2].trim()] : [data.text.trim()];
2567
+ console.log(`πŸ”Šβž‘οΈπŸ“½οΈ tts_say β†’ canvas (${chunks.length} chunk(s), addressed turn) t=${new Date().toISOString()}: "${data.text.slice(0, 60)}"`);
2568
+ for (const s of chunks)
2553
2569
  pushCanvas({ kind: 'say', text: s });
2554
2570
  markMeetingSpeaking(data.text);
2555
2571
  return;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.107",
3
+ "version": "0.9.109",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {