osborn 0.9.121 → 0.9.122

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +53 -75
  2. package/package.json +1 -1
package/dist/index.js CHANGED
@@ -271,38 +271,35 @@ function interruptMeetingSpeech(reason) {
271
271
  // Recall native output_audio, which requires mp3.
272
272
  async function synthMp3(text) {
273
273
  const t0 = Date.now();
274
- const dgKey = process.env.DEEPGRAM_API_KEY;
275
- if (dgKey) {
276
- try {
277
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=mp3&bit_rate=48000', {
278
- method: 'POST',
279
- headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
280
- body: JSON.stringify({ text: text.slice(0, 4000) }),
281
- signal: AbortSignal.timeout(12000),
282
- });
283
- if (dg.ok) {
284
- const buf = Buffer.from(await dg.arrayBuffer());
285
- console.log(`🗣️ synthMp3 deepgram ${buf.length}b in ${Date.now() - t0}ms`);
286
- return buf;
287
- }
288
- }
289
- catch (e) {
290
- console.warn(`⚠️ synthMp3 deepgram: ${e.message}`);
291
- }
292
- }
293
274
  const oa = process.env.OPENAI_API_KEY;
294
- if (oa) {
295
- try {
296
- const r = await fetch('https://api.openai.com/v1/audio/speech', {
297
- method: 'POST',
298
- headers: { 'Authorization': `Bearer ${oa}`, 'Content-Type': 'application/json' },
299
- body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice: 'alloy', input: text.slice(0, 4000), response_format: 'mp3' }),
300
- signal: AbortSignal.timeout(15000),
301
- });
302
- if (r.ok)
303
- return Buffer.from(await r.arrayBuffer());
275
+ if (!oa) {
276
+ console.warn('⚠️ synthMp3: no OPENAI_API_KEY — meeting has no voice');
277
+ return null;
278
+ }
279
+ // Meeting voice = the SAME OpenAI model/voice as the website's regular TTS
280
+ // (DIRECT_MODE_TTS), so the bot sounds IDENTICAL on both fronts (user directive
281
+ // 2026-08-04: Deepgram aura sounded "cheap and inconsistent"). Deepgram removed
282
+ // from the meeting path entirely. Pulls model/voice from DIRECT_MODE_TTS when
283
+ // it's an OpenAI config so the two never drift.
284
+ const model = DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.model || 'tts-1-hd') : 'tts-1-hd';
285
+ const voice = DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.voice || 'fable') : 'fable';
286
+ try {
287
+ const r = await fetch('https://api.openai.com/v1/audio/speech', {
288
+ method: 'POST',
289
+ headers: { 'Authorization': `Bearer ${oa}`, 'Content-Type': 'application/json' },
290
+ body: JSON.stringify({ model, voice, input: text.slice(0, 4000), response_format: 'mp3' }),
291
+ signal: AbortSignal.timeout(20000),
292
+ });
293
+ if (r.ok) {
294
+ const buf = Buffer.from(await r.arrayBuffer());
295
+ console.log(`🗣️ synthMp3 openai ${model}/${voice} ${buf.length}b in ${Date.now() - t0}ms`);
296
+ return buf;
304
297
  }
305
- catch { /* fall through */ }
298
+ const e = await r.text().catch(() => '');
299
+ console.warn(`⚠️ synthMp3 openai ${r.status}: ${e.slice(0, 120)}`);
300
+ }
301
+ catch (e) {
302
+ console.warn(`⚠️ synthMp3 openai: ${e.message}`);
306
303
  }
307
304
  return null;
308
305
  }
@@ -710,46 +707,23 @@ function startApiServer(workingDir, port) {
710
707
  return;
711
708
  }
712
709
  const t0 = Date.now();
713
- // Deepgram FIRST (2026-08-01 latency fix): same TTS family the regular
714
- // voice pipeline uses — ~3-6x faster than the OpenAI full-file synth this
715
- // endpoint used before (measured 2-4s of the meeting reply lag). OpenAI
716
- // stays as fallback.
717
- const dgKey = process.env.DEEPGRAM_API_KEY;
718
- if (dgKey) {
719
- try {
720
- // WAV/linear16, not mp3: mp3 files carry encoder padding (leading/
721
- // trailing silence + boundary click) — back-to-back sentence clips
722
- // were heard as "cracking". WAV is gapless-safe and cheaper to decode.
723
- const dg = await fetch('https://api.deepgram.com/v1/speak?model=aura-2-asteria-en&encoding=linear16&sample_rate=48000&container=wav', {
724
- method: 'POST',
725
- headers: { 'Authorization': `Token ${dgKey}`, 'Content-Type': 'application/json' },
726
- body: JSON.stringify({ text }),
727
- });
728
- if (dg.ok) {
729
- const buf = Buffer.from(await dg.arrayBuffer());
730
- console.log(`🗣️ /tts deepgram wav ${buf.length}b in ${Date.now() - t0}ms (${text.length} chars) t=${new Date().toISOString()}`);
731
- res.writeHead(200, { 'Content-Type': 'audio/wav', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
732
- res.end(buf);
733
- return;
734
- }
735
- console.warn(`⚠️ /tts deepgram ${dg.status} — falling back to OpenAI`);
736
- }
737
- catch (e) {
738
- console.warn(`⚠️ /tts deepgram error: ${e.message} — falling back to OpenAI`);
739
- }
740
- }
741
- const voice = url.searchParams.get('voice') || 'alloy';
710
+ // Meeting voice = the SAME OpenAI model/voice as the website's regular TTS
711
+ // (DIRECT_MODE_TTS) — user directive 2026-08-04: Deepgram aura removed, it
712
+ // sounded cheap/inconsistent. Consistency over the ~2-4s latency Deepgram
713
+ // saved. mp3 out (the canvas <audio> element plays it into the meeting).
742
714
  const key = process.env.OPENAI_API_KEY;
743
715
  if (!key) {
744
716
  res.writeHead(400, { 'Content-Type': 'application/json' });
745
- res.end(JSON.stringify({ error: 'no TTS provider keys' }));
717
+ res.end(JSON.stringify({ error: 'no OPENAI_API_KEY' }));
746
718
  return;
747
719
  }
720
+ const model = DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.model || 'tts-1-hd') : 'tts-1-hd';
721
+ const voice = url.searchParams.get('voice') || (DIRECT_MODE_TTS.provider === 'openai' ? (DIRECT_MODE_TTS.voice || 'fable') : 'fable');
748
722
  try {
749
723
  const tts = await fetch('https://api.openai.com/v1/audio/speech', {
750
724
  method: 'POST',
751
725
  headers: { 'Authorization': `Bearer ${key}`, 'Content-Type': 'application/json' },
752
- body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice, input: text, response_format: 'mp3' }),
726
+ body: JSON.stringify({ model, voice, input: text, response_format: 'mp3' }),
753
727
  });
754
728
  if (!tts.ok) {
755
729
  const e = await tts.text().catch(() => '');
@@ -758,7 +732,7 @@ function startApiServer(workingDir, port) {
758
732
  return;
759
733
  }
760
734
  const buf = Buffer.from(await tts.arrayBuffer());
761
- console.log(`🗣️ /tts openai ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
735
+ console.log(`🗣️ /tts openai ${model}/${voice} ${buf.length}b in ${Date.now() - t0}ms t=${new Date().toISOString()}`);
762
736
  res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
763
737
  res.end(buf);
764
738
  }
@@ -1829,13 +1803,14 @@ async function main() {
1829
1803
  // from muting the audio path. Reset by endMeeting.
1830
1804
  meetingAddressedUntil = Date.now() + 6 * 60 * 60 * 1000;
1831
1805
  }
1832
- // Addressed turns get a MINIMAL tag + a hard brevity rule (0.9.121): the
1833
- // reply is SPOKEN into a live meeting, so long answers (a) take 8s+ to
1834
- // synthesize and (b) pile up / talk over the next turn. 1–2 sentences keeps
1835
- // it conversational and fast; the agent can offer to go deeper if asked.
1836
- const header = addressed
1837
- ? `[MEETING — ${botId}] (addressed — your reply is SPOKEN OUT LOUD into the meeting: keep it to 1–2 short sentences, conversational, no lists/markdown; offer to elaborate only if they want more):`
1838
- : `[MEETING — ${botId}]:`;
1806
+ // NO meeting-specific reply coaching (user directive 2026-08-04): the reply
1807
+ // must be exactly what the main Claude Code agent would say — no brevity
1808
+ // rules, no "spoken out loud" framing. Just a bare [MEETING — id] routing
1809
+ // tag, which is all the plumbing needs: PipelineDirectLLM keys
1810
+ // suppressMeetingTTS off the `[MEETING` prefix, and whether the reply is
1811
+ // SPOKEN is decided in CODE (activeMeetingBotId + meetingAddressedUntil),
1812
+ // not by any prompt instruction. Addressed vs observer is the code boolean.
1813
+ const header = `[MEETING — ${botId}]:`;
1839
1814
  // Prepend + consume any interruption context (bot was cut off mid-sentence).
1840
1815
  const interrupt = meetingInterruptContext ? `${meetingInterruptContext}\n` : '';
1841
1816
  meetingInterruptContext = '';
@@ -5352,19 +5327,22 @@ async function main() {
5352
5327
  recallJoin.registerBot(botId, sessionId);
5353
5328
  activeMeetingBotId = botId;
5354
5329
  await sendToFrontend({ type: 'meeting_joined', botId, message: 'Osborn has joined the meeting' });
5355
- // System injection so the LLM knows it's in a meeting and which
5356
- // skill to apply. The meetings skill (agent/.claude/skills/meetings/SKILL.md)
5357
- // teaches the agent: don't speak in response to [MEETING — *]:
5358
- // messages, keep meeting-todos.md updated in the workspace, etc.
5330
+ // Minimal awareness injection (user directive 2026-08-04): tell the
5331
+ // LLM it's in a meeting and how transcripts are tagged — nothing
5332
+ // more. NO "do NOT speak / silent observer" coaching (that fought
5333
+ // the goal of the reply being exactly the main agent's response).
5334
+ // The agent responds to meeting turns the same way it responds on
5335
+ // the website; note-taking is an optional background task, not a
5336
+ // replacement for responding.
5359
5337
  if (currentLLM) {
5360
5338
  try {
5361
5339
  const sysCtx = new llm.ChatContext();
5362
5340
  sysCtx.addMessage({
5363
5341
  role: 'user',
5364
- content: `[SYSTEM] You are now in a meeting (Recall bot ID: ${botId}, URL: ${meetingUrl}). Transcript chunks will arrive every ~30 seconds tagged \`[MEETING — ${botId}]:\`. Follow the meetings skill: do NOT speak in response (no TTS output), instead maintain meeting-todos.md in the session workspace, optionally trigger background research silently. The voice-native user can still interact normally — only the meeting-tagged messages are the silent-observer path. Acknowledge by writing the initial meeting-todos.md skeleton.`,
5342
+ content: `[SYSTEM] You are now in a meeting (Recall bot ID: ${botId}, URL: ${meetingUrl}). Live transcript arrives tagged \`[MEETING — ${botId}]:\`. Respond to what's said exactly as you naturally would — same as on the website, no special meeting phrasing. You may keep meeting-todos.md updated in the workspace in the background, but responding comes first.`,
5365
5343
  });
5366
5344
  currentLLM.chat({ chatCtx: sysCtx });
5367
- console.log('📓 Meeting system injection sent to LLM');
5345
+ console.log('📓 Meeting awareness injection sent to LLM');
5368
5346
  }
5369
5347
  catch (sysErr) {
5370
5348
  console.warn('⚠️ Meeting system injection failed:', sysErr.message);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.121",
3
+ "version": "0.9.122",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {