osborn 0.9.88 → 0.9.91
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/skills/meetings/SKILL.md +49 -3
- package/Dockerfile.sandbox +8 -1
- package/dist/index.js +84 -22
- package/package.json +1 -1
|
@@ -22,18 +22,64 @@ Two trigger patterns:
|
|
|
22
22
|
|
|
23
23
|
**Do NOT use this skill** for normal user voice-native messages that don't fit those patterns — those get spoken responses as usual.
|
|
24
24
|
|
|
25
|
+
## CRITICAL: delegate the file + transcript work to the `writer` sub-agent
|
|
26
|
+
|
|
27
|
+
The main orchestrator agent has a **hard limit of 3 direct tool calls per turn** (enforced in PreToolUse — Read/Write/Bash/Glob are DENIED after the 3rd call). Writing `meeting-todos.md` and pulling transcripts (curl + jq + Write) is far more than 3 calls, so **doing it directly gets you blocked** ("all tools blocked").
|
|
28
|
+
|
|
29
|
+
**Sub-agents are exempt from this budget and have full permissions.** So for ALL meeting file/transcript work, **delegate to the `writer` sub-agent in ONE `Task` call** and let it do the whole job (fetch transcript, parse, write `meeting-todos.md`). That's a single tool call for you, and the writer has no budget cap.
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
Task(
|
|
33
|
+
subagent_type: 'writer',
|
|
34
|
+
run_in_background: true, // silent — don't block voice
|
|
35
|
+
description: 'update meeting-todos.md',
|
|
36
|
+
prompt: '<the full instructions below: workspace path, bot ID, what to fetch/parse/write>'
|
|
37
|
+
)
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Give the writer everything it needs in the prompt: the session-workspace path, the bot ID, the `us-west-2.recall.ai` endpoint rule, and the `meeting-todos.md` structure. The writer runs the curl/jq/Write steps itself. For research, delegate to the `researcher` sub-agent the same way.
|
|
41
|
+
|
|
42
|
+
## How to SPEAK INTO the meeting (out of silent mode)
|
|
43
|
+
|
|
44
|
+
You can talk directly into the Google Meet / Zoom — your words play as the bot's
|
|
45
|
+
voice. The bot casts a "meeting canvas" webpage as its camera+mic; POSTing to
|
|
46
|
+
your own HTTP API pushes speech (and visuals) to it, and Recall pipes the
|
|
47
|
+
canvas audio into the call. This is ONE Bash call, within budget.
|
|
48
|
+
|
|
49
|
+
**Speak into the meeting:**
|
|
50
|
+
```bash
|
|
51
|
+
curl -sS -X POST http://localhost:8741/canvas -H 'Content-Type: application/json' \
|
|
52
|
+
-d '{"kind":"say","text":"YOUR WORDS HERE"}'
|
|
53
|
+
```
|
|
54
|
+
(Port is `OSBORN_API_PORT`, default 8741.)
|
|
55
|
+
|
|
56
|
+
**Show a visual on the bot's camera** (notes, a link, a title):
|
|
57
|
+
```bash
|
|
58
|
+
curl -sS -X POST http://localhost:8741/canvas -H 'Content-Type: application/json' \
|
|
59
|
+
-d '{"kind":"show","mode":"notes","title":"...","items":["...","..."]}'
|
|
60
|
+
```
|
|
61
|
+
`mode` = `idle` | `notes` (title+items) | `link` (url) | `web` (iframe url) | `text` (title+text).
|
|
62
|
+
|
|
63
|
+
**When to speak into the meeting:** By DEFAULT stay silent (observer) for
|
|
64
|
+
`[MEETING — *]:` chunks — take notes, don't interrupt. Speak into the meeting
|
|
65
|
+
ONLY when: (a) the voice-native user explicitly tells you to say something to the
|
|
66
|
+
meeting / "tell them X" / "answer that", or (b) you're directly addressed by name
|
|
67
|
+
in the meeting and the user has enabled active mode. When you do speak, keep it
|
|
68
|
+
short and let the room continue. This is the toggle between silent-observer and
|
|
69
|
+
active-participant.
|
|
70
|
+
|
|
25
71
|
## How to behave (auto-tagged chunks)
|
|
26
72
|
|
|
27
73
|
For every `[MEETING — *]:` message:
|
|
28
74
|
|
|
29
75
|
1. **Do NOT speak.** No TTS output. No conversational reply.
|
|
30
|
-
2. **
|
|
31
|
-
3. **
|
|
76
|
+
2. **Delegate to the `writer` sub-agent** (see above) to append new action items, decisions, and open questions to `meeting-todos.md`. Do NOT write the file yourself — you'll hit the 3-call budget and get blocked. Batch chunks if they arrive faster than the writer finishes; one evolving file.
|
|
77
|
+
3. **Delegate research to the `researcher` sub-agent** via `Task` (background, silent) when a chunk warrants it.
|
|
32
78
|
4. **Don't consume voice-native attention.** The user can interrupt with a voice-native message at any time — that's the only kind that gets spoken responses.
|
|
33
79
|
|
|
34
80
|
## How to pull transcripts on demand (Bash + curl)
|
|
35
81
|
|
|
36
|
-
When the user explicitly asks (see triggers above)
|
|
82
|
+
When the user explicitly asks (see triggers above): speak briefly first ("On it"), then **delegate the fetch+parse+write to the `writer` sub-agent** in one `Task` call (the steps below are what you put in the writer's prompt — they're 4+ Bash/Write calls, over your 3-call budget). When the writer finishes, speak the result. The commands below are the recipe the writer runs, not calls you make directly.
|
|
37
83
|
|
|
38
84
|
### Step 1: Get the bot ID
|
|
39
85
|
|
package/Dockerfile.sandbox
CHANGED
|
@@ -164,7 +164,14 @@ PKG_SKILLS_DIR="/usr/local/lib/node_modules/osborn/.claude/skills"
|
|
|
164
164
|
SEED_VERSION_FILE="${HOME_SKILLS_DIR}/.seed-version"
|
|
165
165
|
mkdir -p "$HOME_SKILLS_DIR"
|
|
166
166
|
CURRENT_SEED_VERSION=$(cat "$SEED_VERSION_FILE" 2>/dev/null | tr -d '[:space:]' || echo "")
|
|
167
|
-
|
|
167
|
+
# Seed version = the INSTALLED package's own version (reliable on every boot),
|
|
168
|
+
# NOT the OSBORN_IMAGE_VERSION env — that env isn't threaded through
|
|
169
|
+
# `fly machine update`, so it went stale and left the marker permanently pinned
|
|
170
|
+
# at 0.9.48, meaning the refresh below never fired and new-version skills never
|
|
171
|
+
# reached the load path. Reading package.json makes CURRENT != IMAGE trip on a
|
|
172
|
+
# real version bump. Falls back to the env / "latest" if the package is unreadable.
|
|
173
|
+
IMAGE_SEED_VERSION=$(grep -m1 '"version"' /usr/local/lib/node_modules/osborn/package.json 2>/dev/null | sed 's/.*"version"[^"]*"\([^"]*\)".*/\1/')
|
|
174
|
+
[ -z "$IMAGE_SEED_VERSION" ] && IMAGE_SEED_VERSION="${OSBORN_IMAGE_VERSION:-latest}"
|
|
168
175
|
if [ -d "$PKG_SKILLS_DIR" ]; then
|
|
169
176
|
REFRESHED=0
|
|
170
177
|
SEEDED=0
|
package/dist/index.js
CHANGED
|
@@ -515,6 +515,41 @@ function startApiServer(workingDir, port) {
|
|
|
515
515
|
console.log(`[canvas] client connected (${canvasClients.size} total)`);
|
|
516
516
|
return;
|
|
517
517
|
}
|
|
518
|
+
// ── Meeting canvas: TTS audio for speaking INTO the meeting ──────────────
|
|
519
|
+
// GET /tts?text=... → mp3 (OpenAI TTS). The canvas plays this as a real
|
|
520
|
+
// <audio> element so Recall's webpage output pipes it into the meeting —
|
|
521
|
+
// speechSynthesis is NOT captured by Recall, a media element IS.
|
|
522
|
+
if (req.method === 'GET' && url.pathname === '/tts') {
|
|
523
|
+
const text = (url.searchParams.get('text') || '').slice(0, 4000);
|
|
524
|
+
const voice = url.searchParams.get('voice') || 'alloy';
|
|
525
|
+
const key = process.env.OPENAI_API_KEY;
|
|
526
|
+
if (!text || !key) {
|
|
527
|
+
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
528
|
+
res.end(JSON.stringify({ error: !key ? 'no OPENAI_API_KEY' : 'no text' }));
|
|
529
|
+
return;
|
|
530
|
+
}
|
|
531
|
+
try {
|
|
532
|
+
const tts = await fetch('https://api.openai.com/v1/audio/speech', {
|
|
533
|
+
method: 'POST',
|
|
534
|
+
headers: { 'Authorization': `Bearer ${key}`, 'Content-Type': 'application/json' },
|
|
535
|
+
body: JSON.stringify({ model: 'gpt-4o-mini-tts', voice, input: text, response_format: 'mp3' }),
|
|
536
|
+
});
|
|
537
|
+
if (!tts.ok) {
|
|
538
|
+
const e = await tts.text().catch(() => '');
|
|
539
|
+
res.writeHead(502, { 'Content-Type': 'application/json' });
|
|
540
|
+
res.end(JSON.stringify({ error: `tts ${tts.status}`, detail: e.slice(0, 200) }));
|
|
541
|
+
return;
|
|
542
|
+
}
|
|
543
|
+
const buf = Buffer.from(await tts.arrayBuffer());
|
|
544
|
+
res.writeHead(200, { 'Content-Type': 'audio/mpeg', 'Cache-Control': 'no-store', 'Content-Length': buf.length });
|
|
545
|
+
res.end(buf);
|
|
546
|
+
}
|
|
547
|
+
catch (e) {
|
|
548
|
+
res.writeHead(500, { 'Content-Type': 'application/json' });
|
|
549
|
+
res.end(JSON.stringify({ error: e.message }));
|
|
550
|
+
}
|
|
551
|
+
return;
|
|
552
|
+
}
|
|
518
553
|
// ── Meeting canvas: control endpoint (director / agent tool) ─────────────
|
|
519
554
|
// POST /canvas { kind:'say', text } | { kind:'show', mode, title?, items?, url?, text? }
|
|
520
555
|
if (req.method === 'POST' && url.pathname === '/canvas') {
|
|
@@ -1521,6 +1556,36 @@ async function main() {
|
|
|
1521
1556
|
let currentUserId = '';
|
|
1522
1557
|
let activeMeetingBotId = null; // Recall.ai bot ID if in a meeting
|
|
1523
1558
|
let activeMeetingPoller = null; // Transcript poller bound to that bot
|
|
1559
|
+
// LIVE meeting transcript → LLM (buffered webhook finals). See recall.on('transcript').
|
|
1560
|
+
const meetingTranscriptBuffer = [];
|
|
1561
|
+
let meetingFlushTimer = null;
|
|
1562
|
+
const startMeetingFlush = (botId) => {
|
|
1563
|
+
stopMeetingFlush();
|
|
1564
|
+
console.log(`📓 Meeting transcript flush timer started (bot ${botId}, 20s)`);
|
|
1565
|
+
meetingFlushTimer = setInterval(() => {
|
|
1566
|
+
if (!meetingTranscriptBuffer.length || !currentLLM)
|
|
1567
|
+
return;
|
|
1568
|
+
const turns = meetingTranscriptBuffer.splice(0); // drain
|
|
1569
|
+
const tagged = `[MEETING — ${botId}]:\n${turns.join('\n')}`;
|
|
1570
|
+
try {
|
|
1571
|
+
const ctx = new llm.ChatContext();
|
|
1572
|
+
ctx.addMessage({ role: 'user', content: tagged });
|
|
1573
|
+
currentLLM.chat({ chatCtx: ctx });
|
|
1574
|
+
console.log(`📓 Flushed ${turns.length} meeting turn(s) to LLM`);
|
|
1575
|
+
}
|
|
1576
|
+
catch (err) {
|
|
1577
|
+
console.warn(`⚠️ Meeting flush failed: ${err.message}`);
|
|
1578
|
+
}
|
|
1579
|
+
}, 20_000);
|
|
1580
|
+
};
|
|
1581
|
+
const stopMeetingFlush = () => {
|
|
1582
|
+
if (meetingFlushTimer) {
|
|
1583
|
+
clearInterval(meetingFlushTimer);
|
|
1584
|
+
meetingFlushTimer = null;
|
|
1585
|
+
console.log('📓 Meeting flush timer stopped');
|
|
1586
|
+
}
|
|
1587
|
+
meetingTranscriptBuffer.length = 0;
|
|
1588
|
+
};
|
|
1524
1589
|
// Track the active resume session ID across scopes (ParticipantConnected + DataReceived)
|
|
1525
1590
|
// Updated by resume_session, session_selected, continue_session, switch_session handlers
|
|
1526
1591
|
let currentResumeSessionId;
|
|
@@ -1558,26 +1623,18 @@ async function main() {
|
|
|
1558
1623
|
// "what was said in the meeting" display, separate from the LLM input path).
|
|
1559
1624
|
const recall = getRecallClient();
|
|
1560
1625
|
if (recall) {
|
|
1561
|
-
console.log('🎥 Recall.ai client initialized
|
|
1562
|
-
|
|
1563
|
-
|
|
1564
|
-
|
|
1565
|
-
|
|
1566
|
-
|
|
1567
|
-
|
|
1568
|
-
|
|
1569
|
-
|
|
1570
|
-
|
|
1571
|
-
|
|
1572
|
-
|
|
1573
|
-
// const chatCtx = new llm.ChatContext()
|
|
1574
|
-
// chatCtx.addMessage({ role: 'user', content: meetingText })
|
|
1575
|
-
// ;(currentLLM as any).chat({ chatCtx })
|
|
1576
|
-
// }
|
|
1577
|
-
// } catch (err) {
|
|
1578
|
-
// console.error('❌ Failed to route meeting transcript:', err)
|
|
1579
|
-
// }
|
|
1580
|
-
// }
|
|
1626
|
+
console.log('🎥 Recall.ai client initialized — LIVE webhook finals buffered → LLM (batch poller download_url is empty mid-call, so the webhook is the only live transcript source)');
|
|
1627
|
+
// LIVE meeting transcript → LLM. The realtime webhook (handleWebhook) emits
|
|
1628
|
+
// every partial + final. We buffer FINALS (partials are too noisy) and a
|
|
1629
|
+
// flush timer (started on join_meeting) batches them to currentLLM.chat()
|
|
1630
|
+
// as [MEETING — botId]: every ~20s — so the agent actually sees the meeting
|
|
1631
|
+
// and can take notes / delegate to the writer. Previously this was DISABLED
|
|
1632
|
+
// and the poller (batch endpoint) was the only LLM path — but that endpoint
|
|
1633
|
+
// is empty until the meeting ENDS, so mid-call the LLM saw nothing.
|
|
1634
|
+
recall.on('transcript', ({ botId, speaker, text, partial }) => {
|
|
1635
|
+
console.log(`📝 Meeting transcript [${speaker}]${partial ? ' (partial)' : ''}: ${text}`);
|
|
1636
|
+
if (!partial && text.trim())
|
|
1637
|
+
meetingTranscriptBuffer.push(`${speaker}: ${text.trim()}`);
|
|
1581
1638
|
});
|
|
1582
1639
|
}
|
|
1583
1640
|
// ============================================================
|
|
@@ -4066,6 +4123,7 @@ async function main() {
|
|
|
4066
4123
|
clearFastBrainSession();
|
|
4067
4124
|
clearPipelineFastBrainSession();
|
|
4068
4125
|
// Auto-leave any active meeting bot when user disconnects from the room
|
|
4126
|
+
stopMeetingFlush();
|
|
4069
4127
|
if (activeMeetingPoller) {
|
|
4070
4128
|
activeMeetingPoller.stop();
|
|
4071
4129
|
activeMeetingPoller = null;
|
|
@@ -4751,6 +4809,9 @@ async function main() {
|
|
|
4751
4809
|
},
|
|
4752
4810
|
});
|
|
4753
4811
|
activeMeetingPoller.start();
|
|
4812
|
+
// LIVE path: buffer webhook finals + flush to the LLM every 20s.
|
|
4813
|
+
// (The poller above only lands data after the meeting ENDS.)
|
|
4814
|
+
startMeetingFlush(botId);
|
|
4754
4815
|
}
|
|
4755
4816
|
catch (err) {
|
|
4756
4817
|
console.error('❌ Recall.ai join error:', err);
|
|
@@ -4764,8 +4825,9 @@ async function main() {
|
|
|
4764
4825
|
const recallLeave = getRecallClient();
|
|
4765
4826
|
if (recallLeave && botId) {
|
|
4766
4827
|
try {
|
|
4767
|
-
// Stop the transcript poller FIRST so no more
|
|
4768
|
-
// forwarded to the LLM during the leave.
|
|
4828
|
+
// Stop the transcript poller + live flush FIRST so no more chunks
|
|
4829
|
+
// get forwarded to the LLM during the leave.
|
|
4830
|
+
stopMeetingFlush();
|
|
4769
4831
|
if (activeMeetingPoller) {
|
|
4770
4832
|
activeMeetingPoller.stop();
|
|
4771
4833
|
activeMeetingPoller = null;
|