osborn 0.9.134 → 0.9.136
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/Dockerfile.sandbox +14 -0
- package/dist/claude-llm.js +31 -0
- package/dist/pipeline-direct-llm.js +75 -3
- package/dist/prompts/direct-mode-research.md +7 -0
- package/dist/prompts.js +7 -0
- package/package.json +1 -1
package/Dockerfile.sandbox
CHANGED
|
@@ -205,6 +205,20 @@ if [ -d "$HOME/.claude/projects" ]; then
|
|
|
205
205
|
done
|
|
206
206
|
fi
|
|
207
207
|
|
|
208
|
+
|
|
209
|
+
# === Sync volume npm-global with image osborn version ===
|
|
210
|
+
# PATH prefers /workspace/.npm-global/bin over /usr/local/bin — so after an
|
|
211
|
+
# image-swap the volume copy is stale. Detect mismatch and sync on boot so
|
|
212
|
+
# the dashboard update button (image-swap + restart) actually takes effect.
|
|
213
|
+
VOL_OSBORN_VER=$(grep -m1 '"version"' /workspace/.npm-global/lib/node_modules/osborn/package.json 2>/dev/null | sed 's/.*"version"[^"]*"\([^"]*\)".*/\1/' || echo "none")
|
|
214
|
+
if [ -n "$IMAGE_SEED_VERSION" ] && [ "$IMAGE_SEED_VERSION" != "$VOL_OSBORN_VER" ]; then
|
|
215
|
+
echo "[sandbox-d] volume osborn@${VOL_OSBORN_VER} != image@${IMAGE_SEED_VERSION} -- syncing"
|
|
216
|
+
rm -rf /workspace/.npm-global/lib/node_modules/.osborn-* 2>/dev/null
|
|
217
|
+
npm install -g "osborn@${IMAGE_SEED_VERSION}" --no-audit --no-fund 2>&1 | tail -3
|
|
218
|
+
echo "[sandbox-d] sync done: osborn@${IMAGE_SEED_VERSION} on volume"
|
|
219
|
+
else
|
|
220
|
+
echo "[sandbox-d] volume osborn@${VOL_OSBORN_VER} matches image -- no sync needed"
|
|
221
|
+
fi
|
|
208
222
|
echo "[sandbox-d] exec'ing osborn (no chroot, HOME=$HOME)"
|
|
209
223
|
exec osborn
|
|
210
224
|
ENTRYPOINT
|
package/dist/claude-llm.js
CHANGED
|
@@ -199,6 +199,12 @@ export const NAMED_AGENTS = {
|
|
|
199
199
|
'- Do NOT edit or write any files',
|
|
200
200
|
'- Do NOT run destructive commands (no rm, no git push, no npm publish)',
|
|
201
201
|
'- If you need clarification, ask the main agent — it will relay to the user if needed',
|
|
202
|
+
'',
|
|
203
|
+
'## When to use / handoff',
|
|
204
|
+
'Invoked FIRST for any task requiring facts, codebase exploration, or web research.',
|
|
205
|
+
'Return findings to the orchestrator — never directly to the user.',
|
|
206
|
+
'Run several researchers in parallel when there are independent threads to investigate.',
|
|
207
|
+
'Hand off back to the orchestrator; it decides whether to invoke planner or writer next.',
|
|
202
208
|
].join('\n'),
|
|
203
209
|
},
|
|
204
210
|
reasoner: {
|
|
@@ -235,6 +241,11 @@ export const NAMED_AGENTS = {
|
|
|
235
241
|
'- Do NOT edit or write files — return a plan for the writer agent',
|
|
236
242
|
'- Do NOT give wishy-washy "both options are valid" non-answers — commit to a recommendation',
|
|
237
243
|
'- If you need more information, ask the main agent to delegate to the researcher',
|
|
244
|
+
'',
|
|
245
|
+
'## When to use / handoff',
|
|
246
|
+
'Invoked for hard architecture or tradeoff decisions — read-only, returns a plan.',
|
|
247
|
+
'Use AFTER researcher has gathered facts but BEFORE writer touches any files.',
|
|
248
|
+
'Return a clear recommendation and implementation plan; the orchestrator passes it to the writer.',
|
|
238
249
|
].join('\n'),
|
|
239
250
|
},
|
|
240
251
|
writer: {
|
|
@@ -279,6 +290,11 @@ export const NAMED_AGENTS = {
|
|
|
279
290
|
'2. Run the build if applicable (npm run build, tsc --noEmit, etc.).',
|
|
280
291
|
'3. If tests or build fail: attempt to fix the issue you introduced. Re-run.',
|
|
281
292
|
'4. Report: files changed, what changed in each, test results, any failures.',
|
|
293
|
+
'',
|
|
294
|
+
'## When to use / handoff',
|
|
295
|
+
'Invoked AFTER the planner produces a written plan — the writer is the SOLE agent that edits files.',
|
|
296
|
+
'Do not invoke writer until a plan exists for any multi-step change.',
|
|
297
|
+
'When the writer returns, the orchestrator invokes tester AND reviewer in parallel before surfacing results.',
|
|
282
298
|
].join('\n'),
|
|
283
299
|
},
|
|
284
300
|
tester: {
|
|
@@ -313,6 +329,11 @@ export const NAMED_AGENTS = {
|
|
|
313
329
|
'- Do NOT edit or write files — report failures so the writer agent can fix them',
|
|
314
330
|
'- Do NOT run destructive commands (no rm, no git push, no npm publish)',
|
|
315
331
|
'- Do NOT guess at fixes — diagnose only',
|
|
332
|
+
'',
|
|
333
|
+
'## When to use / handoff',
|
|
334
|
+
'Invoked in PARALLEL with reviewer, immediately after the writer returns a change.',
|
|
335
|
+
'Return a PASS or FAIL verdict with exact output; the orchestrator waits for both tester and reviewer.',
|
|
336
|
+
'NEVER skip for a code change — the orchestrator synthesizes and speaks only after both return.',
|
|
316
337
|
].join('\n'),
|
|
317
338
|
},
|
|
318
339
|
planner: {
|
|
@@ -349,6 +370,11 @@ export const NAMED_AGENTS = {
|
|
|
349
370
|
'- Do NOT edit or write files — produce a plan only',
|
|
350
371
|
'- Do NOT leave steps vague ("update the config" → say which file, which key, what value)',
|
|
351
372
|
'- Do NOT include steps that depend on runtime information you do not have',
|
|
373
|
+
'',
|
|
374
|
+
'## When to use / handoff',
|
|
375
|
+
'Invoked AFTER researcher gathers facts and BEFORE writer touches any files, when the task has multiple steps.',
|
|
376
|
+
'Do not skip on multi-step or multi-file changes — vague delegation to writer without a plan produces worse output.',
|
|
377
|
+
'Return a self-contained brief; the orchestrator passes it directly to the writer.',
|
|
352
378
|
].join('\n'),
|
|
353
379
|
},
|
|
354
380
|
reviewer: {
|
|
@@ -398,6 +424,11 @@ export const NAMED_AGENTS = {
|
|
|
398
424
|
'- Do NOT run destructive commands (no rm, no git push, no git commit, no npm publish)',
|
|
399
425
|
'- Do NOT approve a change that has a real defect just to be agreeable',
|
|
400
426
|
'- Do NOT raise trivial style nits as REJECT-worthy issues unless they break functionality',
|
|
427
|
+
'',
|
|
428
|
+
'## When to use / handoff',
|
|
429
|
+
'Invoked in PARALLEL with tester, immediately after the writer returns a change.',
|
|
430
|
+
'Return an ACCEPT or REJECT verdict with specific, actionable issue reports.',
|
|
431
|
+
'The orchestrator waits for both reviewer and tester before synthesizing and speaking to the user.',
|
|
401
432
|
].join('\n'),
|
|
402
433
|
},
|
|
403
434
|
};
|
|
@@ -9,9 +9,54 @@
|
|
|
9
9
|
* Phase 2 (future): Gemini speaks first, Claude suppressed when Gemini has HIGH confidence
|
|
10
10
|
*/
|
|
11
11
|
import { llm, DEFAULT_API_CONNECT_OPTIONS } from '@livekit/agents';
|
|
12
|
+
import { readFileSync } from 'fs';
|
|
12
13
|
import { ClaudeLLM } from './claude-llm.js';
|
|
13
14
|
import { askPipelineFastBrain } from './pipeline-fastbrain.js';
|
|
14
|
-
import { buildSummaryIndex, startIndexWatcher } from './summary-index.js';
|
|
15
|
+
import { buildSummaryIndex, startIndexWatcher, getIndexPath } from './summary-index.js';
|
|
16
|
+
// ============================================================
|
|
17
|
+
// SESSION TAIL — compact recent-conversation block prepended to every user turn
|
|
18
|
+
// so the orchestrator never loses the thread across context compaction.
|
|
19
|
+
// Controlled by:
|
|
20
|
+
// OSBORN_SESSION_TAIL — set to "0" to disable (default ON)
|
|
21
|
+
// OSBORN_SESSION_TAIL_COUNT — max user+assistant lines to keep (default 200)
|
|
22
|
+
// ============================================================
|
|
23
|
+
function buildSessionTail(sessionId, workingDir) {
|
|
24
|
+
if (process.env.OSBORN_SESSION_TAIL === '0')
|
|
25
|
+
return '';
|
|
26
|
+
if (!sessionId)
|
|
27
|
+
return '';
|
|
28
|
+
const maxLines = parseInt(process.env.OSBORN_SESSION_TAIL_COUNT || '200', 10);
|
|
29
|
+
try {
|
|
30
|
+
const indexPath = getIndexPath(sessionId, workingDir);
|
|
31
|
+
if (!indexPath)
|
|
32
|
+
return '';
|
|
33
|
+
const raw = readFileSync(indexPath, 'utf-8');
|
|
34
|
+
const allLines = raw.split('\n').filter(Boolean);
|
|
35
|
+
// Read a larger raw window (4× target) then keep the last maxLines user+assistant entries
|
|
36
|
+
const rawWindow = allLines.slice(-maxLines * 4);
|
|
37
|
+
const convLines = [];
|
|
38
|
+
for (const line of rawWindow) {
|
|
39
|
+
const parts = line.split('|');
|
|
40
|
+
// format: lineNum|byteOffset|timestamp|source|msgType|summary
|
|
41
|
+
if (parts.length < 6)
|
|
42
|
+
continue;
|
|
43
|
+
const msgType = parts[4];
|
|
44
|
+
if (msgType !== 'user' && msgType !== 'assistant')
|
|
45
|
+
continue;
|
|
46
|
+
const timestamp = parts[2];
|
|
47
|
+
const speaker = msgType === 'user' ? 'User' : 'Assistant';
|
|
48
|
+
const summary = parts.slice(5).join('|').substring(0, 150);
|
|
49
|
+
convLines.push(`${timestamp} ${speaker}: ${summary}`);
|
|
50
|
+
}
|
|
51
|
+
const kept = convLines.slice(-maxLines);
|
|
52
|
+
if (kept.length === 0)
|
|
53
|
+
return '';
|
|
54
|
+
return ['<session_tail>', ...kept, '</session_tail>'].join('\n');
|
|
55
|
+
}
|
|
56
|
+
catch {
|
|
57
|
+
return '';
|
|
58
|
+
}
|
|
59
|
+
}
|
|
15
60
|
export class PipelineDirectLLM extends llm.LLM {
|
|
16
61
|
#claudeLLM;
|
|
17
62
|
#opts;
|
|
@@ -88,6 +133,17 @@ export class PipelineDirectLLM extends llm.LLM {
|
|
|
88
133
|
}
|
|
89
134
|
}
|
|
90
135
|
console.log(`📥 [pipeline] chat() call #${callN} (${userText.length} chars): "${userText}"`);
|
|
136
|
+
// ── SESSION TAIL injection ────────────────────────────────────────────────
|
|
137
|
+
// Prepend a compact recent-conversation block so the orchestrator never
|
|
138
|
+
// loses thread across compaction. Runs synchronously (readFileSync) but the
|
|
139
|
+
// index file is ≤1MB and kept hot in the OS page cache — negligible latency.
|
|
140
|
+
// Skipped for meeting chunks (they are not real user turns).
|
|
141
|
+
const sessionTailBlock = (!userText.startsWith('[MEETING'))
|
|
142
|
+
? buildSessionTail(this.#claudeLLM.sessionId, this.#opts.workingDirectory || process.cwd())
|
|
143
|
+
: '';
|
|
144
|
+
if (sessionTailBlock) {
|
|
145
|
+
console.log(`🧵 [pipeline] Session tail: ${sessionTailBlock.split('\n').length - 2} lines ready`);
|
|
146
|
+
}
|
|
91
147
|
// Always check the pending playback context — it can carry two independent
|
|
92
148
|
// signals: (a) an actual interruption (spokenText + recentMessages) when the
|
|
93
149
|
// user cut Osborn off mid-TTS, OR (b) suppressed text generated by the SDK
|
|
@@ -171,11 +227,27 @@ export class PipelineDirectLLM extends llm.LLM {
|
|
|
171
227
|
// when text is non-empty.
|
|
172
228
|
enrichedMessage = userText;
|
|
173
229
|
}
|
|
174
|
-
// Modify the last user message in chatCtx
|
|
230
|
+
// Modify the last user message in chatCtx — prepend session tail if available
|
|
231
|
+
const finalMessage = sessionTailBlock
|
|
232
|
+
? sessionTailBlock + '\n\n' + enrichedMessage
|
|
233
|
+
: enrichedMessage;
|
|
234
|
+
for (let i = chatCtx.items.length - 1; i >= 0; i--) {
|
|
235
|
+
const item = chatCtx.items[i];
|
|
236
|
+
if (item.type === 'message' && item.role === 'user') {
|
|
237
|
+
item.content = [finalMessage];
|
|
238
|
+
break;
|
|
239
|
+
}
|
|
240
|
+
}
|
|
241
|
+
}
|
|
242
|
+
else if (sessionTailBlock && userText.trim()) {
|
|
243
|
+
// No interrupt context — inject session tail directly into the user message
|
|
175
244
|
for (let i = chatCtx.items.length - 1; i >= 0; i--) {
|
|
176
245
|
const item = chatCtx.items[i];
|
|
177
246
|
if (item.type === 'message' && item.role === 'user') {
|
|
178
|
-
item.content
|
|
247
|
+
const original = Array.isArray(item.content)
|
|
248
|
+
? item.content.filter((c) => typeof c === 'string').join('\n')
|
|
249
|
+
: String(item.content ?? '');
|
|
250
|
+
item.content = [sessionTailBlock + '\n\n' + original];
|
|
179
251
|
break;
|
|
180
252
|
}
|
|
181
253
|
}
|
|
@@ -109,6 +109,13 @@ THE SUB-AGENTS:
|
|
|
109
109
|
· reasoner (Opus) — architecture decisions, complex tradeoffs, implementation planning. Read-only.
|
|
110
110
|
· writer (Sonnet) — ALL file changes outside the workspace. Verifies before, runs tests after. The ONLY agent with write access outside the workspace.
|
|
111
111
|
· NEVER use the SDK's built-in 'general-purpose' agent — it is not configured for this project and will hit write blocks. Always pick researcher, reasoner, or writer explicitly.
|
|
112
|
+
|
|
113
|
+
DIVISION OF LABOR — follow this chain by default for any substantive or code task:
|
|
114
|
+
researcher (gather facts) → planner (write a step plan for multi-step work) → writer (execute the plan) → tester AND reviewer in parallel (verify the change) → you synthesize and speak.
|
|
115
|
+
For a quick factual query: researcher only → you speak.
|
|
116
|
+
NEVER skip tester and reviewer after a code change.
|
|
117
|
+
You may OVERRIDE this chain at any time — stop a running agent, inject, or reorder — because you can see each task's live state. The chain is the default, not a cage.
|
|
118
|
+
Surfacing findings and communicating to the user is YOUR job, not a sub-agent's.
|
|
112
119
|
</turn-shape>
|
|
113
120
|
|
|
114
121
|
<co-direction>
|
package/dist/prompts.js
CHANGED
|
@@ -548,6 +548,13 @@ You have three named agents available via the Task tool:
|
|
|
548
548
|
· reasoner — Opus, deep analysis. Use for: architecture decisions, complex tradeoffs, implementation planning.
|
|
549
549
|
· writer — Sonnet, file changes. Use for: ALL file creation, editing, modification. Verifies before and after changes.
|
|
550
550
|
|
|
551
|
+
DIVISION OF LABOR — follow this chain by default for any substantive or code task:
|
|
552
|
+
researcher (gather facts) → planner (write a step plan for multi-step work) → writer (execute the plan) → tester AND reviewer in parallel (verify the change) → you synthesize and speak.
|
|
553
|
+
For a quick factual query: researcher only → you speak.
|
|
554
|
+
NEVER skip tester and reviewer after a code change.
|
|
555
|
+
You may OVERRIDE this chain at any time — stop a running agent, inject a step, or reorder — because you can see each task's live state. The chain is the default, not a cage.
|
|
556
|
+
Surfacing findings and communicating to the user is YOUR job, not a sub-agent's.
|
|
557
|
+
|
|
551
558
|
DELEGATION: For any task needing 3+ tool calls, delegate to the appropriate agent instead of doing it yourself.
|
|
552
559
|
Quick lookups (1-2 calls) you can do directly. Everything else goes to an agent.
|
|
553
560
|
|