osborn 0.9.134 → 0.9.136

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -205,6 +205,20 @@ if [ -d "$HOME/.claude/projects" ]; then
205
205
  done
206
206
  fi
207
207
 
208
+
209
+ # === Sync volume npm-global with image osborn version ===
210
+ # PATH prefers /workspace/.npm-global/bin over /usr/local/bin — so after an
211
+ # image-swap the volume copy is stale. Detect mismatch and sync on boot so
212
+ # the dashboard update button (image-swap + restart) actually takes effect.
213
+ VOL_OSBORN_VER=$(grep -m1 '"version"' /workspace/.npm-global/lib/node_modules/osborn/package.json 2>/dev/null | sed 's/.*"version"[^"]*"\([^"]*\)".*/\1/' || echo "none")
214
+ if [ -n "$IMAGE_SEED_VERSION" ] && [ "$IMAGE_SEED_VERSION" != "$VOL_OSBORN_VER" ]; then
215
+ echo "[sandbox-d] volume osborn@${VOL_OSBORN_VER} != image@${IMAGE_SEED_VERSION} -- syncing"
216
+ rm -rf /workspace/.npm-global/lib/node_modules/.osborn-* 2>/dev/null
217
+ npm install -g "osborn@${IMAGE_SEED_VERSION}" --no-audit --no-fund 2>&1 | tail -3
218
+ echo "[sandbox-d] sync done: osborn@${IMAGE_SEED_VERSION} on volume"
219
+ else
220
+ echo "[sandbox-d] volume osborn@${VOL_OSBORN_VER} matches image -- no sync needed"
221
+ fi
208
222
  echo "[sandbox-d] exec'ing osborn (no chroot, HOME=$HOME)"
209
223
  exec osborn
210
224
  ENTRYPOINT
@@ -199,6 +199,12 @@ export const NAMED_AGENTS = {
199
199
  '- Do NOT edit or write any files',
200
200
  '- Do NOT run destructive commands (no rm, no git push, no npm publish)',
201
201
  '- If you need clarification, ask the main agent — it will relay to the user if needed',
202
+ '',
203
+ '## When to use / handoff',
204
+ 'Invoked FIRST for any task requiring facts, codebase exploration, or web research.',
205
+ 'Return findings to the orchestrator — never directly to the user.',
206
+ 'Run several researchers in parallel when there are independent threads to investigate.',
207
+ 'Hand off back to the orchestrator; it decides whether to invoke planner or writer next.',
202
208
  ].join('\n'),
203
209
  },
204
210
  reasoner: {
@@ -235,6 +241,11 @@ export const NAMED_AGENTS = {
235
241
  '- Do NOT edit or write files — return a plan for the writer agent',
236
242
  '- Do NOT give wishy-washy "both options are valid" non-answers — commit to a recommendation',
237
243
  '- If you need more information, ask the main agent to delegate to the researcher',
244
+ '',
245
+ '## When to use / handoff',
246
+ 'Invoked for hard architecture or tradeoff decisions — read-only, returns a plan.',
247
+ 'Use AFTER researcher has gathered facts but BEFORE writer touches any files.',
248
+ 'Return a clear recommendation and implementation plan; the orchestrator passes it to the writer.',
238
249
  ].join('\n'),
239
250
  },
240
251
  writer: {
@@ -279,6 +290,11 @@ export const NAMED_AGENTS = {
279
290
  '2. Run the build if applicable (npm run build, tsc --noEmit, etc.).',
280
291
  '3. If tests or build fail: attempt to fix the issue you introduced. Re-run.',
281
292
  '4. Report: files changed, what changed in each, test results, any failures.',
293
+ '',
294
+ '## When to use / handoff',
295
+ 'Invoked AFTER the planner produces a written plan — the writer is the SOLE agent that edits files.',
296
+ 'Do not invoke writer until a plan exists for any multi-step change.',
297
+ 'When the writer returns, the orchestrator invokes tester AND reviewer in parallel before surfacing results.',
282
298
  ].join('\n'),
283
299
  },
284
300
  tester: {
@@ -313,6 +329,11 @@ export const NAMED_AGENTS = {
313
329
  '- Do NOT edit or write files — report failures so the writer agent can fix them',
314
330
  '- Do NOT run destructive commands (no rm, no git push, no npm publish)',
315
331
  '- Do NOT guess at fixes — diagnose only',
332
+ '',
333
+ '## When to use / handoff',
334
+ 'Invoked in PARALLEL with reviewer, immediately after the writer returns a change.',
335
+ 'Return a PASS or FAIL verdict with exact output; the orchestrator waits for both tester and reviewer.',
336
+ 'NEVER skip for a code change — the orchestrator synthesizes and speaks only after both return.',
316
337
  ].join('\n'),
317
338
  },
318
339
  planner: {
@@ -349,6 +370,11 @@ export const NAMED_AGENTS = {
349
370
  '- Do NOT edit or write files — produce a plan only',
350
371
  '- Do NOT leave steps vague ("update the config" → say which file, which key, what value)',
351
372
  '- Do NOT include steps that depend on runtime information you do not have',
373
+ '',
374
+ '## When to use / handoff',
375
+ 'Invoked AFTER researcher gathers facts and BEFORE writer touches any files, when the task has multiple steps.',
376
+ 'Do not skip on multi-step or multi-file changes — vague delegation to writer without a plan produces worse output.',
377
+ 'Return a self-contained brief; the orchestrator passes it directly to the writer.',
352
378
  ].join('\n'),
353
379
  },
354
380
  reviewer: {
@@ -398,6 +424,11 @@ export const NAMED_AGENTS = {
398
424
  '- Do NOT run destructive commands (no rm, no git push, no git commit, no npm publish)',
399
425
  '- Do NOT approve a change that has a real defect just to be agreeable',
400
426
  '- Do NOT raise trivial style nits as REJECT-worthy issues unless they break functionality',
427
+ '',
428
+ '## When to use / handoff',
429
+ 'Invoked in PARALLEL with tester, immediately after the writer returns a change.',
430
+ 'Return an ACCEPT or REJECT verdict with specific, actionable issue reports.',
431
+ 'The orchestrator waits for both reviewer and tester before synthesizing and speaking to the user.',
401
432
  ].join('\n'),
402
433
  },
403
434
  };
@@ -9,9 +9,54 @@
9
9
  * Phase 2 (future): Gemini speaks first, Claude suppressed when Gemini has HIGH confidence
10
10
  */
11
11
  import { llm, DEFAULT_API_CONNECT_OPTIONS } from '@livekit/agents';
12
+ import { readFileSync } from 'fs';
12
13
  import { ClaudeLLM } from './claude-llm.js';
13
14
  import { askPipelineFastBrain } from './pipeline-fastbrain.js';
14
- import { buildSummaryIndex, startIndexWatcher } from './summary-index.js';
15
+ import { buildSummaryIndex, startIndexWatcher, getIndexPath } from './summary-index.js';
16
+ // ============================================================
17
+ // SESSION TAIL — compact recent-conversation block prepended to every user turn
18
+ // so the orchestrator never loses the thread across context compaction.
19
+ // Controlled by:
20
+ // OSBORN_SESSION_TAIL — set to "0" to disable (default ON)
21
+ // OSBORN_SESSION_TAIL_COUNT — max user+assistant lines to keep (default 200)
22
+ // ============================================================
23
+ function buildSessionTail(sessionId, workingDir) {
24
+ if (process.env.OSBORN_SESSION_TAIL === '0')
25
+ return '';
26
+ if (!sessionId)
27
+ return '';
28
+ const maxLines = parseInt(process.env.OSBORN_SESSION_TAIL_COUNT || '200', 10);
29
+ try {
30
+ const indexPath = getIndexPath(sessionId, workingDir);
31
+ if (!indexPath)
32
+ return '';
33
+ const raw = readFileSync(indexPath, 'utf-8');
34
+ const allLines = raw.split('\n').filter(Boolean);
35
+ // Read a larger raw window (4× target) then keep the last maxLines user+assistant entries
36
+ const rawWindow = allLines.slice(-maxLines * 4);
37
+ const convLines = [];
38
+ for (const line of rawWindow) {
39
+ const parts = line.split('|');
40
+ // format: lineNum|byteOffset|timestamp|source|msgType|summary
41
+ if (parts.length < 6)
42
+ continue;
43
+ const msgType = parts[4];
44
+ if (msgType !== 'user' && msgType !== 'assistant')
45
+ continue;
46
+ const timestamp = parts[2];
47
+ const speaker = msgType === 'user' ? 'User' : 'Assistant';
48
+ const summary = parts.slice(5).join('|').substring(0, 150);
49
+ convLines.push(`${timestamp} ${speaker}: ${summary}`);
50
+ }
51
+ const kept = convLines.slice(-maxLines);
52
+ if (kept.length === 0)
53
+ return '';
54
+ return ['<session_tail>', ...kept, '</session_tail>'].join('\n');
55
+ }
56
+ catch {
57
+ return '';
58
+ }
59
+ }
15
60
  export class PipelineDirectLLM extends llm.LLM {
16
61
  #claudeLLM;
17
62
  #opts;
@@ -88,6 +133,17 @@ export class PipelineDirectLLM extends llm.LLM {
88
133
  }
89
134
  }
90
135
  console.log(`📥 [pipeline] chat() call #${callN} (${userText.length} chars): "${userText}"`);
136
+ // ── SESSION TAIL injection ────────────────────────────────────────────────
137
+ // Prepend a compact recent-conversation block so the orchestrator never
138
+ // loses thread across compaction. Runs synchronously (readFileSync) but the
139
+ // index file is ≤1MB and kept hot in the OS page cache — negligible latency.
140
+ // Skipped for meeting chunks (they are not real user turns).
141
+ const sessionTailBlock = (!userText.startsWith('[MEETING'))
142
+ ? buildSessionTail(this.#claudeLLM.sessionId, this.#opts.workingDirectory || process.cwd())
143
+ : '';
144
+ if (sessionTailBlock) {
145
+ console.log(`🧵 [pipeline] Session tail: ${sessionTailBlock.split('\n').length - 2} lines ready`);
146
+ }
91
147
  // Always check the pending playback context — it can carry two independent
92
148
  // signals: (a) an actual interruption (spokenText + recentMessages) when the
93
149
  // user cut Osborn off mid-TTS, OR (b) suppressed text generated by the SDK
@@ -171,11 +227,27 @@ export class PipelineDirectLLM extends llm.LLM {
171
227
  // when text is non-empty.
172
228
  enrichedMessage = userText;
173
229
  }
174
- // Modify the last user message in chatCtx
230
+ // Modify the last user message in chatCtx — prepend session tail if available
231
+ const finalMessage = sessionTailBlock
232
+ ? sessionTailBlock + '\n\n' + enrichedMessage
233
+ : enrichedMessage;
234
+ for (let i = chatCtx.items.length - 1; i >= 0; i--) {
235
+ const item = chatCtx.items[i];
236
+ if (item.type === 'message' && item.role === 'user') {
237
+ item.content = [finalMessage];
238
+ break;
239
+ }
240
+ }
241
+ }
242
+ else if (sessionTailBlock && userText.trim()) {
243
+ // No interrupt context — inject session tail directly into the user message
175
244
  for (let i = chatCtx.items.length - 1; i >= 0; i--) {
176
245
  const item = chatCtx.items[i];
177
246
  if (item.type === 'message' && item.role === 'user') {
178
- item.content = [enrichedMessage];
247
+ const original = Array.isArray(item.content)
248
+ ? item.content.filter((c) => typeof c === 'string').join('\n')
249
+ : String(item.content ?? '');
250
+ item.content = [sessionTailBlock + '\n\n' + original];
179
251
  break;
180
252
  }
181
253
  }
@@ -109,6 +109,13 @@ THE SUB-AGENTS:
109
109
  · reasoner (Opus) — architecture decisions, complex tradeoffs, implementation planning. Read-only.
110
110
  · writer (Sonnet) — ALL file changes outside the workspace. Verifies before, runs tests after. The ONLY agent with write access outside the workspace.
111
111
  · NEVER use the SDK's built-in 'general-purpose' agent — it is not configured for this project and will hit write blocks. Always pick researcher, reasoner, or writer explicitly.
112
+
113
+ DIVISION OF LABOR — follow this chain by default for any substantive or code task:
114
+ researcher (gather facts) → planner (write a step plan for multi-step work) → writer (execute the plan) → tester AND reviewer in parallel (verify the change) → you synthesize and speak.
115
+ For a quick factual query: researcher only → you speak.
116
+ NEVER skip tester and reviewer after a code change.
117
+ You may OVERRIDE this chain at any time — stop a running agent, inject, or reorder — because you can see each task's live state. The chain is the default, not a cage.
118
+ Surfacing findings and communicating to the user is YOUR job, not a sub-agent's.
112
119
  </turn-shape>
113
120
 
114
121
  <co-direction>
package/dist/prompts.js CHANGED
@@ -548,6 +548,13 @@ You have three named agents available via the Task tool:
548
548
  · reasoner — Opus, deep analysis. Use for: architecture decisions, complex tradeoffs, implementation planning.
549
549
  · writer — Sonnet, file changes. Use for: ALL file creation, editing, modification. Verifies before and after changes.
550
550
 
551
+ DIVISION OF LABOR — follow this chain by default for any substantive or code task:
552
+ researcher (gather facts) → planner (write a step plan for multi-step work) → writer (execute the plan) → tester AND reviewer in parallel (verify the change) → you synthesize and speak.
553
+ For a quick factual query: researcher only → you speak.
554
+ NEVER skip tester and reviewer after a code change.
555
+ You may OVERRIDE this chain at any time — stop a running agent, inject a step, or reorder — because you can see each task's live state. The chain is the default, not a cage.
556
+ Surfacing findings and communicating to the user is YOUR job, not a sub-agent's.
557
+
551
558
  DELEGATION: For any task needing 3+ tool calls, delegate to the appropriate agent instead of doing it yourself.
552
559
  Quick lookups (1-2 calls) you can do directly. Everything else goes to an agent.
553
560
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.134",
3
+ "version": "0.9.136",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {