osborn 0.9.132 → 0.9.134

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -60,6 +60,24 @@ export declare const NAMED_AGENTS: {
60
60
  model: string;
61
61
  prompt: string;
62
62
  };
63
+ tester: {
64
+ description: string;
65
+ tools: string[];
66
+ model: string;
67
+ prompt: string;
68
+ };
69
+ planner: {
70
+ description: string;
71
+ tools: string[];
72
+ model: string;
73
+ prompt: string;
74
+ };
75
+ reviewer: {
76
+ description: string;
77
+ tools: string[];
78
+ model: string;
79
+ prompt: string;
80
+ };
63
81
  };
64
82
  /**
65
83
  * Claude LLM - Wraps Claude Agent SDK for LiveKit
@@ -281,6 +281,125 @@ export const NAMED_AGENTS = {
281
281
  '4. Report: files changed, what changed in each, test results, any failures.',
282
282
  ].join('\n'),
283
283
  },
284
+ tester: {
285
+ description: [
286
+ 'Test-runner agent (Sonnet). Use for: running test suites, executing builds, interpreting',
287
+ 'CI failures, checking compilation errors, verifying that a change did not break anything.',
288
+ 'Returns structured pass/fail results with exact output — does NOT edit files.',
289
+ ].join(' '),
290
+ tools: ['Bash', 'Read', 'Glob', 'Grep'],
291
+ model: 'sonnet',
292
+ prompt: [
293
+ 'You are Osborn\'s tester agent. Your job is running tests and builds, then reporting results.',
294
+ '',
295
+ '## Your role',
296
+ 'Execute test suites, build commands, and linters. Interpret failures clearly.',
297
+ 'You are a quality gate — find out whether the code works, and say exactly what broke.',
298
+ '',
299
+ '## How to work',
300
+ '1. Identify the correct test / build command from package.json, Makefile, or the task brief.',
301
+ '2. Run it with Bash. Capture stdout + stderr in full.',
302
+ '3. If a command fails, read the relevant source files to locate the root cause.',
303
+ '4. Cap yourself at 6-8 tool calls unless the investigation clearly requires more.',
304
+ '',
305
+ '## What to return',
306
+ '- RESULT: PASS or FAIL (one word, first line)',
307
+ '- COMMAND: the exact command you ran',
308
+ '- OUTPUT: relevant excerpt (errors, failing test names, line numbers)',
309
+ '- ROOT CAUSE: your diagnosis of why it failed (if applicable)',
310
+ '- What you checked but found to be unrelated',
311
+ '',
312
+ '## What NOT to do',
313
+ '- Do NOT edit or write files — report failures so the writer agent can fix them',
314
+ '- Do NOT run destructive commands (no rm, no git push, no npm publish)',
315
+ '- Do NOT guess at fixes — diagnose only',
316
+ ].join('\n'),
317
+ },
318
+ planner: {
319
+ description: [
320
+ 'Planning agent (Opus). Use for: decomposing a large or ambiguous request into a concrete,',
321
+ 'ordered sequence of atomic writer-safe steps. Returns a self-contained brief the writer',
322
+ 'can execute without further clarification. Slow but thorough — only use for genuinely',
323
+ 'complex multi-file changes or when the approach is uncertain.',
324
+ ].join(' '),
325
+ tools: ['Read', 'Glob', 'Grep', 'WebSearch'],
326
+ model: 'opus',
327
+ prompt: [
328
+ 'You are Osborn\'s planning agent. Your job is to decompose complex tasks into clear, atomic steps.',
329
+ '',
330
+ '## Your role',
331
+ 'Turn a vague or large request into a precise, ordered implementation plan the writer can execute',
332
+ 'step by step without guessing. You are the bridge between "what" and "how".',
333
+ '',
334
+ '## How to work',
335
+ '1. Read enough of the codebase to understand the current structure (Glob, Grep, Read).',
336
+ '2. Identify every file that needs to change and why.',
337
+ '3. Order the steps so each one is independently safe (no step depends on a later one).',
338
+ '4. Flag any decision the writer should NOT make alone — surface it as an open question.',
339
+ '',
340
+ '## What to return',
341
+ 'A self-contained brief with:',
342
+ '- GOAL: one sentence summary of the outcome',
343
+ '- CONTEXT: relevant file paths, existing patterns, constraints the writer must respect',
344
+ '- STEPS: numbered, atomic steps (one logical change per step; include file path + what to change)',
345
+ '- OPEN QUESTIONS: anything genuinely ambiguous that needs user input before proceeding',
346
+ '- VERIFY: how the writer should confirm the change worked (test command, manual check, etc.)',
347
+ '',
348
+ '## What NOT to do',
349
+ '- Do NOT edit or write files — produce a plan only',
350
+ '- Do NOT leave steps vague ("update the config" → say which file, which key, what value)',
351
+ '- Do NOT include steps that depend on runtime information you do not have',
352
+ ].join('\n'),
353
+ },
354
+ reviewer: {
355
+ description: [
356
+ 'Code-review agent (Opus). Use for: the VERIFY step in a generator-verifier loop — after the',
357
+ 'writer completes a change, the reviewer reads the diff, checks correctness, spec/requirement',
358
+ 'adherence, obvious bugs, and security issues, then returns an ACCEPT or REJECT verdict with',
359
+ 'specific, actionable feedback. Does NOT edit files — reports so the writer can fix.',
360
+ ].join(' '),
361
+ tools: ['Read', 'Glob', 'Grep', 'Bash'],
362
+ model: 'opus',
363
+ prompt: [
364
+ 'You are Osborn\'s reviewer agent. You are the VERIFY step in a generator-verifier loop.',
365
+ '',
366
+ '## Your role',
367
+ 'Read the writer\'s completed change (via git diff or by reading modified files), then produce',
368
+ 'a structured verdict: ACCEPT or REJECT. You do NOT edit files — you report findings so the',
369
+ 'single writer agent can fix them. You are the quality gate between a change and merge.',
370
+ '',
371
+ '## Bash is read-only inspection only',
372
+ 'You may run: git diff, git log, git status, git show, npm run build, npm test, eslint,',
373
+ 'tsc --noEmit, and similar lint/test/security-scan commands.',
374
+ 'You must NOT run: rm, git push, git commit, git add, npm publish, or any destructive command.',
375
+ '',
376
+ '## How to work',
377
+ '1. Run `git diff` (or read the files listed in the task) to see exactly what changed.',
378
+ '2. Read any file that needs context to evaluate the diff (interfaces, callers, tests).',
379
+ '3. Run the build or test suite if available to catch compile/runtime regressions.',
380
+ '4. Check against the spec or requirement provided in the task brief.',
381
+ '5. Look for: logic errors, missing edge cases, security issues (injection, path traversal,',
382
+ ' credential exposure), broken types, spec deviations, unintended side-effects.',
383
+ '6. Cap yourself at 10 tool calls unless the review clearly requires more.',
384
+ '',
385
+ '## What to return',
386
+ 'Structure your response EXACTLY as follows:',
387
+ '',
388
+ 'VERDICT: ACCEPT | REJECT',
389
+ '',
390
+ 'ISSUES (each on its own line, only present if VERDICT is REJECT):',
391
+ ' - <file>:<line> — <concise description of the problem and why it matters>',
392
+ '',
393
+ 'WHAT IT CHECKED-AND-CLEARED:',
394
+ ' - <each item you verified and found correct — be specific, not generic>',
395
+ '',
396
+ '## What NOT to do',
397
+ '- Do NOT edit or write any files — issue reports only; the writer fixes',
398
+ '- Do NOT run destructive commands (no rm, no git push, no git commit, no npm publish)',
399
+ '- Do NOT approve a change that has a real defect just to be agreeable',
400
+ '- Do NOT raise trivial style nits as REJECT-worthy issues unless they break functionality',
401
+ ].join('\n'),
402
+ },
284
403
  };
285
404
  const RESEARCH_TOOLS = [
286
405
  'Read', 'Write', 'Edit', 'Glob', 'Grep',
package/dist/index.js CHANGED
@@ -1689,6 +1689,8 @@ async function main() {
1689
1689
  let participantConnectedHandler = null;
1690
1690
  let currentAgent = null; // For updateChatCtx() context injection
1691
1691
  let currentLLM = null;
1692
+ const slots = new Map();
1693
+ let focusedSlotId = null; // reserved for future focused-slot wiring; not used yet
1692
1694
  // Agent-side "alone in room" leave timer (see Room-presence lifecycle note up
1693
1695
  // top). Armed in ParticipantDisconnected once a user has left; if no one
1694
1696
  // rejoins within the grace window the agent leaves LiveKit on its own.
@@ -1857,6 +1859,100 @@ async function main() {
1857
1859
  console.error(`❌ killCurrentLLM(${reason}) failed:`, err instanceof Error ? err.message : err);
1858
1860
  }
1859
1861
  }
1862
+ // ============================================================
1863
+ // SLICE 1: spawnBackgroundSession
1864
+ // Creates a fully headless ClaudeLLM subprocess that resumes an
1865
+ // existing session. No AgentSession, no room, no TTS/voice.
1866
+ // Output events are forwarded to the frontend data channel only,
1867
+ // tagged with the slot id so the frontend can distinguish them.
1868
+ // ============================================================
1869
+ async function spawnBackgroundSession(sessionId, bgWorkingDir) {
1870
+ // Hard cap: focused session + background slots must not exceed 3 total.
1871
+ // slots Map holds ONLY background slots; focused session is separate.
1872
+ if (slots.size >= 2) {
1873
+ const msg = `Background session cap reached (${slots.size}/2 slots in use). Kill an existing background slot before spawning a new one.`;
1874
+ console.warn(`⚠️ spawnBackgroundSession: ${msg}`);
1875
+ await sendToFrontend({ type: 'background_session_error', slotId: sessionId, error: msg });
1876
+ return;
1877
+ }
1878
+ const slotWorkingDir = bgWorkingDir || workingDir;
1879
+ const slotId = sessionId; // use sessionId as the slot id for traceability
1880
+ console.log(`🔲 Spawning background slot: sessionId=${slotId.substring(0, 8)} cwd=${slotWorkingDir}`);
1881
+ // Create a headless LLM instance — reuse the same factory, no voice options.
1882
+ // permissionMode bypassPermissions: no dialogs (headless — nobody to click Allow).
1883
+ // No AgentSession, no room involvement.
1884
+ const bgLLM = createClaudeLLM({
1885
+ workingDirectory: slotWorkingDir,
1886
+ sessionBaseDir: slotWorkingDir,
1887
+ resumeSessionId: sessionId,
1888
+ voiceMode: 'direct', // picks up the research system prompt
1889
+ skipTTSQueue: false, // not going through TTS at all
1890
+ permissionMode: 'bypassPermissions',
1891
+ });
1892
+ // Register slot BEFORE cold-starting so list_slots is accurate from the
1893
+ // first moment even if the subprocess takes a few seconds to init.
1894
+ const slot = { id: slotId, llm: bgLLM, isFocused: false, workingDir: slotWorkingDir };
1895
+ slots.set(slotId, slot);
1896
+ // Dedicated EventEmitter for this background slot's SDK events.
1897
+ // Forwards every event to the frontend data channel tagged with slotId.
1898
+ // NEVER touches session.say() / TTS / voice.
1899
+ const { EventEmitter: SlotEventEmitter } = await import('node:events');
1900
+ const bgEmitter = new SlotEventEmitter();
1901
+ bgEmitter.on('assistant_text', ({ text }) => {
1902
+ sendToFrontend({ type: 'claude_output', slotId, text }).catch(() => { });
1903
+ });
1904
+ bgEmitter.on('tool_use', ({ name, input, agentRole }) => {
1905
+ sendToFrontend({ type: 'tool_use', slotId, name, input, agentRole }).catch(() => { });
1906
+ });
1907
+ bgEmitter.on('tool_result', ({ name, response, agentRole }) => {
1908
+ sendToFrontend({ type: 'tool_result', slotId, name, response, agentRole }).catch(() => { });
1909
+ });
1910
+ bgEmitter.on('tool_blocked', ({ name, reason }) => {
1911
+ sendToFrontend({ type: 'tool_blocked', slotId, name, reason }).catch(() => { });
1912
+ });
1913
+ bgEmitter.on('session_id', ({ sessionId: sid }) => {
1914
+ sendToFrontend({ type: 'background_session_ready', slotId, sessionId: sid }).catch(() => { });
1915
+ });
1916
+ bgEmitter.on('session_resume_failed', ({ requestedSessionId, actualSessionId }) => {
1917
+ console.error(`❌ Background slot ${slotId.substring(0, 8)}: resume failed — requested ${requestedSessionId?.substring(0, 8)} got ${actualSessionId?.substring(0, 8)}`);
1918
+ sendToFrontend({ type: 'background_session_error', slotId, error: 'Session resume failed' }).catch(() => { });
1919
+ });
1920
+ // Wire up the LLM's internal EventEmitter so hook-emitted events also flow.
1921
+ bgLLM.events.on('session_id', (data) => bgEmitter.emit('session_id', data));
1922
+ bgLLM.events.on('tool_blocked', (data) => bgEmitter.emit('tool_blocked', data));
1923
+ // Build minimal sdkOptions for the headless cold start.
1924
+ // resume: sessionId brings up real prior context from JSONL.
1925
+ // env: CLAUDE_CODE_DISABLE_AUTO_MEMORY prevents concurrent subprocesses
1926
+ // from racing the shared ~/.claude/CLAUDE.md memory file.
1927
+ // CLAUDE_CONFIG_DIR is intentionally NOT overridden — shared config is fine.
1928
+ const bgEnv = {};
1929
+ for (const [k, v] of Object.entries(process.env)) {
1930
+ if (v !== undefined)
1931
+ bgEnv[k] = v;
1932
+ }
1933
+ bgEnv['CLAUDE_CODE_DISABLE_AUTO_MEMORY'] = '1';
1934
+ const bgSdkOptions = {
1935
+ cwd: slotWorkingDir,
1936
+ resume: sessionId,
1937
+ permissionMode: 'bypassPermissions',
1938
+ enableFileCheckpointing: false,
1939
+ settingSources: ['project', 'user'],
1940
+ env: bgEnv,
1941
+ };
1942
+ // Cold-start: deliver an init message so the subprocess actually spawns
1943
+ // and begins consuming the session JSONL.
1944
+ const KICKOFF_TEXT = '[BACKGROUND_INIT] You have been resumed in headless background mode. No voice session is active. Acknowledge receipt with a single brief line.';
1945
+ bgLLM.pushMessage(KICKOFF_TEXT, bgSdkOptions, {
1946
+ onSessionId: (sid) => {
1947
+ console.log(`✅ Background slot ${slotId.substring(0, 8)}: session confirmed ${sid.substring(0, 8)}`);
1948
+ bgEmitter.emit('session_id', { sessionId: sid });
1949
+ },
1950
+ onCheckpoint: (_ckpt) => { },
1951
+ eventEmitter: bgEmitter,
1952
+ });
1953
+ console.log(`✅ Background slot registered: id=${slotId.substring(0, 8)} workingDir=${slotWorkingDir}`);
1954
+ await sendToFrontend({ type: 'background_session_spawned', slotId, workingDir: slotWorkingDir });
1955
+ }
1860
1956
  let localParticipant = null;
1861
1957
  let agentState = 'initializing';
1862
1958
  // Session-level always-allow list: paths the user has approved for this session without prompting
@@ -5607,6 +5703,71 @@ async function main() {
5607
5703
  }
5608
5704
  }
5609
5705
  }
5706
+ // ============================================================
5707
+ // SLICE 1: HEADLESS BACKGROUND SESSION COMMANDS
5708
+ // spawn_background_session — boot a second Claude subprocess that
5709
+ // resumes an existing session with NO voice involvement.
5710
+ // list_slots — return all live slots (focused + background) so the
5711
+ // frontend can verify both are alive.
5712
+ // ============================================================
5713
+ else if (data.type === 'spawn_background_session') {
5714
+ const targetSessionId = data.sessionId;
5715
+ const bgDir = data.workingDir;
5716
+ let resolvedSessionId = targetSessionId;
5717
+ if (!resolvedSessionId) {
5718
+ // Pick the most recently modified session, excluding the focused one.
5719
+ const allSess = await listAllClaudeSessions(50);
5720
+ const focusedId = currentLLM?.sessionId || currentResumeSessionId || null;
5721
+ const candidate = allSess.find(s => s.sessionId !== focusedId);
5722
+ resolvedSessionId = candidate?.sessionId ?? undefined;
5723
+ }
5724
+ if (!resolvedSessionId) {
5725
+ await sendToFrontend({
5726
+ type: 'background_session_error',
5727
+ slotId: null,
5728
+ error: 'No eligible session found to spawn as background slot',
5729
+ });
5730
+ }
5731
+ else if (slots.has(resolvedSessionId)) {
5732
+ await sendToFrontend({
5733
+ type: 'background_session_error',
5734
+ slotId: resolvedSessionId,
5735
+ error: `Session ${resolvedSessionId.substring(0, 8)} is already running as a background slot`,
5736
+ });
5737
+ }
5738
+ else {
5739
+ spawnBackgroundSession(resolvedSessionId, bgDir).catch((err) => {
5740
+ console.error('❌ spawnBackgroundSession failed:', err instanceof Error ? err.message : err);
5741
+ sendToFrontend({
5742
+ type: 'background_session_error',
5743
+ slotId: resolvedSessionId,
5744
+ error: err instanceof Error ? err.message : String(err),
5745
+ }).catch(() => { });
5746
+ });
5747
+ }
5748
+ }
5749
+ else if (data.type === 'list_slots') {
5750
+ // Return all live slots: the focused session + every background slot.
5751
+ const focusedId = currentLLM?.sessionId || currentResumeSessionId || null;
5752
+ const focusedEntry = currentLLM
5753
+ ? [{
5754
+ id: focusedId || '(pending)',
5755
+ isFocused: true,
5756
+ workingDir,
5757
+ hasSession: currentLLM.hasSession?.() ?? false,
5758
+ }]
5759
+ : [];
5760
+ const bgEntries = [...slots.values()].map(s => ({
5761
+ id: s.id,
5762
+ isFocused: s.isFocused,
5763
+ workingDir: s.workingDir,
5764
+ hasSession: s.llm.hasSession?.() ?? false,
5765
+ }));
5766
+ await sendToFrontend({
5767
+ type: 'slots_list',
5768
+ slots: [...focusedEntry, ...bgEntries],
5769
+ });
5770
+ }
5610
5771
  }
5611
5772
  catch { }
5612
5773
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.132",
3
+ "version": "0.9.134",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {