osborn 0.9.142 → 0.9.144
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/claude-llm.d.ts +13 -0
- package/dist/claude-llm.js +179 -58
- package/dist/pipeline-direct-llm.js +1 -1
- package/package.json +1 -1
package/dist/claude-llm.d.ts
CHANGED
|
@@ -231,6 +231,19 @@ export declare class ClaudeLLM extends llm.LLM {
|
|
|
231
231
|
onCheckpoint: (checkpointId: string) => void;
|
|
232
232
|
eventEmitter: EventEmitter;
|
|
233
233
|
}): void;
|
|
234
|
+
/**
|
|
235
|
+
* Dispatcher v1 — auto-spawn a reviewer after a writer sub-agent completes.
|
|
236
|
+
* Runs a one-shot query() with the reviewer agent and emits dispatch_rejected
|
|
237
|
+
* if the verdict is REJECT. A reviewer failure must never crash the consumer.
|
|
238
|
+
* Public so ClaudeLLMStream can call it via this.#llmRef.spawnReviewer().
|
|
239
|
+
*/
|
|
240
|
+
spawnReviewer(agentId: string, writerOutput: string, emitter: EventEmitter): Promise<void>;
|
|
241
|
+
/**
|
|
242
|
+
* Dispatcher v1 — research gate: vet a researcher sub-agent's output before
|
|
243
|
+
* it reaches the main agent. Emits dispatch_rejected with verdict 'NEEDS-MORE'
|
|
244
|
+
* (distinct from reviewer's 'REJECT') so the frontend can tell them apart.
|
|
245
|
+
*/
|
|
246
|
+
spawnResearchGate(agentId: string, researchOutput: string, emitter: EventEmitter): Promise<void>;
|
|
234
247
|
chat({ chatCtx, toolCtx, connOptions, abortController, }: {
|
|
235
248
|
chatCtx: llm.ChatContext;
|
|
236
249
|
toolCtx?: llm.ToolContext;
|
package/dist/claude-llm.js
CHANGED
|
@@ -218,29 +218,41 @@ export const NAMED_AGENTS = {
|
|
|
218
218
|
tools: ['Read', 'Glob', 'Grep', 'WebSearch', 'WebFetch'],
|
|
219
219
|
model: 'opus',
|
|
220
220
|
prompt: [
|
|
221
|
-
'You are Osborn\'s reasoning agent
|
|
221
|
+
'You are Osborn\'s reasoning agent — the "smart model" seat for hard tradeoffs, architecture decisions, and vetting research.',
|
|
222
222
|
'',
|
|
223
223
|
'## Your role',
|
|
224
|
+
'You DECIDE. You do not route work — that is the orchestrator\'s job. You receive structured summaries (not raw dumps) and return clear, opinionated decisions with full rationale.',
|
|
224
225
|
'Think hard about complex problems. Consider multiple approaches. Identify risks and edge cases.',
|
|
225
|
-
'
|
|
226
|
+
'',
|
|
227
|
+
'## Session context',
|
|
228
|
+
'The orchestrator provides, as an artifact in your brief, the PATH to the session index file (search-index.txt — the running index of this session/mission). You MUST actually READ it — do not rely on a summary or a preloaded window. Read the full index (or the portions you need) directly to ground your analysis, understand the mission, and refine/manage the researchers\' work. Reach into the full index whenever a decision or a research review needs the fuller history — that direct reading is what sets your judgment apart.',
|
|
226
229
|
'',
|
|
227
230
|
'## How to work',
|
|
228
231
|
'1. Read and understand the full context before forming an opinion.',
|
|
229
232
|
'2. If the main agent provided researcher findings, use them as your starting point.',
|
|
230
|
-
'3.
|
|
231
|
-
'4.
|
|
233
|
+
'3. Enumerate at least 2-3 alternative approaches before recommending one.',
|
|
234
|
+
'4. For each option consider: pros, cons, risks, and reversibility.',
|
|
232
235
|
'5. Use Read/Grep to verify assumptions against the actual codebase when relevant.',
|
|
236
|
+
'6. Think about: correctness, maintainability, performance, failure modes, migration path.',
|
|
233
237
|
'',
|
|
234
|
-
'##
|
|
235
|
-
'
|
|
236
|
-
'-
|
|
238
|
+
'## Decision output format',
|
|
239
|
+
'For a decision task, structure your response as:',
|
|
240
|
+
'- OPTIONS: for each option — pros / cons / risks / reversibility',
|
|
241
|
+
'- RECOMMENDATION: the chosen option (one clear answer, not "it depends")',
|
|
242
|
+
'- RATIONALE: one paragraph — why this option wins and what assumptions you are making',
|
|
237
243
|
'- PLAN: step-by-step implementation instructions specific enough for the writer agent',
|
|
238
244
|
'- RISKS: what could go wrong and how to mitigate',
|
|
239
|
-
'
|
|
245
|
+
'',
|
|
246
|
+
'## Research-review gate',
|
|
247
|
+
'When reviewing a researcher\'s findings, judge whether the research is COMPLETE and well-sourced against the original task. If it is, pass it. If it\'s thin, missing sources, or the answer likely lives somewhere the researcher didn\'t look, report that it needs more — so the orchestrator sends the researcher back.',
|
|
248
|
+
'End every research review with exactly one of:',
|
|
249
|
+
' GATE: PASS',
|
|
250
|
+
' GATE: NEEDS-MORE — <what\'s missing / where to look>',
|
|
240
251
|
'',
|
|
241
252
|
'## What NOT to do',
|
|
242
253
|
'- Do NOT edit or write files — return a plan for the writer agent',
|
|
243
254
|
'- Do NOT give wishy-washy "both options are valid" non-answers — commit to a recommendation',
|
|
255
|
+
'- Do NOT consume raw session dumps; ask the orchestrator for a structured summary instead',
|
|
244
256
|
'- If you need more information, ask the main agent to delegate to the researcher',
|
|
245
257
|
'',
|
|
246
258
|
'## When to use / handoff',
|
|
@@ -313,23 +325,35 @@ export const NAMED_AGENTS = {
|
|
|
313
325
|
'Execute test suites, build commands, and linters. Interpret failures clearly.',
|
|
314
326
|
'You are a quality gate — find out whether the code works, and say exactly what broke.',
|
|
315
327
|
'',
|
|
328
|
+
'## Backward-compatibility / regression mandate (CRITICAL)',
|
|
329
|
+
'Existing test suites MUST still pass — any pre-existing test that breaks is a BLOCKER; report it as such.',
|
|
330
|
+
'The public API surface (function signatures, exported types, return shapes, behavior) must NOT silently change.',
|
|
331
|
+
'Flag any change that could break existing callers, even if no test currently covers it.',
|
|
332
|
+
'',
|
|
316
333
|
'## How to work',
|
|
317
334
|
'1. Identify the correct test / build command from package.json, Makefile, or the task brief.',
|
|
318
|
-
'2. Run
|
|
319
|
-
'3.
|
|
320
|
-
'4.
|
|
335
|
+
'2. Run the FULL existing test suite first to establish the regression baseline.',
|
|
336
|
+
'3. Generate and run tests targeted at the SPECIFIC diff/change provided — at both unit and integration levels where relevant.',
|
|
337
|
+
'4. Exercise edge cases: boundary values, empty inputs, error/exception paths, null/undefined.',
|
|
338
|
+
'5. Execution loop: write test → run it → read failure output → fix the test OR flag as a real bug in the code. Do NOT silently paper over a real defect.',
|
|
339
|
+
'6. If a command fails, read the relevant source files to locate the root cause.',
|
|
340
|
+
'7. Cap yourself at 6-8 tool calls unless the investigation clearly requires more.',
|
|
321
341
|
'',
|
|
322
342
|
'## What to return',
|
|
323
343
|
'- RESULT: PASS or FAIL (one word, first line)',
|
|
324
|
-
'- COMMAND: the exact command you ran',
|
|
344
|
+
'- COMMAND: the exact command(s) you ran',
|
|
325
345
|
'- OUTPUT: relevant excerpt (errors, failing test names, line numbers)',
|
|
326
346
|
'- ROOT CAUSE: your diagnosis of why it failed (if applicable)',
|
|
347
|
+
'- TEST FILES: path(s) to any test files written or modified',
|
|
348
|
+
'- COVERAGE DELTA: what the change adds or leaves uncovered (before vs after where determinable); list notable uncovered lines/paths',
|
|
349
|
+
'- REGRESSIONS / COMPAT BREAKS: explicit list of any pre-existing tests that now fail or API changes that could break existing callers — tag each as BLOCKER',
|
|
327
350
|
'- What you checked but found to be unrelated',
|
|
328
351
|
'',
|
|
329
352
|
'## What NOT to do',
|
|
330
|
-
'- Do NOT edit or write files — report failures so the writer agent can fix them',
|
|
353
|
+
'- Do NOT edit or write production files — report failures so the writer agent can fix them',
|
|
331
354
|
'- Do NOT run destructive commands (no rm, no git push, no npm publish)',
|
|
332
355
|
'- Do NOT guess at fixes — diagnose only',
|
|
356
|
+
'- Do NOT paper over a real defect by weakening or skipping a test',
|
|
333
357
|
'',
|
|
334
358
|
'## When to use / handoff',
|
|
335
359
|
'Invoked in PARALLEL with reviewer, immediately after the writer returns a change.',
|
|
@@ -382,53 +406,78 @@ export const NAMED_AGENTS = {
|
|
|
382
406
|
description: [
|
|
383
407
|
'Code-review agent (Opus). Use for: the VERIFY step in a generator-verifier loop — after the',
|
|
384
408
|
'writer completes a change, the reviewer reads the diff, checks correctness, spec/requirement',
|
|
385
|
-
'adherence, obvious bugs, and security issues, then
|
|
386
|
-
'specific, actionable feedback.
|
|
409
|
+
'adherence, obvious bugs, and security issues, then tags each finding BLOCKER/MAJOR/MINOR/NIT',
|
|
410
|
+
'and returns an ACCEPT or REJECT verdict with specific, actionable feedback. May write documentation files (.md etc.) only.',
|
|
387
411
|
].join(' '),
|
|
388
|
-
tools: ['Read', 'Glob', 'Grep', 'Bash'],
|
|
412
|
+
tools: ['Read', 'Glob', 'Grep', 'Bash', 'Write', 'Edit'],
|
|
389
413
|
model: 'opus',
|
|
390
414
|
prompt: [
|
|
391
415
|
'You are Osborn\'s reviewer agent. You are the VERIFY step in a generator-verifier loop.',
|
|
392
416
|
'',
|
|
393
417
|
'## Your role',
|
|
394
418
|
'Read the writer\'s completed change (via git diff or by reading modified files), then produce',
|
|
395
|
-
'a structured verdict: ACCEPT or REJECT. You
|
|
396
|
-
'
|
|
419
|
+
'a structured verdict: ACCEPT or REJECT. You report findings so the single writer agent can fix code.',
|
|
420
|
+
'You may write documentation files (.md/.txt/etc.) only. You are the quality gate between a change and merge.',
|
|
397
421
|
'',
|
|
398
422
|
'## Bash is read-only inspection only',
|
|
399
423
|
'You may run: git diff, git log, git status, git show, npm run build, npm test, eslint,',
|
|
400
424
|
'tsc --noEmit, and similar lint/test/security-scan commands.',
|
|
401
425
|
'You must NOT run: rm, git push, git commit, git add, npm publish, or any destructive command.',
|
|
402
426
|
'',
|
|
427
|
+
'## Step 0 — Discover and adopt project standards (before reviewing)',
|
|
428
|
+
'Look for existing project standards, conventions, and documentation: CLAUDE.md, AGENTS.md,',
|
|
429
|
+
'docs/, README, style guides, and any gotchas/anti-pattern/decision notes in the repo or',
|
|
430
|
+
'session memory. If present, ADOPT them as the standard you review against — check the change',
|
|
431
|
+
'against these project-specific conventions and known past gotchas, not just generic best-practice.',
|
|
432
|
+
'If NO such files exist, note that in your report and fall back to the task spec + general best-practice.',
|
|
433
|
+
'',
|
|
434
|
+
'## Maintain documentation',
|
|
435
|
+
'You MAY create and maintain project standards and conventions files. Specifically: adopt an existing',
|
|
436
|
+
'standards doc if one is found, or create one (e.g. CONVENTIONS.md, docs/standards.md) when none exists',
|
|
437
|
+
'and the review reveals patterns worth recording. Keep documentation current as you review.',
|
|
438
|
+
'RESTRICTION: you may ONLY write files with documentation extensions: .md, .markdown, .mdx, .txt, .rst, .adoc.',
|
|
439
|
+
'You must NEVER write code, config, or source files (.ts, .js, .json, .env, etc.) — the write gate',
|
|
440
|
+
'enforces this at the system level and will deny any such attempt.',
|
|
441
|
+
'',
|
|
403
442
|
'## How to work',
|
|
404
443
|
'1. Run `git diff` (or read the files listed in the task) to see exactly what changed.',
|
|
405
444
|
'2. Read any file that needs context to evaluate the diff (interfaces, callers, tests).',
|
|
406
445
|
'3. Run the build or test suite if available to catch compile/runtime regressions.',
|
|
407
|
-
'4. Check against the spec or requirement provided in the task brief.',
|
|
446
|
+
'4. Check against the spec or requirement provided in the task brief AND any project standards found above.',
|
|
408
447
|
'5. Look for: logic errors, missing edge cases, security issues (injection, path traversal,',
|
|
409
448
|
' credential exposure), broken types, spec deviations, unintended side-effects.',
|
|
410
449
|
'6. Cap yourself at 10 tool calls unless the review clearly requires more.',
|
|
411
450
|
'',
|
|
451
|
+
'## Severity taxonomy',
|
|
452
|
+
'Tag EVERY finding with exactly one of:',
|
|
453
|
+
' BLOCKER — incorrect behavior, data loss, security hole, broken build; must fix before merge',
|
|
454
|
+
' MAJOR — significant bug or spec deviation that will likely cause real problems in use',
|
|
455
|
+
' MINOR — non-critical defect or missed edge case worth fixing but not blocking',
|
|
456
|
+
' NIT — style, naming, or polish; never a reason to REJECT on its own',
|
|
457
|
+
'',
|
|
412
458
|
'## What to return',
|
|
413
459
|
'Structure your response EXACTLY as follows:',
|
|
414
460
|
'',
|
|
415
|
-
'VERDICT: ACCEPT | REJECT',
|
|
461
|
+
'VERDICT: ACCEPT | REJECT — <one-line rationale>',
|
|
416
462
|
'',
|
|
417
|
-
'ISSUES (
|
|
418
|
-
'
|
|
463
|
+
'ISSUES (omit section entirely if VERDICT is ACCEPT):',
|
|
464
|
+
' [SEVERITY] <file>:<line>',
|
|
465
|
+
' Evidence: "<short quote of the offending code>"',
|
|
466
|
+
' Impact: <what breaks or why it matters>',
|
|
467
|
+
' Fix: <recommended change>',
|
|
468
|
+
' Verify: <how to confirm the fix is correct>',
|
|
419
469
|
'',
|
|
420
470
|
'WHAT IT CHECKED-AND-CLEARED:',
|
|
421
471
|
' - <each item you verified and found correct — be specific, not generic>',
|
|
422
472
|
'',
|
|
423
473
|
'## What NOT to do',
|
|
424
|
-
'- Do NOT
|
|
474
|
+
'- Do NOT write code, config, or source files — only documentation-extension files (.md/.markdown/.mdx/.txt/.rst/.adoc) are permitted',
|
|
425
475
|
'- Do NOT run destructive commands (no rm, no git push, no git commit, no npm publish)',
|
|
426
476
|
'- Do NOT approve a change that has a real defect just to be agreeable',
|
|
427
|
-
'- Do NOT
|
|
477
|
+
'- Do NOT REJECT solely on NIT-level findings',
|
|
428
478
|
'',
|
|
429
479
|
'## When to use / handoff',
|
|
430
480
|
'Invoked in PARALLEL with tester, immediately after the writer returns a change.',
|
|
431
|
-
'Return an ACCEPT or REJECT verdict with specific, actionable issue reports.',
|
|
432
481
|
'The orchestrator waits for both reviewer and tester before synthesizing and speaking to the user.',
|
|
433
482
|
].join('\n'),
|
|
434
483
|
},
|
|
@@ -506,9 +555,9 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
506
555
|
// Active queries — multiple can be running (SDK queues them internally).
|
|
507
556
|
// We keep ALL references so interrupt() can stop whatever is currently executing.
|
|
508
557
|
#activeQueries = new Set();
|
|
509
|
-
//
|
|
510
|
-
//
|
|
511
|
-
#
|
|
558
|
+
// Dedup guard — prevents double-firing reviewer/gate if SubagentStop fires
|
|
559
|
+
// more than once for the same agent_id (e.g. retry edge cases).
|
|
560
|
+
#dispatchedFor = new Set();
|
|
512
561
|
constructor(opts = {}) {
|
|
513
562
|
super();
|
|
514
563
|
// Session resume/continue options
|
|
@@ -921,6 +970,7 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
921
970
|
this.#persistentQuery = null;
|
|
922
971
|
this.#messageChannel = null;
|
|
923
972
|
this.#backgroundConsumerRunning = false;
|
|
973
|
+
this.#dispatchedFor.clear();
|
|
924
974
|
console.log('🔒 Persistent session closed');
|
|
925
975
|
}
|
|
926
976
|
/**
|
|
@@ -1043,29 +1093,6 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
1043
1093
|
callbacks.eventEmitter.emit('tts_say', { text: ttsChunk });
|
|
1044
1094
|
}
|
|
1045
1095
|
}
|
|
1046
|
-
// Dispatcher v1 — record Task tool_use blocks so we can correlate
|
|
1047
|
-
// the matching task_summary message to the subagent type.
|
|
1048
|
-
if (block.type === 'tool_use' && block.name === 'Task') {
|
|
1049
|
-
console.log('[DISPATCH-PROBE] task tool_use', JSON.stringify({ id: block.id, subagent_type: block.input?.subagent_type }));
|
|
1050
|
-
this.#dispatchAgentTypes.set(block.id, block.input?.subagent_type);
|
|
1051
|
-
statusManager.upsertDispatch(block.id, {
|
|
1052
|
-
owner: 'orchestrator',
|
|
1053
|
-
subagentType: block.input?.subagent_type,
|
|
1054
|
-
dispatchState: 'running',
|
|
1055
|
-
});
|
|
1056
|
-
}
|
|
1057
|
-
}
|
|
1058
|
-
}
|
|
1059
|
-
// Dispatcher v1 — catch sub-agent completion signals
|
|
1060
|
-
if (msg.type === 'system' && msg.subtype === 'task_summary') {
|
|
1061
|
-
console.log('[DISPATCH-PROBE] task_summary raw:', JSON.stringify(msg).slice(0, 500));
|
|
1062
|
-
const tuid = msg.tool_use_id ?? msg.toolUseId;
|
|
1063
|
-
const output = msg.summary ?? msg.result ?? msg.output ?? '';
|
|
1064
|
-
const subType = tuid ? this.#dispatchAgentTypes.get(tuid) : undefined;
|
|
1065
|
-
statusManager.upsertDispatch(tuid ?? `ts-${Date.now()}`, { subagentType: subType, dispatchState: 'completed', artifact: String(output) });
|
|
1066
|
-
console.log(`[DISPATCH] completed type=${subType ?? '?'} tuid=${(tuid ?? '?').slice(0, 8)} len=${String(output).length}`);
|
|
1067
|
-
if (subType === 'writer' && output) {
|
|
1068
|
-
this.#spawnReviewer(tuid, String(output), callbacks.eventEmitter);
|
|
1069
1096
|
}
|
|
1070
1097
|
}
|
|
1071
1098
|
// Result — marks end of a turn (but we keep consuming for next turn)
|
|
@@ -1098,8 +1125,13 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
1098
1125
|
* Dispatcher v1 — auto-spawn a reviewer after a writer sub-agent completes.
|
|
1099
1126
|
* Runs a one-shot query() with the reviewer agent and emits dispatch_rejected
|
|
1100
1127
|
* if the verdict is REJECT. A reviewer failure must never crash the consumer.
|
|
1128
|
+
* Public so ClaudeLLMStream can call it via this.#llmRef.spawnReviewer().
|
|
1101
1129
|
*/
|
|
1102
|
-
async
|
|
1130
|
+
async spawnReviewer(agentId, writerOutput, emitter) {
|
|
1131
|
+
// Dedup guard — SubagentStop may fire more than once for the same agent_id.
|
|
1132
|
+
if (this.#dispatchedFor.has(agentId))
|
|
1133
|
+
return;
|
|
1134
|
+
this.#dispatchedFor.add(agentId);
|
|
1103
1135
|
try {
|
|
1104
1136
|
const prompt = [
|
|
1105
1137
|
'Use the reviewer sub-agent to review this writer output for correctness/spec-adherence/obvious bugs.',
|
|
@@ -1109,12 +1141,15 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
1109
1141
|
writerOutput.slice(0, 8000),
|
|
1110
1142
|
'</writer_output>',
|
|
1111
1143
|
].join('\n');
|
|
1144
|
+
// Do NOT pass agents here — the reviewer must be review-only and must not
|
|
1145
|
+
// be able to spawn writer/researcher/reasoner sub-agents. Passing an empty
|
|
1146
|
+
// agents roster prevents any SubagentStop(agent_type==='writer') from
|
|
1147
|
+
// firing inside this one-shot query and re-arming the backstop loop.
|
|
1112
1148
|
const reviewerOptions = {
|
|
1113
1149
|
cwd: this.#opts.workingDirectory,
|
|
1114
1150
|
permissionMode: 'default',
|
|
1115
|
-
agents: NAMED_AGENTS,
|
|
1116
1151
|
};
|
|
1117
|
-
console.log(`[DISPATCH] spawning reviewer for
|
|
1152
|
+
console.log(`[DISPATCH] spawning reviewer for agentId=${agentId.slice(0, 8)}`);
|
|
1118
1153
|
const reviewerQuery = query({ prompt, options: reviewerOptions });
|
|
1119
1154
|
this.#activeQueries.add(reviewerQuery);
|
|
1120
1155
|
let reviewerText = '';
|
|
@@ -1132,18 +1167,76 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
1132
1167
|
const verdictMatch = reviewerText.match(/VERDICT:\s*(ACCEPT|REJECT)/i);
|
|
1133
1168
|
const verdict = verdictMatch ? verdictMatch[1].toUpperCase() : null;
|
|
1134
1169
|
if (verdict === 'REJECT') {
|
|
1135
|
-
console.log(`[DISPATCH] review REJECT for
|
|
1136
|
-
statusManager.upsertDispatch(
|
|
1137
|
-
emitter.emit('dispatch_rejected', { tuid, verdict: 'REJECT', review: reviewerText });
|
|
1170
|
+
console.log(`[DISPATCH] review REJECT for agentId=${agentId.slice(0, 8)}`);
|
|
1171
|
+
statusManager.upsertDispatch(agentId, { dispatchState: 'rejected', artifact: reviewerText });
|
|
1172
|
+
emitter.emit('dispatch_rejected', { tuid: agentId, verdict: 'REJECT', review: reviewerText });
|
|
1138
1173
|
}
|
|
1139
1174
|
else {
|
|
1140
|
-
console.log(`[DISPATCH] review ACCEPT for
|
|
1175
|
+
console.log(`[DISPATCH] review ACCEPT for agentId=${agentId.slice(0, 8)}`);
|
|
1141
1176
|
}
|
|
1142
1177
|
}
|
|
1143
1178
|
catch (err) {
|
|
1144
1179
|
console.error('[DISPATCH] reviewer spawn failed (non-fatal):', err);
|
|
1145
1180
|
}
|
|
1146
1181
|
}
|
|
1182
|
+
/**
|
|
1183
|
+
* Dispatcher v1 — research gate: vet a researcher sub-agent's output before
|
|
1184
|
+
* it reaches the main agent. Emits dispatch_rejected with verdict 'NEEDS-MORE'
|
|
1185
|
+
* (distinct from reviewer's 'REJECT') so the frontend can tell them apart.
|
|
1186
|
+
*/
|
|
1187
|
+
async spawnResearchGate(agentId, researchOutput, emitter) {
|
|
1188
|
+
// Dedup guard — SubagentStop may fire more than once for the same agent_id.
|
|
1189
|
+
if (this.#dispatchedFor.has(agentId))
|
|
1190
|
+
return;
|
|
1191
|
+
this.#dispatchedFor.add(agentId);
|
|
1192
|
+
try {
|
|
1193
|
+
const prompt = [
|
|
1194
|
+
'Use the reasoner sub-agent to vet this research output.',
|
|
1195
|
+
'Determine whether the research is sufficient to answer the original question.',
|
|
1196
|
+
'End your reply with exactly `GATE: PASS` or `GATE: NEEDS-MORE`.',
|
|
1197
|
+
'',
|
|
1198
|
+
'<research_output>',
|
|
1199
|
+
researchOutput.slice(0, 8000),
|
|
1200
|
+
'</research_output>',
|
|
1201
|
+
].join('\n');
|
|
1202
|
+
// Do NOT pass agents here — the research-gate reasoner must be review-only
|
|
1203
|
+
// and must not be able to spawn sub-agents. Same rationale as spawnReviewer:
|
|
1204
|
+
// an agents roster would allow delegation back to the writer, which would
|
|
1205
|
+
// fire SubagentStop(agent_type==='writer') and re-arm the backstop.
|
|
1206
|
+
const gateOptions = {
|
|
1207
|
+
cwd: this.#opts.workingDirectory,
|
|
1208
|
+
permissionMode: 'default',
|
|
1209
|
+
};
|
|
1210
|
+
console.log(`[DISPATCH] spawning research-gate for agentId=${agentId.slice(0, 8)}`);
|
|
1211
|
+
const gateQuery = query({ prompt, options: gateOptions });
|
|
1212
|
+
this.#activeQueries.add(gateQuery);
|
|
1213
|
+
let review = '';
|
|
1214
|
+
try {
|
|
1215
|
+
for await (const msg of gateQuery) {
|
|
1216
|
+
const m = msg;
|
|
1217
|
+
if (m.type === 'result' && m.result) {
|
|
1218
|
+
review = String(m.result);
|
|
1219
|
+
}
|
|
1220
|
+
}
|
|
1221
|
+
}
|
|
1222
|
+
finally {
|
|
1223
|
+
this.#activeQueries.delete(gateQuery);
|
|
1224
|
+
}
|
|
1225
|
+
const gateMatch = review.match(/GATE:\s*(PASS|NEEDS-MORE)/i);
|
|
1226
|
+
const gateVerdict = gateMatch ? gateMatch[1].toUpperCase() : null;
|
|
1227
|
+
if (gateVerdict === 'NEEDS-MORE') {
|
|
1228
|
+
console.log(`[DISPATCH] research-gate NEEDS-MORE for agentId=${agentId.slice(0, 8)}`);
|
|
1229
|
+
statusManager.upsertDispatch(agentId, { dispatchState: 'rejected', artifact: review });
|
|
1230
|
+
emitter.emit('dispatch_rejected', { tuid: agentId, verdict: 'NEEDS-MORE', review });
|
|
1231
|
+
}
|
|
1232
|
+
else {
|
|
1233
|
+
console.log(`[DISPATCH] research-gate PASS for agentId=${agentId.slice(0, 8)}`);
|
|
1234
|
+
}
|
|
1235
|
+
}
|
|
1236
|
+
catch (err) {
|
|
1237
|
+
console.error('[DISPATCH] research-gate spawn failed (non-fatal):', err);
|
|
1238
|
+
}
|
|
1239
|
+
}
|
|
1147
1240
|
chat({ chatCtx, toolCtx, connOptions = DEFAULT_API_CONNECT_OPTIONS, abortController, }) {
|
|
1148
1241
|
return new ClaudeLLMStream(this, {
|
|
1149
1242
|
chatCtx,
|
|
@@ -1362,6 +1455,21 @@ class ClaudeLLMStream extends llm.LLMStream {
|
|
|
1362
1455
|
this.#eventEmitter.emit('tool_use', { name: toolName, input: toolInput, agentRole: agentType || 'main' });
|
|
1363
1456
|
return { hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'ask' } };
|
|
1364
1457
|
}
|
|
1458
|
+
// Reviewer agent: ONLY documentation-extension files allowed — fail closed
|
|
1459
|
+
if (agentType === 'reviewer') {
|
|
1460
|
+
const reviewerPath = String(toolInput.file_path || '');
|
|
1461
|
+
const DOC_EXTENSIONS = /\.(md|markdown|mdx|txt|rst|adoc)$/i;
|
|
1462
|
+
if (!reviewerPath || !DOC_EXTENSIONS.test(reviewerPath)) {
|
|
1463
|
+
const reason = reviewerPath
|
|
1464
|
+
? `Reviewer write denied: ${reviewerPath} is not a documentation file (.md/.markdown/.mdx/.txt/.rst/.adoc). Reviewer may only write documentation.`
|
|
1465
|
+
: 'Reviewer write denied: could not determine target file path. Failing closed.';
|
|
1466
|
+
console.log(`🚫 Reviewer write blocked: ${reviewerPath || '(no path)'} — not a doc extension`);
|
|
1467
|
+
this.#eventEmitter.emit('tool_blocked', { name: toolName, reason });
|
|
1468
|
+
return { hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'deny' }, reason };
|
|
1469
|
+
}
|
|
1470
|
+
console.log(`📝 Reviewer doc write allowed: ${reviewerPath}`);
|
|
1471
|
+
return { hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'ask' } };
|
|
1472
|
+
}
|
|
1365
1473
|
// All other agents (main, researcher, reasoner, etc.): workspace only
|
|
1366
1474
|
const filePath = String(toolInput.file_path || '');
|
|
1367
1475
|
if (filePath && !filePath.includes('/osb/') && !filePath.includes('.osborn/sessions/') && !filePath.includes('.osborn/research/')) {
|
|
@@ -1606,6 +1714,19 @@ class ClaudeLLMStream extends llm.LLMStream {
|
|
|
1606
1714
|
matcher: '.*',
|
|
1607
1715
|
hooks: [async (input) => {
|
|
1608
1716
|
console.log('[LIFECYCLE-PROBE] SubagentStop', JSON.stringify(input));
|
|
1717
|
+
const at = input?.agent_type;
|
|
1718
|
+
const msg = String(input?.last_assistant_message ?? '');
|
|
1719
|
+
const aid = input?.agent_id ?? ('sa-' + Date.now());
|
|
1720
|
+
statusManager.upsertDispatch(aid, { subagentType: at, dispatchState: 'completed', artifact: msg });
|
|
1721
|
+
// Infinite-loop guard — never re-dispatch the reviewer or reasoner.
|
|
1722
|
+
if (at === 'reviewer' || at === 'reasoner')
|
|
1723
|
+
return {};
|
|
1724
|
+
if (at === 'writer' && msg) {
|
|
1725
|
+
void this.#llmRef.spawnReviewer(aid, msg, this.#eventEmitter);
|
|
1726
|
+
}
|
|
1727
|
+
else if (at === 'researcher' && msg) {
|
|
1728
|
+
void this.#llmRef.spawnResearchGate(aid, msg, this.#eventEmitter);
|
|
1729
|
+
}
|
|
1609
1730
|
return {};
|
|
1610
1731
|
}]
|
|
1611
1732
|
}],
|
|
@@ -25,7 +25,7 @@ function buildSessionTail(sessionId, workingDir) {
|
|
|
25
25
|
return '';
|
|
26
26
|
if (!sessionId)
|
|
27
27
|
return '';
|
|
28
|
-
const maxLines = parseInt(process.env.OSBORN_SESSION_TAIL_COUNT || '
|
|
28
|
+
const maxLines = parseInt(process.env.OSBORN_SESSION_TAIL_COUNT || '3000', 10);
|
|
29
29
|
try {
|
|
30
30
|
const indexPath = getIndexPath(sessionId, workingDir);
|
|
31
31
|
if (!indexPath)
|