agent-nuvira 3.3.3 → 3.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -5
- package/dist/agent-sdk/src/agent.d.ts +2 -0
- package/dist/agent-sdk/src/agent.d.ts.map +1 -1
- package/dist/agent-sdk/src/define.d.ts +64 -0
- package/dist/agent-sdk/src/define.d.ts.map +1 -0
- package/dist/agent-sdk/src/define.js +76 -0
- package/dist/agent-sdk/src/define.js.map +1 -0
- package/dist/agent-sdk/src/index.d.ts +9 -0
- package/dist/agent-sdk/src/index.d.ts.map +1 -1
- package/dist/agent-sdk/src/index.js +9 -0
- package/dist/agent-sdk/src/index.js.map +1 -1
- package/dist/agent-sdk/src/scaffold.d.ts +11 -0
- package/dist/agent-sdk/src/scaffold.d.ts.map +1 -1
- package/dist/agent-sdk/src/scaffold.js +16 -6
- package/dist/agent-sdk/src/scaffold.js.map +1 -1
- package/dist/agents/long-form-plan.d.ts.map +1 -1
- package/dist/agents/long-form-plan.js +2 -1
- package/dist/agents/long-form-plan.js.map +1 -1
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +5 -4
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/cli/agent.d.ts +2 -2
- package/dist/cli/agent.js +10 -10
- package/dist/cli/chat.d.ts +94 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +354 -20
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/cli-program.d.ts.map +1 -1
- package/dist/cli/cli-program.js +5 -0
- package/dist/cli/cli-program.js.map +1 -1
- package/dist/cli/config.d.ts.map +1 -1
- package/dist/cli/config.js +9 -1
- package/dist/cli/config.js.map +1 -1
- package/dist/cli/doctor.d.ts.map +1 -1
- package/dist/cli/doctor.js +3 -2
- package/dist/cli/doctor.js.map +1 -1
- package/dist/cli/edit.js +2 -2
- package/dist/cli/eval.d.ts +17 -0
- package/dist/cli/eval.d.ts.map +1 -1
- package/dist/cli/eval.js +105 -2
- package/dist/cli/eval.js.map +1 -1
- package/dist/cli/execute.d.ts +16 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +192 -22
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/loop-executor.d.ts +55 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +212 -13
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/model.d.ts.map +1 -1
- package/dist/cli/model.js +9 -8
- package/dist/cli/model.js.map +1 -1
- package/dist/cli/models.d.ts +1 -1
- package/dist/cli/models.js +4 -4
- package/dist/cli/parity.d.ts +85 -0
- package/dist/cli/parity.d.ts.map +1 -0
- package/dist/cli/parity.js +506 -0
- package/dist/cli/parity.js.map +1 -0
- package/dist/cli/plan.d.ts.map +1 -1
- package/dist/cli/plan.js +2 -1
- package/dist/cli/plan.js.map +1 -1
- package/dist/cli/retrieval.d.ts.map +1 -1
- package/dist/cli/retrieval.js +5 -4
- package/dist/cli/retrieval.js.map +1 -1
- package/dist/cli/sdk.js +4 -4
- package/dist/cli/sdk.js.map +1 -1
- package/dist/cli/trace.d.ts.map +1 -1
- package/dist/cli/trace.js +2 -1
- package/dist/cli/trace.js.map +1 -1
- package/dist/cli/workflow.js +2 -2
- package/dist/cli/workflow.js.map +1 -1
- package/dist/config/process-env.d.ts +136 -0
- package/dist/config/process-env.d.ts.map +1 -0
- package/dist/config/process-env.js +217 -0
- package/dist/config/process-env.js.map +1 -0
- package/dist/config/types.d.ts +44 -0
- package/dist/config/types.d.ts.map +1 -1
- package/dist/findings/verdicts.d.ts +241 -0
- package/dist/findings/verdicts.d.ts.map +1 -0
- package/dist/findings/verdicts.js +284 -0
- package/dist/findings/verdicts.js.map +1 -0
- package/dist/gateway/adapters.d.ts +65 -0
- package/dist/gateway/adapters.d.ts.map +1 -1
- package/dist/gateway/adapters.js +216 -10
- package/dist/gateway/adapters.js.map +1 -1
- package/dist/gateway/channel-directory.d.ts +31 -0
- package/dist/gateway/channel-directory.d.ts.map +1 -1
- package/dist/gateway/channel-directory.js +40 -0
- package/dist/gateway/channel-directory.js.map +1 -1
- package/dist/gateway/gateway-log.d.ts +1 -1
- package/dist/gateway/gateway-log.d.ts.map +1 -1
- package/dist/gateway/gateway-log.js.map +1 -1
- package/dist/gateway/hooks.d.ts +87 -19
- package/dist/gateway/hooks.d.ts.map +1 -1
- package/dist/gateway/hooks.js +62 -23
- package/dist/gateway/hooks.js.map +1 -1
- package/dist/gateway/inbound-media.d.ts +147 -0
- package/dist/gateway/inbound-media.d.ts.map +1 -0
- package/dist/gateway/inbound-media.js +317 -0
- package/dist/gateway/inbound-media.js.map +1 -0
- package/dist/gateway/inbox.d.ts +8 -1
- package/dist/gateway/inbox.d.ts.map +1 -1
- package/dist/gateway/inbox.js.map +1 -1
- package/dist/gateway/platform-config.d.ts +14 -0
- package/dist/gateway/platform-config.d.ts.map +1 -1
- package/dist/gateway/platform-config.js +26 -8
- package/dist/gateway/platform-config.js.map +1 -1
- package/dist/gateway/realtime.d.ts +114 -0
- package/dist/gateway/realtime.d.ts.map +1 -0
- package/dist/gateway/realtime.js +402 -0
- package/dist/gateway/realtime.js.map +1 -0
- package/dist/gateway/registry.d.ts +31 -0
- package/dist/gateway/registry.d.ts.map +1 -1
- package/dist/gateway/registry.js +224 -4
- package/dist/gateway/registry.js.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts +11 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.js +123 -4
- package/dist/gateway/whatsapp/baileys-bridge.js.map +1 -1
- package/dist/gateway/whatsapp/bridge.d.ts +6 -2
- package/dist/gateway/whatsapp/bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/bridge.js.map +1 -1
- package/dist/index.js +8 -0
- package/dist/index.js.map +1 -1
- package/dist/inference/factory.d.ts +14 -0
- package/dist/inference/factory.d.ts.map +1 -1
- package/dist/inference/factory.js +17 -0
- package/dist/inference/factory.js.map +1 -1
- package/dist/inference/groq-adapter.d.ts +2 -0
- package/dist/inference/groq-adapter.d.ts.map +1 -1
- package/dist/inference/groq-adapter.js +16 -6
- package/dist/inference/groq-adapter.js.map +1 -1
- package/dist/inference/tools.d.ts.map +1 -1
- package/dist/inference/tools.js +29 -0
- package/dist/inference/tools.js.map +1 -1
- package/dist/learning/benchmark.d.ts.map +1 -1
- package/dist/learning/benchmark.js +3 -2
- package/dist/learning/benchmark.js.map +1 -1
- package/dist/learning/continuation.d.ts.map +1 -1
- package/dist/learning/continuation.js +2 -1
- package/dist/learning/continuation.js.map +1 -1
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +2 -1
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/deferred-task.d.ts.map +1 -1
- package/dist/learning/deferred-task.js +13 -4
- package/dist/learning/deferred-task.js.map +1 -1
- package/dist/learning/eval-framework.d.ts.map +1 -1
- package/dist/learning/eval-framework.js +2 -1
- package/dist/learning/eval-framework.js.map +1 -1
- package/dist/learning/long-form.d.ts.map +1 -1
- package/dist/learning/long-form.js +2 -1
- package/dist/learning/long-form.js.map +1 -1
- package/dist/learning/model-reachability.d.ts +95 -0
- package/dist/learning/model-reachability.d.ts.map +1 -0
- package/dist/learning/model-reachability.js +111 -0
- package/dist/learning/model-reachability.js.map +1 -0
- package/dist/learning/model-registry.d.ts +22 -0
- package/dist/learning/model-registry.d.ts.map +1 -1
- package/dist/learning/model-registry.js +26 -1
- package/dist/learning/model-registry.js.map +1 -1
- package/dist/learning/model-verify-job.d.ts +131 -0
- package/dist/learning/model-verify-job.d.ts.map +1 -0
- package/dist/learning/model-verify-job.js +221 -0
- package/dist/learning/model-verify-job.js.map +1 -0
- package/dist/learning/reasoning-cache.d.ts.map +1 -1
- package/dist/learning/reasoning-cache.js +2 -1
- package/dist/learning/reasoning-cache.js.map +1 -1
- package/dist/learning/reasoning-trace.d.ts +37 -1
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +66 -0
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +2 -1
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/learning/retrieval.d.ts.map +1 -1
- package/dist/learning/retrieval.js +2 -1
- package/dist/learning/retrieval.js.map +1 -1
- package/dist/learning/seeded-benchmark.d.ts +160 -0
- package/dist/learning/seeded-benchmark.d.ts.map +1 -0
- package/dist/learning/seeded-benchmark.js +321 -0
- package/dist/learning/seeded-benchmark.js.map +1 -0
- package/dist/learning/seeded-bugs.d.ts +142 -0
- package/dist/learning/seeded-bugs.d.ts.map +1 -0
- package/dist/learning/seeded-bugs.js +535 -0
- package/dist/learning/seeded-bugs.js.map +1 -0
- package/dist/learning/step-checkpoint.d.ts +127 -0
- package/dist/learning/step-checkpoint.d.ts.map +1 -0
- package/dist/learning/step-checkpoint.js +244 -0
- package/dist/learning/step-checkpoint.js.map +1 -0
- package/dist/nlu/intent-confirm.d.ts +23 -1
- package/dist/nlu/intent-confirm.d.ts.map +1 -1
- package/dist/nlu/intent-confirm.js +85 -2
- package/dist/nlu/intent-confirm.js.map +1 -1
- package/dist/nlu/learnings.d.ts +7 -0
- package/dist/nlu/learnings.d.ts.map +1 -1
- package/dist/nlu/learnings.js +7 -0
- package/dist/nlu/learnings.js.map +1 -1
- package/dist/observability/debug-log.d.ts +250 -0
- package/dist/observability/debug-log.d.ts.map +1 -0
- package/dist/observability/debug-log.js +500 -0
- package/dist/observability/debug-log.js.map +1 -0
- package/dist/observability/event-bus.d.ts.map +1 -1
- package/dist/observability/event-bus.js +4 -1
- package/dist/observability/event-bus.js.map +1 -1
- package/dist/observability/otel.d.ts +278 -0
- package/dist/observability/otel.d.ts.map +1 -0
- package/dist/observability/otel.js +590 -0
- package/dist/observability/otel.js.map +1 -0
- package/dist/parity/drivers.d.ts +99 -0
- package/dist/parity/drivers.d.ts.map +1 -0
- package/dist/parity/drivers.js +1362 -0
- package/dist/parity/drivers.js.map +1 -0
- package/dist/parity/graph.d.ts +73 -0
- package/dist/parity/graph.d.ts.map +1 -0
- package/dist/parity/graph.js +162 -0
- package/dist/parity/graph.js.map +1 -0
- package/dist/parity/matrix.d.ts +105 -0
- package/dist/parity/matrix.d.ts.map +1 -0
- package/dist/parity/matrix.js +352 -0
- package/dist/parity/matrix.js.map +1 -0
- package/dist/parity/observation.d.ts +444 -0
- package/dist/parity/observation.d.ts.map +1 -0
- package/dist/parity/observation.js +333 -0
- package/dist/parity/observation.js.map +1 -0
- package/dist/parity/scenarios.d.ts +229 -0
- package/dist/parity/scenarios.d.ts.map +1 -0
- package/dist/parity/scenarios.js +175 -0
- package/dist/parity/scenarios.js.map +1 -0
- package/dist/parity/surfaces.d.ts +122 -0
- package/dist/parity/surfaces.d.ts.map +1 -0
- package/dist/parity/surfaces.js +190 -0
- package/dist/parity/surfaces.js.map +1 -0
- package/dist/runtime/fault-injection.d.ts +173 -0
- package/dist/runtime/fault-injection.d.ts.map +1 -0
- package/dist/runtime/fault-injection.js +281 -0
- package/dist/runtime/fault-injection.js.map +1 -0
- package/dist/tools/child-agent-entry.d.ts +23 -0
- package/dist/tools/child-agent-entry.d.ts.map +1 -0
- package/dist/tools/child-agent-entry.js +129 -0
- package/dist/tools/child-agent-entry.js.map +1 -0
- package/dist/tools/child-agent-runtime.d.ts +124 -0
- package/dist/tools/child-agent-runtime.d.ts.map +1 -0
- package/dist/tools/child-agent-runtime.js +704 -0
- package/dist/tools/child-agent-runtime.js.map +1 -0
- package/dist/tools/coding-tools.d.ts.map +1 -1
- package/dist/tools/coding-tools.js +82 -10
- package/dist/tools/coding-tools.js.map +1 -1
- package/dist/tools/delegation-system.d.ts +31 -0
- package/dist/tools/delegation-system.d.ts.map +1 -1
- package/dist/tools/delegation-system.js +70 -9
- package/dist/tools/delegation-system.js.map +1 -1
- package/dist/tools/extract/docx.d.ts +27 -0
- package/dist/tools/extract/docx.d.ts.map +1 -0
- package/dist/tools/extract/docx.js +48 -0
- package/dist/tools/extract/docx.js.map +1 -0
- package/dist/tools/extract/html-text.d.ts +27 -0
- package/dist/tools/extract/html-text.d.ts.map +1 -0
- package/dist/tools/extract/html-text.js +86 -0
- package/dist/tools/extract/html-text.js.map +1 -0
- package/dist/tools/extract/pdf-ocr.d.ts +58 -0
- package/dist/tools/extract/pdf-ocr.d.ts.map +1 -0
- package/dist/tools/extract/pdf-ocr.js +116 -0
- package/dist/tools/extract/pdf-ocr.js.map +1 -0
- package/dist/tools/extract/pdf.d.ts +65 -0
- package/dist/tools/extract/pdf.d.ts.map +1 -0
- package/dist/tools/extract/pdf.js +197 -0
- package/dist/tools/extract/pdf.js.map +1 -0
- package/dist/tools/extract/pptx.d.ts +32 -0
- package/dist/tools/extract/pptx.d.ts.map +1 -0
- package/dist/tools/extract/pptx.js +77 -0
- package/dist/tools/extract/pptx.js.map +1 -0
- package/dist/tools/extract/xlsx.d.ts +47 -0
- package/dist/tools/extract/xlsx.d.ts.map +1 -0
- package/dist/tools/extract/xlsx.js +111 -0
- package/dist/tools/extract/xlsx.js.map +1 -0
- package/dist/tools/finding-tool.d.ts +76 -0
- package/dist/tools/finding-tool.d.ts.map +1 -0
- package/dist/tools/finding-tool.js +125 -0
- package/dist/tools/finding-tool.js.map +1 -0
- package/dist/tools/messaging-tools.d.ts +41 -11
- package/dist/tools/messaging-tools.d.ts.map +1 -1
- package/dist/tools/messaging-tools.js +104 -55
- package/dist/tools/messaging-tools.js.map +1 -1
- package/dist/tools/neutts-synth.d.ts +27 -3
- package/dist/tools/neutts-synth.d.ts.map +1 -1
- package/dist/tools/neutts-synth.js +57 -13
- package/dist/tools/neutts-synth.js.map +1 -1
- package/dist/tools/pipeline-tool.d.ts +43 -1
- package/dist/tools/pipeline-tool.d.ts.map +1 -1
- package/dist/tools/pipeline-tool.js +13 -2
- package/dist/tools/pipeline-tool.js.map +1 -1
- package/dist/tools/read-extract.d.ts +116 -45
- package/dist/tools/read-extract.d.ts.map +1 -1
- package/dist/tools/read-extract.js +494 -158
- package/dist/tools/read-extract.js.map +1 -1
- package/dist/tools/registry.d.ts +2 -2
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +152 -23
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/subagent-refusal.d.ts +15 -0
- package/dist/tools/subagent-refusal.d.ts.map +1 -0
- package/dist/tools/subagent-refusal.js +18 -0
- package/dist/tools/subagent-refusal.js.map +1 -0
- package/dist/tools/subagent-spawner.d.ts +122 -0
- package/dist/tools/subagent-spawner.d.ts.map +1 -1
- package/dist/tools/subagent-spawner.js +249 -28
- package/dist/tools/subagent-spawner.js.map +1 -1
- package/dist/tools/tool-hooks.d.ts +177 -0
- package/dist/tools/tool-hooks.d.ts.map +1 -0
- package/dist/tools/tool-hooks.js +427 -0
- package/dist/tools/tool-hooks.js.map +1 -0
- package/dist/tools/tool-loop.d.ts +82 -0
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +167 -9
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/tools/tool-refusal.d.ts +68 -0
- package/dist/tools/tool-refusal.d.ts.map +1 -0
- package/dist/tools/tool-refusal.js +78 -0
- package/dist/tools/tool-refusal.js.map +1 -0
- package/dist/tools/toolsets.d.ts +8 -0
- package/dist/tools/toolsets.d.ts.map +1 -1
- package/dist/tools/toolsets.js +12 -2
- package/dist/tools/toolsets.js.map +1 -1
- package/dist/tools/vision-tools.d.ts +88 -83
- package/dist/tools/vision-tools.d.ts.map +1 -1
- package/dist/tools/vision-tools.js +134 -103
- package/dist/tools/vision-tools.js.map +1 -1
- package/dist/tools/worktree.d.ts +210 -0
- package/dist/tools/worktree.d.ts.map +1 -0
- package/dist/tools/worktree.js +374 -0
- package/dist/tools/worktree.js.map +1 -0
- package/dist/utils/format.d.ts +3 -0
- package/dist/utils/format.d.ts.map +1 -0
- package/dist/utils/format.js +32 -0
- package/dist/utils/format.js.map +1 -0
- package/dist/web-dashboard/attachment-extract.d.ts +64 -0
- package/dist/web-dashboard/attachment-extract.d.ts.map +1 -0
- package/dist/web-dashboard/attachment-extract.js +154 -0
- package/dist/web-dashboard/attachment-extract.js.map +1 -0
- package/dist/web-dashboard/chat-console.d.ts +108 -1
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +36 -0
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/hub-data.d.ts +37 -0
- package/dist/web-dashboard/hub-data.d.ts.map +1 -1
- package/dist/web-dashboard/hub-data.js +61 -1
- package/dist/web-dashboard/hub-data.js.map +1 -1
- package/dist/web-dashboard/process-env-inventory.d.ts +57 -0
- package/dist/web-dashboard/process-env-inventory.d.ts.map +1 -0
- package/dist/web-dashboard/process-env-inventory.js +96 -0
- package/dist/web-dashboard/process-env-inventory.js.map +1 -0
- package/dist/web-dashboard/server.d.ts +14 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +493 -30
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +261 -48
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/package.json +18 -3
- package/src/web-dashboard/public/assets/{index-Cyd6tIew.css → index-CIJ6FHHZ.css} +1 -1
- package/src/web-dashboard/public/assets/index-CsySzR41.js +207 -0
- package/src/web-dashboard/public/assets/index-CsySzR41.js.map +1 -0
- package/src/web-dashboard/public/index.html +2 -2
- package/dist/tools/child-agent-worker.js +0 -212
- package/src/web-dashboard/public/assets/index-CxDj7p6i.js +0 -207
- package/src/web-dashboard/public/assets/index-CxDj7p6i.js.map +0 -1
|
@@ -0,0 +1,704 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Child-agent runtime (`src/tools/child-agent-runtime.ts`).
|
|
3
|
+
*
|
|
4
|
+
* The work a forked subagent process actually does: resolve a REAL inference
|
|
5
|
+
* provider from the user's own configuration, run a bounded think → act →
|
|
6
|
+
* observe loop, execute any tool the model asks for through the REAL registry,
|
|
7
|
+
* and return the model's own output.
|
|
8
|
+
*
|
|
9
|
+
* ── What this replaces ──────────────────────────────────────────────────────
|
|
10
|
+
* `child-agent-worker.ts` was a simulation wearing the shape of an agent loop:
|
|
11
|
+
* its `LLMClient.call()` switched on keywords in the goal and returned canned
|
|
12
|
+
* strings ("I've analyzed the task: …"), and its `ToolExecutor.execute()`
|
|
13
|
+
* returned `Executed <tool>` without running anything. A subagent built on it
|
|
14
|
+
* reported work that never happened — the finding-#4 defect in the same
|
|
15
|
+
* workstream, which is why the tool was marked NOT CONNECTED in the registry.
|
|
16
|
+
*
|
|
17
|
+
* ── The rule this module follows ────────────────────────────────────────────
|
|
18
|
+
* Nothing is synthesised. If no provider can be constructed, or the backend is
|
|
19
|
+
* not reachable, or the provider cannot do tool-calling while tools were
|
|
20
|
+
* requested, this THROWS a typed refusal instead of returning a plausible
|
|
21
|
+
* answer. `SubagentRefusalError.code` carries the reason so the parent records
|
|
22
|
+
* it as the failure it is.
|
|
23
|
+
*
|
|
24
|
+
* The provider factory and the tool runner are injectable so the loop is
|
|
25
|
+
* testable without a network; production passes neither and gets the real ones.
|
|
26
|
+
*/
|
|
27
|
+
import { ConfigManager } from '../config/manager.js';
|
|
28
|
+
import { ProviderFactory } from '../inference/factory.js';
|
|
29
|
+
import { resolveAdapterDefault } from '../learning/model-selection.js';
|
|
30
|
+
import { SubagentRefusalError } from './subagent-refusal.js';
|
|
31
|
+
import { getTool, toolJsonSchemas, TOOL_CONTRACT_JSON } from './registry.js';
|
|
32
|
+
// WS1 — the finding tool's bus event; forwarded to the parent as its own frame.
|
|
33
|
+
import { FINDING_EVENT } from './finding-tool.js';
|
|
34
|
+
import { sessionDebugLog } from '../observability/debug-log.js';
|
|
35
|
+
// WS3 (#25) — the child's own span tree, exported over OTLP when the operator
|
|
36
|
+
// asked for it. This is a SECOND process with its own provider, so its trace
|
|
37
|
+
// only joins the parent's when the parent handed it a `traceparent`.
|
|
38
|
+
import { flushSpans, parentContextFromEnv, shutdownSpans, startTurnSpan, TOOL_SPAN_PREFIX, withSpanActive, } from '../observability/otel.js';
|
|
39
|
+
// WS4 (#26) — the SAME operator hooks, in the child's own process. The child
|
|
40
|
+
// reads its own config and inherits the parent's environment, so a hook declared
|
|
41
|
+
// either way applies here too; without this, a veto that holds on every
|
|
42
|
+
// in-process surface would leak through a forked subagent.
|
|
43
|
+
import { runBeforeToolHooks, runToolOutcomeHooks, toolHookRefusalText, } from './tool-hooks.js';
|
|
44
|
+
// WS6 (#28) — the declared fault seam, read in the child's own process (the
|
|
45
|
+
// declaration arrives through the environment the parent handed it).
|
|
46
|
+
import { faultAt } from '../runtime/fault-injection.js';
|
|
47
|
+
// WS5 (#27) — the child's OWN partial resume. The model calls happen in THIS
|
|
48
|
+
// process, so a resume that only the parent could do would replay nothing: the
|
|
49
|
+
// child records its steps into its own store and replays the ones whose input is
|
|
50
|
+
// unchanged, exactly as the in-process loop does (`tools/tool-loop.ts`).
|
|
51
|
+
import { closeResume, openResume, resolveResumeRequest, stepDigest, } from '../learning/step-checkpoint.js';
|
|
52
|
+
// G1 — the verification gate, in the CHILD's own engine. `write_file`/`edit_file`
|
|
53
|
+
// are mutations (`edit-verification.ts`), so an in-process turn that writes one
|
|
54
|
+
// gets one bounded nudge before it can answer and reports the residual honestly.
|
|
55
|
+
// The forked child had no gate at all, so the SAME write inside a delegated run
|
|
56
|
+
// was never followed by "nothing observed the result, run a check" — measured by
|
|
57
|
+
// the parity harness as `modelCalls 3 vs 2` across surfaces
|
|
58
|
+
// (see docs/TOOL_TRUTHFULNESS_TRACKER.md). The gate is imported, not reimplemented,
|
|
59
|
+
// so the nudge text, the tool classification and the honesty flags cannot drift
|
|
60
|
+
// between the loop that runs in this process and the loop that runs in the parent's.
|
|
61
|
+
import { assessEditActivity, detectUnverifiedEditClaim, isMutationTool, isVerificationTool, verificationNudgeFor, } from './edit-verification.js';
|
|
62
|
+
/**
|
|
63
|
+
* Which transport carries tool calls for this provider.
|
|
64
|
+
*
|
|
65
|
+
* Most hosted providers speak the OpenAI `tools` protocol. A local Ollama model
|
|
66
|
+
* (the `local` adapter) does not, and that used to make the subagent REFUSE when
|
|
67
|
+
* tools were asked for — honest, but it meant a local-only setup could not use
|
|
68
|
+
* tools at all. The fallback closes that: the tool names, argument shapes and a
|
|
69
|
+
* `{"tool":…,"arguments":…}` contract ride in the prompt, and the reply is parsed
|
|
70
|
+
* with the same `extractFallbackToolCalls` the chat loop uses for exactly this.
|
|
71
|
+
*/
|
|
72
|
+
function transportFor(provider) {
|
|
73
|
+
return typeof provider.generateTools === 'function' ? 'native' : 'json';
|
|
74
|
+
}
|
|
75
|
+
const DEFAULT_MAX_LLM_CALLS = 25;
|
|
76
|
+
const DEFAULT_MAX_ITERATIONS = 12;
|
|
77
|
+
/** Tools a subagent must never call, whatever the caller asks for. */
|
|
78
|
+
const ALWAYS_BLOCKED = new Set(['ask_user', 'respond']);
|
|
79
|
+
function buildSystemPrompt(config, tools, transport = 'none') {
|
|
80
|
+
const lines = [
|
|
81
|
+
`You are a subagent. Your task: ${config.goal}`,
|
|
82
|
+
'',
|
|
83
|
+
tools.length
|
|
84
|
+
? `You have these tools: ${tools.join(', ')}. Use them to gather what you need, then answer.`
|
|
85
|
+
: 'You have no tools. Answer from what you already know and say so if you cannot.',
|
|
86
|
+
];
|
|
87
|
+
if (transport === 'json') {
|
|
88
|
+
// No tool protocol on this provider, so the contract is stated in the prompt
|
|
89
|
+
// — the same text the chat and execute loops use, never a private re-wording.
|
|
90
|
+
lines.push('', TOOL_CONTRACT_JSON);
|
|
91
|
+
}
|
|
92
|
+
lines.push('', 'Work in steps. When the task is done, reply with the final answer as plain text —', 'no preamble, no instructions to the user.');
|
|
93
|
+
return lines.join('\n');
|
|
94
|
+
}
|
|
95
|
+
/**
|
|
96
|
+
* Resolve the tool allow-list: the request, minus anything always blocked, minus
|
|
97
|
+
* anything the caller listed as blocked, keeping only names the registry knows.
|
|
98
|
+
* A name that does not exist is dropped loudly (returned) rather than silently.
|
|
99
|
+
*/
|
|
100
|
+
export function resolveToolAllowList(config) {
|
|
101
|
+
const blocked = new Set([...(config.blockedTools ?? []), ...ALWAYS_BLOCKED]);
|
|
102
|
+
const unknown = [];
|
|
103
|
+
const allowed = [];
|
|
104
|
+
for (const name of config.tools ?? []) {
|
|
105
|
+
if (blocked.has(name))
|
|
106
|
+
continue;
|
|
107
|
+
if (!getTool(name)) {
|
|
108
|
+
unknown.push(name);
|
|
109
|
+
continue;
|
|
110
|
+
}
|
|
111
|
+
allowed.push(name);
|
|
112
|
+
}
|
|
113
|
+
return { allowed, unknown };
|
|
114
|
+
}
|
|
115
|
+
/**
|
|
116
|
+
* Run the subagent to completion.
|
|
117
|
+
*
|
|
118
|
+
* @throws SubagentRefusalError when nothing real can be run — no constructible
|
|
119
|
+
* provider, an unreachable backend, or tools requested on a provider that
|
|
120
|
+
* cannot call them.
|
|
121
|
+
*/
|
|
122
|
+
export async function runSubagent(config, hooks = {}) {
|
|
123
|
+
const send = hooks.send ?? (() => { });
|
|
124
|
+
const maxLlmCalls = config.maxLlmCalls ?? DEFAULT_MAX_LLM_CALLS;
|
|
125
|
+
const maxIterations = config.maxIterations ?? DEFAULT_MAX_ITERATIONS;
|
|
126
|
+
const { allowed, unknown } = resolveToolAllowList(config);
|
|
127
|
+
if (unknown.length > 0) {
|
|
128
|
+
throw new SubagentRefusalError('unsupported_format', `Unknown tool(s) requested for this subagent: ${unknown.join(', ')}. ` +
|
|
129
|
+
`Available: ${toolJsonSchemas().length} registered tools.`);
|
|
130
|
+
}
|
|
131
|
+
const { provider, type, model } = await createProvider(hooks, config);
|
|
132
|
+
// Which provider/model/transport is about to serve this run, announced BEFORE
|
|
133
|
+
// anything can fail. The parent records these on the run, so a subagent that
|
|
134
|
+
// dies on its very first model call is still attributable — "groq rejected
|
|
135
|
+
// gemma-4-26b-a4b-it over the native tool protocol" — instead of leaving a
|
|
136
|
+
// bare error string and no way to tell which backend produced it.
|
|
137
|
+
const hasTools = allowed.length > 0;
|
|
138
|
+
const transport = transportFor(provider);
|
|
139
|
+
send({
|
|
140
|
+
type: 'progress',
|
|
141
|
+
phase: 'starting',
|
|
142
|
+
provider: type,
|
|
143
|
+
...(model ? { model } : {}),
|
|
144
|
+
tools: allowed,
|
|
145
|
+
transport: hasTools ? transport : 'none',
|
|
146
|
+
});
|
|
147
|
+
// WS2 (#24) — the optional session debug log for this child process.
|
|
148
|
+
// Opened HERE, after the provider is constructed, so its header can name the
|
|
149
|
+
// backend from the first line; `null` unless `NUVIRA_DEBUG_LOG` is set (the
|
|
150
|
+
// parent's process env is inherited across the fork).
|
|
151
|
+
const debugLog = sessionDebugLog({
|
|
152
|
+
surface: 'subagent',
|
|
153
|
+
goal: config.goal,
|
|
154
|
+
backend: {
|
|
155
|
+
engine: 'loop',
|
|
156
|
+
provider: type,
|
|
157
|
+
...(model ? { model } : {}),
|
|
158
|
+
transport: hasTools ? transport : 'none',
|
|
159
|
+
},
|
|
160
|
+
});
|
|
161
|
+
debugLog?.event('turn.start', { tools: allowed.length, transport: hasTools ? transport : 'none' });
|
|
162
|
+
// WS3 (#25) — the child's turn span. Started from the `traceparent` the parent
|
|
163
|
+
// injected into this process's environment, so the child hangs off the tool
|
|
164
|
+
// call that spawned it and the whole turn is ONE trace. With no `traceparent`
|
|
165
|
+
// (a child started by hand, or the parent not tracing) this starts a trace of
|
|
166
|
+
// its own — which is the honest outcome, not a fabricated parent id.
|
|
167
|
+
const otelSpan = await startTurnSpan({
|
|
168
|
+
surface: 'subagent',
|
|
169
|
+
goal: config.goal,
|
|
170
|
+
parent: parentContextFromEnv(),
|
|
171
|
+
});
|
|
172
|
+
/**
|
|
173
|
+
* Finish the span tree and let the process go.
|
|
174
|
+
*
|
|
175
|
+
* The forked child is a ONE-SHOT process: unlike the dashboard or the
|
|
176
|
+
* gateway, nothing else will use this provider after the answer is reported,
|
|
177
|
+
* so shutting it down here is what stops a lingering exporter socket or batch
|
|
178
|
+
* timer from keeping the child alive after it has said what it did.
|
|
179
|
+
*/
|
|
180
|
+
let spansFinished = false;
|
|
181
|
+
const finishSpans = async (outcome) => {
|
|
182
|
+
if (!otelSpan || spansFinished)
|
|
183
|
+
return;
|
|
184
|
+
spansFinished = true;
|
|
185
|
+
otelSpan.end(outcome);
|
|
186
|
+
await flushSpans();
|
|
187
|
+
await shutdownSpans();
|
|
188
|
+
};
|
|
189
|
+
/**
|
|
190
|
+
* Write the child's debug log and announce where it landed.
|
|
191
|
+
*
|
|
192
|
+
* The path travels as a `progress` frame (the same channel `starting` and
|
|
193
|
+
* `finding` use) because the child has no console of its own worth reading —
|
|
194
|
+
* an unattended fork's stdout is easy to lose, and a log nobody can find is
|
|
195
|
+
* not an attachable artifact. The parent ignores phases it does not know.
|
|
196
|
+
*/
|
|
197
|
+
let debugLogFinished = false;
|
|
198
|
+
const finishDebugLog = (detail = {}) => {
|
|
199
|
+
if (!debugLog || debugLogFinished)
|
|
200
|
+
return;
|
|
201
|
+
debugLogFinished = true;
|
|
202
|
+
debugLog.event('turn.end', detail);
|
|
203
|
+
const path = debugLog.write();
|
|
204
|
+
if (path)
|
|
205
|
+
send({ type: 'progress', phase: 'debug_log', path });
|
|
206
|
+
};
|
|
207
|
+
// WS5 (#27) — the child's own resume ledger, and the frame that reports what it
|
|
208
|
+
// did with it. Opened before the loop (a record is read once, not per step) and
|
|
209
|
+
// closed on every path that produces a result, so a child that ran is a child
|
|
210
|
+
// whose steps the NEXT resume can replay — and a child that could not write the
|
|
211
|
+
// record says so instead of reporting a resume that silently did nothing.
|
|
212
|
+
const resumeRequest = resolveResumeRequest({ resume: config.resume });
|
|
213
|
+
const resumeCwd = config.cwd ?? process.cwd();
|
|
214
|
+
const resume = resumeRequest
|
|
215
|
+
? openResume({ goal: config.goal, cwd: resumeCwd, resume: resumeRequest })
|
|
216
|
+
: null;
|
|
217
|
+
const finishResume = (out) => {
|
|
218
|
+
if (!resume)
|
|
219
|
+
return out;
|
|
220
|
+
const outcome = closeResume(resume, { goal: config.goal, cwd: resumeCwd });
|
|
221
|
+
send({ type: 'progress', phase: 'resume', resume: outcome });
|
|
222
|
+
return { ...out, resume: outcome };
|
|
223
|
+
};
|
|
224
|
+
// A backend that cannot be reached is a refusal, not an empty result.
|
|
225
|
+
const available = await provider.isAvailable().catch(() => false);
|
|
226
|
+
if (!available) {
|
|
227
|
+
// WS2 — a REFUSAL is exactly the run a bug report is about, so the log is
|
|
228
|
+
// still written (with the backend already in its header) before throwing.
|
|
229
|
+
finishDebugLog({ refused: 'provider_not_reachable', provider: type });
|
|
230
|
+
// WS3 — and the span is shipped red. A child whose provider was unreachable
|
|
231
|
+
// is the shape an operator most needs to see in a trace, not an absence.
|
|
232
|
+
const refused = `Provider '${type}' is not reachable.`;
|
|
233
|
+
await finishSpans({ ok: false, message: refused });
|
|
234
|
+
throw new SubagentRefusalError('not_configured', `Provider '${type}' is not reachable. Configure it (or start its backend, e.g. \`ollama serve\`) and re-run.`);
|
|
235
|
+
}
|
|
236
|
+
// ── Tool loop (native protocol, or the shared JSON fallback) ──────────────
|
|
237
|
+
//
|
|
238
|
+
// WS6 (#28) — EVERYTHING below runs under one guard, because a call that throws
|
|
239
|
+
// mid-run used to leave NOTHING behind: the unreachable-provider refusal above
|
|
240
|
+
// writes its debug log and ships a red span before it throws, while a provider
|
|
241
|
+
// that failed INSIDE a call skipped both. Measured by the provider-fault parity
|
|
242
|
+
// row: the child reported `written: false` and `exported: false`, so a crashed
|
|
243
|
+
// subagent produced no attachable log and no trace at all — the two artifacts
|
|
244
|
+
// WS2 and WS3 exist to guarantee, missing on exactly the run an operator most
|
|
245
|
+
// needs them for.
|
|
246
|
+
/**
|
|
247
|
+
* Leave the same evidence a refusal leaves, then rethrow.
|
|
248
|
+
*
|
|
249
|
+
* WS6 (#28) — a provider that failed INSIDE a call used to leave NOTHING
|
|
250
|
+
* behind: the unreachable-provider refusal above writes its debug log and ships
|
|
251
|
+
* a red span before it throws, while a call that threw skipped both. Measured by
|
|
252
|
+
* the provider-fault parity row: the child reported `written: false` and
|
|
253
|
+
* `exported: false`, so a crashed subagent produced no attachable log and no
|
|
254
|
+
* trace at all — the two artifacts WS2 and WS3 exist to guarantee, missing on
|
|
255
|
+
* exactly the run an operator most needs them for. The error is rethrown
|
|
256
|
+
* unchanged, so the parent still receives the honest failure frame.
|
|
257
|
+
*/
|
|
258
|
+
const failWithEvidence = async (err) => {
|
|
259
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
260
|
+
finishDebugLog({ failed: message, provider: type });
|
|
261
|
+
await finishSpans({ ok: false, message });
|
|
262
|
+
throw err;
|
|
263
|
+
};
|
|
264
|
+
if (hasTools) {
|
|
265
|
+
const out = await runToolLoop(config, allowed, provider, type, transport, {
|
|
266
|
+
send,
|
|
267
|
+
runTool: hooks.runTool,
|
|
268
|
+
maxLlmCalls,
|
|
269
|
+
maxIterations,
|
|
270
|
+
model,
|
|
271
|
+
debug: debugLog,
|
|
272
|
+
otel: otelSpan,
|
|
273
|
+
resume: resume?.ledger ?? null,
|
|
274
|
+
}).catch(failWithEvidence);
|
|
275
|
+
finishDebugLog({
|
|
276
|
+
llmCalls: out.llmCalls,
|
|
277
|
+
toolCalls: out.toolCalls,
|
|
278
|
+
truncated: out.truncated,
|
|
279
|
+
transport: out.transport,
|
|
280
|
+
});
|
|
281
|
+
otelSpan?.attr('nuvira.llmCalls', out.llmCalls);
|
|
282
|
+
otelSpan?.attr('nuvira.toolCalls', out.toolCalls);
|
|
283
|
+
await finishSpans({
|
|
284
|
+
// A run that hit its ceiling before the model stopped is not a failure —
|
|
285
|
+
// it is a bounded run, and the loop reports that fact rather than an error.
|
|
286
|
+
ok: true,
|
|
287
|
+
...(out.truncated ? { message: 'the subagent reached its ceiling' } : {}),
|
|
288
|
+
});
|
|
289
|
+
return finishResume(out);
|
|
290
|
+
}
|
|
291
|
+
// ── Plain completion (no tools) ───────────────────────────────────────────
|
|
292
|
+
const prompt = [
|
|
293
|
+
buildSystemPrompt(config, []),
|
|
294
|
+
'',
|
|
295
|
+
`Task: ${config.goal}`,
|
|
296
|
+
].join('\n');
|
|
297
|
+
const text = await provider.generate(prompt, modelOption(model)).catch(failWithEvidence);
|
|
298
|
+
finishDebugLog({ llmCalls: 1, toolCalls: 0, transport: 'none' });
|
|
299
|
+
await finishSpans({ ok: true });
|
|
300
|
+
return finishResume({
|
|
301
|
+
result: text.trim(),
|
|
302
|
+
llmCalls: 1,
|
|
303
|
+
toolCalls: 0,
|
|
304
|
+
provider: type,
|
|
305
|
+
...(model ? { model } : {}),
|
|
306
|
+
transport: 'none',
|
|
307
|
+
truncated: false,
|
|
308
|
+
});
|
|
309
|
+
}
|
|
310
|
+
/**
|
|
311
|
+
* One model call, on whichever transport this provider speaks.
|
|
312
|
+
*
|
|
313
|
+
* The fallback path reuses `buildJsonFallbackPrompt` and
|
|
314
|
+
* `extractFallbackToolCalls` — the same pair the chat loop and the execute loop
|
|
315
|
+
* use — so a subagent speaks the dialect the rest of the system already parses.
|
|
316
|
+
* They are imported lazily because `tool-loop.ts` pulls in the registry and the
|
|
317
|
+
* event bus, which a plain completion should not pay for.
|
|
318
|
+
*/
|
|
319
|
+
async function callModel(provider, transport, messages, schemas, model) {
|
|
320
|
+
if (transport === 'native') {
|
|
321
|
+
return provider.generateTools(messages, schemas, modelOption(model));
|
|
322
|
+
}
|
|
323
|
+
const { buildJsonFallbackPrompt } = await import('../inference/tool-call-utils.js');
|
|
324
|
+
const { extractFallbackToolCalls } = await import('./tool-loop.js');
|
|
325
|
+
const raw = await provider.generate(buildJsonFallbackPrompt(messages, schemas), modelOption(model));
|
|
326
|
+
const { text, calls } = extractFallbackToolCalls(raw);
|
|
327
|
+
return {
|
|
328
|
+
content: text,
|
|
329
|
+
toolCalls: calls.map((c) => ({ id: c.id, name: c.name, arguments: c.arguments })),
|
|
330
|
+
};
|
|
331
|
+
}
|
|
332
|
+
/** Only pass a model when one was actually resolved — 'auto' is not a model id. */
|
|
333
|
+
function modelOption(model) {
|
|
334
|
+
return model ? { model } : undefined;
|
|
335
|
+
}
|
|
336
|
+
/**
|
|
337
|
+
* The model id the adapter will put on the wire for this subagent.
|
|
338
|
+
*
|
|
339
|
+
* Resolved with `resolveAdapterDefault` — the SAME function every adapter calls
|
|
340
|
+
* for itself when the caller passes no model — so the name recorded on a run
|
|
341
|
+
* cannot disagree with the name actually sent. It is then passed EXPLICITLY to
|
|
342
|
+
* every model call, which makes that guarantee structural rather than a promise:
|
|
343
|
+
* `options.model` wins over the adapter's own fallback, so a run that reports
|
|
344
|
+
* `model: X` really was sent X.
|
|
345
|
+
*
|
|
346
|
+
* Returns undefined when nothing can be resolved; the adapter then raises its own
|
|
347
|
+
* clear "no model resolved — run `nuvira models refresh`" error, which is the
|
|
348
|
+
* failure a run should carry rather than an invented name.
|
|
349
|
+
*/
|
|
350
|
+
function effectiveModel(providerType, configuredModel) {
|
|
351
|
+
return resolveAdapterDefault(providerType, configuredModel === 'auto' ? undefined : configuredModel);
|
|
352
|
+
}
|
|
353
|
+
async function createProvider(hooks, config) {
|
|
354
|
+
if (hooks.createProvider) {
|
|
355
|
+
const { provider, type } = await hooks.createProvider(config.provider, config.model);
|
|
356
|
+
// An injected factory names its own model; there is no adapter registry to
|
|
357
|
+
// consult, so a caller-named model is reported as-is and nothing is invented.
|
|
358
|
+
const named = config.model && config.model !== 'auto' ? config.model : undefined;
|
|
359
|
+
return { provider, type, ...(named ? { model: named } : {}) };
|
|
360
|
+
}
|
|
361
|
+
const configManager = new ConfigManager();
|
|
362
|
+
const requested = config.provider && config.provider !== 'auto' ? config.provider : 'auto';
|
|
363
|
+
const { type, config: providerConfig } = configManager.getProviderConfig(requested);
|
|
364
|
+
if (!ProviderFactory.isConstructible(type)) {
|
|
365
|
+
throw new SubagentRefusalError('not_configured', `Provider '${type}' has no adapter, so no subagent call can be made with it. Configure a supported provider.`);
|
|
366
|
+
}
|
|
367
|
+
// A model named for the subagent wins over the provider's default; the adapter
|
|
368
|
+
// still receives a real id because 'auto' is filtered out above.
|
|
369
|
+
const merged = config.model && config.model !== 'auto' ? { ...providerConfig, model: config.model } : providerConfig;
|
|
370
|
+
const model = effectiveModel(type, merged.model);
|
|
371
|
+
return { provider: ProviderFactory.createProvider(type, merged), type, ...(model ? { model } : {}) };
|
|
372
|
+
}
|
|
373
|
+
async function runToolLoop(config, allowed, provider, type, transport, loop) {
|
|
374
|
+
const schemas = toolJsonSchemas(allowed).map((t) => ({
|
|
375
|
+
name: t.name,
|
|
376
|
+
description: t.description,
|
|
377
|
+
parameters: t.parameters,
|
|
378
|
+
}));
|
|
379
|
+
const messages = [
|
|
380
|
+
{ role: 'system', content: buildSystemPrompt(config, allowed, transport) },
|
|
381
|
+
{ role: 'user', content: config.goal },
|
|
382
|
+
];
|
|
383
|
+
const model = loop.model;
|
|
384
|
+
const base = { provider: type, ...(model ? { model } : {}), transport };
|
|
385
|
+
let llmCalls = 0;
|
|
386
|
+
let toolCalls = 0;
|
|
387
|
+
// G1 — the accumulators the verification gate reads, the same three the
|
|
388
|
+
// in-process loop keeps (`successfulToolCalls` / `mutatedPaths` /
|
|
389
|
+
// `verificationEvidence`). Only calls that actually RAN SUCCEEDED are recorded:
|
|
390
|
+
// a refusal is neither a mutation nor a verification.
|
|
391
|
+
const successfulToolCalls = [];
|
|
392
|
+
const mutatedPaths = [];
|
|
393
|
+
const verificationEvidence = [];
|
|
394
|
+
// Bounded exactly once, like the in-process gate (`verificationNudges < 1`).
|
|
395
|
+
let verificationNudges = 0;
|
|
396
|
+
// The iteration ceiling is the loop's `loop.maxIterations` plus one for each
|
|
397
|
+
// nudge spent: the ceiling bounds the WORK, and a nudge the loop itself asked
|
|
398
|
+
// for must not eat the budget the goal was owed (in-process does the same by
|
|
399
|
+
// raising its step limit).
|
|
400
|
+
let iterationLimit = loop.maxIterations;
|
|
401
|
+
// One config read per loop rather than per call: the hooks are declared in the
|
|
402
|
+
// child's own configuration (this process's, not the parent's), and the
|
|
403
|
+
// declarations are resolved through the same manager the tools use.
|
|
404
|
+
const hookConfigManager = new ConfigManager();
|
|
405
|
+
// The in-process loop's refusal classifier is the authority for "this call did
|
|
406
|
+
// NOT run": `write_file` refusing a path outside the workspace returns
|
|
407
|
+
// "… escapes the workspace … — denied" with NO `Error:` prefix, so a bare prefix
|
|
408
|
+
// check would count a refused write as a mutation and nudge the child to verify
|
|
409
|
+
// a file that was never written. Imported lazily for the same reason
|
|
410
|
+
// `extractFallbackToolCalls` is: `tool-loop.ts` pulls in the registry and the
|
|
411
|
+
// event bus.
|
|
412
|
+
const { classifyToolRefusal } = await import('./tool-loop.js');
|
|
413
|
+
/**
|
|
414
|
+
* Finish the loop, annotating the result with the honesty flags.
|
|
415
|
+
*
|
|
416
|
+
* Computed from the SAME accumulators the gate reads, so the nudge and the flag
|
|
417
|
+
* can never disagree — and computed on EVERY exit (including a ceiling), since a
|
|
418
|
+
* run cut short with a mutation unobserved is exactly the one that needs the
|
|
419
|
+
* flag. Nothing here depends on whether the nudge fired.
|
|
420
|
+
*/
|
|
421
|
+
const finish = (result, truncated) => {
|
|
422
|
+
const activity = assessEditActivity(successfulToolCalls, verificationEvidence, mutatedPaths);
|
|
423
|
+
const out = { result, llmCalls, toolCalls, ...base, truncated };
|
|
424
|
+
if (activity.needsVerification)
|
|
425
|
+
out.unverifiedEdit = true;
|
|
426
|
+
if (detectUnverifiedEditClaim(result, activity.mutations, activity.verifications)) {
|
|
427
|
+
out.unverifiedEditClaim = true;
|
|
428
|
+
}
|
|
429
|
+
return out;
|
|
430
|
+
};
|
|
431
|
+
for (let iteration = 0; iteration < iterationLimit; iteration += 1) {
|
|
432
|
+
if (llmCalls >= loop.maxLlmCalls) {
|
|
433
|
+
return finish('Subagent stopped: reached its model-call ceiling before finishing.', true);
|
|
434
|
+
}
|
|
435
|
+
// WS5 (#27) — a resumed child replays this step when its input is unchanged.
|
|
436
|
+
// The digest is over the WHOLE input (the thread AND the schema), so a step
|
|
437
|
+
// whose tool result or tool list differs MISSES and is paid for again — the
|
|
438
|
+
// property that makes a replay an answer to the same question rather than to
|
|
439
|
+
// the same step number.
|
|
440
|
+
const stepKey = `model:${iteration + 1}`;
|
|
441
|
+
const stepHash = loop.resume ? stepDigest(messages, schemas) : '';
|
|
442
|
+
const replayed = loop.resume?.replay(stepKey, stepHash) ?? null;
|
|
443
|
+
let response;
|
|
444
|
+
if (replayed) {
|
|
445
|
+
response = replayed;
|
|
446
|
+
// A REPLAYED step is not a model call, so `llmCalls` is not incremented —
|
|
447
|
+
// the count the parent records and the debug log carry is the number of
|
|
448
|
+
// calls this run actually made (see `ResumeOutcome.modelCalls`).
|
|
449
|
+
loop.send({ type: 'progress', phase: 'resume_step', step: stepKey });
|
|
450
|
+
loop.send({ type: 'progress', phase: 'thinking', iteration: iteration + 1, llmCalls, toolCalls });
|
|
451
|
+
}
|
|
452
|
+
else {
|
|
453
|
+
// WS6 (#28) — the ATTEMPT is counted BEFORE it is made, and that ordering is
|
|
454
|
+
// the fix for a measured dishonesty: a model call that THREW used to leave
|
|
455
|
+
// `llmCalls` unchanged, so a child whose provider failed every call reported
|
|
456
|
+
// ZERO model calls — as if it never reached a model at all. The in-process
|
|
457
|
+
// surfaces' counts come from the wire and already include failed attempts, so
|
|
458
|
+
// this is also what makes the child's count comparable across the fork; and
|
|
459
|
+
// the frame that announces the call now carries the incremented count, which
|
|
460
|
+
// is the only channel the parent has (the child dies before a result frame).
|
|
461
|
+
llmCalls += 1;
|
|
462
|
+
loop.send({ type: 'progress', phase: 'thinking', iteration: iteration + 1, llmCalls, toolCalls });
|
|
463
|
+
response = await callModel(provider, transport, messages, schemas, model);
|
|
464
|
+
loop.resume?.record(stepKey, stepHash, response);
|
|
465
|
+
}
|
|
466
|
+
if (response.toolCalls.length === 0) {
|
|
467
|
+
// ── G1 — VERIFICATION GATE ────────────────────────────────────────────
|
|
468
|
+
// The model is about to answer, but this run MUTATED the workspace and
|
|
469
|
+
// nothing observed the result: spend ONE bounded nudge asking for the check
|
|
470
|
+
// (the same nudge the in-process loop sends, naming THIS workspace's
|
|
471
|
+
// strongest real command). A nudge, not a hard block — a task with no
|
|
472
|
+
// runnable check must still finish, and the residual `unverifiedEdit` flag
|
|
473
|
+
// carries the honesty for that case.
|
|
474
|
+
if (verificationNudges < 1 &&
|
|
475
|
+
assessEditActivity(successfulToolCalls, verificationEvidence, mutatedPaths).needsVerification) {
|
|
476
|
+
verificationNudges += 1;
|
|
477
|
+
iterationLimit += 1;
|
|
478
|
+
const mutations = successfulToolCalls.filter(isMutationTool);
|
|
479
|
+
// A frame, because the child's stdout is easy to lose and the parent is
|
|
480
|
+
// its only channel; the GATE is reported for the same reason the
|
|
481
|
+
// in-process loop records a `gate` trace event — a run that asked for a
|
|
482
|
+
// check and one that never needed to are otherwise indistinguishable.
|
|
483
|
+
loop.send({
|
|
484
|
+
type: 'progress',
|
|
485
|
+
phase: 'gate',
|
|
486
|
+
gate: 'verification',
|
|
487
|
+
summary: 'the subagent mutated the workspace and nothing observed the result — one bounded nudge to verify',
|
|
488
|
+
mutations,
|
|
489
|
+
llmCalls,
|
|
490
|
+
toolCalls,
|
|
491
|
+
});
|
|
492
|
+
loop.debug?.event('gate.verification', { mutations });
|
|
493
|
+
messages.push({ role: 'assistant', content: response.content });
|
|
494
|
+
messages.push({
|
|
495
|
+
role: 'user',
|
|
496
|
+
content: verificationNudgeFor(config.cwd ?? process.cwd(), mutatedPaths),
|
|
497
|
+
});
|
|
498
|
+
continue;
|
|
499
|
+
}
|
|
500
|
+
return finish(response.content.trim(), false);
|
|
501
|
+
}
|
|
502
|
+
// Replay the assistant turn exactly as the provider asked for it — the
|
|
503
|
+
// tool-call ids and any providerMeta (Gemini's thoughtSignature) must go back
|
|
504
|
+
// verbatim or the next turn is rejected.
|
|
505
|
+
messages.push({
|
|
506
|
+
role: 'assistant',
|
|
507
|
+
content: response.content,
|
|
508
|
+
toolCalls: response.toolCalls.map((call) => ({
|
|
509
|
+
id: call.id,
|
|
510
|
+
name: call.name,
|
|
511
|
+
arguments: JSON.stringify(call.arguments),
|
|
512
|
+
...(call.providerMeta ? { providerMeta: call.providerMeta } : {}),
|
|
513
|
+
})),
|
|
514
|
+
});
|
|
515
|
+
for (const call of response.toolCalls) {
|
|
516
|
+
if (!allowed.includes(call.name)) {
|
|
517
|
+
loop.debug?.event('tool.refused', { tool: call.name });
|
|
518
|
+
messages.push({
|
|
519
|
+
role: 'tool',
|
|
520
|
+
content: `Refused: '${call.name}' is not available to this subagent.`,
|
|
521
|
+
toolCallId: call.id,
|
|
522
|
+
});
|
|
523
|
+
continue;
|
|
524
|
+
}
|
|
525
|
+
// Two frames, mirroring the main loop's `tool:started` → `tool:called`
|
|
526
|
+
// pair: the CALL, then its OUTCOME. The parent used to hear the name and
|
|
527
|
+
// nothing else, so a run's tool calls were unattributable — a call that
|
|
528
|
+
// FAILED looked exactly like one that worked (recorded on #22 as
|
|
529
|
+
// tool-call-lifecycle@subagent).
|
|
530
|
+
loop.send({ type: 'progress', phase: 'tool_call', tool: call.name });
|
|
531
|
+
loop.debug?.event('tool.start', { tool: call.name });
|
|
532
|
+
const toolStartedAt = Date.now();
|
|
533
|
+
// WS4 (#26) — the `before` hooks. A veto stops the call here, so a hook
|
|
534
|
+
// that holds on the CLI holds for a forked subagent too — the child is a
|
|
535
|
+
// separate PROCESS, and a policy that stopped at the process boundary would
|
|
536
|
+
// be worse than no policy, because it would look enforced.
|
|
537
|
+
const beforeHooks = await runBeforeToolHooks({
|
|
538
|
+
tool: call.name,
|
|
539
|
+
args: call.arguments,
|
|
540
|
+
callId: call.id,
|
|
541
|
+
surface: 'subagent',
|
|
542
|
+
...(config.cwd ? { cwd: config.cwd } : {}),
|
|
543
|
+
configManager: hookConfigManager,
|
|
544
|
+
});
|
|
545
|
+
for (const problem of beforeHooks.problems) {
|
|
546
|
+
// A failed hook is reported to the parent on its own frame rather than
|
|
547
|
+
// swallowed: the seam fails open, so a broken policy would otherwise be
|
|
548
|
+
// indistinguishable from one that approved every call.
|
|
549
|
+
loop.send({ type: 'progress', phase: 'hook_problem', problem });
|
|
550
|
+
}
|
|
551
|
+
if (beforeHooks.denied) {
|
|
552
|
+
const refusal = toolHookRefusalText(beforeHooks);
|
|
553
|
+
loop.debug?.event('tool.refused', { tool: call.name, by: 'tool-hook' });
|
|
554
|
+
// The same pair of frames a call that ran and failed sends, so the parent
|
|
555
|
+
// records the attempt and its outcome exactly as it does everywhere else.
|
|
556
|
+
loop.send({
|
|
557
|
+
type: 'progress',
|
|
558
|
+
phase: 'tool_result',
|
|
559
|
+
tool: call.name,
|
|
560
|
+
ok: false,
|
|
561
|
+
llmCalls,
|
|
562
|
+
toolCalls,
|
|
563
|
+
});
|
|
564
|
+
messages.push({ role: 'tool', content: refusal, toolCallId: call.id });
|
|
565
|
+
continue;
|
|
566
|
+
}
|
|
567
|
+
// WS3 (#25) — the call as a child span of the child's turn, created where
|
|
568
|
+
// the call actually RUNS (the allow-list check above is a refusal, not a
|
|
569
|
+
// call). Same name shape as every other surface's tool span, so the tree
|
|
570
|
+
// an operator reads is the same tree wherever the tool ran.
|
|
571
|
+
const toolSpan = loop.otel?.child(`${TOOL_SPAN_PREFIX}${call.name}`, { 'nuvira.tool': call.name }) ?? null;
|
|
572
|
+
try {
|
|
573
|
+
const output = await withSpanActive(toolSpan, () => executeTool(config, call.name, call.arguments, loop.runTool, (event, data) => {
|
|
574
|
+
// WS1 — a finding the child recorded, shipped on its own frame (the
|
|
575
|
+
// same way its tool lifecycle crosses IPC). The parent records it, so a
|
|
576
|
+
// subagent run reports the verdicts it produced like every other
|
|
577
|
+
// surface instead of leaving them inside the child's process.
|
|
578
|
+
if (event === FINDING_EVENT) {
|
|
579
|
+
loop.send({ type: 'progress', phase: 'finding', finding: data, llmCalls, toolCalls });
|
|
580
|
+
// WS2 — and into the child's own debug log, so a bug report from a
|
|
581
|
+
// forked run carries the verdicts it recorded.
|
|
582
|
+
const finding = data;
|
|
583
|
+
loop.debug?.event('finding', `${finding?.verdict ?? '?'} ${finding?.claim ?? ''}`);
|
|
584
|
+
}
|
|
585
|
+
}));
|
|
586
|
+
toolCalls += 1;
|
|
587
|
+
// The same convention the main loop and the tool registry use: a tool
|
|
588
|
+
// signals failure by returning text that starts with `Error:`.
|
|
589
|
+
const ok = !output.startsWith('Error:');
|
|
590
|
+
// HONEST ACCOUNTING for the gate. A DECLINED call is not a success
|
|
591
|
+
// whatever prefix it used, so it may not count as a mutation (or a
|
|
592
|
+
// verification) — the same rule the in-process loop applies, and the
|
|
593
|
+
// reason `classifyToolRefusal` is consulted here rather than the prefix
|
|
594
|
+
// alone. `ok` stays the parent-facing outcome (unchanged wire shape).
|
|
595
|
+
const ranOk = ok && classifyToolRefusal(output) === null;
|
|
596
|
+
if (ranOk) {
|
|
597
|
+
successfulToolCalls.push(call.name);
|
|
598
|
+
if (isMutationTool(call.name)) {
|
|
599
|
+
const a = call.arguments;
|
|
600
|
+
const p = a?.path ?? a?.file_path ?? a?.file;
|
|
601
|
+
if (typeof p === 'string' && p)
|
|
602
|
+
mutatedPaths.push(p);
|
|
603
|
+
}
|
|
604
|
+
else if (isVerificationTool(call.name)) {
|
|
605
|
+
verificationEvidence.push({ tool: call.name, args: call.arguments, result: output });
|
|
606
|
+
}
|
|
607
|
+
}
|
|
608
|
+
loop.debug?.event('tool.end', { tool: call.name, ok });
|
|
609
|
+
loop.send({
|
|
610
|
+
type: 'progress',
|
|
611
|
+
phase: 'tool_result',
|
|
612
|
+
tool: call.name,
|
|
613
|
+
ok,
|
|
614
|
+
llmCalls,
|
|
615
|
+
toolCalls,
|
|
616
|
+
});
|
|
617
|
+
toolSpan?.attr('nuvira.ok', ok);
|
|
618
|
+
toolSpan?.end({ ok });
|
|
619
|
+
// WS4 (#26) — `after` for a call that succeeded, `failed` for one that
|
|
620
|
+
// did not. Exactly one of the two, because a hook that counts failures
|
|
621
|
+
// must not be told about a success.
|
|
622
|
+
const outcomeHooks = await runToolOutcomeHooks({
|
|
623
|
+
tool: call.name,
|
|
624
|
+
args: call.arguments,
|
|
625
|
+
callId: call.id,
|
|
626
|
+
surface: 'subagent',
|
|
627
|
+
...(config.cwd ? { cwd: config.cwd } : {}),
|
|
628
|
+
configManager: hookConfigManager,
|
|
629
|
+
ok,
|
|
630
|
+
result: output,
|
|
631
|
+
durationMs: Date.now() - toolStartedAt,
|
|
632
|
+
});
|
|
633
|
+
for (const problem of outcomeHooks.problems) {
|
|
634
|
+
loop.send({ type: 'progress', phase: 'hook_problem', problem });
|
|
635
|
+
}
|
|
636
|
+
messages.push({ role: 'tool', content: output, toolCallId: call.id });
|
|
637
|
+
}
|
|
638
|
+
catch (err) {
|
|
639
|
+
// The injected `runTool` override can throw where the registry's own
|
|
640
|
+
// executor would have returned an `Error:` result — both are a failure
|
|
641
|
+
// the `failed` phase is owed.
|
|
642
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
643
|
+
const outcomeHooks = await runToolOutcomeHooks({
|
|
644
|
+
tool: call.name,
|
|
645
|
+
args: call.arguments,
|
|
646
|
+
callId: call.id,
|
|
647
|
+
surface: 'subagent',
|
|
648
|
+
...(config.cwd ? { cwd: config.cwd } : {}),
|
|
649
|
+
configManager: hookConfigManager,
|
|
650
|
+
ok: false,
|
|
651
|
+
error: message,
|
|
652
|
+
durationMs: Date.now() - toolStartedAt,
|
|
653
|
+
});
|
|
654
|
+
for (const problem of outcomeHooks.problems) {
|
|
655
|
+
loop.send({ type: 'progress', phase: 'hook_problem', problem });
|
|
656
|
+
}
|
|
657
|
+
throw err;
|
|
658
|
+
}
|
|
659
|
+
finally {
|
|
660
|
+
// A safety net, not a second report: an injected `runTool` can throw,
|
|
661
|
+
// and a span left open would hang off the turn span for ever.
|
|
662
|
+
toolSpan?.end({ ok: false, message: 'tool outcome was never reported' });
|
|
663
|
+
}
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
return finish('Subagent stopped: reached its iteration ceiling before finishing.', true);
|
|
667
|
+
}
|
|
668
|
+
/**
|
|
669
|
+
* Execute one tool through the REAL registry. Errors come back as text for the
|
|
670
|
+
* model to read (a tool that failed is information, not a crash), which is also
|
|
671
|
+
* what the main tool loop does.
|
|
672
|
+
*/
|
|
673
|
+
async function executeTool(config, name, args, override, emit) {
|
|
674
|
+
// WS6 (#28) — a DECLARED fault, injected before every path (including an
|
|
675
|
+
// injected `runTool`), so a fault declared for a turn reaches a FORKED CHILD the
|
|
676
|
+
// same way it reaches the in-process loop. The child inherits the declaration
|
|
677
|
+
// through its environment, which is why one declaration covers all five
|
|
678
|
+
// surfaces and the parity row can assert it across the process boundary.
|
|
679
|
+
const injected = faultAt('tool', name);
|
|
680
|
+
if (injected)
|
|
681
|
+
return `Error: ${injected.message}`;
|
|
682
|
+
if (override)
|
|
683
|
+
return override(name, args);
|
|
684
|
+
const tool = getTool(name);
|
|
685
|
+
if (!tool)
|
|
686
|
+
return `Error: unknown tool '${name}'.`;
|
|
687
|
+
const ctx = {
|
|
688
|
+
configManager: new ConfigManager(),
|
|
689
|
+
cwd: config.cwd ?? process.cwd(),
|
|
690
|
+
// A tool that reports through the bus (the finding tool emits
|
|
691
|
+
// `finding:recorded`) needs a sink on this side of the fork; the caller
|
|
692
|
+
// forwards it as a frame. Absent for a direct/test invocation, exactly like
|
|
693
|
+
// the other optional context fields.
|
|
694
|
+
...(emit ? { emit } : {}),
|
|
695
|
+
};
|
|
696
|
+
try {
|
|
697
|
+
const out = await tool.run(args, ctx);
|
|
698
|
+
return typeof out === 'string' ? out : JSON.stringify(out);
|
|
699
|
+
}
|
|
700
|
+
catch (err) {
|
|
701
|
+
return `Error: ${err instanceof Error ? err.message : String(err)}`;
|
|
702
|
+
}
|
|
703
|
+
}
|
|
704
|
+
//# sourceMappingURL=child-agent-runtime.js.map
|