agent-nuvira 3.3.1 → 3.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -13
- package/dist/agent-sdk/src/agent.d.ts +2 -0
- package/dist/agent-sdk/src/agent.d.ts.map +1 -1
- package/dist/agent-sdk/src/define.d.ts +64 -0
- package/dist/agent-sdk/src/define.d.ts.map +1 -0
- package/dist/agent-sdk/src/define.js +76 -0
- package/dist/agent-sdk/src/define.js.map +1 -0
- package/dist/agent-sdk/src/index.d.ts +9 -0
- package/dist/agent-sdk/src/index.d.ts.map +1 -1
- package/dist/agent-sdk/src/index.js +9 -0
- package/dist/agent-sdk/src/index.js.map +1 -1
- package/dist/agent-sdk/src/scaffold.d.ts +11 -0
- package/dist/agent-sdk/src/scaffold.d.ts.map +1 -1
- package/dist/agent-sdk/src/scaffold.js +16 -6
- package/dist/agent-sdk/src/scaffold.js.map +1 -1
- package/dist/agents/agents/runner.d.ts +9 -0
- package/dist/agents/agents/runner.d.ts.map +1 -1
- package/dist/agents/agents/runner.js +34 -12
- package/dist/agents/agents/runner.js.map +1 -1
- package/dist/agents/agents/writer-tool-calling.d.ts.map +1 -1
- package/dist/agents/agents/writer-tool-calling.js +7 -2
- package/dist/agents/agents/writer-tool-calling.js.map +1 -1
- package/dist/agents/agents/writer.d.ts.map +1 -1
- package/dist/agents/agents/writer.js +7 -1
- package/dist/agents/agents/writer.js.map +1 -1
- package/dist/agents/artifact-verification.d.ts +102 -0
- package/dist/agents/artifact-verification.d.ts.map +1 -0
- package/dist/agents/artifact-verification.js +216 -0
- package/dist/agents/artifact-verification.js.map +1 -0
- package/dist/agents/checkpoint-store.d.ts +73 -0
- package/dist/agents/checkpoint-store.d.ts.map +1 -1
- package/dist/agents/checkpoint-store.js +161 -0
- package/dist/agents/checkpoint-store.js.map +1 -1
- package/dist/agents/credential-store.d.ts +120 -0
- package/dist/agents/credential-store.d.ts.map +1 -1
- package/dist/agents/credential-store.js +358 -19
- package/dist/agents/credential-store.js.map +1 -1
- package/dist/agents/long-form-plan.d.ts.map +1 -1
- package/dist/agents/long-form-plan.js +2 -1
- package/dist/agents/long-form-plan.js.map +1 -1
- package/dist/agents/orchestrator.d.ts +10 -1
- package/dist/agents/orchestrator.d.ts.map +1 -1
- package/dist/agents/orchestrator.js +239 -79
- package/dist/agents/orchestrator.js.map +1 -1
- package/dist/agents/phase-engine.d.ts +61 -0
- package/dist/agents/phase-engine.d.ts.map +1 -1
- package/dist/agents/phase-engine.js +68 -2
- package/dist/agents/phase-engine.js.map +1 -1
- package/dist/agents/prompt-assembly.d.ts +11 -0
- package/dist/agents/prompt-assembly.d.ts.map +1 -1
- package/dist/agents/prompt-assembly.js +17 -0
- package/dist/agents/prompt-assembly.js.map +1 -1
- package/dist/agents/release-preflight.d.ts +121 -0
- package/dist/agents/release-preflight.d.ts.map +1 -0
- package/dist/agents/release-preflight.js +259 -0
- package/dist/agents/release-preflight.js.map +1 -0
- package/dist/agents/release-runner.d.ts +157 -0
- package/dist/agents/release-runner.d.ts.map +1 -0
- package/dist/agents/release-runner.js +719 -0
- package/dist/agents/release-runner.js.map +1 -0
- package/dist/agents/step-handoff.d.ts +190 -0
- package/dist/agents/step-handoff.d.ts.map +1 -0
- package/dist/agents/step-handoff.js +443 -0
- package/dist/agents/step-handoff.js.map +1 -0
- package/dist/cli/agent.d.ts +2 -2
- package/dist/cli/agent.js +10 -10
- package/dist/cli/benchmark.d.ts.map +1 -1
- package/dist/cli/benchmark.js +11 -4
- package/dist/cli/benchmark.js.map +1 -1
- package/dist/cli/chat.d.ts +94 -0
- package/dist/cli/chat.d.ts.map +1 -1
- package/dist/cli/chat.js +379 -24
- package/dist/cli/chat.js.map +1 -1
- package/dist/cli/cli-program.d.ts.map +1 -1
- package/dist/cli/cli-program.js +8 -0
- package/dist/cli/cli-program.js.map +1 -1
- package/dist/cli/config.d.ts.map +1 -1
- package/dist/cli/config.js +9 -1
- package/dist/cli/config.js.map +1 -1
- package/dist/cli/credentials.d.ts +28 -0
- package/dist/cli/credentials.d.ts.map +1 -0
- package/dist/cli/credentials.js +213 -0
- package/dist/cli/credentials.js.map +1 -0
- package/dist/cli/doctor.d.ts.map +1 -1
- package/dist/cli/doctor.js +3 -2
- package/dist/cli/doctor.js.map +1 -1
- package/dist/cli/edit.js +2 -2
- package/dist/cli/eval.d.ts +17 -0
- package/dist/cli/eval.d.ts.map +1 -1
- package/dist/cli/eval.js +105 -2
- package/dist/cli/eval.js.map +1 -1
- package/dist/cli/execute.d.ts +16 -1
- package/dist/cli/execute.d.ts.map +1 -1
- package/dist/cli/execute.js +192 -22
- package/dist/cli/execute.js.map +1 -1
- package/dist/cli/failover-runner.d.ts.map +1 -1
- package/dist/cli/failover-runner.js +9 -3
- package/dist/cli/failover-runner.js.map +1 -1
- package/dist/cli/loop-executor.d.ts +55 -0
- package/dist/cli/loop-executor.d.ts.map +1 -1
- package/dist/cli/loop-executor.js +285 -19
- package/dist/cli/loop-executor.js.map +1 -1
- package/dist/cli/model.d.ts.map +1 -1
- package/dist/cli/model.js +9 -8
- package/dist/cli/model.js.map +1 -1
- package/dist/cli/models.d.ts +1 -1
- package/dist/cli/models.js +4 -4
- package/dist/cli/nlu.d.ts.map +1 -1
- package/dist/cli/nlu.js +8 -2
- package/dist/cli/nlu.js.map +1 -1
- package/dist/cli/parity.d.ts +85 -0
- package/dist/cli/parity.d.ts.map +1 -0
- package/dist/cli/parity.js +506 -0
- package/dist/cli/parity.js.map +1 -0
- package/dist/cli/plan.d.ts.map +1 -1
- package/dist/cli/plan.js +2 -1
- package/dist/cli/plan.js.map +1 -1
- package/dist/cli/publish.d.ts +22 -0
- package/dist/cli/publish.d.ts.map +1 -1
- package/dist/cli/publish.js +109 -13
- package/dist/cli/publish.js.map +1 -1
- package/dist/cli/retrieval.d.ts.map +1 -1
- package/dist/cli/retrieval.js +5 -4
- package/dist/cli/retrieval.js.map +1 -1
- package/dist/cli/sdk.js +4 -4
- package/dist/cli/sdk.js.map +1 -1
- package/dist/cli/trace.d.ts.map +1 -1
- package/dist/cli/trace.js +2 -1
- package/dist/cli/trace.js.map +1 -1
- package/dist/cli/workflow.js +2 -2
- package/dist/cli/workflow.js.map +1 -1
- package/dist/config/live-credentials.d.ts +57 -0
- package/dist/config/live-credentials.d.ts.map +1 -0
- package/dist/config/live-credentials.js +126 -0
- package/dist/config/live-credentials.js.map +1 -0
- package/dist/config/paths.d.ts +18 -0
- package/dist/config/paths.d.ts.map +1 -1
- package/dist/config/paths.js +25 -0
- package/dist/config/paths.js.map +1 -1
- package/dist/config/types.d.ts +44 -0
- package/dist/config/types.d.ts.map +1 -1
- package/dist/federation/a2a-types.js +1 -1
- package/dist/federation/a2a-types.js.map +1 -1
- package/dist/findings/verdicts.d.ts +241 -0
- package/dist/findings/verdicts.d.ts.map +1 -0
- package/dist/findings/verdicts.js +284 -0
- package/dist/findings/verdicts.js.map +1 -0
- package/dist/gateway/adapters.d.ts +65 -0
- package/dist/gateway/adapters.d.ts.map +1 -1
- package/dist/gateway/adapters.js +216 -10
- package/dist/gateway/adapters.js.map +1 -1
- package/dist/gateway/channel-directory.d.ts +31 -0
- package/dist/gateway/channel-directory.d.ts.map +1 -1
- package/dist/gateway/channel-directory.js +40 -0
- package/dist/gateway/channel-directory.js.map +1 -1
- package/dist/gateway/gateway-log.d.ts +1 -1
- package/dist/gateway/gateway-log.d.ts.map +1 -1
- package/dist/gateway/gateway-log.js.map +1 -1
- package/dist/gateway/hooks.d.ts +87 -19
- package/dist/gateway/hooks.d.ts.map +1 -1
- package/dist/gateway/hooks.js +62 -23
- package/dist/gateway/hooks.js.map +1 -1
- package/dist/gateway/inbound-media.d.ts +147 -0
- package/dist/gateway/inbound-media.d.ts.map +1 -0
- package/dist/gateway/inbound-media.js +317 -0
- package/dist/gateway/inbound-media.js.map +1 -0
- package/dist/gateway/inbox.d.ts +8 -1
- package/dist/gateway/inbox.d.ts.map +1 -1
- package/dist/gateway/inbox.js.map +1 -1
- package/dist/gateway/platform-config.d.ts +14 -0
- package/dist/gateway/platform-config.d.ts.map +1 -1
- package/dist/gateway/platform-config.js +26 -8
- package/dist/gateway/platform-config.js.map +1 -1
- package/dist/gateway/realtime.d.ts +114 -0
- package/dist/gateway/realtime.d.ts.map +1 -0
- package/dist/gateway/realtime.js +402 -0
- package/dist/gateway/realtime.js.map +1 -0
- package/dist/gateway/registry.d.ts +31 -0
- package/dist/gateway/registry.d.ts.map +1 -1
- package/dist/gateway/registry.js +224 -4
- package/dist/gateway/registry.js.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts +11 -1
- package/dist/gateway/whatsapp/baileys-bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/baileys-bridge.js +123 -4
- package/dist/gateway/whatsapp/baileys-bridge.js.map +1 -1
- package/dist/gateway/whatsapp/bridge.d.ts +6 -2
- package/dist/gateway/whatsapp/bridge.d.ts.map +1 -1
- package/dist/gateway/whatsapp/bridge.js.map +1 -1
- package/dist/index.js +8 -0
- package/dist/index.js.map +1 -1
- package/dist/inference/factory.d.ts +14 -0
- package/dist/inference/factory.d.ts.map +1 -1
- package/dist/inference/factory.js +17 -0
- package/dist/inference/factory.js.map +1 -1
- package/dist/inference/groq-adapter.d.ts +2 -0
- package/dist/inference/groq-adapter.d.ts.map +1 -1
- package/dist/inference/groq-adapter.js +16 -6
- package/dist/inference/groq-adapter.js.map +1 -1
- package/dist/inference/model-validator.d.ts +6 -1
- package/dist/inference/model-validator.d.ts.map +1 -1
- package/dist/inference/model-validator.js +7 -2
- package/dist/inference/model-validator.js.map +1 -1
- package/dist/inference/route-resolver.d.ts +116 -0
- package/dist/inference/route-resolver.d.ts.map +1 -0
- package/dist/inference/route-resolver.js +159 -0
- package/dist/inference/route-resolver.js.map +1 -0
- package/dist/inference/tools.d.ts.map +1 -1
- package/dist/inference/tools.js +29 -0
- package/dist/inference/tools.js.map +1 -1
- package/dist/learning/autonomy-policy.d.ts +18 -0
- package/dist/learning/autonomy-policy.d.ts.map +1 -1
- package/dist/learning/autonomy-policy.js +35 -0
- package/dist/learning/autonomy-policy.js.map +1 -1
- package/dist/learning/benchmark.d.ts.map +1 -1
- package/dist/learning/benchmark.js +3 -2
- package/dist/learning/benchmark.js.map +1 -1
- package/dist/learning/continuation.d.ts.map +1 -1
- package/dist/learning/continuation.js +2 -1
- package/dist/learning/continuation.js.map +1 -1
- package/dist/learning/cost-tracker.d.ts.map +1 -1
- package/dist/learning/cost-tracker.js +2 -1
- package/dist/learning/cost-tracker.js.map +1 -1
- package/dist/learning/deferred-task.d.ts.map +1 -1
- package/dist/learning/deferred-task.js +13 -4
- package/dist/learning/deferred-task.js.map +1 -1
- package/dist/learning/eval-framework.d.ts.map +1 -1
- package/dist/learning/eval-framework.js +2 -1
- package/dist/learning/eval-framework.js.map +1 -1
- package/dist/learning/long-form.d.ts.map +1 -1
- package/dist/learning/long-form.js +2 -1
- package/dist/learning/long-form.js.map +1 -1
- package/dist/learning/model-registry.d.ts.map +1 -1
- package/dist/learning/model-registry.js +2 -1
- package/dist/learning/model-registry.js.map +1 -1
- package/dist/learning/reasoning-cache.d.ts.map +1 -1
- package/dist/learning/reasoning-cache.js +2 -1
- package/dist/learning/reasoning-cache.js.map +1 -1
- package/dist/learning/reasoning-trace.d.ts +38 -2
- package/dist/learning/reasoning-trace.d.ts.map +1 -1
- package/dist/learning/reasoning-trace.js +91 -1
- package/dist/learning/reasoning-trace.js.map +1 -1
- package/dist/learning/resilient-call.d.ts.map +1 -1
- package/dist/learning/resilient-call.js +31 -18
- package/dist/learning/resilient-call.js.map +1 -1
- package/dist/learning/retrieval.d.ts.map +1 -1
- package/dist/learning/retrieval.js +2 -1
- package/dist/learning/retrieval.js.map +1 -1
- package/dist/learning/seeded-benchmark.d.ts +160 -0
- package/dist/learning/seeded-benchmark.d.ts.map +1 -0
- package/dist/learning/seeded-benchmark.js +321 -0
- package/dist/learning/seeded-benchmark.js.map +1 -0
- package/dist/learning/seeded-bugs.d.ts +142 -0
- package/dist/learning/seeded-bugs.d.ts.map +1 -0
- package/dist/learning/seeded-bugs.js +535 -0
- package/dist/learning/seeded-bugs.js.map +1 -0
- package/dist/learning/skill-store.d.ts.map +1 -1
- package/dist/learning/skill-store.js +30 -9
- package/dist/learning/skill-store.js.map +1 -1
- package/dist/learning/step-checkpoint.d.ts +127 -0
- package/dist/learning/step-checkpoint.d.ts.map +1 -0
- package/dist/learning/step-checkpoint.js +244 -0
- package/dist/learning/step-checkpoint.js.map +1 -0
- package/dist/mcp/catalog.js +2 -3
- package/dist/mcp/catalog.js.map +1 -1
- package/dist/mcp/manager.d.ts.map +1 -1
- package/dist/mcp/manager.js +5 -3
- package/dist/mcp/manager.js.map +1 -1
- package/dist/nlu/intent-confirm.d.ts +23 -1
- package/dist/nlu/intent-confirm.d.ts.map +1 -1
- package/dist/nlu/intent-confirm.js +85 -2
- package/dist/nlu/intent-confirm.js.map +1 -1
- package/dist/nlu/learnings.d.ts +7 -0
- package/dist/nlu/learnings.d.ts.map +1 -1
- package/dist/nlu/learnings.js +7 -0
- package/dist/nlu/learnings.js.map +1 -1
- package/dist/observability/debug-log.d.ts +250 -0
- package/dist/observability/debug-log.d.ts.map +1 -0
- package/dist/observability/debug-log.js +500 -0
- package/dist/observability/debug-log.js.map +1 -0
- package/dist/observability/event-bus.d.ts.map +1 -1
- package/dist/observability/event-bus.js +4 -1
- package/dist/observability/event-bus.js.map +1 -1
- package/dist/observability/otel.d.ts +278 -0
- package/dist/observability/otel.d.ts.map +1 -0
- package/dist/observability/otel.js +590 -0
- package/dist/observability/otel.js.map +1 -0
- package/dist/parity/drivers.d.ts +99 -0
- package/dist/parity/drivers.d.ts.map +1 -0
- package/dist/parity/drivers.js +1362 -0
- package/dist/parity/drivers.js.map +1 -0
- package/dist/parity/graph.d.ts +73 -0
- package/dist/parity/graph.d.ts.map +1 -0
- package/dist/parity/graph.js +162 -0
- package/dist/parity/graph.js.map +1 -0
- package/dist/parity/matrix.d.ts +105 -0
- package/dist/parity/matrix.d.ts.map +1 -0
- package/dist/parity/matrix.js +352 -0
- package/dist/parity/matrix.js.map +1 -0
- package/dist/parity/observation.d.ts +444 -0
- package/dist/parity/observation.d.ts.map +1 -0
- package/dist/parity/observation.js +333 -0
- package/dist/parity/observation.js.map +1 -0
- package/dist/parity/scenarios.d.ts +229 -0
- package/dist/parity/scenarios.d.ts.map +1 -0
- package/dist/parity/scenarios.js +175 -0
- package/dist/parity/scenarios.js.map +1 -0
- package/dist/parity/surfaces.d.ts +122 -0
- package/dist/parity/surfaces.d.ts.map +1 -0
- package/dist/parity/surfaces.js +190 -0
- package/dist/parity/surfaces.js.map +1 -0
- package/dist/runtime/fault-injection.d.ts +173 -0
- package/dist/runtime/fault-injection.d.ts.map +1 -0
- package/dist/runtime/fault-injection.js +281 -0
- package/dist/runtime/fault-injection.js.map +1 -0
- package/dist/skills/secret-capture.d.ts.map +1 -1
- package/dist/skills/secret-capture.js +9 -4
- package/dist/skills/secret-capture.js.map +1 -1
- package/dist/tools/child-agent-entry.d.ts +23 -0
- package/dist/tools/child-agent-entry.d.ts.map +1 -0
- package/dist/tools/child-agent-entry.js +129 -0
- package/dist/tools/child-agent-entry.js.map +1 -0
- package/dist/tools/child-agent-runtime.d.ts +124 -0
- package/dist/tools/child-agent-runtime.d.ts.map +1 -0
- package/dist/tools/child-agent-runtime.js +704 -0
- package/dist/tools/child-agent-runtime.js.map +1 -0
- package/dist/tools/coding-tools.d.ts.map +1 -1
- package/dist/tools/coding-tools.js +82 -10
- package/dist/tools/coding-tools.js.map +1 -1
- package/dist/tools/credentials-tool.d.ts +24 -0
- package/dist/tools/credentials-tool.d.ts.map +1 -0
- package/dist/tools/credentials-tool.js +125 -0
- package/dist/tools/credentials-tool.js.map +1 -0
- package/dist/tools/delegation-system.d.ts +31 -0
- package/dist/tools/delegation-system.d.ts.map +1 -1
- package/dist/tools/delegation-system.js +70 -9
- package/dist/tools/delegation-system.js.map +1 -1
- package/dist/tools/extract/docx.d.ts +27 -0
- package/dist/tools/extract/docx.d.ts.map +1 -0
- package/dist/tools/extract/docx.js +48 -0
- package/dist/tools/extract/docx.js.map +1 -0
- package/dist/tools/extract/html-text.d.ts +27 -0
- package/dist/tools/extract/html-text.d.ts.map +1 -0
- package/dist/tools/extract/html-text.js +86 -0
- package/dist/tools/extract/html-text.js.map +1 -0
- package/dist/tools/extract/pdf-ocr.d.ts +58 -0
- package/dist/tools/extract/pdf-ocr.d.ts.map +1 -0
- package/dist/tools/extract/pdf-ocr.js +116 -0
- package/dist/tools/extract/pdf-ocr.js.map +1 -0
- package/dist/tools/extract/pdf.d.ts +65 -0
- package/dist/tools/extract/pdf.d.ts.map +1 -0
- package/dist/tools/extract/pdf.js +197 -0
- package/dist/tools/extract/pdf.js.map +1 -0
- package/dist/tools/extract/pptx.d.ts +32 -0
- package/dist/tools/extract/pptx.d.ts.map +1 -0
- package/dist/tools/extract/pptx.js +77 -0
- package/dist/tools/extract/pptx.js.map +1 -0
- package/dist/tools/extract/xlsx.d.ts +47 -0
- package/dist/tools/extract/xlsx.d.ts.map +1 -0
- package/dist/tools/extract/xlsx.js +111 -0
- package/dist/tools/extract/xlsx.js.map +1 -0
- package/dist/tools/finding-tool.d.ts +76 -0
- package/dist/tools/finding-tool.d.ts.map +1 -0
- package/dist/tools/finding-tool.js +125 -0
- package/dist/tools/finding-tool.js.map +1 -0
- package/dist/tools/git-tool.d.ts +25 -7
- package/dist/tools/git-tool.d.ts.map +1 -1
- package/dist/tools/git-tool.js +155 -15
- package/dist/tools/git-tool.js.map +1 -1
- package/dist/tools/loop-project-context.d.ts.map +1 -1
- package/dist/tools/loop-project-context.js +16 -0
- package/dist/tools/loop-project-context.js.map +1 -1
- package/dist/tools/loop-route-feed.d.ts +57 -0
- package/dist/tools/loop-route-feed.d.ts.map +1 -0
- package/dist/tools/loop-route-feed.js +101 -0
- package/dist/tools/loop-route-feed.js.map +1 -0
- package/dist/tools/messaging-tools.d.ts +41 -11
- package/dist/tools/messaging-tools.d.ts.map +1 -1
- package/dist/tools/messaging-tools.js +104 -55
- package/dist/tools/messaging-tools.js.map +1 -1
- package/dist/tools/neutts-synth.d.ts +27 -3
- package/dist/tools/neutts-synth.d.ts.map +1 -1
- package/dist/tools/neutts-synth.js +57 -13
- package/dist/tools/neutts-synth.js.map +1 -1
- package/dist/tools/pipeline-tool.d.ts +43 -1
- package/dist/tools/pipeline-tool.d.ts.map +1 -1
- package/dist/tools/pipeline-tool.js +21 -4
- package/dist/tools/pipeline-tool.js.map +1 -1
- package/dist/tools/publish-tool.d.ts.map +1 -1
- package/dist/tools/publish-tool.js +76 -10
- package/dist/tools/publish-tool.js.map +1 -1
- package/dist/tools/read-extract.d.ts +116 -45
- package/dist/tools/read-extract.d.ts.map +1 -1
- package/dist/tools/read-extract.js +494 -158
- package/dist/tools/read-extract.js.map +1 -1
- package/dist/tools/registry.d.ts +23 -3
- package/dist/tools/registry.d.ts.map +1 -1
- package/dist/tools/registry.js +196 -29
- package/dist/tools/registry.js.map +1 -1
- package/dist/tools/run-terminal.d.ts +9 -0
- package/dist/tools/run-terminal.d.ts.map +1 -1
- package/dist/tools/run-terminal.js +80 -2
- package/dist/tools/run-terminal.js.map +1 -1
- package/dist/tools/subagent-refusal.d.ts +15 -0
- package/dist/tools/subagent-refusal.d.ts.map +1 -0
- package/dist/tools/subagent-refusal.js +18 -0
- package/dist/tools/subagent-refusal.js.map +1 -0
- package/dist/tools/subagent-spawner.d.ts +122 -0
- package/dist/tools/subagent-spawner.d.ts.map +1 -1
- package/dist/tools/subagent-spawner.js +249 -28
- package/dist/tools/subagent-spawner.js.map +1 -1
- package/dist/tools/tool-hooks.d.ts +177 -0
- package/dist/tools/tool-hooks.d.ts.map +1 -0
- package/dist/tools/tool-hooks.js +427 -0
- package/dist/tools/tool-hooks.js.map +1 -0
- package/dist/tools/tool-loop.d.ts +94 -5
- package/dist/tools/tool-loop.d.ts.map +1 -1
- package/dist/tools/tool-loop.js +315 -9
- package/dist/tools/tool-loop.js.map +1 -1
- package/dist/tools/tool-refusal.d.ts +68 -0
- package/dist/tools/tool-refusal.d.ts.map +1 -0
- package/dist/tools/tool-refusal.js +78 -0
- package/dist/tools/tool-refusal.js.map +1 -0
- package/dist/tools/toolsets.d.ts +8 -0
- package/dist/tools/toolsets.d.ts.map +1 -1
- package/dist/tools/toolsets.js +14 -4
- package/dist/tools/toolsets.js.map +1 -1
- package/dist/tools/vision-tools.d.ts +88 -83
- package/dist/tools/vision-tools.d.ts.map +1 -1
- package/dist/tools/vision-tools.js +134 -103
- package/dist/tools/vision-tools.js.map +1 -1
- package/dist/tools/worktree.d.ts +210 -0
- package/dist/tools/worktree.d.ts.map +1 -0
- package/dist/tools/worktree.js +374 -0
- package/dist/tools/worktree.js.map +1 -0
- package/dist/utils/format.d.ts +3 -0
- package/dist/utils/format.d.ts.map +1 -0
- package/dist/utils/format.js +32 -0
- package/dist/utils/format.js.map +1 -0
- package/dist/web-dashboard/attachment-extract.d.ts +64 -0
- package/dist/web-dashboard/attachment-extract.d.ts.map +1 -0
- package/dist/web-dashboard/attachment-extract.js +154 -0
- package/dist/web-dashboard/attachment-extract.js.map +1 -0
- package/dist/web-dashboard/chat-console.d.ts +108 -1
- package/dist/web-dashboard/chat-console.d.ts.map +1 -1
- package/dist/web-dashboard/chat-console.js +36 -0
- package/dist/web-dashboard/chat-console.js.map +1 -1
- package/dist/web-dashboard/hub-data.d.ts +37 -0
- package/dist/web-dashboard/hub-data.d.ts.map +1 -1
- package/dist/web-dashboard/hub-data.js +61 -1
- package/dist/web-dashboard/hub-data.js.map +1 -1
- package/dist/web-dashboard/server.d.ts +14 -0
- package/dist/web-dashboard/server.d.ts.map +1 -1
- package/dist/web-dashboard/server.js +247 -13
- package/dist/web-dashboard/server.js.map +1 -1
- package/dist/web-dashboard/src/types.d.ts +196 -48
- package/dist/web-dashboard/src/types.d.ts.map +1 -1
- package/dist/workflow/registry.js +1 -1
- package/dist/workflow/registry.js.map +1 -1
- package/package.json +22 -6
- package/src/web-dashboard/public/assets/{index-Cyd6tIew.css → index-XJjj2cBX.css} +1 -1
- package/src/web-dashboard/public/assets/index-YWF9FpwQ.js +207 -0
- package/src/web-dashboard/public/assets/index-YWF9FpwQ.js.map +1 -0
- package/src/web-dashboard/public/index.html +2 -2
- package/dist/tools/child-agent-worker.js +0 -212
- package/src/web-dashboard/public/assets/index-CjvoBhlz.js +0 -207
- package/src/web-dashboard/public/assets/index-CjvoBhlz.js.map +0 -1
|
@@ -0,0 +1,1362 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* WS0 (#22) — the drivers that actually run a turn on each surface.
|
|
3
|
+
*
|
|
4
|
+
* WHY THESE LIVE IN `src/` NOW. They used to live in `tests/parity/drivers.ts`
|
|
5
|
+
* and reached the stub through `vi.spyOn` on the real engine — which made the
|
|
6
|
+
* harness unrunnable outside the test runner, and left "prove every surface at
|
|
7
|
+
* par" as something only CI could do. The requirement is a `nuvira parity`
|
|
8
|
+
* command, so the drivers had to become a thing production code can call.
|
|
9
|
+
*
|
|
10
|
+
* The seam is CONFIGURATION, not a mock. Every surface resolves its provider
|
|
11
|
+
* through the SAME shared `resolveProvider` (`src/cli/router.ts`), so a temp
|
|
12
|
+
* `buffconfig.json` naming `groq` with `providers.groq.baseUrl` pointed at a
|
|
13
|
+
* loopback OpenAI-compatible stub makes the whole stack run for real: the real
|
|
14
|
+
* ChatCommand, the real console, the real gateway registry, the real execute
|
|
15
|
+
* command and the real forked child, each with a REAL provider object (the Groq
|
|
16
|
+
* adapter). Only the server on the other end of the socket is a stub — the
|
|
17
|
+
* definition of `transport` depth in `./scenarios.ts`, and the same depth the
|
|
18
|
+
* forked child was already driven at.
|
|
19
|
+
*
|
|
20
|
+
* NO TEST SEAM WAS ADDED TO PRODUCTION CODE. The surfaces already accept what is
|
|
21
|
+
* needed: `answerOnce`/`ChatConsole.answer` take `provider`/`model`, the gateway
|
|
22
|
+
* derives the pair from its own config (`registry.ts:1797`), and the child reads
|
|
23
|
+
* its own `buffconfig.json`. Nothing in `chat.ts`, `chat-console.ts`,
|
|
24
|
+
* `gateway/registry.ts`, `execute.ts` or the spawner changed to make this work.
|
|
25
|
+
*
|
|
26
|
+
* ISOLATION IS THE POINT, AND IT IS PROCESS-LOCAL. A run points
|
|
27
|
+
* `NUVIRA_CONFIG_DIR` and `NUVIRA_MEMORY_DIR` at a throwaway directory for its
|
|
28
|
+
* whole life, so the stub never touches the developer's real profile — no real
|
|
29
|
+
* API key, no real response cache, no real model registry, no real gateway log.
|
|
30
|
+
* That is the same convention `src/config/paths.ts` documents and the same one
|
|
31
|
+
* the test drivers relied on. It cannot leak into a separately-running dashboard
|
|
32
|
+
* or gateway: a `nuvira parity` invocation is its own process.
|
|
33
|
+
*
|
|
34
|
+
* EVERY SURFACE'S OBSERVATION IS READ FROM THAT SURFACE'S OWN REPORT — the
|
|
35
|
+
* engine's return for the CLI, the console's result for the dashboard, the
|
|
36
|
+
* gateway's `inbound.chat` log record for the gateway, the command's result for
|
|
37
|
+
* execute, the child's own progress frames for the subagent. Nothing is
|
|
38
|
+
* reconstructed here, so a surface that stops reporting its status, attribution
|
|
39
|
+
* or tool calls goes red instead of quietly losing the fact.
|
|
40
|
+
*/
|
|
41
|
+
import { createServer } from 'node:http';
|
|
42
|
+
import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
|
|
43
|
+
import { tmpdir } from 'node:os';
|
|
44
|
+
import { join } from 'node:path';
|
|
45
|
+
import { gunzipSync } from 'node:zlib';
|
|
46
|
+
import { getCache } from '../context/cache.js';
|
|
47
|
+
import { readGatewayLog } from '../gateway/gateway-log.js';
|
|
48
|
+
import { debugLogDir, readLatestDebugLog } from '../observability/debug-log.js';
|
|
49
|
+
// WS3 (#25) — the far end of the export the surfaces run, and the two names that
|
|
50
|
+
// define the portable span tree (`src/observability/otel.ts`).
|
|
51
|
+
import { otelEnableVarName, shutdownSpans, TOOL_SPAN_PREFIX, TURN_SPAN_NAME, } from '../observability/otel.js';
|
|
52
|
+
// WS4 (#26) — the hook ENV names the harness declares through, and the phases
|
|
53
|
+
// they exist for. Imported from the module that defines them so a renamed
|
|
54
|
+
// variable cannot leave the harness silently declaring nothing (which would read
|
|
55
|
+
// as "no surface fired a hook", the failure this row is meant to catch).
|
|
56
|
+
import { TOOL_HOOK_ENV, TOOL_HOOK_PHASES } from '../tools/tool-hooks.js';
|
|
57
|
+
import { noDebugLog, noFault, noIsolation, noOtelExport, noResume, noToolHooks, turnStatus, } from './observation.js';
|
|
58
|
+
// WS6 (#28) — the fault protocol. The harness OWNS the provider faults (the stub
|
|
59
|
+
// answers them) and DECLARES the seam faults (`tool`/`ipc`), so both halves of the
|
|
60
|
+
// workstream are driven through the same run rather than two harnesses.
|
|
61
|
+
import { FAULT_ENV, faultMessage, formatFaultPlan, resetFaultInjector } from '../runtime/fault-injection.js';
|
|
62
|
+
// WS5 (#27) — the two env keys the harness declares these capabilities THROUGH,
|
|
63
|
+
// imported from the modules that define them so a renamed variable cannot leave
|
|
64
|
+
// the harness silently declaring nothing (which would read as "no surface
|
|
65
|
+
// isolated its turn", the failure this row is meant to catch).
|
|
66
|
+
import { WORKTREE_ENABLE_ENV } from '../tools/worktree.js';
|
|
67
|
+
import { RESUME_ENABLE_ENV } from '../learning/step-checkpoint.js';
|
|
68
|
+
/**
|
|
69
|
+
* Where every driver stubs. Transport depth is not a compromise here — it is
|
|
70
|
+
* what makes the comparison honest: the provider OBJECT is the real adapter and
|
|
71
|
+
* the turn code above it is untouched. The runner folds `provider` and
|
|
72
|
+
* `transport` into one comparable class (`./scenarios.ts`, rule 1), so these
|
|
73
|
+
* observations compare with anything else that runs the real turn code.
|
|
74
|
+
*/
|
|
75
|
+
export const DRIVER_DEPTH = 'transport';
|
|
76
|
+
/** The provider id every surface is configured with, so the comparison is at-par. */
|
|
77
|
+
export const PARITY_PROVIDER_TYPE = 'groq';
|
|
78
|
+
/** The one model the stub serves. The surfaces are pinned to it, so all five agree. */
|
|
79
|
+
export const PARITY_MODEL = 'parity-stub-model';
|
|
80
|
+
/** The stub's API key. Never a real credential — the stub never validates it. */
|
|
81
|
+
const PARITY_API_KEY = 'parity-stub-key';
|
|
82
|
+
/**
|
|
83
|
+
* The surfaces this module can drive. Declared as data so a caller can assert
|
|
84
|
+
* the driver list covers the registry without paying for a live harness — a
|
|
85
|
+
* surface missing here would otherwise be counted as neither covered nor
|
|
86
|
+
* blocked, the silent hole this harness exists to prevent.
|
|
87
|
+
*/
|
|
88
|
+
export const PARITY_DRIVER_SURFACES = [
|
|
89
|
+
'cli-chat',
|
|
90
|
+
'dashboard-chat',
|
|
91
|
+
'gateway-chat',
|
|
92
|
+
'cli-execute',
|
|
93
|
+
'subagent',
|
|
94
|
+
];
|
|
95
|
+
async function startStub(scenario) {
|
|
96
|
+
let chatCalls = 0;
|
|
97
|
+
let faultsServed = 0;
|
|
98
|
+
// Only a PROVIDER-site fault is the stub's to serve. A `tool`/`ipc` fault is
|
|
99
|
+
// declared to the running agent instead (`withTurnEnvelope`), which is what
|
|
100
|
+
// makes the seam, rather than this stub, the thing under test there.
|
|
101
|
+
const providerFault = scenario.fault?.site === 'provider' ? scenario.fault : null;
|
|
102
|
+
const server = createServer((req, res) => {
|
|
103
|
+
const chunks = [];
|
|
104
|
+
req.on('data', (chunk) => chunks.push(chunk));
|
|
105
|
+
req.on('end', () => {
|
|
106
|
+
const json = (body) => {
|
|
107
|
+
res.writeHead(200, { 'content-type': 'application/json' });
|
|
108
|
+
res.end(JSON.stringify(body));
|
|
109
|
+
};
|
|
110
|
+
if (req.url?.endsWith('/models')) {
|
|
111
|
+
// A reachability probe or a model-list validation against an
|
|
112
|
+
// OpenAI-compatible endpoint asks here. The stub serves the ONE model
|
|
113
|
+
// the surfaces are pinned to, so `resolveWorkingModel` keeps the pin
|
|
114
|
+
// instead of repairing it to some other provider's default.
|
|
115
|
+
// DELIBERATELY NOT FAULTED: a faulted probe would make the surfaces fail
|
|
116
|
+
// in their ROUTING rather than in their turn, which is a different row
|
|
117
|
+
// (and would let a surface pass without ever attempting the call).
|
|
118
|
+
return json({ data: [{ id: PARITY_MODEL, object: 'model' }] });
|
|
119
|
+
}
|
|
120
|
+
if (req.url?.endsWith('/chat/completions')) {
|
|
121
|
+
chatCalls += 1;
|
|
122
|
+
// WS6 (#28) — the DECLARED provider fault, served on the wire so the REAL
|
|
123
|
+
// adapter's error mapping is what runs. `faultMessage` is shared with the
|
|
124
|
+
// seam, so a fault reads the same words wherever it came from, and a body
|
|
125
|
+
// that cannot be parsed is a distinct kind rather than a second flavour of
|
|
126
|
+
// "error" — a response that arrives but says nothing is its own failure.
|
|
127
|
+
if (providerFault && faultsServed < providerFault.times) {
|
|
128
|
+
faultsServed += 1;
|
|
129
|
+
if (providerFault.kind === 'malformed') {
|
|
130
|
+
res.writeHead(200, { 'content-type': 'application/json' });
|
|
131
|
+
res.end('{"choices": [{"message": {"content": '); // truncated on purpose
|
|
132
|
+
return;
|
|
133
|
+
}
|
|
134
|
+
res.writeHead(providerFault.kind === 'unavailable' ? 503 : 500, {
|
|
135
|
+
'content-type': 'application/json',
|
|
136
|
+
});
|
|
137
|
+
res.end(JSON.stringify({
|
|
138
|
+
error: { message: faultMessage(providerFault, 'the model call') },
|
|
139
|
+
}));
|
|
140
|
+
return;
|
|
141
|
+
}
|
|
142
|
+
let body = {};
|
|
143
|
+
try {
|
|
144
|
+
body = JSON.parse(Buffer.concat(chunks).toString('utf8') || '{}');
|
|
145
|
+
}
|
|
146
|
+
catch {
|
|
147
|
+
body = {};
|
|
148
|
+
}
|
|
149
|
+
const toolAlreadyRan = (body.messages ?? []).some((m) => m?.role === 'tool');
|
|
150
|
+
const wantTool = Boolean(scenario.toolCall) && !toolAlreadyRan;
|
|
151
|
+
const content = wantTool ? '' : scenario.answer;
|
|
152
|
+
const toolCalls = wantTool
|
|
153
|
+
? [
|
|
154
|
+
{
|
|
155
|
+
id: 'parity_call_1',
|
|
156
|
+
type: 'function',
|
|
157
|
+
function: {
|
|
158
|
+
name: scenario.toolCall.tool,
|
|
159
|
+
arguments: JSON.stringify(scenario.toolCall.args),
|
|
160
|
+
},
|
|
161
|
+
},
|
|
162
|
+
]
|
|
163
|
+
: undefined;
|
|
164
|
+
// The dashboard console passes `onToken`, so the real Groq adapter takes
|
|
165
|
+
// the STREAMING path (`generateToolsStream` -> SSE). Answering that with
|
|
166
|
+
// a JSON body reads as an empty response and the loop retries until its
|
|
167
|
+
// budget — measured, and the reason this branch exists. Both shapes are
|
|
168
|
+
// served so every surface can be driven the way it really talks.
|
|
169
|
+
if (body.stream) {
|
|
170
|
+
res.writeHead(200, { 'content-type': 'text/event-stream', 'cache-control': 'no-cache' });
|
|
171
|
+
const chunk = (delta, finish) => `data: ${JSON.stringify({
|
|
172
|
+
choices: [{ delta, ...(finish ? { finish_reason: finish } : {}) }],
|
|
173
|
+
})}\n\n`;
|
|
174
|
+
const fragments = [];
|
|
175
|
+
if (wantTool) {
|
|
176
|
+
fragments.push(chunk({
|
|
177
|
+
role: 'assistant',
|
|
178
|
+
content: '',
|
|
179
|
+
tool_calls: toolCalls.map((call, index) => ({ index, ...call })),
|
|
180
|
+
}));
|
|
181
|
+
}
|
|
182
|
+
else {
|
|
183
|
+
fragments.push(chunk({ role: 'assistant', content }));
|
|
184
|
+
}
|
|
185
|
+
fragments.push(chunk({}, wantTool ? 'tool_calls' : 'stop'));
|
|
186
|
+
fragments.push('data: [DONE]\n\n');
|
|
187
|
+
res.end(fragments.join(''));
|
|
188
|
+
return;
|
|
189
|
+
}
|
|
190
|
+
return json({
|
|
191
|
+
choices: [
|
|
192
|
+
{
|
|
193
|
+
message: {
|
|
194
|
+
content,
|
|
195
|
+
...(toolCalls ? { tool_calls: toolCalls } : {}),
|
|
196
|
+
},
|
|
197
|
+
},
|
|
198
|
+
],
|
|
199
|
+
});
|
|
200
|
+
}
|
|
201
|
+
res.writeHead(404, { 'content-type': 'application/json' });
|
|
202
|
+
res.end('{}');
|
|
203
|
+
});
|
|
204
|
+
});
|
|
205
|
+
await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve));
|
|
206
|
+
const { port } = server.address();
|
|
207
|
+
return {
|
|
208
|
+
baseUrl: `http://127.0.0.1:${port}/v1`,
|
|
209
|
+
chatCalls: () => chatCalls,
|
|
210
|
+
faultsServed: () => faultsServed,
|
|
211
|
+
close: () => new Promise((resolve) => server.close(() => resolve())),
|
|
212
|
+
};
|
|
213
|
+
}
|
|
214
|
+
/** Pull the comparable facts out of one OTLP JSON payload, ignoring the rest. */
|
|
215
|
+
function collectSpansInto(payload, into, onServiceName) {
|
|
216
|
+
const resourceSpans = payload?.resourceSpans;
|
|
217
|
+
if (!Array.isArray(resourceSpans))
|
|
218
|
+
return;
|
|
219
|
+
for (const entry of resourceSpans) {
|
|
220
|
+
const attributes = entry?.resource?.attributes;
|
|
221
|
+
if (Array.isArray(attributes)) {
|
|
222
|
+
for (const attribute of attributes) {
|
|
223
|
+
const a = attribute;
|
|
224
|
+
if (a?.key === 'service.name' && typeof a.value?.stringValue === 'string') {
|
|
225
|
+
onServiceName(a.value.stringValue);
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
const scopeSpans = entry?.scopeSpans;
|
|
230
|
+
if (!Array.isArray(scopeSpans))
|
|
231
|
+
continue;
|
|
232
|
+
for (const scope of scopeSpans) {
|
|
233
|
+
const spans = scope?.spans;
|
|
234
|
+
if (!Array.isArray(spans))
|
|
235
|
+
continue;
|
|
236
|
+
for (const span of spans) {
|
|
237
|
+
const s = span;
|
|
238
|
+
if (typeof s?.name !== 'string')
|
|
239
|
+
continue;
|
|
240
|
+
into.push({
|
|
241
|
+
name: s.name,
|
|
242
|
+
traceId: typeof s.traceId === 'string' ? s.traceId : '',
|
|
243
|
+
spanId: typeof s.spanId === 'string' ? s.spanId : '',
|
|
244
|
+
parentSpanId: typeof s.parentSpanId === 'string' ? s.parentSpanId : '',
|
|
245
|
+
});
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
async function startOtlpCollector() {
|
|
251
|
+
const received = [];
|
|
252
|
+
let service = null;
|
|
253
|
+
let requests = 0;
|
|
254
|
+
const server = createServer((req, res) => {
|
|
255
|
+
const chunks = [];
|
|
256
|
+
req.on('data', (chunk) => chunks.push(chunk));
|
|
257
|
+
req.on('end', () => {
|
|
258
|
+
requests += 1;
|
|
259
|
+
try {
|
|
260
|
+
const raw = Buffer.concat(chunks);
|
|
261
|
+
// The SDK gzips when it is told to; decoding it here means a run that
|
|
262
|
+
// enables compression is MEASURED rather than silently read as empty.
|
|
263
|
+
const body = req.headers['content-encoding'] === 'gzip' ? gunzipSync(raw) : raw;
|
|
264
|
+
collectSpansInto(JSON.parse(body.toString('utf8')), received, (name) => {
|
|
265
|
+
service ??= name;
|
|
266
|
+
});
|
|
267
|
+
}
|
|
268
|
+
catch {
|
|
269
|
+
// A body we cannot parse is a span we report as MISSING — never a crash
|
|
270
|
+
// in the harness, and never a silent pass.
|
|
271
|
+
}
|
|
272
|
+
res.writeHead(200, { 'content-type': 'application/json' });
|
|
273
|
+
res.end('{}');
|
|
274
|
+
});
|
|
275
|
+
});
|
|
276
|
+
await new Promise((resolve) => server.listen(0, '127.0.0.1', resolve));
|
|
277
|
+
const { port } = server.address();
|
|
278
|
+
return {
|
|
279
|
+
endpoint: `http://127.0.0.1:${port}/v1/traces`,
|
|
280
|
+
spans: () => [...received],
|
|
281
|
+
serviceName: () => service,
|
|
282
|
+
requests: () => requests,
|
|
283
|
+
close: () => new Promise((resolve) => {
|
|
284
|
+
// The exporter holds keep-alive sockets and `close()` alone waits for
|
|
285
|
+
// them. Tearing them down explicitly is what stops a harness run from
|
|
286
|
+
// hanging on its own collector after the last span arrived.
|
|
287
|
+
server.close(() => resolve());
|
|
288
|
+
server.closeAllConnections?.();
|
|
289
|
+
}),
|
|
290
|
+
};
|
|
291
|
+
}
|
|
292
|
+
/**
|
|
293
|
+
* Reduce the spans a collector received to the compared projection.
|
|
294
|
+
*
|
|
295
|
+
* The collector hands spans over in COMPLETION order (measured), so the whole
|
|
296
|
+
* reduction is order-insensitive: sorted names, sorted edges. A `parentSpanId`
|
|
297
|
+
* matching no span in this collector is a REMOTE parent (a child process that
|
|
298
|
+
* continued a trace begun elsewhere) — it contributes no edge, because an edge to
|
|
299
|
+
* a name we never received cannot compare, and it is recorded on `remoteParent`
|
|
300
|
+
* for the scenario's own assertion instead.
|
|
301
|
+
*/
|
|
302
|
+
function otelObsOf(collector) {
|
|
303
|
+
const received = collector.spans();
|
|
304
|
+
if (received.length === 0)
|
|
305
|
+
return noOtelExport();
|
|
306
|
+
const nameById = new Map(received.filter((s) => s.spanId).map((s) => [s.spanId, s.name]));
|
|
307
|
+
const names = received.map((s) => s.name).sort();
|
|
308
|
+
const edges = new Set();
|
|
309
|
+
for (const span of received) {
|
|
310
|
+
if (!span.parentSpanId)
|
|
311
|
+
continue;
|
|
312
|
+
const parent = nameById.get(span.parentSpanId);
|
|
313
|
+
if (parent)
|
|
314
|
+
edges.add(`${parent} → ${span.name}`);
|
|
315
|
+
}
|
|
316
|
+
const traceIds = [...new Set(received.map((s) => s.traceId).filter(Boolean))];
|
|
317
|
+
const turn = received.find((s) => s.name === TURN_SPAN_NAME) ?? null;
|
|
318
|
+
return {
|
|
319
|
+
exported: true,
|
|
320
|
+
spans: names,
|
|
321
|
+
edges: [...edges].sort(),
|
|
322
|
+
turnSpans: received.filter((s) => s.name === TURN_SPAN_NAME).length,
|
|
323
|
+
toolSpans: names.filter((name) => name.startsWith(TOOL_SPAN_PREFIX)),
|
|
324
|
+
singleTrace: traceIds.length === 1,
|
|
325
|
+
serviceName: collector.serviceName(),
|
|
326
|
+
traceId: traceIds.length === 1 ? traceIds[0] : (turn?.traceId ?? null),
|
|
327
|
+
remoteParent: turn && turn.parentSpanId && !nameById.has(turn.parentSpanId) ? turn.parentSpanId : null,
|
|
328
|
+
};
|
|
329
|
+
}
|
|
330
|
+
/**
|
|
331
|
+
* Run one surface's turn with a fresh collector in front of the export path.
|
|
332
|
+
*
|
|
333
|
+
* Returns the turn's own value AND the projection read from the collector after
|
|
334
|
+
* it finished, so a driver wraps exactly the call that talks to the model and
|
|
335
|
+
* nothing else. The endpoint is set for the duration of that call and restored
|
|
336
|
+
* after, because the SDK reads it when it BUILDS the provider — which is why the
|
|
337
|
+
* harness resets the provider before each surface (`shutdownSpans` in the driver
|
|
338
|
+
* wrapper) instead of trusting one endpoint to serve them all.
|
|
339
|
+
*/
|
|
340
|
+
async function withOtlpCollector(body) {
|
|
341
|
+
const collector = await startOtlpCollector();
|
|
342
|
+
const previousEndpoint = process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT;
|
|
343
|
+
process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = collector.endpoint;
|
|
344
|
+
try {
|
|
345
|
+
const value = await body();
|
|
346
|
+
return { value, otel: otelObsOf(collector) };
|
|
347
|
+
}
|
|
348
|
+
finally {
|
|
349
|
+
if (previousEndpoint === undefined)
|
|
350
|
+
delete process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT;
|
|
351
|
+
else
|
|
352
|
+
process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = previousEndpoint;
|
|
353
|
+
await collector.close();
|
|
354
|
+
}
|
|
355
|
+
}
|
|
356
|
+
// ─── The operator's tool hooks ──────────────────────────────────────────────
|
|
357
|
+
/** Where a declared hook appends the invocations it received (harness-owned). */
|
|
358
|
+
export const TOOL_HOOK_LOG_VAR = 'NUVIRA_TOOL_HOOK_LOG';
|
|
359
|
+
/** Which tools a declared `before` hook vetoes (harness-owned). */
|
|
360
|
+
export const TOOL_HOOK_DENY_VAR = 'NUVIRA_TOOL_HOOK_DENY';
|
|
361
|
+
/**
|
|
362
|
+
* WS4 (#26) — the hook script the harness DECLARES, exactly as written.
|
|
363
|
+
*
|
|
364
|
+
* It is a real operator hook: a command that reads the call as JSON on stdin and
|
|
365
|
+
* may veto it on stdout. It reads two variables the harness sets — where to
|
|
366
|
+
* append the invocation it received, and which tools this scenario's `before`
|
|
367
|
+
* hook denies — so ONE script serves every scenario and every phase, and the log
|
|
368
|
+
* it writes is the record this row is compared on.
|
|
369
|
+
*
|
|
370
|
+
* Why a script at all, rather than a spy on the hook seam: the claim is "an
|
|
371
|
+
* operator's declared command runs", and only a real process proves the path an
|
|
372
|
+
* operator would actually take — the spawn, the stdin pipe, the JSON contract, the
|
|
373
|
+
* exit code. A spy would only prove the surface called its own helper.
|
|
374
|
+
*/
|
|
375
|
+
const TOOL_HOOK_SCRIPT = `#!/usr/bin/env node
|
|
376
|
+
// WS4 (#26) — the operator hook the parity harness declares. See
|
|
377
|
+
// src/parity/drivers.ts for the contract and why it is a real process.
|
|
378
|
+
import { appendFileSync } from 'node:fs';
|
|
379
|
+
|
|
380
|
+
let raw = '';
|
|
381
|
+
process.stdin.setEncoding('utf8');
|
|
382
|
+
for await (const chunk of process.stdin) raw += chunk;
|
|
383
|
+
|
|
384
|
+
let payload = {};
|
|
385
|
+
try {
|
|
386
|
+
payload = JSON.parse(raw);
|
|
387
|
+
} catch {
|
|
388
|
+
// Not JSON means this was never handed a real payload: exit non-zero so the
|
|
389
|
+
// surface reports a problem instead of reading silence as a decision.
|
|
390
|
+
process.stderr.write('parity hook: stdin was not the documented JSON payload');
|
|
391
|
+
process.exit(2);
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
const denyList = (process.env.${TOOL_HOOK_DENY_VAR} ?? '')
|
|
395
|
+
.split(',')
|
|
396
|
+
.map((name) => name.trim())
|
|
397
|
+
.filter(Boolean);
|
|
398
|
+
const denied = payload.phase === 'before' && denyList.includes(payload.tool);
|
|
399
|
+
|
|
400
|
+
const log = process.env.${TOOL_HOOK_LOG_VAR};
|
|
401
|
+
if (log) {
|
|
402
|
+
appendFileSync(
|
|
403
|
+
log,
|
|
404
|
+
JSON.stringify({
|
|
405
|
+
phase: payload.phase,
|
|
406
|
+
tool: payload.tool,
|
|
407
|
+
surface: payload.surface ?? null,
|
|
408
|
+
ok: typeof payload.ok === 'boolean' ? payload.ok : null,
|
|
409
|
+
decision: denied ? 'deny' : null,
|
|
410
|
+
hook: payload.hook,
|
|
411
|
+
}) + '\\n',
|
|
412
|
+
);
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
if (denied) {
|
|
416
|
+
process.stdout.write(
|
|
417
|
+
JSON.stringify({ decision: 'deny', reason: 'parity: ' + payload.tool + ' is not allowed to run' }),
|
|
418
|
+
);
|
|
419
|
+
}
|
|
420
|
+
`;
|
|
421
|
+
/**
|
|
422
|
+
* Reduce the hook's log AND the surface's own tool lifecycle to the projection.
|
|
423
|
+
*
|
|
424
|
+
* Two witnesses, deliberately: the log says what the hook was asked and how it
|
|
425
|
+
* answered, and `observation.toolCalls` says what the SURFACE did with that
|
|
426
|
+
* answer. `vetoReported` needs both to agree; `vetoLeaked` needs only the surface
|
|
427
|
+
* to show a denied tool succeeding. See `ToolHooksObs` for why neither side alone
|
|
428
|
+
* is trusted.
|
|
429
|
+
*/
|
|
430
|
+
function toolHooksObsOf(logPath, observation) {
|
|
431
|
+
let lines;
|
|
432
|
+
try {
|
|
433
|
+
lines = readFileSync(logPath, 'utf8').split('\n').filter((line) => line.trim() !== '');
|
|
434
|
+
}
|
|
435
|
+
catch {
|
|
436
|
+
// No log means the hook never ran (or never wrote) — the honest value, which
|
|
437
|
+
// a scenario that declared hooks reads as a failure rather than a neutral.
|
|
438
|
+
return noToolHooks();
|
|
439
|
+
}
|
|
440
|
+
const invocations = new Set();
|
|
441
|
+
const denied = new Set();
|
|
442
|
+
const surfacesSeen = new Set();
|
|
443
|
+
for (const line of lines) {
|
|
444
|
+
let entry;
|
|
445
|
+
try {
|
|
446
|
+
entry = JSON.parse(line);
|
|
447
|
+
}
|
|
448
|
+
catch {
|
|
449
|
+
continue;
|
|
450
|
+
}
|
|
451
|
+
const phase = typeof entry.phase === 'string' ? entry.phase : 'unknown';
|
|
452
|
+
const tool = typeof entry.tool === 'string' ? entry.tool : 'unknown';
|
|
453
|
+
invocations.add(`${phase}:${tool}`);
|
|
454
|
+
if (typeof entry.surface === 'string' && entry.surface !== '')
|
|
455
|
+
surfacesSeen.add(entry.surface);
|
|
456
|
+
if (phase === 'before' && entry.decision === 'deny')
|
|
457
|
+
denied.add(tool);
|
|
458
|
+
}
|
|
459
|
+
const deniedTools = [...denied].sort();
|
|
460
|
+
// The surface's OWN outcomes for the denied tools: `true` = reported success
|
|
461
|
+
// (so the call ran — a leak), `false` = reported as a failed call.
|
|
462
|
+
const outcomes = deniedTools.map((tool) => observation.toolCalls.filter((call) => call.tool === tool).map((call) => call.ok === true));
|
|
463
|
+
return {
|
|
464
|
+
invocations: [...invocations].sort(),
|
|
465
|
+
denied: deniedTools,
|
|
466
|
+
vetoReported: deniedTools.length > 0 && outcomes.every((perTool) => perTool.some((ok) => ok === false)),
|
|
467
|
+
vetoLeaked: outcomes.some((perTool) => perTool.some((ok) => ok === true)),
|
|
468
|
+
surfacesSeen: [...surfacesSeen].sort(),
|
|
469
|
+
};
|
|
470
|
+
}
|
|
471
|
+
/**
|
|
472
|
+
* Run one surface's turn with this scenario's hooks DECLARED, and read back what
|
|
473
|
+
* the hook received.
|
|
474
|
+
*
|
|
475
|
+
* The declarations are environment variables for the duration of the call and are
|
|
476
|
+
* restored afterwards, for the same reason the OTLP endpoint is: the surfaces
|
|
477
|
+
* read them at call time, so a scenario that declared a hook must not leave it
|
|
478
|
+
* declared for the next one — a veto that leaked into the next scenario would
|
|
479
|
+
* look like that scenario's own behaviour.
|
|
480
|
+
*
|
|
481
|
+
* The log file is per SURFACE as well as per scenario, so the child's invocations
|
|
482
|
+
* (written from its own process, which inherits the path) cannot be attributed to
|
|
483
|
+
* an in-process surface that ran before it.
|
|
484
|
+
*/
|
|
485
|
+
async function withToolHooks(ws, surface, scenario, body) {
|
|
486
|
+
const declared = scenario.hooks;
|
|
487
|
+
const logPath = join(ws.root, `tool-hook-${scenario.id}-${surface}.jsonl`);
|
|
488
|
+
rmSync(logPath, { force: true });
|
|
489
|
+
const previous = new Map();
|
|
490
|
+
const set = (name, value) => {
|
|
491
|
+
previous.set(name, process.env[name]);
|
|
492
|
+
if (value === undefined)
|
|
493
|
+
delete process.env[name];
|
|
494
|
+
else
|
|
495
|
+
process.env[name] = value;
|
|
496
|
+
};
|
|
497
|
+
const deny = declared?.deny?.filter((tool) => tool.trim() !== '') ?? [];
|
|
498
|
+
for (const phase of TOOL_HOOK_PHASES) {
|
|
499
|
+
set(TOOL_HOOK_ENV[phase], declared?.phases.includes(phase) ? `node ${ws.hookScript}` : undefined);
|
|
500
|
+
}
|
|
501
|
+
set(TOOL_HOOK_LOG_VAR, declared ? logPath : undefined);
|
|
502
|
+
set(TOOL_HOOK_DENY_VAR, deny.length > 0 ? deny.join(',') : undefined);
|
|
503
|
+
try {
|
|
504
|
+
const observation = await body();
|
|
505
|
+
return { ...observation, hooks: toolHooksObsOf(logPath, observation) };
|
|
506
|
+
}
|
|
507
|
+
finally {
|
|
508
|
+
for (const [name, value] of previous) {
|
|
509
|
+
if (value === undefined)
|
|
510
|
+
delete process.env[name];
|
|
511
|
+
else
|
|
512
|
+
process.env[name] = value;
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
// ─── The turn envelope (WS5: isolation and resume) ──────────────────────────
|
|
517
|
+
/**
|
|
518
|
+
* Run one surface's turn with this scenario's isolation and resume DECLARED, and
|
|
519
|
+
* reduce the surface's own reports to what is compared.
|
|
520
|
+
*
|
|
521
|
+
* THE RESUME PROBE IS A PAIR OF TURNS, and it has to be: "a resumed run reuses
|
|
522
|
+
* unchanged steps instead of re-paying for every model call" is a statement about
|
|
523
|
+
* two runs — one that writes the record and one that reads it — so a single turn
|
|
524
|
+
* cannot produce the fact. The FIRST turn is the one everything else about the
|
|
525
|
+
* scenario is read from (a fully replayed turn reaches no model at all, and the
|
|
526
|
+
* runner refuses to compare such a turn); the SECOND contributes only its resume
|
|
527
|
+
* fields.
|
|
528
|
+
*
|
|
529
|
+
* THE RESPONSE CACHE IS CLEARED BEFORE EACH TURN, and skipping that would make
|
|
530
|
+
* this row lie in the most convenient direction: the second turn sends the same
|
|
531
|
+
* message as the first, so a cache hit would answer it without reaching the loop
|
|
532
|
+
* at all — no record read, no step replayed, and a `resuming: false` that a
|
|
533
|
+
* harness comparing only model counts would read as agreement. The driver's own
|
|
534
|
+
* turns clear it too; this is the second, independent guard.
|
|
535
|
+
*
|
|
536
|
+
* Both declarations are the ENVIRONMENT (`NUVIRA_ISOLATE` / `NUVIRA_RESUME`) and
|
|
537
|
+
* are restored afterwards, exactly like the OTLP endpoint and the hooks: the
|
|
538
|
+
* surfaces read them at call time, so a declaration that leaked into the next
|
|
539
|
+
* scenario would look like that scenario's own behaviour. The environment is also
|
|
540
|
+
* what makes ONE declaration cover all five surfaces — the dashboard server, the
|
|
541
|
+
* gateway and the forked child have no flags to carry.
|
|
542
|
+
*/
|
|
543
|
+
async function withTurnEnvelope(scenario, surface, body) {
|
|
544
|
+
const askedIsolation = scenario.isolation === true;
|
|
545
|
+
const askedResume = scenario.resume === true;
|
|
546
|
+
// WS6 (#28) — a `tool`/`ipc` fault is DECLARED to the running agent (the seam),
|
|
547
|
+
// while a `provider` fault is served by the stub (so the real adapter's error
|
|
548
|
+
// mapping runs and the model call still happens). Both are set explicitly,
|
|
549
|
+
// including the `undefined` case: a declaration inherited from the developer's
|
|
550
|
+
// shell would make the scenarios that DO NOT ask for a fault inject one anyway.
|
|
551
|
+
const seamFault = scenario.fault && scenario.fault.site !== 'provider' ? scenario.fault : null;
|
|
552
|
+
const previousIsolation = process.env[WORKTREE_ENABLE_ENV];
|
|
553
|
+
const previousResume = process.env[RESUME_ENABLE_ENV];
|
|
554
|
+
const previousFault = process.env[FAULT_ENV];
|
|
555
|
+
const set = (name, value) => {
|
|
556
|
+
if (value === undefined)
|
|
557
|
+
delete process.env[name];
|
|
558
|
+
else
|
|
559
|
+
process.env[name] = value;
|
|
560
|
+
};
|
|
561
|
+
/**
|
|
562
|
+
* The record this surface's probe uses — NAMED, and namespaced by surface.
|
|
563
|
+
*
|
|
564
|
+
* The auto id is `checkpointIdFor(goal, cwd)`, which is the right default for a
|
|
565
|
+
* human (`--resume` means "the last run of this ask, here") and exactly wrong for
|
|
566
|
+
* this harness: every in-process surface runs the SAME ask in the SAME directory,
|
|
567
|
+
* so they would all read ONE record and a surface could "replay" a tree another
|
|
568
|
+
* surface recorded — measured, and it made two surfaces pass for a reason that had
|
|
569
|
+
* nothing to do with them. A per-surface id keeps each probe's record its own,
|
|
570
|
+
* which is also the path an operator uses to resume a named run.
|
|
571
|
+
*/
|
|
572
|
+
const resumeId = `parity-${scenario.id}-${surface}`;
|
|
573
|
+
try {
|
|
574
|
+
set(WORKTREE_ENABLE_ENV, askedIsolation ? '1' : undefined);
|
|
575
|
+
// BOTH turns are asked to resume, and that is not a formality: the FIRST one is
|
|
576
|
+
// what WRITES the record the second replays. A probe whose first turn ran
|
|
577
|
+
// without a ledger would compare a resumed turn against an empty record —
|
|
578
|
+
// replayed 0, model calls unchanged — and report that as the capability working.
|
|
579
|
+
set(RESUME_ENABLE_ENV, askedResume ? resumeId : undefined);
|
|
580
|
+
set(FAULT_ENV, seamFault ? formatFaultPlan(seamFault) : undefined);
|
|
581
|
+
// The injector is cached per declaration, and this declaration is fresh for
|
|
582
|
+
// this surface — so it starts with a full allowance either way.
|
|
583
|
+
resetFaultInjector();
|
|
584
|
+
await clearResponseCache();
|
|
585
|
+
const first = await body();
|
|
586
|
+
if (!askedResume)
|
|
587
|
+
return first;
|
|
588
|
+
await clearResponseCache();
|
|
589
|
+
const second = await body();
|
|
590
|
+
// Only the resume fields come from the second turn: everything else about the
|
|
591
|
+
// scenario is read from the first one, which is the turn that reached a model.
|
|
592
|
+
return { ...first, resume: second.resume };
|
|
593
|
+
}
|
|
594
|
+
finally {
|
|
595
|
+
set(WORKTREE_ENABLE_ENV, previousIsolation);
|
|
596
|
+
set(RESUME_ENABLE_ENV, previousResume);
|
|
597
|
+
set(FAULT_ENV, previousFault);
|
|
598
|
+
resetFaultInjector();
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
/**
|
|
602
|
+
* The response cache is SHARED, ON DISK, AND ON BY DEFAULT for `answerOnce`
|
|
603
|
+
* (`src/cli/chat.ts:690`), keyed by `provider:model:prompt`. Every surface of a
|
|
604
|
+
* run sends the SAME scenario message, so without clearing between surfaces the
|
|
605
|
+
* second one would be served from the first one's entry — same answer, no model
|
|
606
|
+
* call, no tool lifecycle, and a flawless "agreement" between two replays.
|
|
607
|
+
*
|
|
608
|
+
* Isolating `NUVIRA_MEMORY_DIR` already makes the cache file fresh per run, but
|
|
609
|
+
* within a run the entries still collide, so this clears before each surface.
|
|
610
|
+
* The runner's `unreached-model` refusal (`./scenarios.ts`, rule 4) is the
|
|
611
|
+
* second, independent guard: a zero from a replay is refused, not compared.
|
|
612
|
+
*/
|
|
613
|
+
async function clearResponseCache() {
|
|
614
|
+
try {
|
|
615
|
+
await getCache().clear();
|
|
616
|
+
}
|
|
617
|
+
catch {
|
|
618
|
+
// The cache may never break a turn — and the `modelCalls` guard catches a
|
|
619
|
+
// stale entry anyway.
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
function createWorkspace() {
|
|
623
|
+
const root = mkdtempSync(join(tmpdir(), 'buff-parity-'));
|
|
624
|
+
const configDir = join(root, 'config');
|
|
625
|
+
const memoryDir = join(root, 'memory');
|
|
626
|
+
mkdirSync(configDir, { recursive: true });
|
|
627
|
+
mkdirSync(memoryDir, { recursive: true });
|
|
628
|
+
// WS4 — the hook command the harness declares. Written once per run, so every
|
|
629
|
+
// scenario's hooks are the SAME program and a difference between scenarios can
|
|
630
|
+
// only come from the declaration, not from the script.
|
|
631
|
+
const hookScript = join(root, 'tool-hook.mjs');
|
|
632
|
+
writeFileSync(hookScript, TOOL_HOOK_SCRIPT);
|
|
633
|
+
const workspace = {
|
|
634
|
+
root,
|
|
635
|
+
configDir,
|
|
636
|
+
memoryDir,
|
|
637
|
+
hookScript,
|
|
638
|
+
useStub(baseUrl) {
|
|
639
|
+
// `buffconfig.json` is the file `ConfigManager` reads (`config/manager.ts`
|
|
640
|
+
// -> `config/paths.ts`). The provider object the surfaces build from it is
|
|
641
|
+
// the REAL Groq adapter; `baseUrl` is the override it honours.
|
|
642
|
+
writeFileSync(join(configDir, 'buffconfig.json'), JSON.stringify({
|
|
643
|
+
defaultProvider: PARITY_PROVIDER_TYPE,
|
|
644
|
+
providers: {
|
|
645
|
+
[PARITY_PROVIDER_TYPE]: { apiKey: PARITY_API_KEY, model: PARITY_MODEL, baseUrl },
|
|
646
|
+
},
|
|
647
|
+
}, null, 2));
|
|
648
|
+
},
|
|
649
|
+
};
|
|
650
|
+
return workspace;
|
|
651
|
+
}
|
|
652
|
+
/**
|
|
653
|
+
* Reduce a surface's own isolation report to the compared projection.
|
|
654
|
+
*
|
|
655
|
+
* Takes the shape both the in-process surfaces and the subagent manager report
|
|
656
|
+
* (a base commit, a file list and whether the directory was removed) rather than
|
|
657
|
+
* one of their concrete types, because the two are produced by different code on
|
|
658
|
+
* purpose: the in-process turn makes its own worktree, the parent makes the
|
|
659
|
+
* child's. What must agree is the RESULT, and that is what this reduces.
|
|
660
|
+
*/
|
|
661
|
+
function isolationObsOf(asked, report) {
|
|
662
|
+
if (!report)
|
|
663
|
+
return { ...noIsolation(), asked };
|
|
664
|
+
return {
|
|
665
|
+
asked,
|
|
666
|
+
isolated: true,
|
|
667
|
+
// Sorted, so the comparison is over the SET of files the run changed: a diff
|
|
668
|
+
// is measured from `git diff`, whose order is the repository's, not the
|
|
669
|
+
// surface's.
|
|
670
|
+
files: [...report.diff.files].sort(),
|
|
671
|
+
removed: report.removed,
|
|
672
|
+
base: report.base,
|
|
673
|
+
};
|
|
674
|
+
}
|
|
675
|
+
/**
|
|
676
|
+
* Reduce a surface's own resume report to the compared projection.
|
|
677
|
+
*
|
|
678
|
+
* `asked` with no report is the honest description of a surface that ignored the
|
|
679
|
+
* request: `resuming: false` against every other surface's `true`, which
|
|
680
|
+
* `compare` reports as a difference rather than as agreement about nothing.
|
|
681
|
+
*/
|
|
682
|
+
function resumeObsOf(asked, report) {
|
|
683
|
+
if (!report)
|
|
684
|
+
return { ...noResume(), asked };
|
|
685
|
+
return {
|
|
686
|
+
asked,
|
|
687
|
+
resuming: true,
|
|
688
|
+
replayed: report.replayed,
|
|
689
|
+
modelCalls: report.modelCalls,
|
|
690
|
+
saved: report.saved,
|
|
691
|
+
};
|
|
692
|
+
}
|
|
693
|
+
/**
|
|
694
|
+
* Read a surface's OWN session debug log and reduce its header to the compared
|
|
695
|
+
* facts — WS2.
|
|
696
|
+
*
|
|
697
|
+
* Read from DISK rather than from a return value, for the same reason the
|
|
698
|
+
* gateway driver reads `inbound.chat` from the gateway log: the capability is
|
|
699
|
+
* "a file you can attach to a bug report", so the artifact itself is the
|
|
700
|
+
* evidence. `written: false` when the surface produced nothing, which the
|
|
701
|
+
* harness refuses to read as agreement (the run turns logging on for every
|
|
702
|
+
* surface).
|
|
703
|
+
*/
|
|
704
|
+
function debugLogOf(surface, dir = debugLogDir()) {
|
|
705
|
+
const found = readLatestDebugLog(surface, dir);
|
|
706
|
+
if (!found)
|
|
707
|
+
return noDebugLog();
|
|
708
|
+
return {
|
|
709
|
+
written: true,
|
|
710
|
+
provider: found.header.provider,
|
|
711
|
+
model: found.header.model,
|
|
712
|
+
transport: found.header.transport,
|
|
713
|
+
};
|
|
714
|
+
}
|
|
715
|
+
/**
|
|
716
|
+
* Reduce a surface's own answer to the shared observation.
|
|
717
|
+
*
|
|
718
|
+
* MEASURED, and found by this harness on its first real run: the surfaces do NOT
|
|
719
|
+
* share a result contract. `ChatCommand.answerOnce` reports `generationFailed`
|
|
720
|
+
* and no `ok`; the console SYNTHESISES `ok`. Mapping each surface's OWN contract
|
|
721
|
+
* rather than inventing a common one keeps that seam visible instead of hiding
|
|
722
|
+
* it — which is what a parity harness is for.
|
|
723
|
+
*/
|
|
724
|
+
/**
|
|
725
|
+
* WS6 (#28) — whether this surface's OWN turn shows the fault it was given.
|
|
726
|
+
*
|
|
727
|
+
* Derived rather than counted, because the counter cannot cross the fork (see
|
|
728
|
+
* `FaultObs`). The rule is the fault's own contract:
|
|
729
|
+
*
|
|
730
|
+
* - `tool` — the named call (or any call, when none is named) is reported
|
|
731
|
+
* FAILED. A surface that swallowed the injected `Error:` and
|
|
732
|
+
* reported the call as ok reads as `took: false`.
|
|
733
|
+
* - `provider` / `ipc` — the turn did not COMPLETE. A surface that produced an
|
|
734
|
+
* answer anyway reads as `took: false`, which is precisely the
|
|
735
|
+
* false-success shape this workstream exists to catch.
|
|
736
|
+
*/
|
|
737
|
+
function faultObsOf(scenario, toolCalls, status) {
|
|
738
|
+
const plan = scenario.fault;
|
|
739
|
+
if (!plan)
|
|
740
|
+
return noFault();
|
|
741
|
+
const took = plan.site === 'tool'
|
|
742
|
+
? toolCalls.some((call) => (plan.match === undefined || call.tool === plan.match) && call.ok === false)
|
|
743
|
+
: status !== 'completed';
|
|
744
|
+
return { asked: true, site: plan.site, kind: plan.kind, took };
|
|
745
|
+
}
|
|
746
|
+
function toObservation(surface, scenario, answer, toolCalls, modelCalls) {
|
|
747
|
+
// WS6 (#28) — `generationFailed` outranks `ok`, through the ONE helper every
|
|
748
|
+
// surface's read goes through (see `turnStatus`). A served-but-failed turn is a
|
|
749
|
+
// failure; reading the request-level `ok` first is what let the provider-fault
|
|
750
|
+
// row catch two surfaces calling an empty generation a completion.
|
|
751
|
+
const status = turnStatus(answer);
|
|
752
|
+
const succeeded = status === 'completed';
|
|
753
|
+
return {
|
|
754
|
+
surface,
|
|
755
|
+
engine: 'loop',
|
|
756
|
+
status,
|
|
757
|
+
// Recorded, never inferred: the runner refuses a zero (rule 4 in
|
|
758
|
+
// ./scenarios.ts) instead of reading a cache replay as agreement.
|
|
759
|
+
modelCalls,
|
|
760
|
+
...(answer.provider ? { provider: answer.provider } : {}),
|
|
761
|
+
...(answer.model ? { model: answer.model } : {}),
|
|
762
|
+
...(answer.transport ? { transport: answer.transport } : {}),
|
|
763
|
+
toolCalls,
|
|
764
|
+
// WS1 — recorded findings, in order. `[]` when the surface reported none.
|
|
765
|
+
findings: answer.findings ?? [],
|
|
766
|
+
// WS2 — the session debug log's header. `written: false` when the surface
|
|
767
|
+
// produced none, which the harness (logging ON) reads as a failure.
|
|
768
|
+
debugLog: answer.debugLog ?? noDebugLog(),
|
|
769
|
+
// WS3 — the span tree the collector received. `exported: false` when nothing
|
|
770
|
+
// arrived, which the harness (export ON) reads as a failure.
|
|
771
|
+
otel: answer.otel ?? noOtelExport(),
|
|
772
|
+
// WS4 — replaced by the driver wrapper with what the operator's declared hook
|
|
773
|
+
// actually received (`withToolHooks`): the log is written by the HOOK, and
|
|
774
|
+
// only the wrapper knows which file this surface's run was pointed at.
|
|
775
|
+
hooks: noToolHooks(),
|
|
776
|
+
// WS5 — the surface's OWN report of the isolation it ran with, and of what its
|
|
777
|
+
// resume replayed. `asked` comes from the scenario, which is what makes
|
|
778
|
+
// "asked for it and did not do it" a difference rather than a tautology. The
|
|
779
|
+
// resume field is REPLACED by `withTurnEnvelope` for the resumed scenario,
|
|
780
|
+
// whose second turn is the one that can answer it.
|
|
781
|
+
isolation: isolationObsOf(scenario.isolation === true, answer.worktree),
|
|
782
|
+
resume: resumeObsOf(scenario.resume === true, answer.resume),
|
|
783
|
+
// WS6 (#28) — the declared fault, and whether THIS surface's turn shows it.
|
|
784
|
+
fault: faultObsOf(scenario, toolCalls, status),
|
|
785
|
+
...(typeof answer.content === 'string' ? { answer: answer.content } : {}),
|
|
786
|
+
...(succeeded
|
|
787
|
+
? {}
|
|
788
|
+
: { errorCode: answer.generationFailed ? 'generation_failed' : 'turn_failed' }),
|
|
789
|
+
noise: { at: Date.now() },
|
|
790
|
+
};
|
|
791
|
+
}
|
|
792
|
+
/**
|
|
793
|
+
* Collect one surface's tool-call lifecycle: the `called` phase only, which is
|
|
794
|
+
* the one that carries an outcome (`ok`). A call that never reached `called` did
|
|
795
|
+
* not happen, and an outcome that is absent is recorded as absent, not guessed.
|
|
796
|
+
*/
|
|
797
|
+
function collectCalled(target, phase, info) {
|
|
798
|
+
if (phase !== 'called')
|
|
799
|
+
return;
|
|
800
|
+
target.push({ tool: info.tool, ...(typeof info.ok === 'boolean' ? { ok: info.ok } : {}) });
|
|
801
|
+
}
|
|
802
|
+
// ─── The drivers ────────────────────────────────────────────────────────────
|
|
803
|
+
/** CLI chat: the shared engine, called directly, pinned to the stub's model. */
|
|
804
|
+
async function runViaChatOnce(ws, scenario) {
|
|
805
|
+
const stub = await startStub(scenario);
|
|
806
|
+
try {
|
|
807
|
+
ws.useStub(stub.baseUrl);
|
|
808
|
+
await clearResponseCache();
|
|
809
|
+
const { ChatCommand } = await import('../cli/chat.js');
|
|
810
|
+
const command = new ChatCommand();
|
|
811
|
+
const toolCalls = [];
|
|
812
|
+
const { value: answer, otel } = await withOtlpCollector(() => command.answerOnce(scenario.message, {
|
|
813
|
+
provider: PARITY_PROVIDER_TYPE,
|
|
814
|
+
model: PARITY_MODEL,
|
|
815
|
+
onToolCall: (phase, info) => collectCalled(toolCalls, phase, info),
|
|
816
|
+
}));
|
|
817
|
+
return toObservation('cli-chat', scenario, { ...answer, debugLog: debugLogOf('cli-chat'), otel }, toolCalls, stub.chatCalls());
|
|
818
|
+
}
|
|
819
|
+
finally {
|
|
820
|
+
await stub.close();
|
|
821
|
+
}
|
|
822
|
+
}
|
|
823
|
+
/**
|
|
824
|
+
* Dashboard chat: the real console, with NO injected engine and no injected
|
|
825
|
+
* provider, so `ensureEngine()` lazily loads the real `ChatCommand` and the
|
|
826
|
+
* console's own turn plumbing (session record, busy guard, turn telemetry,
|
|
827
|
+
* progress emission) runs rather than being bypassed. The provider/model are
|
|
828
|
+
* pinned through the console's public options, which is exactly the surface
|
|
829
|
+
* handing the pin to its engine.
|
|
830
|
+
*/
|
|
831
|
+
async function runViaConsole(ws, scenario) {
|
|
832
|
+
const stub = await startStub(scenario);
|
|
833
|
+
try {
|
|
834
|
+
ws.useStub(stub.baseUrl);
|
|
835
|
+
await clearResponseCache();
|
|
836
|
+
const { ChatConsole } = await import('../web-dashboard/chat-console.js');
|
|
837
|
+
const console_ = new ChatConsole({});
|
|
838
|
+
const toolCalls = [];
|
|
839
|
+
// The console's own subscription seam — the same one the GUI uses — rather
|
|
840
|
+
// than reaching into the engine, so this observes what a dashboard user sees.
|
|
841
|
+
const off = console_.onEvent((_sessionId, event) => {
|
|
842
|
+
if (event.kind !== 'tool')
|
|
843
|
+
return;
|
|
844
|
+
collectCalled(toolCalls, event.phase, {
|
|
845
|
+
tool: event.tool,
|
|
846
|
+
...(event.ok === undefined ? {} : { ok: event.ok }),
|
|
847
|
+
});
|
|
848
|
+
});
|
|
849
|
+
try {
|
|
850
|
+
// WS5 (#27) — one session PER TURN, counter and all. The resume probe runs
|
|
851
|
+
// the same ask twice, and a second turn in the SAME conversation carries the
|
|
852
|
+
// first answer in its history — so its input genuinely differs, the replay
|
|
853
|
+
// correctly misses, and the row would compare two different questions. A
|
|
854
|
+
// fresh conversation each time is what the other surfaces do anyway (a
|
|
855
|
+
// one-shot CLI answer, a fresh child process). The counter keeps the ids
|
|
856
|
+
// unique within a run; nothing else reads them.
|
|
857
|
+
consoleRun += 1;
|
|
858
|
+
const { value: result, otel } = await withOtlpCollector(() => console_.answer(`parity-${scenario.id}-${consoleRun}`, scenario.message, {
|
|
859
|
+
provider: PARITY_PROVIDER_TYPE,
|
|
860
|
+
model: PARITY_MODEL,
|
|
861
|
+
}));
|
|
862
|
+
return toObservation('dashboard-chat', scenario, { ...result, debugLog: debugLogOf('dashboard-chat'), otel }, toolCalls, stub.chatCalls());
|
|
863
|
+
}
|
|
864
|
+
finally {
|
|
865
|
+
off();
|
|
866
|
+
}
|
|
867
|
+
}
|
|
868
|
+
finally {
|
|
869
|
+
await stub.close();
|
|
870
|
+
}
|
|
871
|
+
}
|
|
872
|
+
/** A channel adapter that records every send, so the gateway driver sees a real reply. */
|
|
873
|
+
class RecordingAdapter {
|
|
874
|
+
platform = 'mock';
|
|
875
|
+
configured = true;
|
|
876
|
+
sent = [];
|
|
877
|
+
describe() {
|
|
878
|
+
return 'Parity recorder';
|
|
879
|
+
}
|
|
880
|
+
async start() { }
|
|
881
|
+
async stop() { }
|
|
882
|
+
async send(_channelId, text) {
|
|
883
|
+
this.sent.push(text);
|
|
884
|
+
return true;
|
|
885
|
+
}
|
|
886
|
+
}
|
|
887
|
+
/** Unique per gateway run: the gateway dedups a re-delivered message. */
|
|
888
|
+
let gatewayRun = 0;
|
|
889
|
+
/**
|
|
890
|
+
* Unique per console turn (WS5): the resume probe drives the same ask twice, and
|
|
891
|
+
* two turns in one conversation would carry the first answer in the second's
|
|
892
|
+
* history (see the comment at the call site).
|
|
893
|
+
*/
|
|
894
|
+
let consoleRun = 0;
|
|
895
|
+
/**
|
|
896
|
+
* Gateway (WhatsApp / Telegram / …): the REAL registry handler, with no chat
|
|
897
|
+
* engine injected, so `runInboundChat` lazily imports the real `ChatCommand`.
|
|
898
|
+
* The gateway derives the provider/model pair from its own config
|
|
899
|
+
* (`registry.ts:1797`) — which the harness has pointed at the stub — so this is
|
|
900
|
+
* the surface's own resolution, not something forced in from here.
|
|
901
|
+
*
|
|
902
|
+
* The observation is read from the gateway's OWN durable record, the
|
|
903
|
+
* `inbound.chat` log entry: a messaging surface has no terminal to scroll, so
|
|
904
|
+
* the log is where its attribution and tool lifecycle have to live. Reading the
|
|
905
|
+
* log rather than a return value is the honest test of that claim — if the
|
|
906
|
+
* gateway stops writing the triple, this goes red.
|
|
907
|
+
*/
|
|
908
|
+
async function runViaGateway(ws, scenario) {
|
|
909
|
+
const stub = await startStub(scenario);
|
|
910
|
+
const deliveryDir = mkdtempSync(join(ws.root, 'gateway-'));
|
|
911
|
+
try {
|
|
912
|
+
ws.useStub(stub.baseUrl);
|
|
913
|
+
await clearResponseCache();
|
|
914
|
+
const { GatewayRegistry } = await import('../gateway/registry.js');
|
|
915
|
+
const adapter = new RecordingAdapter();
|
|
916
|
+
const registry = new GatewayRegistry({ streamEvents: false, deliveryConfigDir: deliveryDir });
|
|
917
|
+
registry.register(adapter);
|
|
918
|
+
gatewayRun += 1;
|
|
919
|
+
const channelId = `parity-${scenario.id}-${gatewayRun}`;
|
|
920
|
+
const { otel } = await withOtlpCollector(() => registry.handleInbound({
|
|
921
|
+
platform: 'mock',
|
|
922
|
+
channelId,
|
|
923
|
+
text: scenario.message,
|
|
924
|
+
from: 'parity',
|
|
925
|
+
senderId: 'parity',
|
|
926
|
+
}, { forceKind: 'chat' }));
|
|
927
|
+
const record = readGatewayLog(50).find((r) => r.event === 'inbound.chat' && r.channelId === channelId);
|
|
928
|
+
if (!record) {
|
|
929
|
+
throw new Error('the gateway did not record an inbound.chat turn for this message');
|
|
930
|
+
}
|
|
931
|
+
// The log carries the surface's own `{ tool, ok }` lifecycle; absence is
|
|
932
|
+
// preserved as absence (a call whose outcome frame never arrived).
|
|
933
|
+
const toolCalls = Array.isArray(record.toolCalls)
|
|
934
|
+
? record.toolCalls.map((call) => ({
|
|
935
|
+
tool: call.tool,
|
|
936
|
+
...(typeof call.ok === 'boolean' ? { ok: call.ok } : {}),
|
|
937
|
+
}))
|
|
938
|
+
: [];
|
|
939
|
+
const reply = [...adapter.sent].reverse().find((line) => line.includes(scenario.answer));
|
|
940
|
+
// WS1 — the findings the gateway recorded for this turn, read from its own
|
|
941
|
+
// durable record for the same reason the tool lifecycle is: a messaging
|
|
942
|
+
// surface has no terminal, so `inbound.chat` IS where this surface said it.
|
|
943
|
+
const findings = Array.isArray(record.findings)
|
|
944
|
+
? record.findings
|
|
945
|
+
: [];
|
|
946
|
+
return toObservation('gateway-chat', scenario, {
|
|
947
|
+
content: reply,
|
|
948
|
+
findings,
|
|
949
|
+
provider: typeof record.provider === 'string' ? record.provider : undefined,
|
|
950
|
+
model: typeof record.model === 'string' ? record.model : undefined,
|
|
951
|
+
transport: record.transport === 'native' || record.transport === 'json'
|
|
952
|
+
? record.transport
|
|
953
|
+
: record.transport === 'none'
|
|
954
|
+
? 'none'
|
|
955
|
+
: undefined,
|
|
956
|
+
generationFailed: record.generationFailed === true,
|
|
957
|
+
// WS2 — read from the same isolated profile the gateway wrote into.
|
|
958
|
+
debugLog: debugLogOf('gateway-chat'),
|
|
959
|
+
// WS3 — the span tree the collector received from this surface.
|
|
960
|
+
otel,
|
|
961
|
+
// WS5 — read from the gateway's own durable record, for the same reason
|
|
962
|
+
// the tool lifecycle and the findings are: a messaging surface has no
|
|
963
|
+
// terminal, so `inbound.chat` IS where this surface said what it did.
|
|
964
|
+
...(record.worktree && typeof record.worktree === 'object'
|
|
965
|
+
? { worktree: record.worktree }
|
|
966
|
+
: {}),
|
|
967
|
+
...(record.resume && typeof record.resume === 'object'
|
|
968
|
+
? { resume: record.resume }
|
|
969
|
+
: {}),
|
|
970
|
+
}, toolCalls, stub.chatCalls());
|
|
971
|
+
}
|
|
972
|
+
finally {
|
|
973
|
+
await stub.close();
|
|
974
|
+
rmSync(deliveryDir, { recursive: true, force: true });
|
|
975
|
+
}
|
|
976
|
+
}
|
|
977
|
+
/**
|
|
978
|
+
* CLI execute / one-shot: the COMMAND's own single-goal path, with the provider
|
|
979
|
+
* served by the shared factory pointed at the stub. Both of the command's arms
|
|
980
|
+
* report now — the loop engine through `runLoopExecutor` and the direct chat
|
|
981
|
+
* answer through the engine's `onToolCall` — so the observation is what the
|
|
982
|
+
* COMMAND returns, not what the engine inside it happened to know.
|
|
983
|
+
*
|
|
984
|
+
* `runSingleGoal` is `private` in `src/cli/execute.ts`; the cast below is the
|
|
985
|
+
* same typed seam the tests already use, and it goes red the day the command
|
|
986
|
+
* stops handing its own result back.
|
|
987
|
+
*/
|
|
988
|
+
async function runViaExecuteCommand(ws, scenario) {
|
|
989
|
+
const stub = await startStub(scenario);
|
|
990
|
+
try {
|
|
991
|
+
ws.useStub(stub.baseUrl);
|
|
992
|
+
await clearResponseCache();
|
|
993
|
+
const { ExecuteCommand } = await import('../cli/execute.js');
|
|
994
|
+
const command = new ExecuteCommand();
|
|
995
|
+
const { value: result, otel } = await withOtlpCollector(() => command.runSingleGoal(scenario.message, PARITY_PROVIDER_TYPE, PARITY_MODEL, {
|
|
996
|
+
engine: 'loop',
|
|
997
|
+
}));
|
|
998
|
+
return toObservation('cli-execute', scenario, {
|
|
999
|
+
content: result.content,
|
|
1000
|
+
provider: result.provider,
|
|
1001
|
+
model: result.model,
|
|
1002
|
+
transport: result.transport,
|
|
1003
|
+
generationFailed: !result.success,
|
|
1004
|
+
...(result.findings ? { findings: result.findings } : {}),
|
|
1005
|
+
debugLog: debugLogOf('cli-execute'),
|
|
1006
|
+
// WS3 — the span tree this command's loop exported for the turn.
|
|
1007
|
+
otel,
|
|
1008
|
+
// WS5 — the command's own report, on either engine arm.
|
|
1009
|
+
...(result.worktree ? { worktree: result.worktree } : {}),
|
|
1010
|
+
...(result.resume ? { resume: result.resume } : {}),
|
|
1011
|
+
},
|
|
1012
|
+
// The command's own per-call outcomes (captured from the loop's
|
|
1013
|
+
// `tool`/`refusal` events). The fallback keeps a name-only result honest:
|
|
1014
|
+
// `ok` stays absent rather than being guessed.
|
|
1015
|
+
result.toolOutcomes
|
|
1016
|
+
? result.toolOutcomes.map((call) => ({
|
|
1017
|
+
tool: call.tool,
|
|
1018
|
+
...(typeof call.ok === 'boolean' ? { ok: call.ok } : {}),
|
|
1019
|
+
}))
|
|
1020
|
+
: (result.toolCalls ?? []).map((tool) => ({ tool })), stub.chatCalls());
|
|
1021
|
+
}
|
|
1022
|
+
finally {
|
|
1023
|
+
await stub.close();
|
|
1024
|
+
}
|
|
1025
|
+
}
|
|
1026
|
+
/**
|
|
1027
|
+
* Subagent: a REAL forked child process over its IPC channel, with the model
|
|
1028
|
+
* served by the same loopback stub. The child resolves its OWN provider from
|
|
1029
|
+
* config — that IS the isolation boundary — but the harness points that provider
|
|
1030
|
+
* at the same `groq` id and the same stub the in-process surfaces use, so
|
|
1031
|
+
* agreement is evidence rather than a stub coincidence.
|
|
1032
|
+
*
|
|
1033
|
+
* The observation is read from what the child itself reported (provider, model,
|
|
1034
|
+
* transport, llm calls, and each tool call + outcome on its progress frames),
|
|
1035
|
+
* never reconstructed here.
|
|
1036
|
+
*/
|
|
1037
|
+
async function runViaSubagent(ws, scenario) {
|
|
1038
|
+
const stub = await startStub(scenario);
|
|
1039
|
+
// The child gets its OWN config/memory dirs, because it is its own process —
|
|
1040
|
+
// but the file it reads points at the SAME stub server.
|
|
1041
|
+
//
|
|
1042
|
+
// DETERMINISTIC, not `mkdtemp`: a resume probe drives the same scenario twice,
|
|
1043
|
+
// and the child's step record lives in the memory dir it inherits. A throwaway
|
|
1044
|
+
// dir per call would put the second turn's record in a different place from the
|
|
1045
|
+
// first one's, so the child could never replay anything and the probe would be
|
|
1046
|
+
// measuring the harness's own bookkeeping. Keyed by SCENARIO (not by surface) on
|
|
1047
|
+
// purpose: the child of one surface must not read another's record, and each
|
|
1048
|
+
// surface's own two turns must share one.
|
|
1049
|
+
const childRoot = join(ws.root, `subagent-${scenario.id}`);
|
|
1050
|
+
const childConfig = join(childRoot, 'config');
|
|
1051
|
+
mkdirSync(childConfig, { recursive: true });
|
|
1052
|
+
writeFileSync(join(childConfig, 'buffconfig.json'), JSON.stringify({
|
|
1053
|
+
defaultProvider: PARITY_PROVIDER_TYPE,
|
|
1054
|
+
providers: {
|
|
1055
|
+
[PARITY_PROVIDER_TYPE]: {
|
|
1056
|
+
apiKey: PARITY_API_KEY,
|
|
1057
|
+
model: PARITY_MODEL,
|
|
1058
|
+
baseUrl: stub.baseUrl,
|
|
1059
|
+
},
|
|
1060
|
+
},
|
|
1061
|
+
}));
|
|
1062
|
+
try {
|
|
1063
|
+
const { getSubagentManager } = await import('../tools/subagent-spawner.js');
|
|
1064
|
+
const manager = getSubagentManager();
|
|
1065
|
+
const callsByRun = new Map();
|
|
1066
|
+
// The listener is attached before the id exists, and a progress frame can
|
|
1067
|
+
// arrive before `spawn()` resolves — hence keying by run id.
|
|
1068
|
+
const onProgress = (id, msg) => {
|
|
1069
|
+
// The OUTCOME frame: the child reports `tool_call` then `tool_result`, and
|
|
1070
|
+
// the driver records the one that carries WHAT HAPPENED, exactly as the
|
|
1071
|
+
// `called` phase is read in-process. A missing outcome stays absent.
|
|
1072
|
+
if (msg?.phase !== 'tool_result' || typeof msg.tool !== 'string')
|
|
1073
|
+
return;
|
|
1074
|
+
const list = callsByRun.get(id) ?? [];
|
|
1075
|
+
list.push({ tool: msg.tool, ...(typeof msg.ok === 'boolean' ? { ok: msg.ok } : {}) });
|
|
1076
|
+
callsByRun.set(id, list);
|
|
1077
|
+
};
|
|
1078
|
+
manager.on('progress', onProgress);
|
|
1079
|
+
try {
|
|
1080
|
+
// A tool is ALWAYS offered to the child, even for a scenario that calls
|
|
1081
|
+
// none: the in-process surfaces always have their tools available, so
|
|
1082
|
+
// offering the child one keeps its transport attribution comparable. The
|
|
1083
|
+
// stub only asks for the tool when the scenario does.
|
|
1084
|
+
const availableTool = scenario.toolCall?.tool ?? 'list_dir';
|
|
1085
|
+
// WS3 — one collector for this surface, held across the WHOLE child run:
|
|
1086
|
+
// the fork inherits `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT` (and `NUVIRA_OTEL`)
|
|
1087
|
+
// from this process's environment, and the child flushes into it before it
|
|
1088
|
+
// reports its result — so what is read here is the child's OWN export,
|
|
1089
|
+
// produced in its own process, rather than something reconstructed here.
|
|
1090
|
+
const spawnConfig = {
|
|
1091
|
+
goal: scenario.message,
|
|
1092
|
+
provider: PARITY_PROVIDER_TYPE,
|
|
1093
|
+
model: PARITY_MODEL,
|
|
1094
|
+
tools: [availableTool],
|
|
1095
|
+
// WS5 (#27) — delegation-level isolation: the PARENT makes the worktree,
|
|
1096
|
+
// forks the child INTO it and measures the diff itself. That is the
|
|
1097
|
+
// mechanism a delegating caller actually has (the child resolves its own
|
|
1098
|
+
// config and could decline), and it is why this surface is asked through
|
|
1099
|
+
// the spawn config rather than through the environment the in-process
|
|
1100
|
+
// surfaces read.
|
|
1101
|
+
...(scenario.isolation === true ? { worktree: true } : {}),
|
|
1102
|
+
env: {
|
|
1103
|
+
NUVIRA_CONFIG_DIR: childConfig,
|
|
1104
|
+
NUVIRA_MEMORY_DIR: join(childRoot, 'memory'),
|
|
1105
|
+
// WS2 — the child keeps its OWN config/memory (that IS the isolation
|
|
1106
|
+
// boundary) but writes its debug log into the run's throwaway debug
|
|
1107
|
+
// dir, so this driver and a test can read the artifact the child
|
|
1108
|
+
// actually produced instead of re-deriving a path inside the child's
|
|
1109
|
+
// private profile.
|
|
1110
|
+
NUVIRA_DEBUG_LOG_DIR: debugLogDir(),
|
|
1111
|
+
},
|
|
1112
|
+
};
|
|
1113
|
+
const { value, otel } = await withOtlpCollector(async () => {
|
|
1114
|
+
const state = await manager.spawn(spawnConfig);
|
|
1115
|
+
try {
|
|
1116
|
+
return { state, result: await manager.waitForCompletion(state.id, 60_000) };
|
|
1117
|
+
}
|
|
1118
|
+
catch {
|
|
1119
|
+
// A CHILD THAT FAILED IS AN OBSERVATION, NOT A HARNESS CRASH — found by
|
|
1120
|
+
// the WS6 provider-fault row, which is the first scenario where the child
|
|
1121
|
+
// legitimately ends in failure. `waitForCompletion` REJECTS on a failed
|
|
1122
|
+
// run, so this throw used to escape the driver, abort `runParityScenario`
|
|
1123
|
+
// and fail the harness with an exception instead of a verdict. A harness
|
|
1124
|
+
// that cannot say "every surface failed honestly" cannot measure fault
|
|
1125
|
+
// handling at all.
|
|
1126
|
+
//
|
|
1127
|
+
// The rejection is DISCARDED in favour of the manager's own recorded
|
|
1128
|
+
// state: what is reported is the child's own report (its error, its
|
|
1129
|
+
// attribution, its call counts), not the exception this driver happened
|
|
1130
|
+
// to catch — the same "read the surface's own record" rule the gateway
|
|
1131
|
+
// and debug-log drivers follow.
|
|
1132
|
+
const failed = manager.getState(state.id) ?? state;
|
|
1133
|
+
return {
|
|
1134
|
+
state: failed,
|
|
1135
|
+
result: {
|
|
1136
|
+
id: failed.id,
|
|
1137
|
+
success: false,
|
|
1138
|
+
result: failed.result ?? '',
|
|
1139
|
+
...(failed.error ? { error: failed.error } : {}),
|
|
1140
|
+
...(failed.refusalCode ? { refusalCode: failed.refusalCode } : {}),
|
|
1141
|
+
...(failed.provider ? { provider: failed.provider } : {}),
|
|
1142
|
+
...(failed.model ? { model: failed.model } : {}),
|
|
1143
|
+
...(failed.transport ? { transport: failed.transport } : {}),
|
|
1144
|
+
...(failed.findings ? { findings: failed.findings } : {}),
|
|
1145
|
+
...(failed.resume ? { resume: failed.resume } : {}),
|
|
1146
|
+
llmCalls: failed.llmCalls,
|
|
1147
|
+
tokensUsed: failed.tokensUsed,
|
|
1148
|
+
toolCalls: failed.toolCalls,
|
|
1149
|
+
durationMs: failed.durationMs ?? 0,
|
|
1150
|
+
log: manager.getLog(failed.id),
|
|
1151
|
+
},
|
|
1152
|
+
};
|
|
1153
|
+
}
|
|
1154
|
+
});
|
|
1155
|
+
const { state, result } = value;
|
|
1156
|
+
const transport = result.transport === 'native' || result.transport === 'json' ? result.transport : 'none';
|
|
1157
|
+
// The child's own verdict, through the shared helper (it reports success
|
|
1158
|
+
// directly, so `ok` is the flag it has) — the same rule across the fork
|
|
1159
|
+
// rather than a second one that could disagree.
|
|
1160
|
+
const status = turnStatus({ ok: result.success });
|
|
1161
|
+
const childCalls = callsByRun.get(state.id) ?? [];
|
|
1162
|
+
return {
|
|
1163
|
+
surface: 'subagent',
|
|
1164
|
+
engine: 'loop',
|
|
1165
|
+
status,
|
|
1166
|
+
modelCalls: result.llmCalls,
|
|
1167
|
+
...(result.provider ? { provider: result.provider } : {}),
|
|
1168
|
+
...(result.model ? { model: result.model } : {}),
|
|
1169
|
+
transport,
|
|
1170
|
+
toolCalls: childCalls,
|
|
1171
|
+
// WS1 — the child's own findings, read from the frames it sent.
|
|
1172
|
+
findings: result.findings ?? [],
|
|
1173
|
+
// WS2 — the child's own debug log, read back from the file it wrote
|
|
1174
|
+
// (its dir is pinned into the child's env above, so this is the CHILD's
|
|
1175
|
+
// artifact — produced in its own process, with its own provider object —
|
|
1176
|
+
// and not something reconstructed here).
|
|
1177
|
+
debugLog: debugLogOf('subagent'),
|
|
1178
|
+
// WS3 — the span tree the CHILD exported, read back from the collector it
|
|
1179
|
+
// was pointed at through its inherited environment.
|
|
1180
|
+
otel,
|
|
1181
|
+
// WS4 — replaced by the driver wrapper with what the operator's declared
|
|
1182
|
+
// hook received: the child runs the hook in its OWN process and appends
|
|
1183
|
+
// to the log file it inherited the path of (`withToolHooks`).
|
|
1184
|
+
hooks: noToolHooks(),
|
|
1185
|
+
// WS5 — the isolation the PARENT made for this child, and what the diff
|
|
1186
|
+
// against the base commit was. Read from the result the manager hands its
|
|
1187
|
+
// caller, which is the artifact a delegating caller actually receives.
|
|
1188
|
+
isolation: isolationObsOf(scenario.isolation === true, result.worktree),
|
|
1189
|
+
// WS5 — the CHILD's own resume report, read from the frame it sent (the
|
|
1190
|
+
// child is a separate process; this is its only channel back).
|
|
1191
|
+
resume: resumeObsOf(scenario.resume === true, result.resume),
|
|
1192
|
+
// WS6 (#28) — the declared fault, derived from what the CHILD's own frames
|
|
1193
|
+
// and result say. This is the field that proves a declaration crosses the
|
|
1194
|
+
// fork: a `tool` fault reaches this child through the environment the
|
|
1195
|
+
// parent handed it, and the failed call is reported back on a frame.
|
|
1196
|
+
fault: faultObsOf(scenario, childCalls, status),
|
|
1197
|
+
...(result.result ? { answer: result.result } : {}),
|
|
1198
|
+
...(result.success ? {} : { errorCode: result.refusalCode ?? 'turn_failed' }),
|
|
1199
|
+
noise: { at: Date.now() },
|
|
1200
|
+
};
|
|
1201
|
+
}
|
|
1202
|
+
finally {
|
|
1203
|
+
manager.off('progress', onProgress);
|
|
1204
|
+
}
|
|
1205
|
+
}
|
|
1206
|
+
finally {
|
|
1207
|
+
// The child's root is left in place (the workspace's own teardown removes it):
|
|
1208
|
+
// a resume probe's two turns must share it, so deleting it here would delete the
|
|
1209
|
+
// record the second turn is about to read.
|
|
1210
|
+
await stub.close();
|
|
1211
|
+
}
|
|
1212
|
+
}
|
|
1213
|
+
/**
|
|
1214
|
+
* A driver that exists only to record why it cannot run. Exported so a caller
|
|
1215
|
+
* can pin the guard that a blocked surface is never silently invoked — the real
|
|
1216
|
+
* driver list currently needs no blocks.
|
|
1217
|
+
*/
|
|
1218
|
+
export function blockedDriver(surface, reason) {
|
|
1219
|
+
return {
|
|
1220
|
+
surface,
|
|
1221
|
+
depth: DRIVER_DEPTH,
|
|
1222
|
+
available: false,
|
|
1223
|
+
blockedBy: reason,
|
|
1224
|
+
run: async () => {
|
|
1225
|
+
throw new Error(`parity driver for ${surface} is blocked and must never be invoked: ${reason}`);
|
|
1226
|
+
},
|
|
1227
|
+
};
|
|
1228
|
+
}
|
|
1229
|
+
/**
|
|
1230
|
+
* Build the real drivers.
|
|
1231
|
+
*
|
|
1232
|
+
* The environment is switched to a throwaway profile here and restored in
|
|
1233
|
+
* `dispose()`. Every module that resolves `NUVIRA_CONFIG_DIR` / `NUVIRA_MEMORY_DIR`
|
|
1234
|
+
* at call time (the config manager, the cache, the gateway log) therefore reads
|
|
1235
|
+
* the isolated profile for the whole run, and the developer's real profile is
|
|
1236
|
+
* untouched — the same hermetic convention the test suite uses.
|
|
1237
|
+
*/
|
|
1238
|
+
export async function createParityHarness() {
|
|
1239
|
+
const workspace = createWorkspace();
|
|
1240
|
+
const previous = {
|
|
1241
|
+
configDir: process.env.NUVIRA_CONFIG_DIR,
|
|
1242
|
+
memoryDir: process.env.NUVIRA_MEMORY_DIR,
|
|
1243
|
+
buffConfigDir: process.env.BUFF_CONFIG_DIR,
|
|
1244
|
+
buffMemoryDir: process.env.BUFF_MEMORY_DIR,
|
|
1245
|
+
debugLog: process.env.NUVIRA_DEBUG_LOG,
|
|
1246
|
+
otel: process.env[otelEnableVarName],
|
|
1247
|
+
otelEndpoint: process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT,
|
|
1248
|
+
// WS5 (#27) — never the developer's own request. The harness DECLARES these
|
|
1249
|
+
// per scenario (`withTurnEnvelope`); a value inherited from the shell would
|
|
1250
|
+
// make the scenarios that do not ask for the capability measure it anyway.
|
|
1251
|
+
isolation: process.env[WORKTREE_ENABLE_ENV],
|
|
1252
|
+
resume: process.env[RESUME_ENABLE_ENV],
|
|
1253
|
+
// WS6 (#28) — same rule: the harness DECLARES a fault per scenario, so a
|
|
1254
|
+
// declaration inherited from the shell must not arm one for every scenario.
|
|
1255
|
+
fault: process.env[FAULT_ENV],
|
|
1256
|
+
};
|
|
1257
|
+
process.env.NUVIRA_CONFIG_DIR = workspace.configDir;
|
|
1258
|
+
process.env.NUVIRA_MEMORY_DIR = workspace.memoryDir;
|
|
1259
|
+
// WS2 — logging is turned ON for the whole run, so "this surface wrote a log
|
|
1260
|
+
// whose header names the backend" is an ASSERTION rather than something the
|
|
1261
|
+
// harness never asked for. Every surface writes into the isolated profile
|
|
1262
|
+
// above (a forked child into its own), so nothing reaches the developer's
|
|
1263
|
+
// real `~/.nuvira`.
|
|
1264
|
+
process.env.NUVIRA_DEBUG_LOG = '1';
|
|
1265
|
+
// WS3 — span export is turned ON for the whole run, so "this surface exported
|
|
1266
|
+
// its turn" is an ASSERTION rather than something the harness never asked for.
|
|
1267
|
+
// Only the GATE belongs here; the endpoint is set per driver
|
|
1268
|
+
// (`withOtlpCollector`), because each surface gets its own collector.
|
|
1269
|
+
process.env[otelEnableVarName] = '1';
|
|
1270
|
+
// WS6 (#28) — and no fault, until a scenario declares one.
|
|
1271
|
+
delete process.env[FAULT_ENV];
|
|
1272
|
+
resetFaultInjector();
|
|
1273
|
+
// The legacy aliases would otherwise win on the modules that check them, and
|
|
1274
|
+
// point half the run back at the developer's real profile.
|
|
1275
|
+
delete process.env.BUFF_CONFIG_DIR;
|
|
1276
|
+
delete process.env.BUFF_MEMORY_DIR;
|
|
1277
|
+
const driver = (surface, run) => ({
|
|
1278
|
+
surface,
|
|
1279
|
+
depth: DRIVER_DEPTH,
|
|
1280
|
+
available: true,
|
|
1281
|
+
run: async (scenario) => {
|
|
1282
|
+
// WS3 — the SDK keeps ONE provider per process and reads the endpoint when
|
|
1283
|
+
// it builds it, so the previous surface's provider (pointing at the
|
|
1284
|
+
// previous collector, now closed) has to go before this surface starts.
|
|
1285
|
+
// Without this reset the second surface would export into the first
|
|
1286
|
+
// surface's collector, and every surface after that would read as having
|
|
1287
|
+
// exported nothing — a silent, permanent green on one surface only.
|
|
1288
|
+
await shutdownSpans();
|
|
1289
|
+
// WS5 — isolation and resume are DECLARED around the turn (and undeclared
|
|
1290
|
+
// again after it), so "this surface isolated its turn / replayed its steps"
|
|
1291
|
+
// is an assertion about a declaration the harness actually made. OUTSIDE
|
|
1292
|
+
// the hook wrapper because the resume probe runs the turn twice, and each
|
|
1293
|
+
// of those two turns is a whole turn of its own.
|
|
1294
|
+
return withTurnEnvelope(scenario, surface, () =>
|
|
1295
|
+
// WS4 — the operator's hooks are DECLARED around the turn (and undeclared
|
|
1296
|
+
// again after it), so "this surface ran the hook" is an assertion about a
|
|
1297
|
+
// declaration the harness actually made.
|
|
1298
|
+
withToolHooks(workspace, surface, scenario, () => run(workspace, scenario)));
|
|
1299
|
+
},
|
|
1300
|
+
});
|
|
1301
|
+
const inProcess = [
|
|
1302
|
+
driver('cli-chat', runViaChatOnce),
|
|
1303
|
+
driver('dashboard-chat', runViaConsole),
|
|
1304
|
+
driver('gateway-chat', runViaGateway),
|
|
1305
|
+
driver('cli-execute', runViaExecuteCommand),
|
|
1306
|
+
];
|
|
1307
|
+
const subagent = [driver('subagent', runViaSubagent)];
|
|
1308
|
+
return {
|
|
1309
|
+
drivers: [...inProcess, ...subagent],
|
|
1310
|
+
inProcess,
|
|
1311
|
+
subagent,
|
|
1312
|
+
async dispose() {
|
|
1313
|
+
if (previous.configDir === undefined)
|
|
1314
|
+
delete process.env.NUVIRA_CONFIG_DIR;
|
|
1315
|
+
else
|
|
1316
|
+
process.env.NUVIRA_CONFIG_DIR = previous.configDir;
|
|
1317
|
+
if (previous.memoryDir === undefined)
|
|
1318
|
+
delete process.env.NUVIRA_MEMORY_DIR;
|
|
1319
|
+
else
|
|
1320
|
+
process.env.NUVIRA_MEMORY_DIR = previous.memoryDir;
|
|
1321
|
+
if (previous.buffConfigDir === undefined)
|
|
1322
|
+
delete process.env.BUFF_CONFIG_DIR;
|
|
1323
|
+
else
|
|
1324
|
+
process.env.BUFF_CONFIG_DIR = previous.buffConfigDir;
|
|
1325
|
+
if (previous.buffMemoryDir === undefined)
|
|
1326
|
+
delete process.env.BUFF_MEMORY_DIR;
|
|
1327
|
+
else
|
|
1328
|
+
process.env.BUFF_MEMORY_DIR = previous.buffMemoryDir;
|
|
1329
|
+
if (previous.debugLog === undefined)
|
|
1330
|
+
delete process.env.NUVIRA_DEBUG_LOG;
|
|
1331
|
+
else
|
|
1332
|
+
process.env.NUVIRA_DEBUG_LOG = previous.debugLog;
|
|
1333
|
+
if (previous.otel === undefined)
|
|
1334
|
+
delete process.env[otelEnableVarName];
|
|
1335
|
+
else
|
|
1336
|
+
process.env[otelEnableVarName] = previous.otel;
|
|
1337
|
+
if (previous.otelEndpoint === undefined)
|
|
1338
|
+
delete process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT;
|
|
1339
|
+
else
|
|
1340
|
+
process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = previous.otelEndpoint;
|
|
1341
|
+
if (previous.isolation === undefined)
|
|
1342
|
+
delete process.env[WORKTREE_ENABLE_ENV];
|
|
1343
|
+
else
|
|
1344
|
+
process.env[WORKTREE_ENABLE_ENV] = previous.isolation;
|
|
1345
|
+
if (previous.resume === undefined)
|
|
1346
|
+
delete process.env[RESUME_ENABLE_ENV];
|
|
1347
|
+
else
|
|
1348
|
+
process.env[RESUME_ENABLE_ENV] = previous.resume;
|
|
1349
|
+
if (previous.fault === undefined)
|
|
1350
|
+
delete process.env[FAULT_ENV];
|
|
1351
|
+
else
|
|
1352
|
+
process.env[FAULT_ENV] = previous.fault;
|
|
1353
|
+
resetFaultInjector();
|
|
1354
|
+
// Tear the LAST surface's provider down too: a test file that runs several
|
|
1355
|
+
// parity runs in a row would otherwise inherit the first one's provider,
|
|
1356
|
+
// still pointing at a collector that has been closed.
|
|
1357
|
+
await shutdownSpans();
|
|
1358
|
+
rmSync(workspace.root, { recursive: true, force: true });
|
|
1359
|
+
},
|
|
1360
|
+
};
|
|
1361
|
+
}
|
|
1362
|
+
//# sourceMappingURL=drivers.js.map
|