@danceiny/gotry 0.0.1-rc.16 → 0.0.1-rc.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +47 -13
- package/README.zh-CN.md +20 -10
- package/bin/gotry-booking-copilot.js +53 -0
- package/bin/gotry-bootstrap.js +111 -51
- package/bin/gotry-inner.js +442 -60
- package/bin/gotry-runtime-resolution.d.ts +27 -0
- package/bin/gotry-runtime-resolution.js +50 -0
- package/bin/gotry.js +1 -1
- package/cordis.gotry-patch.yml +5 -1
- package/dist/capabilities/agent-reach-deep.js +1 -1
- package/dist/capabilities/agent-reach.js +1 -1
- package/dist/capabilities/anything.js +1 -1
- package/dist/capabilities/artifacts.js +1 -1
- package/dist/capabilities/effect.js +1 -1
- package/dist/capabilities/fact-log.js +1 -1
- package/dist/capabilities/flyai.js +20 -6
- package/dist/capabilities/hbcli.js +1 -1
- package/dist/capabilities/incident-log.js +1 -1
- package/dist/capabilities/model-override.js +18 -0
- package/dist/capabilities/opensky.js +1 -1
- package/dist/capabilities/resilience.js +1 -1
- package/dist/capabilities/session/action-cache.js +1 -1
- package/dist/capabilities/session/adapters/ctrip-flight.js +1 -1
- package/dist/capabilities/session/adapters/meituan-local.js +1 -1
- package/dist/capabilities/session/benchmark.js +1 -1
- package/dist/capabilities/session/extension-bridge.js +19 -6
- package/dist/capabilities/session/extension-channel.js +3 -3
- package/dist/capabilities/session/extension-distribution.js +234 -0
- package/dist/capabilities/session/extract.js +1 -1
- package/dist/capabilities/session/golden-score.js +92 -0
- package/dist/capabilities/session/health-watch.js +1 -1
- package/dist/capabilities/session/read-guard.js +1 -1
- package/dist/capabilities/session/static-flight-golden.js +137 -0
- package/dist/capabilities/session/transport.js +1 -1
- package/dist/capabilities/session/wizard.js +21 -251
- package/dist/capabilities/session-consent.js +1 -1
- package/dist/capabilities/session-login.js +11 -5
- package/dist/capabilities/session-search.js +46 -4
- package/dist/capabilities/weather.js +168 -46
- package/dist/data/session-golden-20.json +25 -0
- package/dist/data/sf-golden-manifest.json +102 -0
- package/dist/data/sf-static-routes.json +91 -0
- package/dist/scripts/action-cache-tests.js +1 -1
- package/dist/scripts/agent-planning-budget-e2e.js +227 -0
- package/dist/scripts/agent-planning-budget-tests.js +173 -0
- package/dist/scripts/agent-reach-deep-tests.js +1 -1
- package/dist/scripts/agent-reach-tests.js +1 -1
- package/dist/scripts/agent-reach-wrapper-tests.js +1 -1
- package/dist/scripts/anything-tests.js +1 -1
- package/dist/scripts/async-collect.js +1 -1
- package/dist/scripts/benchmark-environment-bridge-e2e.js +981 -0
- package/dist/scripts/benchmark-environment-bridge-tests.js +2544 -0
- package/dist/scripts/booking-copilot-availability-ledger-binding-tests.js +335 -0
- package/dist/scripts/booking-copilot-availability-policy-v2-tests.js +1355 -0
- package/dist/scripts/booking-copilot-bin-proof-tests.js +89 -0
- package/dist/scripts/booking-copilot-crossrepo-fixture-server.js +113 -0
- package/dist/scripts/booking-copilot-dsh-core-proof-tests.js +356 -0
- package/dist/scripts/booking-copilot-dsh-planner-proof-tests.js +557 -0
- package/dist/scripts/booking-copilot-dsh-plugin-proof-tests.js +140 -0
- package/dist/scripts/booking-copilot-event-sequence-concurrency-proof-tests.js +202 -0
- package/dist/scripts/booking-copilot-gap-code-contract-proof-tests.js +102 -0
- package/dist/scripts/booking-copilot-operation-ledger-concurrency-proof-tests.js +374 -0
- package/dist/scripts/booking-copilot-receipt-ledger-concurrency-proof-tests.js +447 -0
- package/dist/scripts/booking-copilot-runtime-proof-tests.js +205 -0
- package/dist/scripts/booking-copilot-server-proof-tests.js +317 -0
- package/dist/scripts/booking-copilot-startup-proof-tests.js +283 -0
- package/dist/scripts/booking-copilot-v2-runtime-proof-tests.js +3935 -0
- package/dist/scripts/booking-saga-tests.js +1 -1
- package/dist/scripts/booking-surface-contract-proof-tests.js +393 -0
- package/dist/scripts/booking-surface-v2-contract-proof-tests.js +1174 -0
- package/dist/scripts/bootstrap-tests.js +30 -10
- package/dist/scripts/build-changelog.js +1 -1
- package/dist/scripts/changelog-tests.js +1 -1
- package/dist/scripts/companion-tests.js +1 -1
- package/dist/scripts/diff-test.js +1 -1
- package/dist/scripts/dsh-runtime-closure-tests.js +225 -0
- package/dist/scripts/dsh-runtime-closure.js +150 -0
- package/dist/scripts/effect-tests.js +1 -1
- package/dist/scripts/engine-run.js +1 -1
- package/dist/scripts/engine-tests.js +1 -1
- package/dist/scripts/evaluation-cadence-tests.js +347 -0
- package/dist/scripts/evaluation-contract-tests.js +574 -0
- package/dist/scripts/extension-distribution-cli.js +35 -0
- package/dist/scripts/extension-distribution-tests.js +341 -0
- package/dist/scripts/extension-tests.js +128 -23
- package/dist/scripts/fact-gate-tests.js +1 -1
- package/dist/scripts/flyai-tests.js +1 -1
- package/dist/scripts/hbcli-e2e-tests.js +1 -1
- package/dist/scripts/hbcli-tests.js +1 -1
- package/dist/scripts/health-watch-cli.js +1 -1
- package/dist/scripts/i18n-tests.js +1 -1
- package/dist/scripts/incident-tests.js +1 -1
- package/dist/scripts/journey-tests.js +1 -1
- package/dist/scripts/ledger-tests.js +1 -1
- package/dist/scripts/ledger-workflow-crash.js +1 -1
- package/dist/scripts/memory-capture-tests.js +1 -1
- package/dist/scripts/memory-decay-tests.js +1 -1
- package/dist/scripts/memory-metrics.js +1 -1
- package/dist/scripts/memory-value-report.js +1 -1
- package/dist/scripts/model-override-e2e.js +176 -0
- package/dist/scripts/nightly-evidence-tests.js +1 -1
- package/dist/scripts/nightly-evidence.js +1 -1
- package/dist/scripts/nudge-digest.js +1 -1
- package/dist/scripts/onboarding-tests.js +21 -53
- package/dist/scripts/opensky-check.js +1 -1
- package/dist/scripts/opensky-tests.js +1 -1
- package/dist/scripts/pnpm-dsh-closure-proof.js +20 -0
- package/dist/scripts/price-drift-tests.js +1 -1
- package/dist/scripts/price-drift-watch.js +1 -1
- package/dist/scripts/probe-poi-tests.js +1 -1
- package/dist/scripts/product-metrics.js +1 -1
- package/dist/scripts/publish-preverify.js +40 -4
- package/dist/scripts/realtime-pricing-tests.js +1 -1
- package/dist/scripts/replay-async.js +1 -1
- package/dist/scripts/replay-real.js +1 -1
- package/dist/scripts/replay.js +1 -1
- package/dist/scripts/session-attach-diagnose.js +1 -1
- package/dist/scripts/session-attach-poc.js +1 -1
- package/dist/scripts/session-benchmark.js +1 -1
- package/dist/scripts/session-extract-tests.js +1 -1
- package/dist/scripts/session-login.js +1 -1
- package/dist/scripts/session-tests.js +70 -20
- package/dist/scripts/sf-live-benchmark.js +338 -0
- package/dist/scripts/sf-live-cli-tests.js +21 -0
- package/dist/scripts/sf-soft-score-tests.js +108 -0
- package/dist/scripts/sf-summary.js +93 -0
- package/dist/scripts/skeleton-check.js +1 -1
- package/dist/scripts/skeleton-integration-test.js +1 -1
- package/dist/scripts/skills-contract-tests.js +1 -1
- package/dist/scripts/smoke-session-gate-tests.js +29 -0
- package/dist/scripts/smoke.js +79 -33
- package/dist/scripts/state-cli-tests.js +1 -1
- package/dist/scripts/state-cli.js +1 -1
- package/dist/scripts/static-golden-tests.js +299 -0
- package/dist/scripts/time-eval-tests.js +1 -1
- package/dist/scripts/travel-timeline-tests.js +1 -1
- package/dist/scripts/unified-tests.js +1 -1
- package/dist/scripts/weather-tests.js +694 -44
- package/dist/scripts/z3-race-tests.js +1 -1
- package/dist/src/artifact-gate.js +1 -1
- package/dist/src/benchmark-agent-conformance.js +370 -0
- package/dist/src/benchmark-environment-bridge.js +384 -0
- package/dist/src/benchmark-headless-child-diagnostics.js +173 -0
- package/dist/src/benchmark-tool-isolation.js +124 -0
- package/dist/src/bookable-facts.js +1 -1
- package/dist/src/booking-saga.js +1 -1
- package/dist/src/booking-surface/availability-policy-v2.js +830 -0
- package/dist/src/booking-surface/canonical-schema.js +113 -0
- package/dist/src/booking-surface/contracts-v2.js +89 -0
- package/dist/src/booking-surface/contracts.js +46 -0
- package/dist/src/booking-surface/dsh-planner.js +453 -0
- package/dist/src/booking-surface/dsh-plugin.js +93 -0
- package/dist/src/booking-surface/error-codes.js +94 -0
- package/dist/src/booking-surface/index.js +15 -0
- package/dist/src/booking-surface/profile.js +68 -0
- package/dist/src/booking-surface/runtime-v2.js +1771 -0
- package/dist/src/booking-surface/runtime.js +351 -0
- package/dist/src/booking-surface/server-v2.js +334 -0
- package/dist/src/booking-surface/server.js +302 -0
- package/dist/src/booking-surface/startup.js +159 -0
- package/dist/src/booking-surface/validation-v2.js +319 -0
- package/dist/src/booking-surface/validation.js +809 -0
- package/dist/src/bridge.js +1 -1
- package/dist/src/companions.js +1 -1
- package/dist/src/contracts.js +1 -1
- package/dist/src/dsh-llm.js +1 -1
- package/dist/src/engine.js +1 -1
- package/dist/src/evaluation-cadence.js +234 -0
- package/dist/src/evaluation-contracts.js +906 -0
- package/dist/src/i18n.js +1 -1
- package/dist/src/index.js +50 -19
- package/dist/src/journey.js +1 -1
- package/dist/src/loop.js +1 -1
- package/dist/src/memory-capture.js +1 -1
- package/dist/src/memory-decay.js +1 -1
- package/dist/src/memory-utility.js +1 -1
- package/dist/src/mock-llm.js +1 -1
- package/dist/src/model.js +1 -1
- package/dist/src/realtime-pricing.js +1 -1
- package/dist/src/slot-spec.js +1 -1
- package/dist/src/state-ledger.js +2 -1
- package/dist/src/time-anchor.js +1 -1
- package/dist/src/tool-budget.js +136 -0
- package/dist/src/tool-packet.js +1 -1
- package/dist/src/travel-slots.js +1 -1
- package/dist/src/travel-timeline.js +1 -1
- package/dist/src/unified.js +1 -1
- package/dist/src/wish-pool.js +1 -1
- package/dist/src/z3-shared.js +1 -1
- package/extension/README.md +31 -7
- package/package.json +286 -11
- package/schemas/booking.surface.v1.schema.json +927 -0
- package/schemas/booking.surface.v2.schema.json +61 -0
- package/ts/capabilities/flyai.ts +16 -3
- package/ts/capabilities/session/extension-bridge.ts +37 -14
- package/ts/capabilities/session/extension-channel.ts +6 -3
- package/ts/capabilities/session/extension-distribution.ts +264 -0
- package/ts/capabilities/session/golden-score.ts +139 -0
- package/ts/capabilities/session/health-watch.ts +1 -1
- package/ts/capabilities/session/static-flight-golden.ts +209 -0
- package/ts/capabilities/session/wizard.ts +34 -176
- package/ts/capabilities/session-login.ts +15 -5
- package/ts/capabilities/session-search.ts +40 -3
- package/ts/capabilities/weather.ts +141 -52
- package/ts/package.json +3 -3
- package/ts/src/benchmark-agent-conformance.ts +448 -0
- package/ts/src/benchmark-environment-bridge.ts +348 -0
- package/ts/src/benchmark-headless-child-diagnostics.ts +184 -0
- package/ts/src/benchmark-tool-isolation.ts +166 -0
- package/ts/src/booking-surface/availability-policy-v2.ts +523 -0
- package/ts/src/booking-surface/canonical-schema.js +113 -0
- package/ts/src/booking-surface/contracts-v2.ts +118 -0
- package/ts/src/booking-surface/contracts.ts +380 -0
- package/ts/src/booking-surface/dsh-planner.ts +452 -0
- package/ts/src/booking-surface/dsh-plugin.js +93 -0
- package/ts/src/booking-surface/error-codes.ts +101 -0
- package/ts/src/booking-surface/index.ts +12 -0
- package/ts/src/booking-surface/profile.ts +42 -0
- package/ts/src/booking-surface/runtime-v2.ts +1466 -0
- package/ts/src/booking-surface/runtime.ts +483 -0
- package/ts/src/booking-surface/server-v2.ts +247 -0
- package/ts/src/booking-surface/server.ts +324 -0
- package/ts/src/booking-surface/startup.ts +196 -0
- package/ts/src/booking-surface/validation-v2.ts +205 -0
- package/ts/src/booking-surface/validation.ts +453 -0
- package/ts/src/index.ts +64 -11
- package/ts/src/state-ledger.ts +1 -0
- package/ts/src/tool-budget.ts +165 -0
- package/dist/scripts/wizard-bootstrap.js +0 -32
|
@@ -0,0 +1,2544 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import { chmodSync, lstatSync, mkdtempSync, readFileSync, rmSync, symlinkSync, writeFileSync } from 'node:fs';
|
|
3
|
+
import { tmpdir } from 'node:os';
|
|
4
|
+
import { join } from 'node:path';
|
|
5
|
+
import { Context } from '@deepseek-ai/cordis';
|
|
6
|
+
import { apply } from '../src/index.js';
|
|
7
|
+
import { registerBenchmarkEnvironmentBridge } from '../src/benchmark-environment-bridge.js';
|
|
8
|
+
import { installBenchmarkToolIsolation } from '../src/benchmark-tool-isolation.js';
|
|
9
|
+
import { BENCHMARK_BRIDGE_CALL_FAILED, BENCHMARK_BRIDGE_CALL_REQUIRED, BENCHMARK_BRIDGE_RETRY_CALL_NOT_ALLOWED, BENCHMARK_BRIDGE_OUTPUT_TRUNCATED, BENCHMARK_BRIDGE_RUNNER_FAILED, BENCHMARK_BRIDGE_SPAWN_FAILED, BENCHMARK_BRIDGE_TIMED_OUT, BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE, BENCHMARK_TERMINAL_INVALID, MAX_CONFORMANCE_RETRIES, benchmarkChildFailureForConformanceCode, createBenchmarkAgentConformance, installBenchmarkAgentConformance, parseBenchmarkTerminal, validateTerminalOutputConfig } from '../src/benchmark-agent-conformance.js';
|
|
10
|
+
import { BENCHMARK_CHILD_DIAGNOSTIC_MAX_BYTES, BENCHMARK_CHILD_DIAGNOSTIC_SCHEMA, appendBoundedChildDiagnostic, classifyBenchmarkChildFailure, classifyBenchmarkTurnEnd, createBenchmarkDiagnosticArbiter, parseBenchmarkChildDiagnostic } from '../src/benchmark-headless-child-diagnostics.js';
|
|
11
|
+
{
|
|
12
|
+
const exactFamilies = [
|
|
13
|
+
[
|
|
14
|
+
[
|
|
15
|
+
'AUTH',
|
|
16
|
+
'INVALID_CREDENTIAL',
|
|
17
|
+
'MISSING_CREDENTIAL'
|
|
18
|
+
],
|
|
19
|
+
'child_model_auth'
|
|
20
|
+
],
|
|
21
|
+
[
|
|
22
|
+
[
|
|
23
|
+
'QUOTA',
|
|
24
|
+
'RATE_LIMIT'
|
|
25
|
+
],
|
|
26
|
+
'child_model_capacity'
|
|
27
|
+
],
|
|
28
|
+
[
|
|
29
|
+
[
|
|
30
|
+
'SERVER'
|
|
31
|
+
],
|
|
32
|
+
'child_model_server'
|
|
33
|
+
],
|
|
34
|
+
[
|
|
35
|
+
[
|
|
36
|
+
'TRANSPORT',
|
|
37
|
+
'TIMEOUT'
|
|
38
|
+
],
|
|
39
|
+
'child_model_transport'
|
|
40
|
+
],
|
|
41
|
+
[
|
|
42
|
+
[
|
|
43
|
+
'EMPTY_RESPONSE',
|
|
44
|
+
'STREAM_CLOSED',
|
|
45
|
+
'MALFORMED_RESPONSE',
|
|
46
|
+
'INVALID_RESPONSE'
|
|
47
|
+
],
|
|
48
|
+
'child_model_stream'
|
|
49
|
+
],
|
|
50
|
+
[
|
|
51
|
+
[
|
|
52
|
+
'INVALID_REQUEST',
|
|
53
|
+
'CONTEXT_WINDOW_EXCEEDED',
|
|
54
|
+
'NO_ADAPTER',
|
|
55
|
+
'UNKNOWN_MODEL',
|
|
56
|
+
'UNSUPPORTED_OPTION'
|
|
57
|
+
],
|
|
58
|
+
'child_model_request'
|
|
59
|
+
],
|
|
60
|
+
[
|
|
61
|
+
[
|
|
62
|
+
'ABORTED'
|
|
63
|
+
],
|
|
64
|
+
'child_aborted'
|
|
65
|
+
]
|
|
66
|
+
];
|
|
67
|
+
for (const [codes, expected] of exactFamilies){
|
|
68
|
+
for (const code of codes)assert.equal(classifyBenchmarkTurnEnd({
|
|
69
|
+
kind: 'error',
|
|
70
|
+
error: {
|
|
71
|
+
code
|
|
72
|
+
}
|
|
73
|
+
}), expected);
|
|
74
|
+
}
|
|
75
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
76
|
+
kind: 'error',
|
|
77
|
+
error: {
|
|
78
|
+
code: 'RATE_LIMIT',
|
|
79
|
+
message: 'sentinel'
|
|
80
|
+
}
|
|
81
|
+
}), 'child_model_capacity');
|
|
82
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
83
|
+
kind: 'error',
|
|
84
|
+
error: {
|
|
85
|
+
code: 'AUTH',
|
|
86
|
+
message: 'sentinel'
|
|
87
|
+
}
|
|
88
|
+
}), 'child_model_auth');
|
|
89
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
90
|
+
kind: 'error',
|
|
91
|
+
error: {
|
|
92
|
+
code: 'MISSING_CREDENTIAL',
|
|
93
|
+
message: 'sentinel'
|
|
94
|
+
}
|
|
95
|
+
}), 'child_model_auth');
|
|
96
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
97
|
+
kind: 'error',
|
|
98
|
+
error: {
|
|
99
|
+
code: 'QUOTA',
|
|
100
|
+
message: 'sentinel'
|
|
101
|
+
}
|
|
102
|
+
}), 'child_model_capacity');
|
|
103
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
104
|
+
kind: 'error',
|
|
105
|
+
error: {
|
|
106
|
+
code: 'SERVER',
|
|
107
|
+
message: 'sentinel'
|
|
108
|
+
}
|
|
109
|
+
}), 'child_model_server');
|
|
110
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
111
|
+
kind: 'error',
|
|
112
|
+
error: {
|
|
113
|
+
code: 'TRANSPORT',
|
|
114
|
+
message: 'api-key sentinel'
|
|
115
|
+
}
|
|
116
|
+
}), 'child_model_transport');
|
|
117
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
118
|
+
kind: 'error',
|
|
119
|
+
error: {
|
|
120
|
+
code: 'OTHER',
|
|
121
|
+
status: 401,
|
|
122
|
+
message: 'key sentinel'
|
|
123
|
+
}
|
|
124
|
+
}), 'child_model_auth');
|
|
125
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
126
|
+
kind: 'error',
|
|
127
|
+
error: {
|
|
128
|
+
code: 'OTHER',
|
|
129
|
+
status: 403,
|
|
130
|
+
message: 'key sentinel'
|
|
131
|
+
}
|
|
132
|
+
}), 'child_model_auth');
|
|
133
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
134
|
+
kind: 'error',
|
|
135
|
+
error: {
|
|
136
|
+
code: 'OTHER',
|
|
137
|
+
status: 429,
|
|
138
|
+
message: 'quota sentinel'
|
|
139
|
+
}
|
|
140
|
+
}), 'child_model_capacity');
|
|
141
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
142
|
+
kind: 'error',
|
|
143
|
+
error: {
|
|
144
|
+
code: 'OTHER',
|
|
145
|
+
status: 503,
|
|
146
|
+
message: 'server sentinel'
|
|
147
|
+
}
|
|
148
|
+
}), 'child_model_server');
|
|
149
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
150
|
+
kind: 'error',
|
|
151
|
+
error: {
|
|
152
|
+
code: 'OTHER',
|
|
153
|
+
status: 500,
|
|
154
|
+
message: 'server sentinel'
|
|
155
|
+
}
|
|
156
|
+
}), 'child_model_server');
|
|
157
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
158
|
+
kind: 'error',
|
|
159
|
+
error: {
|
|
160
|
+
code: 'OTHER',
|
|
161
|
+
status: 599,
|
|
162
|
+
message: 'server sentinel'
|
|
163
|
+
}
|
|
164
|
+
}), 'child_model_server');
|
|
165
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
166
|
+
kind: 'error',
|
|
167
|
+
error: {
|
|
168
|
+
code: 'PI_AI_ERROR',
|
|
169
|
+
message: 'opaque'
|
|
170
|
+
}
|
|
171
|
+
}), 'child_runtime_error');
|
|
172
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
173
|
+
kind: 'error',
|
|
174
|
+
error: {
|
|
175
|
+
code: 'INVALID_RESPONSE',
|
|
176
|
+
message: 'opaque'
|
|
177
|
+
}
|
|
178
|
+
}), 'child_model_stream');
|
|
179
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
180
|
+
kind: 'error',
|
|
181
|
+
error: {
|
|
182
|
+
code: 'UNSUPPORTED_OPTION',
|
|
183
|
+
message: 'opaque'
|
|
184
|
+
}
|
|
185
|
+
}), 'child_model_request');
|
|
186
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
187
|
+
kind: 'error',
|
|
188
|
+
error: {
|
|
189
|
+
code: 'OTHER',
|
|
190
|
+
status: 503.5,
|
|
191
|
+
message: 'opaque'
|
|
192
|
+
}
|
|
193
|
+
}), 'child_runtime_error');
|
|
194
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
195
|
+
kind: 'error',
|
|
196
|
+
error: {
|
|
197
|
+
code: 'OTHER',
|
|
198
|
+
status: 99,
|
|
199
|
+
message: 'opaque'
|
|
200
|
+
}
|
|
201
|
+
}), 'child_runtime_error');
|
|
202
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
203
|
+
kind: 'error',
|
|
204
|
+
error: {
|
|
205
|
+
code: 'OTHER',
|
|
206
|
+
status: 600,
|
|
207
|
+
message: 'opaque'
|
|
208
|
+
}
|
|
209
|
+
}), 'child_runtime_error');
|
|
210
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
211
|
+
kind: 'error',
|
|
212
|
+
error: {
|
|
213
|
+
code: 'rate_limit',
|
|
214
|
+
message: 'opaque'
|
|
215
|
+
}
|
|
216
|
+
}), 'child_runtime_error');
|
|
217
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
218
|
+
kind: 'error',
|
|
219
|
+
error: {
|
|
220
|
+
code: 'UNKNOWN',
|
|
221
|
+
message: 'opaque'
|
|
222
|
+
}
|
|
223
|
+
}), 'child_runtime_error');
|
|
224
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
225
|
+
kind: 'error',
|
|
226
|
+
error: {
|
|
227
|
+
code: 'UNKNOWN',
|
|
228
|
+
message: 'message sentinel',
|
|
229
|
+
requestId: 'request sentinel',
|
|
230
|
+
path: 'path sentinel',
|
|
231
|
+
prompt: 'prompt sentinel',
|
|
232
|
+
key: 'key sentinel'
|
|
233
|
+
}
|
|
234
|
+
}), 'child_runtime_error');
|
|
235
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
236
|
+
kind: 'blocked'
|
|
237
|
+
}), 'child_blocked');
|
|
238
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
239
|
+
kind: 'max-tokens'
|
|
240
|
+
}), 'child_max_tokens');
|
|
241
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
242
|
+
kind: 'aborted'
|
|
243
|
+
}), 'child_aborted');
|
|
244
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
245
|
+
kind: 'interrupted'
|
|
246
|
+
}), 'child_interrupted');
|
|
247
|
+
assert.equal(classifyBenchmarkTurnEnd({
|
|
248
|
+
kind: 'completed'
|
|
249
|
+
}), undefined);
|
|
250
|
+
const writes = [];
|
|
251
|
+
const arbiter = createBenchmarkDiagnosticArbiter((code)=>writes.push(code));
|
|
252
|
+
arbiter.offer('session-a', 'child_conformance_failure');
|
|
253
|
+
arbiter.offer('session-a', 'child_runtime_error');
|
|
254
|
+
arbiter.offer('session-a', 'child_bridge_failure');
|
|
255
|
+
arbiter.offer('session-b', 'child_model_server');
|
|
256
|
+
arbiter.flush('session-a');
|
|
257
|
+
arbiter.flush('session-a');
|
|
258
|
+
arbiter.flush('session-b');
|
|
259
|
+
arbiter.flush('session-b');
|
|
260
|
+
assert.deepEqual(writes, [
|
|
261
|
+
'child_bridge_failure',
|
|
262
|
+
'child_model_server'
|
|
263
|
+
]);
|
|
264
|
+
}assert.equal(MAX_CONFORMANCE_RETRIES, 1);
|
|
265
|
+
const projection = {
|
|
266
|
+
toolName: 'gotry_benchmark_environment',
|
|
267
|
+
allowedTools: [
|
|
268
|
+
'lookup'
|
|
269
|
+
],
|
|
270
|
+
terminal: {
|
|
271
|
+
tag: 'done',
|
|
272
|
+
max_bytes: 1024
|
|
273
|
+
}
|
|
274
|
+
};
|
|
275
|
+
assert.equal(validateTerminalOutputConfig(projection.terminal), true);
|
|
276
|
+
for (const invalid of [
|
|
277
|
+
null,
|
|
278
|
+
{
|
|
279
|
+
tag: 'done'
|
|
280
|
+
},
|
|
281
|
+
{
|
|
282
|
+
tag: '1bad',
|
|
283
|
+
max_bytes: 1024
|
|
284
|
+
},
|
|
285
|
+
{
|
|
286
|
+
tag: 'done',
|
|
287
|
+
max_bytes: 0
|
|
288
|
+
},
|
|
289
|
+
{
|
|
290
|
+
tag: 'done',
|
|
291
|
+
max_bytes: 1024 * 1024 + 1
|
|
292
|
+
},
|
|
293
|
+
{
|
|
294
|
+
tag: 'done',
|
|
295
|
+
max_bytes: 1024,
|
|
296
|
+
extra: true
|
|
297
|
+
}
|
|
298
|
+
])assert.equal(validateTerminalOutputConfig(invalid), false);
|
|
299
|
+
assert.deepEqual(parseBenchmarkTerminal(' \n<done>{"ok":true}</done>\n', projection.terminal), {
|
|
300
|
+
ok: true,
|
|
301
|
+
value: {
|
|
302
|
+
ok: true
|
|
303
|
+
}
|
|
304
|
+
});
|
|
305
|
+
for (const invalid of [
|
|
306
|
+
'prose <done>{"ok":true}</done>',
|
|
307
|
+
'<done>{"ok":true}</done> trailing',
|
|
308
|
+
'<done>```json\n{"ok":true}\n```</done>',
|
|
309
|
+
'<done>{"ok":true}</done><done>{"ok":true}</done>',
|
|
310
|
+
'<done>{"value":"</done><done>"}</done>',
|
|
311
|
+
'<wrong>{"ok":true}</wrong>',
|
|
312
|
+
'<done>[{"ok":true}]</done>',
|
|
313
|
+
'<done>true</done>',
|
|
314
|
+
'<done>{"ok":</done>'
|
|
315
|
+
])assert.equal(parseBenchmarkTerminal(invalid, projection.terminal).ok, false);
|
|
316
|
+
assert.equal(parseBenchmarkTerminal(`<done>{"x":"${'y'.repeat(1024)}"}</done>`, projection.terminal).ok, false);
|
|
317
|
+
function turnStart(turn = 1) {
|
|
318
|
+
return {
|
|
319
|
+
type: 'turn/start',
|
|
320
|
+
data: {
|
|
321
|
+
turn
|
|
322
|
+
}
|
|
323
|
+
};
|
|
324
|
+
}
|
|
325
|
+
function turnEnd(turn = 1) {
|
|
326
|
+
return {
|
|
327
|
+
type: 'turn/end',
|
|
328
|
+
data: {
|
|
329
|
+
turn,
|
|
330
|
+
reason: {
|
|
331
|
+
kind: 'completed'
|
|
332
|
+
}
|
|
333
|
+
}
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
function toolCall(callId = 'call-1', options = {}) {
|
|
337
|
+
const { turn = 1, step = 1, action = 'call', tool = 'lookup' } = options;
|
|
338
|
+
return {
|
|
339
|
+
type: 'tool/call',
|
|
340
|
+
data: {
|
|
341
|
+
turn,
|
|
342
|
+
step,
|
|
343
|
+
callId,
|
|
344
|
+
name: projection.toolName,
|
|
345
|
+
arguments: JSON.stringify({
|
|
346
|
+
query: {
|
|
347
|
+
action,
|
|
348
|
+
tool,
|
|
349
|
+
arguments: {}
|
|
350
|
+
}
|
|
351
|
+
})
|
|
352
|
+
}
|
|
353
|
+
};
|
|
354
|
+
}
|
|
355
|
+
function toolResult(callId = 'call-1', options = {}) {
|
|
356
|
+
const { turn = 1, step = 1, ok = true, isError = false, error = 'runner_failed' } = options;
|
|
357
|
+
return {
|
|
358
|
+
type: 'tool/result',
|
|
359
|
+
data: {
|
|
360
|
+
turn,
|
|
361
|
+
step,
|
|
362
|
+
message: {
|
|
363
|
+
source: {
|
|
364
|
+
kind: 'tool',
|
|
365
|
+
callId
|
|
366
|
+
},
|
|
367
|
+
content: [
|
|
368
|
+
{
|
|
369
|
+
type: 'tool-result',
|
|
370
|
+
toolCallId: callId,
|
|
371
|
+
isError,
|
|
372
|
+
content: [
|
|
373
|
+
{
|
|
374
|
+
type: 'text',
|
|
375
|
+
text: JSON.stringify(ok ? {
|
|
376
|
+
ok: true,
|
|
377
|
+
result: {}
|
|
378
|
+
} : {
|
|
379
|
+
ok: false,
|
|
380
|
+
error
|
|
381
|
+
})
|
|
382
|
+
}
|
|
383
|
+
]
|
|
384
|
+
}
|
|
385
|
+
]
|
|
386
|
+
}
|
|
387
|
+
}
|
|
388
|
+
};
|
|
389
|
+
}
|
|
390
|
+
function assistant(text, options = {}) {
|
|
391
|
+
const { turn = 1, step = 2, interrupted = false } = options;
|
|
392
|
+
return {
|
|
393
|
+
type: 'assistant/message',
|
|
394
|
+
data: {
|
|
395
|
+
turn,
|
|
396
|
+
step,
|
|
397
|
+
message: {
|
|
398
|
+
content: [
|
|
399
|
+
{
|
|
400
|
+
type: 'text',
|
|
401
|
+
text
|
|
402
|
+
}
|
|
403
|
+
]
|
|
404
|
+
},
|
|
405
|
+
...interrupted ? {
|
|
406
|
+
interrupted: true
|
|
407
|
+
} : {}
|
|
408
|
+
}
|
|
409
|
+
};
|
|
410
|
+
}
|
|
411
|
+
{
|
|
412
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
413
|
+
state.observe(turnStart());
|
|
414
|
+
state.observe(assistant('I would run the CLI.', {
|
|
415
|
+
step: 1
|
|
416
|
+
}));
|
|
417
|
+
assert.deepEqual(state.stopping(1), {
|
|
418
|
+
kind: 'steer',
|
|
419
|
+
mode: 'call'
|
|
420
|
+
}, 'no-call first stop gets one correction');
|
|
421
|
+
assert.equal(state.guardBridgeExecution(), undefined, 'call correction still permits the first real bridge dispatch');
|
|
422
|
+
state.observe(toolCall('call-a', {
|
|
423
|
+
step: 2
|
|
424
|
+
}));
|
|
425
|
+
state.observe(toolResult('call-a', {
|
|
426
|
+
step: 2
|
|
427
|
+
}));
|
|
428
|
+
state.observe(assistant('<done>{"status":"succeeded"}</done>', {
|
|
429
|
+
step: 3
|
|
430
|
+
}));
|
|
431
|
+
assert.deepEqual(state.stopping(1), {
|
|
432
|
+
kind: 'accept'
|
|
433
|
+
}, 'call correction may converge to one successful terminal');
|
|
434
|
+
}{
|
|
435
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
436
|
+
state.observe(turnStart());
|
|
437
|
+
state.observe(assistant('<done>{"status":"succeeded"}</done>', {
|
|
438
|
+
step: 1
|
|
439
|
+
}));
|
|
440
|
+
assert.deepEqual(state.stopping(1), {
|
|
441
|
+
kind: 'steer',
|
|
442
|
+
mode: 'call'
|
|
443
|
+
}, 'valid terminal without a call still needs a call');
|
|
444
|
+
state.observe(assistant('<done>{"status":"succeeded"}</done>', {
|
|
445
|
+
step: 2
|
|
446
|
+
}));
|
|
447
|
+
assert.deepEqual(state.stopping(1), {
|
|
448
|
+
kind: 'reject',
|
|
449
|
+
code: BENCHMARK_BRIDGE_CALL_REQUIRED
|
|
450
|
+
});
|
|
451
|
+
}{
|
|
452
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
453
|
+
state.observe(turnStart());
|
|
454
|
+
state.observe(toolCall());
|
|
455
|
+
state.observe(toolResult());
|
|
456
|
+
state.observe(assistant('bad terminal'));
|
|
457
|
+
assert.deepEqual(state.stopping(1), {
|
|
458
|
+
kind: 'steer',
|
|
459
|
+
mode: 'terminal'
|
|
460
|
+
}, 'bad terminal gets one format-only correction');
|
|
461
|
+
assert.equal(state.guardBridgeExecution(), BENCHMARK_BRIDGE_RETRY_CALL_NOT_ALLOWED);
|
|
462
|
+
state.observe(assistant('<done>{"status":"succeeded"}</done>', {
|
|
463
|
+
step: 3
|
|
464
|
+
}));
|
|
465
|
+
assert.deepEqual(state.stopping(1), {
|
|
466
|
+
kind: 'accept'
|
|
467
|
+
}, 'format-only correction can reuse the successful result');
|
|
468
|
+
}{
|
|
469
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
470
|
+
state.observe(turnStart());
|
|
471
|
+
state.observe(toolCall());
|
|
472
|
+
state.observe(toolResult());
|
|
473
|
+
state.observe(assistant('bad terminal'));
|
|
474
|
+
assert.deepEqual(state.stopping(1), {
|
|
475
|
+
kind: 'steer',
|
|
476
|
+
mode: 'terminal'
|
|
477
|
+
});
|
|
478
|
+
state.observe(toolCall('call-2', {
|
|
479
|
+
step: 3
|
|
480
|
+
}));
|
|
481
|
+
assert.deepEqual(state.stopping(1), {
|
|
482
|
+
kind: 'reject',
|
|
483
|
+
code: BENCHMARK_BRIDGE_RETRY_CALL_NOT_ALLOWED
|
|
484
|
+
});
|
|
485
|
+
}{
|
|
486
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
487
|
+
state.observe(turnStart());
|
|
488
|
+
state.observe(toolCall());
|
|
489
|
+
state.observe(toolResult());
|
|
490
|
+
state.observe(assistant('bad terminal'));
|
|
491
|
+
assert.deepEqual(state.stopping(1), {
|
|
492
|
+
kind: 'steer',
|
|
493
|
+
mode: 'terminal'
|
|
494
|
+
});
|
|
495
|
+
state.observe(assistant('still bad', {
|
|
496
|
+
step: 3
|
|
497
|
+
}));
|
|
498
|
+
assert.deepEqual(state.stopping(1), {
|
|
499
|
+
kind: 'reject',
|
|
500
|
+
code: BENCHMARK_TERMINAL_INVALID
|
|
501
|
+
});
|
|
502
|
+
}{
|
|
503
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
504
|
+
state.observe(turnStart());
|
|
505
|
+
state.observe(toolCall());
|
|
506
|
+
state.observe(toolResult('call-1', {
|
|
507
|
+
ok: false
|
|
508
|
+
}));
|
|
509
|
+
assert.deepEqual(state.stopping(1), {
|
|
510
|
+
kind: 'reject',
|
|
511
|
+
code: BENCHMARK_BRIDGE_RUNNER_FAILED
|
|
512
|
+
}, 'structured runner failure is not retried');
|
|
513
|
+
}{
|
|
514
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
515
|
+
state.observe(turnStart());
|
|
516
|
+
state.observe(toolCall());
|
|
517
|
+
state.observe(toolResult('call-1', {
|
|
518
|
+
ok: false,
|
|
519
|
+
error: 'output_truncated'
|
|
520
|
+
}));
|
|
521
|
+
assert.deepEqual(state.stopping(1), {
|
|
522
|
+
kind: 'reject',
|
|
523
|
+
code: BENCHMARK_BRIDGE_OUTPUT_TRUNCATED
|
|
524
|
+
}, 'structured runner truncation has a distinct reason');
|
|
525
|
+
}{
|
|
526
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
527
|
+
state.observe(turnStart());
|
|
528
|
+
state.observe(toolCall('failed', {
|
|
529
|
+
step: 1
|
|
530
|
+
}));
|
|
531
|
+
state.observe(toolResult('failed', {
|
|
532
|
+
step: 1,
|
|
533
|
+
ok: false
|
|
534
|
+
}));
|
|
535
|
+
state.observe(toolCall('successful', {
|
|
536
|
+
step: 2
|
|
537
|
+
}));
|
|
538
|
+
state.observe(toolResult('successful', {
|
|
539
|
+
step: 2
|
|
540
|
+
}));
|
|
541
|
+
state.observe(assistant('<done>{"status":"succeeded"}</done>', {
|
|
542
|
+
step: 3
|
|
543
|
+
}));
|
|
544
|
+
assert.deepEqual(state.stopping(1), {
|
|
545
|
+
kind: 'accept'
|
|
546
|
+
}, 'a model-owned later success can recover from an earlier failed call without a conformance retry');
|
|
547
|
+
}{
|
|
548
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
549
|
+
state.observe(turnStart());
|
|
550
|
+
state.observe(toolCall('successful', {
|
|
551
|
+
step: 1
|
|
552
|
+
}));
|
|
553
|
+
state.observe(toolResult('successful', {
|
|
554
|
+
step: 1
|
|
555
|
+
}));
|
|
556
|
+
state.observe(toolCall('failed', {
|
|
557
|
+
step: 2
|
|
558
|
+
}));
|
|
559
|
+
state.observe(toolResult('failed', {
|
|
560
|
+
step: 2,
|
|
561
|
+
ok: false
|
|
562
|
+
}));
|
|
563
|
+
state.observe(assistant('<done>{"status":"succeeded"}</done>', {
|
|
564
|
+
step: 3
|
|
565
|
+
}));
|
|
566
|
+
assert.deepEqual(state.stopping(1), {
|
|
567
|
+
kind: 'accept'
|
|
568
|
+
}, 'a later failed optional call does not erase an already paired successful result');
|
|
569
|
+
}{
|
|
570
|
+
const state = createBenchmarkAgentConformance(projection);
|
|
571
|
+
state.observe(turnStart());
|
|
572
|
+
state.observe(toolCall('discovery', {
|
|
573
|
+
action: 'tools'
|
|
574
|
+
}));
|
|
575
|
+
state.observe(toolResult('discovery'));
|
|
576
|
+
state.observe(assistant('<done>{"status":"succeeded"}</done>'));
|
|
577
|
+
assert.deepEqual(state.stopping(1), {
|
|
578
|
+
kind: 'steer',
|
|
579
|
+
mode: 'call'
|
|
580
|
+
}, 'action tools does not satisfy the call gate');
|
|
581
|
+
state.observe(turnEnd());
|
|
582
|
+
assert.deepEqual(state.stopping(1), {
|
|
583
|
+
kind: 'reject',
|
|
584
|
+
code: BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE
|
|
585
|
+
});
|
|
586
|
+
state.observe(turnStart(2));
|
|
587
|
+
assert.deepEqual(state.stopping(1), {
|
|
588
|
+
kind: 'reject',
|
|
589
|
+
code: BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE
|
|
590
|
+
}, 'turn state cannot leak across turns');
|
|
591
|
+
}{
|
|
592
|
+
const rootListeners = new Map();
|
|
593
|
+
const scopedListeners = new Map();
|
|
594
|
+
const guards = [];
|
|
595
|
+
const steers = [];
|
|
596
|
+
const runtimeWrites = [];
|
|
597
|
+
const runEffect = (action)=>{
|
|
598
|
+
const disposers = [];
|
|
599
|
+
const value = action();
|
|
600
|
+
if (value && typeof value.next === 'function') {
|
|
601
|
+
let item = value.next();
|
|
602
|
+
while(!item.done){
|
|
603
|
+
if (typeof item.value === 'function') disposers.push(item.value);
|
|
604
|
+
item = value.next();
|
|
605
|
+
}
|
|
606
|
+
}
|
|
607
|
+
return ()=>{
|
|
608
|
+
for (const dispose of disposers.reverse())dispose();
|
|
609
|
+
};
|
|
610
|
+
};
|
|
611
|
+
const add = (target, name, listener)=>{
|
|
612
|
+
const list = target.get(name) ?? [];
|
|
613
|
+
list.push(listener);
|
|
614
|
+
target.set(name, list);
|
|
615
|
+
return ()=>target.set(name, list.filter((candidate)=>candidate !== listener));
|
|
616
|
+
};
|
|
617
|
+
const session = {};
|
|
618
|
+
const agent = {
|
|
619
|
+
session,
|
|
620
|
+
steer (message) {
|
|
621
|
+
steers.push(message);
|
|
622
|
+
},
|
|
623
|
+
ctx: {
|
|
624
|
+
tools: {
|
|
625
|
+
guard (check) {
|
|
626
|
+
guards.push(check);
|
|
627
|
+
return ()=>{};
|
|
628
|
+
}
|
|
629
|
+
},
|
|
630
|
+
effect: runEffect,
|
|
631
|
+
on (name, listener) {
|
|
632
|
+
return add(scopedListeners, name, listener);
|
|
633
|
+
}
|
|
634
|
+
}
|
|
635
|
+
};
|
|
636
|
+
const ctx = {
|
|
637
|
+
on (name, listener) {
|
|
638
|
+
return add(rootListeners, name, listener);
|
|
639
|
+
}
|
|
640
|
+
};
|
|
641
|
+
installBenchmarkAgentConformance(ctx, projection, (code)=>runtimeWrites.push(code));
|
|
642
|
+
rootListeners.get('agent/created')[0]({
|
|
643
|
+
agent
|
|
644
|
+
});
|
|
645
|
+
rootListeners.get('session/event')[0](session, turnStart());
|
|
646
|
+
rootListeners.get('session/event')[0](session, {
|
|
647
|
+
type: 'llm/retry',
|
|
648
|
+
data: {
|
|
649
|
+
turn: 1,
|
|
650
|
+
reason: {
|
|
651
|
+
kind: 'error',
|
|
652
|
+
error: {
|
|
653
|
+
code: 'RATE_LIMIT'
|
|
654
|
+
}
|
|
655
|
+
}
|
|
656
|
+
}
|
|
657
|
+
});
|
|
658
|
+
rootListeners.get('session/event')[0](session, {
|
|
659
|
+
type: 'agent/request-error',
|
|
660
|
+
data: {
|
|
661
|
+
turn: 1,
|
|
662
|
+
error: {
|
|
663
|
+
code: 'SERVER'
|
|
664
|
+
}
|
|
665
|
+
}
|
|
666
|
+
});
|
|
667
|
+
rootListeners.get('session/event')[0](session, assistant('prose only', {
|
|
668
|
+
step: 1
|
|
669
|
+
}));
|
|
670
|
+
rootListeners.get('agent/turn-stopping')[0]({
|
|
671
|
+
agent,
|
|
672
|
+
turn: 1
|
|
673
|
+
});
|
|
674
|
+
assert.equal(steers.length, 1, 'runtime wiring steers exactly once at the stop boundary');
|
|
675
|
+
assert.equal(steers[0].role, 'user');
|
|
676
|
+
assert.equal(Object.isFrozen(steers[0]), true, 'correction uses the official immutable DSH user message');
|
|
677
|
+
assert.equal(Object.isFrozen(steers[0].content), true, 'correction content is deeply frozen');
|
|
678
|
+
assert.equal(guards[0]({
|
|
679
|
+
name: projection.toolName
|
|
680
|
+
}), undefined, 'call correction leaves bridge execution available');
|
|
681
|
+
const assembled = await scopedListeners.get('system-prompt/assemble')[0]({}, {}, async ()=>({
|
|
682
|
+
sections: [],
|
|
683
|
+
tools: []
|
|
684
|
+
}));
|
|
685
|
+
assert.match(assembled.sections[0].text, /agent_env\.cli/);
|
|
686
|
+
assert.match(assembled.sections[0].text, /\"action\":\"call\"/);
|
|
687
|
+
assert.match(assembled.sections[0].text, /<done>/);
|
|
688
|
+
assert.equal(assembled.sections[0].text.includes('/tmp/'), false);
|
|
689
|
+
rootListeners.get('session/event')[0](session, turnEnd());
|
|
690
|
+
assert.deepEqual(runtimeWrites, []);
|
|
691
|
+
rootListeners.get('session/event')[0](session, turnStart(2));
|
|
692
|
+
rootListeners.get('session/event')[0](session, {
|
|
693
|
+
type: 'turn/end',
|
|
694
|
+
data: {
|
|
695
|
+
turn: 2,
|
|
696
|
+
reason: {
|
|
697
|
+
kind: 'error',
|
|
698
|
+
error: {
|
|
699
|
+
code: 'SERVER',
|
|
700
|
+
message: 'sentinel'
|
|
701
|
+
}
|
|
702
|
+
}
|
|
703
|
+
}
|
|
704
|
+
});
|
|
705
|
+
rootListeners.get('session/event')[0](session, {
|
|
706
|
+
type: 'turn/end',
|
|
707
|
+
data: {
|
|
708
|
+
turn: 2,
|
|
709
|
+
reason: {
|
|
710
|
+
kind: 'error',
|
|
711
|
+
error: {
|
|
712
|
+
code: 'UNKNOWN_MODEL'
|
|
713
|
+
}
|
|
714
|
+
}
|
|
715
|
+
}
|
|
716
|
+
});
|
|
717
|
+
assert.deepEqual(runtimeWrites, [
|
|
718
|
+
'child_model_server'
|
|
719
|
+
]);
|
|
720
|
+
rootListeners.get('session/disposed')[0](session);
|
|
721
|
+
const session2 = {};
|
|
722
|
+
const agent2 = {
|
|
723
|
+
session: session2,
|
|
724
|
+
steer () {},
|
|
725
|
+
ctx: agent.ctx
|
|
726
|
+
};
|
|
727
|
+
rootListeners.get('agent/created')[0]({
|
|
728
|
+
agent: agent2
|
|
729
|
+
});
|
|
730
|
+
rootListeners.get('session/event')[0](session2, turnStart());
|
|
731
|
+
assert.throws(()=>rootListeners.get('agent/turn-stopping')[0]({
|
|
732
|
+
agent: agent2
|
|
733
|
+
}), new RegExp(BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE));
|
|
734
|
+
rootListeners.get('session/event')[0](session2, {
|
|
735
|
+
type: 'turn/end',
|
|
736
|
+
data: {
|
|
737
|
+
turn: 1,
|
|
738
|
+
reason: {
|
|
739
|
+
kind: 'error',
|
|
740
|
+
error: {
|
|
741
|
+
code: 'UNKNOWN'
|
|
742
|
+
}
|
|
743
|
+
}
|
|
744
|
+
}
|
|
745
|
+
});
|
|
746
|
+
assert.deepEqual(runtimeWrites, [
|
|
747
|
+
'child_model_server',
|
|
748
|
+
'child_conformance_failure'
|
|
749
|
+
]);
|
|
750
|
+
rootListeners.get('session/disposed')[0](session2);
|
|
751
|
+
const session3 = {};
|
|
752
|
+
const agent3 = {
|
|
753
|
+
session: session3,
|
|
754
|
+
steer () {},
|
|
755
|
+
ctx: agent.ctx
|
|
756
|
+
};
|
|
757
|
+
rootListeners.get('agent/created')[0]({
|
|
758
|
+
agent: agent3
|
|
759
|
+
});
|
|
760
|
+
rootListeners.get('session/event')[0](session3, turnStart());
|
|
761
|
+
rootListeners.get('session/event')[0](session3, {
|
|
762
|
+
type: 'turn/end',
|
|
763
|
+
data: {
|
|
764
|
+
turn: 1,
|
|
765
|
+
reason: {
|
|
766
|
+
kind: 'error',
|
|
767
|
+
error: {
|
|
768
|
+
code: 'TIMEOUT'
|
|
769
|
+
}
|
|
770
|
+
}
|
|
771
|
+
}
|
|
772
|
+
});
|
|
773
|
+
assert.deepEqual(runtimeWrites, [
|
|
774
|
+
'child_model_server',
|
|
775
|
+
'child_conformance_failure',
|
|
776
|
+
'child_model_transport'
|
|
777
|
+
], 'session diagnostics remain isolated');
|
|
778
|
+
rootListeners.get('session/disposed')[0](session3);
|
|
779
|
+
}function fakeHandle(outcome) {
|
|
780
|
+
const stdout = outcome.stdout ?? '';
|
|
781
|
+
const stderr = outcome.stderr ?? '';
|
|
782
|
+
const reader = {
|
|
783
|
+
readFrom: (_offset)=>({
|
|
784
|
+
text: stdout,
|
|
785
|
+
nextOffset: Buffer.byteLength(stdout),
|
|
786
|
+
lossy: outcome.lossy ?? false
|
|
787
|
+
})
|
|
788
|
+
};
|
|
789
|
+
const errorReader = {
|
|
790
|
+
readFrom: (_offset)=>({
|
|
791
|
+
text: stderr,
|
|
792
|
+
nextOffset: Buffer.byteLength(stderr),
|
|
793
|
+
lossy: false
|
|
794
|
+
})
|
|
795
|
+
};
|
|
796
|
+
let rejectDone;
|
|
797
|
+
const done = outcome.spawnReject ? Promise.reject(new Error('spawn rejected')) : outcome.waitForAbort ? new Promise((_resolve, reject)=>{
|
|
798
|
+
rejectDone = reject;
|
|
799
|
+
}) : Promise.resolve({
|
|
800
|
+
exitCode: outcome.exitCode ?? 0,
|
|
801
|
+
signal: outcome.signal ?? null
|
|
802
|
+
});
|
|
803
|
+
return {
|
|
804
|
+
pid: outcome.spawnReject ? -1 : 4242,
|
|
805
|
+
stdin: undefined,
|
|
806
|
+
stdout: undefined,
|
|
807
|
+
stderr: undefined,
|
|
808
|
+
collected: {
|
|
809
|
+
stdout: reader,
|
|
810
|
+
stderr: errorReader
|
|
811
|
+
},
|
|
812
|
+
done,
|
|
813
|
+
terminate () {
|
|
814
|
+
if (outcome.waitForAbort) rejectDone?.(new Error('timed out'));
|
|
815
|
+
},
|
|
816
|
+
waitForExit: async ()=>true
|
|
817
|
+
};
|
|
818
|
+
}
|
|
819
|
+
async function assertRealCordisWaterfallOrdering() {
|
|
820
|
+
const ctx = new Context();
|
|
821
|
+
const bridge = {
|
|
822
|
+
name: 'gotry_benchmark_environment',
|
|
823
|
+
description: 'benchmark bridge',
|
|
824
|
+
parameters: {
|
|
825
|
+
query: {
|
|
826
|
+
type: 'json',
|
|
827
|
+
required: true
|
|
828
|
+
}
|
|
829
|
+
}
|
|
830
|
+
};
|
|
831
|
+
const exactSchema = structuredClone(bridge);
|
|
832
|
+
let addPreStepTool = false;
|
|
833
|
+
const rootTools = {
|
|
834
|
+
get (name) {
|
|
835
|
+
return name === bridge.name ? bridge : undefined;
|
|
836
|
+
},
|
|
837
|
+
schemas (agent) {
|
|
838
|
+
return agent && addPreStepTool ? [
|
|
839
|
+
exactSchema,
|
|
840
|
+
{
|
|
841
|
+
name: 'non_bridge'
|
|
842
|
+
}
|
|
843
|
+
] : [
|
|
844
|
+
exactSchema
|
|
845
|
+
];
|
|
846
|
+
}
|
|
847
|
+
};
|
|
848
|
+
ctx.provide('tools', rootTools);
|
|
849
|
+
ctx.provide('agents', {
|
|
850
|
+
list: ()=>[]
|
|
851
|
+
});
|
|
852
|
+
const bus = ctx;
|
|
853
|
+
bus.on('system-prompt/assemble', async (_assembly, _context, next)=>{
|
|
854
|
+
const result = await next();
|
|
855
|
+
return {
|
|
856
|
+
...result,
|
|
857
|
+
tools: [
|
|
858
|
+
...result.tools,
|
|
859
|
+
{
|
|
860
|
+
name: 'non_bridge'
|
|
861
|
+
}
|
|
862
|
+
]
|
|
863
|
+
};
|
|
864
|
+
});
|
|
865
|
+
bus.on('agent/pre-step', async (_payload, next)=>{
|
|
866
|
+
const result = await next();
|
|
867
|
+
addPreStepTool = true;
|
|
868
|
+
return result;
|
|
869
|
+
});
|
|
870
|
+
installBenchmarkToolIsolation(ctx);
|
|
871
|
+
const scopedEffect = (action, label)=>ctx.effect(action, label);
|
|
872
|
+
const scopedTools = {
|
|
873
|
+
guard: ()=>ctx.effect(()=>()=>undefined),
|
|
874
|
+
presentAs: ()=>ctx.effect(()=>()=>undefined),
|
|
875
|
+
restrict: ()=>ctx.effect(()=>()=>undefined)
|
|
876
|
+
};
|
|
877
|
+
const agent = {
|
|
878
|
+
ctx: {
|
|
879
|
+
tools: scopedTools,
|
|
880
|
+
effect: scopedEffect,
|
|
881
|
+
on: bus.on
|
|
882
|
+
}
|
|
883
|
+
};
|
|
884
|
+
bus.emit('agent/created', {
|
|
885
|
+
agent
|
|
886
|
+
});
|
|
887
|
+
await assert.rejects(bus.waterfall('system-prompt/assemble', {
|
|
888
|
+
tools: []
|
|
889
|
+
}, {
|
|
890
|
+
agent,
|
|
891
|
+
scope: agent
|
|
892
|
+
}, async ()=>({
|
|
893
|
+
tools: [
|
|
894
|
+
exactSchema
|
|
895
|
+
]
|
|
896
|
+
})), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'prepend assembly guard observes an earlier listener post-next mutation');
|
|
897
|
+
await assert.rejects(bus.waterfall('agent/pre-step', {
|
|
898
|
+
agent
|
|
899
|
+
}, async ()=>({
|
|
900
|
+
kind: 'enter'
|
|
901
|
+
})), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'prepend pre-step guard observes an earlier listener post-next scope mutation');
|
|
902
|
+
await ctx.fiber.dispose();
|
|
903
|
+
}
|
|
904
|
+
await assertRealCordisWaterfallOrdering();
|
|
905
|
+
const root = mkdtempSync(join(tmpdir(), 'gotry-benchmark-bridge-test-'));
|
|
906
|
+
const ambientSentinelNames = [
|
|
907
|
+
'GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL',
|
|
908
|
+
'GOTRY_LLM_MODEL',
|
|
909
|
+
'DATABASE_URL',
|
|
910
|
+
'SSH_AUTH_SOCK',
|
|
911
|
+
'AWS_PROFILE',
|
|
912
|
+
'HTTPS_PROXY'
|
|
913
|
+
];
|
|
914
|
+
const ambientSentinels = new Map(ambientSentinelNames.map((name)=>[
|
|
915
|
+
name,
|
|
916
|
+
process.env[name]
|
|
917
|
+
]));
|
|
918
|
+
try {
|
|
919
|
+
delete process.env.GOTRY_LLM_MODEL;
|
|
920
|
+
const timedOutDiagnostic = '\n' + JSON.stringify({
|
|
921
|
+
schema_version: BENCHMARK_CHILD_DIAGNOSTIC_SCHEMA,
|
|
922
|
+
code: 'child_bridge_timed_out'
|
|
923
|
+
}) + '\n';
|
|
924
|
+
assert.equal(parseBenchmarkChildDiagnostic(timedOutDiagnostic), 'child_bridge_timed_out', 'strict control record parses to its allowlisted reason code');
|
|
925
|
+
assert.equal(classifyBenchmarkChildFailure({
|
|
926
|
+
code: 1,
|
|
927
|
+
diagnostic: timedOutDiagnostic
|
|
928
|
+
}), 'child_bridge_timed_out', 'structured control data maps to a stable bridge timeout code');
|
|
929
|
+
assert.equal(classifyBenchmarkChildFailure({
|
|
930
|
+
code: 1,
|
|
931
|
+
diagnostic: `provider text contains child_bridge_timed_out and ${BENCHMARK_BRIDGE_TIMED_OUT}`
|
|
932
|
+
}), 'child_nonzero_exit', 'free text cannot impersonate a structured bridge reason');
|
|
933
|
+
assert.equal(parseBenchmarkChildDiagnostic(JSON.stringify({
|
|
934
|
+
schema_version: BENCHMARK_CHILD_DIAGNOSTIC_SCHEMA,
|
|
935
|
+
code: 'child_bridge_timed_out',
|
|
936
|
+
extra: 'rejected'
|
|
937
|
+
})), undefined, 'control records with extra keys fail closed');
|
|
938
|
+
assert.equal(benchmarkChildFailureForConformanceCode(BENCHMARK_BRIDGE_RUNNER_FAILED), 'child_bridge_runner_failed', 'runner failure maps to a stable structured child reason');
|
|
939
|
+
assert.equal(benchmarkChildFailureForConformanceCode(BENCHMARK_BRIDGE_SPAWN_FAILED), 'child_bridge_spawn_failed', 'spawn failure maps to a stable structured child reason');
|
|
940
|
+
assert.equal(benchmarkChildFailureForConformanceCode(BENCHMARK_BRIDGE_OUTPUT_TRUNCATED), 'child_bridge_output_truncated', 'runner output truncation maps to a stable structured child reason');
|
|
941
|
+
assert.equal(classifyBenchmarkChildFailure({
|
|
942
|
+
code: 0,
|
|
943
|
+
outputTruncated: true
|
|
944
|
+
}), 'child_output_truncated', 'truncated terminal output takes precedence over control data');
|
|
945
|
+
assert.equal(classifyBenchmarkChildFailure({
|
|
946
|
+
code: null,
|
|
947
|
+
signal: 'SIGTERM',
|
|
948
|
+
diagnostic: timedOutDiagnostic
|
|
949
|
+
}), 'child_signaled', 'outer child signal takes precedence over an inner bridge diagnostic');
|
|
950
|
+
assert.equal(classifyBenchmarkChildFailure({
|
|
951
|
+
code: 0,
|
|
952
|
+
diagnostic: ''
|
|
953
|
+
}), 'child_lifecycle_failure', 'unexpected zero-exit diagnostic path remains a stable lifecycle failure');
|
|
954
|
+
const noisyPrefix = Buffer.alloc(BENCHMARK_CHILD_DIAGNOSTIC_MAX_BYTES + 10, 'x');
|
|
955
|
+
const boundedControl = appendBoundedChildDiagnostic(appendBoundedChildDiagnostic(Buffer.alloc(0), noisyPrefix), timedOutDiagnostic);
|
|
956
|
+
assert.equal(boundedControl.length, BENCHMARK_CHILD_DIAGNOSTIC_MAX_BYTES, 'control capture is bounded');
|
|
957
|
+
assert.equal(parseBenchmarkChildDiagnostic(boundedControl.toString('utf8')), 'child_bridge_timed_out', 'rolling tail preserves a final structured reason after bounded noise');
|
|
958
|
+
const configPath = join(root, 'bridge.json');
|
|
959
|
+
writeFileSync(configPath, JSON.stringify({
|
|
960
|
+
schema_version: 'gotry_benchmark_environment_bridge_v2',
|
|
961
|
+
enabled: true,
|
|
962
|
+
executable: process.execPath,
|
|
963
|
+
cwd: root,
|
|
964
|
+
argv_prefix: [
|
|
965
|
+
'-m',
|
|
966
|
+
'agent_env.cli',
|
|
967
|
+
'--lang',
|
|
968
|
+
'en'
|
|
969
|
+
],
|
|
970
|
+
allowed_tools: [
|
|
971
|
+
'lookup',
|
|
972
|
+
'constructor',
|
|
973
|
+
'toString'
|
|
974
|
+
],
|
|
975
|
+
allowed_output_keys: {
|
|
976
|
+
lookup: [
|
|
977
|
+
'city',
|
|
978
|
+
'nested'
|
|
979
|
+
],
|
|
980
|
+
constructor: [
|
|
981
|
+
'legacy'
|
|
982
|
+
]
|
|
983
|
+
},
|
|
984
|
+
timeout_ms: 20,
|
|
985
|
+
max_output_bytes: 4_096,
|
|
986
|
+
terminal_output: {
|
|
987
|
+
tag: 'done',
|
|
988
|
+
max_bytes: 4_096
|
|
989
|
+
},
|
|
990
|
+
isolation: {
|
|
991
|
+
mode: 'host-enforced',
|
|
992
|
+
writes: 'forbidden',
|
|
993
|
+
network: 'denied'
|
|
994
|
+
}
|
|
995
|
+
}));
|
|
996
|
+
const registered = [];
|
|
997
|
+
let visibleBridge;
|
|
998
|
+
let shadowedAgent;
|
|
999
|
+
const scopedExtraSchemas = new Map();
|
|
1000
|
+
const spawnSpecs = [];
|
|
1001
|
+
const agentCreatedListeners = [];
|
|
1002
|
+
const eventNames = [];
|
|
1003
|
+
const promptVariables = [];
|
|
1004
|
+
const assemblyListeners = [];
|
|
1005
|
+
const preStepListeners = [];
|
|
1006
|
+
const disposedListeners = [];
|
|
1007
|
+
const eventOptions = new Map();
|
|
1008
|
+
const runEffect = (action)=>{
|
|
1009
|
+
const yielded = [];
|
|
1010
|
+
const value = action();
|
|
1011
|
+
if (value && typeof value.next === 'function') {
|
|
1012
|
+
let step = value.next();
|
|
1013
|
+
while(!step.done){
|
|
1014
|
+
if (typeof step.value === 'function') yielded.push(step.value);
|
|
1015
|
+
step = value.next();
|
|
1016
|
+
}
|
|
1017
|
+
}
|
|
1018
|
+
let active = true;
|
|
1019
|
+
return async ()=>{
|
|
1020
|
+
if (!active) return;
|
|
1021
|
+
active = false;
|
|
1022
|
+
for (const dispose of yielded.reverse())await dispose();
|
|
1023
|
+
};
|
|
1024
|
+
};
|
|
1025
|
+
let disposeRootIsolation;
|
|
1026
|
+
let timeoutSignal;
|
|
1027
|
+
const outcomes = [
|
|
1028
|
+
{
|
|
1029
|
+
stdout: '{"result":[{"city":"Dubai"}]}'
|
|
1030
|
+
}
|
|
1031
|
+
];
|
|
1032
|
+
process.env.GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL = 'must-not-cross-boundary';
|
|
1033
|
+
process.env.DATABASE_URL = 'postgres://secret';
|
|
1034
|
+
process.env.SSH_AUTH_SOCK = '/tmp/secret.sock';
|
|
1035
|
+
process.env.AWS_PROFILE = 'secret-profile';
|
|
1036
|
+
process.env.HTTPS_PROXY = 'https://secret-proxy';
|
|
1037
|
+
const ctx = {
|
|
1038
|
+
tools: {
|
|
1039
|
+
register (tool) {
|
|
1040
|
+
registered.push(tool);
|
|
1041
|
+
return ()=>{};
|
|
1042
|
+
},
|
|
1043
|
+
get (name, agent) {
|
|
1044
|
+
return name === 'gotry_benchmark_environment' ? agent !== undefined && agent === shadowedAgent ? {
|
|
1045
|
+
name
|
|
1046
|
+
} : registered.find((tool)=>tool.name === name) : undefined;
|
|
1047
|
+
},
|
|
1048
|
+
schemas (agent) {
|
|
1049
|
+
const project = (tool)=>({
|
|
1050
|
+
name: tool.name,
|
|
1051
|
+
description: tool.description,
|
|
1052
|
+
parameters: structuredClone(tool.parameters)
|
|
1053
|
+
});
|
|
1054
|
+
const schemas = registered.map(project);
|
|
1055
|
+
if (agent === undefined) return schemas;
|
|
1056
|
+
return [
|
|
1057
|
+
...schemas.filter((schema)=>schema.name === 'gotry_benchmark_environment'),
|
|
1058
|
+
...scopedExtraSchemas.get(agent) ?? []
|
|
1059
|
+
];
|
|
1060
|
+
}
|
|
1061
|
+
},
|
|
1062
|
+
systemPrompt: {
|
|
1063
|
+
variable (name) {
|
|
1064
|
+
promptVariables.push(name);
|
|
1065
|
+
}
|
|
1066
|
+
},
|
|
1067
|
+
on (event, listener, options) {
|
|
1068
|
+
eventNames.push(event);
|
|
1069
|
+
eventOptions.set(event, options);
|
|
1070
|
+
if (event === 'agent/created') agentCreatedListeners.push(listener);
|
|
1071
|
+
if (event === 'agent/disposed') disposedListeners.push(listener);
|
|
1072
|
+
if (event === 'system-prompt/assemble') assemblyListeners.push(listener);
|
|
1073
|
+
if (event === 'agent/pre-step') preStepListeners.push(listener);
|
|
1074
|
+
return ()=>{};
|
|
1075
|
+
},
|
|
1076
|
+
effect (action, label) {
|
|
1077
|
+
const dispose = runEffect(action);
|
|
1078
|
+
if (label === 'benchmark-environment-tool-isolation') disposeRootIsolation = dispose;
|
|
1079
|
+
return dispose;
|
|
1080
|
+
},
|
|
1081
|
+
get (name) {
|
|
1082
|
+
if (name === 'subprocess') return this.subprocess;
|
|
1083
|
+
if (name === 'agents') return this.agents;
|
|
1084
|
+
},
|
|
1085
|
+
subprocess: {
|
|
1086
|
+
spawn (spec) {
|
|
1087
|
+
spawnSpecs.push(spec);
|
|
1088
|
+
const outcome = outcomes.shift() ?? {
|
|
1089
|
+
stdout: '{"result":{}}'
|
|
1090
|
+
};
|
|
1091
|
+
if (outcome.spawnError) throw new Error('fake spawn failed');
|
|
1092
|
+
if (outcome.waitForAbort) {
|
|
1093
|
+
timeoutSignal = spec.signal;
|
|
1094
|
+
const handle = fakeHandle(outcome);
|
|
1095
|
+
spec.signal?.addEventListener('abort', ()=>handle.terminate?.(), {
|
|
1096
|
+
once: true
|
|
1097
|
+
});
|
|
1098
|
+
return handle;
|
|
1099
|
+
}
|
|
1100
|
+
return fakeHandle(outcome);
|
|
1101
|
+
}
|
|
1102
|
+
},
|
|
1103
|
+
agents: {
|
|
1104
|
+
list () {
|
|
1105
|
+
return [];
|
|
1106
|
+
}
|
|
1107
|
+
}
|
|
1108
|
+
};
|
|
1109
|
+
const coldStartListeners = [];
|
|
1110
|
+
const coldStartRoot = {
|
|
1111
|
+
name: 'gotry_benchmark_environment'
|
|
1112
|
+
};
|
|
1113
|
+
assert.throws(()=>installBenchmarkToolIsolation({
|
|
1114
|
+
tools: {
|
|
1115
|
+
get (name) {
|
|
1116
|
+
return name === 'gotry_benchmark_environment' ? coldStartRoot : undefined;
|
|
1117
|
+
},
|
|
1118
|
+
schemas () {
|
|
1119
|
+
return [
|
|
1120
|
+
{
|
|
1121
|
+
name: 'gotry_benchmark_environment'
|
|
1122
|
+
}
|
|
1123
|
+
];
|
|
1124
|
+
}
|
|
1125
|
+
},
|
|
1126
|
+
agents: {
|
|
1127
|
+
list () {
|
|
1128
|
+
return [
|
|
1129
|
+
{
|
|
1130
|
+
id: 'already-live'
|
|
1131
|
+
}
|
|
1132
|
+
];
|
|
1133
|
+
}
|
|
1134
|
+
},
|
|
1135
|
+
on (event) {
|
|
1136
|
+
coldStartListeners.push(event);
|
|
1137
|
+
return ()=>{};
|
|
1138
|
+
}
|
|
1139
|
+
}), /cold-start|live-agent/i, 'installing benchmark isolation with an existing agent fails hard');
|
|
1140
|
+
assert.deepEqual(coldStartListeners, [], 'cold-start rejection does not register an isolation listener');
|
|
1141
|
+
const processListenersBeforeBenchmark = {
|
|
1142
|
+
uncaughtException: process.listenerCount('uncaughtException'),
|
|
1143
|
+
unhandledRejection: process.listenerCount('unhandledRejection')
|
|
1144
|
+
};
|
|
1145
|
+
apply(ctx, {
|
|
1146
|
+
stateRoot: root,
|
|
1147
|
+
timeoutMs: 20,
|
|
1148
|
+
hbcliBin: '',
|
|
1149
|
+
sessionAccess: 'off',
|
|
1150
|
+
benchmarkEnvironmentConfigPath: configPath
|
|
1151
|
+
});
|
|
1152
|
+
assert.ok(registered.some((tool)=>tool.name === 'gotry_benchmark_environment'), 'an explicit valid owner-local config registers the benchmark environment bridge');
|
|
1153
|
+
assert.deepEqual(registered.map((tool)=>tool.name), [
|
|
1154
|
+
'gotry_benchmark_environment'
|
|
1155
|
+
], 'benchmark mode boots only the single bridge tool instead of the product tool catalog');
|
|
1156
|
+
assert.deepEqual(promptVariables, [], 'benchmark mode does not install product prompt variables');
|
|
1157
|
+
assert.equal(eventNames.includes('tools/pre-execute'), false, 'benchmark mode does not install the product session-consent hook');
|
|
1158
|
+
assert.deepEqual({
|
|
1159
|
+
uncaughtException: process.listenerCount('uncaughtException'),
|
|
1160
|
+
unhandledRejection: process.listenerCount('unhandledRejection')
|
|
1161
|
+
}, processListenersBeforeBenchmark, 'benchmark mode does not install product process incident guards');
|
|
1162
|
+
assert.deepEqual([
|
|
1163
|
+
...eventNames
|
|
1164
|
+
].sort(), [
|
|
1165
|
+
'agent/created',
|
|
1166
|
+
'agent/created',
|
|
1167
|
+
'agent/disposed',
|
|
1168
|
+
'agent/disposed',
|
|
1169
|
+
'agent/pre-step',
|
|
1170
|
+
'agent/turn-stopping',
|
|
1171
|
+
'session/disposed',
|
|
1172
|
+
'session/disposed',
|
|
1173
|
+
'session/event',
|
|
1174
|
+
'session/event',
|
|
1175
|
+
'system-prompt/assemble',
|
|
1176
|
+
'tools/execute',
|
|
1177
|
+
'tools/post-execute'
|
|
1178
|
+
], 'benchmark root listeners come only from budget, isolation, and conformance when model override is unset');
|
|
1179
|
+
visibleBridge = registered.find((tool)=>tool.name === 'gotry_benchmark_environment');
|
|
1180
|
+
const restrictions = [];
|
|
1181
|
+
const guards = [];
|
|
1182
|
+
const presentations = [];
|
|
1183
|
+
const cleanupCounts = {
|
|
1184
|
+
restrict: 0,
|
|
1185
|
+
guard: 0,
|
|
1186
|
+
presentAs: 0,
|
|
1187
|
+
assembly: 0
|
|
1188
|
+
};
|
|
1189
|
+
const firstScopedAssemblyListeners = [];
|
|
1190
|
+
const scopedTools = {
|
|
1191
|
+
restrict (filter) {
|
|
1192
|
+
restrictions.push(filter);
|
|
1193
|
+
return ()=>{
|
|
1194
|
+
cleanupCounts.restrict += 1;
|
|
1195
|
+
};
|
|
1196
|
+
},
|
|
1197
|
+
guard (check) {
|
|
1198
|
+
guards.push(check);
|
|
1199
|
+
return ()=>{
|
|
1200
|
+
cleanupCounts.guard += 1;
|
|
1201
|
+
};
|
|
1202
|
+
},
|
|
1203
|
+
presentAs (mode) {
|
|
1204
|
+
presentations.push(mode);
|
|
1205
|
+
return ()=>{
|
|
1206
|
+
cleanupCounts.presentAs += 1;
|
|
1207
|
+
};
|
|
1208
|
+
}
|
|
1209
|
+
};
|
|
1210
|
+
assert.equal(agentCreatedListeners.length, 2, 'opt-in bridge installs exactly isolation and conformance agent listeners');
|
|
1211
|
+
const isolatedAgentEffects = [];
|
|
1212
|
+
const isolatedAgent = {
|
|
1213
|
+
session: {},
|
|
1214
|
+
steer (_message) {},
|
|
1215
|
+
ctx: {
|
|
1216
|
+
tools: scopedTools,
|
|
1217
|
+
effect: (action)=>{
|
|
1218
|
+
const dispose = runEffect(action);
|
|
1219
|
+
isolatedAgentEffects.push(dispose);
|
|
1220
|
+
return dispose;
|
|
1221
|
+
},
|
|
1222
|
+
on: (event, listener, options)=>{
|
|
1223
|
+
assert.equal(event, 'system-prompt/assemble');
|
|
1224
|
+
assert.deepEqual(options, {
|
|
1225
|
+
prepend: true
|
|
1226
|
+
});
|
|
1227
|
+
firstScopedAssemblyListeners.push(listener);
|
|
1228
|
+
return ()=>{
|
|
1229
|
+
cleanupCounts.assembly += 1;
|
|
1230
|
+
};
|
|
1231
|
+
}
|
|
1232
|
+
}
|
|
1233
|
+
};
|
|
1234
|
+
for (const listener of agentCreatedListeners)listener({
|
|
1235
|
+
agent: isolatedAgent
|
|
1236
|
+
});
|
|
1237
|
+
assert.deepEqual(restrictions, [
|
|
1238
|
+
{
|
|
1239
|
+
allow: [
|
|
1240
|
+
'gotry_benchmark_environment'
|
|
1241
|
+
]
|
|
1242
|
+
}
|
|
1243
|
+
], 'agent scope allows only the bridge tool');
|
|
1244
|
+
assert.deepEqual(presentations, [
|
|
1245
|
+
'native'
|
|
1246
|
+
], 'agent scope forces native tool presentation');
|
|
1247
|
+
assert.ok(eventNames.includes('agent/pre-step'), 'isolation observes pre-step before each request');
|
|
1248
|
+
assert.deepEqual(eventOptions.get('agent/pre-step'), {
|
|
1249
|
+
prepend: true
|
|
1250
|
+
}, 'pre-step isolation wraps every previously registered listener');
|
|
1251
|
+
assert.ok(eventNames.includes('agent/disposed'), 'isolation cleans up on agent disposal');
|
|
1252
|
+
assert.ok(assemblyListeners.length > 0, 'isolation validates final assembled tool surface');
|
|
1253
|
+
assert.equal(firstScopedAssemblyListeners.length, 2, 'agent owns one isolation assembly listener and one conformance section listener');
|
|
1254
|
+
assert.deepEqual(eventOptions.get('system-prompt/assemble'), {
|
|
1255
|
+
prepend: true
|
|
1256
|
+
}, 'assembly isolation wraps every previously registered listener');
|
|
1257
|
+
assert.equal(disposedListeners.length, 2, 'isolation and conformance each own one agent/disposed listener');
|
|
1258
|
+
assert.ok(disposeRootIsolation, 'root isolation effect exposes plugin-lifecycle cleanup');
|
|
1259
|
+
const exactSchema = {
|
|
1260
|
+
name: visibleBridge.name,
|
|
1261
|
+
description: visibleBridge.description,
|
|
1262
|
+
parameters: structuredClone(visibleBridge.parameters)
|
|
1263
|
+
};
|
|
1264
|
+
const assemble = assemblyListeners[0];
|
|
1265
|
+
const nextExact = async ()=>({
|
|
1266
|
+
tools: [
|
|
1267
|
+
exactSchema
|
|
1268
|
+
]
|
|
1269
|
+
});
|
|
1270
|
+
assert.deepEqual(await assemble({
|
|
1271
|
+
tools: []
|
|
1272
|
+
}, {
|
|
1273
|
+
agent: isolatedAgent,
|
|
1274
|
+
scope: isolatedAgent
|
|
1275
|
+
}, nextExact), {
|
|
1276
|
+
tools: [
|
|
1277
|
+
exactSchema
|
|
1278
|
+
]
|
|
1279
|
+
}, 'legal final assembly passes unchanged');
|
|
1280
|
+
await assert.rejects(async ()=>await assemble({
|
|
1281
|
+
tools: []
|
|
1282
|
+
}, {
|
|
1283
|
+
agent: isolatedAgent,
|
|
1284
|
+
scope: isolatedAgent
|
|
1285
|
+
}, async ()=>({
|
|
1286
|
+
tools: [
|
|
1287
|
+
exactSchema,
|
|
1288
|
+
{
|
|
1289
|
+
name: 'own_side_effect'
|
|
1290
|
+
}
|
|
1291
|
+
]
|
|
1292
|
+
})), /BENCHMARK_TOOL_SURFACE_VIOLATION/);
|
|
1293
|
+
await assert.rejects(async ()=>await assemble({
|
|
1294
|
+
tools: []
|
|
1295
|
+
}, {
|
|
1296
|
+
agent: isolatedAgent,
|
|
1297
|
+
scope: isolatedAgent
|
|
1298
|
+
}, async ()=>({
|
|
1299
|
+
tools: [
|
|
1300
|
+
{
|
|
1301
|
+
...exactSchema,
|
|
1302
|
+
description: `${exactSchema.description ?? ''} tampered`
|
|
1303
|
+
}
|
|
1304
|
+
]
|
|
1305
|
+
})), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'final assembly rejects same-name schema mutation');
|
|
1306
|
+
const diagnosticAssembly = {
|
|
1307
|
+
tools: [
|
|
1308
|
+
{
|
|
1309
|
+
name: 'diagnostic_tool'
|
|
1310
|
+
}
|
|
1311
|
+
]
|
|
1312
|
+
};
|
|
1313
|
+
assert.equal(await assemble({
|
|
1314
|
+
tools: []
|
|
1315
|
+
}, {
|
|
1316
|
+
scope: {
|
|
1317
|
+
id: 'diagnostic-scope'
|
|
1318
|
+
}
|
|
1319
|
+
}, async ()=>diagnosticAssembly), diagnosticAssembly, 'non-agent diagnostic assembly passes through');
|
|
1320
|
+
await assert.rejects(async ()=>await assemble({
|
|
1321
|
+
tools: []
|
|
1322
|
+
}, {
|
|
1323
|
+
agent: isolatedAgent,
|
|
1324
|
+
scope: {
|
|
1325
|
+
id: 'wrong-scope'
|
|
1326
|
+
}
|
|
1327
|
+
}, nextExact), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'agent assembly requires the same agent and scope');
|
|
1328
|
+
const preStep = preStepListeners[0];
|
|
1329
|
+
assert.deepEqual(await preStep({
|
|
1330
|
+
agent: isolatedAgent
|
|
1331
|
+
}, async ()=>({
|
|
1332
|
+
decision: 'continue'
|
|
1333
|
+
})), {
|
|
1334
|
+
decision: 'continue'
|
|
1335
|
+
}, 'legal pre-step decision passes unchanged');
|
|
1336
|
+
scopedExtraSchemas.set(isolatedAgent, [
|
|
1337
|
+
{
|
|
1338
|
+
name: 'own_side_effect'
|
|
1339
|
+
}
|
|
1340
|
+
]);
|
|
1341
|
+
await assert.rejects(async ()=>await preStep({
|
|
1342
|
+
agent: isolatedAgent
|
|
1343
|
+
}, async ()=>({
|
|
1344
|
+
decision: 'continue'
|
|
1345
|
+
})), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'downstream pre-step scope expansion fails closed');
|
|
1346
|
+
scopedExtraSchemas.delete(isolatedAgent);
|
|
1347
|
+
shadowedAgent = isolatedAgent;
|
|
1348
|
+
await assert.rejects(async ()=>await assemble({
|
|
1349
|
+
tools: []
|
|
1350
|
+
}, {
|
|
1351
|
+
agent: isolatedAgent,
|
|
1352
|
+
scope: isolatedAgent
|
|
1353
|
+
}, nextExact), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'same-name scoped shadow fails final assembly identity check');
|
|
1354
|
+
shadowedAgent = undefined;
|
|
1355
|
+
assert.equal(guards.length, 2, 'agent scope installs exact isolation and conformance guards');
|
|
1356
|
+
const guard = guards[0];
|
|
1357
|
+
const originalAgent = {
|
|
1358
|
+
ctx: {
|
|
1359
|
+
tools: {
|
|
1360
|
+
get: (name)=>name === 'gotry_benchmark_environment' ? visibleBridge : undefined
|
|
1361
|
+
}
|
|
1362
|
+
}
|
|
1363
|
+
};
|
|
1364
|
+
const shadowAgent = {
|
|
1365
|
+
ctx: {
|
|
1366
|
+
tools: {
|
|
1367
|
+
get: (name)=>name === 'gotry_benchmark_environment' ? {
|
|
1368
|
+
name
|
|
1369
|
+
} : undefined
|
|
1370
|
+
}
|
|
1371
|
+
}
|
|
1372
|
+
};
|
|
1373
|
+
shadowedAgent = shadowAgent;
|
|
1374
|
+
assert.equal(guard({
|
|
1375
|
+
name: 'gotry_benchmark_environment',
|
|
1376
|
+
args: {},
|
|
1377
|
+
agent: originalAgent
|
|
1378
|
+
}), undefined, 'original bridge definition is allowed');
|
|
1379
|
+
assert.equal(guard({
|
|
1380
|
+
name: 'gotry_benchmark_environment',
|
|
1381
|
+
args: {},
|
|
1382
|
+
agent: shadowAgent
|
|
1383
|
+
}), 'BENCHMARK_TOOL_NOT_ALLOWED', 'same-name shadow definition is denied');
|
|
1384
|
+
assert.equal(guard({
|
|
1385
|
+
name: 'gotry_benchmark_environment',
|
|
1386
|
+
args: {
|
|
1387
|
+
path: '/private/secret'
|
|
1388
|
+
},
|
|
1389
|
+
agent: isolatedAgent
|
|
1390
|
+
}), undefined, 'bridge tool is allowed');
|
|
1391
|
+
const denied = guard({
|
|
1392
|
+
name: 'other_tool',
|
|
1393
|
+
args: {
|
|
1394
|
+
path: '/private/secret',
|
|
1395
|
+
token: 'secret'
|
|
1396
|
+
}
|
|
1397
|
+
});
|
|
1398
|
+
assert.equal(denied, 'BENCHMARK_TOOL_NOT_ALLOWED', 'non-bridge tools are denied without argument/path echo');
|
|
1399
|
+
assert.equal(denied?.includes('/private/secret'), false);
|
|
1400
|
+
assert.equal(denied?.includes('secret'), false);
|
|
1401
|
+
assert.equal(guard({
|
|
1402
|
+
name: 'other_tool',
|
|
1403
|
+
args: {
|
|
1404
|
+
different: true
|
|
1405
|
+
}
|
|
1406
|
+
}), denied, 'denial reason is stable');
|
|
1407
|
+
assert.throws(()=>agentCreatedListeners[0]({
|
|
1408
|
+
agent: {
|
|
1409
|
+
ctx: {
|
|
1410
|
+
tools: {
|
|
1411
|
+
restrict (filter) {
|
|
1412
|
+
void filter;
|
|
1413
|
+
},
|
|
1414
|
+
guard (check) {
|
|
1415
|
+
void check;
|
|
1416
|
+
}
|
|
1417
|
+
}
|
|
1418
|
+
}
|
|
1419
|
+
}
|
|
1420
|
+
}), /presentAs/, 'missing scoped presentAs fails hard');
|
|
1421
|
+
assert.throws(()=>agentCreatedListeners[0]({
|
|
1422
|
+
agent: {
|
|
1423
|
+
ctx: {
|
|
1424
|
+
tools: scopedTools
|
|
1425
|
+
}
|
|
1426
|
+
}
|
|
1427
|
+
}), /effect/, 'missing scoped effect fails hard');
|
|
1428
|
+
assert.throws(()=>agentCreatedListeners[0]({
|
|
1429
|
+
agent: {
|
|
1430
|
+
ctx: {
|
|
1431
|
+
tools: scopedTools,
|
|
1432
|
+
effect: (action)=>runEffect(action)
|
|
1433
|
+
}
|
|
1434
|
+
}
|
|
1435
|
+
}), /event bus/, 'missing scoped event bus fails hard');
|
|
1436
|
+
assert.throws(()=>agentCreatedListeners[0]({
|
|
1437
|
+
agent: {
|
|
1438
|
+
ctx: {
|
|
1439
|
+
tools: {
|
|
1440
|
+
guard: guards[0]
|
|
1441
|
+
}
|
|
1442
|
+
}
|
|
1443
|
+
}
|
|
1444
|
+
}), /restrict/, 'missing scoped restrict fails hard');
|
|
1445
|
+
assert.throws(()=>agentCreatedListeners[0]({
|
|
1446
|
+
agent: {
|
|
1447
|
+
ctx: {
|
|
1448
|
+
tools: {
|
|
1449
|
+
restrict (filter) {
|
|
1450
|
+
void filter;
|
|
1451
|
+
}
|
|
1452
|
+
}
|
|
1453
|
+
}
|
|
1454
|
+
}
|
|
1455
|
+
}), /guard/, 'missing scoped guard fails hard');
|
|
1456
|
+
assert.throws(()=>installBenchmarkToolIsolation({
|
|
1457
|
+
tools: {
|
|
1458
|
+
get (name) {
|
|
1459
|
+
return name === 'gotry_benchmark_environment' ? {
|
|
1460
|
+
name: 'gotry_benchmark_environment'
|
|
1461
|
+
} : undefined;
|
|
1462
|
+
},
|
|
1463
|
+
schemas () {
|
|
1464
|
+
return [
|
|
1465
|
+
{
|
|
1466
|
+
name: 'gotry_benchmark_environment'
|
|
1467
|
+
}
|
|
1468
|
+
];
|
|
1469
|
+
}
|
|
1470
|
+
},
|
|
1471
|
+
agents: {
|
|
1472
|
+
list () {
|
|
1473
|
+
return [];
|
|
1474
|
+
}
|
|
1475
|
+
},
|
|
1476
|
+
on () {
|
|
1477
|
+
return ()=>{};
|
|
1478
|
+
}
|
|
1479
|
+
}), /effect/, 'missing ctx.effect fails hard');
|
|
1480
|
+
await disposedListeners[0]({
|
|
1481
|
+
agent: isolatedAgent
|
|
1482
|
+
});
|
|
1483
|
+
assert.deepEqual(cleanupCounts, {
|
|
1484
|
+
restrict: 1,
|
|
1485
|
+
guard: 1,
|
|
1486
|
+
presentAs: 1,
|
|
1487
|
+
assembly: 1
|
|
1488
|
+
}, 'agent disposal releases every scoped isolation effect exactly once');
|
|
1489
|
+
await disposedListeners[1]({
|
|
1490
|
+
agent: isolatedAgent
|
|
1491
|
+
});
|
|
1492
|
+
await isolatedAgentEffects[1]();
|
|
1493
|
+
assert.deepEqual(cleanupCounts, {
|
|
1494
|
+
restrict: 1,
|
|
1495
|
+
guard: 2,
|
|
1496
|
+
presentAs: 1,
|
|
1497
|
+
assembly: 2
|
|
1498
|
+
}, 'agent-scope disposal also releases conformance guard and prompt section');
|
|
1499
|
+
const secondCleanup = {
|
|
1500
|
+
restrict: 0,
|
|
1501
|
+
guard: 0,
|
|
1502
|
+
presentAs: 0,
|
|
1503
|
+
assembly: 0
|
|
1504
|
+
};
|
|
1505
|
+
const secondGuards = [];
|
|
1506
|
+
const secondScopedAssemblyListeners = [];
|
|
1507
|
+
const secondTools = {
|
|
1508
|
+
restrict () {
|
|
1509
|
+
return ()=>{
|
|
1510
|
+
secondCleanup.restrict += 1;
|
|
1511
|
+
};
|
|
1512
|
+
},
|
|
1513
|
+
guard (check) {
|
|
1514
|
+
secondGuards.push(check);
|
|
1515
|
+
return ()=>{
|
|
1516
|
+
secondCleanup.guard += 1;
|
|
1517
|
+
};
|
|
1518
|
+
},
|
|
1519
|
+
presentAs () {
|
|
1520
|
+
return ()=>{
|
|
1521
|
+
secondCleanup.presentAs += 1;
|
|
1522
|
+
};
|
|
1523
|
+
}
|
|
1524
|
+
};
|
|
1525
|
+
const secondAgentEffects = [];
|
|
1526
|
+
const secondAgent = {
|
|
1527
|
+
session: {},
|
|
1528
|
+
steer (_message) {},
|
|
1529
|
+
ctx: {
|
|
1530
|
+
tools: secondTools,
|
|
1531
|
+
effect: (action)=>{
|
|
1532
|
+
const dispose = runEffect(action);
|
|
1533
|
+
secondAgentEffects.push(dispose);
|
|
1534
|
+
return dispose;
|
|
1535
|
+
},
|
|
1536
|
+
on: (_event, listener)=>{
|
|
1537
|
+
secondScopedAssemblyListeners.push(listener);
|
|
1538
|
+
return ()=>{
|
|
1539
|
+
secondCleanup.assembly += 1;
|
|
1540
|
+
};
|
|
1541
|
+
}
|
|
1542
|
+
}
|
|
1543
|
+
};
|
|
1544
|
+
for (const listener of agentCreatedListeners)listener({
|
|
1545
|
+
agent: secondAgent
|
|
1546
|
+
});
|
|
1547
|
+
let inFlightNextCalls = 0;
|
|
1548
|
+
await assert.rejects(async ()=>await secondScopedAssemblyListeners[0]({}, {
|
|
1549
|
+
agent: secondAgent,
|
|
1550
|
+
scope: secondAgent
|
|
1551
|
+
}, async ()=>{
|
|
1552
|
+
inFlightNextCalls += 1;
|
|
1553
|
+
await disposeRootIsolation();
|
|
1554
|
+
return {
|
|
1555
|
+
tools: [
|
|
1556
|
+
exactSchema
|
|
1557
|
+
]
|
|
1558
|
+
};
|
|
1559
|
+
}), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'plugin unload during assembly fails closed before returning a model-visible result');
|
|
1560
|
+
assert.equal(inFlightNextCalls, 1, 'in-flight quarantine test reaches the controlled unload point');
|
|
1561
|
+
assert.deepEqual(secondCleanup, {
|
|
1562
|
+
restrict: 0,
|
|
1563
|
+
guard: 0,
|
|
1564
|
+
presentAs: 0,
|
|
1565
|
+
assembly: 0
|
|
1566
|
+
}, 'plugin unload keeps a live agent quarantined');
|
|
1567
|
+
let quarantineNextCalls = 0;
|
|
1568
|
+
await assert.rejects(async ()=>await secondScopedAssemblyListeners[0]({}, {
|
|
1569
|
+
agent: secondAgent,
|
|
1570
|
+
scope: secondAgent
|
|
1571
|
+
}, async ()=>{
|
|
1572
|
+
quarantineNextCalls += 1;
|
|
1573
|
+
return {
|
|
1574
|
+
tools: [
|
|
1575
|
+
exactSchema
|
|
1576
|
+
]
|
|
1577
|
+
};
|
|
1578
|
+
}), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'plugin-unloaded live agent rejects assembly before model request');
|
|
1579
|
+
assert.equal(quarantineNextCalls, 0, 'quarantine does not enter the remaining assembly chain');
|
|
1580
|
+
assert.equal(secondGuards[0]({
|
|
1581
|
+
name: 'other_tool',
|
|
1582
|
+
agent: secondAgent
|
|
1583
|
+
}), 'BENCHMARK_TOOL_NOT_ALLOWED', 'quarantined agent still denies non-bridge dispatch after plugin unload');
|
|
1584
|
+
assert.equal(secondGuards[0]({
|
|
1585
|
+
name: 'gotry_benchmark_environment',
|
|
1586
|
+
agent: secondAgent
|
|
1587
|
+
}), 'BENCHMARK_TOOL_NOT_ALLOWED', 'quarantined agent also denies bridge dispatch after plugin unload');
|
|
1588
|
+
assert.deepEqual(cleanupCounts, {
|
|
1589
|
+
restrict: 1,
|
|
1590
|
+
guard: 2,
|
|
1591
|
+
presentAs: 1,
|
|
1592
|
+
assembly: 2
|
|
1593
|
+
}, 'plugin unload does not double-dispose an already removed agent');
|
|
1594
|
+
assert.throws(()=>agentCreatedListeners[0]({
|
|
1595
|
+
agent: secondAgent
|
|
1596
|
+
}), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'agent creation during plugin stop fails closed');
|
|
1597
|
+
for (const dispose of secondAgentEffects)await dispose();
|
|
1598
|
+
assert.deepEqual(secondCleanup, {
|
|
1599
|
+
restrict: 1,
|
|
1600
|
+
guard: 2,
|
|
1601
|
+
presentAs: 1,
|
|
1602
|
+
assembly: 2
|
|
1603
|
+
}, 'agent disposal releases its quarantined isolation and conformance effects');
|
|
1604
|
+
const bridge = registered.find((tool)=>tool.name === 'gotry_benchmark_environment');
|
|
1605
|
+
assert.ok(bridge.execute, 'registered bridge exposes execute');
|
|
1606
|
+
const args = {
|
|
1607
|
+
city: 'Dubai',
|
|
1608
|
+
payload: '$(touch /tmp/nope)'
|
|
1609
|
+
};
|
|
1610
|
+
const result = await bridge.execute({
|
|
1611
|
+
query: {
|
|
1612
|
+
action: 'call',
|
|
1613
|
+
tool: 'lookup',
|
|
1614
|
+
arguments: args
|
|
1615
|
+
}
|
|
1616
|
+
}, null);
|
|
1617
|
+
assert.deepEqual(spawnSpecs[0]?.argv, [
|
|
1618
|
+
process.execPath,
|
|
1619
|
+
'-m',
|
|
1620
|
+
'agent_env.cli',
|
|
1621
|
+
'--lang',
|
|
1622
|
+
'en',
|
|
1623
|
+
'call',
|
|
1624
|
+
'lookup',
|
|
1625
|
+
JSON.stringify(args)
|
|
1626
|
+
], 'call uses only the configured executable/prefix and fixed lookup subcommand argv');
|
|
1627
|
+
assert.equal(spawnSpecs[0]?.cwd, root, 'call uses configured cwd');
|
|
1628
|
+
assert.equal(spawnSpecs[0]?.stdio.stdin, 'ignore', 'call never exposes stdin');
|
|
1629
|
+
assert.equal(spawnSpecs[0]?.stdio.stdout.maxBytes, 4_096, 'stdout cap comes from config');
|
|
1630
|
+
assert.equal(spawnSpecs[0]?.stdio.stderr.maxBytes, 4_096, 'stderr cap comes from config');
|
|
1631
|
+
assert.ok((spawnSpecs[0]?.graceMs ?? 0) > 0 && (spawnSpecs[0]?.graceMs ?? Infinity) <= 1_000, 'graceMs is bounded');
|
|
1632
|
+
assert.equal(spawnSpecs[0]?.env?.GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL, undefined, 'parent sentinel is not inherited');
|
|
1633
|
+
for (const key of [
|
|
1634
|
+
'DATABASE_URL',
|
|
1635
|
+
'SSH_AUTH_SOCK',
|
|
1636
|
+
'AWS_PROFILE',
|
|
1637
|
+
'HTTPS_PROXY',
|
|
1638
|
+
'GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL'
|
|
1639
|
+
]){
|
|
1640
|
+
assert.equal(spawnSpecs[0]?.env?.[key], undefined, `${key} is tombstoned in bridge subprocess env`);
|
|
1641
|
+
}
|
|
1642
|
+
assert.equal(spawnSpecs[0]?.env?.PYTHONDONTWRITEBYTECODE, '1');
|
|
1643
|
+
assert.equal(spawnSpecs[0]?.env?.PYTHONNOUSERSITE, '1');
|
|
1644
|
+
assert.deepEqual(result, {
|
|
1645
|
+
ok: true,
|
|
1646
|
+
result: [
|
|
1647
|
+
{
|
|
1648
|
+
city: 'Dubai'
|
|
1649
|
+
}
|
|
1650
|
+
]
|
|
1651
|
+
}, 'one-line JSON stdout becomes structured result');
|
|
1652
|
+
outcomes.push({
|
|
1653
|
+
stdout: '{"result":{"city":"Dubai"}}'
|
|
1654
|
+
});
|
|
1655
|
+
const unmappedResult = await bridge.execute({
|
|
1656
|
+
query: {
|
|
1657
|
+
action: 'call',
|
|
1658
|
+
tool: 'toString',
|
|
1659
|
+
arguments: {
|
|
1660
|
+
city: 'Dubai'
|
|
1661
|
+
}
|
|
1662
|
+
}
|
|
1663
|
+
}, null);
|
|
1664
|
+
assert.deepEqual(unmappedResult, {
|
|
1665
|
+
ok: true,
|
|
1666
|
+
result: {
|
|
1667
|
+
city: 'Dubai'
|
|
1668
|
+
}
|
|
1669
|
+
}, 'allowed tool without a positive mapping retains the recursive denylist only');
|
|
1670
|
+
outcomes.push({
|
|
1671
|
+
stdout: '{"result":{"legacy":"value"}}'
|
|
1672
|
+
});
|
|
1673
|
+
const legacyResult = await bridge.execute({
|
|
1674
|
+
query: {
|
|
1675
|
+
action: 'call',
|
|
1676
|
+
tool: 'constructor',
|
|
1677
|
+
arguments: {
|
|
1678
|
+
city: 'Dubai'
|
|
1679
|
+
}
|
|
1680
|
+
}
|
|
1681
|
+
}, null);
|
|
1682
|
+
assert.deepEqual(legacyResult, {
|
|
1683
|
+
ok: true,
|
|
1684
|
+
result: {
|
|
1685
|
+
legacy: 'value'
|
|
1686
|
+
}
|
|
1687
|
+
}, 'mapped constructor accepts its declared positive key');
|
|
1688
|
+
outcomes.push({
|
|
1689
|
+
stdout: '{"result":{"city":"Dubai"}}'
|
|
1690
|
+
});
|
|
1691
|
+
const constructorUnexpected = await bridge.execute({
|
|
1692
|
+
query: {
|
|
1693
|
+
action: 'call',
|
|
1694
|
+
tool: 'constructor',
|
|
1695
|
+
arguments: {
|
|
1696
|
+
city: 'Dubai'
|
|
1697
|
+
}
|
|
1698
|
+
}
|
|
1699
|
+
}, null);
|
|
1700
|
+
assert.deepEqual(constructorUnexpected, {
|
|
1701
|
+
ok: false,
|
|
1702
|
+
error: 'forbidden_output'
|
|
1703
|
+
}, 'mapped constructor rejects undeclared positive keys');
|
|
1704
|
+
outcomes.push({
|
|
1705
|
+
stdout: '{"result":{"nested":{"city":"Dubai"}}}'
|
|
1706
|
+
});
|
|
1707
|
+
const nestedAllowedResult = await bridge.execute({
|
|
1708
|
+
query: {
|
|
1709
|
+
action: 'call',
|
|
1710
|
+
tool: 'lookup',
|
|
1711
|
+
arguments: {
|
|
1712
|
+
city: 'Dubai'
|
|
1713
|
+
}
|
|
1714
|
+
}
|
|
1715
|
+
}, null);
|
|
1716
|
+
assert.deepEqual(nestedAllowedResult, {
|
|
1717
|
+
ok: true,
|
|
1718
|
+
result: {
|
|
1719
|
+
nested: {
|
|
1720
|
+
city: 'Dubai'
|
|
1721
|
+
}
|
|
1722
|
+
}
|
|
1723
|
+
}, 'configured positive output allowlist accepts declared nested keys');
|
|
1724
|
+
outcomes.push({
|
|
1725
|
+
stdout: '{"result":{"city":"Dubai","unexpected":"secret"}}'
|
|
1726
|
+
});
|
|
1727
|
+
const unexpectedResult = await bridge.execute({
|
|
1728
|
+
query: {
|
|
1729
|
+
action: 'call',
|
|
1730
|
+
tool: 'lookup',
|
|
1731
|
+
arguments: {
|
|
1732
|
+
city: 'Dubai'
|
|
1733
|
+
}
|
|
1734
|
+
}
|
|
1735
|
+
}, null);
|
|
1736
|
+
assert.deepEqual(unexpectedResult, {
|
|
1737
|
+
ok: false,
|
|
1738
|
+
error: 'forbidden_output'
|
|
1739
|
+
}, 'configured positive output allowlist rejects unexpected keys without reflecting them');
|
|
1740
|
+
outcomes.push({
|
|
1741
|
+
stdout: '{"result":[{"city":"Dubai","nested":{"unexpected":"secret"}}]}'
|
|
1742
|
+
});
|
|
1743
|
+
const nestedUnexpectedResult = await bridge.execute({
|
|
1744
|
+
query: {
|
|
1745
|
+
action: 'call',
|
|
1746
|
+
tool: 'lookup',
|
|
1747
|
+
arguments: {
|
|
1748
|
+
city: 'Dubai'
|
|
1749
|
+
}
|
|
1750
|
+
}
|
|
1751
|
+
}, null);
|
|
1752
|
+
assert.deepEqual(nestedUnexpectedResult, {
|
|
1753
|
+
ok: false,
|
|
1754
|
+
error: 'forbidden_output'
|
|
1755
|
+
}, 'configured positive output allowlist recurses through arrays and objects');
|
|
1756
|
+
for (const forbidden of [
|
|
1757
|
+
'gold',
|
|
1758
|
+
'goldAnswer',
|
|
1759
|
+
'oracle',
|
|
1760
|
+
'expected',
|
|
1761
|
+
'expected_answer',
|
|
1762
|
+
'answer',
|
|
1763
|
+
'label',
|
|
1764
|
+
'score',
|
|
1765
|
+
'reward',
|
|
1766
|
+
'ground_truth',
|
|
1767
|
+
'groundTruth',
|
|
1768
|
+
'hidden_query',
|
|
1769
|
+
'hidden-query',
|
|
1770
|
+
'loader_metadata',
|
|
1771
|
+
'loaderMetadata',
|
|
1772
|
+
'reference',
|
|
1773
|
+
'gоld'
|
|
1774
|
+
]){
|
|
1775
|
+
outcomes.push({
|
|
1776
|
+
stdout: JSON.stringify({
|
|
1777
|
+
result: {
|
|
1778
|
+
nested: {
|
|
1779
|
+
[forbidden]: 'secret'
|
|
1780
|
+
}
|
|
1781
|
+
}
|
|
1782
|
+
})
|
|
1783
|
+
});
|
|
1784
|
+
const forbiddenResult = await bridge.execute({
|
|
1785
|
+
query: {
|
|
1786
|
+
action: 'call',
|
|
1787
|
+
tool: 'lookup',
|
|
1788
|
+
arguments: {
|
|
1789
|
+
city: 'Dubai'
|
|
1790
|
+
}
|
|
1791
|
+
}
|
|
1792
|
+
}, null);
|
|
1793
|
+
assert.deepEqual(forbiddenResult, {
|
|
1794
|
+
ok: false,
|
|
1795
|
+
error: 'forbidden_output'
|
|
1796
|
+
}, `recursive no-oracle key ${forbidden} is rejected without reflecting its name`);
|
|
1797
|
+
}
|
|
1798
|
+
outcomes.push({
|
|
1799
|
+
stdout: '{"result":"the hidden answer"}'
|
|
1800
|
+
});
|
|
1801
|
+
const primitiveOutput = await bridge.execute({
|
|
1802
|
+
query: {
|
|
1803
|
+
action: 'call',
|
|
1804
|
+
tool: 'lookup',
|
|
1805
|
+
arguments: {
|
|
1806
|
+
city: 'Dubai'
|
|
1807
|
+
}
|
|
1808
|
+
}
|
|
1809
|
+
}, null);
|
|
1810
|
+
assert.deepEqual(primitiveOutput, {
|
|
1811
|
+
ok: false,
|
|
1812
|
+
error: 'invalid_output'
|
|
1813
|
+
}, 'primitive result strings cannot bypass the structured visible-output boundary');
|
|
1814
|
+
const beforeOversized = spawnSpecs.length;
|
|
1815
|
+
const oversized = await bridge.execute({
|
|
1816
|
+
query: {
|
|
1817
|
+
action: 'call',
|
|
1818
|
+
tool: 'lookup',
|
|
1819
|
+
arguments: {
|
|
1820
|
+
blob: 'x'.repeat(65_537)
|
|
1821
|
+
}
|
|
1822
|
+
}
|
|
1823
|
+
}, null);
|
|
1824
|
+
assert.deepEqual(oversized, {
|
|
1825
|
+
ok: false,
|
|
1826
|
+
error: 'invalid_arguments',
|
|
1827
|
+
reason: 'serialization_limit'
|
|
1828
|
+
});
|
|
1829
|
+
assert.equal(spawnSpecs.length, beforeOversized, 'oversized serialized arguments are rejected before spawn');
|
|
1830
|
+
const beforeDeep = spawnSpecs.length;
|
|
1831
|
+
let deep = {};
|
|
1832
|
+
for(let index = 0; index < 13; index++)deep = {
|
|
1833
|
+
next: deep
|
|
1834
|
+
};
|
|
1835
|
+
const deepResult = await bridge.execute({
|
|
1836
|
+
query: {
|
|
1837
|
+
action: 'call',
|
|
1838
|
+
tool: 'lookup',
|
|
1839
|
+
arguments: deep
|
|
1840
|
+
}
|
|
1841
|
+
}, null);
|
|
1842
|
+
assert.deepEqual(deepResult, {
|
|
1843
|
+
ok: false,
|
|
1844
|
+
error: 'invalid_arguments',
|
|
1845
|
+
reason: 'serialization_limit'
|
|
1846
|
+
});
|
|
1847
|
+
assert.equal(spawnSpecs.length, beforeDeep, 'over-deep arguments are rejected before spawn');
|
|
1848
|
+
const beforeOverride = spawnSpecs.length;
|
|
1849
|
+
const overrideResult = await bridge.execute({
|
|
1850
|
+
query: {
|
|
1851
|
+
action: 'call',
|
|
1852
|
+
tool: 'lookup',
|
|
1853
|
+
arguments: {
|
|
1854
|
+
city: 'Dubai',
|
|
1855
|
+
executable: '/tmp/evil',
|
|
1856
|
+
cwd: '/tmp/evil',
|
|
1857
|
+
argv: [
|
|
1858
|
+
'--unsafe'
|
|
1859
|
+
]
|
|
1860
|
+
}
|
|
1861
|
+
}
|
|
1862
|
+
}, null);
|
|
1863
|
+
assert.deepEqual(spawnSpecs[beforeOverride]?.argv, [
|
|
1864
|
+
process.execPath,
|
|
1865
|
+
'-m',
|
|
1866
|
+
'agent_env.cli',
|
|
1867
|
+
'--lang',
|
|
1868
|
+
'en',
|
|
1869
|
+
'call',
|
|
1870
|
+
'lookup',
|
|
1871
|
+
JSON.stringify({
|
|
1872
|
+
city: 'Dubai',
|
|
1873
|
+
executable: '/tmp/evil',
|
|
1874
|
+
cwd: '/tmp/evil',
|
|
1875
|
+
argv: [
|
|
1876
|
+
'--unsafe'
|
|
1877
|
+
]
|
|
1878
|
+
})
|
|
1879
|
+
], 'model executable/cwd/argv fields remain data and cannot override config');
|
|
1880
|
+
assert.equal(overrideResult.ok, true, 'override-shaped arguments still use the configured bridge');
|
|
1881
|
+
outcomes.push({
|
|
1882
|
+
stdout: '{"result":{"city":"Dubai"}}',
|
|
1883
|
+
lossy: true
|
|
1884
|
+
});
|
|
1885
|
+
const truncated = await bridge.execute({
|
|
1886
|
+
query: {
|
|
1887
|
+
action: 'call',
|
|
1888
|
+
tool: 'lookup',
|
|
1889
|
+
arguments: {
|
|
1890
|
+
city: 'Dubai'
|
|
1891
|
+
}
|
|
1892
|
+
}
|
|
1893
|
+
}, null);
|
|
1894
|
+
assert.deepEqual(truncated, {
|
|
1895
|
+
ok: false,
|
|
1896
|
+
error: 'output_truncated'
|
|
1897
|
+
}, 'lossy stdout is rejected without parsing partial output');
|
|
1898
|
+
outcomes.push({
|
|
1899
|
+
stdout: '{"result":{"city":"Dubai"}}',
|
|
1900
|
+
stderr: 'private runner diagnostic',
|
|
1901
|
+
exitCode: 17,
|
|
1902
|
+
signal: 'SIGTERM'
|
|
1903
|
+
});
|
|
1904
|
+
const failed = await bridge.execute({
|
|
1905
|
+
query: {
|
|
1906
|
+
action: 'call',
|
|
1907
|
+
tool: 'lookup',
|
|
1908
|
+
arguments: {
|
|
1909
|
+
city: 'Dubai'
|
|
1910
|
+
}
|
|
1911
|
+
}
|
|
1912
|
+
}, null);
|
|
1913
|
+
assert.deepEqual(failed, {
|
|
1914
|
+
ok: false,
|
|
1915
|
+
error: 'runner_failed',
|
|
1916
|
+
exit_code: 17,
|
|
1917
|
+
signal: 'SIGTERM'
|
|
1918
|
+
}, 'nonzero runner result is structured without stderr echo');
|
|
1919
|
+
for (const stdout of [
|
|
1920
|
+
'not-json',
|
|
1921
|
+
'{"result":1}{"result":2}'
|
|
1922
|
+
]){
|
|
1923
|
+
outcomes.push({
|
|
1924
|
+
stdout
|
|
1925
|
+
});
|
|
1926
|
+
const malformed = await bridge.execute({
|
|
1927
|
+
query: {
|
|
1928
|
+
action: 'call',
|
|
1929
|
+
tool: 'lookup',
|
|
1930
|
+
arguments: {
|
|
1931
|
+
city: 'Dubai'
|
|
1932
|
+
}
|
|
1933
|
+
}
|
|
1934
|
+
}, null);
|
|
1935
|
+
assert.deepEqual(malformed, {
|
|
1936
|
+
ok: false,
|
|
1937
|
+
error: 'invalid_json'
|
|
1938
|
+
}, 'malformed or multi-value JSON is rejected');
|
|
1939
|
+
}
|
|
1940
|
+
outcomes.push({
|
|
1941
|
+
spawnError: true
|
|
1942
|
+
});
|
|
1943
|
+
const spawnFailed = await bridge.execute({
|
|
1944
|
+
query: {
|
|
1945
|
+
action: 'call',
|
|
1946
|
+
tool: 'lookup',
|
|
1947
|
+
arguments: {
|
|
1948
|
+
city: 'Dubai'
|
|
1949
|
+
}
|
|
1950
|
+
}
|
|
1951
|
+
}, null);
|
|
1952
|
+
assert.deepEqual(spawnFailed, {
|
|
1953
|
+
ok: false,
|
|
1954
|
+
error: 'spawn_failed'
|
|
1955
|
+
}, 'spawn infrastructure failure is structured');
|
|
1956
|
+
outcomes.push({
|
|
1957
|
+
spawnReject: true
|
|
1958
|
+
});
|
|
1959
|
+
const asyncSpawnFailed = await bridge.execute({
|
|
1960
|
+
query: {
|
|
1961
|
+
action: 'call',
|
|
1962
|
+
tool: 'lookup',
|
|
1963
|
+
arguments: {
|
|
1964
|
+
city: 'Dubai'
|
|
1965
|
+
}
|
|
1966
|
+
}
|
|
1967
|
+
}, null);
|
|
1968
|
+
assert.deepEqual(asyncSpawnFailed, {
|
|
1969
|
+
ok: false,
|
|
1970
|
+
error: 'spawn_failed'
|
|
1971
|
+
}, 'DSH pid=-1 spawn rejection is distinct from a started runner failure');
|
|
1972
|
+
outcomes.push({
|
|
1973
|
+
waitForAbort: true
|
|
1974
|
+
});
|
|
1975
|
+
const timedOut = await bridge.execute({
|
|
1976
|
+
query: {
|
|
1977
|
+
action: 'call',
|
|
1978
|
+
tool: 'lookup',
|
|
1979
|
+
arguments: {
|
|
1980
|
+
city: 'Dubai'
|
|
1981
|
+
}
|
|
1982
|
+
}
|
|
1983
|
+
}, null);
|
|
1984
|
+
assert.deepEqual(timedOut, {
|
|
1985
|
+
ok: false,
|
|
1986
|
+
error: 'timed_out'
|
|
1987
|
+
}, 'deadline abort is surfaced as timed_out');
|
|
1988
|
+
assert.equal(timeoutSignal?.aborted, true, 'timeout abort signal is fired');
|
|
1989
|
+
const beforeInvalidArgs = spawnSpecs.length;
|
|
1990
|
+
assert.deepEqual(await bridge.execute({
|
|
1991
|
+
query: {
|
|
1992
|
+
action: 'call',
|
|
1993
|
+
tool: 'lookup',
|
|
1994
|
+
arguments: [
|
|
1995
|
+
'not',
|
|
1996
|
+
'plain'
|
|
1997
|
+
]
|
|
1998
|
+
}
|
|
1999
|
+
}, null), {
|
|
2000
|
+
ok: false,
|
|
2001
|
+
error: 'invalid_arguments'
|
|
2002
|
+
});
|
|
2003
|
+
assert.equal(spawnSpecs.length, beforeInvalidArgs, 'non-object arguments are rejected before spawn');
|
|
2004
|
+
assert.deepEqual(await bridge.execute({
|
|
2005
|
+
query: {
|
|
2006
|
+
action: 'inspect'
|
|
2007
|
+
}
|
|
2008
|
+
}, null), {
|
|
2009
|
+
ok: false,
|
|
2010
|
+
error: 'invalid_action'
|
|
2011
|
+
}, 'unknown action is rejected structurally');
|
|
2012
|
+
const beforeRejected = spawnSpecs.length;
|
|
2013
|
+
const rejected = await bridge.execute({
|
|
2014
|
+
query: {
|
|
2015
|
+
action: 'call',
|
|
2016
|
+
tool: 'delete_all',
|
|
2017
|
+
arguments: {}
|
|
2018
|
+
}
|
|
2019
|
+
}, null);
|
|
2020
|
+
assert.deepEqual(rejected, {
|
|
2021
|
+
ok: false,
|
|
2022
|
+
error: 'disallowed_tool'
|
|
2023
|
+
}, 'disallowed tool is rejected structurally');
|
|
2024
|
+
assert.equal(spawnSpecs.length, beforeRejected, 'disallowed tool is rejected before spawn');
|
|
2025
|
+
const spacedConfigPath = join(root, ' benchmark-environment-config.json ');
|
|
2026
|
+
writeFileSync(spacedConfigPath, readFileSync(configPath));
|
|
2027
|
+
const spacedTools = [];
|
|
2028
|
+
const spacedCtx = {
|
|
2029
|
+
tools: {
|
|
2030
|
+
register (tool) {
|
|
2031
|
+
spacedTools.push(tool);
|
|
2032
|
+
return ()=>{};
|
|
2033
|
+
},
|
|
2034
|
+
get (name) {
|
|
2035
|
+
return spacedTools.find((tool)=>tool.name === name);
|
|
2036
|
+
},
|
|
2037
|
+
schemas () {
|
|
2038
|
+
return spacedTools.map((tool)=>({
|
|
2039
|
+
name: tool.name
|
|
2040
|
+
}));
|
|
2041
|
+
}
|
|
2042
|
+
},
|
|
2043
|
+
agents: {
|
|
2044
|
+
list () {
|
|
2045
|
+
return [];
|
|
2046
|
+
}
|
|
2047
|
+
},
|
|
2048
|
+
systemPrompt: {
|
|
2049
|
+
variable () {}
|
|
2050
|
+
},
|
|
2051
|
+
on () {
|
|
2052
|
+
return ()=>{};
|
|
2053
|
+
},
|
|
2054
|
+
effect () {
|
|
2055
|
+
return ()=>{};
|
|
2056
|
+
},
|
|
2057
|
+
get (name) {
|
|
2058
|
+
if (name === 'subprocess') return {
|
|
2059
|
+
spawn: (_spec)=>fakeHandle({
|
|
2060
|
+
stdout: '{}'
|
|
2061
|
+
})
|
|
2062
|
+
};
|
|
2063
|
+
if (name === 'agents') return {
|
|
2064
|
+
list () {
|
|
2065
|
+
return [];
|
|
2066
|
+
}
|
|
2067
|
+
};
|
|
2068
|
+
return undefined;
|
|
2069
|
+
}
|
|
2070
|
+
};
|
|
2071
|
+
apply(spacedCtx, {
|
|
2072
|
+
stateRoot: root,
|
|
2073
|
+
timeoutMs: 20,
|
|
2074
|
+
hbcliBin: '',
|
|
2075
|
+
sessionAccess: 'off',
|
|
2076
|
+
benchmarkEnvironmentConfigPath: spacedConfigPath
|
|
2077
|
+
});
|
|
2078
|
+
assert.deepEqual(spacedTools.map((tool)=>tool.name), [
|
|
2079
|
+
'gotry_benchmark_environment'
|
|
2080
|
+
], 'benchmark bridge loads a valid raw path with whitespace basename');
|
|
2081
|
+
const disabled = [];
|
|
2082
|
+
const disabledVariables = [];
|
|
2083
|
+
const disabledCtx = {
|
|
2084
|
+
tools: {
|
|
2085
|
+
register (tool) {
|
|
2086
|
+
disabled.push(tool);
|
|
2087
|
+
return ()=>{};
|
|
2088
|
+
}
|
|
2089
|
+
},
|
|
2090
|
+
systemPrompt: {
|
|
2091
|
+
variable (name) {
|
|
2092
|
+
disabledVariables.push(name);
|
|
2093
|
+
}
|
|
2094
|
+
}
|
|
2095
|
+
};
|
|
2096
|
+
apply(disabledCtx, {
|
|
2097
|
+
stateRoot: root,
|
|
2098
|
+
timeoutMs: 1_000,
|
|
2099
|
+
hbcliBin: '',
|
|
2100
|
+
sessionAccess: 'off',
|
|
2101
|
+
benchmarkEnvironmentConfigPath: ''
|
|
2102
|
+
});
|
|
2103
|
+
assert.equal(disabled.some((tool)=>tool.name === 'gotry_benchmark_environment'), false, 'empty config path keeps bridge default-off');
|
|
2104
|
+
assert.ok(disabled.length > 1, 'normal product mode keeps the full GoTry tool catalog');
|
|
2105
|
+
assert.deepEqual(disabledVariables, [
|
|
2106
|
+
'current_date',
|
|
2107
|
+
'time_anchor_card',
|
|
2108
|
+
'motivation_brief'
|
|
2109
|
+
], 'normal product mode keeps its prompt variables');
|
|
2110
|
+
const whitespace = [];
|
|
2111
|
+
const whitespaceCtx = {
|
|
2112
|
+
tools: {
|
|
2113
|
+
register (tool) {
|
|
2114
|
+
whitespace.push(tool);
|
|
2115
|
+
return ()=>{};
|
|
2116
|
+
}
|
|
2117
|
+
},
|
|
2118
|
+
systemPrompt: {
|
|
2119
|
+
variable () {}
|
|
2120
|
+
}
|
|
2121
|
+
};
|
|
2122
|
+
apply(whitespaceCtx, {
|
|
2123
|
+
stateRoot: root,
|
|
2124
|
+
timeoutMs: 1_000,
|
|
2125
|
+
hbcliBin: '',
|
|
2126
|
+
sessionAccess: 'off',
|
|
2127
|
+
benchmarkEnvironmentConfigPath: ' \t '
|
|
2128
|
+
});
|
|
2129
|
+
assert.equal(whitespace.some((tool)=>tool.name === 'gotry_benchmark_environment'), false, 'whitespace config path keeps benchmark mode default-off');
|
|
2130
|
+
assert.ok(whitespace.length > 1, 'whitespace path preserves the ordinary product tool catalog');
|
|
2131
|
+
const originalModelOverride = process.env.GOTRY_LLM_MODEL;
|
|
2132
|
+
process.env.GOTRY_LLM_MODEL = 'round7-model-preserved';
|
|
2133
|
+
try {
|
|
2134
|
+
const modelTools = [];
|
|
2135
|
+
let modelRequest;
|
|
2136
|
+
const modelEvents = [];
|
|
2137
|
+
const modelCtx = {
|
|
2138
|
+
tools: {
|
|
2139
|
+
register (tool) {
|
|
2140
|
+
modelTools.push(tool);
|
|
2141
|
+
return ()=>{};
|
|
2142
|
+
},
|
|
2143
|
+
get (name) {
|
|
2144
|
+
return modelTools.find((tool)=>tool.name === name);
|
|
2145
|
+
},
|
|
2146
|
+
schemas () {
|
|
2147
|
+
return modelTools.map((tool)=>({
|
|
2148
|
+
name: tool.name
|
|
2149
|
+
}));
|
|
2150
|
+
}
|
|
2151
|
+
},
|
|
2152
|
+
agents: {
|
|
2153
|
+
list () {
|
|
2154
|
+
return [];
|
|
2155
|
+
}
|
|
2156
|
+
},
|
|
2157
|
+
systemPrompt: {
|
|
2158
|
+
variable () {}
|
|
2159
|
+
},
|
|
2160
|
+
on (event, listener) {
|
|
2161
|
+
modelEvents.push(event);
|
|
2162
|
+
if (event === 'agent/request') modelRequest = listener;
|
|
2163
|
+
return ()=>{};
|
|
2164
|
+
},
|
|
2165
|
+
effect () {
|
|
2166
|
+
return ()=>{};
|
|
2167
|
+
},
|
|
2168
|
+
get (name) {
|
|
2169
|
+
if (name === 'subprocess') return {
|
|
2170
|
+
spawn: (_spec)=>fakeHandle({
|
|
2171
|
+
stdout: '{}'
|
|
2172
|
+
})
|
|
2173
|
+
};
|
|
2174
|
+
if (name === 'agents') return {
|
|
2175
|
+
list () {
|
|
2176
|
+
return [];
|
|
2177
|
+
}
|
|
2178
|
+
};
|
|
2179
|
+
return undefined;
|
|
2180
|
+
}
|
|
2181
|
+
};
|
|
2182
|
+
apply(modelCtx, {
|
|
2183
|
+
stateRoot: root,
|
|
2184
|
+
timeoutMs: 20,
|
|
2185
|
+
hbcliBin: '',
|
|
2186
|
+
sessionAccess: 'off',
|
|
2187
|
+
benchmarkEnvironmentConfigPath: configPath
|
|
2188
|
+
});
|
|
2189
|
+
assert.deepEqual(modelTools.map((tool)=>tool.name), [
|
|
2190
|
+
'gotry_benchmark_environment'
|
|
2191
|
+
], 'benchmark model override does not re-enable product tools');
|
|
2192
|
+
assert.ok(modelEvents.includes('agent/request'), 'benchmark mode preserves the model override hook');
|
|
2193
|
+
assert.ok(modelRequest);
|
|
2194
|
+
assert.deepEqual(await modelRequest({}, async ()=>({
|
|
2195
|
+
provider: 'persisted',
|
|
2196
|
+
model: 'old',
|
|
2197
|
+
reasoningEffort: 'high',
|
|
2198
|
+
marker: 'kept'
|
|
2199
|
+
})), {
|
|
2200
|
+
provider: 'deepseek-official',
|
|
2201
|
+
model: 'round7-model-preserved',
|
|
2202
|
+
marker: 'kept'
|
|
2203
|
+
}, 'benchmark model override remains effective and replaces the persisted model only');
|
|
2204
|
+
} finally{
|
|
2205
|
+
if (originalModelOverride === undefined) delete process.env.GOTRY_LLM_MODEL;
|
|
2206
|
+
else process.env.GOTRY_LLM_MODEL = originalModelOverride;
|
|
2207
|
+
}
|
|
2208
|
+
const validConfig = {
|
|
2209
|
+
schema_version: 'gotry_benchmark_environment_bridge_v2',
|
|
2210
|
+
enabled: true,
|
|
2211
|
+
executable: process.execPath,
|
|
2212
|
+
cwd: root,
|
|
2213
|
+
argv_prefix: [
|
|
2214
|
+
'-m',
|
|
2215
|
+
'agent_env.cli',
|
|
2216
|
+
'--lang',
|
|
2217
|
+
'en'
|
|
2218
|
+
],
|
|
2219
|
+
allowed_tools: [
|
|
2220
|
+
'lookup'
|
|
2221
|
+
],
|
|
2222
|
+
timeout_ms: 20,
|
|
2223
|
+
max_output_bytes: 4_096,
|
|
2224
|
+
terminal_output: {
|
|
2225
|
+
tag: 'done',
|
|
2226
|
+
max_bytes: 4_096
|
|
2227
|
+
},
|
|
2228
|
+
isolation: {
|
|
2229
|
+
mode: 'host-enforced',
|
|
2230
|
+
writes: 'forbidden',
|
|
2231
|
+
network: 'denied'
|
|
2232
|
+
}
|
|
2233
|
+
};
|
|
2234
|
+
const frozenProjection = registerBenchmarkEnvironmentBridge(configPath, ()=>{}, {
|
|
2235
|
+
spawn: (_spec)=>fakeHandle({
|
|
2236
|
+
stdout: '{"result":{}}'
|
|
2237
|
+
})
|
|
2238
|
+
});
|
|
2239
|
+
assert.equal(Object.isFrozen(frozenProjection), true, 'bridge projection is frozen');
|
|
2240
|
+
assert.equal(Object.isFrozen(frozenProjection.allowedTools), true, 'projected allowlist is frozen');
|
|
2241
|
+
assert.equal(Object.isFrozen(frozenProjection.terminal), true, 'projected terminal contract is frozen');
|
|
2242
|
+
assert.throws(()=>{
|
|
2243
|
+
frozenProjection.allowedTools.push('escape');
|
|
2244
|
+
}, TypeError);
|
|
2245
|
+
assert.throws(()=>{
|
|
2246
|
+
frozenProjection.terminal.tag = 'escape';
|
|
2247
|
+
}, TypeError);
|
|
2248
|
+
const bridgeRegistrationFor = (config, withSubprocess = true)=>{
|
|
2249
|
+
const path = join(root, `config-${Math.random().toString(36).slice(2)}.json`);
|
|
2250
|
+
writeFileSync(path, JSON.stringify(config));
|
|
2251
|
+
const tools = [];
|
|
2252
|
+
const freshCtx = {
|
|
2253
|
+
tools: {
|
|
2254
|
+
register (tool) {
|
|
2255
|
+
tools.push(tool);
|
|
2256
|
+
return ()=>{};
|
|
2257
|
+
},
|
|
2258
|
+
get (name) {
|
|
2259
|
+
return tools.find((tool)=>tool.name === name);
|
|
2260
|
+
},
|
|
2261
|
+
schemas () {
|
|
2262
|
+
return tools.map((tool)=>({
|
|
2263
|
+
name: tool.name
|
|
2264
|
+
}));
|
|
2265
|
+
}
|
|
2266
|
+
},
|
|
2267
|
+
systemPrompt: {
|
|
2268
|
+
variable () {}
|
|
2269
|
+
},
|
|
2270
|
+
on () {
|
|
2271
|
+
return ()=>{};
|
|
2272
|
+
},
|
|
2273
|
+
effect (action) {
|
|
2274
|
+
return action();
|
|
2275
|
+
},
|
|
2276
|
+
agents: {
|
|
2277
|
+
list () {
|
|
2278
|
+
return [];
|
|
2279
|
+
}
|
|
2280
|
+
},
|
|
2281
|
+
get (name) {
|
|
2282
|
+
if (name === 'subprocess') return this.subprocess;
|
|
2283
|
+
if (name === 'agents') return this.agents;
|
|
2284
|
+
},
|
|
2285
|
+
...withSubprocess ? {
|
|
2286
|
+
subprocess: {
|
|
2287
|
+
spawn: (_spec)=>fakeHandle({
|
|
2288
|
+
stdout: '{"result":{}}'
|
|
2289
|
+
})
|
|
2290
|
+
}
|
|
2291
|
+
} : {}
|
|
2292
|
+
};
|
|
2293
|
+
apply(freshCtx, {
|
|
2294
|
+
stateRoot: root,
|
|
2295
|
+
timeoutMs: 20,
|
|
2296
|
+
hbcliBin: '',
|
|
2297
|
+
sessionAccess: 'off',
|
|
2298
|
+
benchmarkEnvironmentConfigPath: path
|
|
2299
|
+
});
|
|
2300
|
+
return tools.some((tool)=>tool.name === 'gotry_benchmark_environment');
|
|
2301
|
+
};
|
|
2302
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2303
|
+
...validConfig,
|
|
2304
|
+
unknown: true
|
|
2305
|
+
}), /benchmark environment bridge configuration unavailable/, 'unknown top-level config key fails hard');
|
|
2306
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2307
|
+
...validConfig,
|
|
2308
|
+
schema_version: 'gotry_benchmark_environment_bridge_v1'
|
|
2309
|
+
}), /benchmark environment bridge configuration unavailable/, 'v1 config cannot silently omit the Round 3 terminal semantics');
|
|
2310
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2311
|
+
...validConfig,
|
|
2312
|
+
allowed_tools: [
|
|
2313
|
+
'lookup',
|
|
2314
|
+
'lookup'
|
|
2315
|
+
]
|
|
2316
|
+
}), /benchmark environment bridge configuration unavailable/, 'duplicate allowed tool fails hard');
|
|
2317
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2318
|
+
...validConfig,
|
|
2319
|
+
allowed_output_keys: {}
|
|
2320
|
+
}), /benchmark environment bridge configuration unavailable/, 'empty output-key mapping fails hard');
|
|
2321
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2322
|
+
...validConfig,
|
|
2323
|
+
allowed_output_keys: {
|
|
2324
|
+
lookup: []
|
|
2325
|
+
}
|
|
2326
|
+
}), /benchmark environment bridge configuration unavailable/, 'empty output-key allowlist fails hard');
|
|
2327
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2328
|
+
...validConfig,
|
|
2329
|
+
allowed_output_keys: {
|
|
2330
|
+
unknown: [
|
|
2331
|
+
'city'
|
|
2332
|
+
]
|
|
2333
|
+
}
|
|
2334
|
+
}), /benchmark environment bridge configuration unavailable/, 'output-key mapping for unknown tool fails hard');
|
|
2335
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2336
|
+
...validConfig,
|
|
2337
|
+
allowed_output_keys: {
|
|
2338
|
+
lookup: [
|
|
2339
|
+
'city',
|
|
2340
|
+
'city'
|
|
2341
|
+
]
|
|
2342
|
+
}
|
|
2343
|
+
}), /benchmark environment bridge configuration unavailable/, 'duplicate output key fails hard');
|
|
2344
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2345
|
+
...validConfig,
|
|
2346
|
+
allowed_output_keys: {
|
|
2347
|
+
lookup: [
|
|
2348
|
+
'not a key'
|
|
2349
|
+
]
|
|
2350
|
+
}
|
|
2351
|
+
}), /benchmark environment bridge configuration unavailable/, 'non-identifier output key fails hard');
|
|
2352
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2353
|
+
...validConfig,
|
|
2354
|
+
argv_prefix: [
|
|
2355
|
+
'agent\n--unsafe'
|
|
2356
|
+
]
|
|
2357
|
+
}), /benchmark environment bridge configuration unavailable/, 'argv control separator fails hard');
|
|
2358
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2359
|
+
...validConfig,
|
|
2360
|
+
allowed_tools: [
|
|
2361
|
+
'lookup;rm'
|
|
2362
|
+
]
|
|
2363
|
+
}), /benchmark environment bridge configuration unavailable/, 'shell metacharacter tool identifier fails hard');
|
|
2364
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2365
|
+
...validConfig,
|
|
2366
|
+
isolation: {
|
|
2367
|
+
mode: 'host-enforced',
|
|
2368
|
+
writes: 'forbidden'
|
|
2369
|
+
}
|
|
2370
|
+
}), /benchmark environment bridge configuration unavailable/, 'incomplete isolation policy fails hard');
|
|
2371
|
+
const { terminal_output: _terminalOutput, ...missingTerminalConfig } = validConfig;
|
|
2372
|
+
assert.throws(()=>bridgeRegistrationFor(missingTerminalConfig), /benchmark environment bridge configuration unavailable/, 'terminal output contract is required');
|
|
2373
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2374
|
+
...validConfig,
|
|
2375
|
+
terminal_output: {
|
|
2376
|
+
tag: '1invalid',
|
|
2377
|
+
max_bytes: 4_096
|
|
2378
|
+
}
|
|
2379
|
+
}), /benchmark environment bridge configuration unavailable/, 'terminal tag must be an identifier');
|
|
2380
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2381
|
+
...validConfig,
|
|
2382
|
+
terminal_output: {
|
|
2383
|
+
tag: 'done',
|
|
2384
|
+
max_bytes: 0
|
|
2385
|
+
}
|
|
2386
|
+
}), /benchmark environment bridge configuration unavailable/, 'terminal output lower bound fails hard');
|
|
2387
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2388
|
+
...validConfig,
|
|
2389
|
+
terminal_output: {
|
|
2390
|
+
tag: 'done',
|
|
2391
|
+
max_bytes: 1024 * 1024 + 1
|
|
2392
|
+
}
|
|
2393
|
+
}), /benchmark environment bridge configuration unavailable/, 'terminal output upper bound fails hard');
|
|
2394
|
+
assert.throws(()=>bridgeRegistrationFor({
|
|
2395
|
+
...validConfig,
|
|
2396
|
+
terminal_output: {
|
|
2397
|
+
tag: 'done',
|
|
2398
|
+
max_bytes: 4_096,
|
|
2399
|
+
extra: true
|
|
2400
|
+
}
|
|
2401
|
+
}), /benchmark environment bridge configuration unavailable/, 'terminal output rejects unknown keys');
|
|
2402
|
+
assert.throws(()=>bridgeRegistrationFor(validConfig, false), /benchmark environment bridge subprocess unavailable/, 'explicit opt-in without an active subprocess provider fails hard');
|
|
2403
|
+
assert.equal(loadConfigRegistration(configPath, root), true, 'owner-local regular 0644 config remains valid');
|
|
2404
|
+
assert.throws(()=>loadConfigRegistration('bridge.json', root), /benchmark environment bridge configuration unavailable/, 'relative config path fails hard');
|
|
2405
|
+
const symlinkPath = join(root, 'bridge-symlink.json');
|
|
2406
|
+
symlinkSync(configPath, symlinkPath);
|
|
2407
|
+
assert.equal(lstatSync(symlinkPath).isSymbolicLink(), true);
|
|
2408
|
+
assert.throws(()=>loadConfigRegistration(symlinkPath, root), /benchmark environment bridge configuration unavailable/, 'symlink config path fails hard');
|
|
2409
|
+
const widePath = join(root, 'bridge-wide.json');
|
|
2410
|
+
writeFileSync(widePath, JSON.stringify(validConfig));
|
|
2411
|
+
chmodSync(widePath, 0o666);
|
|
2412
|
+
assert.throws(()=>loadConfigRegistration(widePath, root), /benchmark environment bridge configuration unavailable/, 'group/world writable config fails hard');
|
|
2413
|
+
const legacyConfigPath = join(root, 'legacy-bridge.json');
|
|
2414
|
+
writeFileSync(legacyConfigPath, JSON.stringify(validConfig));
|
|
2415
|
+
const legacyTools = [];
|
|
2416
|
+
const legacyCtx = {
|
|
2417
|
+
tools: {
|
|
2418
|
+
register (tool) {
|
|
2419
|
+
legacyTools.push(tool);
|
|
2420
|
+
return ()=>{};
|
|
2421
|
+
},
|
|
2422
|
+
get (name) {
|
|
2423
|
+
return legacyTools.find((tool)=>tool.name === name);
|
|
2424
|
+
},
|
|
2425
|
+
schemas () {
|
|
2426
|
+
return legacyTools.map((tool)=>({
|
|
2427
|
+
name: tool.name
|
|
2428
|
+
}));
|
|
2429
|
+
}
|
|
2430
|
+
},
|
|
2431
|
+
systemPrompt: {
|
|
2432
|
+
variable () {}
|
|
2433
|
+
},
|
|
2434
|
+
on () {
|
|
2435
|
+
return ()=>{};
|
|
2436
|
+
},
|
|
2437
|
+
effect (_action) {
|
|
2438
|
+
return ()=>{};
|
|
2439
|
+
},
|
|
2440
|
+
agents: {
|
|
2441
|
+
list () {
|
|
2442
|
+
return [];
|
|
2443
|
+
}
|
|
2444
|
+
},
|
|
2445
|
+
get (name) {
|
|
2446
|
+
if (name === 'subprocess') return this.subprocess;
|
|
2447
|
+
if (name === 'agents') return this.agents;
|
|
2448
|
+
},
|
|
2449
|
+
subprocess: {
|
|
2450
|
+
spawn: (_spec)=>fakeHandle({
|
|
2451
|
+
stdout: '{"result":{"city":"Dubai"}}'
|
|
2452
|
+
})
|
|
2453
|
+
}
|
|
2454
|
+
};
|
|
2455
|
+
apply(legacyCtx, {
|
|
2456
|
+
stateRoot: root,
|
|
2457
|
+
timeoutMs: 20,
|
|
2458
|
+
hbcliBin: '',
|
|
2459
|
+
sessionAccess: 'off',
|
|
2460
|
+
benchmarkEnvironmentConfigPath: legacyConfigPath
|
|
2461
|
+
});
|
|
2462
|
+
const legacyBridge = legacyTools.find((tool)=>tool.name === 'gotry_benchmark_environment');
|
|
2463
|
+
assert.ok(legacyBridge?.execute, 'legacy config without allowed_output_keys registers the bridge');
|
|
2464
|
+
assert.deepEqual(await legacyBridge.execute({
|
|
2465
|
+
query: {
|
|
2466
|
+
action: 'call',
|
|
2467
|
+
tool: 'lookup',
|
|
2468
|
+
arguments: {
|
|
2469
|
+
city: 'Dubai'
|
|
2470
|
+
}
|
|
2471
|
+
}
|
|
2472
|
+
}, null), {
|
|
2473
|
+
ok: true,
|
|
2474
|
+
result: {
|
|
2475
|
+
city: 'Dubai'
|
|
2476
|
+
}
|
|
2477
|
+
}, 'legacy config without allowed_output_keys still executes safe structured output');
|
|
2478
|
+
console.log('BENCHMARK ENVIRONMENT BRIDGE TESTS: registration + TDD bridge contract assertions');
|
|
2479
|
+
} catch (error) {
|
|
2480
|
+
console.error(error);
|
|
2481
|
+
process.exitCode = 1;
|
|
2482
|
+
} finally{
|
|
2483
|
+
for (const [name, value] of ambientSentinels){
|
|
2484
|
+
if (value === undefined) delete process.env[name];
|
|
2485
|
+
else process.env[name] = value;
|
|
2486
|
+
}
|
|
2487
|
+
rmSync(root, {
|
|
2488
|
+
recursive: true,
|
|
2489
|
+
force: true
|
|
2490
|
+
});
|
|
2491
|
+
}
|
|
2492
|
+
function loadConfigRegistration(path, stateRoot) {
|
|
2493
|
+
const tools = [];
|
|
2494
|
+
const freshCtx = {
|
|
2495
|
+
tools: {
|
|
2496
|
+
register (tool) {
|
|
2497
|
+
tools.push(tool);
|
|
2498
|
+
return ()=>{};
|
|
2499
|
+
},
|
|
2500
|
+
get (name) {
|
|
2501
|
+
return tools.find((tool)=>tool.name === name);
|
|
2502
|
+
},
|
|
2503
|
+
schemas () {
|
|
2504
|
+
return tools.map((tool)=>({
|
|
2505
|
+
name: tool.name
|
|
2506
|
+
}));
|
|
2507
|
+
}
|
|
2508
|
+
},
|
|
2509
|
+
systemPrompt: {
|
|
2510
|
+
variable () {}
|
|
2511
|
+
},
|
|
2512
|
+
agents: {
|
|
2513
|
+
list () {
|
|
2514
|
+
return [];
|
|
2515
|
+
}
|
|
2516
|
+
},
|
|
2517
|
+
on () {
|
|
2518
|
+
return ()=>{};
|
|
2519
|
+
},
|
|
2520
|
+
effect (_action) {
|
|
2521
|
+
return ()=>{};
|
|
2522
|
+
},
|
|
2523
|
+
get (name) {
|
|
2524
|
+
if (name === 'subprocess') return this.subprocess;
|
|
2525
|
+
if (name === 'agents') return this.agents;
|
|
2526
|
+
},
|
|
2527
|
+
subprocess: {
|
|
2528
|
+
spawn: (_spec)=>fakeHandle({
|
|
2529
|
+
stdout: '{"result":{}}'
|
|
2530
|
+
})
|
|
2531
|
+
}
|
|
2532
|
+
};
|
|
2533
|
+
apply(freshCtx, {
|
|
2534
|
+
stateRoot,
|
|
2535
|
+
timeoutMs: 20,
|
|
2536
|
+
hbcliBin: '',
|
|
2537
|
+
sessionAccess: 'off',
|
|
2538
|
+
benchmarkEnvironmentConfigPath: path
|
|
2539
|
+
});
|
|
2540
|
+
return tools.some((tool)=>tool.name === 'gotry_benchmark_environment');
|
|
2541
|
+
}
|
|
2542
|
+
|
|
2543
|
+
|
|
2544
|
+
//# sourceURL=ts/scripts/benchmark-environment-bridge-tests.ts
|