@danceiny/gotry 0.0.1-rc.16 → 0.0.1-rc.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (229) hide show
  1. package/README.md +47 -13
  2. package/README.zh-CN.md +20 -10
  3. package/bin/gotry-booking-copilot.js +53 -0
  4. package/bin/gotry-bootstrap.js +111 -51
  5. package/bin/gotry-inner.js +442 -60
  6. package/bin/gotry-runtime-resolution.d.ts +27 -0
  7. package/bin/gotry-runtime-resolution.js +50 -0
  8. package/bin/gotry.js +1 -1
  9. package/cordis.gotry-patch.yml +5 -1
  10. package/dist/capabilities/agent-reach-deep.js +1 -1
  11. package/dist/capabilities/agent-reach.js +1 -1
  12. package/dist/capabilities/anything.js +1 -1
  13. package/dist/capabilities/artifacts.js +1 -1
  14. package/dist/capabilities/effect.js +1 -1
  15. package/dist/capabilities/fact-log.js +1 -1
  16. package/dist/capabilities/flyai.js +20 -6
  17. package/dist/capabilities/hbcli.js +1 -1
  18. package/dist/capabilities/incident-log.js +1 -1
  19. package/dist/capabilities/model-override.js +18 -0
  20. package/dist/capabilities/opensky.js +1 -1
  21. package/dist/capabilities/resilience.js +1 -1
  22. package/dist/capabilities/session/action-cache.js +1 -1
  23. package/dist/capabilities/session/adapters/ctrip-flight.js +1 -1
  24. package/dist/capabilities/session/adapters/meituan-local.js +1 -1
  25. package/dist/capabilities/session/benchmark.js +1 -1
  26. package/dist/capabilities/session/extension-bridge.js +19 -6
  27. package/dist/capabilities/session/extension-channel.js +3 -3
  28. package/dist/capabilities/session/extension-distribution.js +234 -0
  29. package/dist/capabilities/session/extract.js +1 -1
  30. package/dist/capabilities/session/golden-score.js +92 -0
  31. package/dist/capabilities/session/health-watch.js +1 -1
  32. package/dist/capabilities/session/read-guard.js +1 -1
  33. package/dist/capabilities/session/static-flight-golden.js +137 -0
  34. package/dist/capabilities/session/transport.js +1 -1
  35. package/dist/capabilities/session/wizard.js +21 -251
  36. package/dist/capabilities/session-consent.js +1 -1
  37. package/dist/capabilities/session-login.js +11 -5
  38. package/dist/capabilities/session-search.js +46 -4
  39. package/dist/capabilities/weather.js +168 -46
  40. package/dist/data/session-golden-20.json +25 -0
  41. package/dist/data/sf-golden-manifest.json +102 -0
  42. package/dist/data/sf-static-routes.json +91 -0
  43. package/dist/scripts/action-cache-tests.js +1 -1
  44. package/dist/scripts/agent-planning-budget-e2e.js +227 -0
  45. package/dist/scripts/agent-planning-budget-tests.js +173 -0
  46. package/dist/scripts/agent-reach-deep-tests.js +1 -1
  47. package/dist/scripts/agent-reach-tests.js +1 -1
  48. package/dist/scripts/agent-reach-wrapper-tests.js +1 -1
  49. package/dist/scripts/anything-tests.js +1 -1
  50. package/dist/scripts/async-collect.js +1 -1
  51. package/dist/scripts/benchmark-environment-bridge-e2e.js +981 -0
  52. package/dist/scripts/benchmark-environment-bridge-tests.js +2544 -0
  53. package/dist/scripts/booking-copilot-availability-ledger-binding-tests.js +335 -0
  54. package/dist/scripts/booking-copilot-availability-policy-v2-tests.js +1355 -0
  55. package/dist/scripts/booking-copilot-bin-proof-tests.js +89 -0
  56. package/dist/scripts/booking-copilot-crossrepo-fixture-server.js +113 -0
  57. package/dist/scripts/booking-copilot-dsh-core-proof-tests.js +356 -0
  58. package/dist/scripts/booking-copilot-dsh-planner-proof-tests.js +557 -0
  59. package/dist/scripts/booking-copilot-dsh-plugin-proof-tests.js +140 -0
  60. package/dist/scripts/booking-copilot-event-sequence-concurrency-proof-tests.js +202 -0
  61. package/dist/scripts/booking-copilot-gap-code-contract-proof-tests.js +102 -0
  62. package/dist/scripts/booking-copilot-operation-ledger-concurrency-proof-tests.js +374 -0
  63. package/dist/scripts/booking-copilot-receipt-ledger-concurrency-proof-tests.js +447 -0
  64. package/dist/scripts/booking-copilot-runtime-proof-tests.js +205 -0
  65. package/dist/scripts/booking-copilot-server-proof-tests.js +317 -0
  66. package/dist/scripts/booking-copilot-startup-proof-tests.js +283 -0
  67. package/dist/scripts/booking-copilot-v2-runtime-proof-tests.js +3935 -0
  68. package/dist/scripts/booking-saga-tests.js +1 -1
  69. package/dist/scripts/booking-surface-contract-proof-tests.js +393 -0
  70. package/dist/scripts/booking-surface-v2-contract-proof-tests.js +1174 -0
  71. package/dist/scripts/bootstrap-tests.js +30 -10
  72. package/dist/scripts/build-changelog.js +1 -1
  73. package/dist/scripts/changelog-tests.js +1 -1
  74. package/dist/scripts/companion-tests.js +1 -1
  75. package/dist/scripts/diff-test.js +1 -1
  76. package/dist/scripts/dsh-runtime-closure-tests.js +225 -0
  77. package/dist/scripts/dsh-runtime-closure.js +150 -0
  78. package/dist/scripts/effect-tests.js +1 -1
  79. package/dist/scripts/engine-run.js +1 -1
  80. package/dist/scripts/engine-tests.js +1 -1
  81. package/dist/scripts/evaluation-cadence-tests.js +347 -0
  82. package/dist/scripts/evaluation-contract-tests.js +574 -0
  83. package/dist/scripts/extension-distribution-cli.js +35 -0
  84. package/dist/scripts/extension-distribution-tests.js +341 -0
  85. package/dist/scripts/extension-tests.js +128 -23
  86. package/dist/scripts/fact-gate-tests.js +1 -1
  87. package/dist/scripts/flyai-tests.js +1 -1
  88. package/dist/scripts/hbcli-e2e-tests.js +1 -1
  89. package/dist/scripts/hbcli-tests.js +1 -1
  90. package/dist/scripts/health-watch-cli.js +1 -1
  91. package/dist/scripts/i18n-tests.js +1 -1
  92. package/dist/scripts/incident-tests.js +1 -1
  93. package/dist/scripts/journey-tests.js +1 -1
  94. package/dist/scripts/ledger-tests.js +1 -1
  95. package/dist/scripts/ledger-workflow-crash.js +1 -1
  96. package/dist/scripts/memory-capture-tests.js +1 -1
  97. package/dist/scripts/memory-decay-tests.js +1 -1
  98. package/dist/scripts/memory-metrics.js +1 -1
  99. package/dist/scripts/memory-value-report.js +1 -1
  100. package/dist/scripts/model-override-e2e.js +176 -0
  101. package/dist/scripts/nightly-evidence-tests.js +1 -1
  102. package/dist/scripts/nightly-evidence.js +1 -1
  103. package/dist/scripts/nudge-digest.js +1 -1
  104. package/dist/scripts/onboarding-tests.js +21 -53
  105. package/dist/scripts/opensky-check.js +1 -1
  106. package/dist/scripts/opensky-tests.js +1 -1
  107. package/dist/scripts/pnpm-dsh-closure-proof.js +20 -0
  108. package/dist/scripts/price-drift-tests.js +1 -1
  109. package/dist/scripts/price-drift-watch.js +1 -1
  110. package/dist/scripts/probe-poi-tests.js +1 -1
  111. package/dist/scripts/product-metrics.js +1 -1
  112. package/dist/scripts/publish-preverify.js +40 -4
  113. package/dist/scripts/realtime-pricing-tests.js +1 -1
  114. package/dist/scripts/replay-async.js +1 -1
  115. package/dist/scripts/replay-real.js +1 -1
  116. package/dist/scripts/replay.js +1 -1
  117. package/dist/scripts/session-attach-diagnose.js +1 -1
  118. package/dist/scripts/session-attach-poc.js +1 -1
  119. package/dist/scripts/session-benchmark.js +1 -1
  120. package/dist/scripts/session-extract-tests.js +1 -1
  121. package/dist/scripts/session-login.js +1 -1
  122. package/dist/scripts/session-tests.js +70 -20
  123. package/dist/scripts/sf-live-benchmark.js +338 -0
  124. package/dist/scripts/sf-live-cli-tests.js +21 -0
  125. package/dist/scripts/sf-soft-score-tests.js +108 -0
  126. package/dist/scripts/sf-summary.js +93 -0
  127. package/dist/scripts/skeleton-check.js +1 -1
  128. package/dist/scripts/skeleton-integration-test.js +1 -1
  129. package/dist/scripts/skills-contract-tests.js +1 -1
  130. package/dist/scripts/smoke-session-gate-tests.js +29 -0
  131. package/dist/scripts/smoke.js +79 -33
  132. package/dist/scripts/state-cli-tests.js +1 -1
  133. package/dist/scripts/state-cli.js +1 -1
  134. package/dist/scripts/static-golden-tests.js +299 -0
  135. package/dist/scripts/time-eval-tests.js +1 -1
  136. package/dist/scripts/travel-timeline-tests.js +1 -1
  137. package/dist/scripts/unified-tests.js +1 -1
  138. package/dist/scripts/weather-tests.js +694 -44
  139. package/dist/scripts/z3-race-tests.js +1 -1
  140. package/dist/src/artifact-gate.js +1 -1
  141. package/dist/src/benchmark-agent-conformance.js +370 -0
  142. package/dist/src/benchmark-environment-bridge.js +384 -0
  143. package/dist/src/benchmark-headless-child-diagnostics.js +173 -0
  144. package/dist/src/benchmark-tool-isolation.js +124 -0
  145. package/dist/src/bookable-facts.js +1 -1
  146. package/dist/src/booking-saga.js +1 -1
  147. package/dist/src/booking-surface/availability-policy-v2.js +830 -0
  148. package/dist/src/booking-surface/canonical-schema.js +113 -0
  149. package/dist/src/booking-surface/contracts-v2.js +89 -0
  150. package/dist/src/booking-surface/contracts.js +46 -0
  151. package/dist/src/booking-surface/dsh-planner.js +453 -0
  152. package/dist/src/booking-surface/dsh-plugin.js +93 -0
  153. package/dist/src/booking-surface/error-codes.js +94 -0
  154. package/dist/src/booking-surface/index.js +15 -0
  155. package/dist/src/booking-surface/profile.js +68 -0
  156. package/dist/src/booking-surface/runtime-v2.js +1771 -0
  157. package/dist/src/booking-surface/runtime.js +351 -0
  158. package/dist/src/booking-surface/server-v2.js +334 -0
  159. package/dist/src/booking-surface/server.js +302 -0
  160. package/dist/src/booking-surface/startup.js +159 -0
  161. package/dist/src/booking-surface/validation-v2.js +319 -0
  162. package/dist/src/booking-surface/validation.js +809 -0
  163. package/dist/src/bridge.js +1 -1
  164. package/dist/src/companions.js +1 -1
  165. package/dist/src/contracts.js +1 -1
  166. package/dist/src/dsh-llm.js +1 -1
  167. package/dist/src/engine.js +1 -1
  168. package/dist/src/evaluation-cadence.js +234 -0
  169. package/dist/src/evaluation-contracts.js +906 -0
  170. package/dist/src/i18n.js +1 -1
  171. package/dist/src/index.js +50 -19
  172. package/dist/src/journey.js +1 -1
  173. package/dist/src/loop.js +1 -1
  174. package/dist/src/memory-capture.js +1 -1
  175. package/dist/src/memory-decay.js +1 -1
  176. package/dist/src/memory-utility.js +1 -1
  177. package/dist/src/mock-llm.js +1 -1
  178. package/dist/src/model.js +1 -1
  179. package/dist/src/realtime-pricing.js +1 -1
  180. package/dist/src/slot-spec.js +1 -1
  181. package/dist/src/state-ledger.js +2 -1
  182. package/dist/src/time-anchor.js +1 -1
  183. package/dist/src/tool-budget.js +136 -0
  184. package/dist/src/tool-packet.js +1 -1
  185. package/dist/src/travel-slots.js +1 -1
  186. package/dist/src/travel-timeline.js +1 -1
  187. package/dist/src/unified.js +1 -1
  188. package/dist/src/wish-pool.js +1 -1
  189. package/dist/src/z3-shared.js +1 -1
  190. package/extension/README.md +31 -7
  191. package/package.json +286 -11
  192. package/schemas/booking.surface.v1.schema.json +927 -0
  193. package/schemas/booking.surface.v2.schema.json +61 -0
  194. package/ts/capabilities/flyai.ts +16 -3
  195. package/ts/capabilities/session/extension-bridge.ts +37 -14
  196. package/ts/capabilities/session/extension-channel.ts +6 -3
  197. package/ts/capabilities/session/extension-distribution.ts +264 -0
  198. package/ts/capabilities/session/golden-score.ts +139 -0
  199. package/ts/capabilities/session/health-watch.ts +1 -1
  200. package/ts/capabilities/session/static-flight-golden.ts +209 -0
  201. package/ts/capabilities/session/wizard.ts +34 -176
  202. package/ts/capabilities/session-login.ts +15 -5
  203. package/ts/capabilities/session-search.ts +40 -3
  204. package/ts/capabilities/weather.ts +141 -52
  205. package/ts/package.json +3 -3
  206. package/ts/src/benchmark-agent-conformance.ts +448 -0
  207. package/ts/src/benchmark-environment-bridge.ts +348 -0
  208. package/ts/src/benchmark-headless-child-diagnostics.ts +184 -0
  209. package/ts/src/benchmark-tool-isolation.ts +166 -0
  210. package/ts/src/booking-surface/availability-policy-v2.ts +523 -0
  211. package/ts/src/booking-surface/canonical-schema.js +113 -0
  212. package/ts/src/booking-surface/contracts-v2.ts +118 -0
  213. package/ts/src/booking-surface/contracts.ts +380 -0
  214. package/ts/src/booking-surface/dsh-planner.ts +452 -0
  215. package/ts/src/booking-surface/dsh-plugin.js +93 -0
  216. package/ts/src/booking-surface/error-codes.ts +101 -0
  217. package/ts/src/booking-surface/index.ts +12 -0
  218. package/ts/src/booking-surface/profile.ts +42 -0
  219. package/ts/src/booking-surface/runtime-v2.ts +1466 -0
  220. package/ts/src/booking-surface/runtime.ts +483 -0
  221. package/ts/src/booking-surface/server-v2.ts +247 -0
  222. package/ts/src/booking-surface/server.ts +324 -0
  223. package/ts/src/booking-surface/startup.ts +196 -0
  224. package/ts/src/booking-surface/validation-v2.ts +205 -0
  225. package/ts/src/booking-surface/validation.ts +453 -0
  226. package/ts/src/index.ts +64 -11
  227. package/ts/src/state-ledger.ts +1 -0
  228. package/ts/src/tool-budget.ts +165 -0
  229. package/dist/scripts/wizard-bootstrap.js +0 -32
@@ -0,0 +1,2544 @@
1
+ import assert from 'node:assert/strict';
2
+ import { chmodSync, lstatSync, mkdtempSync, readFileSync, rmSync, symlinkSync, writeFileSync } from 'node:fs';
3
+ import { tmpdir } from 'node:os';
4
+ import { join } from 'node:path';
5
+ import { Context } from '@deepseek-ai/cordis';
6
+ import { apply } from '../src/index.js';
7
+ import { registerBenchmarkEnvironmentBridge } from '../src/benchmark-environment-bridge.js';
8
+ import { installBenchmarkToolIsolation } from '../src/benchmark-tool-isolation.js';
9
+ import { BENCHMARK_BRIDGE_CALL_FAILED, BENCHMARK_BRIDGE_CALL_REQUIRED, BENCHMARK_BRIDGE_RETRY_CALL_NOT_ALLOWED, BENCHMARK_BRIDGE_OUTPUT_TRUNCATED, BENCHMARK_BRIDGE_RUNNER_FAILED, BENCHMARK_BRIDGE_SPAWN_FAILED, BENCHMARK_BRIDGE_TIMED_OUT, BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE, BENCHMARK_TERMINAL_INVALID, MAX_CONFORMANCE_RETRIES, benchmarkChildFailureForConformanceCode, createBenchmarkAgentConformance, installBenchmarkAgentConformance, parseBenchmarkTerminal, validateTerminalOutputConfig } from '../src/benchmark-agent-conformance.js';
10
+ import { BENCHMARK_CHILD_DIAGNOSTIC_MAX_BYTES, BENCHMARK_CHILD_DIAGNOSTIC_SCHEMA, appendBoundedChildDiagnostic, classifyBenchmarkChildFailure, classifyBenchmarkTurnEnd, createBenchmarkDiagnosticArbiter, parseBenchmarkChildDiagnostic } from '../src/benchmark-headless-child-diagnostics.js';
11
+ {
12
+ const exactFamilies = [
13
+ [
14
+ [
15
+ 'AUTH',
16
+ 'INVALID_CREDENTIAL',
17
+ 'MISSING_CREDENTIAL'
18
+ ],
19
+ 'child_model_auth'
20
+ ],
21
+ [
22
+ [
23
+ 'QUOTA',
24
+ 'RATE_LIMIT'
25
+ ],
26
+ 'child_model_capacity'
27
+ ],
28
+ [
29
+ [
30
+ 'SERVER'
31
+ ],
32
+ 'child_model_server'
33
+ ],
34
+ [
35
+ [
36
+ 'TRANSPORT',
37
+ 'TIMEOUT'
38
+ ],
39
+ 'child_model_transport'
40
+ ],
41
+ [
42
+ [
43
+ 'EMPTY_RESPONSE',
44
+ 'STREAM_CLOSED',
45
+ 'MALFORMED_RESPONSE',
46
+ 'INVALID_RESPONSE'
47
+ ],
48
+ 'child_model_stream'
49
+ ],
50
+ [
51
+ [
52
+ 'INVALID_REQUEST',
53
+ 'CONTEXT_WINDOW_EXCEEDED',
54
+ 'NO_ADAPTER',
55
+ 'UNKNOWN_MODEL',
56
+ 'UNSUPPORTED_OPTION'
57
+ ],
58
+ 'child_model_request'
59
+ ],
60
+ [
61
+ [
62
+ 'ABORTED'
63
+ ],
64
+ 'child_aborted'
65
+ ]
66
+ ];
67
+ for (const [codes, expected] of exactFamilies){
68
+ for (const code of codes)assert.equal(classifyBenchmarkTurnEnd({
69
+ kind: 'error',
70
+ error: {
71
+ code
72
+ }
73
+ }), expected);
74
+ }
75
+ assert.equal(classifyBenchmarkTurnEnd({
76
+ kind: 'error',
77
+ error: {
78
+ code: 'RATE_LIMIT',
79
+ message: 'sentinel'
80
+ }
81
+ }), 'child_model_capacity');
82
+ assert.equal(classifyBenchmarkTurnEnd({
83
+ kind: 'error',
84
+ error: {
85
+ code: 'AUTH',
86
+ message: 'sentinel'
87
+ }
88
+ }), 'child_model_auth');
89
+ assert.equal(classifyBenchmarkTurnEnd({
90
+ kind: 'error',
91
+ error: {
92
+ code: 'MISSING_CREDENTIAL',
93
+ message: 'sentinel'
94
+ }
95
+ }), 'child_model_auth');
96
+ assert.equal(classifyBenchmarkTurnEnd({
97
+ kind: 'error',
98
+ error: {
99
+ code: 'QUOTA',
100
+ message: 'sentinel'
101
+ }
102
+ }), 'child_model_capacity');
103
+ assert.equal(classifyBenchmarkTurnEnd({
104
+ kind: 'error',
105
+ error: {
106
+ code: 'SERVER',
107
+ message: 'sentinel'
108
+ }
109
+ }), 'child_model_server');
110
+ assert.equal(classifyBenchmarkTurnEnd({
111
+ kind: 'error',
112
+ error: {
113
+ code: 'TRANSPORT',
114
+ message: 'api-key sentinel'
115
+ }
116
+ }), 'child_model_transport');
117
+ assert.equal(classifyBenchmarkTurnEnd({
118
+ kind: 'error',
119
+ error: {
120
+ code: 'OTHER',
121
+ status: 401,
122
+ message: 'key sentinel'
123
+ }
124
+ }), 'child_model_auth');
125
+ assert.equal(classifyBenchmarkTurnEnd({
126
+ kind: 'error',
127
+ error: {
128
+ code: 'OTHER',
129
+ status: 403,
130
+ message: 'key sentinel'
131
+ }
132
+ }), 'child_model_auth');
133
+ assert.equal(classifyBenchmarkTurnEnd({
134
+ kind: 'error',
135
+ error: {
136
+ code: 'OTHER',
137
+ status: 429,
138
+ message: 'quota sentinel'
139
+ }
140
+ }), 'child_model_capacity');
141
+ assert.equal(classifyBenchmarkTurnEnd({
142
+ kind: 'error',
143
+ error: {
144
+ code: 'OTHER',
145
+ status: 503,
146
+ message: 'server sentinel'
147
+ }
148
+ }), 'child_model_server');
149
+ assert.equal(classifyBenchmarkTurnEnd({
150
+ kind: 'error',
151
+ error: {
152
+ code: 'OTHER',
153
+ status: 500,
154
+ message: 'server sentinel'
155
+ }
156
+ }), 'child_model_server');
157
+ assert.equal(classifyBenchmarkTurnEnd({
158
+ kind: 'error',
159
+ error: {
160
+ code: 'OTHER',
161
+ status: 599,
162
+ message: 'server sentinel'
163
+ }
164
+ }), 'child_model_server');
165
+ assert.equal(classifyBenchmarkTurnEnd({
166
+ kind: 'error',
167
+ error: {
168
+ code: 'PI_AI_ERROR',
169
+ message: 'opaque'
170
+ }
171
+ }), 'child_runtime_error');
172
+ assert.equal(classifyBenchmarkTurnEnd({
173
+ kind: 'error',
174
+ error: {
175
+ code: 'INVALID_RESPONSE',
176
+ message: 'opaque'
177
+ }
178
+ }), 'child_model_stream');
179
+ assert.equal(classifyBenchmarkTurnEnd({
180
+ kind: 'error',
181
+ error: {
182
+ code: 'UNSUPPORTED_OPTION',
183
+ message: 'opaque'
184
+ }
185
+ }), 'child_model_request');
186
+ assert.equal(classifyBenchmarkTurnEnd({
187
+ kind: 'error',
188
+ error: {
189
+ code: 'OTHER',
190
+ status: 503.5,
191
+ message: 'opaque'
192
+ }
193
+ }), 'child_runtime_error');
194
+ assert.equal(classifyBenchmarkTurnEnd({
195
+ kind: 'error',
196
+ error: {
197
+ code: 'OTHER',
198
+ status: 99,
199
+ message: 'opaque'
200
+ }
201
+ }), 'child_runtime_error');
202
+ assert.equal(classifyBenchmarkTurnEnd({
203
+ kind: 'error',
204
+ error: {
205
+ code: 'OTHER',
206
+ status: 600,
207
+ message: 'opaque'
208
+ }
209
+ }), 'child_runtime_error');
210
+ assert.equal(classifyBenchmarkTurnEnd({
211
+ kind: 'error',
212
+ error: {
213
+ code: 'rate_limit',
214
+ message: 'opaque'
215
+ }
216
+ }), 'child_runtime_error');
217
+ assert.equal(classifyBenchmarkTurnEnd({
218
+ kind: 'error',
219
+ error: {
220
+ code: 'UNKNOWN',
221
+ message: 'opaque'
222
+ }
223
+ }), 'child_runtime_error');
224
+ assert.equal(classifyBenchmarkTurnEnd({
225
+ kind: 'error',
226
+ error: {
227
+ code: 'UNKNOWN',
228
+ message: 'message sentinel',
229
+ requestId: 'request sentinel',
230
+ path: 'path sentinel',
231
+ prompt: 'prompt sentinel',
232
+ key: 'key sentinel'
233
+ }
234
+ }), 'child_runtime_error');
235
+ assert.equal(classifyBenchmarkTurnEnd({
236
+ kind: 'blocked'
237
+ }), 'child_blocked');
238
+ assert.equal(classifyBenchmarkTurnEnd({
239
+ kind: 'max-tokens'
240
+ }), 'child_max_tokens');
241
+ assert.equal(classifyBenchmarkTurnEnd({
242
+ kind: 'aborted'
243
+ }), 'child_aborted');
244
+ assert.equal(classifyBenchmarkTurnEnd({
245
+ kind: 'interrupted'
246
+ }), 'child_interrupted');
247
+ assert.equal(classifyBenchmarkTurnEnd({
248
+ kind: 'completed'
249
+ }), undefined);
250
+ const writes = [];
251
+ const arbiter = createBenchmarkDiagnosticArbiter((code)=>writes.push(code));
252
+ arbiter.offer('session-a', 'child_conformance_failure');
253
+ arbiter.offer('session-a', 'child_runtime_error');
254
+ arbiter.offer('session-a', 'child_bridge_failure');
255
+ arbiter.offer('session-b', 'child_model_server');
256
+ arbiter.flush('session-a');
257
+ arbiter.flush('session-a');
258
+ arbiter.flush('session-b');
259
+ arbiter.flush('session-b');
260
+ assert.deepEqual(writes, [
261
+ 'child_bridge_failure',
262
+ 'child_model_server'
263
+ ]);
264
+ }assert.equal(MAX_CONFORMANCE_RETRIES, 1);
265
+ const projection = {
266
+ toolName: 'gotry_benchmark_environment',
267
+ allowedTools: [
268
+ 'lookup'
269
+ ],
270
+ terminal: {
271
+ tag: 'done',
272
+ max_bytes: 1024
273
+ }
274
+ };
275
+ assert.equal(validateTerminalOutputConfig(projection.terminal), true);
276
+ for (const invalid of [
277
+ null,
278
+ {
279
+ tag: 'done'
280
+ },
281
+ {
282
+ tag: '1bad',
283
+ max_bytes: 1024
284
+ },
285
+ {
286
+ tag: 'done',
287
+ max_bytes: 0
288
+ },
289
+ {
290
+ tag: 'done',
291
+ max_bytes: 1024 * 1024 + 1
292
+ },
293
+ {
294
+ tag: 'done',
295
+ max_bytes: 1024,
296
+ extra: true
297
+ }
298
+ ])assert.equal(validateTerminalOutputConfig(invalid), false);
299
+ assert.deepEqual(parseBenchmarkTerminal(' \n<done>{"ok":true}</done>\n', projection.terminal), {
300
+ ok: true,
301
+ value: {
302
+ ok: true
303
+ }
304
+ });
305
+ for (const invalid of [
306
+ 'prose <done>{"ok":true}</done>',
307
+ '<done>{"ok":true}</done> trailing',
308
+ '<done>```json\n{"ok":true}\n```</done>',
309
+ '<done>{"ok":true}</done><done>{"ok":true}</done>',
310
+ '<done>{"value":"</done><done>"}</done>',
311
+ '<wrong>{"ok":true}</wrong>',
312
+ '<done>[{"ok":true}]</done>',
313
+ '<done>true</done>',
314
+ '<done>{"ok":</done>'
315
+ ])assert.equal(parseBenchmarkTerminal(invalid, projection.terminal).ok, false);
316
+ assert.equal(parseBenchmarkTerminal(`<done>{"x":"${'y'.repeat(1024)}"}</done>`, projection.terminal).ok, false);
317
+ function turnStart(turn = 1) {
318
+ return {
319
+ type: 'turn/start',
320
+ data: {
321
+ turn
322
+ }
323
+ };
324
+ }
325
+ function turnEnd(turn = 1) {
326
+ return {
327
+ type: 'turn/end',
328
+ data: {
329
+ turn,
330
+ reason: {
331
+ kind: 'completed'
332
+ }
333
+ }
334
+ };
335
+ }
336
+ function toolCall(callId = 'call-1', options = {}) {
337
+ const { turn = 1, step = 1, action = 'call', tool = 'lookup' } = options;
338
+ return {
339
+ type: 'tool/call',
340
+ data: {
341
+ turn,
342
+ step,
343
+ callId,
344
+ name: projection.toolName,
345
+ arguments: JSON.stringify({
346
+ query: {
347
+ action,
348
+ tool,
349
+ arguments: {}
350
+ }
351
+ })
352
+ }
353
+ };
354
+ }
355
+ function toolResult(callId = 'call-1', options = {}) {
356
+ const { turn = 1, step = 1, ok = true, isError = false, error = 'runner_failed' } = options;
357
+ return {
358
+ type: 'tool/result',
359
+ data: {
360
+ turn,
361
+ step,
362
+ message: {
363
+ source: {
364
+ kind: 'tool',
365
+ callId
366
+ },
367
+ content: [
368
+ {
369
+ type: 'tool-result',
370
+ toolCallId: callId,
371
+ isError,
372
+ content: [
373
+ {
374
+ type: 'text',
375
+ text: JSON.stringify(ok ? {
376
+ ok: true,
377
+ result: {}
378
+ } : {
379
+ ok: false,
380
+ error
381
+ })
382
+ }
383
+ ]
384
+ }
385
+ ]
386
+ }
387
+ }
388
+ };
389
+ }
390
+ function assistant(text, options = {}) {
391
+ const { turn = 1, step = 2, interrupted = false } = options;
392
+ return {
393
+ type: 'assistant/message',
394
+ data: {
395
+ turn,
396
+ step,
397
+ message: {
398
+ content: [
399
+ {
400
+ type: 'text',
401
+ text
402
+ }
403
+ ]
404
+ },
405
+ ...interrupted ? {
406
+ interrupted: true
407
+ } : {}
408
+ }
409
+ };
410
+ }
411
+ {
412
+ const state = createBenchmarkAgentConformance(projection);
413
+ state.observe(turnStart());
414
+ state.observe(assistant('I would run the CLI.', {
415
+ step: 1
416
+ }));
417
+ assert.deepEqual(state.stopping(1), {
418
+ kind: 'steer',
419
+ mode: 'call'
420
+ }, 'no-call first stop gets one correction');
421
+ assert.equal(state.guardBridgeExecution(), undefined, 'call correction still permits the first real bridge dispatch');
422
+ state.observe(toolCall('call-a', {
423
+ step: 2
424
+ }));
425
+ state.observe(toolResult('call-a', {
426
+ step: 2
427
+ }));
428
+ state.observe(assistant('<done>{"status":"succeeded"}</done>', {
429
+ step: 3
430
+ }));
431
+ assert.deepEqual(state.stopping(1), {
432
+ kind: 'accept'
433
+ }, 'call correction may converge to one successful terminal');
434
+ }{
435
+ const state = createBenchmarkAgentConformance(projection);
436
+ state.observe(turnStart());
437
+ state.observe(assistant('<done>{"status":"succeeded"}</done>', {
438
+ step: 1
439
+ }));
440
+ assert.deepEqual(state.stopping(1), {
441
+ kind: 'steer',
442
+ mode: 'call'
443
+ }, 'valid terminal without a call still needs a call');
444
+ state.observe(assistant('<done>{"status":"succeeded"}</done>', {
445
+ step: 2
446
+ }));
447
+ assert.deepEqual(state.stopping(1), {
448
+ kind: 'reject',
449
+ code: BENCHMARK_BRIDGE_CALL_REQUIRED
450
+ });
451
+ }{
452
+ const state = createBenchmarkAgentConformance(projection);
453
+ state.observe(turnStart());
454
+ state.observe(toolCall());
455
+ state.observe(toolResult());
456
+ state.observe(assistant('bad terminal'));
457
+ assert.deepEqual(state.stopping(1), {
458
+ kind: 'steer',
459
+ mode: 'terminal'
460
+ }, 'bad terminal gets one format-only correction');
461
+ assert.equal(state.guardBridgeExecution(), BENCHMARK_BRIDGE_RETRY_CALL_NOT_ALLOWED);
462
+ state.observe(assistant('<done>{"status":"succeeded"}</done>', {
463
+ step: 3
464
+ }));
465
+ assert.deepEqual(state.stopping(1), {
466
+ kind: 'accept'
467
+ }, 'format-only correction can reuse the successful result');
468
+ }{
469
+ const state = createBenchmarkAgentConformance(projection);
470
+ state.observe(turnStart());
471
+ state.observe(toolCall());
472
+ state.observe(toolResult());
473
+ state.observe(assistant('bad terminal'));
474
+ assert.deepEqual(state.stopping(1), {
475
+ kind: 'steer',
476
+ mode: 'terminal'
477
+ });
478
+ state.observe(toolCall('call-2', {
479
+ step: 3
480
+ }));
481
+ assert.deepEqual(state.stopping(1), {
482
+ kind: 'reject',
483
+ code: BENCHMARK_BRIDGE_RETRY_CALL_NOT_ALLOWED
484
+ });
485
+ }{
486
+ const state = createBenchmarkAgentConformance(projection);
487
+ state.observe(turnStart());
488
+ state.observe(toolCall());
489
+ state.observe(toolResult());
490
+ state.observe(assistant('bad terminal'));
491
+ assert.deepEqual(state.stopping(1), {
492
+ kind: 'steer',
493
+ mode: 'terminal'
494
+ });
495
+ state.observe(assistant('still bad', {
496
+ step: 3
497
+ }));
498
+ assert.deepEqual(state.stopping(1), {
499
+ kind: 'reject',
500
+ code: BENCHMARK_TERMINAL_INVALID
501
+ });
502
+ }{
503
+ const state = createBenchmarkAgentConformance(projection);
504
+ state.observe(turnStart());
505
+ state.observe(toolCall());
506
+ state.observe(toolResult('call-1', {
507
+ ok: false
508
+ }));
509
+ assert.deepEqual(state.stopping(1), {
510
+ kind: 'reject',
511
+ code: BENCHMARK_BRIDGE_RUNNER_FAILED
512
+ }, 'structured runner failure is not retried');
513
+ }{
514
+ const state = createBenchmarkAgentConformance(projection);
515
+ state.observe(turnStart());
516
+ state.observe(toolCall());
517
+ state.observe(toolResult('call-1', {
518
+ ok: false,
519
+ error: 'output_truncated'
520
+ }));
521
+ assert.deepEqual(state.stopping(1), {
522
+ kind: 'reject',
523
+ code: BENCHMARK_BRIDGE_OUTPUT_TRUNCATED
524
+ }, 'structured runner truncation has a distinct reason');
525
+ }{
526
+ const state = createBenchmarkAgentConformance(projection);
527
+ state.observe(turnStart());
528
+ state.observe(toolCall('failed', {
529
+ step: 1
530
+ }));
531
+ state.observe(toolResult('failed', {
532
+ step: 1,
533
+ ok: false
534
+ }));
535
+ state.observe(toolCall('successful', {
536
+ step: 2
537
+ }));
538
+ state.observe(toolResult('successful', {
539
+ step: 2
540
+ }));
541
+ state.observe(assistant('<done>{"status":"succeeded"}</done>', {
542
+ step: 3
543
+ }));
544
+ assert.deepEqual(state.stopping(1), {
545
+ kind: 'accept'
546
+ }, 'a model-owned later success can recover from an earlier failed call without a conformance retry');
547
+ }{
548
+ const state = createBenchmarkAgentConformance(projection);
549
+ state.observe(turnStart());
550
+ state.observe(toolCall('successful', {
551
+ step: 1
552
+ }));
553
+ state.observe(toolResult('successful', {
554
+ step: 1
555
+ }));
556
+ state.observe(toolCall('failed', {
557
+ step: 2
558
+ }));
559
+ state.observe(toolResult('failed', {
560
+ step: 2,
561
+ ok: false
562
+ }));
563
+ state.observe(assistant('<done>{"status":"succeeded"}</done>', {
564
+ step: 3
565
+ }));
566
+ assert.deepEqual(state.stopping(1), {
567
+ kind: 'accept'
568
+ }, 'a later failed optional call does not erase an already paired successful result');
569
+ }{
570
+ const state = createBenchmarkAgentConformance(projection);
571
+ state.observe(turnStart());
572
+ state.observe(toolCall('discovery', {
573
+ action: 'tools'
574
+ }));
575
+ state.observe(toolResult('discovery'));
576
+ state.observe(assistant('<done>{"status":"succeeded"}</done>'));
577
+ assert.deepEqual(state.stopping(1), {
578
+ kind: 'steer',
579
+ mode: 'call'
580
+ }, 'action tools does not satisfy the call gate');
581
+ state.observe(turnEnd());
582
+ assert.deepEqual(state.stopping(1), {
583
+ kind: 'reject',
584
+ code: BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE
585
+ });
586
+ state.observe(turnStart(2));
587
+ assert.deepEqual(state.stopping(1), {
588
+ kind: 'reject',
589
+ code: BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE
590
+ }, 'turn state cannot leak across turns');
591
+ }{
592
+ const rootListeners = new Map();
593
+ const scopedListeners = new Map();
594
+ const guards = [];
595
+ const steers = [];
596
+ const runtimeWrites = [];
597
+ const runEffect = (action)=>{
598
+ const disposers = [];
599
+ const value = action();
600
+ if (value && typeof value.next === 'function') {
601
+ let item = value.next();
602
+ while(!item.done){
603
+ if (typeof item.value === 'function') disposers.push(item.value);
604
+ item = value.next();
605
+ }
606
+ }
607
+ return ()=>{
608
+ for (const dispose of disposers.reverse())dispose();
609
+ };
610
+ };
611
+ const add = (target, name, listener)=>{
612
+ const list = target.get(name) ?? [];
613
+ list.push(listener);
614
+ target.set(name, list);
615
+ return ()=>target.set(name, list.filter((candidate)=>candidate !== listener));
616
+ };
617
+ const session = {};
618
+ const agent = {
619
+ session,
620
+ steer (message) {
621
+ steers.push(message);
622
+ },
623
+ ctx: {
624
+ tools: {
625
+ guard (check) {
626
+ guards.push(check);
627
+ return ()=>{};
628
+ }
629
+ },
630
+ effect: runEffect,
631
+ on (name, listener) {
632
+ return add(scopedListeners, name, listener);
633
+ }
634
+ }
635
+ };
636
+ const ctx = {
637
+ on (name, listener) {
638
+ return add(rootListeners, name, listener);
639
+ }
640
+ };
641
+ installBenchmarkAgentConformance(ctx, projection, (code)=>runtimeWrites.push(code));
642
+ rootListeners.get('agent/created')[0]({
643
+ agent
644
+ });
645
+ rootListeners.get('session/event')[0](session, turnStart());
646
+ rootListeners.get('session/event')[0](session, {
647
+ type: 'llm/retry',
648
+ data: {
649
+ turn: 1,
650
+ reason: {
651
+ kind: 'error',
652
+ error: {
653
+ code: 'RATE_LIMIT'
654
+ }
655
+ }
656
+ }
657
+ });
658
+ rootListeners.get('session/event')[0](session, {
659
+ type: 'agent/request-error',
660
+ data: {
661
+ turn: 1,
662
+ error: {
663
+ code: 'SERVER'
664
+ }
665
+ }
666
+ });
667
+ rootListeners.get('session/event')[0](session, assistant('prose only', {
668
+ step: 1
669
+ }));
670
+ rootListeners.get('agent/turn-stopping')[0]({
671
+ agent,
672
+ turn: 1
673
+ });
674
+ assert.equal(steers.length, 1, 'runtime wiring steers exactly once at the stop boundary');
675
+ assert.equal(steers[0].role, 'user');
676
+ assert.equal(Object.isFrozen(steers[0]), true, 'correction uses the official immutable DSH user message');
677
+ assert.equal(Object.isFrozen(steers[0].content), true, 'correction content is deeply frozen');
678
+ assert.equal(guards[0]({
679
+ name: projection.toolName
680
+ }), undefined, 'call correction leaves bridge execution available');
681
+ const assembled = await scopedListeners.get('system-prompt/assemble')[0]({}, {}, async ()=>({
682
+ sections: [],
683
+ tools: []
684
+ }));
685
+ assert.match(assembled.sections[0].text, /agent_env\.cli/);
686
+ assert.match(assembled.sections[0].text, /\"action\":\"call\"/);
687
+ assert.match(assembled.sections[0].text, /<done>/);
688
+ assert.equal(assembled.sections[0].text.includes('/tmp/'), false);
689
+ rootListeners.get('session/event')[0](session, turnEnd());
690
+ assert.deepEqual(runtimeWrites, []);
691
+ rootListeners.get('session/event')[0](session, turnStart(2));
692
+ rootListeners.get('session/event')[0](session, {
693
+ type: 'turn/end',
694
+ data: {
695
+ turn: 2,
696
+ reason: {
697
+ kind: 'error',
698
+ error: {
699
+ code: 'SERVER',
700
+ message: 'sentinel'
701
+ }
702
+ }
703
+ }
704
+ });
705
+ rootListeners.get('session/event')[0](session, {
706
+ type: 'turn/end',
707
+ data: {
708
+ turn: 2,
709
+ reason: {
710
+ kind: 'error',
711
+ error: {
712
+ code: 'UNKNOWN_MODEL'
713
+ }
714
+ }
715
+ }
716
+ });
717
+ assert.deepEqual(runtimeWrites, [
718
+ 'child_model_server'
719
+ ]);
720
+ rootListeners.get('session/disposed')[0](session);
721
+ const session2 = {};
722
+ const agent2 = {
723
+ session: session2,
724
+ steer () {},
725
+ ctx: agent.ctx
726
+ };
727
+ rootListeners.get('agent/created')[0]({
728
+ agent: agent2
729
+ });
730
+ rootListeners.get('session/event')[0](session2, turnStart());
731
+ assert.throws(()=>rootListeners.get('agent/turn-stopping')[0]({
732
+ agent: agent2
733
+ }), new RegExp(BENCHMARK_CONFORMANCE_STATE_UNAVAILABLE));
734
+ rootListeners.get('session/event')[0](session2, {
735
+ type: 'turn/end',
736
+ data: {
737
+ turn: 1,
738
+ reason: {
739
+ kind: 'error',
740
+ error: {
741
+ code: 'UNKNOWN'
742
+ }
743
+ }
744
+ }
745
+ });
746
+ assert.deepEqual(runtimeWrites, [
747
+ 'child_model_server',
748
+ 'child_conformance_failure'
749
+ ]);
750
+ rootListeners.get('session/disposed')[0](session2);
751
+ const session3 = {};
752
+ const agent3 = {
753
+ session: session3,
754
+ steer () {},
755
+ ctx: agent.ctx
756
+ };
757
+ rootListeners.get('agent/created')[0]({
758
+ agent: agent3
759
+ });
760
+ rootListeners.get('session/event')[0](session3, turnStart());
761
+ rootListeners.get('session/event')[0](session3, {
762
+ type: 'turn/end',
763
+ data: {
764
+ turn: 1,
765
+ reason: {
766
+ kind: 'error',
767
+ error: {
768
+ code: 'TIMEOUT'
769
+ }
770
+ }
771
+ }
772
+ });
773
+ assert.deepEqual(runtimeWrites, [
774
+ 'child_model_server',
775
+ 'child_conformance_failure',
776
+ 'child_model_transport'
777
+ ], 'session diagnostics remain isolated');
778
+ rootListeners.get('session/disposed')[0](session3);
779
+ }function fakeHandle(outcome) {
780
+ const stdout = outcome.stdout ?? '';
781
+ const stderr = outcome.stderr ?? '';
782
+ const reader = {
783
+ readFrom: (_offset)=>({
784
+ text: stdout,
785
+ nextOffset: Buffer.byteLength(stdout),
786
+ lossy: outcome.lossy ?? false
787
+ })
788
+ };
789
+ const errorReader = {
790
+ readFrom: (_offset)=>({
791
+ text: stderr,
792
+ nextOffset: Buffer.byteLength(stderr),
793
+ lossy: false
794
+ })
795
+ };
796
+ let rejectDone;
797
+ const done = outcome.spawnReject ? Promise.reject(new Error('spawn rejected')) : outcome.waitForAbort ? new Promise((_resolve, reject)=>{
798
+ rejectDone = reject;
799
+ }) : Promise.resolve({
800
+ exitCode: outcome.exitCode ?? 0,
801
+ signal: outcome.signal ?? null
802
+ });
803
+ return {
804
+ pid: outcome.spawnReject ? -1 : 4242,
805
+ stdin: undefined,
806
+ stdout: undefined,
807
+ stderr: undefined,
808
+ collected: {
809
+ stdout: reader,
810
+ stderr: errorReader
811
+ },
812
+ done,
813
+ terminate () {
814
+ if (outcome.waitForAbort) rejectDone?.(new Error('timed out'));
815
+ },
816
+ waitForExit: async ()=>true
817
+ };
818
+ }
819
+ async function assertRealCordisWaterfallOrdering() {
820
+ const ctx = new Context();
821
+ const bridge = {
822
+ name: 'gotry_benchmark_environment',
823
+ description: 'benchmark bridge',
824
+ parameters: {
825
+ query: {
826
+ type: 'json',
827
+ required: true
828
+ }
829
+ }
830
+ };
831
+ const exactSchema = structuredClone(bridge);
832
+ let addPreStepTool = false;
833
+ const rootTools = {
834
+ get (name) {
835
+ return name === bridge.name ? bridge : undefined;
836
+ },
837
+ schemas (agent) {
838
+ return agent && addPreStepTool ? [
839
+ exactSchema,
840
+ {
841
+ name: 'non_bridge'
842
+ }
843
+ ] : [
844
+ exactSchema
845
+ ];
846
+ }
847
+ };
848
+ ctx.provide('tools', rootTools);
849
+ ctx.provide('agents', {
850
+ list: ()=>[]
851
+ });
852
+ const bus = ctx;
853
+ bus.on('system-prompt/assemble', async (_assembly, _context, next)=>{
854
+ const result = await next();
855
+ return {
856
+ ...result,
857
+ tools: [
858
+ ...result.tools,
859
+ {
860
+ name: 'non_bridge'
861
+ }
862
+ ]
863
+ };
864
+ });
865
+ bus.on('agent/pre-step', async (_payload, next)=>{
866
+ const result = await next();
867
+ addPreStepTool = true;
868
+ return result;
869
+ });
870
+ installBenchmarkToolIsolation(ctx);
871
+ const scopedEffect = (action, label)=>ctx.effect(action, label);
872
+ const scopedTools = {
873
+ guard: ()=>ctx.effect(()=>()=>undefined),
874
+ presentAs: ()=>ctx.effect(()=>()=>undefined),
875
+ restrict: ()=>ctx.effect(()=>()=>undefined)
876
+ };
877
+ const agent = {
878
+ ctx: {
879
+ tools: scopedTools,
880
+ effect: scopedEffect,
881
+ on: bus.on
882
+ }
883
+ };
884
+ bus.emit('agent/created', {
885
+ agent
886
+ });
887
+ await assert.rejects(bus.waterfall('system-prompt/assemble', {
888
+ tools: []
889
+ }, {
890
+ agent,
891
+ scope: agent
892
+ }, async ()=>({
893
+ tools: [
894
+ exactSchema
895
+ ]
896
+ })), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'prepend assembly guard observes an earlier listener post-next mutation');
897
+ await assert.rejects(bus.waterfall('agent/pre-step', {
898
+ agent
899
+ }, async ()=>({
900
+ kind: 'enter'
901
+ })), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'prepend pre-step guard observes an earlier listener post-next scope mutation');
902
+ await ctx.fiber.dispose();
903
+ }
904
+ await assertRealCordisWaterfallOrdering();
905
+ const root = mkdtempSync(join(tmpdir(), 'gotry-benchmark-bridge-test-'));
906
+ const ambientSentinelNames = [
907
+ 'GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL',
908
+ 'GOTRY_LLM_MODEL',
909
+ 'DATABASE_URL',
910
+ 'SSH_AUTH_SOCK',
911
+ 'AWS_PROFILE',
912
+ 'HTTPS_PROXY'
913
+ ];
914
+ const ambientSentinels = new Map(ambientSentinelNames.map((name)=>[
915
+ name,
916
+ process.env[name]
917
+ ]));
918
+ try {
919
+ delete process.env.GOTRY_LLM_MODEL;
920
+ const timedOutDiagnostic = '\n' + JSON.stringify({
921
+ schema_version: BENCHMARK_CHILD_DIAGNOSTIC_SCHEMA,
922
+ code: 'child_bridge_timed_out'
923
+ }) + '\n';
924
+ assert.equal(parseBenchmarkChildDiagnostic(timedOutDiagnostic), 'child_bridge_timed_out', 'strict control record parses to its allowlisted reason code');
925
+ assert.equal(classifyBenchmarkChildFailure({
926
+ code: 1,
927
+ diagnostic: timedOutDiagnostic
928
+ }), 'child_bridge_timed_out', 'structured control data maps to a stable bridge timeout code');
929
+ assert.equal(classifyBenchmarkChildFailure({
930
+ code: 1,
931
+ diagnostic: `provider text contains child_bridge_timed_out and ${BENCHMARK_BRIDGE_TIMED_OUT}`
932
+ }), 'child_nonzero_exit', 'free text cannot impersonate a structured bridge reason');
933
+ assert.equal(parseBenchmarkChildDiagnostic(JSON.stringify({
934
+ schema_version: BENCHMARK_CHILD_DIAGNOSTIC_SCHEMA,
935
+ code: 'child_bridge_timed_out',
936
+ extra: 'rejected'
937
+ })), undefined, 'control records with extra keys fail closed');
938
+ assert.equal(benchmarkChildFailureForConformanceCode(BENCHMARK_BRIDGE_RUNNER_FAILED), 'child_bridge_runner_failed', 'runner failure maps to a stable structured child reason');
939
+ assert.equal(benchmarkChildFailureForConformanceCode(BENCHMARK_BRIDGE_SPAWN_FAILED), 'child_bridge_spawn_failed', 'spawn failure maps to a stable structured child reason');
940
+ assert.equal(benchmarkChildFailureForConformanceCode(BENCHMARK_BRIDGE_OUTPUT_TRUNCATED), 'child_bridge_output_truncated', 'runner output truncation maps to a stable structured child reason');
941
+ assert.equal(classifyBenchmarkChildFailure({
942
+ code: 0,
943
+ outputTruncated: true
944
+ }), 'child_output_truncated', 'truncated terminal output takes precedence over control data');
945
+ assert.equal(classifyBenchmarkChildFailure({
946
+ code: null,
947
+ signal: 'SIGTERM',
948
+ diagnostic: timedOutDiagnostic
949
+ }), 'child_signaled', 'outer child signal takes precedence over an inner bridge diagnostic');
950
+ assert.equal(classifyBenchmarkChildFailure({
951
+ code: 0,
952
+ diagnostic: ''
953
+ }), 'child_lifecycle_failure', 'unexpected zero-exit diagnostic path remains a stable lifecycle failure');
954
+ const noisyPrefix = Buffer.alloc(BENCHMARK_CHILD_DIAGNOSTIC_MAX_BYTES + 10, 'x');
955
+ const boundedControl = appendBoundedChildDiagnostic(appendBoundedChildDiagnostic(Buffer.alloc(0), noisyPrefix), timedOutDiagnostic);
956
+ assert.equal(boundedControl.length, BENCHMARK_CHILD_DIAGNOSTIC_MAX_BYTES, 'control capture is bounded');
957
+ assert.equal(parseBenchmarkChildDiagnostic(boundedControl.toString('utf8')), 'child_bridge_timed_out', 'rolling tail preserves a final structured reason after bounded noise');
958
+ const configPath = join(root, 'bridge.json');
959
+ writeFileSync(configPath, JSON.stringify({
960
+ schema_version: 'gotry_benchmark_environment_bridge_v2',
961
+ enabled: true,
962
+ executable: process.execPath,
963
+ cwd: root,
964
+ argv_prefix: [
965
+ '-m',
966
+ 'agent_env.cli',
967
+ '--lang',
968
+ 'en'
969
+ ],
970
+ allowed_tools: [
971
+ 'lookup',
972
+ 'constructor',
973
+ 'toString'
974
+ ],
975
+ allowed_output_keys: {
976
+ lookup: [
977
+ 'city',
978
+ 'nested'
979
+ ],
980
+ constructor: [
981
+ 'legacy'
982
+ ]
983
+ },
984
+ timeout_ms: 20,
985
+ max_output_bytes: 4_096,
986
+ terminal_output: {
987
+ tag: 'done',
988
+ max_bytes: 4_096
989
+ },
990
+ isolation: {
991
+ mode: 'host-enforced',
992
+ writes: 'forbidden',
993
+ network: 'denied'
994
+ }
995
+ }));
996
+ const registered = [];
997
+ let visibleBridge;
998
+ let shadowedAgent;
999
+ const scopedExtraSchemas = new Map();
1000
+ const spawnSpecs = [];
1001
+ const agentCreatedListeners = [];
1002
+ const eventNames = [];
1003
+ const promptVariables = [];
1004
+ const assemblyListeners = [];
1005
+ const preStepListeners = [];
1006
+ const disposedListeners = [];
1007
+ const eventOptions = new Map();
1008
+ const runEffect = (action)=>{
1009
+ const yielded = [];
1010
+ const value = action();
1011
+ if (value && typeof value.next === 'function') {
1012
+ let step = value.next();
1013
+ while(!step.done){
1014
+ if (typeof step.value === 'function') yielded.push(step.value);
1015
+ step = value.next();
1016
+ }
1017
+ }
1018
+ let active = true;
1019
+ return async ()=>{
1020
+ if (!active) return;
1021
+ active = false;
1022
+ for (const dispose of yielded.reverse())await dispose();
1023
+ };
1024
+ };
1025
+ let disposeRootIsolation;
1026
+ let timeoutSignal;
1027
+ const outcomes = [
1028
+ {
1029
+ stdout: '{"result":[{"city":"Dubai"}]}'
1030
+ }
1031
+ ];
1032
+ process.env.GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL = 'must-not-cross-boundary';
1033
+ process.env.DATABASE_URL = 'postgres://secret';
1034
+ process.env.SSH_AUTH_SOCK = '/tmp/secret.sock';
1035
+ process.env.AWS_PROFILE = 'secret-profile';
1036
+ process.env.HTTPS_PROXY = 'https://secret-proxy';
1037
+ const ctx = {
1038
+ tools: {
1039
+ register (tool) {
1040
+ registered.push(tool);
1041
+ return ()=>{};
1042
+ },
1043
+ get (name, agent) {
1044
+ return name === 'gotry_benchmark_environment' ? agent !== undefined && agent === shadowedAgent ? {
1045
+ name
1046
+ } : registered.find((tool)=>tool.name === name) : undefined;
1047
+ },
1048
+ schemas (agent) {
1049
+ const project = (tool)=>({
1050
+ name: tool.name,
1051
+ description: tool.description,
1052
+ parameters: structuredClone(tool.parameters)
1053
+ });
1054
+ const schemas = registered.map(project);
1055
+ if (agent === undefined) return schemas;
1056
+ return [
1057
+ ...schemas.filter((schema)=>schema.name === 'gotry_benchmark_environment'),
1058
+ ...scopedExtraSchemas.get(agent) ?? []
1059
+ ];
1060
+ }
1061
+ },
1062
+ systemPrompt: {
1063
+ variable (name) {
1064
+ promptVariables.push(name);
1065
+ }
1066
+ },
1067
+ on (event, listener, options) {
1068
+ eventNames.push(event);
1069
+ eventOptions.set(event, options);
1070
+ if (event === 'agent/created') agentCreatedListeners.push(listener);
1071
+ if (event === 'agent/disposed') disposedListeners.push(listener);
1072
+ if (event === 'system-prompt/assemble') assemblyListeners.push(listener);
1073
+ if (event === 'agent/pre-step') preStepListeners.push(listener);
1074
+ return ()=>{};
1075
+ },
1076
+ effect (action, label) {
1077
+ const dispose = runEffect(action);
1078
+ if (label === 'benchmark-environment-tool-isolation') disposeRootIsolation = dispose;
1079
+ return dispose;
1080
+ },
1081
+ get (name) {
1082
+ if (name === 'subprocess') return this.subprocess;
1083
+ if (name === 'agents') return this.agents;
1084
+ },
1085
+ subprocess: {
1086
+ spawn (spec) {
1087
+ spawnSpecs.push(spec);
1088
+ const outcome = outcomes.shift() ?? {
1089
+ stdout: '{"result":{}}'
1090
+ };
1091
+ if (outcome.spawnError) throw new Error('fake spawn failed');
1092
+ if (outcome.waitForAbort) {
1093
+ timeoutSignal = spec.signal;
1094
+ const handle = fakeHandle(outcome);
1095
+ spec.signal?.addEventListener('abort', ()=>handle.terminate?.(), {
1096
+ once: true
1097
+ });
1098
+ return handle;
1099
+ }
1100
+ return fakeHandle(outcome);
1101
+ }
1102
+ },
1103
+ agents: {
1104
+ list () {
1105
+ return [];
1106
+ }
1107
+ }
1108
+ };
1109
+ const coldStartListeners = [];
1110
+ const coldStartRoot = {
1111
+ name: 'gotry_benchmark_environment'
1112
+ };
1113
+ assert.throws(()=>installBenchmarkToolIsolation({
1114
+ tools: {
1115
+ get (name) {
1116
+ return name === 'gotry_benchmark_environment' ? coldStartRoot : undefined;
1117
+ },
1118
+ schemas () {
1119
+ return [
1120
+ {
1121
+ name: 'gotry_benchmark_environment'
1122
+ }
1123
+ ];
1124
+ }
1125
+ },
1126
+ agents: {
1127
+ list () {
1128
+ return [
1129
+ {
1130
+ id: 'already-live'
1131
+ }
1132
+ ];
1133
+ }
1134
+ },
1135
+ on (event) {
1136
+ coldStartListeners.push(event);
1137
+ return ()=>{};
1138
+ }
1139
+ }), /cold-start|live-agent/i, 'installing benchmark isolation with an existing agent fails hard');
1140
+ assert.deepEqual(coldStartListeners, [], 'cold-start rejection does not register an isolation listener');
1141
+ const processListenersBeforeBenchmark = {
1142
+ uncaughtException: process.listenerCount('uncaughtException'),
1143
+ unhandledRejection: process.listenerCount('unhandledRejection')
1144
+ };
1145
+ apply(ctx, {
1146
+ stateRoot: root,
1147
+ timeoutMs: 20,
1148
+ hbcliBin: '',
1149
+ sessionAccess: 'off',
1150
+ benchmarkEnvironmentConfigPath: configPath
1151
+ });
1152
+ assert.ok(registered.some((tool)=>tool.name === 'gotry_benchmark_environment'), 'an explicit valid owner-local config registers the benchmark environment bridge');
1153
+ assert.deepEqual(registered.map((tool)=>tool.name), [
1154
+ 'gotry_benchmark_environment'
1155
+ ], 'benchmark mode boots only the single bridge tool instead of the product tool catalog');
1156
+ assert.deepEqual(promptVariables, [], 'benchmark mode does not install product prompt variables');
1157
+ assert.equal(eventNames.includes('tools/pre-execute'), false, 'benchmark mode does not install the product session-consent hook');
1158
+ assert.deepEqual({
1159
+ uncaughtException: process.listenerCount('uncaughtException'),
1160
+ unhandledRejection: process.listenerCount('unhandledRejection')
1161
+ }, processListenersBeforeBenchmark, 'benchmark mode does not install product process incident guards');
1162
+ assert.deepEqual([
1163
+ ...eventNames
1164
+ ].sort(), [
1165
+ 'agent/created',
1166
+ 'agent/created',
1167
+ 'agent/disposed',
1168
+ 'agent/disposed',
1169
+ 'agent/pre-step',
1170
+ 'agent/turn-stopping',
1171
+ 'session/disposed',
1172
+ 'session/disposed',
1173
+ 'session/event',
1174
+ 'session/event',
1175
+ 'system-prompt/assemble',
1176
+ 'tools/execute',
1177
+ 'tools/post-execute'
1178
+ ], 'benchmark root listeners come only from budget, isolation, and conformance when model override is unset');
1179
+ visibleBridge = registered.find((tool)=>tool.name === 'gotry_benchmark_environment');
1180
+ const restrictions = [];
1181
+ const guards = [];
1182
+ const presentations = [];
1183
+ const cleanupCounts = {
1184
+ restrict: 0,
1185
+ guard: 0,
1186
+ presentAs: 0,
1187
+ assembly: 0
1188
+ };
1189
+ const firstScopedAssemblyListeners = [];
1190
+ const scopedTools = {
1191
+ restrict (filter) {
1192
+ restrictions.push(filter);
1193
+ return ()=>{
1194
+ cleanupCounts.restrict += 1;
1195
+ };
1196
+ },
1197
+ guard (check) {
1198
+ guards.push(check);
1199
+ return ()=>{
1200
+ cleanupCounts.guard += 1;
1201
+ };
1202
+ },
1203
+ presentAs (mode) {
1204
+ presentations.push(mode);
1205
+ return ()=>{
1206
+ cleanupCounts.presentAs += 1;
1207
+ };
1208
+ }
1209
+ };
1210
+ assert.equal(agentCreatedListeners.length, 2, 'opt-in bridge installs exactly isolation and conformance agent listeners');
1211
+ const isolatedAgentEffects = [];
1212
+ const isolatedAgent = {
1213
+ session: {},
1214
+ steer (_message) {},
1215
+ ctx: {
1216
+ tools: scopedTools,
1217
+ effect: (action)=>{
1218
+ const dispose = runEffect(action);
1219
+ isolatedAgentEffects.push(dispose);
1220
+ return dispose;
1221
+ },
1222
+ on: (event, listener, options)=>{
1223
+ assert.equal(event, 'system-prompt/assemble');
1224
+ assert.deepEqual(options, {
1225
+ prepend: true
1226
+ });
1227
+ firstScopedAssemblyListeners.push(listener);
1228
+ return ()=>{
1229
+ cleanupCounts.assembly += 1;
1230
+ };
1231
+ }
1232
+ }
1233
+ };
1234
+ for (const listener of agentCreatedListeners)listener({
1235
+ agent: isolatedAgent
1236
+ });
1237
+ assert.deepEqual(restrictions, [
1238
+ {
1239
+ allow: [
1240
+ 'gotry_benchmark_environment'
1241
+ ]
1242
+ }
1243
+ ], 'agent scope allows only the bridge tool');
1244
+ assert.deepEqual(presentations, [
1245
+ 'native'
1246
+ ], 'agent scope forces native tool presentation');
1247
+ assert.ok(eventNames.includes('agent/pre-step'), 'isolation observes pre-step before each request');
1248
+ assert.deepEqual(eventOptions.get('agent/pre-step'), {
1249
+ prepend: true
1250
+ }, 'pre-step isolation wraps every previously registered listener');
1251
+ assert.ok(eventNames.includes('agent/disposed'), 'isolation cleans up on agent disposal');
1252
+ assert.ok(assemblyListeners.length > 0, 'isolation validates final assembled tool surface');
1253
+ assert.equal(firstScopedAssemblyListeners.length, 2, 'agent owns one isolation assembly listener and one conformance section listener');
1254
+ assert.deepEqual(eventOptions.get('system-prompt/assemble'), {
1255
+ prepend: true
1256
+ }, 'assembly isolation wraps every previously registered listener');
1257
+ assert.equal(disposedListeners.length, 2, 'isolation and conformance each own one agent/disposed listener');
1258
+ assert.ok(disposeRootIsolation, 'root isolation effect exposes plugin-lifecycle cleanup');
1259
+ const exactSchema = {
1260
+ name: visibleBridge.name,
1261
+ description: visibleBridge.description,
1262
+ parameters: structuredClone(visibleBridge.parameters)
1263
+ };
1264
+ const assemble = assemblyListeners[0];
1265
+ const nextExact = async ()=>({
1266
+ tools: [
1267
+ exactSchema
1268
+ ]
1269
+ });
1270
+ assert.deepEqual(await assemble({
1271
+ tools: []
1272
+ }, {
1273
+ agent: isolatedAgent,
1274
+ scope: isolatedAgent
1275
+ }, nextExact), {
1276
+ tools: [
1277
+ exactSchema
1278
+ ]
1279
+ }, 'legal final assembly passes unchanged');
1280
+ await assert.rejects(async ()=>await assemble({
1281
+ tools: []
1282
+ }, {
1283
+ agent: isolatedAgent,
1284
+ scope: isolatedAgent
1285
+ }, async ()=>({
1286
+ tools: [
1287
+ exactSchema,
1288
+ {
1289
+ name: 'own_side_effect'
1290
+ }
1291
+ ]
1292
+ })), /BENCHMARK_TOOL_SURFACE_VIOLATION/);
1293
+ await assert.rejects(async ()=>await assemble({
1294
+ tools: []
1295
+ }, {
1296
+ agent: isolatedAgent,
1297
+ scope: isolatedAgent
1298
+ }, async ()=>({
1299
+ tools: [
1300
+ {
1301
+ ...exactSchema,
1302
+ description: `${exactSchema.description ?? ''} tampered`
1303
+ }
1304
+ ]
1305
+ })), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'final assembly rejects same-name schema mutation');
1306
+ const diagnosticAssembly = {
1307
+ tools: [
1308
+ {
1309
+ name: 'diagnostic_tool'
1310
+ }
1311
+ ]
1312
+ };
1313
+ assert.equal(await assemble({
1314
+ tools: []
1315
+ }, {
1316
+ scope: {
1317
+ id: 'diagnostic-scope'
1318
+ }
1319
+ }, async ()=>diagnosticAssembly), diagnosticAssembly, 'non-agent diagnostic assembly passes through');
1320
+ await assert.rejects(async ()=>await assemble({
1321
+ tools: []
1322
+ }, {
1323
+ agent: isolatedAgent,
1324
+ scope: {
1325
+ id: 'wrong-scope'
1326
+ }
1327
+ }, nextExact), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'agent assembly requires the same agent and scope');
1328
+ const preStep = preStepListeners[0];
1329
+ assert.deepEqual(await preStep({
1330
+ agent: isolatedAgent
1331
+ }, async ()=>({
1332
+ decision: 'continue'
1333
+ })), {
1334
+ decision: 'continue'
1335
+ }, 'legal pre-step decision passes unchanged');
1336
+ scopedExtraSchemas.set(isolatedAgent, [
1337
+ {
1338
+ name: 'own_side_effect'
1339
+ }
1340
+ ]);
1341
+ await assert.rejects(async ()=>await preStep({
1342
+ agent: isolatedAgent
1343
+ }, async ()=>({
1344
+ decision: 'continue'
1345
+ })), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'downstream pre-step scope expansion fails closed');
1346
+ scopedExtraSchemas.delete(isolatedAgent);
1347
+ shadowedAgent = isolatedAgent;
1348
+ await assert.rejects(async ()=>await assemble({
1349
+ tools: []
1350
+ }, {
1351
+ agent: isolatedAgent,
1352
+ scope: isolatedAgent
1353
+ }, nextExact), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'same-name scoped shadow fails final assembly identity check');
1354
+ shadowedAgent = undefined;
1355
+ assert.equal(guards.length, 2, 'agent scope installs exact isolation and conformance guards');
1356
+ const guard = guards[0];
1357
+ const originalAgent = {
1358
+ ctx: {
1359
+ tools: {
1360
+ get: (name)=>name === 'gotry_benchmark_environment' ? visibleBridge : undefined
1361
+ }
1362
+ }
1363
+ };
1364
+ const shadowAgent = {
1365
+ ctx: {
1366
+ tools: {
1367
+ get: (name)=>name === 'gotry_benchmark_environment' ? {
1368
+ name
1369
+ } : undefined
1370
+ }
1371
+ }
1372
+ };
1373
+ shadowedAgent = shadowAgent;
1374
+ assert.equal(guard({
1375
+ name: 'gotry_benchmark_environment',
1376
+ args: {},
1377
+ agent: originalAgent
1378
+ }), undefined, 'original bridge definition is allowed');
1379
+ assert.equal(guard({
1380
+ name: 'gotry_benchmark_environment',
1381
+ args: {},
1382
+ agent: shadowAgent
1383
+ }), 'BENCHMARK_TOOL_NOT_ALLOWED', 'same-name shadow definition is denied');
1384
+ assert.equal(guard({
1385
+ name: 'gotry_benchmark_environment',
1386
+ args: {
1387
+ path: '/private/secret'
1388
+ },
1389
+ agent: isolatedAgent
1390
+ }), undefined, 'bridge tool is allowed');
1391
+ const denied = guard({
1392
+ name: 'other_tool',
1393
+ args: {
1394
+ path: '/private/secret',
1395
+ token: 'secret'
1396
+ }
1397
+ });
1398
+ assert.equal(denied, 'BENCHMARK_TOOL_NOT_ALLOWED', 'non-bridge tools are denied without argument/path echo');
1399
+ assert.equal(denied?.includes('/private/secret'), false);
1400
+ assert.equal(denied?.includes('secret'), false);
1401
+ assert.equal(guard({
1402
+ name: 'other_tool',
1403
+ args: {
1404
+ different: true
1405
+ }
1406
+ }), denied, 'denial reason is stable');
1407
+ assert.throws(()=>agentCreatedListeners[0]({
1408
+ agent: {
1409
+ ctx: {
1410
+ tools: {
1411
+ restrict (filter) {
1412
+ void filter;
1413
+ },
1414
+ guard (check) {
1415
+ void check;
1416
+ }
1417
+ }
1418
+ }
1419
+ }
1420
+ }), /presentAs/, 'missing scoped presentAs fails hard');
1421
+ assert.throws(()=>agentCreatedListeners[0]({
1422
+ agent: {
1423
+ ctx: {
1424
+ tools: scopedTools
1425
+ }
1426
+ }
1427
+ }), /effect/, 'missing scoped effect fails hard');
1428
+ assert.throws(()=>agentCreatedListeners[0]({
1429
+ agent: {
1430
+ ctx: {
1431
+ tools: scopedTools,
1432
+ effect: (action)=>runEffect(action)
1433
+ }
1434
+ }
1435
+ }), /event bus/, 'missing scoped event bus fails hard');
1436
+ assert.throws(()=>agentCreatedListeners[0]({
1437
+ agent: {
1438
+ ctx: {
1439
+ tools: {
1440
+ guard: guards[0]
1441
+ }
1442
+ }
1443
+ }
1444
+ }), /restrict/, 'missing scoped restrict fails hard');
1445
+ assert.throws(()=>agentCreatedListeners[0]({
1446
+ agent: {
1447
+ ctx: {
1448
+ tools: {
1449
+ restrict (filter) {
1450
+ void filter;
1451
+ }
1452
+ }
1453
+ }
1454
+ }
1455
+ }), /guard/, 'missing scoped guard fails hard');
1456
+ assert.throws(()=>installBenchmarkToolIsolation({
1457
+ tools: {
1458
+ get (name) {
1459
+ return name === 'gotry_benchmark_environment' ? {
1460
+ name: 'gotry_benchmark_environment'
1461
+ } : undefined;
1462
+ },
1463
+ schemas () {
1464
+ return [
1465
+ {
1466
+ name: 'gotry_benchmark_environment'
1467
+ }
1468
+ ];
1469
+ }
1470
+ },
1471
+ agents: {
1472
+ list () {
1473
+ return [];
1474
+ }
1475
+ },
1476
+ on () {
1477
+ return ()=>{};
1478
+ }
1479
+ }), /effect/, 'missing ctx.effect fails hard');
1480
+ await disposedListeners[0]({
1481
+ agent: isolatedAgent
1482
+ });
1483
+ assert.deepEqual(cleanupCounts, {
1484
+ restrict: 1,
1485
+ guard: 1,
1486
+ presentAs: 1,
1487
+ assembly: 1
1488
+ }, 'agent disposal releases every scoped isolation effect exactly once');
1489
+ await disposedListeners[1]({
1490
+ agent: isolatedAgent
1491
+ });
1492
+ await isolatedAgentEffects[1]();
1493
+ assert.deepEqual(cleanupCounts, {
1494
+ restrict: 1,
1495
+ guard: 2,
1496
+ presentAs: 1,
1497
+ assembly: 2
1498
+ }, 'agent-scope disposal also releases conformance guard and prompt section');
1499
+ const secondCleanup = {
1500
+ restrict: 0,
1501
+ guard: 0,
1502
+ presentAs: 0,
1503
+ assembly: 0
1504
+ };
1505
+ const secondGuards = [];
1506
+ const secondScopedAssemblyListeners = [];
1507
+ const secondTools = {
1508
+ restrict () {
1509
+ return ()=>{
1510
+ secondCleanup.restrict += 1;
1511
+ };
1512
+ },
1513
+ guard (check) {
1514
+ secondGuards.push(check);
1515
+ return ()=>{
1516
+ secondCleanup.guard += 1;
1517
+ };
1518
+ },
1519
+ presentAs () {
1520
+ return ()=>{
1521
+ secondCleanup.presentAs += 1;
1522
+ };
1523
+ }
1524
+ };
1525
+ const secondAgentEffects = [];
1526
+ const secondAgent = {
1527
+ session: {},
1528
+ steer (_message) {},
1529
+ ctx: {
1530
+ tools: secondTools,
1531
+ effect: (action)=>{
1532
+ const dispose = runEffect(action);
1533
+ secondAgentEffects.push(dispose);
1534
+ return dispose;
1535
+ },
1536
+ on: (_event, listener)=>{
1537
+ secondScopedAssemblyListeners.push(listener);
1538
+ return ()=>{
1539
+ secondCleanup.assembly += 1;
1540
+ };
1541
+ }
1542
+ }
1543
+ };
1544
+ for (const listener of agentCreatedListeners)listener({
1545
+ agent: secondAgent
1546
+ });
1547
+ let inFlightNextCalls = 0;
1548
+ await assert.rejects(async ()=>await secondScopedAssemblyListeners[0]({}, {
1549
+ agent: secondAgent,
1550
+ scope: secondAgent
1551
+ }, async ()=>{
1552
+ inFlightNextCalls += 1;
1553
+ await disposeRootIsolation();
1554
+ return {
1555
+ tools: [
1556
+ exactSchema
1557
+ ]
1558
+ };
1559
+ }), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'plugin unload during assembly fails closed before returning a model-visible result');
1560
+ assert.equal(inFlightNextCalls, 1, 'in-flight quarantine test reaches the controlled unload point');
1561
+ assert.deepEqual(secondCleanup, {
1562
+ restrict: 0,
1563
+ guard: 0,
1564
+ presentAs: 0,
1565
+ assembly: 0
1566
+ }, 'plugin unload keeps a live agent quarantined');
1567
+ let quarantineNextCalls = 0;
1568
+ await assert.rejects(async ()=>await secondScopedAssemblyListeners[0]({}, {
1569
+ agent: secondAgent,
1570
+ scope: secondAgent
1571
+ }, async ()=>{
1572
+ quarantineNextCalls += 1;
1573
+ return {
1574
+ tools: [
1575
+ exactSchema
1576
+ ]
1577
+ };
1578
+ }), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'plugin-unloaded live agent rejects assembly before model request');
1579
+ assert.equal(quarantineNextCalls, 0, 'quarantine does not enter the remaining assembly chain');
1580
+ assert.equal(secondGuards[0]({
1581
+ name: 'other_tool',
1582
+ agent: secondAgent
1583
+ }), 'BENCHMARK_TOOL_NOT_ALLOWED', 'quarantined agent still denies non-bridge dispatch after plugin unload');
1584
+ assert.equal(secondGuards[0]({
1585
+ name: 'gotry_benchmark_environment',
1586
+ agent: secondAgent
1587
+ }), 'BENCHMARK_TOOL_NOT_ALLOWED', 'quarantined agent also denies bridge dispatch after plugin unload');
1588
+ assert.deepEqual(cleanupCounts, {
1589
+ restrict: 1,
1590
+ guard: 2,
1591
+ presentAs: 1,
1592
+ assembly: 2
1593
+ }, 'plugin unload does not double-dispose an already removed agent');
1594
+ assert.throws(()=>agentCreatedListeners[0]({
1595
+ agent: secondAgent
1596
+ }), /BENCHMARK_TOOL_SURFACE_VIOLATION/, 'agent creation during plugin stop fails closed');
1597
+ for (const dispose of secondAgentEffects)await dispose();
1598
+ assert.deepEqual(secondCleanup, {
1599
+ restrict: 1,
1600
+ guard: 2,
1601
+ presentAs: 1,
1602
+ assembly: 2
1603
+ }, 'agent disposal releases its quarantined isolation and conformance effects');
1604
+ const bridge = registered.find((tool)=>tool.name === 'gotry_benchmark_environment');
1605
+ assert.ok(bridge.execute, 'registered bridge exposes execute');
1606
+ const args = {
1607
+ city: 'Dubai',
1608
+ payload: '$(touch /tmp/nope)'
1609
+ };
1610
+ const result = await bridge.execute({
1611
+ query: {
1612
+ action: 'call',
1613
+ tool: 'lookup',
1614
+ arguments: args
1615
+ }
1616
+ }, null);
1617
+ assert.deepEqual(spawnSpecs[0]?.argv, [
1618
+ process.execPath,
1619
+ '-m',
1620
+ 'agent_env.cli',
1621
+ '--lang',
1622
+ 'en',
1623
+ 'call',
1624
+ 'lookup',
1625
+ JSON.stringify(args)
1626
+ ], 'call uses only the configured executable/prefix and fixed lookup subcommand argv');
1627
+ assert.equal(spawnSpecs[0]?.cwd, root, 'call uses configured cwd');
1628
+ assert.equal(spawnSpecs[0]?.stdio.stdin, 'ignore', 'call never exposes stdin');
1629
+ assert.equal(spawnSpecs[0]?.stdio.stdout.maxBytes, 4_096, 'stdout cap comes from config');
1630
+ assert.equal(spawnSpecs[0]?.stdio.stderr.maxBytes, 4_096, 'stderr cap comes from config');
1631
+ assert.ok((spawnSpecs[0]?.graceMs ?? 0) > 0 && (spawnSpecs[0]?.graceMs ?? Infinity) <= 1_000, 'graceMs is bounded');
1632
+ assert.equal(spawnSpecs[0]?.env?.GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL, undefined, 'parent sentinel is not inherited');
1633
+ for (const key of [
1634
+ 'DATABASE_URL',
1635
+ 'SSH_AUTH_SOCK',
1636
+ 'AWS_PROFILE',
1637
+ 'HTTPS_PROXY',
1638
+ 'GOTRY_BENCHMARK_BRIDGE_PARENT_SENTINEL'
1639
+ ]){
1640
+ assert.equal(spawnSpecs[0]?.env?.[key], undefined, `${key} is tombstoned in bridge subprocess env`);
1641
+ }
1642
+ assert.equal(spawnSpecs[0]?.env?.PYTHONDONTWRITEBYTECODE, '1');
1643
+ assert.equal(spawnSpecs[0]?.env?.PYTHONNOUSERSITE, '1');
1644
+ assert.deepEqual(result, {
1645
+ ok: true,
1646
+ result: [
1647
+ {
1648
+ city: 'Dubai'
1649
+ }
1650
+ ]
1651
+ }, 'one-line JSON stdout becomes structured result');
1652
+ outcomes.push({
1653
+ stdout: '{"result":{"city":"Dubai"}}'
1654
+ });
1655
+ const unmappedResult = await bridge.execute({
1656
+ query: {
1657
+ action: 'call',
1658
+ tool: 'toString',
1659
+ arguments: {
1660
+ city: 'Dubai'
1661
+ }
1662
+ }
1663
+ }, null);
1664
+ assert.deepEqual(unmappedResult, {
1665
+ ok: true,
1666
+ result: {
1667
+ city: 'Dubai'
1668
+ }
1669
+ }, 'allowed tool without a positive mapping retains the recursive denylist only');
1670
+ outcomes.push({
1671
+ stdout: '{"result":{"legacy":"value"}}'
1672
+ });
1673
+ const legacyResult = await bridge.execute({
1674
+ query: {
1675
+ action: 'call',
1676
+ tool: 'constructor',
1677
+ arguments: {
1678
+ city: 'Dubai'
1679
+ }
1680
+ }
1681
+ }, null);
1682
+ assert.deepEqual(legacyResult, {
1683
+ ok: true,
1684
+ result: {
1685
+ legacy: 'value'
1686
+ }
1687
+ }, 'mapped constructor accepts its declared positive key');
1688
+ outcomes.push({
1689
+ stdout: '{"result":{"city":"Dubai"}}'
1690
+ });
1691
+ const constructorUnexpected = await bridge.execute({
1692
+ query: {
1693
+ action: 'call',
1694
+ tool: 'constructor',
1695
+ arguments: {
1696
+ city: 'Dubai'
1697
+ }
1698
+ }
1699
+ }, null);
1700
+ assert.deepEqual(constructorUnexpected, {
1701
+ ok: false,
1702
+ error: 'forbidden_output'
1703
+ }, 'mapped constructor rejects undeclared positive keys');
1704
+ outcomes.push({
1705
+ stdout: '{"result":{"nested":{"city":"Dubai"}}}'
1706
+ });
1707
+ const nestedAllowedResult = await bridge.execute({
1708
+ query: {
1709
+ action: 'call',
1710
+ tool: 'lookup',
1711
+ arguments: {
1712
+ city: 'Dubai'
1713
+ }
1714
+ }
1715
+ }, null);
1716
+ assert.deepEqual(nestedAllowedResult, {
1717
+ ok: true,
1718
+ result: {
1719
+ nested: {
1720
+ city: 'Dubai'
1721
+ }
1722
+ }
1723
+ }, 'configured positive output allowlist accepts declared nested keys');
1724
+ outcomes.push({
1725
+ stdout: '{"result":{"city":"Dubai","unexpected":"secret"}}'
1726
+ });
1727
+ const unexpectedResult = await bridge.execute({
1728
+ query: {
1729
+ action: 'call',
1730
+ tool: 'lookup',
1731
+ arguments: {
1732
+ city: 'Dubai'
1733
+ }
1734
+ }
1735
+ }, null);
1736
+ assert.deepEqual(unexpectedResult, {
1737
+ ok: false,
1738
+ error: 'forbidden_output'
1739
+ }, 'configured positive output allowlist rejects unexpected keys without reflecting them');
1740
+ outcomes.push({
1741
+ stdout: '{"result":[{"city":"Dubai","nested":{"unexpected":"secret"}}]}'
1742
+ });
1743
+ const nestedUnexpectedResult = await bridge.execute({
1744
+ query: {
1745
+ action: 'call',
1746
+ tool: 'lookup',
1747
+ arguments: {
1748
+ city: 'Dubai'
1749
+ }
1750
+ }
1751
+ }, null);
1752
+ assert.deepEqual(nestedUnexpectedResult, {
1753
+ ok: false,
1754
+ error: 'forbidden_output'
1755
+ }, 'configured positive output allowlist recurses through arrays and objects');
1756
+ for (const forbidden of [
1757
+ 'gold',
1758
+ 'goldAnswer',
1759
+ 'oracle',
1760
+ 'expected',
1761
+ 'expected_answer',
1762
+ 'answer',
1763
+ 'label',
1764
+ 'score',
1765
+ 'reward',
1766
+ 'ground_truth',
1767
+ 'groundTruth',
1768
+ 'hidden_query',
1769
+ 'hidden-query',
1770
+ 'loader_metadata',
1771
+ 'loaderMetadata',
1772
+ 'reference',
1773
+ 'gоld'
1774
+ ]){
1775
+ outcomes.push({
1776
+ stdout: JSON.stringify({
1777
+ result: {
1778
+ nested: {
1779
+ [forbidden]: 'secret'
1780
+ }
1781
+ }
1782
+ })
1783
+ });
1784
+ const forbiddenResult = await bridge.execute({
1785
+ query: {
1786
+ action: 'call',
1787
+ tool: 'lookup',
1788
+ arguments: {
1789
+ city: 'Dubai'
1790
+ }
1791
+ }
1792
+ }, null);
1793
+ assert.deepEqual(forbiddenResult, {
1794
+ ok: false,
1795
+ error: 'forbidden_output'
1796
+ }, `recursive no-oracle key ${forbidden} is rejected without reflecting its name`);
1797
+ }
1798
+ outcomes.push({
1799
+ stdout: '{"result":"the hidden answer"}'
1800
+ });
1801
+ const primitiveOutput = await bridge.execute({
1802
+ query: {
1803
+ action: 'call',
1804
+ tool: 'lookup',
1805
+ arguments: {
1806
+ city: 'Dubai'
1807
+ }
1808
+ }
1809
+ }, null);
1810
+ assert.deepEqual(primitiveOutput, {
1811
+ ok: false,
1812
+ error: 'invalid_output'
1813
+ }, 'primitive result strings cannot bypass the structured visible-output boundary');
1814
+ const beforeOversized = spawnSpecs.length;
1815
+ const oversized = await bridge.execute({
1816
+ query: {
1817
+ action: 'call',
1818
+ tool: 'lookup',
1819
+ arguments: {
1820
+ blob: 'x'.repeat(65_537)
1821
+ }
1822
+ }
1823
+ }, null);
1824
+ assert.deepEqual(oversized, {
1825
+ ok: false,
1826
+ error: 'invalid_arguments',
1827
+ reason: 'serialization_limit'
1828
+ });
1829
+ assert.equal(spawnSpecs.length, beforeOversized, 'oversized serialized arguments are rejected before spawn');
1830
+ const beforeDeep = spawnSpecs.length;
1831
+ let deep = {};
1832
+ for(let index = 0; index < 13; index++)deep = {
1833
+ next: deep
1834
+ };
1835
+ const deepResult = await bridge.execute({
1836
+ query: {
1837
+ action: 'call',
1838
+ tool: 'lookup',
1839
+ arguments: deep
1840
+ }
1841
+ }, null);
1842
+ assert.deepEqual(deepResult, {
1843
+ ok: false,
1844
+ error: 'invalid_arguments',
1845
+ reason: 'serialization_limit'
1846
+ });
1847
+ assert.equal(spawnSpecs.length, beforeDeep, 'over-deep arguments are rejected before spawn');
1848
+ const beforeOverride = spawnSpecs.length;
1849
+ const overrideResult = await bridge.execute({
1850
+ query: {
1851
+ action: 'call',
1852
+ tool: 'lookup',
1853
+ arguments: {
1854
+ city: 'Dubai',
1855
+ executable: '/tmp/evil',
1856
+ cwd: '/tmp/evil',
1857
+ argv: [
1858
+ '--unsafe'
1859
+ ]
1860
+ }
1861
+ }
1862
+ }, null);
1863
+ assert.deepEqual(spawnSpecs[beforeOverride]?.argv, [
1864
+ process.execPath,
1865
+ '-m',
1866
+ 'agent_env.cli',
1867
+ '--lang',
1868
+ 'en',
1869
+ 'call',
1870
+ 'lookup',
1871
+ JSON.stringify({
1872
+ city: 'Dubai',
1873
+ executable: '/tmp/evil',
1874
+ cwd: '/tmp/evil',
1875
+ argv: [
1876
+ '--unsafe'
1877
+ ]
1878
+ })
1879
+ ], 'model executable/cwd/argv fields remain data and cannot override config');
1880
+ assert.equal(overrideResult.ok, true, 'override-shaped arguments still use the configured bridge');
1881
+ outcomes.push({
1882
+ stdout: '{"result":{"city":"Dubai"}}',
1883
+ lossy: true
1884
+ });
1885
+ const truncated = await bridge.execute({
1886
+ query: {
1887
+ action: 'call',
1888
+ tool: 'lookup',
1889
+ arguments: {
1890
+ city: 'Dubai'
1891
+ }
1892
+ }
1893
+ }, null);
1894
+ assert.deepEqual(truncated, {
1895
+ ok: false,
1896
+ error: 'output_truncated'
1897
+ }, 'lossy stdout is rejected without parsing partial output');
1898
+ outcomes.push({
1899
+ stdout: '{"result":{"city":"Dubai"}}',
1900
+ stderr: 'private runner diagnostic',
1901
+ exitCode: 17,
1902
+ signal: 'SIGTERM'
1903
+ });
1904
+ const failed = await bridge.execute({
1905
+ query: {
1906
+ action: 'call',
1907
+ tool: 'lookup',
1908
+ arguments: {
1909
+ city: 'Dubai'
1910
+ }
1911
+ }
1912
+ }, null);
1913
+ assert.deepEqual(failed, {
1914
+ ok: false,
1915
+ error: 'runner_failed',
1916
+ exit_code: 17,
1917
+ signal: 'SIGTERM'
1918
+ }, 'nonzero runner result is structured without stderr echo');
1919
+ for (const stdout of [
1920
+ 'not-json',
1921
+ '{"result":1}{"result":2}'
1922
+ ]){
1923
+ outcomes.push({
1924
+ stdout
1925
+ });
1926
+ const malformed = await bridge.execute({
1927
+ query: {
1928
+ action: 'call',
1929
+ tool: 'lookup',
1930
+ arguments: {
1931
+ city: 'Dubai'
1932
+ }
1933
+ }
1934
+ }, null);
1935
+ assert.deepEqual(malformed, {
1936
+ ok: false,
1937
+ error: 'invalid_json'
1938
+ }, 'malformed or multi-value JSON is rejected');
1939
+ }
1940
+ outcomes.push({
1941
+ spawnError: true
1942
+ });
1943
+ const spawnFailed = await bridge.execute({
1944
+ query: {
1945
+ action: 'call',
1946
+ tool: 'lookup',
1947
+ arguments: {
1948
+ city: 'Dubai'
1949
+ }
1950
+ }
1951
+ }, null);
1952
+ assert.deepEqual(spawnFailed, {
1953
+ ok: false,
1954
+ error: 'spawn_failed'
1955
+ }, 'spawn infrastructure failure is structured');
1956
+ outcomes.push({
1957
+ spawnReject: true
1958
+ });
1959
+ const asyncSpawnFailed = await bridge.execute({
1960
+ query: {
1961
+ action: 'call',
1962
+ tool: 'lookup',
1963
+ arguments: {
1964
+ city: 'Dubai'
1965
+ }
1966
+ }
1967
+ }, null);
1968
+ assert.deepEqual(asyncSpawnFailed, {
1969
+ ok: false,
1970
+ error: 'spawn_failed'
1971
+ }, 'DSH pid=-1 spawn rejection is distinct from a started runner failure');
1972
+ outcomes.push({
1973
+ waitForAbort: true
1974
+ });
1975
+ const timedOut = await bridge.execute({
1976
+ query: {
1977
+ action: 'call',
1978
+ tool: 'lookup',
1979
+ arguments: {
1980
+ city: 'Dubai'
1981
+ }
1982
+ }
1983
+ }, null);
1984
+ assert.deepEqual(timedOut, {
1985
+ ok: false,
1986
+ error: 'timed_out'
1987
+ }, 'deadline abort is surfaced as timed_out');
1988
+ assert.equal(timeoutSignal?.aborted, true, 'timeout abort signal is fired');
1989
+ const beforeInvalidArgs = spawnSpecs.length;
1990
+ assert.deepEqual(await bridge.execute({
1991
+ query: {
1992
+ action: 'call',
1993
+ tool: 'lookup',
1994
+ arguments: [
1995
+ 'not',
1996
+ 'plain'
1997
+ ]
1998
+ }
1999
+ }, null), {
2000
+ ok: false,
2001
+ error: 'invalid_arguments'
2002
+ });
2003
+ assert.equal(spawnSpecs.length, beforeInvalidArgs, 'non-object arguments are rejected before spawn');
2004
+ assert.deepEqual(await bridge.execute({
2005
+ query: {
2006
+ action: 'inspect'
2007
+ }
2008
+ }, null), {
2009
+ ok: false,
2010
+ error: 'invalid_action'
2011
+ }, 'unknown action is rejected structurally');
2012
+ const beforeRejected = spawnSpecs.length;
2013
+ const rejected = await bridge.execute({
2014
+ query: {
2015
+ action: 'call',
2016
+ tool: 'delete_all',
2017
+ arguments: {}
2018
+ }
2019
+ }, null);
2020
+ assert.deepEqual(rejected, {
2021
+ ok: false,
2022
+ error: 'disallowed_tool'
2023
+ }, 'disallowed tool is rejected structurally');
2024
+ assert.equal(spawnSpecs.length, beforeRejected, 'disallowed tool is rejected before spawn');
2025
+ const spacedConfigPath = join(root, ' benchmark-environment-config.json ');
2026
+ writeFileSync(spacedConfigPath, readFileSync(configPath));
2027
+ const spacedTools = [];
2028
+ const spacedCtx = {
2029
+ tools: {
2030
+ register (tool) {
2031
+ spacedTools.push(tool);
2032
+ return ()=>{};
2033
+ },
2034
+ get (name) {
2035
+ return spacedTools.find((tool)=>tool.name === name);
2036
+ },
2037
+ schemas () {
2038
+ return spacedTools.map((tool)=>({
2039
+ name: tool.name
2040
+ }));
2041
+ }
2042
+ },
2043
+ agents: {
2044
+ list () {
2045
+ return [];
2046
+ }
2047
+ },
2048
+ systemPrompt: {
2049
+ variable () {}
2050
+ },
2051
+ on () {
2052
+ return ()=>{};
2053
+ },
2054
+ effect () {
2055
+ return ()=>{};
2056
+ },
2057
+ get (name) {
2058
+ if (name === 'subprocess') return {
2059
+ spawn: (_spec)=>fakeHandle({
2060
+ stdout: '{}'
2061
+ })
2062
+ };
2063
+ if (name === 'agents') return {
2064
+ list () {
2065
+ return [];
2066
+ }
2067
+ };
2068
+ return undefined;
2069
+ }
2070
+ };
2071
+ apply(spacedCtx, {
2072
+ stateRoot: root,
2073
+ timeoutMs: 20,
2074
+ hbcliBin: '',
2075
+ sessionAccess: 'off',
2076
+ benchmarkEnvironmentConfigPath: spacedConfigPath
2077
+ });
2078
+ assert.deepEqual(spacedTools.map((tool)=>tool.name), [
2079
+ 'gotry_benchmark_environment'
2080
+ ], 'benchmark bridge loads a valid raw path with whitespace basename');
2081
+ const disabled = [];
2082
+ const disabledVariables = [];
2083
+ const disabledCtx = {
2084
+ tools: {
2085
+ register (tool) {
2086
+ disabled.push(tool);
2087
+ return ()=>{};
2088
+ }
2089
+ },
2090
+ systemPrompt: {
2091
+ variable (name) {
2092
+ disabledVariables.push(name);
2093
+ }
2094
+ }
2095
+ };
2096
+ apply(disabledCtx, {
2097
+ stateRoot: root,
2098
+ timeoutMs: 1_000,
2099
+ hbcliBin: '',
2100
+ sessionAccess: 'off',
2101
+ benchmarkEnvironmentConfigPath: ''
2102
+ });
2103
+ assert.equal(disabled.some((tool)=>tool.name === 'gotry_benchmark_environment'), false, 'empty config path keeps bridge default-off');
2104
+ assert.ok(disabled.length > 1, 'normal product mode keeps the full GoTry tool catalog');
2105
+ assert.deepEqual(disabledVariables, [
2106
+ 'current_date',
2107
+ 'time_anchor_card',
2108
+ 'motivation_brief'
2109
+ ], 'normal product mode keeps its prompt variables');
2110
+ const whitespace = [];
2111
+ const whitespaceCtx = {
2112
+ tools: {
2113
+ register (tool) {
2114
+ whitespace.push(tool);
2115
+ return ()=>{};
2116
+ }
2117
+ },
2118
+ systemPrompt: {
2119
+ variable () {}
2120
+ }
2121
+ };
2122
+ apply(whitespaceCtx, {
2123
+ stateRoot: root,
2124
+ timeoutMs: 1_000,
2125
+ hbcliBin: '',
2126
+ sessionAccess: 'off',
2127
+ benchmarkEnvironmentConfigPath: ' \t '
2128
+ });
2129
+ assert.equal(whitespace.some((tool)=>tool.name === 'gotry_benchmark_environment'), false, 'whitespace config path keeps benchmark mode default-off');
2130
+ assert.ok(whitespace.length > 1, 'whitespace path preserves the ordinary product tool catalog');
2131
+ const originalModelOverride = process.env.GOTRY_LLM_MODEL;
2132
+ process.env.GOTRY_LLM_MODEL = 'round7-model-preserved';
2133
+ try {
2134
+ const modelTools = [];
2135
+ let modelRequest;
2136
+ const modelEvents = [];
2137
+ const modelCtx = {
2138
+ tools: {
2139
+ register (tool) {
2140
+ modelTools.push(tool);
2141
+ return ()=>{};
2142
+ },
2143
+ get (name) {
2144
+ return modelTools.find((tool)=>tool.name === name);
2145
+ },
2146
+ schemas () {
2147
+ return modelTools.map((tool)=>({
2148
+ name: tool.name
2149
+ }));
2150
+ }
2151
+ },
2152
+ agents: {
2153
+ list () {
2154
+ return [];
2155
+ }
2156
+ },
2157
+ systemPrompt: {
2158
+ variable () {}
2159
+ },
2160
+ on (event, listener) {
2161
+ modelEvents.push(event);
2162
+ if (event === 'agent/request') modelRequest = listener;
2163
+ return ()=>{};
2164
+ },
2165
+ effect () {
2166
+ return ()=>{};
2167
+ },
2168
+ get (name) {
2169
+ if (name === 'subprocess') return {
2170
+ spawn: (_spec)=>fakeHandle({
2171
+ stdout: '{}'
2172
+ })
2173
+ };
2174
+ if (name === 'agents') return {
2175
+ list () {
2176
+ return [];
2177
+ }
2178
+ };
2179
+ return undefined;
2180
+ }
2181
+ };
2182
+ apply(modelCtx, {
2183
+ stateRoot: root,
2184
+ timeoutMs: 20,
2185
+ hbcliBin: '',
2186
+ sessionAccess: 'off',
2187
+ benchmarkEnvironmentConfigPath: configPath
2188
+ });
2189
+ assert.deepEqual(modelTools.map((tool)=>tool.name), [
2190
+ 'gotry_benchmark_environment'
2191
+ ], 'benchmark model override does not re-enable product tools');
2192
+ assert.ok(modelEvents.includes('agent/request'), 'benchmark mode preserves the model override hook');
2193
+ assert.ok(modelRequest);
2194
+ assert.deepEqual(await modelRequest({}, async ()=>({
2195
+ provider: 'persisted',
2196
+ model: 'old',
2197
+ reasoningEffort: 'high',
2198
+ marker: 'kept'
2199
+ })), {
2200
+ provider: 'deepseek-official',
2201
+ model: 'round7-model-preserved',
2202
+ marker: 'kept'
2203
+ }, 'benchmark model override remains effective and replaces the persisted model only');
2204
+ } finally{
2205
+ if (originalModelOverride === undefined) delete process.env.GOTRY_LLM_MODEL;
2206
+ else process.env.GOTRY_LLM_MODEL = originalModelOverride;
2207
+ }
2208
+ const validConfig = {
2209
+ schema_version: 'gotry_benchmark_environment_bridge_v2',
2210
+ enabled: true,
2211
+ executable: process.execPath,
2212
+ cwd: root,
2213
+ argv_prefix: [
2214
+ '-m',
2215
+ 'agent_env.cli',
2216
+ '--lang',
2217
+ 'en'
2218
+ ],
2219
+ allowed_tools: [
2220
+ 'lookup'
2221
+ ],
2222
+ timeout_ms: 20,
2223
+ max_output_bytes: 4_096,
2224
+ terminal_output: {
2225
+ tag: 'done',
2226
+ max_bytes: 4_096
2227
+ },
2228
+ isolation: {
2229
+ mode: 'host-enforced',
2230
+ writes: 'forbidden',
2231
+ network: 'denied'
2232
+ }
2233
+ };
2234
+ const frozenProjection = registerBenchmarkEnvironmentBridge(configPath, ()=>{}, {
2235
+ spawn: (_spec)=>fakeHandle({
2236
+ stdout: '{"result":{}}'
2237
+ })
2238
+ });
2239
+ assert.equal(Object.isFrozen(frozenProjection), true, 'bridge projection is frozen');
2240
+ assert.equal(Object.isFrozen(frozenProjection.allowedTools), true, 'projected allowlist is frozen');
2241
+ assert.equal(Object.isFrozen(frozenProjection.terminal), true, 'projected terminal contract is frozen');
2242
+ assert.throws(()=>{
2243
+ frozenProjection.allowedTools.push('escape');
2244
+ }, TypeError);
2245
+ assert.throws(()=>{
2246
+ frozenProjection.terminal.tag = 'escape';
2247
+ }, TypeError);
2248
+ const bridgeRegistrationFor = (config, withSubprocess = true)=>{
2249
+ const path = join(root, `config-${Math.random().toString(36).slice(2)}.json`);
2250
+ writeFileSync(path, JSON.stringify(config));
2251
+ const tools = [];
2252
+ const freshCtx = {
2253
+ tools: {
2254
+ register (tool) {
2255
+ tools.push(tool);
2256
+ return ()=>{};
2257
+ },
2258
+ get (name) {
2259
+ return tools.find((tool)=>tool.name === name);
2260
+ },
2261
+ schemas () {
2262
+ return tools.map((tool)=>({
2263
+ name: tool.name
2264
+ }));
2265
+ }
2266
+ },
2267
+ systemPrompt: {
2268
+ variable () {}
2269
+ },
2270
+ on () {
2271
+ return ()=>{};
2272
+ },
2273
+ effect (action) {
2274
+ return action();
2275
+ },
2276
+ agents: {
2277
+ list () {
2278
+ return [];
2279
+ }
2280
+ },
2281
+ get (name) {
2282
+ if (name === 'subprocess') return this.subprocess;
2283
+ if (name === 'agents') return this.agents;
2284
+ },
2285
+ ...withSubprocess ? {
2286
+ subprocess: {
2287
+ spawn: (_spec)=>fakeHandle({
2288
+ stdout: '{"result":{}}'
2289
+ })
2290
+ }
2291
+ } : {}
2292
+ };
2293
+ apply(freshCtx, {
2294
+ stateRoot: root,
2295
+ timeoutMs: 20,
2296
+ hbcliBin: '',
2297
+ sessionAccess: 'off',
2298
+ benchmarkEnvironmentConfigPath: path
2299
+ });
2300
+ return tools.some((tool)=>tool.name === 'gotry_benchmark_environment');
2301
+ };
2302
+ assert.throws(()=>bridgeRegistrationFor({
2303
+ ...validConfig,
2304
+ unknown: true
2305
+ }), /benchmark environment bridge configuration unavailable/, 'unknown top-level config key fails hard');
2306
+ assert.throws(()=>bridgeRegistrationFor({
2307
+ ...validConfig,
2308
+ schema_version: 'gotry_benchmark_environment_bridge_v1'
2309
+ }), /benchmark environment bridge configuration unavailable/, 'v1 config cannot silently omit the Round 3 terminal semantics');
2310
+ assert.throws(()=>bridgeRegistrationFor({
2311
+ ...validConfig,
2312
+ allowed_tools: [
2313
+ 'lookup',
2314
+ 'lookup'
2315
+ ]
2316
+ }), /benchmark environment bridge configuration unavailable/, 'duplicate allowed tool fails hard');
2317
+ assert.throws(()=>bridgeRegistrationFor({
2318
+ ...validConfig,
2319
+ allowed_output_keys: {}
2320
+ }), /benchmark environment bridge configuration unavailable/, 'empty output-key mapping fails hard');
2321
+ assert.throws(()=>bridgeRegistrationFor({
2322
+ ...validConfig,
2323
+ allowed_output_keys: {
2324
+ lookup: []
2325
+ }
2326
+ }), /benchmark environment bridge configuration unavailable/, 'empty output-key allowlist fails hard');
2327
+ assert.throws(()=>bridgeRegistrationFor({
2328
+ ...validConfig,
2329
+ allowed_output_keys: {
2330
+ unknown: [
2331
+ 'city'
2332
+ ]
2333
+ }
2334
+ }), /benchmark environment bridge configuration unavailable/, 'output-key mapping for unknown tool fails hard');
2335
+ assert.throws(()=>bridgeRegistrationFor({
2336
+ ...validConfig,
2337
+ allowed_output_keys: {
2338
+ lookup: [
2339
+ 'city',
2340
+ 'city'
2341
+ ]
2342
+ }
2343
+ }), /benchmark environment bridge configuration unavailable/, 'duplicate output key fails hard');
2344
+ assert.throws(()=>bridgeRegistrationFor({
2345
+ ...validConfig,
2346
+ allowed_output_keys: {
2347
+ lookup: [
2348
+ 'not a key'
2349
+ ]
2350
+ }
2351
+ }), /benchmark environment bridge configuration unavailable/, 'non-identifier output key fails hard');
2352
+ assert.throws(()=>bridgeRegistrationFor({
2353
+ ...validConfig,
2354
+ argv_prefix: [
2355
+ 'agent\n--unsafe'
2356
+ ]
2357
+ }), /benchmark environment bridge configuration unavailable/, 'argv control separator fails hard');
2358
+ assert.throws(()=>bridgeRegistrationFor({
2359
+ ...validConfig,
2360
+ allowed_tools: [
2361
+ 'lookup;rm'
2362
+ ]
2363
+ }), /benchmark environment bridge configuration unavailable/, 'shell metacharacter tool identifier fails hard');
2364
+ assert.throws(()=>bridgeRegistrationFor({
2365
+ ...validConfig,
2366
+ isolation: {
2367
+ mode: 'host-enforced',
2368
+ writes: 'forbidden'
2369
+ }
2370
+ }), /benchmark environment bridge configuration unavailable/, 'incomplete isolation policy fails hard');
2371
+ const { terminal_output: _terminalOutput, ...missingTerminalConfig } = validConfig;
2372
+ assert.throws(()=>bridgeRegistrationFor(missingTerminalConfig), /benchmark environment bridge configuration unavailable/, 'terminal output contract is required');
2373
+ assert.throws(()=>bridgeRegistrationFor({
2374
+ ...validConfig,
2375
+ terminal_output: {
2376
+ tag: '1invalid',
2377
+ max_bytes: 4_096
2378
+ }
2379
+ }), /benchmark environment bridge configuration unavailable/, 'terminal tag must be an identifier');
2380
+ assert.throws(()=>bridgeRegistrationFor({
2381
+ ...validConfig,
2382
+ terminal_output: {
2383
+ tag: 'done',
2384
+ max_bytes: 0
2385
+ }
2386
+ }), /benchmark environment bridge configuration unavailable/, 'terminal output lower bound fails hard');
2387
+ assert.throws(()=>bridgeRegistrationFor({
2388
+ ...validConfig,
2389
+ terminal_output: {
2390
+ tag: 'done',
2391
+ max_bytes: 1024 * 1024 + 1
2392
+ }
2393
+ }), /benchmark environment bridge configuration unavailable/, 'terminal output upper bound fails hard');
2394
+ assert.throws(()=>bridgeRegistrationFor({
2395
+ ...validConfig,
2396
+ terminal_output: {
2397
+ tag: 'done',
2398
+ max_bytes: 4_096,
2399
+ extra: true
2400
+ }
2401
+ }), /benchmark environment bridge configuration unavailable/, 'terminal output rejects unknown keys');
2402
+ assert.throws(()=>bridgeRegistrationFor(validConfig, false), /benchmark environment bridge subprocess unavailable/, 'explicit opt-in without an active subprocess provider fails hard');
2403
+ assert.equal(loadConfigRegistration(configPath, root), true, 'owner-local regular 0644 config remains valid');
2404
+ assert.throws(()=>loadConfigRegistration('bridge.json', root), /benchmark environment bridge configuration unavailable/, 'relative config path fails hard');
2405
+ const symlinkPath = join(root, 'bridge-symlink.json');
2406
+ symlinkSync(configPath, symlinkPath);
2407
+ assert.equal(lstatSync(symlinkPath).isSymbolicLink(), true);
2408
+ assert.throws(()=>loadConfigRegistration(symlinkPath, root), /benchmark environment bridge configuration unavailable/, 'symlink config path fails hard');
2409
+ const widePath = join(root, 'bridge-wide.json');
2410
+ writeFileSync(widePath, JSON.stringify(validConfig));
2411
+ chmodSync(widePath, 0o666);
2412
+ assert.throws(()=>loadConfigRegistration(widePath, root), /benchmark environment bridge configuration unavailable/, 'group/world writable config fails hard');
2413
+ const legacyConfigPath = join(root, 'legacy-bridge.json');
2414
+ writeFileSync(legacyConfigPath, JSON.stringify(validConfig));
2415
+ const legacyTools = [];
2416
+ const legacyCtx = {
2417
+ tools: {
2418
+ register (tool) {
2419
+ legacyTools.push(tool);
2420
+ return ()=>{};
2421
+ },
2422
+ get (name) {
2423
+ return legacyTools.find((tool)=>tool.name === name);
2424
+ },
2425
+ schemas () {
2426
+ return legacyTools.map((tool)=>({
2427
+ name: tool.name
2428
+ }));
2429
+ }
2430
+ },
2431
+ systemPrompt: {
2432
+ variable () {}
2433
+ },
2434
+ on () {
2435
+ return ()=>{};
2436
+ },
2437
+ effect (_action) {
2438
+ return ()=>{};
2439
+ },
2440
+ agents: {
2441
+ list () {
2442
+ return [];
2443
+ }
2444
+ },
2445
+ get (name) {
2446
+ if (name === 'subprocess') return this.subprocess;
2447
+ if (name === 'agents') return this.agents;
2448
+ },
2449
+ subprocess: {
2450
+ spawn: (_spec)=>fakeHandle({
2451
+ stdout: '{"result":{"city":"Dubai"}}'
2452
+ })
2453
+ }
2454
+ };
2455
+ apply(legacyCtx, {
2456
+ stateRoot: root,
2457
+ timeoutMs: 20,
2458
+ hbcliBin: '',
2459
+ sessionAccess: 'off',
2460
+ benchmarkEnvironmentConfigPath: legacyConfigPath
2461
+ });
2462
+ const legacyBridge = legacyTools.find((tool)=>tool.name === 'gotry_benchmark_environment');
2463
+ assert.ok(legacyBridge?.execute, 'legacy config without allowed_output_keys registers the bridge');
2464
+ assert.deepEqual(await legacyBridge.execute({
2465
+ query: {
2466
+ action: 'call',
2467
+ tool: 'lookup',
2468
+ arguments: {
2469
+ city: 'Dubai'
2470
+ }
2471
+ }
2472
+ }, null), {
2473
+ ok: true,
2474
+ result: {
2475
+ city: 'Dubai'
2476
+ }
2477
+ }, 'legacy config without allowed_output_keys still executes safe structured output');
2478
+ console.log('BENCHMARK ENVIRONMENT BRIDGE TESTS: registration + TDD bridge contract assertions');
2479
+ } catch (error) {
2480
+ console.error(error);
2481
+ process.exitCode = 1;
2482
+ } finally{
2483
+ for (const [name, value] of ambientSentinels){
2484
+ if (value === undefined) delete process.env[name];
2485
+ else process.env[name] = value;
2486
+ }
2487
+ rmSync(root, {
2488
+ recursive: true,
2489
+ force: true
2490
+ });
2491
+ }
2492
+ function loadConfigRegistration(path, stateRoot) {
2493
+ const tools = [];
2494
+ const freshCtx = {
2495
+ tools: {
2496
+ register (tool) {
2497
+ tools.push(tool);
2498
+ return ()=>{};
2499
+ },
2500
+ get (name) {
2501
+ return tools.find((tool)=>tool.name === name);
2502
+ },
2503
+ schemas () {
2504
+ return tools.map((tool)=>({
2505
+ name: tool.name
2506
+ }));
2507
+ }
2508
+ },
2509
+ systemPrompt: {
2510
+ variable () {}
2511
+ },
2512
+ agents: {
2513
+ list () {
2514
+ return [];
2515
+ }
2516
+ },
2517
+ on () {
2518
+ return ()=>{};
2519
+ },
2520
+ effect (_action) {
2521
+ return ()=>{};
2522
+ },
2523
+ get (name) {
2524
+ if (name === 'subprocess') return this.subprocess;
2525
+ if (name === 'agents') return this.agents;
2526
+ },
2527
+ subprocess: {
2528
+ spawn: (_spec)=>fakeHandle({
2529
+ stdout: '{"result":{}}'
2530
+ })
2531
+ }
2532
+ };
2533
+ apply(freshCtx, {
2534
+ stateRoot,
2535
+ timeoutMs: 20,
2536
+ hbcliBin: '',
2537
+ sessionAccess: 'off',
2538
+ benchmarkEnvironmentConfigPath: path
2539
+ });
2540
+ return tools.some((tool)=>tool.name === 'gotry_benchmark_environment');
2541
+ }
2542
+
2543
+
2544
+ //# sourceURL=ts/scripts/benchmark-environment-bridge-tests.ts