@danceiny/gotry 0.0.1-rc.16 → 0.0.1-rc.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +47 -13
- package/README.zh-CN.md +20 -10
- package/bin/gotry-booking-copilot.js +53 -0
- package/bin/gotry-bootstrap.js +111 -51
- package/bin/gotry-inner.js +442 -60
- package/bin/gotry-runtime-resolution.d.ts +27 -0
- package/bin/gotry-runtime-resolution.js +50 -0
- package/bin/gotry.js +1 -1
- package/cordis.gotry-patch.yml +5 -1
- package/dist/capabilities/agent-reach-deep.js +1 -1
- package/dist/capabilities/agent-reach.js +1 -1
- package/dist/capabilities/anything.js +1 -1
- package/dist/capabilities/artifacts.js +1 -1
- package/dist/capabilities/effect.js +1 -1
- package/dist/capabilities/fact-log.js +1 -1
- package/dist/capabilities/flyai.js +20 -6
- package/dist/capabilities/hbcli.js +1 -1
- package/dist/capabilities/incident-log.js +1 -1
- package/dist/capabilities/model-override.js +18 -0
- package/dist/capabilities/opensky.js +1 -1
- package/dist/capabilities/resilience.js +1 -1
- package/dist/capabilities/session/action-cache.js +1 -1
- package/dist/capabilities/session/adapters/ctrip-flight.js +1 -1
- package/dist/capabilities/session/adapters/meituan-local.js +1 -1
- package/dist/capabilities/session/benchmark.js +1 -1
- package/dist/capabilities/session/extension-bridge.js +19 -6
- package/dist/capabilities/session/extension-channel.js +3 -3
- package/dist/capabilities/session/extension-distribution.js +234 -0
- package/dist/capabilities/session/extract.js +1 -1
- package/dist/capabilities/session/golden-score.js +92 -0
- package/dist/capabilities/session/health-watch.js +1 -1
- package/dist/capabilities/session/read-guard.js +1 -1
- package/dist/capabilities/session/static-flight-golden.js +137 -0
- package/dist/capabilities/session/transport.js +1 -1
- package/dist/capabilities/session/wizard.js +21 -251
- package/dist/capabilities/session-consent.js +1 -1
- package/dist/capabilities/session-login.js +11 -5
- package/dist/capabilities/session-search.js +46 -4
- package/dist/capabilities/weather.js +168 -46
- package/dist/data/session-golden-20.json +25 -0
- package/dist/data/sf-golden-manifest.json +102 -0
- package/dist/data/sf-static-routes.json +91 -0
- package/dist/scripts/action-cache-tests.js +1 -1
- package/dist/scripts/agent-planning-budget-e2e.js +227 -0
- package/dist/scripts/agent-planning-budget-tests.js +173 -0
- package/dist/scripts/agent-reach-deep-tests.js +1 -1
- package/dist/scripts/agent-reach-tests.js +1 -1
- package/dist/scripts/agent-reach-wrapper-tests.js +1 -1
- package/dist/scripts/anything-tests.js +1 -1
- package/dist/scripts/async-collect.js +1 -1
- package/dist/scripts/benchmark-environment-bridge-e2e.js +981 -0
- package/dist/scripts/benchmark-environment-bridge-tests.js +2544 -0
- package/dist/scripts/booking-copilot-availability-ledger-binding-tests.js +335 -0
- package/dist/scripts/booking-copilot-availability-policy-v2-tests.js +1355 -0
- package/dist/scripts/booking-copilot-bin-proof-tests.js +89 -0
- package/dist/scripts/booking-copilot-crossrepo-fixture-server.js +113 -0
- package/dist/scripts/booking-copilot-dsh-core-proof-tests.js +356 -0
- package/dist/scripts/booking-copilot-dsh-planner-proof-tests.js +557 -0
- package/dist/scripts/booking-copilot-dsh-plugin-proof-tests.js +140 -0
- package/dist/scripts/booking-copilot-event-sequence-concurrency-proof-tests.js +202 -0
- package/dist/scripts/booking-copilot-gap-code-contract-proof-tests.js +102 -0
- package/dist/scripts/booking-copilot-operation-ledger-concurrency-proof-tests.js +374 -0
- package/dist/scripts/booking-copilot-receipt-ledger-concurrency-proof-tests.js +447 -0
- package/dist/scripts/booking-copilot-runtime-proof-tests.js +205 -0
- package/dist/scripts/booking-copilot-server-proof-tests.js +317 -0
- package/dist/scripts/booking-copilot-startup-proof-tests.js +283 -0
- package/dist/scripts/booking-copilot-v2-runtime-proof-tests.js +3935 -0
- package/dist/scripts/booking-saga-tests.js +1 -1
- package/dist/scripts/booking-surface-contract-proof-tests.js +393 -0
- package/dist/scripts/booking-surface-v2-contract-proof-tests.js +1174 -0
- package/dist/scripts/bootstrap-tests.js +30 -10
- package/dist/scripts/build-changelog.js +1 -1
- package/dist/scripts/changelog-tests.js +1 -1
- package/dist/scripts/companion-tests.js +1 -1
- package/dist/scripts/diff-test.js +1 -1
- package/dist/scripts/dsh-runtime-closure-tests.js +225 -0
- package/dist/scripts/dsh-runtime-closure.js +150 -0
- package/dist/scripts/effect-tests.js +1 -1
- package/dist/scripts/engine-run.js +1 -1
- package/dist/scripts/engine-tests.js +1 -1
- package/dist/scripts/evaluation-cadence-tests.js +347 -0
- package/dist/scripts/evaluation-contract-tests.js +574 -0
- package/dist/scripts/extension-distribution-cli.js +35 -0
- package/dist/scripts/extension-distribution-tests.js +341 -0
- package/dist/scripts/extension-tests.js +128 -23
- package/dist/scripts/fact-gate-tests.js +1 -1
- package/dist/scripts/flyai-tests.js +1 -1
- package/dist/scripts/hbcli-e2e-tests.js +1 -1
- package/dist/scripts/hbcli-tests.js +1 -1
- package/dist/scripts/health-watch-cli.js +1 -1
- package/dist/scripts/i18n-tests.js +1 -1
- package/dist/scripts/incident-tests.js +1 -1
- package/dist/scripts/journey-tests.js +1 -1
- package/dist/scripts/ledger-tests.js +1 -1
- package/dist/scripts/ledger-workflow-crash.js +1 -1
- package/dist/scripts/memory-capture-tests.js +1 -1
- package/dist/scripts/memory-decay-tests.js +1 -1
- package/dist/scripts/memory-metrics.js +1 -1
- package/dist/scripts/memory-value-report.js +1 -1
- package/dist/scripts/model-override-e2e.js +176 -0
- package/dist/scripts/nightly-evidence-tests.js +1 -1
- package/dist/scripts/nightly-evidence.js +1 -1
- package/dist/scripts/nudge-digest.js +1 -1
- package/dist/scripts/onboarding-tests.js +21 -53
- package/dist/scripts/opensky-check.js +1 -1
- package/dist/scripts/opensky-tests.js +1 -1
- package/dist/scripts/pnpm-dsh-closure-proof.js +20 -0
- package/dist/scripts/price-drift-tests.js +1 -1
- package/dist/scripts/price-drift-watch.js +1 -1
- package/dist/scripts/probe-poi-tests.js +1 -1
- package/dist/scripts/product-metrics.js +1 -1
- package/dist/scripts/publish-preverify.js +40 -4
- package/dist/scripts/realtime-pricing-tests.js +1 -1
- package/dist/scripts/replay-async.js +1 -1
- package/dist/scripts/replay-real.js +1 -1
- package/dist/scripts/replay.js +1 -1
- package/dist/scripts/session-attach-diagnose.js +1 -1
- package/dist/scripts/session-attach-poc.js +1 -1
- package/dist/scripts/session-benchmark.js +1 -1
- package/dist/scripts/session-extract-tests.js +1 -1
- package/dist/scripts/session-login.js +1 -1
- package/dist/scripts/session-tests.js +70 -20
- package/dist/scripts/sf-live-benchmark.js +338 -0
- package/dist/scripts/sf-live-cli-tests.js +21 -0
- package/dist/scripts/sf-soft-score-tests.js +108 -0
- package/dist/scripts/sf-summary.js +93 -0
- package/dist/scripts/skeleton-check.js +1 -1
- package/dist/scripts/skeleton-integration-test.js +1 -1
- package/dist/scripts/skills-contract-tests.js +1 -1
- package/dist/scripts/smoke-session-gate-tests.js +29 -0
- package/dist/scripts/smoke.js +79 -33
- package/dist/scripts/state-cli-tests.js +1 -1
- package/dist/scripts/state-cli.js +1 -1
- package/dist/scripts/static-golden-tests.js +299 -0
- package/dist/scripts/time-eval-tests.js +1 -1
- package/dist/scripts/travel-timeline-tests.js +1 -1
- package/dist/scripts/unified-tests.js +1 -1
- package/dist/scripts/weather-tests.js +694 -44
- package/dist/scripts/z3-race-tests.js +1 -1
- package/dist/src/artifact-gate.js +1 -1
- package/dist/src/benchmark-agent-conformance.js +370 -0
- package/dist/src/benchmark-environment-bridge.js +384 -0
- package/dist/src/benchmark-headless-child-diagnostics.js +173 -0
- package/dist/src/benchmark-tool-isolation.js +124 -0
- package/dist/src/bookable-facts.js +1 -1
- package/dist/src/booking-saga.js +1 -1
- package/dist/src/booking-surface/availability-policy-v2.js +830 -0
- package/dist/src/booking-surface/canonical-schema.js +113 -0
- package/dist/src/booking-surface/contracts-v2.js +89 -0
- package/dist/src/booking-surface/contracts.js +46 -0
- package/dist/src/booking-surface/dsh-planner.js +453 -0
- package/dist/src/booking-surface/dsh-plugin.js +93 -0
- package/dist/src/booking-surface/error-codes.js +94 -0
- package/dist/src/booking-surface/index.js +15 -0
- package/dist/src/booking-surface/profile.js +68 -0
- package/dist/src/booking-surface/runtime-v2.js +1771 -0
- package/dist/src/booking-surface/runtime.js +351 -0
- package/dist/src/booking-surface/server-v2.js +334 -0
- package/dist/src/booking-surface/server.js +302 -0
- package/dist/src/booking-surface/startup.js +159 -0
- package/dist/src/booking-surface/validation-v2.js +319 -0
- package/dist/src/booking-surface/validation.js +809 -0
- package/dist/src/bridge.js +1 -1
- package/dist/src/companions.js +1 -1
- package/dist/src/contracts.js +1 -1
- package/dist/src/dsh-llm.js +1 -1
- package/dist/src/engine.js +1 -1
- package/dist/src/evaluation-cadence.js +234 -0
- package/dist/src/evaluation-contracts.js +906 -0
- package/dist/src/i18n.js +1 -1
- package/dist/src/index.js +50 -19
- package/dist/src/journey.js +1 -1
- package/dist/src/loop.js +1 -1
- package/dist/src/memory-capture.js +1 -1
- package/dist/src/memory-decay.js +1 -1
- package/dist/src/memory-utility.js +1 -1
- package/dist/src/mock-llm.js +1 -1
- package/dist/src/model.js +1 -1
- package/dist/src/realtime-pricing.js +1 -1
- package/dist/src/slot-spec.js +1 -1
- package/dist/src/state-ledger.js +2 -1
- package/dist/src/time-anchor.js +1 -1
- package/dist/src/tool-budget.js +136 -0
- package/dist/src/tool-packet.js +1 -1
- package/dist/src/travel-slots.js +1 -1
- package/dist/src/travel-timeline.js +1 -1
- package/dist/src/unified.js +1 -1
- package/dist/src/wish-pool.js +1 -1
- package/dist/src/z3-shared.js +1 -1
- package/extension/README.md +31 -7
- package/package.json +286 -11
- package/schemas/booking.surface.v1.schema.json +927 -0
- package/schemas/booking.surface.v2.schema.json +61 -0
- package/ts/capabilities/flyai.ts +16 -3
- package/ts/capabilities/session/extension-bridge.ts +37 -14
- package/ts/capabilities/session/extension-channel.ts +6 -3
- package/ts/capabilities/session/extension-distribution.ts +264 -0
- package/ts/capabilities/session/golden-score.ts +139 -0
- package/ts/capabilities/session/health-watch.ts +1 -1
- package/ts/capabilities/session/static-flight-golden.ts +209 -0
- package/ts/capabilities/session/wizard.ts +34 -176
- package/ts/capabilities/session-login.ts +15 -5
- package/ts/capabilities/session-search.ts +40 -3
- package/ts/capabilities/weather.ts +141 -52
- package/ts/package.json +3 -3
- package/ts/src/benchmark-agent-conformance.ts +448 -0
- package/ts/src/benchmark-environment-bridge.ts +348 -0
- package/ts/src/benchmark-headless-child-diagnostics.ts +184 -0
- package/ts/src/benchmark-tool-isolation.ts +166 -0
- package/ts/src/booking-surface/availability-policy-v2.ts +523 -0
- package/ts/src/booking-surface/canonical-schema.js +113 -0
- package/ts/src/booking-surface/contracts-v2.ts +118 -0
- package/ts/src/booking-surface/contracts.ts +380 -0
- package/ts/src/booking-surface/dsh-planner.ts +452 -0
- package/ts/src/booking-surface/dsh-plugin.js +93 -0
- package/ts/src/booking-surface/error-codes.ts +101 -0
- package/ts/src/booking-surface/index.ts +12 -0
- package/ts/src/booking-surface/profile.ts +42 -0
- package/ts/src/booking-surface/runtime-v2.ts +1466 -0
- package/ts/src/booking-surface/runtime.ts +483 -0
- package/ts/src/booking-surface/server-v2.ts +247 -0
- package/ts/src/booking-surface/server.ts +324 -0
- package/ts/src/booking-surface/startup.ts +196 -0
- package/ts/src/booking-surface/validation-v2.ts +205 -0
- package/ts/src/booking-surface/validation.ts +453 -0
- package/ts/src/index.ts +64 -11
- package/ts/src/state-ledger.ts +1 -0
- package/ts/src/tool-budget.ts +165 -0
- package/dist/scripts/wizard-bootstrap.js +0 -32
|
@@ -0,0 +1,981 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import { createServer } from 'node:http';
|
|
3
|
+
import { chmodSync, cpSync, existsSync, mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
|
|
4
|
+
import { createRequire } from 'node:module';
|
|
5
|
+
import { tmpdir } from 'node:os';
|
|
6
|
+
import { dirname, join } from 'node:path';
|
|
7
|
+
import { pathToFileURL } from 'node:url';
|
|
8
|
+
import { spawn } from 'node:child_process';
|
|
9
|
+
import { benchmarkRuntimeSupported, selectDshCwd, selectDshRuntime, supportsNodeVersion } from '../../bin/gotry-runtime-resolution.js';
|
|
10
|
+
const ROOT = join(import.meta.dirname, '..', '..');
|
|
11
|
+
const BIN = join(ROOT, 'bin', 'gotry-inner.js');
|
|
12
|
+
const TOOL = 'gotry_benchmark_environment';
|
|
13
|
+
const MARKER = 'BENCHMARK_BRIDGE_LOOKUP_OK';
|
|
14
|
+
const TIMEOUT_MS = 30_000;
|
|
15
|
+
const TSX_LOADER = pathToFileURL(createRequire(import.meta.url).resolve('tsx')).href;
|
|
16
|
+
function runRuntimeProbe(options) {
|
|
17
|
+
const fixture = mkdtempSync(join(tmpdir(), 'gotry-runtime-probe-'));
|
|
18
|
+
try {
|
|
19
|
+
writeFileSync(join(fixture, 'package.json'), JSON.stringify({
|
|
20
|
+
name: 'runtime-probe',
|
|
21
|
+
type: 'module'
|
|
22
|
+
}));
|
|
23
|
+
const writeDsh = (root, version)=>{
|
|
24
|
+
mkdirSync(join(root, 'lib'), {
|
|
25
|
+
recursive: true
|
|
26
|
+
});
|
|
27
|
+
writeFileSync(join(root, 'package.json'), JSON.stringify({
|
|
28
|
+
name: '@deepseek-ai/dsh',
|
|
29
|
+
version,
|
|
30
|
+
type: 'module'
|
|
31
|
+
}));
|
|
32
|
+
writeFileSync(join(root, 'lib', 'bin.js'), 'export {}\n');
|
|
33
|
+
};
|
|
34
|
+
if (options.rootVersion) writeDsh(join(fixture, 'node_modules', '@deepseek-ai', 'dsh'), options.rootVersion);
|
|
35
|
+
if (options.vendorVersion) writeDsh(join(fixture, 'ts', 'dsh-runtime', 'node_modules', '@deepseek-ai', 'dsh'), options.vendorVersion);
|
|
36
|
+
const runtime = selectDshRuntime({
|
|
37
|
+
repoRoot: fixture,
|
|
38
|
+
rootResolver: createRequire(join(fixture, 'package.json')),
|
|
39
|
+
benchmark: options.benchmark === true
|
|
40
|
+
});
|
|
41
|
+
return runtime ? {
|
|
42
|
+
source: runtime.source,
|
|
43
|
+
version: runtime.version
|
|
44
|
+
} : null;
|
|
45
|
+
} finally{
|
|
46
|
+
rmSync(fixture, {
|
|
47
|
+
recursive: true,
|
|
48
|
+
force: true
|
|
49
|
+
});
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
function assertRuntimeSelectionAndVersionGuards() {
|
|
53
|
+
const sourcePriority = runRuntimeProbe({
|
|
54
|
+
rootVersion: '0.1.2-alpha.3',
|
|
55
|
+
vendorVersion: '0.1.2-alpha.1'
|
|
56
|
+
});
|
|
57
|
+
assert.deepEqual(sourcePriority, {
|
|
58
|
+
source: 'root',
|
|
59
|
+
version: '0.1.2-alpha.3'
|
|
60
|
+
}, 'source checkout uses the root dsh package even when legacy vendor is alpha.1');
|
|
61
|
+
const legacyFallback = runRuntimeProbe({
|
|
62
|
+
vendorVersion: '0.1.2-alpha.1'
|
|
63
|
+
});
|
|
64
|
+
assert.deepEqual(legacyFallback, {
|
|
65
|
+
source: 'legacy-vendored',
|
|
66
|
+
version: '0.1.2-alpha.1'
|
|
67
|
+
}, 'non-benchmark source checkout may use the legacy vendored dsh fallback');
|
|
68
|
+
const wrongBenchmarkVersion = runRuntimeProbe({
|
|
69
|
+
rootVersion: '0.1.2-alpha.1',
|
|
70
|
+
vendorVersion: '0.1.2-alpha.1',
|
|
71
|
+
benchmark: true
|
|
72
|
+
});
|
|
73
|
+
assert.deepEqual(wrongBenchmarkVersion, {
|
|
74
|
+
source: 'root',
|
|
75
|
+
version: '0.1.2-alpha.1'
|
|
76
|
+
});
|
|
77
|
+
assert.equal(benchmarkRuntimeSupported(wrongBenchmarkVersion), false, 'benchmark mode rejects a non-alpha.3 dsh runtime before spawn');
|
|
78
|
+
assert.equal(runRuntimeProbe({
|
|
79
|
+
vendorVersion: '0.1.2-alpha.1',
|
|
80
|
+
benchmark: true
|
|
81
|
+
}), null, 'benchmark mode never falls back to legacy vendored dsh');
|
|
82
|
+
assert.equal(supportsNodeVersion('22.14.0'), false, 'Node 22.14 is rejected before dsh resolution/spawn');
|
|
83
|
+
assert.equal(supportsNodeVersion('22.15.0'), true, 'Node 22.15 is the accepted minimum');
|
|
84
|
+
assert.equal(supportsNodeVersion('24.0.0'), true, 'newer Node majors remain accepted');
|
|
85
|
+
const invocationCwd = join(tmpdir(), 'gotry-runtime-invocation-cwd');
|
|
86
|
+
const sourceStateRoot = join(ROOT, 'ts/dsh-runtime');
|
|
87
|
+
assert.equal(selectDshCwd({
|
|
88
|
+
repoRoot: ROOT,
|
|
89
|
+
invocationCwd,
|
|
90
|
+
sourceCheckoutMode: true,
|
|
91
|
+
benchmark: false
|
|
92
|
+
}), sourceStateRoot, 'source checkout normal mode keeps dsh cwd at ts/dsh-runtime for gotry-state continuity');
|
|
93
|
+
assert.equal(selectDshCwd({
|
|
94
|
+
repoRoot: ROOT,
|
|
95
|
+
invocationCwd,
|
|
96
|
+
sourceCheckoutMode: true,
|
|
97
|
+
benchmark: true
|
|
98
|
+
}), invocationCwd, 'source checkout benchmark mode uses the isolated invocation cwd');
|
|
99
|
+
assert.equal(selectDshCwd({
|
|
100
|
+
repoRoot: ROOT,
|
|
101
|
+
invocationCwd,
|
|
102
|
+
sourceCheckoutMode: false,
|
|
103
|
+
benchmark: false
|
|
104
|
+
}), invocationCwd, 'installed package normal mode uses the user invocation cwd');
|
|
105
|
+
}
|
|
106
|
+
function writeResolutionProbe(path, resultPath, block) {
|
|
107
|
+
writeFileSync(path, `
|
|
108
|
+
const fs = require('node:fs')
|
|
109
|
+
const Module = require('node:module')
|
|
110
|
+
const resultPath = ${JSON.stringify(resultPath)}
|
|
111
|
+
const isOptional = value => typeof value === 'string' && (value.includes('dsh-calendar') || value.includes('dsh-map-tools'))
|
|
112
|
+
const record = (kind, target) => {
|
|
113
|
+
if (isOptional(target)) fs.appendFileSync(resultPath, JSON.stringify({ pid: process.pid, kind, target }) + '\\n')
|
|
114
|
+
}
|
|
115
|
+
const observe = request => {
|
|
116
|
+
if (typeof request !== 'string') return
|
|
117
|
+
record('resolve', request)
|
|
118
|
+
if (${JSON.stringify(block)} && isOptional(request)) {
|
|
119
|
+
const error = new Error('optional benchmark probe isolation')
|
|
120
|
+
error.code = 'MODULE_NOT_FOUND'
|
|
121
|
+
throw error
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
const originalResolve = Module._resolveFilename
|
|
125
|
+
Module._resolveFilename = function (request, parent, isMain, options) {
|
|
126
|
+
observe(request)
|
|
127
|
+
return originalResolve.call(this, request, parent, isMain, options)
|
|
128
|
+
}
|
|
129
|
+
const originalFindPath = Module._findPath
|
|
130
|
+
Module._findPath = function (request, paths, isMain) {
|
|
131
|
+
observe(request)
|
|
132
|
+
return originalFindPath.call(this, request, paths, isMain)
|
|
133
|
+
}
|
|
134
|
+
const originalExistsSync = fs.existsSync
|
|
135
|
+
fs.existsSync = function (target) {
|
|
136
|
+
record('existsSync', target)
|
|
137
|
+
if (${JSON.stringify(block)} && isOptional(target)) return false
|
|
138
|
+
return originalExistsSync.call(this, target)
|
|
139
|
+
}
|
|
140
|
+
const originalCreateRequire = Module.createRequire
|
|
141
|
+
Module.createRequire = function (...args) {
|
|
142
|
+
const required = originalCreateRequire.apply(this, args)
|
|
143
|
+
const originalRequiredResolve = required.resolve
|
|
144
|
+
required.resolve = function (request, ...args) {
|
|
145
|
+
record('resolve', request)
|
|
146
|
+
if (${JSON.stringify(block)} && isOptional(request)) {
|
|
147
|
+
const error = new Error('optional benchmark probe isolation')
|
|
148
|
+
error.code = 'MODULE_NOT_FOUND'
|
|
149
|
+
throw error
|
|
150
|
+
}
|
|
151
|
+
return originalRequiredResolve.call(required, request, ...args)
|
|
152
|
+
}
|
|
153
|
+
required.resolve.paths = originalRequiredResolve.paths
|
|
154
|
+
return required
|
|
155
|
+
}
|
|
156
|
+
Module.syncBuiltinESMExports()
|
|
157
|
+
`, {
|
|
158
|
+
mode: 0o600,
|
|
159
|
+
flag: 'wx'
|
|
160
|
+
});
|
|
161
|
+
}
|
|
162
|
+
function sse(payload) {
|
|
163
|
+
return `data: ${JSON.stringify(payload)}\n\n`;
|
|
164
|
+
}
|
|
165
|
+
function finalText(text) {
|
|
166
|
+
return sse({
|
|
167
|
+
id: 'bridge-final',
|
|
168
|
+
object: 'chat.completion.chunk',
|
|
169
|
+
choices: [
|
|
170
|
+
{
|
|
171
|
+
delta: {
|
|
172
|
+
role: 'assistant',
|
|
173
|
+
content: text
|
|
174
|
+
},
|
|
175
|
+
finish_reason: null
|
|
176
|
+
}
|
|
177
|
+
]
|
|
178
|
+
}) + sse({
|
|
179
|
+
id: 'bridge-final-stop',
|
|
180
|
+
object: 'chat.completion.chunk',
|
|
181
|
+
choices: [
|
|
182
|
+
{
|
|
183
|
+
delta: {},
|
|
184
|
+
finish_reason: 'stop'
|
|
185
|
+
}
|
|
186
|
+
]
|
|
187
|
+
}) + 'data: [DONE]\n\n';
|
|
188
|
+
}
|
|
189
|
+
function toolCall(callId = 'bridge-call-1') {
|
|
190
|
+
return sse({
|
|
191
|
+
id: `bridge-${callId}`,
|
|
192
|
+
object: 'chat.completion.chunk',
|
|
193
|
+
choices: [
|
|
194
|
+
{
|
|
195
|
+
delta: {
|
|
196
|
+
role: 'assistant',
|
|
197
|
+
tool_calls: [
|
|
198
|
+
{
|
|
199
|
+
index: 0,
|
|
200
|
+
id: callId,
|
|
201
|
+
type: 'function',
|
|
202
|
+
function: {
|
|
203
|
+
name: TOOL,
|
|
204
|
+
arguments: JSON.stringify({
|
|
205
|
+
query: {
|
|
206
|
+
action: 'call',
|
|
207
|
+
tool: 'lookup',
|
|
208
|
+
arguments: {
|
|
209
|
+
city: 'Dubai'
|
|
210
|
+
}
|
|
211
|
+
}
|
|
212
|
+
})
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
]
|
|
216
|
+
},
|
|
217
|
+
finish_reason: null
|
|
218
|
+
}
|
|
219
|
+
]
|
|
220
|
+
}) + sse({
|
|
221
|
+
id: 'bridge-call-stop',
|
|
222
|
+
object: 'chat.completion.chunk',
|
|
223
|
+
choices: [
|
|
224
|
+
{
|
|
225
|
+
delta: {},
|
|
226
|
+
finish_reason: 'tool_calls'
|
|
227
|
+
}
|
|
228
|
+
]
|
|
229
|
+
}) + 'data: [DONE]\n\n';
|
|
230
|
+
}
|
|
231
|
+
function names(body) {
|
|
232
|
+
return (body.tools ?? []).map((t)=>{
|
|
233
|
+
const f = t.function;
|
|
234
|
+
return String(f?.name ?? t.name ?? '');
|
|
235
|
+
}).filter(Boolean);
|
|
236
|
+
}
|
|
237
|
+
function toolResultPresent(body) {
|
|
238
|
+
return (body.messages ?? []).some((m)=>m.role === 'tool' && JSON.stringify(m).includes(MARKER));
|
|
239
|
+
}
|
|
240
|
+
function anyToolResultPresent(body) {
|
|
241
|
+
return (body.messages ?? []).some((m)=>m.role === 'tool');
|
|
242
|
+
}
|
|
243
|
+
async function runCase(mode, executableOverride, extraEnv = {}) {
|
|
244
|
+
const requests = [];
|
|
245
|
+
let spawnTarget = '';
|
|
246
|
+
const server = createServer((req, res)=>{
|
|
247
|
+
const chunks = [];
|
|
248
|
+
req.on('data', (c)=>chunks.push(Buffer.from(c)));
|
|
249
|
+
req.on('end', ()=>{
|
|
250
|
+
let body = {};
|
|
251
|
+
try {
|
|
252
|
+
body = JSON.parse(Buffer.concat(chunks).toString());
|
|
253
|
+
} catch {}
|
|
254
|
+
requests.push(body);
|
|
255
|
+
res.writeHead(200, {
|
|
256
|
+
'content-type': 'text/event-stream'
|
|
257
|
+
});
|
|
258
|
+
if (mode === 'spawn-failed' && names(body).includes(TOOL) && !anyToolResultPresent(body) && spawnTarget) rmSync(spawnTarget, {
|
|
259
|
+
force: true
|
|
260
|
+
});
|
|
261
|
+
if (mode !== 'disabled' && mode !== 'invalid-path' && mode !== 'invalid-schema' && mode !== 'unsafe-config' && names(body).includes(TOOL) && !anyToolResultPresent(body)) res.end(toolCall());
|
|
262
|
+
else res.end(finalText(mode === 'enabled' ? '<benchmark_terminal>{"status":"succeeded"}</benchmark_terminal>' : '<benchmark_terminal>{"status":"succeeded"}</benchmark_terminal>'));
|
|
263
|
+
});
|
|
264
|
+
});
|
|
265
|
+
await new Promise((resolve)=>server.listen(0, '127.0.0.1', resolve));
|
|
266
|
+
const port = server.address().port;
|
|
267
|
+
const cwd = mkdtempSync(join(tmpdir(), 'gotry-bridge-cwd-'));
|
|
268
|
+
const dsh = mkdtempSync(join(tmpdir(), 'gotry-bridge-dsh-'));
|
|
269
|
+
const probe = join(cwd, 'benchmark-resolution-probe.cjs');
|
|
270
|
+
const probeResult = join(cwd, 'benchmark-resolution-hits.json');
|
|
271
|
+
writeFileSync(probeResult, '', {
|
|
272
|
+
mode: 0o600,
|
|
273
|
+
flag: 'wx'
|
|
274
|
+
});
|
|
275
|
+
writeResolutionProbe(probe, probeResult, mode === 'disabled');
|
|
276
|
+
const runner = join(cwd, 'synthetic-runner.js');
|
|
277
|
+
spawnTarget = join(cwd, 'synthetic-spawn-target.js');
|
|
278
|
+
const configPath = join(cwd, mode === 'invalid-path' ? 'benchmark-env-config-\n.json' : 'benchmark-env-config.json');
|
|
279
|
+
const runnerBody = mode === 'timeout' ? `setTimeout(() => {}, 60_000)` : mode === 'runner-failed' ? `process.stderr.write('PRIVATE_RUNNER_DIAGNOSTIC_DO_NOT_REFLECT'); process.exit(17)` : mode === 'output-truncated' ? `process.stdout.write('x'.repeat(20_000))` : mode === 'unexpected-output' ? `process.stdout.write(JSON.stringify({ result: { marker: '${MARKER}', leaked: [], unexpected: 'must-not-reflect' } }))` : `const forbidden = ['GOTRY_BENCHMARK_ENV_CONFIG', 'GOTRY_BENCHMARK_BRIDGE_PARENT_SECRET', 'LLM_API_KEY', 'LLM_BASE_URL', 'LLM_MODEL', 'DEEPSEEK_BASE_URL', 'GOTRY_LLM_MODEL', 'DATABASE_URL', 'SSH_AUTH_SOCK', 'AWS_PROFILE', 'HTTPS_PROXY']; const leaked = forbidden.filter(name => process.env[name] !== undefined); process.stdout.write(JSON.stringify({ result: { marker: '${MARKER}', leaked } }))`;
|
|
280
|
+
writeFileSync(runner, `if (process.argv.length !== 5 || process.argv[2] !== 'call' || process.argv[3] !== 'lookup' || JSON.parse(process.argv[4]).city !== 'Dubai') process.exit(2); ${runnerBody}`);
|
|
281
|
+
writeFileSync(spawnTarget, '#!/usr/bin/env node\nprocess.exit(0)\n', {
|
|
282
|
+
mode: 0o700
|
|
283
|
+
});
|
|
284
|
+
writeFileSync(configPath, JSON.stringify({
|
|
285
|
+
schema_version: mode === 'invalid-schema' ? 'invalid' : 'gotry_benchmark_environment_bridge_v2',
|
|
286
|
+
enabled: true,
|
|
287
|
+
executable: mode === 'spawn-failed' ? spawnTarget : process.execPath,
|
|
288
|
+
cwd,
|
|
289
|
+
argv_prefix: mode === 'spawn-failed' ? [
|
|
290
|
+
'placeholder'
|
|
291
|
+
] : [
|
|
292
|
+
runner
|
|
293
|
+
],
|
|
294
|
+
allowed_tools: [
|
|
295
|
+
'lookup'
|
|
296
|
+
],
|
|
297
|
+
allowed_output_keys: {
|
|
298
|
+
lookup: [
|
|
299
|
+
'marker',
|
|
300
|
+
'leaked'
|
|
301
|
+
]
|
|
302
|
+
},
|
|
303
|
+
timeout_ms: mode === 'timeout' ? 50 : 10_000,
|
|
304
|
+
max_output_bytes: mode === 'output-truncated' ? 1_024 : 4_096,
|
|
305
|
+
terminal_output: {
|
|
306
|
+
tag: 'benchmark_terminal',
|
|
307
|
+
max_bytes: 4_096
|
|
308
|
+
},
|
|
309
|
+
isolation: {
|
|
310
|
+
mode: 'host-enforced',
|
|
311
|
+
writes: 'forbidden',
|
|
312
|
+
network: 'denied'
|
|
313
|
+
}
|
|
314
|
+
}));
|
|
315
|
+
if (mode === 'unsafe-config') chmodSync(configPath, 0o666);
|
|
316
|
+
const env = {
|
|
317
|
+
...process.env,
|
|
318
|
+
DSH_TOOLS_MODE: 'both',
|
|
319
|
+
DSH_HOME: dsh,
|
|
320
|
+
LLM_API_KEY: 'synthetic-bridge-key',
|
|
321
|
+
LLM_BASE_URL: `http://127.0.0.1:${port}/v1`,
|
|
322
|
+
LLM_MODEL: 'synthetic-bridge-model',
|
|
323
|
+
DEEPSEEK_API_KEY: 'synthetic-bridge-key',
|
|
324
|
+
DEEPSEEK_BASE_URL: `http://127.0.0.1:${port}/v1`,
|
|
325
|
+
...mode !== 'disabled' ? {
|
|
326
|
+
DATABASE_URL: 'postgres://sentinel',
|
|
327
|
+
SSH_AUTH_SOCK: '/tmp/sentinel.sock',
|
|
328
|
+
AWS_PROFILE: 'sentinel-profile'
|
|
329
|
+
} : {},
|
|
330
|
+
GOTRY_BENCHMARK_ENV_CONFIG: mode === 'disabled' ? '' : configPath,
|
|
331
|
+
...mode !== 'disabled' ? {
|
|
332
|
+
GOTRY_BENCHMARK_BRIDGE_PARENT_SECRET: 'do-not-leak'
|
|
333
|
+
} : {},
|
|
334
|
+
...mode === 'debug-redaction' ? {
|
|
335
|
+
GOTRY_DEBUG: '1'
|
|
336
|
+
} : {},
|
|
337
|
+
...extraEnv,
|
|
338
|
+
NODE_OPTIONS: [
|
|
339
|
+
process.env.NODE_OPTIONS,
|
|
340
|
+
...!executableOverride ? [
|
|
341
|
+
`--import=${TSX_LOADER}`
|
|
342
|
+
] : [],
|
|
343
|
+
`--require=${probe}`
|
|
344
|
+
].filter(Boolean).join(' ')
|
|
345
|
+
};
|
|
346
|
+
for (const key of [
|
|
347
|
+
'GOTRY_LLM_MODEL',
|
|
348
|
+
'HTTP_PROXY',
|
|
349
|
+
'HTTPS_PROXY',
|
|
350
|
+
'ALL_PROXY',
|
|
351
|
+
'http_proxy',
|
|
352
|
+
'https_proxy',
|
|
353
|
+
'all_proxy'
|
|
354
|
+
])delete env[key];
|
|
355
|
+
env.NO_PROXY = '127.0.0.1,localhost';
|
|
356
|
+
if (mode !== 'disabled') env.HTTPS_PROXY = 'https://sentinel-proxy';
|
|
357
|
+
if (mode === 'disabled') delete env.GOTRY_BENCHMARK_ENV_CONFIG;
|
|
358
|
+
const executable = executableOverride || process.execPath;
|
|
359
|
+
const invocation = mode === 'web-mode' ? [
|
|
360
|
+
'web',
|
|
361
|
+
'--no-open'
|
|
362
|
+
] : [
|
|
363
|
+
mode === 'debug-redaction' ? 'PRIVATE_QUERY_SENTINEL_DO_NOT_REFLECT' : 'bridge smoke'
|
|
364
|
+
];
|
|
365
|
+
const argv = executableOverride ? invocation : [
|
|
366
|
+
BIN,
|
|
367
|
+
...invocation
|
|
368
|
+
];
|
|
369
|
+
try {
|
|
370
|
+
const child = spawn(executable, argv, {
|
|
371
|
+
cwd,
|
|
372
|
+
env,
|
|
373
|
+
stdio: [
|
|
374
|
+
'ignore',
|
|
375
|
+
'pipe',
|
|
376
|
+
'pipe'
|
|
377
|
+
]
|
|
378
|
+
});
|
|
379
|
+
let stdout = '';
|
|
380
|
+
let stderr = '';
|
|
381
|
+
child.stdout.on('data', (c)=>{
|
|
382
|
+
stdout += c;
|
|
383
|
+
});
|
|
384
|
+
child.stderr.on('data', (c)=>{
|
|
385
|
+
stderr += c;
|
|
386
|
+
});
|
|
387
|
+
const exit = await new Promise((resolve)=>{
|
|
388
|
+
const timer = setTimeout(()=>{
|
|
389
|
+
child.kill('SIGKILL');
|
|
390
|
+
resolve(null);
|
|
391
|
+
}, TIMEOUT_MS);
|
|
392
|
+
child.once('close', (code)=>{
|
|
393
|
+
clearTimeout(timer);
|
|
394
|
+
resolve(code);
|
|
395
|
+
});
|
|
396
|
+
});
|
|
397
|
+
const optionalResolutionHits = readFileSync(probeResult, 'utf8').split('\n').filter(Boolean).reduce((hits, line)=>{
|
|
398
|
+
const event = JSON.parse(line);
|
|
399
|
+
if (event.target?.includes('dsh-calendar')) hits.calendar += 1;
|
|
400
|
+
if (event.target?.includes('dsh-map-tools')) hits.map += 1;
|
|
401
|
+
return hits;
|
|
402
|
+
}, {
|
|
403
|
+
calendar: 0,
|
|
404
|
+
map: 0
|
|
405
|
+
});
|
|
406
|
+
return {
|
|
407
|
+
exit,
|
|
408
|
+
stdout,
|
|
409
|
+
stderr,
|
|
410
|
+
output: stdout + stderr,
|
|
411
|
+
requests,
|
|
412
|
+
optionalResolutionHits
|
|
413
|
+
};
|
|
414
|
+
} finally{
|
|
415
|
+
await new Promise((resolve)=>server.close(()=>resolve()));
|
|
416
|
+
rmSync(dsh, {
|
|
417
|
+
recursive: true,
|
|
418
|
+
force: true
|
|
419
|
+
});
|
|
420
|
+
rmSync(cwd, {
|
|
421
|
+
recursive: true,
|
|
422
|
+
force: true
|
|
423
|
+
});
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
async function assertRuntimeContract(executableOverride) {
|
|
427
|
+
const target = executableOverride ? 'packaged' : 'source';
|
|
428
|
+
const binSource = readFileSync(executableOverride ?? BIN, 'utf8');
|
|
429
|
+
assert.equal(binSource.includes('headless-keepalive'), false, `${target} bin must not inject a headless keepalive preload`);
|
|
430
|
+
assert.equal(binSource.includes('--require=${headlessKeepalivePath}'), false, `${target} bin must not mutate NODE_OPTIONS with a keepalive preload`);
|
|
431
|
+
assert.equal(binSource.includes('headless-startup-hold'), false, `${target} bin must not inject a timer-based startup hold`);
|
|
432
|
+
const disabled = await runCase('disabled', executableOverride);
|
|
433
|
+
assert.equal(disabled.exit, 0, `${target} default-off child exit=${disabled.exit}; output tail=${disabled.output.slice(-2_000)}`);
|
|
434
|
+
assert.ok(disabled.requests.length > 0 && !disabled.requests.some((r)=>names(r).includes(TOOL)), `${target} default-off must reach the relay without exposing benchmark tool; exit=${disabled.exit}; requests=${disabled.requests.length}; output=${disabled.output.slice(-2_000)}`);
|
|
435
|
+
assert.ok(disabled.optionalResolutionHits.calendar > 0 || disabled.optionalResolutionHits.map > 0, `${target} default-off must retain optional plugin resolution as a counter-proof`);
|
|
436
|
+
const enabled = await runCase('enabled', executableOverride);
|
|
437
|
+
assert.equal(enabled.exit, 0, `${target} opt-in child exit=${enabled.exit}; requests=${enabled.requests.length}; tool surfaces=${JSON.stringify(enabled.requests.map(names))}; output tail=${enabled.output.slice(-10000)}`);
|
|
438
|
+
assert.ok(enabled.requests.some((r)=>names(r).includes(TOOL)), `${target} opt-in planner request must expose benchmark tool; schemas=${JSON.stringify(enabled.requests.map(names))}; output=${enabled.output.slice(-4_000)}`);
|
|
439
|
+
const enabledToolNames = [
|
|
440
|
+
...new Set(enabled.requests.flatMap(names))
|
|
441
|
+
].sort();
|
|
442
|
+
assert.deepEqual(enabledToolNames, [
|
|
443
|
+
TOOL
|
|
444
|
+
], `${target} enabled runtime must expose exactly the benchmark tool; observed tool names=${JSON.stringify(enabledToolNames)}`);
|
|
445
|
+
assert.equal(enabledToolNames.some((name)=>name.startsWith('calendar_') || name.startsWith('map_')), false, `${target} benchmark projection must not expose calendar/map tools`);
|
|
446
|
+
assert.deepEqual(enabled.optionalResolutionHits, {
|
|
447
|
+
calendar: 0,
|
|
448
|
+
map: 0
|
|
449
|
+
}, `${target} benchmark mode must not resolve optional calendar/map plugins`);
|
|
450
|
+
assert.ok(enabled.requests.some(toolResultPresent), `${target} marker must enter model history as tool result`);
|
|
451
|
+
const leakedReport = enabled.requests.map((r)=>JSON.stringify(r).match(/\\?"leaked\\?":\[(.*?)\]/)?.[1]).filter(Boolean).join('|');
|
|
452
|
+
assert.ok(enabled.requests.some((r)=>/\\?"leaked\\?":\[\]/.test(JSON.stringify(r))), `${target} tool result must report no forbidden environment names; observed names=${leakedReport || '(none)'}`);
|
|
453
|
+
assert.equal(enabled.requests.some((r)=>JSON.stringify(r).includes('do-not-leak')), false, `${target} tool result must not expose the parent secret value`);
|
|
454
|
+
const enabledPrompt = enabled.requests.map((r)=>JSON.stringify(r)).join('\n');
|
|
455
|
+
for (const variable of [
|
|
456
|
+
'{{current_date}}',
|
|
457
|
+
'{{time_anchor_card}}',
|
|
458
|
+
'{{motivation_brief}}'
|
|
459
|
+
]){
|
|
460
|
+
assert.equal(enabledPrompt.includes(variable), false, `${target} benchmark persona must not retain ${variable}`);
|
|
461
|
+
}
|
|
462
|
+
assert.equal(enabledPrompt.includes('gotry_feasibility_check'), false, `${target} benchmark persona must not retain ordinary GoTry tool instructions`);
|
|
463
|
+
assert.ok(enabled.requests.length > 0 && enabled.requests.some((request)=>{
|
|
464
|
+
const prompt = JSON.stringify(request);
|
|
465
|
+
return (prompt.match(/You are GoTry, a task-agnostic travel planning assistant\./g) ?? []).length === 1 && (prompt.match(/Use only the current conversation and tools available in this benchmark session\./g) ?? []).length === 1;
|
|
466
|
+
}), `${target} benchmark persona has each stable sentence exactly once per request`);
|
|
467
|
+
assert.match(enabled.output, /benchmark_terminal/);
|
|
468
|
+
const debugRedaction = await runCase('debug-redaction', executableOverride);
|
|
469
|
+
assert.equal(debugRedaction.exit, 0, `${target} benchmark debug mode preserves successful execution`);
|
|
470
|
+
assert.equal(debugRedaction.output.includes('PRIVATE_QUERY_SENTINEL_DO_NOT_REFLECT'), false, `${target} benchmark debug output never reflects the private task`);
|
|
471
|
+
const unexpected = await runCase('unexpected-output', executableOverride);
|
|
472
|
+
assert.equal(unexpected.exit, 1, `${target} unexpected output must fail the successful-call conformance gate; exit=${unexpected.exit}; output tail=${unexpected.output.slice(-1000)}`);
|
|
473
|
+
assert.equal(unexpected.stdout, '', `${target} rejected runner output keeps stdout empty`);
|
|
474
|
+
assert.ok(unexpected.requests.some((r)=>JSON.stringify(r).includes('forbidden_output')), `${target} positive output allowlist must reject unexpected output`);
|
|
475
|
+
assert.equal(unexpected.requests.some((r)=>JSON.stringify(r).includes('must-not-reflect')), false, `${target} positive output rejection must not reflect unexpected key/value`);
|
|
476
|
+
assert.equal(unexpected.output.includes('<benchmark_terminal>'), false, `${target} failed bridge result cannot release a terminal body`);
|
|
477
|
+
const invalidPath = await runCase('invalid-path', executableOverride);
|
|
478
|
+
assert.equal(invalidPath.exit, 1, `${target} invalid config basename child exit=${invalidPath.exit}`);
|
|
479
|
+
assert.equal(invalidPath.requests.length, 0, `${target} invalid config must fail before relay/network activity`);
|
|
480
|
+
assert.match(invalidPath.output, /benchmark environment configuration unavailable/);
|
|
481
|
+
assert.equal(invalidPath.output.includes('benchmark-env-config-'), false, `${target} invalid config error must not expose path or filename`);
|
|
482
|
+
const unsafeConfig = await runCase('unsafe-config', executableOverride);
|
|
483
|
+
assert.equal(unsafeConfig.exit, 1, `${target} group/world-writable config child exit=${unsafeConfig.exit}; requests=${unsafeConfig.requests.length}; output=${unsafeConfig.output.slice(-2_000)}`);
|
|
484
|
+
assert.equal(unsafeConfig.requests.length, 0, `${target} unsafe config must fail before relay/network activity`);
|
|
485
|
+
assert.match(unsafeConfig.output, /benchmark environment configuration unavailable/);
|
|
486
|
+
const invalidSchema = await runCase('invalid-schema', executableOverride);
|
|
487
|
+
assert.equal(invalidSchema.exit, 1, `${target} invalid config schema child exit=${invalidSchema.exit}`);
|
|
488
|
+
assert.equal(invalidSchema.requests.length, 0, `${target} invalid schema must fail before relay/network activity`);
|
|
489
|
+
assert.match(invalidSchema.output, /benchmark environment configuration unavailable/);
|
|
490
|
+
const webMode = await runCase('web-mode', executableOverride);
|
|
491
|
+
assert.equal(webMode.exit, 1, `${target} benchmark opt-in rejects web mode`);
|
|
492
|
+
assert.equal(webMode.stdout, '', `${target} web-mode rejection keeps stdout empty`);
|
|
493
|
+
assert.equal(webMode.requests.length, 0, `${target} web rejection occurs before any model request`);
|
|
494
|
+
assert.match(webMode.output, /benchmark environment requires headless mode/);
|
|
495
|
+
const truncated = await runCase('output-truncated', executableOverride);
|
|
496
|
+
assert.equal(truncated.exit, 1, `${target} output truncation fails the successful-call conformance gate`);
|
|
497
|
+
assert.ok(truncated.requests.some((r)=>JSON.stringify(r).includes('output_truncated')), `${target} real runner output over the configured cap is rejected`);
|
|
498
|
+
assert.equal(truncated.stdout, '', `${target} output truncation keeps stdout empty`);
|
|
499
|
+
assert.match(truncated.stderr, /benchmark terminal output unavailable \(child_bridge_output_truncated\)/, `${target} output truncation emits a stable bridge reason code`);
|
|
500
|
+
const timedOut = await runCase('timeout', executableOverride);
|
|
501
|
+
assert.equal(timedOut.exit, 1, `${target} timeout fails the successful-call conformance gate`);
|
|
502
|
+
assert.ok(timedOut.requests.some((r)=>JSON.stringify(r).includes('timed_out')), `${target} real runner deadline is enforced`);
|
|
503
|
+
assert.equal(timedOut.stdout, '', `${target} timeout keeps stdout empty`);
|
|
504
|
+
assert.match(timedOut.stderr, /benchmark terminal output unavailable \(child_bridge_timed_out\)/, `${target} timeout emits a stable child bridge reason code`);
|
|
505
|
+
const runnerFailed = await runCase('runner-failed', executableOverride);
|
|
506
|
+
assert.equal(runnerFailed.exit, 1, `${target} runner failure fails the successful-call conformance gate`);
|
|
507
|
+
assert.ok(runnerFailed.requests.some((r)=>JSON.stringify(r).includes('runner_failed')), `${target} nonzero runner exit is surfaced structurally`);
|
|
508
|
+
assert.equal(runnerFailed.stdout, '', `${target} runner failure keeps stdout empty`);
|
|
509
|
+
assert.match(runnerFailed.stderr, /benchmark terminal output unavailable \(child_bridge_runner_failed\)/, `${target} runner failure emits a stable child bridge reason code`);
|
|
510
|
+
assert.equal(runnerFailed.output.includes('PRIVATE_RUNNER_DIAGNOSTIC_DO_NOT_REFLECT'), false, `${target} runner stderr is never reflected`);
|
|
511
|
+
const spawnFailed = await runCase('spawn-failed', executableOverride);
|
|
512
|
+
assert.equal(spawnFailed.exit, 1, `${target} runner spawn failure fails the successful-call conformance gate`);
|
|
513
|
+
assert.ok(spawnFailed.requests.some((r)=>JSON.stringify(r).includes('spawn_failed')), `${target} runner spawn failure is surfaced structurally`);
|
|
514
|
+
assert.equal(spawnFailed.stdout, '', `${target} runner spawn failure keeps stdout empty`);
|
|
515
|
+
assert.match(spawnFailed.stderr, /benchmark terminal output unavailable \(child_bridge_spawn_failed\)/, `${target} runner spawn failure emits a stable child bridge reason code`);
|
|
516
|
+
}
|
|
517
|
+
async function assertSourceRuntimeContractWhenAvailable() {
|
|
518
|
+
const vendoredDsh = join(ROOT, 'ts', 'dsh-runtime', 'node_modules', '@deepseek-ai', 'dsh', 'lib', 'bin.js');
|
|
519
|
+
let sourceRuntimeAvailable = existsSync(vendoredDsh);
|
|
520
|
+
if (!sourceRuntimeAvailable) {
|
|
521
|
+
try {
|
|
522
|
+
createRequire(BIN).resolve('@deepseek-ai/dsh/lib/bin.js');
|
|
523
|
+
sourceRuntimeAvailable = true;
|
|
524
|
+
} catch {
|
|
525
|
+
sourceRuntimeAvailable = false;
|
|
526
|
+
}
|
|
527
|
+
}
|
|
528
|
+
if (!sourceRuntimeAvailable) {
|
|
529
|
+
const binSource = readFileSync(BIN, 'utf8');
|
|
530
|
+
assert.equal(binSource.includes('headless-keepalive'), false, 'source bin must not inject a headless keepalive preload');
|
|
531
|
+
assert.equal(binSource.includes('--require=${headlessKeepalivePath}'), false, 'source bin must not mutate NODE_OPTIONS with a keepalive preload');
|
|
532
|
+
assert.equal(binSource.includes('headless-startup-hold'), false, 'source bin must not inject a timer-based startup hold');
|
|
533
|
+
return false;
|
|
534
|
+
}
|
|
535
|
+
await assertRuntimeContract();
|
|
536
|
+
return true;
|
|
537
|
+
}
|
|
538
|
+
function installedPackageRoot(executable) {
|
|
539
|
+
const consumerRoot = dirname(dirname(dirname(executable)));
|
|
540
|
+
const packageMain = createRequire(join(consumerRoot, 'package.json')).resolve('@danceiny/gotry');
|
|
541
|
+
return dirname(dirname(dirname(packageMain)));
|
|
542
|
+
}
|
|
543
|
+
async function assertPackagedPatchProjection(executable) {
|
|
544
|
+
const sourcePackageRoot = installedPackageRoot(executable);
|
|
545
|
+
const packageScope = dirname(sourcePackageRoot);
|
|
546
|
+
const probeParent = mkdtempSync(join(packageScope, 'round4-projection-'));
|
|
547
|
+
const probePackageRoot = join(probeParent, 'gotry');
|
|
548
|
+
const probeExecutable = join(probePackageRoot, 'bin', 'gotry-inner.js');
|
|
549
|
+
const patchPath = join(probePackageRoot, 'cordis.gotry-patch.yml');
|
|
550
|
+
const inlinePoisonModule = join(probePackageRoot, 'future-inline-plugin.cjs');
|
|
551
|
+
const reorderedPoisonModule = join(probePackageRoot, 'future-reordered-plugin.cjs');
|
|
552
|
+
const inlinePoisonProof = join(probeParent, 'future-inline-loaded.txt');
|
|
553
|
+
const reorderedPoisonProof = join(probeParent, 'future-reordered-loaded.txt');
|
|
554
|
+
cpSync(sourcePackageRoot, probePackageRoot, {
|
|
555
|
+
recursive: true
|
|
556
|
+
});
|
|
557
|
+
chmodSync(probeExecutable, 0o755);
|
|
558
|
+
const basePatch = readFileSync(patchPath, 'utf8');
|
|
559
|
+
const stableError = /benchmark environment configuration unavailable/;
|
|
560
|
+
const poisonEnv = {
|
|
561
|
+
GOTRY_FUTURE_INLINE_PROOF: inlinePoisonProof,
|
|
562
|
+
GOTRY_FUTURE_REORDERED_PROOF: reorderedPoisonProof
|
|
563
|
+
};
|
|
564
|
+
const runRejectedPatch = async (label, patch, forbiddenValues = [])=>{
|
|
565
|
+
rmSync(inlinePoisonProof, {
|
|
566
|
+
force: true
|
|
567
|
+
});
|
|
568
|
+
rmSync(reorderedPoisonProof, {
|
|
569
|
+
force: true
|
|
570
|
+
});
|
|
571
|
+
writeFileSync(patchPath, patch);
|
|
572
|
+
const result = await runCase('enabled', probeExecutable, poisonEnv);
|
|
573
|
+
assert.equal(result.exit, 1, `${label} must reject the benchmark startup`);
|
|
574
|
+
assert.equal(result.requests.length, 0, `${label} must fail before relay activity`);
|
|
575
|
+
assert.deepEqual(result.optionalResolutionHits, {
|
|
576
|
+
calendar: 0,
|
|
577
|
+
map: 0
|
|
578
|
+
}, `${label} must fail before optional plugin resolution`);
|
|
579
|
+
assert.match(result.output, stableError, `${label} emits a stable generic error`);
|
|
580
|
+
assert.equal(result.output.includes(probePackageRoot), false, `${label} must not reflect the package path`);
|
|
581
|
+
assert.equal(existsSync(inlinePoisonProof) || existsSync(reorderedPoisonProof), false, `${label} must not execute a poison plugin`);
|
|
582
|
+
for (const value of forbiddenValues)assert.equal(result.output.includes(value), false, `${label} must not reflect rejected input`);
|
|
583
|
+
};
|
|
584
|
+
try {
|
|
585
|
+
writeFileSync(inlinePoisonModule, `const fs = require('node:fs'); fs.appendFileSync(process.env.GOTRY_FUTURE_INLINE_PROOF, 'loaded\\n'); exports.name = 'round4-future-inline'; exports.apply = () => {}`);
|
|
586
|
+
writeFileSync(reorderedPoisonModule, `const fs = require('node:fs'); fs.appendFileSync(process.env.GOTRY_FUTURE_REORDERED_PROOF, 'loaded\\n'); exports.name = 'round4-future-reordered'; exports.apply = () => {}`);
|
|
587
|
+
const futureEntries = [
|
|
588
|
+
` - { id: dsh-future-inline, name: '${inlinePoisonModule}' }`,
|
|
589
|
+
` - name: '${reorderedPoisonModule}'\n id: dsh-future-reordered`
|
|
590
|
+
].join('\n');
|
|
591
|
+
const futurePatch = basePatch.replace(" - id: dsh-map-tools", `${futureEntries}\n - id: dsh-map-tools`);
|
|
592
|
+
assert.notEqual(futurePatch, basePatch, 'future-plugin fixture must enter the insert sequence');
|
|
593
|
+
writeFileSync(patchPath, futurePatch);
|
|
594
|
+
const ordinary = await runCase('disabled', probeExecutable, poisonEnv);
|
|
595
|
+
assert.ok(existsSync(inlinePoisonProof), `default-off must execute the inline future-plugin top level; exit=${ordinary.exit}; output=${ordinary.output.slice(-2_000)}`);
|
|
596
|
+
assert.ok(existsSync(reorderedPoisonProof), `default-off must execute the reordered future-plugin top level; exit=${ordinary.exit}; output=${ordinary.output.slice(-2_000)}`);
|
|
597
|
+
rmSync(inlinePoisonProof, {
|
|
598
|
+
force: true
|
|
599
|
+
});
|
|
600
|
+
rmSync(reorderedPoisonProof, {
|
|
601
|
+
force: true
|
|
602
|
+
});
|
|
603
|
+
const benchmark = await runCase('enabled', probeExecutable, poisonEnv);
|
|
604
|
+
assert.equal(benchmark.exit, 0, `benchmark future-plugin projection exits 0; output=${benchmark.output.slice(-2_000)}`);
|
|
605
|
+
assert.ok(benchmark.requests.some((request)=>names(request).includes(TOOL)), 'benchmark future-plugin projection reaches the bridge relay');
|
|
606
|
+
assert.equal(existsSync(inlinePoisonProof) || existsSync(reorderedPoisonProof), false, 'benchmark projection must not execute inline or reordered future plugins');
|
|
607
|
+
assert.deepEqual(benchmark.optionalResolutionHits, {
|
|
608
|
+
calendar: 0,
|
|
609
|
+
map: 0
|
|
610
|
+
}, 'benchmark future-plugin projection does not resolve optional host plugins');
|
|
611
|
+
await runRejectedPatch('missing gotry-tools', basePatch.replace(' - id: gotry-tools', ' - id: gotry-tools-missing'));
|
|
612
|
+
await runRejectedPatch('duplicate gotry-tools', basePatch.replace(' - id: dsh-map-tools', " - id: gotry-tools\n name: 'duplicate/gotry-tools'\n - id: dsh-map-tools"));
|
|
613
|
+
await runRejectedPatch('second insert block', `${basePatch}\n- insert:\n - id: dsh-second-insert\n name: '${inlinePoisonModule}'\n`, [
|
|
614
|
+
inlinePoisonModule
|
|
615
|
+
]);
|
|
616
|
+
await runRejectedPatch('flow second insert block', `${basePatch}\n- insert: [{ id: dsh-flow-second-insert, name: '${inlinePoisonModule}' }]\n`, [
|
|
617
|
+
inlinePoisonModule
|
|
618
|
+
]);
|
|
619
|
+
await runRejectedPatch('spoofed gotry-tools name', basePatch.replace("name: 'placeholder/ts/src/index.ts'", `name: '${inlinePoisonModule}'`), [
|
|
620
|
+
inlinePoisonModule
|
|
621
|
+
]);
|
|
622
|
+
await runRejectedPatch('spoofed gotry-tools name with decoy anchor', `${basePatch.replace("name: 'placeholder/ts/src/index.ts'", `name: '${inlinePoisonModule}'`)}\n- id: benchmark-name-decoy\n name: 'placeholder/ts/src/index.ts'\n`, [
|
|
623
|
+
inlinePoisonModule
|
|
624
|
+
]);
|
|
625
|
+
await runRejectedPatch('spoofed gotry-tools name with nested decoy anchor', basePatch.replace("name: 'placeholder/ts/src/index.ts'", `name: '${inlinePoisonModule}'`).replace(" stateRoot: '.'", " name: 'placeholder/ts/src/index.ts'\n stateRoot: '.'"), [
|
|
626
|
+
inlinePoisonModule
|
|
627
|
+
]);
|
|
628
|
+
await runRejectedPatch('missing benchmark config anchor', basePatch.replace(/^\s*hbcliBin:.*\n/m, ''));
|
|
629
|
+
await runRejectedPatch('missing benchmark config anchor with decoy', `${basePatch.replace(/^\s*hbcliBin:.*\n/m, '')}\n- id: benchmark-anchor-decoy\n hbcliBin: 'hbcli'\n`);
|
|
630
|
+
await runRejectedPatch('missing benchmark config anchor with nested decoy', basePatch.replace(/^\s*hbcliBin:.*\n/m, '').replace(" stateRoot: '.'", " nestedAnchorDecoy:\n hbcliBin: 'hbcli'\n stateRoot: '.'"));
|
|
631
|
+
await runRejectedPatch('duplicate benchmark config anchor', basePatch.replace(" hbcliBin: 'hbcli'", " hbcliBin: 'hbcli'\n hbcliBin: 'hbcli'"));
|
|
632
|
+
await runRejectedPatch('pre-existing benchmark config path', basePatch.replace(" hbcliBin: 'hbcli'", " hbcliBin: 'hbcli'\n benchmarkEnvironmentConfigPath: '/not/used'"), [
|
|
633
|
+
'/not/used'
|
|
634
|
+
]);
|
|
635
|
+
await runRejectedPatch('missing system-prompt anchor', basePatch.replace(/^- id: system-prompt[\s\S]*$/m, ''));
|
|
636
|
+
await runRejectedPatch('duplicate system-prompt anchor', `${basePatch}\n- id: system-prompt\n config:\n persona: >-\n duplicate\n`);
|
|
637
|
+
await runRejectedPatch('quoted system-prompt duplicate', `${basePatch}\n- id: 'system-prompt'\n config:\n persona: >-\n quoted duplicate\n`);
|
|
638
|
+
const systemPromptMutationSentinel = 'ROUND7_SYSTEM_PROMPT_MUTATION_SENTINEL_DO_NOT_REFLECT';
|
|
639
|
+
await runRejectedPatch('quoted mapping-key system-prompt duplicate', `${basePatch}\n- "id": system-prompt\n config:\n persona: >-\n ${systemPromptMutationSentinel}\n`, [
|
|
640
|
+
systemPromptMutationSentinel
|
|
641
|
+
]);
|
|
642
|
+
await runRejectedPatch('flow quoted-key system-prompt duplicate', `${basePatch}\n- { "id": system-prompt, config: { persona: ${systemPromptMutationSentinel} } }\n`, [
|
|
643
|
+
systemPromptMutationSentinel
|
|
644
|
+
]);
|
|
645
|
+
await runRejectedPatch('reordered system-prompt duplicate', `${basePatch}\n- name: reordered-system-prompt\n id: system-prompt\n config:\n persona: >-\n reordered duplicate\n`);
|
|
646
|
+
await runRejectedPatch('flow system-prompt duplicate', `${basePatch}\n- { id: system-prompt, config: { persona: flow duplicate } }\n`);
|
|
647
|
+
await runRejectedPatch('noncanonical insert id root item', `${basePatch}\n- id: insert\n config:\n persona: >-\n ${systemPromptMutationSentinel}\n`, [
|
|
648
|
+
systemPromptMutationSentinel
|
|
649
|
+
]);
|
|
650
|
+
await runRejectedPatch('malformed system-prompt persona', basePatch.replace(/^ persona: >-$/m, ' persona: plain'));
|
|
651
|
+
} finally{
|
|
652
|
+
rmSync(probeParent, {
|
|
653
|
+
recursive: true,
|
|
654
|
+
force: true
|
|
655
|
+
});
|
|
656
|
+
}
|
|
657
|
+
}
|
|
658
|
+
const LARGE_TERMINAL_PAYLOAD = 'x'.repeat(80 * 1024);
|
|
659
|
+
function taggedTerminal(valid) {
|
|
660
|
+
return `<benchmark_terminal>${valid ? '{"status":"succeeded"}' : '{"status":'}</benchmark_terminal>`;
|
|
661
|
+
}
|
|
662
|
+
function conformanceResponse(mode, request, plannerCount) {
|
|
663
|
+
const hasToolResult = (request.messages ?? []).some((message)=>message.role === 'tool');
|
|
664
|
+
if (mode === 'a' && plannerCount === 1 && !hasToolResult) return finalText('assistant prose without a call');
|
|
665
|
+
const call = mode === 'a' && plannerCount === 2 || [
|
|
666
|
+
'b',
|
|
667
|
+
'd',
|
|
668
|
+
'f',
|
|
669
|
+
'large'
|
|
670
|
+
].includes(mode) && plannerCount === 1;
|
|
671
|
+
if (call && !hasToolResult) return toolCall();
|
|
672
|
+
if (mode === 'f' && plannerCount === 3) return toolCall('bridge-call-retry');
|
|
673
|
+
if (mode === 'large' && hasToolResult) {
|
|
674
|
+
return finalText(`<benchmark_terminal>${JSON.stringify({
|
|
675
|
+
payload: LARGE_TERMINAL_PAYLOAD
|
|
676
|
+
})}</benchmark_terminal>`);
|
|
677
|
+
}
|
|
678
|
+
const valid = mode === 'a' ? hasToolResult : mode === 'b' ? plannerCount >= 3 : mode === 'e' ? true : false;
|
|
679
|
+
return finalText(mode === 'c' || mode === 'd' ? 'bad benchmark body' : taggedTerminal(valid));
|
|
680
|
+
}
|
|
681
|
+
async function runConformanceCase(mode, executableOverride) {
|
|
682
|
+
const requests = [];
|
|
683
|
+
let plannerCount = 0;
|
|
684
|
+
let recoveredAttempts = 0;
|
|
685
|
+
let servedToolCalls = 0;
|
|
686
|
+
const server = createServer((req, res)=>{
|
|
687
|
+
const chunks = [];
|
|
688
|
+
req.on('data', (chunk)=>chunks.push(Buffer.from(chunk)));
|
|
689
|
+
req.on('end', ()=>{
|
|
690
|
+
let body = {};
|
|
691
|
+
try {
|
|
692
|
+
body = JSON.parse(Buffer.concat(chunks).toString());
|
|
693
|
+
} catch {}
|
|
694
|
+
requests.push(body);
|
|
695
|
+
const requestHasToolResult = anyToolResultPresent(body);
|
|
696
|
+
const plannerRequest = names(body).includes(TOOL);
|
|
697
|
+
if (mode === 'exhausted' || mode === 'unknown' || mode === 'post-failure' && requestHasToolResult) {
|
|
698
|
+
res.writeHead(mode === 'exhausted' ? 429 : mode === 'unknown' ? 418 : 500, {
|
|
699
|
+
'content-type': 'application/json'
|
|
700
|
+
});
|
|
701
|
+
res.end(JSON.stringify({
|
|
702
|
+
error: {
|
|
703
|
+
message: 'PRIVATE_SENTINEL_DO_NOT_REFLECT',
|
|
704
|
+
type: 'server_error'
|
|
705
|
+
}
|
|
706
|
+
}));
|
|
707
|
+
return;
|
|
708
|
+
}
|
|
709
|
+
if (mode === 'recovered' && plannerRequest && !requestHasToolResult && recoveredAttempts++ === 0) {
|
|
710
|
+
res.writeHead(503, {
|
|
711
|
+
'content-type': 'application/json'
|
|
712
|
+
});
|
|
713
|
+
res.end(JSON.stringify({
|
|
714
|
+
error: {
|
|
715
|
+
message: 'PRIVATE_SENTINEL_DO_NOT_REFLECT',
|
|
716
|
+
type: 'server_error'
|
|
717
|
+
}
|
|
718
|
+
}));
|
|
719
|
+
return;
|
|
720
|
+
}
|
|
721
|
+
if (plannerRequest) plannerCount += 1;
|
|
722
|
+
res.writeHead(200, {
|
|
723
|
+
'content-type': 'text/event-stream'
|
|
724
|
+
});
|
|
725
|
+
const response = plannerRequest ? conformanceResponse(mode === 'recovered' ? 'b' : mode === 'post-failure' ? 'f' : mode, body, plannerCount) : finalText('auxiliary request');
|
|
726
|
+
if (response.includes(`"name":"${TOOL}"`)) servedToolCalls += 1;
|
|
727
|
+
res.end(response);
|
|
728
|
+
});
|
|
729
|
+
});
|
|
730
|
+
await new Promise((resolve)=>server.listen(0, '127.0.0.1', resolve));
|
|
731
|
+
const port = server.address().port;
|
|
732
|
+
const cwd = mkdtempSync(join(tmpdir(), 'gotry-conformance-cwd-'));
|
|
733
|
+
const dsh = mkdtempSync(join(tmpdir(), 'gotry-conformance-dsh-'));
|
|
734
|
+
const runner = join(cwd, 'synthetic-runner.js');
|
|
735
|
+
const runnerCount = join(cwd, 'runner-count.txt');
|
|
736
|
+
const configPath = join(cwd, 'benchmark-env-config.json');
|
|
737
|
+
writeFileSync(runner, `const fs = require('node:fs'); const path = ${JSON.stringify(runnerCount)}; const count = fs.existsSync(path) ? Number(fs.readFileSync(path, 'utf8')) : 0; fs.writeFileSync(path, String(count + 1)); process.stdout.write(JSON.stringify({ result: { marker: '${MARKER}' } }))`);
|
|
738
|
+
writeFileSync(configPath, JSON.stringify({
|
|
739
|
+
schema_version: 'gotry_benchmark_environment_bridge_v2',
|
|
740
|
+
enabled: true,
|
|
741
|
+
executable: process.execPath,
|
|
742
|
+
cwd,
|
|
743
|
+
argv_prefix: [
|
|
744
|
+
runner
|
|
745
|
+
],
|
|
746
|
+
allowed_tools: [
|
|
747
|
+
'lookup'
|
|
748
|
+
],
|
|
749
|
+
allowed_output_keys: {
|
|
750
|
+
lookup: [
|
|
751
|
+
'marker'
|
|
752
|
+
]
|
|
753
|
+
},
|
|
754
|
+
timeout_ms: 2_000,
|
|
755
|
+
max_output_bytes: 4_096,
|
|
756
|
+
terminal_output: {
|
|
757
|
+
tag: 'benchmark_terminal',
|
|
758
|
+
max_bytes: mode === 'large' ? 128 * 1024 : 4_096
|
|
759
|
+
},
|
|
760
|
+
isolation: {
|
|
761
|
+
mode: 'host-enforced',
|
|
762
|
+
writes: 'forbidden',
|
|
763
|
+
network: 'denied'
|
|
764
|
+
}
|
|
765
|
+
}));
|
|
766
|
+
const env = {
|
|
767
|
+
...process.env,
|
|
768
|
+
DSH_TOOLS_MODE: 'both',
|
|
769
|
+
DSH_HOME: dsh,
|
|
770
|
+
LLM_API_KEY: 'synthetic-conformance-key',
|
|
771
|
+
LLM_BASE_URL: `http://127.0.0.1:${port}/v1`,
|
|
772
|
+
LLM_MODEL: 'synthetic-conformance-model',
|
|
773
|
+
GOTRY_BENCHMARK_ENV_CONFIG: configPath,
|
|
774
|
+
DEEPSEEK_API_KEY: 'synthetic-conformance-key',
|
|
775
|
+
DEEPSEEK_BASE_URL: `http://127.0.0.1:${port}/v1`,
|
|
776
|
+
NODE_OPTIONS: [
|
|
777
|
+
process.env.NODE_OPTIONS,
|
|
778
|
+
...!executableOverride ? [
|
|
779
|
+
`--import=${TSX_LOADER}`
|
|
780
|
+
] : []
|
|
781
|
+
].filter(Boolean).join(' ')
|
|
782
|
+
};
|
|
783
|
+
for (const key of [
|
|
784
|
+
'GOTRY_LLM_MODEL',
|
|
785
|
+
'HTTP_PROXY',
|
|
786
|
+
'HTTPS_PROXY',
|
|
787
|
+
'ALL_PROXY',
|
|
788
|
+
'http_proxy',
|
|
789
|
+
'https_proxy',
|
|
790
|
+
'all_proxy'
|
|
791
|
+
])delete env[key];
|
|
792
|
+
env.NO_PROXY = '127.0.0.1,localhost';
|
|
793
|
+
let stdout = '';
|
|
794
|
+
let stderr = '';
|
|
795
|
+
try {
|
|
796
|
+
const executable = executableOverride || process.execPath;
|
|
797
|
+
const argv = executableOverride ? [
|
|
798
|
+
'conformance smoke'
|
|
799
|
+
] : [
|
|
800
|
+
BIN,
|
|
801
|
+
'conformance smoke'
|
|
802
|
+
];
|
|
803
|
+
const child = spawn(executable, argv, {
|
|
804
|
+
cwd,
|
|
805
|
+
env,
|
|
806
|
+
stdio: [
|
|
807
|
+
'ignore',
|
|
808
|
+
'pipe',
|
|
809
|
+
'pipe'
|
|
810
|
+
]
|
|
811
|
+
});
|
|
812
|
+
child.stdout.on('data', (chunk)=>{
|
|
813
|
+
stdout += chunk.toString();
|
|
814
|
+
});
|
|
815
|
+
child.stderr.on('data', (chunk)=>{
|
|
816
|
+
stderr += chunk.toString();
|
|
817
|
+
});
|
|
818
|
+
const exit = await new Promise((resolve)=>{
|
|
819
|
+
const timer = setTimeout(()=>{
|
|
820
|
+
child.kill('SIGKILL');
|
|
821
|
+
resolve(null);
|
|
822
|
+
}, TIMEOUT_MS);
|
|
823
|
+
child.once('close', (code)=>{
|
|
824
|
+
clearTimeout(timer);
|
|
825
|
+
resolve(code);
|
|
826
|
+
});
|
|
827
|
+
});
|
|
828
|
+
const runnerInvocations = existsSync(runnerCount) ? Number(readFileSync(runnerCount, 'utf8')) : 0;
|
|
829
|
+
return {
|
|
830
|
+
exit,
|
|
831
|
+
stdout,
|
|
832
|
+
stderr,
|
|
833
|
+
requests,
|
|
834
|
+
servedToolCalls,
|
|
835
|
+
runnerInvocations
|
|
836
|
+
};
|
|
837
|
+
} finally{
|
|
838
|
+
await new Promise((resolve)=>server.close(()=>resolve()));
|
|
839
|
+
rmSync(dsh, {
|
|
840
|
+
recursive: true,
|
|
841
|
+
force: true
|
|
842
|
+
});
|
|
843
|
+
rmSync(cwd, {
|
|
844
|
+
recursive: true,
|
|
845
|
+
force: true
|
|
846
|
+
});
|
|
847
|
+
}
|
|
848
|
+
}
|
|
849
|
+
async function assertTerminalDiagnostics(executableOverride) {
|
|
850
|
+
const terminalReasons = (stderr)=>[
|
|
851
|
+
...stderr.matchAll(/benchmark terminal output unavailable \(([^)]+)\)/g)
|
|
852
|
+
].map((match)=>match[1]);
|
|
853
|
+
const exhausted = await runConformanceCase('exhausted', executableOverride);
|
|
854
|
+
assert.notEqual(exhausted.exit, 0, 'exhausted transient model failure exits non-zero');
|
|
855
|
+
assert.match(exhausted.stderr, /child_model_capacity/, 'exhausted transient model failure emits coarse capacity enum');
|
|
856
|
+
assert.ok(exhausted.requests.length > 1, 'exhausted case actually exercises retry attempts');
|
|
857
|
+
assert.equal(exhausted.stdout, '', 'exhausted transient model failure releases no terminal stdout');
|
|
858
|
+
assert.equal(exhausted.stderr.includes('PRIVATE_SENTINEL_DO_NOT_REFLECT'), false, 'exhausted error body is never reflected');
|
|
859
|
+
assert.deepEqual(terminalReasons(exhausted.stderr), [
|
|
860
|
+
'child_model_capacity'
|
|
861
|
+
], 'exhausted emits exactly one terminal reason');
|
|
862
|
+
const recovered = await runConformanceCase('recovered', executableOverride);
|
|
863
|
+
assert.equal(recovered.exit, 0, 'transient model failure followed by valid terminal recovers');
|
|
864
|
+
assert.match(recovered.stdout, /<benchmark_terminal>/, 'recovered run releases terminal stdout');
|
|
865
|
+
assert.equal(recovered.stderr.includes('child_model_'), false, 'recovered run emits no failure enum');
|
|
866
|
+
assert.deepEqual(terminalReasons(recovered.stderr), [], 'recovered emits no terminal reason');
|
|
867
|
+
const postFailure = await runConformanceCase('post-failure', executableOverride);
|
|
868
|
+
assert.notEqual(postFailure.exit, 0, 'model failure after successful bridge exits non-zero');
|
|
869
|
+
assert.equal(postFailure.runnerInvocations, 1, 'post-bridge failure follows exactly one successful bridge invocation');
|
|
870
|
+
assert.match(postFailure.stderr, /child_model_server/, 'post-bridge model failure emits server enum');
|
|
871
|
+
assert.equal(postFailure.stdout, '', 'post-bridge model failure releases no terminal stdout');
|
|
872
|
+
assert.deepEqual(terminalReasons(postFailure.stderr), [
|
|
873
|
+
'child_model_server'
|
|
874
|
+
], 'post-bridge emits exactly one terminal reason');
|
|
875
|
+
const unknown = await runConformanceCase('unknown', executableOverride);
|
|
876
|
+
assert.notEqual(unknown.exit, 0, 'unknown model failure exits non-zero');
|
|
877
|
+
assert.match(unknown.stderr, /child_runtime_error/, 'unknown model failure collapses to generic runtime enum');
|
|
878
|
+
assert.equal(unknown.stdout, '', 'unknown model failure releases no terminal stdout');
|
|
879
|
+
assert.equal(unknown.stderr.includes('PRIVATE_SENTINEL_DO_NOT_REFLECT'), false, 'unknown error body is never reflected');
|
|
880
|
+
assert.deepEqual(terminalReasons(unknown.stderr), [
|
|
881
|
+
'child_runtime_error'
|
|
882
|
+
], 'unknown emits exactly one terminal reason');
|
|
883
|
+
const precedence = await runConformanceCase('f', executableOverride);
|
|
884
|
+
assert.match(precedence.stderr, /child_conformance_failure/, 'conformance-specific failure remains higher precedence than final generic error');
|
|
885
|
+
assert.equal(precedence.stderr.includes('child_runtime_error'), false, 'generic terminal classification does not double-write');
|
|
886
|
+
assert.deepEqual(terminalReasons(precedence.stderr), [
|
|
887
|
+
'child_conformance_failure'
|
|
888
|
+
], 'precedence emits exactly one terminal reason');
|
|
889
|
+
}
|
|
890
|
+
async function assertOutputConformance(executableOverride) {
|
|
891
|
+
const a = await runConformanceCase('a', executableOverride);
|
|
892
|
+
assert.equal(a.exit, 0, 'A prose/no-call correction then one bridge call and valid terminal exits 0');
|
|
893
|
+
assert.equal(a.servedToolCalls, 1, 'A exposes exactly one bridge call');
|
|
894
|
+
assert.equal(a.runnerInvocations, 1, 'A executes the bridge subprocess exactly once');
|
|
895
|
+
assert.ok(a.stdout.includes('<benchmark_terminal>'), 'A forwards only tagged terminal output');
|
|
896
|
+
const b = await runConformanceCase('b', executableOverride);
|
|
897
|
+
assert.equal(b.exit, 0, 'B malformed terminal correction then valid terminal exits 0');
|
|
898
|
+
assert.equal(b.servedToolCalls, 1, `B exposes exactly one bridge call; request shapes=${JSON.stringify(b.requests.map((request)=>({
|
|
899
|
+
tools: names(request),
|
|
900
|
+
roles: (request.messages ?? []).map((message)=>message.role)
|
|
901
|
+
})))}`);
|
|
902
|
+
assert.equal(b.runnerInvocations, 1, 'B format-only correction does not rerun the bridge subprocess');
|
|
903
|
+
for (const mode of [
|
|
904
|
+
'c',
|
|
905
|
+
'd'
|
|
906
|
+
]){
|
|
907
|
+
const result = await runConformanceCase(mode, executableOverride);
|
|
908
|
+
assert.notEqual(result.exit, 0, `${mode.toUpperCase()} repeated invalid output is non-zero`);
|
|
909
|
+
assert.match(result.stderr, /benchmark terminal output unavailable \(child_conformance_failure\)/, `${mode.toUpperCase()} emits a stable conformance reason code`);
|
|
910
|
+
assert.equal(result.stdout.includes('bad benchmark body'), false, `${mode.toUpperCase()} does not forward invalid body to stdout`);
|
|
911
|
+
assert.equal(result.stderr.includes('bad benchmark body'), false, `${mode.toUpperCase()} stable diagnostics do not reflect invalid body`);
|
|
912
|
+
assert.equal(result.runnerInvocations, mode === 'c' ? 0 : 1, `${mode.toUpperCase()} subprocess count matches the accepted call history`);
|
|
913
|
+
}
|
|
914
|
+
const e = await runConformanceCase('e', executableOverride);
|
|
915
|
+
assert.notEqual(e.exit, 0, 'E valid tagged terminal without bridge call is rejected');
|
|
916
|
+
assert.match(e.stderr, /benchmark terminal output unavailable \(child_conformance_failure\)/, 'E emits a stable conformance reason code');
|
|
917
|
+
assert.equal(e.stdout.includes('<benchmark_terminal>'), false, 'E does not forward terminal without call');
|
|
918
|
+
assert.equal(e.runnerInvocations, 0, 'E never executes the bridge subprocess');
|
|
919
|
+
const f = await runConformanceCase('f', executableOverride);
|
|
920
|
+
assert.notEqual(f.exit, 0, 'F format correction that tries another bridge call is rejected');
|
|
921
|
+
assert.match(f.stderr, /benchmark terminal output unavailable \(child_conformance_failure\)/, 'F emits a stable conformance reason code');
|
|
922
|
+
assert.equal(f.servedToolCalls, 2, 'F model attempts a second native call');
|
|
923
|
+
assert.equal(f.runnerInvocations, 1, 'F conformance guard blocks the second subprocess dispatch');
|
|
924
|
+
assert.equal(f.stdout.includes('<benchmark_terminal>'), false, 'F does not release a terminal body after retry redispatch');
|
|
925
|
+
const large = await runConformanceCase('large', executableOverride);
|
|
926
|
+
assert.equal(large.exit, 0, 'large valid terminal flushes before successful process close');
|
|
927
|
+
assert.equal(large.runnerInvocations, 1, 'large terminal still executes the bridge once');
|
|
928
|
+
assert.ok(Buffer.byteLength(large.stdout, 'utf8') > 64 * 1024, 'large terminal exceeds the ordinary pipe buffer');
|
|
929
|
+
assert.match(large.stdout, /<\/benchmark_terminal>\s*$/);
|
|
930
|
+
}
|
|
931
|
+
const packaged = process.env.GOTRY_BRIDGE_E2E_BIN;
|
|
932
|
+
assertRuntimeSelectionAndVersionGuards();
|
|
933
|
+
const sourceRuntimeChecked = await assertSourceRuntimeContractWhenAvailable();
|
|
934
|
+
if (sourceRuntimeChecked) await assertOutputConformance();
|
|
935
|
+
if (sourceRuntimeChecked) await assertTerminalDiagnostics();
|
|
936
|
+
if (packaged) {
|
|
937
|
+
await assertRuntimeContract(packaged);
|
|
938
|
+
await assertOutputConformance(packaged);
|
|
939
|
+
await assertTerminalDiagnostics(packaged);
|
|
940
|
+
await assertPackagedPatchProjection(packaged);
|
|
941
|
+
const packageRoot = installedPackageRoot(packaged);
|
|
942
|
+
const packagedBridge = await import(pathToFileURL(join(packageRoot, 'dist', 'src', 'benchmark-environment-bridge.js')).href);
|
|
943
|
+
const missingServiceRoot = mkdtempSync(join(tmpdir(), 'gotry-bridge-missing-service-'));
|
|
944
|
+
try {
|
|
945
|
+
const configPath = join(missingServiceRoot, 'bridge.json');
|
|
946
|
+
writeFileSync(configPath, JSON.stringify({
|
|
947
|
+
schema_version: 'gotry_benchmark_environment_bridge_v2',
|
|
948
|
+
enabled: true,
|
|
949
|
+
executable: process.execPath,
|
|
950
|
+
cwd: missingServiceRoot,
|
|
951
|
+
argv_prefix: [
|
|
952
|
+
'-e',
|
|
953
|
+
'process.exit(0)'
|
|
954
|
+
],
|
|
955
|
+
allowed_tools: [
|
|
956
|
+
'lookup'
|
|
957
|
+
],
|
|
958
|
+
timeout_ms: 100,
|
|
959
|
+
max_output_bytes: 4_096,
|
|
960
|
+
terminal_output: {
|
|
961
|
+
tag: 'benchmark_terminal',
|
|
962
|
+
max_bytes: 4_096
|
|
963
|
+
},
|
|
964
|
+
isolation: {
|
|
965
|
+
mode: 'host-enforced',
|
|
966
|
+
writes: 'forbidden',
|
|
967
|
+
network: 'denied'
|
|
968
|
+
}
|
|
969
|
+
}));
|
|
970
|
+
assert.throws(()=>packagedBridge.registerBenchmarkEnvironmentBridge(configPath, ()=>{}, undefined), /benchmark environment bridge subprocess unavailable/, 'packaged explicit opt-in fails hard when no active subprocess provider exists');
|
|
971
|
+
} finally{
|
|
972
|
+
rmSync(missingServiceRoot, {
|
|
973
|
+
recursive: true,
|
|
974
|
+
force: true
|
|
975
|
+
});
|
|
976
|
+
}
|
|
977
|
+
}
|
|
978
|
+
console.log(`benchmark environment bridge E2E: OK (${packaged ? sourceRuntimeChecked ? 'source + packaged' : 'source-static + packaged' : sourceRuntimeChecked ? 'source' : 'source-static'})`);
|
|
979
|
+
|
|
980
|
+
|
|
981
|
+
//# sourceURL=ts/scripts/benchmark-environment-bridge-e2e.ts
|