@basein/runner 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +276 -0
- package/dist/auth/client.d.ts +85 -0
- package/dist/auth/client.js +284 -0
- package/dist/bin/bir-hooks.d.ts +48 -0
- package/dist/bin/bir-hooks.js +201 -0
- package/dist/bin/bir-proxy.d.ts +45 -0
- package/dist/bin/bir-proxy.js +207 -0
- package/dist/bin/bir-scenario.d.ts +24 -0
- package/dist/bin/bir-scenario.js +177 -0
- package/dist/bin/bir.d.ts +21 -0
- package/dist/bin/bir.js +876 -0
- package/dist/config/adapters/claude-code.d.ts +76 -0
- package/dist/config/adapters/claude-code.js +181 -0
- package/dist/config/adapters/generic.d.ts +17 -0
- package/dist/config/adapters/generic.js +36 -0
- package/dist/config/generate.d.ts +127 -0
- package/dist/config/generate.js +114 -0
- package/dist/config/resolve.d.ts +68 -0
- package/dist/config/resolve.js +132 -0
- package/dist/control/client.d.ts +56 -0
- package/dist/control/client.js +86 -0
- package/dist/control/correlation.d.ts +86 -0
- package/dist/control/correlation.js +0 -0
- package/dist/control/discovery.d.ts +50 -0
- package/dist/control/discovery.js +123 -0
- package/dist/control/ordering.d.ts +38 -0
- package/dist/control/ordering.js +44 -0
- package/dist/control/paths.d.ts +32 -0
- package/dist/control/paths.js +56 -0
- package/dist/control/server.d.ts +272 -0
- package/dist/control/server.js +1131 -0
- package/dist/control/transcript.d.ts +75 -0
- package/dist/control/transcript.js +241 -0
- package/dist/index.d.ts +37 -0
- package/dist/index.js +32 -0
- package/dist/jsonrpc/framing.d.ts +49 -0
- package/dist/jsonrpc/framing.js +143 -0
- package/dist/jsonrpc/types.d.ts +52 -0
- package/dist/jsonrpc/types.js +46 -0
- package/dist/proxy/intercept.d.ts +55 -0
- package/dist/proxy/intercept.js +147 -0
- package/dist/proxy/relay.d.ts +97 -0
- package/dist/proxy/relay.js +166 -0
- package/dist/proxy/session.d.ts +116 -0
- package/dist/proxy/session.js +319 -0
- package/dist/record/housekeeping.d.ts +34 -0
- package/dist/record/housekeeping.js +39 -0
- package/dist/record/queue.d.ts +48 -0
- package/dist/record/queue.js +96 -0
- package/dist/record/recorder.d.ts +111 -0
- package/dist/record/recorder.js +39 -0
- package/dist/record/redact.d.ts +37 -0
- package/dist/record/redact.js +119 -0
- package/dist/record/remote-recorder.d.ts +110 -0
- package/dist/record/remote-recorder.js +301 -0
- package/dist/record/truncate.d.ts +36 -0
- package/dist/record/truncate.js +85 -0
- package/dist/replay/bundle.d.ts +36 -0
- package/dist/replay/bundle.js +89 -0
- package/dist/replay/controller.d.ts +300 -0
- package/dist/replay/controller.js +807 -0
- package/dist/replay/coverage.d.ts +41 -0
- package/dist/replay/coverage.js +56 -0
- package/dist/replay/derive.d.ts +58 -0
- package/dist/replay/derive.js +166 -0
- package/dist/replay/executor.d.ts +78 -0
- package/dist/replay/executor.js +233 -0
- package/dist/replay/logic.d.ts +31 -0
- package/dist/replay/logic.js +50 -0
- package/dist/replay/plan.d.ts +181 -0
- package/dist/replay/plan.js +397 -0
- package/dist/replay/pricing.d.ts +41 -0
- package/dist/replay/pricing.js +76 -0
- package/dist/replay/source-run.d.ts +50 -0
- package/dist/replay/source-run.js +98 -0
- package/dist/replay/tool-error.d.ts +22 -0
- package/dist/replay/tool-error.js +60 -0
- package/dist/replay/types.d.ts +116 -0
- package/dist/replay/types.js +35 -0
- package/dist/upstream/client.d.ts +78 -0
- package/dist/upstream/client.js +114 -0
- package/dist/upstream/http-client.d.ts +78 -0
- package/dist/upstream/http-client.js +261 -0
- package/dist/upstream/lazy-client.d.ts +31 -0
- package/dist/upstream/lazy-client.js +53 -0
- package/dist/upstream/stdio-client.d.ts +57 -0
- package/dist/upstream/stdio-client.js +203 -0
- package/dist/util/log.d.ts +27 -0
- package/dist/util/log.js +51 -0
- package/dist/util/version.d.ts +2 -0
- package/dist/util/version.js +40 -0
- package/docs/BaseInstRunner.md +621 -0
- package/docs/calculatedReplay.md +1185 -0
- package/docs/calculatedReplayGuide.md +448 -0
- package/docs/installRun.md +413 -0
- package/docs/mcpmark.md +752 -0
- package/docs/quickstart.md +201 -0
- package/docs/t-bench.md +394 -0
- package/package.json +56 -0
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# Setting up a runner machine
|
|
2
|
+
|
|
3
|
+
This guide takes one computer from nothing to recording. It assumes no prior
|
|
4
|
+
knowledge of the project. It should take about ten minutes, most of which is
|
|
5
|
+
waiting for downloads.
|
|
6
|
+
|
|
7
|
+
**What you are setting up.** BaseInstRunner sits quietly between your AI coding
|
|
8
|
+
assistant and the tools it uses, watches what happens, and saves each session to
|
|
9
|
+
the BaseIn service so it can be reviewed or re-run later. It does not change what
|
|
10
|
+
the assistant does.
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## Before you start
|
|
15
|
+
|
|
16
|
+
Three things:
|
|
17
|
+
|
|
18
|
+
1. **Node 20 or newer.** Check by opening a terminal and typing `node -v`. If you
|
|
19
|
+
see something like `v20.11.0` or higher, you are fine. If you see an error or
|
|
20
|
+
a smaller number, install it from [nodejs.org](https://nodejs.org) first.
|
|
21
|
+
2. **The BaseIn service address** — a URL like `https://basein.example.com`.
|
|
22
|
+
Whoever runs the service gives you this.
|
|
23
|
+
3. **Your BaseIn email and password.** You will type these once.
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
## Step 1 — run one command
|
|
28
|
+
|
|
29
|
+
Open a terminal **in the folder where you unpacked this project's `scripts`
|
|
30
|
+
folder**, and run the line for your machine. Replace the two placeholder values
|
|
31
|
+
with your real service address and your real project folder.
|
|
32
|
+
|
|
33
|
+
### Windows
|
|
34
|
+
|
|
35
|
+
```powershell
|
|
36
|
+
.\scripts\install-runner.ps1 C:\path\to\your\project -AuthUrl https://basein.example.com
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
### macOS
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
BIR_AUTH_URL=https://basein.example.com ./scripts/install-runner.sh ~/path/to/your/project
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
> **"Running scripts is disabled on this system"** (Windows only) — Windows blocks
|
|
46
|
+
> unsigned scripts by default. Allow them for your own account with:
|
|
47
|
+
> `Set-ExecutionPolicy -Scope CurrentUser RemoteSigned`, then run the command
|
|
48
|
+
> again.
|
|
49
|
+
|
|
50
|
+
---
|
|
51
|
+
|
|
52
|
+
## Step 2 — watch what it does
|
|
53
|
+
|
|
54
|
+
The script narrates each stage. A healthy run looks like this:
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
==> Checking Node
|
|
58
|
+
node v20.11.0
|
|
59
|
+
==> Installing the runner
|
|
60
|
+
@basein/runner@0.1.0 from the registry
|
|
61
|
+
bir 0.1.0 -- all four binaries on PATH
|
|
62
|
+
==> Configuring the BaseIn service
|
|
63
|
+
BIR_AUTH_URL=https://basein.example.com (persisted for this user)
|
|
64
|
+
==> Signing in
|
|
65
|
+
Email: you@example.com
|
|
66
|
+
Password:
|
|
67
|
+
==> Wrapping MCP servers in C:\path\to\your\project
|
|
68
|
+
+ chrome-devtools -> bir-proxy (project scope, upstream: npx)
|
|
69
|
+
+ hooks -> ...\.claude\settings.json
|
|
70
|
+
|
|
71
|
+
Wrapped 1 server.
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
The password does not appear as you type it. That is normal.
|
|
75
|
+
|
|
76
|
+
If anything goes wrong the script stops immediately and says why — it checks
|
|
77
|
+
everything it can *before* changing your machine, so a failed run leaves nothing
|
|
78
|
+
half-installed.
|
|
79
|
+
|
|
80
|
+
---
|
|
81
|
+
|
|
82
|
+
## Step 3 — start recording
|
|
83
|
+
|
|
84
|
+
Two terminals, in the project folder you gave the script.
|
|
85
|
+
|
|
86
|
+
**First terminal** — leave this one running the whole time:
|
|
87
|
+
|
|
88
|
+
```
|
|
89
|
+
bir-hooks
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
**Second terminal** — your normal work:
|
|
93
|
+
|
|
94
|
+
```
|
|
95
|
+
claude
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
That is it. Everything you do in the second terminal is now recorded.
|
|
99
|
+
|
|
100
|
+
When you are finished for the day, press `Ctrl+C` in the first terminal.
|
|
101
|
+
|
|
102
|
+
---
|
|
103
|
+
|
|
104
|
+
## Step 4 — check it actually worked
|
|
105
|
+
|
|
106
|
+
With `bir-hooks` running, in another terminal:
|
|
107
|
+
|
|
108
|
+
```
|
|
109
|
+
bir doctor
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
This is the one command worth remembering. It does not read a settings file and
|
|
113
|
+
tell you what *should* happen — it asks the running system what is *actually*
|
|
114
|
+
happening, and it fails loudly when something is wrong.
|
|
115
|
+
|
|
116
|
+
To see what is set up without needing anything running:
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
bir status
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
---
|
|
123
|
+
|
|
124
|
+
## When something is wrong
|
|
125
|
+
|
|
126
|
+
### "It says nothing is recorded"
|
|
127
|
+
|
|
128
|
+
`bir status` will show `BaseIn: (BIR_AUTH_URL not set — nothing is recorded)`.
|
|
129
|
+
The service address did not stick. Close the terminal, open a new one, and check
|
|
130
|
+
`echo $env:BIR_AUTH_URL` (Windows) or `echo $BIR_AUTH_URL` (macOS). If it is
|
|
131
|
+
empty, run the install script again with the `-AuthUrl` / `BIR_AUTH_URL` value.
|
|
132
|
+
|
|
133
|
+
### "bir is not recognised as a command"
|
|
134
|
+
|
|
135
|
+
The install worked but your terminal has not noticed yet. **Close the terminal
|
|
136
|
+
and open a new one.** This fixes it almost every time.
|
|
137
|
+
|
|
138
|
+
### "It records the tools but not what I typed"
|
|
139
|
+
|
|
140
|
+
`bir-hooks` is not running, or it is running in a different folder. It has to be
|
|
141
|
+
started **in the same folder** as your session — that is how the two find each
|
|
142
|
+
other. Stop it, `cd` to the project folder, and start it again.
|
|
143
|
+
|
|
144
|
+
### "invalid JSON at ...\.mcp.json"
|
|
145
|
+
|
|
146
|
+
Something has edited that file into a shape that is no longer valid JSON —
|
|
147
|
+
usually a missing comma or a stray bracket. Open it and check. (A file saved by
|
|
148
|
+
Notepad is fine; that case is handled.)
|
|
149
|
+
|
|
150
|
+
### Anything else
|
|
151
|
+
|
|
152
|
+
Run `bir doctor` and keep the output. It says which part of the chain is broken,
|
|
153
|
+
which is most of the way to an answer.
|
|
154
|
+
|
|
155
|
+
---
|
|
156
|
+
|
|
157
|
+
## Updating to a newer version
|
|
158
|
+
|
|
159
|
+
Run the same install script again with the new version number:
|
|
160
|
+
|
|
161
|
+
```powershell
|
|
162
|
+
.\scripts\install-runner.ps1 C:\path\to\your\project -Version 0.2.0 -AuthUrl https://basein.example.com
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
BIR_VERSION=0.2.0 BIR_AUTH_URL=https://basein.example.com ./scripts/install-runner.sh ~/path/to/your/project
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
It is safe to re-run: it will not ask you to sign in again, and it will not
|
|
170
|
+
double-wrap anything.
|
|
171
|
+
|
|
172
|
+
**One thing you must do by hand: stop `bir-hooks` and start it again.** A running
|
|
173
|
+
process keeps using the old version until it is restarted. If you forget, the
|
|
174
|
+
next session logs a `version.skew` warning telling you exactly that.
|
|
175
|
+
|
|
176
|
+
---
|
|
177
|
+
|
|
178
|
+
## Undoing it
|
|
179
|
+
|
|
180
|
+
In the project folder:
|
|
181
|
+
|
|
182
|
+
```
|
|
183
|
+
bir uninstall
|
|
184
|
+
```
|
|
185
|
+
|
|
186
|
+
This puts every file it touched back exactly as it found it, byte for byte.
|
|
187
|
+
|
|
188
|
+
To remove the software entirely as well:
|
|
189
|
+
|
|
190
|
+
```
|
|
191
|
+
npm rm -g @basein/runner
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
---
|
|
195
|
+
|
|
196
|
+
## Where to go next
|
|
197
|
+
|
|
198
|
+
- [installRun.md](installRun.md) — the fleet version of this: packaging, release
|
|
199
|
+
process, offline installs, keeping `bir-hooks` alive across reboots.
|
|
200
|
+
- [../README.md](../README.md) — what is recorded, what cannot be, and every
|
|
201
|
+
configuration switch.
|
package/docs/t-bench.md
ADDED
|
@@ -0,0 +1,394 @@
|
|
|
1
|
+
# τ²-bench — running it against BaseInstRunner, with Claude Code as the agent
|
|
2
|
+
|
|
3
|
+
Companion to [benchmark.md](benchmark.md), which recommends MCPMark first and τ²-bench
|
|
4
|
+
second. This is the second half: what τ²-bench actually is, how to read a trajectory,
|
|
5
|
+
and the three structural facts that decide whether the integration measures our product
|
|
6
|
+
or measures nothing at all.
|
|
7
|
+
|
|
8
|
+
**Read §3 before writing any code.** τ²-bench executes tool calls *inside its own
|
|
9
|
+
orchestrator*. Nothing crosses MCP by default, which means a naive integration produces a
|
|
10
|
+
clean run, a plausible score, and zero recorded steps.
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## 1. Why this benchmark, in one paragraph
|
|
15
|
+
|
|
16
|
+
MCPMark proves a replayed scenario survives a **fresh instance of the same task**.
|
|
17
|
+
τ²-bench proves it survives a **different task of the same shape** — "cancel order W1234
|
|
18
|
+
for user A" against "cancel order W5678 for user B". Its domains are built out of
|
|
19
|
+
policy-constrained task families over a shared database, which is exactly the surface that
|
|
20
|
+
per-step parameter derivation exists for. Nothing else public tests the *calculated* half
|
|
21
|
+
of calculated replay this directly. It also carries the name recognition that makes a
|
|
22
|
+
result travel.
|
|
23
|
+
|
|
24
|
+
Domains: `retail` (115 tasks), `airline` (50), `telecom` (114), plus `mock` and
|
|
25
|
+
`banking_knowledge`. Python ≥3.12 <3.14, `uv sync`, LiteLLM-compatible keys in `.env`.
|
|
26
|
+
|
|
27
|
+
---
|
|
28
|
+
|
|
29
|
+
## 2. Reading a trajectory
|
|
30
|
+
|
|
31
|
+
The link that prompted this doc:
|
|
32
|
+
|
|
33
|
+
```
|
|
34
|
+
https://taubench.com/trajectory-visualizer?model=qwen3.5-397b-a17b-think_sierra_2026-03-02&domain=retail&task=0
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
It is a client-side viewer over the simulation JSON that `tau2 run` writes to
|
|
38
|
+
`data/simulations/`. Two modes — **Trajectories** and **Tasks** — and the URL pins one
|
|
39
|
+
`(model, domain, task)` triple. The same data is browsable locally with `tau2 view`, which
|
|
40
|
+
is the version to use while building the adapter, because it reads your own runs.
|
|
41
|
+
|
|
42
|
+
An episode is a four-party loop, and the visualizer shows it as one interleaved column:
|
|
43
|
+
|
|
44
|
+
```
|
|
45
|
+
Agent ──message──▶ User ──message──▶ Agent ──tool_call──▶ Environment ──result──▶ Agent
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
| Element | What it is | What we care about |
|
|
49
|
+
|---|---|---|
|
|
50
|
+
| **UserMessage** | the *simulated* customer, driven by `--user-llm` | turn 1 carries the intent; later turns supply parameters |
|
|
51
|
+
| **AssistantMessage** | the agent's text, and optionally `tool_calls` | the text is what the COMMUNICATE check reads |
|
|
52
|
+
| **ToolMessage** | the environment's result | executed by the orchestrator, not the agent (§3.1) |
|
|
53
|
+
| **evaluation_criteria.actions** | *one* reference solution, replayed on a fresh env to derive the target DB hash | **not** a required path (§3.2) |
|
|
54
|
+
| reward breakdown | DB / COMMUNICATE / ENV_ASSERTION / NL_ASSERTION | the product of whatever `reward_basis` lists |
|
|
55
|
+
|
|
56
|
+
Episodes end on a stop token, a user-simulator "task complete" transfer, or `max_steps`.
|
|
57
|
+
|
|
58
|
+
**The one habit worth forming:** when a task scores 0, open the trajectory and check
|
|
59
|
+
*which* component was 0. A DB failure and a COMMUNICATE failure look identical in the
|
|
60
|
+
aggregate and mean opposite things for us — see §3.3.
|
|
61
|
+
|
|
62
|
+
---
|
|
63
|
+
|
|
64
|
+
## 3. The three facts that decide the integration
|
|
65
|
+
|
|
66
|
+
### 3.1 The grade comes from the transcript, not from the world
|
|
67
|
+
|
|
68
|
+
*Corrected after the first working run. An earlier version of this section said the
|
|
69
|
+
binding constraint was that the orchestrator executes tool calls. That is true — the agent
|
|
70
|
+
returns an `AssistantMessage` containing `tool_calls` and the orchestrator invokes them
|
|
71
|
+
against the domain environment — but it is not the constraint that decides the design.*
|
|
72
|
+
|
|
73
|
+
The constraint that decides the design is in `evaluator_env.py`: τ² **does not grade the
|
|
74
|
+
environment the episode ran in**. It constructs a *fresh* environment, replays the
|
|
75
|
+
trajectory's `(tool_call, tool_result)` pairs onto it via `set_state`, and hashes that.
|
|
76
|
+
|
|
77
|
+
> **Consequence:** the reward is earned by what appears in the messages. An agent that
|
|
78
|
+
> executes the whole task perfectly against the live environment and reports it in prose
|
|
79
|
+
> scores **zero**, because the graded environment received nothing.
|
|
80
|
+
|
|
81
|
+
We measured exactly this. Claude Code solved retail task 0 with a call sequence matching
|
|
82
|
+
the reference trajectory *exactly* — same tools, same item ids, same payment method — and
|
|
83
|
+
scored `DB 0`, `reward 0`. Nothing in the results table said "your harness is broken"; it
|
|
84
|
+
read like a hard task.
|
|
85
|
+
|
|
86
|
+
τ²-bench also ships no MCP interface: domain tools are Python callables wrapped in `Tool`
|
|
87
|
+
objects and called in-process. So bir sees nothing unless Claude Code makes real MCP calls
|
|
88
|
+
of its own. Those two facts together — the trajectory must carry the calls, *and* Claude
|
|
89
|
+
Code must make them over MCP — are what §4 has to satisfy at once.
|
|
90
|
+
|
|
91
|
+
### 3.2 Reward is outcome-based, so replay is legitimately scoreable
|
|
92
|
+
|
|
93
|
+
`evaluation_criteria.actions` is a reference trajectory, replayed on a fresh environment to
|
|
94
|
+
establish a **target database hash**. The DB check compares hashes. Any sequence of tool
|
|
95
|
+
calls reaching an equivalent end state passes. The docs are emphatic: *actions document one
|
|
96
|
+
working solution, not the only acceptable one.*
|
|
97
|
+
|
|
98
|
+
`ACTION` — exact trajectory match — is a reward component that exists but is **not** in the
|
|
99
|
+
default `reward_basis` for airline, retail or telecom, which is `["DB", "COMMUNICATE"]`.
|
|
100
|
+
|
|
101
|
+
This is the fact that makes the whole exercise honest. A replayed scenario takes the path
|
|
102
|
+
*a previous run* took, which will not be token-identical to the reference actions. Under an
|
|
103
|
+
action-matching benchmark that would be scored as deviation. Under τ²-bench it is scored as
|
|
104
|
+
what it is: the same outcome, reached differently. We are not exploiting a loophole — we
|
|
105
|
+
are using the benchmark's stated philosophy.
|
|
106
|
+
|
|
107
|
+
**Check `reward_basis` per task before running.** If a domain or task set includes `ACTION`,
|
|
108
|
+
exclude it and say so, rather than discovering it in the aggregate.
|
|
109
|
+
|
|
110
|
+
### 3.3 Reward is a *product*, and COMMUNICATE is in the basis
|
|
111
|
+
|
|
112
|
+
`reward = db_reward × communicate_reward`. `communicate_info` is a list of strings the
|
|
113
|
+
agent must have said to the user, matched as substrings.
|
|
114
|
+
|
|
115
|
+
Now read that against what direct-mode replay does. From the README: on a scenario whose
|
|
116
|
+
every step is a wrapped MCP tool, the sequence runs through connections the proxies already
|
|
117
|
+
hold — **zero model tokens** — and the model's only job is one tool call that reads the
|
|
118
|
+
results.
|
|
119
|
+
|
|
120
|
+
> A replay that executes every step perfectly and never *tells the customer the refund
|
|
121
|
+
> amount* scores `1 × 0 = 0`.
|
|
122
|
+
|
|
123
|
+
The design already accounts for this: `steered_full` means every planned step ran with
|
|
124
|
+
scenario-derived inputs **and the response model was produced**. So the machinery should be
|
|
125
|
+
right. But this is the single highest-value assertion to verify early, because its failure
|
|
126
|
+
signature — every task at exactly 0.0 with a healthy step ledger — is easy to misread as a
|
|
127
|
+
plumbing bug and waste a day on.
|
|
128
|
+
|
|
129
|
+
τ-bench is, in this specific sense, a *harder* test of replay than MCPMark: MCPMark grades
|
|
130
|
+
the world, τ-bench grades the world **and** what the agent said about it.
|
|
131
|
+
|
|
132
|
+
---
|
|
133
|
+
|
|
134
|
+
## 4. Architecture — two environments
|
|
135
|
+
|
|
136
|
+
Both constraints in §3.1 have to hold at once: Claude Code must make **real MCP calls**
|
|
137
|
+
(or bir records nothing), and the **trajectory must carry those calls** (or the grade is
|
|
138
|
+
zero). One environment cannot satisfy both — if Claude Code executes a call and the
|
|
139
|
+
adapter also hands it to the orchestrator, the orchestrator executes it a second time,
|
|
140
|
+
the exchange fails as already-exchanged, and the trajectory records an error the evaluator
|
|
141
|
+
cannot replay.
|
|
142
|
+
|
|
143
|
+
Two copies of the domain resolve it. Each executes every call exactly once.
|
|
144
|
+
|
|
145
|
+
```
|
|
146
|
+
tau2 orchestrator ── graded environment ◀── executes the calls the adapter emits
|
|
147
|
+
├── user simulator (--user-llm) (this is what the trajectory records)
|
|
148
|
+
└── ClaudeCodeAgent
|
|
149
|
+
├── ToolBridge ── private copy of the domain ◀── Claude Code acts here
|
|
150
|
+
│ (in-process HTTP, ephemeral port, one per episode)
|
|
151
|
+
└── claude -p / --resume, one call per orchestrator turn
|
|
152
|
+
└── mcp__tau2__* ─▶ bir-proxy ─▶ node mcp-stdio-bridge.mjs ─▶ ToolBridge
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
The copies start identical — taken after `_initialize_environment` has applied the task's
|
|
156
|
+
setup — and replay the same calls in the same order, so they stay in step. Claude Code's
|
|
157
|
+
results are real; the orchestrator's execution is what gets graded.
|
|
158
|
+
|
|
159
|
+
### 4.1 Taking the copy
|
|
160
|
+
|
|
161
|
+
`build_agent` hands the agent `environment.get_tools()`: live `Tool` objects whose `_func`
|
|
162
|
+
is a bound method of the domain's toolkit, which owns the db. Deep-copying **the toolkit**
|
|
163
|
+
yields an independent domain with its own database:
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
toolkit = self.tools[0]._func.__self__ # RetailTools, etc.
|
|
167
|
+
shadow = copy.deepcopy(toolkit)
|
|
168
|
+
shadow_tools = list(shadow.get_tools().values())
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
Copying the toolkit rather than rebuilding the domain from the registry preserves any
|
|
172
|
+
task-specific initialization that would otherwise have to be reconstructed from the task
|
|
173
|
+
object. Verify independence rather than assuming it: write through the copy and confirm
|
|
174
|
+
`environment.get_db_hash()` does not move. `selftest_bridge.py` is that check.
|
|
175
|
+
|
|
176
|
+
### 4.2 One Claude Code turn becomes N+1 messages
|
|
177
|
+
|
|
178
|
+
τ²'s validator (`validate_message_format_default`) rejects a message carrying both text and
|
|
179
|
+
tool calls. So a turn in which Claude Code makes four calls and then replies is *five*
|
|
180
|
+
messages, each separated by the orchestrator executing one call. The adapter queues them
|
|
181
|
+
and hands over one per `generate_next_message`; the tool results the orchestrator returns
|
|
182
|
+
are ignored, because Claude Code already saw the equivalent result from its own copy and
|
|
183
|
+
the run is over by then. Their purpose is the trajectory.
|
|
184
|
+
|
|
185
|
+
> **Stamp each message when you hand it over, not when you build it.** τ² sorts the final
|
|
186
|
+
> trajectory by timestamp. A queue built in one burst carries timestamps *older* than the
|
|
187
|
+
> tool results answering it, so the sort files every assistant message ahead of every tool
|
|
188
|
+
> message and the evaluator refuses to replay it. It is intermittent — when the clock does
|
|
189
|
+
> not tick between batches the timestamps tie and the stable sort happens to be right — so
|
|
190
|
+
> it presents as roughly one run in three dying with an infrastructure error.
|
|
191
|
+
|
|
192
|
+
Two smaller things in the same area:
|
|
193
|
+
|
|
194
|
+
- **Give every assistant message a `cost`, even zero.** `get_cost` discards the whole
|
|
195
|
+
conversation's cost if any single one is `None`, so the tool-call messages carry `0.0`
|
|
196
|
+
and the reply carries the run's total.
|
|
197
|
+
- **Translate the token counts.** τ² sums `prompt_tokens` / `completion_tokens`; Claude
|
|
198
|
+
Code reports `input_tokens`, `output_tokens` and two cache figures. Cached reads are real
|
|
199
|
+
prompt tokens and are most of the prompt on a resumed session, so they belong in the
|
|
200
|
+
total or every turn after the first looks nearly free.
|
|
201
|
+
|
|
202
|
+
### 4.3 Passing the policy
|
|
203
|
+
|
|
204
|
+
The agent constructor receives `domain_policy` as a string. It goes into the session as
|
|
205
|
+
`--append-system-prompt` (or `--append-system-prompt-file` — retail's `policy.md` is long
|
|
206
|
+
enough that a file is the better shape). Do not paste it into the per-turn prompt: it is
|
|
207
|
+
constant across the episode, and repeating it per turn inflates every cost number we are
|
|
208
|
+
trying to measure.
|
|
209
|
+
|
|
210
|
+
### 4.4 Per-turn invocation
|
|
211
|
+
|
|
212
|
+
```bash
|
|
213
|
+
# turn 1
|
|
214
|
+
claude -p "$USER_MESSAGE" \
|
|
215
|
+
--append-system-prompt-file "$POLICY_MD" \
|
|
216
|
+
--settings "$BIR_SETTINGS_JSON" \
|
|
217
|
+
--mcp-config "$EPISODE_MCP_JSON" \
|
|
218
|
+
--output-format json \
|
|
219
|
+
--permission-mode dontAsk \
|
|
220
|
+
--allowedTools "mcp__tau2__*" \
|
|
221
|
+
--max-turns 30
|
|
222
|
+
|
|
223
|
+
# turns 2..n
|
|
224
|
+
claude -p "$USER_MESSAGE" --resume "$SESSION_ID" --output-format json ...
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
`--allowedTools "mcp__tau2__*"` and nothing else. The agent has no business touching the
|
|
228
|
+
filesystem or shell in a customer-service episode, and an allow list that says so turns a
|
|
229
|
+
whole class of confusing failures into a clean permission denial.
|
|
230
|
+
|
|
231
|
+
Check `system/init`'s `mcp_server_errors` on turn 1 of every episode. A `--mcp-config`
|
|
232
|
+
entry that fails validation is skipped *silently*; the episode then runs with no tools and
|
|
233
|
+
scores 0, which reads exactly like a hard task.
|
|
234
|
+
|
|
235
|
+
---
|
|
236
|
+
|
|
237
|
+
## 5. BIR wiring
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
export BIR_AUTH_URL=https://your-basein-service
|
|
241
|
+
export BIR_CONTROL_URL=http://127.0.0.1:53411 # not discovery
|
|
242
|
+
export BIR_CONTROL_TOKEN=...
|
|
243
|
+
export BIR_REPLAY_ALLOW_SERVERS=tau2
|
|
244
|
+
BIR_REPLAY=1 bir-hooks 2>&1 | tee -a results/tau2-audit.log
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
`BIR_CONTROL_URL` / `BIR_CONTROL_TOKEN` are mandatory, not convenient: the README calls
|
|
248
|
+
them the only reliable channel for an SDK session, and every episode turn is one. Discovery
|
|
249
|
+
across a few hundred harness-spawned processes is how an arm silently degrades to Tier 2.
|
|
250
|
+
|
|
251
|
+
`BIR_REPLAY_ALLOW_SERVERS=tau2` is the whole allowance. Direct execution dispatches steps on
|
|
252
|
+
connections that are already open and never reaches the permission system, so the list is
|
|
253
|
+
the blast radius.
|
|
254
|
+
|
|
255
|
+
### 5.1 The per-prompt run boundary — the structural mismatch
|
|
256
|
+
|
|
257
|
+
This is the one that will shape the results, and it is worth understanding before the first
|
|
258
|
+
run rather than after.
|
|
259
|
+
|
|
260
|
+
`onPrompt` rolls a **new run on every prompt**. In τ²-bench, every user-simulator turn is a
|
|
261
|
+
prompt. So an episode of six turns is **six runs**, not one, and matching happens per turn
|
|
262
|
+
against a library of per-turn scenarios.
|
|
263
|
+
|
|
264
|
+
What follows from that:
|
|
265
|
+
|
|
266
|
+
- **Turn 1 carries the intent** — "I want to cancel my order" — and is where a match should
|
|
267
|
+
fire. Later turns are mostly parameter supply ("it's W5678") and are short, cheap, and
|
|
268
|
+
poor matching material.
|
|
269
|
+
- **A scenario is a turn's tool chain, not an episode's.** Cross-episode reuse therefore
|
|
270
|
+
happens turn-by-turn, and the savings ledger sums over turns.
|
|
271
|
+
- **Per-run usage watermarking matters more here than anywhere.** Six runs share one
|
|
272
|
+
Claude Code transcript; summing the file gives turn 6 the cost of turns 1–6. The design
|
|
273
|
+
doc's `markTranscriptUsage` / `usageSince` delta is not an optimisation in this setting,
|
|
274
|
+
it is the difference between a real number and a fabricated one. Check `measured: true`.
|
|
275
|
+
|
|
276
|
+
If the results show matching that fires reliably on turn 1 and rarely after, that is not a
|
|
277
|
+
bug — it is the honest shape of the product against a multi-turn benchmark, and it should
|
|
278
|
+
be reported that way.
|
|
279
|
+
|
|
280
|
+
---
|
|
281
|
+
|
|
282
|
+
## 6. Running it
|
|
283
|
+
|
|
284
|
+
```bash
|
|
285
|
+
git clone https://github.com/sierra-research/tau2-bench && cd tau2-bench
|
|
286
|
+
uv sync && cp .env.example .env # user-simulator key goes here
|
|
287
|
+
|
|
288
|
+
# smoke: one task, one trial
|
|
289
|
+
tau2 run --domain retail --agent claude_code --user-llm <model> \
|
|
290
|
+
--num-trials 1 --task-ids 0
|
|
291
|
+
|
|
292
|
+
tau2 view # read the trajectory you just produced
|
|
293
|
+
|
|
294
|
+
# baseline arm, recording only (BIR_REPLAY unset)
|
|
295
|
+
tau2 run --domain retail --agent claude_code --user-llm <model> \
|
|
296
|
+
--num-trials 4 --max-concurrency 4
|
|
297
|
+
|
|
298
|
+
# replay arm, library seeded from the passing runs above
|
|
299
|
+
BIR_REPLAY=1 tau2 run --domain retail --agent claude_code --user-llm <model> \
|
|
300
|
+
--num-trials 4 --max-concurrency 4
|
|
301
|
+
```
|
|
302
|
+
|
|
303
|
+
Pin the `--user-llm` model and version across all arms. It is the environment; changing it
|
|
304
|
+
changes the task, and a shifted baseline invalidates every comparison.
|
|
305
|
+
|
|
306
|
+
---
|
|
307
|
+
|
|
308
|
+
## 7. Verification gates
|
|
309
|
+
|
|
310
|
+
Run these on `--task-ids 0`, in order. Each has a distinct failure signature, and they are
|
|
311
|
+
ordered so that the cheapest and most misleading failures are caught first.
|
|
312
|
+
|
|
313
|
+
| # | Gate | Failure looks like | Means |
|
|
314
|
+
|---|---|---|---|
|
|
315
|
+
| 1 | `mcp_servers` in `system/init` lists `tau2` as connected | score 0, no tool calls | config skipped silently |
|
|
316
|
+
| 2 | `bir doctor` green, `metadata.tier == 1` inside a harness-spawned run | recording exists but thin | Tier 2 — hooks not loading |
|
|
317
|
+
| 3 | The proxy recorded the episode's tool calls | empty step stream on a passing task | traffic bypassed bir |
|
|
318
|
+
| 4 | Reward > 0 on a task the reference solves | *every* task exactly 0.0 | the trajectory carries no tool calls (§3.1) |
|
|
319
|
+
| 5 | No duplicated side effects | reward 0, doubled refund in the env | one environment, executed twice (§4) |
|
|
320
|
+
| 6 | Zero retries across a multi-trial run | ~1 run in 3 dies as "infrastructure error" | the timestamp race (§4.2) |
|
|
321
|
+
| 7 | `Avg Cost/Conversation` is a number, not `n/a` | costs silently absent | an assistant message with `cost=None` |
|
|
322
|
+
| 8 | On the replay arm: DB=1 **and** COMMUNICATE=1 | reward 0, step ledger healthy | **§3.3** — replay produced no response text |
|
|
323
|
+
|
|
324
|
+
**Every one of these fails quietly.** That is the point of the list. Three of them we hit
|
|
325
|
+
for real, and none announced itself:
|
|
326
|
+
|
|
327
|
+
- **Gate 4** produced a clean table reading `reward 0.0000, DB match 0/1` on a task the
|
|
328
|
+
agent had solved *perfectly* — its calls matched the reference exactly. It reads like a
|
|
329
|
+
hard task, not a broken harness.
|
|
330
|
+
- **Gate 6** was absorbed by τ²'s own retry logic, so the run "succeeded" while costing 3×
|
|
331
|
+
and mixing infrastructure errors into what would look like task variance across a sweep.
|
|
332
|
+
- A fourth, not a gate but worth knowing: on Windows, npm's `claude.cmd` shim routes
|
|
333
|
+
arguments through `cmd.exe`, which re-parses them. Any user turn containing `&`, `|`,
|
|
334
|
+
`%`, `^` or a newline arrives truncated, and the agent answers *"your message came
|
|
335
|
+
through empty."* Call `claude.exe` directly and refuse to fall back to the shim.
|
|
336
|
+
|
|
337
|
+
The lesson generalises past this benchmark: a harness bug and a hard task produce the same
|
|
338
|
+
number. Read one transcript per configuration before trusting any aggregate.
|
|
339
|
+
|
|
340
|
+
---
|
|
341
|
+
|
|
342
|
+
## 8. What to measure
|
|
343
|
+
|
|
344
|
+
Per arm, per domain: reward (mean), pass^k across trials, `$`/episode and `$`/turn from
|
|
345
|
+
`--output-format json`'s `total_cost_usd`, wall-clock per episode, and turn count.
|
|
346
|
+
|
|
347
|
+
Per replayed turn, from the audit log and the execution ledger: outcome
|
|
348
|
+
(`steered_full` / `diverged` / `not_steered` / `failed`), match similarity against
|
|
349
|
+
`BIR_MIN_STEER_SIMILARITY` (default 0.92), direct vs steered mode, `deriveCostUsd` against
|
|
350
|
+
`sessionCostUsd`, and the per-step verdicts.
|
|
351
|
+
|
|
352
|
+
Two τ-specific splits worth reporting that MCPMark cannot give us:
|
|
353
|
+
|
|
354
|
+
- **Reward decomposed.** DB and COMMUNICATE separately, always. An arm where DB holds and
|
|
355
|
+
COMMUNICATE slips is a precise, fixable finding about what replay omits; the product hides
|
|
356
|
+
it completely.
|
|
357
|
+
- **Match rate by turn index.** Turn 1 versus turns 2..n (§5.1). This is the number that
|
|
358
|
+
says whether calculated replay generalises across a *task family* or only across
|
|
359
|
+
restatements of one task, and it is the question τ-bench was chosen to answer.
|
|
360
|
+
|
|
361
|
+
Then the control from [benchmark.md](benchmark.md) §7 Phase 3, which matters more here than
|
|
362
|
+
anywhere: τ-bench task families differ by *entity*, so a matcher that ignores which order
|
|
363
|
+
id the customer named will replay cheerfully against the wrong order. Build the near-miss
|
|
364
|
+
set by swapping entities between same-shape tasks — same words, different customer — and
|
|
365
|
+
report the **false replay rate**. In a retail domain, that number has an obvious real-world
|
|
366
|
+
reading, which is exactly why it should be published whatever it says.
|
|
367
|
+
|
|
368
|
+
---
|
|
369
|
+
|
|
370
|
+
## 9. Known limits
|
|
371
|
+
|
|
372
|
+
- **The MCP wrapper is ours.** τ²-bench has no MCP interface, so the transport under test is
|
|
373
|
+
a shim we wrote. A sceptic can question it; the answer is to publish it and to run gate 3
|
|
374
|
+
and gate 5 in public. MCPMark has no such caveat, which is the other reason it is primary.
|
|
375
|
+
- **The user simulator is an LLM**, so the environment is stochastic across trials. pass^k
|
|
376
|
+
here mixes agent variance with simulator variance. Pin the simulator model and report it
|
|
377
|
+
as a limit rather than pretending the variance is all ours.
|
|
378
|
+
- **Six runs per episode** (§5.1) makes every per-run number a per-turn number. Say so in
|
|
379
|
+
the table headers; a `$`/run figure that a reader takes for `$`/task is off by ~6×.
|
|
380
|
+
- **`banking_knowledge` stays out.** Same reasoning as [benchmark.md](benchmark.md) §5 — it
|
|
381
|
+
is retrieval over a document set, not a repeatable tool chain, and there is nothing there
|
|
382
|
+
for a scenario to be.
|
|
383
|
+
- **Voice mode is out of scope.** Full-duplex needs `FullDuplexAgent` and a realtime
|
|
384
|
+
provider; Claude Code is half-duplex and so is this integration.
|
|
385
|
+
|
|
386
|
+
---
|
|
387
|
+
|
|
388
|
+
## Sources
|
|
389
|
+
|
|
390
|
+
- τ²-bench — [repo](https://github.com/sierra-research/tau2-bench) · [agent developer guide](https://github.com/sierra-research/tau2-bench/blob/main/src/tau2/agent/README.md) · [orchestrator guide](https://github.com/sierra-research/tau2-bench/blob/main/src/tau2/orchestrator/README.md) · [evaluation docs](https://github.com/sierra-research/tau2-bench/blob/main/docs/evaluation.md)
|
|
391
|
+
- Papers — [τ²-bench (arXiv:2506.07982)](https://arxiv.org/pdf/2506.07982) · [τ-bench (arXiv:2406.12045)](https://arxiv.org/pdf/2406.12045)
|
|
392
|
+
- [Trajectory visualizer](https://taubench.com/trajectory-visualizer/?model=qwen3.5-397b-a17b-think_sierra_2026-03-02&domain=retail&task=0) · [leaderboard](https://taubench.com/leaderboard/)
|
|
393
|
+
- [Claude Code headless mode](https://code.claude.com/docs/en/headless)
|
|
394
|
+
- This repo — [benchmark.md](benchmark.md) · [calculatedReplay.md](calculatedReplay.md) · [calculatedReplayGuide.md](calculatedReplayGuide.md)
|
package/package.json
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@basein/runner",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "A recording MCP proxy: sits between any MCP client and its MCP servers, executes each call on the client's behalf, and records the run as a reusable BaseIn scenario.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"repository": {
|
|
8
|
+
"type": "git",
|
|
9
|
+
"url": "git+https://github.com/eran0lavi/BaseInstRunnerMCP.git"
|
|
10
|
+
},
|
|
11
|
+
"publishConfig": {
|
|
12
|
+
"access": "public"
|
|
13
|
+
},
|
|
14
|
+
"engines": {
|
|
15
|
+
"node": ">=20"
|
|
16
|
+
},
|
|
17
|
+
"bin": {
|
|
18
|
+
"bir": "dist/bin/bir.js",
|
|
19
|
+
"bir-proxy": "dist/bin/bir-proxy.js",
|
|
20
|
+
"bir-hooks": "dist/bin/bir-hooks.js",
|
|
21
|
+
"bir-scenario": "dist/bin/bir-scenario.js"
|
|
22
|
+
},
|
|
23
|
+
"main": "dist/index.js",
|
|
24
|
+
"types": "dist/index.d.ts",
|
|
25
|
+
"files": [
|
|
26
|
+
"dist",
|
|
27
|
+
"!dist/**/*.map",
|
|
28
|
+
"README.md",
|
|
29
|
+
"LICENSE",
|
|
30
|
+
"docs",
|
|
31
|
+
"!docs/next.md"
|
|
32
|
+
],
|
|
33
|
+
"scripts": {
|
|
34
|
+
"build": "tsc -p tsconfig.json",
|
|
35
|
+
"typecheck": "tsc -p tsconfig.test.json --noEmit",
|
|
36
|
+
"pretest": "tsc -p tsconfig.test.json",
|
|
37
|
+
"test": "node test/run-tests.mjs",
|
|
38
|
+
"prepublishOnly": "npm run build",
|
|
39
|
+
"test:smoke": "node test/smoke-replay.mjs"
|
|
40
|
+
},
|
|
41
|
+
"keywords": [
|
|
42
|
+
"mcp",
|
|
43
|
+
"proxy",
|
|
44
|
+
"recorder",
|
|
45
|
+
"claude-code",
|
|
46
|
+
"agent",
|
|
47
|
+
"observability",
|
|
48
|
+
"replay",
|
|
49
|
+
"scenario"
|
|
50
|
+
],
|
|
51
|
+
"dependencies": {},
|
|
52
|
+
"devDependencies": {
|
|
53
|
+
"@types/node": "^20.14.0",
|
|
54
|
+
"typescript": "^5.6.0"
|
|
55
|
+
}
|
|
56
|
+
}
|