@basein/runner 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +276 -0
  3. package/dist/auth/client.d.ts +85 -0
  4. package/dist/auth/client.js +284 -0
  5. package/dist/bin/bir-hooks.d.ts +48 -0
  6. package/dist/bin/bir-hooks.js +201 -0
  7. package/dist/bin/bir-proxy.d.ts +45 -0
  8. package/dist/bin/bir-proxy.js +207 -0
  9. package/dist/bin/bir-scenario.d.ts +24 -0
  10. package/dist/bin/bir-scenario.js +177 -0
  11. package/dist/bin/bir.d.ts +21 -0
  12. package/dist/bin/bir.js +876 -0
  13. package/dist/config/adapters/claude-code.d.ts +76 -0
  14. package/dist/config/adapters/claude-code.js +181 -0
  15. package/dist/config/adapters/generic.d.ts +17 -0
  16. package/dist/config/adapters/generic.js +36 -0
  17. package/dist/config/generate.d.ts +127 -0
  18. package/dist/config/generate.js +114 -0
  19. package/dist/config/resolve.d.ts +68 -0
  20. package/dist/config/resolve.js +132 -0
  21. package/dist/control/client.d.ts +56 -0
  22. package/dist/control/client.js +86 -0
  23. package/dist/control/correlation.d.ts +86 -0
  24. package/dist/control/correlation.js +0 -0
  25. package/dist/control/discovery.d.ts +50 -0
  26. package/dist/control/discovery.js +123 -0
  27. package/dist/control/ordering.d.ts +38 -0
  28. package/dist/control/ordering.js +44 -0
  29. package/dist/control/paths.d.ts +32 -0
  30. package/dist/control/paths.js +56 -0
  31. package/dist/control/server.d.ts +272 -0
  32. package/dist/control/server.js +1131 -0
  33. package/dist/control/transcript.d.ts +75 -0
  34. package/dist/control/transcript.js +241 -0
  35. package/dist/index.d.ts +37 -0
  36. package/dist/index.js +32 -0
  37. package/dist/jsonrpc/framing.d.ts +49 -0
  38. package/dist/jsonrpc/framing.js +143 -0
  39. package/dist/jsonrpc/types.d.ts +52 -0
  40. package/dist/jsonrpc/types.js +46 -0
  41. package/dist/proxy/intercept.d.ts +55 -0
  42. package/dist/proxy/intercept.js +147 -0
  43. package/dist/proxy/relay.d.ts +97 -0
  44. package/dist/proxy/relay.js +166 -0
  45. package/dist/proxy/session.d.ts +116 -0
  46. package/dist/proxy/session.js +319 -0
  47. package/dist/record/housekeeping.d.ts +34 -0
  48. package/dist/record/housekeeping.js +39 -0
  49. package/dist/record/queue.d.ts +48 -0
  50. package/dist/record/queue.js +96 -0
  51. package/dist/record/recorder.d.ts +111 -0
  52. package/dist/record/recorder.js +39 -0
  53. package/dist/record/redact.d.ts +37 -0
  54. package/dist/record/redact.js +119 -0
  55. package/dist/record/remote-recorder.d.ts +110 -0
  56. package/dist/record/remote-recorder.js +301 -0
  57. package/dist/record/truncate.d.ts +36 -0
  58. package/dist/record/truncate.js +85 -0
  59. package/dist/replay/bundle.d.ts +36 -0
  60. package/dist/replay/bundle.js +89 -0
  61. package/dist/replay/controller.d.ts +300 -0
  62. package/dist/replay/controller.js +807 -0
  63. package/dist/replay/coverage.d.ts +41 -0
  64. package/dist/replay/coverage.js +56 -0
  65. package/dist/replay/derive.d.ts +58 -0
  66. package/dist/replay/derive.js +166 -0
  67. package/dist/replay/executor.d.ts +78 -0
  68. package/dist/replay/executor.js +233 -0
  69. package/dist/replay/logic.d.ts +31 -0
  70. package/dist/replay/logic.js +50 -0
  71. package/dist/replay/plan.d.ts +181 -0
  72. package/dist/replay/plan.js +397 -0
  73. package/dist/replay/pricing.d.ts +41 -0
  74. package/dist/replay/pricing.js +76 -0
  75. package/dist/replay/source-run.d.ts +50 -0
  76. package/dist/replay/source-run.js +98 -0
  77. package/dist/replay/tool-error.d.ts +22 -0
  78. package/dist/replay/tool-error.js +60 -0
  79. package/dist/replay/types.d.ts +116 -0
  80. package/dist/replay/types.js +35 -0
  81. package/dist/upstream/client.d.ts +78 -0
  82. package/dist/upstream/client.js +114 -0
  83. package/dist/upstream/http-client.d.ts +78 -0
  84. package/dist/upstream/http-client.js +261 -0
  85. package/dist/upstream/lazy-client.d.ts +31 -0
  86. package/dist/upstream/lazy-client.js +53 -0
  87. package/dist/upstream/stdio-client.d.ts +57 -0
  88. package/dist/upstream/stdio-client.js +203 -0
  89. package/dist/util/log.d.ts +27 -0
  90. package/dist/util/log.js +51 -0
  91. package/dist/util/version.d.ts +2 -0
  92. package/dist/util/version.js +40 -0
  93. package/docs/BaseInstRunner.md +621 -0
  94. package/docs/calculatedReplay.md +1185 -0
  95. package/docs/calculatedReplayGuide.md +448 -0
  96. package/docs/installRun.md +413 -0
  97. package/docs/mcpmark.md +752 -0
  98. package/docs/quickstart.md +201 -0
  99. package/docs/t-bench.md +394 -0
  100. package/package.json +56 -0
@@ -0,0 +1,201 @@
1
+ # Setting up a runner machine
2
+
3
+ This guide takes one computer from nothing to recording. It assumes no prior
4
+ knowledge of the project. It should take about ten minutes, most of which is
5
+ waiting for downloads.
6
+
7
+ **What you are setting up.** BaseInstRunner sits quietly between your AI coding
8
+ assistant and the tools it uses, watches what happens, and saves each session to
9
+ the BaseIn service so it can be reviewed or re-run later. It does not change what
10
+ the assistant does.
11
+
12
+ ---
13
+
14
+ ## Before you start
15
+
16
+ Three things:
17
+
18
+ 1. **Node 20 or newer.** Check by opening a terminal and typing `node -v`. If you
19
+ see something like `v20.11.0` or higher, you are fine. If you see an error or
20
+ a smaller number, install it from [nodejs.org](https://nodejs.org) first.
21
+ 2. **The BaseIn service address** — a URL like `https://basein.example.com`.
22
+ Whoever runs the service gives you this.
23
+ 3. **Your BaseIn email and password.** You will type these once.
24
+
25
+ ---
26
+
27
+ ## Step 1 — run one command
28
+
29
+ Open a terminal **in the folder where you unpacked this project's `scripts`
30
+ folder**, and run the line for your machine. Replace the two placeholder values
31
+ with your real service address and your real project folder.
32
+
33
+ ### Windows
34
+
35
+ ```powershell
36
+ .\scripts\install-runner.ps1 C:\path\to\your\project -AuthUrl https://basein.example.com
37
+ ```
38
+
39
+ ### macOS
40
+
41
+ ```bash
42
+ BIR_AUTH_URL=https://basein.example.com ./scripts/install-runner.sh ~/path/to/your/project
43
+ ```
44
+
45
+ > **"Running scripts is disabled on this system"** (Windows only) — Windows blocks
46
+ > unsigned scripts by default. Allow them for your own account with:
47
+ > `Set-ExecutionPolicy -Scope CurrentUser RemoteSigned`, then run the command
48
+ > again.
49
+
50
+ ---
51
+
52
+ ## Step 2 — watch what it does
53
+
54
+ The script narrates each stage. A healthy run looks like this:
55
+
56
+ ```
57
+ ==> Checking Node
58
+ node v20.11.0
59
+ ==> Installing the runner
60
+ @basein/runner@0.1.0 from the registry
61
+ bir 0.1.0 -- all four binaries on PATH
62
+ ==> Configuring the BaseIn service
63
+ BIR_AUTH_URL=https://basein.example.com (persisted for this user)
64
+ ==> Signing in
65
+ Email: you@example.com
66
+ Password:
67
+ ==> Wrapping MCP servers in C:\path\to\your\project
68
+ + chrome-devtools -> bir-proxy (project scope, upstream: npx)
69
+ + hooks -> ...\.claude\settings.json
70
+
71
+ Wrapped 1 server.
72
+ ```
73
+
74
+ The password does not appear as you type it. That is normal.
75
+
76
+ If anything goes wrong the script stops immediately and says why — it checks
77
+ everything it can *before* changing your machine, so a failed run leaves nothing
78
+ half-installed.
79
+
80
+ ---
81
+
82
+ ## Step 3 — start recording
83
+
84
+ Two terminals, in the project folder you gave the script.
85
+
86
+ **First terminal** — leave this one running the whole time:
87
+
88
+ ```
89
+ bir-hooks
90
+ ```
91
+
92
+ **Second terminal** — your normal work:
93
+
94
+ ```
95
+ claude
96
+ ```
97
+
98
+ That is it. Everything you do in the second terminal is now recorded.
99
+
100
+ When you are finished for the day, press `Ctrl+C` in the first terminal.
101
+
102
+ ---
103
+
104
+ ## Step 4 — check it actually worked
105
+
106
+ With `bir-hooks` running, in another terminal:
107
+
108
+ ```
109
+ bir doctor
110
+ ```
111
+
112
+ This is the one command worth remembering. It does not read a settings file and
113
+ tell you what *should* happen — it asks the running system what is *actually*
114
+ happening, and it fails loudly when something is wrong.
115
+
116
+ To see what is set up without needing anything running:
117
+
118
+ ```
119
+ bir status
120
+ ```
121
+
122
+ ---
123
+
124
+ ## When something is wrong
125
+
126
+ ### "It says nothing is recorded"
127
+
128
+ `bir status` will show `BaseIn: (BIR_AUTH_URL not set — nothing is recorded)`.
129
+ The service address did not stick. Close the terminal, open a new one, and check
130
+ `echo $env:BIR_AUTH_URL` (Windows) or `echo $BIR_AUTH_URL` (macOS). If it is
131
+ empty, run the install script again with the `-AuthUrl` / `BIR_AUTH_URL` value.
132
+
133
+ ### "bir is not recognised as a command"
134
+
135
+ The install worked but your terminal has not noticed yet. **Close the terminal
136
+ and open a new one.** This fixes it almost every time.
137
+
138
+ ### "It records the tools but not what I typed"
139
+
140
+ `bir-hooks` is not running, or it is running in a different folder. It has to be
141
+ started **in the same folder** as your session — that is how the two find each
142
+ other. Stop it, `cd` to the project folder, and start it again.
143
+
144
+ ### "invalid JSON at ...\.mcp.json"
145
+
146
+ Something has edited that file into a shape that is no longer valid JSON —
147
+ usually a missing comma or a stray bracket. Open it and check. (A file saved by
148
+ Notepad is fine; that case is handled.)
149
+
150
+ ### Anything else
151
+
152
+ Run `bir doctor` and keep the output. It says which part of the chain is broken,
153
+ which is most of the way to an answer.
154
+
155
+ ---
156
+
157
+ ## Updating to a newer version
158
+
159
+ Run the same install script again with the new version number:
160
+
161
+ ```powershell
162
+ .\scripts\install-runner.ps1 C:\path\to\your\project -Version 0.2.0 -AuthUrl https://basein.example.com
163
+ ```
164
+
165
+ ```bash
166
+ BIR_VERSION=0.2.0 BIR_AUTH_URL=https://basein.example.com ./scripts/install-runner.sh ~/path/to/your/project
167
+ ```
168
+
169
+ It is safe to re-run: it will not ask you to sign in again, and it will not
170
+ double-wrap anything.
171
+
172
+ **One thing you must do by hand: stop `bir-hooks` and start it again.** A running
173
+ process keeps using the old version until it is restarted. If you forget, the
174
+ next session logs a `version.skew` warning telling you exactly that.
175
+
176
+ ---
177
+
178
+ ## Undoing it
179
+
180
+ In the project folder:
181
+
182
+ ```
183
+ bir uninstall
184
+ ```
185
+
186
+ This puts every file it touched back exactly as it found it, byte for byte.
187
+
188
+ To remove the software entirely as well:
189
+
190
+ ```
191
+ npm rm -g @basein/runner
192
+ ```
193
+
194
+ ---
195
+
196
+ ## Where to go next
197
+
198
+ - [installRun.md](installRun.md) — the fleet version of this: packaging, release
199
+ process, offline installs, keeping `bir-hooks` alive across reboots.
200
+ - [../README.md](../README.md) — what is recorded, what cannot be, and every
201
+ configuration switch.
@@ -0,0 +1,394 @@
1
+ # τ²-bench — running it against BaseInstRunner, with Claude Code as the agent
2
+
3
+ Companion to [benchmark.md](benchmark.md), which recommends MCPMark first and τ²-bench
4
+ second. This is the second half: what τ²-bench actually is, how to read a trajectory,
5
+ and the three structural facts that decide whether the integration measures our product
6
+ or measures nothing at all.
7
+
8
+ **Read §3 before writing any code.** τ²-bench executes tool calls *inside its own
9
+ orchestrator*. Nothing crosses MCP by default, which means a naive integration produces a
10
+ clean run, a plausible score, and zero recorded steps.
11
+
12
+ ---
13
+
14
+ ## 1. Why this benchmark, in one paragraph
15
+
16
+ MCPMark proves a replayed scenario survives a **fresh instance of the same task**.
17
+ τ²-bench proves it survives a **different task of the same shape** — "cancel order W1234
18
+ for user A" against "cancel order W5678 for user B". Its domains are built out of
19
+ policy-constrained task families over a shared database, which is exactly the surface that
20
+ per-step parameter derivation exists for. Nothing else public tests the *calculated* half
21
+ of calculated replay this directly. It also carries the name recognition that makes a
22
+ result travel.
23
+
24
+ Domains: `retail` (115 tasks), `airline` (50), `telecom` (114), plus `mock` and
25
+ `banking_knowledge`. Python ≥3.12 <3.14, `uv sync`, LiteLLM-compatible keys in `.env`.
26
+
27
+ ---
28
+
29
+ ## 2. Reading a trajectory
30
+
31
+ The link that prompted this doc:
32
+
33
+ ```
34
+ https://taubench.com/trajectory-visualizer?model=qwen3.5-397b-a17b-think_sierra_2026-03-02&domain=retail&task=0
35
+ ```
36
+
37
+ It is a client-side viewer over the simulation JSON that `tau2 run` writes to
38
+ `data/simulations/`. Two modes — **Trajectories** and **Tasks** — and the URL pins one
39
+ `(model, domain, task)` triple. The same data is browsable locally with `tau2 view`, which
40
+ is the version to use while building the adapter, because it reads your own runs.
41
+
42
+ An episode is a four-party loop, and the visualizer shows it as one interleaved column:
43
+
44
+ ```
45
+ Agent ──message──▶ User ──message──▶ Agent ──tool_call──▶ Environment ──result──▶ Agent
46
+ ```
47
+
48
+ | Element | What it is | What we care about |
49
+ |---|---|---|
50
+ | **UserMessage** | the *simulated* customer, driven by `--user-llm` | turn 1 carries the intent; later turns supply parameters |
51
+ | **AssistantMessage** | the agent's text, and optionally `tool_calls` | the text is what the COMMUNICATE check reads |
52
+ | **ToolMessage** | the environment's result | executed by the orchestrator, not the agent (§3.1) |
53
+ | **evaluation_criteria.actions** | *one* reference solution, replayed on a fresh env to derive the target DB hash | **not** a required path (§3.2) |
54
+ | reward breakdown | DB / COMMUNICATE / ENV_ASSERTION / NL_ASSERTION | the product of whatever `reward_basis` lists |
55
+
56
+ Episodes end on a stop token, a user-simulator "task complete" transfer, or `max_steps`.
57
+
58
+ **The one habit worth forming:** when a task scores 0, open the trajectory and check
59
+ *which* component was 0. A DB failure and a COMMUNICATE failure look identical in the
60
+ aggregate and mean opposite things for us — see §3.3.
61
+
62
+ ---
63
+
64
+ ## 3. The three facts that decide the integration
65
+
66
+ ### 3.1 The grade comes from the transcript, not from the world
67
+
68
+ *Corrected after the first working run. An earlier version of this section said the
69
+ binding constraint was that the orchestrator executes tool calls. That is true — the agent
70
+ returns an `AssistantMessage` containing `tool_calls` and the orchestrator invokes them
71
+ against the domain environment — but it is not the constraint that decides the design.*
72
+
73
+ The constraint that decides the design is in `evaluator_env.py`: τ² **does not grade the
74
+ environment the episode ran in**. It constructs a *fresh* environment, replays the
75
+ trajectory's `(tool_call, tool_result)` pairs onto it via `set_state`, and hashes that.
76
+
77
+ > **Consequence:** the reward is earned by what appears in the messages. An agent that
78
+ > executes the whole task perfectly against the live environment and reports it in prose
79
+ > scores **zero**, because the graded environment received nothing.
80
+
81
+ We measured exactly this. Claude Code solved retail task 0 with a call sequence matching
82
+ the reference trajectory *exactly* — same tools, same item ids, same payment method — and
83
+ scored `DB 0`, `reward 0`. Nothing in the results table said "your harness is broken"; it
84
+ read like a hard task.
85
+
86
+ τ²-bench also ships no MCP interface: domain tools are Python callables wrapped in `Tool`
87
+ objects and called in-process. So bir sees nothing unless Claude Code makes real MCP calls
88
+ of its own. Those two facts together — the trajectory must carry the calls, *and* Claude
89
+ Code must make them over MCP — are what §4 has to satisfy at once.
90
+
91
+ ### 3.2 Reward is outcome-based, so replay is legitimately scoreable
92
+
93
+ `evaluation_criteria.actions` is a reference trajectory, replayed on a fresh environment to
94
+ establish a **target database hash**. The DB check compares hashes. Any sequence of tool
95
+ calls reaching an equivalent end state passes. The docs are emphatic: *actions document one
96
+ working solution, not the only acceptable one.*
97
+
98
+ `ACTION` — exact trajectory match — is a reward component that exists but is **not** in the
99
+ default `reward_basis` for airline, retail or telecom, which is `["DB", "COMMUNICATE"]`.
100
+
101
+ This is the fact that makes the whole exercise honest. A replayed scenario takes the path
102
+ *a previous run* took, which will not be token-identical to the reference actions. Under an
103
+ action-matching benchmark that would be scored as deviation. Under τ²-bench it is scored as
104
+ what it is: the same outcome, reached differently. We are not exploiting a loophole — we
105
+ are using the benchmark's stated philosophy.
106
+
107
+ **Check `reward_basis` per task before running.** If a domain or task set includes `ACTION`,
108
+ exclude it and say so, rather than discovering it in the aggregate.
109
+
110
+ ### 3.3 Reward is a *product*, and COMMUNICATE is in the basis
111
+
112
+ `reward = db_reward × communicate_reward`. `communicate_info` is a list of strings the
113
+ agent must have said to the user, matched as substrings.
114
+
115
+ Now read that against what direct-mode replay does. From the README: on a scenario whose
116
+ every step is a wrapped MCP tool, the sequence runs through connections the proxies already
117
+ hold — **zero model tokens** — and the model's only job is one tool call that reads the
118
+ results.
119
+
120
+ > A replay that executes every step perfectly and never *tells the customer the refund
121
+ > amount* scores `1 × 0 = 0`.
122
+
123
+ The design already accounts for this: `steered_full` means every planned step ran with
124
+ scenario-derived inputs **and the response model was produced**. So the machinery should be
125
+ right. But this is the single highest-value assertion to verify early, because its failure
126
+ signature — every task at exactly 0.0 with a healthy step ledger — is easy to misread as a
127
+ plumbing bug and waste a day on.
128
+
129
+ τ-bench is, in this specific sense, a *harder* test of replay than MCPMark: MCPMark grades
130
+ the world, τ-bench grades the world **and** what the agent said about it.
131
+
132
+ ---
133
+
134
+ ## 4. Architecture — two environments
135
+
136
+ Both constraints in §3.1 have to hold at once: Claude Code must make **real MCP calls**
137
+ (or bir records nothing), and the **trajectory must carry those calls** (or the grade is
138
+ zero). One environment cannot satisfy both — if Claude Code executes a call and the
139
+ adapter also hands it to the orchestrator, the orchestrator executes it a second time,
140
+ the exchange fails as already-exchanged, and the trajectory records an error the evaluator
141
+ cannot replay.
142
+
143
+ Two copies of the domain resolve it. Each executes every call exactly once.
144
+
145
+ ```
146
+ tau2 orchestrator ── graded environment ◀── executes the calls the adapter emits
147
+ ├── user simulator (--user-llm) (this is what the trajectory records)
148
+ └── ClaudeCodeAgent
149
+ ├── ToolBridge ── private copy of the domain ◀── Claude Code acts here
150
+ │ (in-process HTTP, ephemeral port, one per episode)
151
+ └── claude -p / --resume, one call per orchestrator turn
152
+ └── mcp__tau2__* ─▶ bir-proxy ─▶ node mcp-stdio-bridge.mjs ─▶ ToolBridge
153
+ ```
154
+
155
+ The copies start identical — taken after `_initialize_environment` has applied the task's
156
+ setup — and replay the same calls in the same order, so they stay in step. Claude Code's
157
+ results are real; the orchestrator's execution is what gets graded.
158
+
159
+ ### 4.1 Taking the copy
160
+
161
+ `build_agent` hands the agent `environment.get_tools()`: live `Tool` objects whose `_func`
162
+ is a bound method of the domain's toolkit, which owns the db. Deep-copying **the toolkit**
163
+ yields an independent domain with its own database:
164
+
165
+ ```python
166
+ toolkit = self.tools[0]._func.__self__ # RetailTools, etc.
167
+ shadow = copy.deepcopy(toolkit)
168
+ shadow_tools = list(shadow.get_tools().values())
169
+ ```
170
+
171
+ Copying the toolkit rather than rebuilding the domain from the registry preserves any
172
+ task-specific initialization that would otherwise have to be reconstructed from the task
173
+ object. Verify independence rather than assuming it: write through the copy and confirm
174
+ `environment.get_db_hash()` does not move. `selftest_bridge.py` is that check.
175
+
176
+ ### 4.2 One Claude Code turn becomes N+1 messages
177
+
178
+ τ²'s validator (`validate_message_format_default`) rejects a message carrying both text and
179
+ tool calls. So a turn in which Claude Code makes four calls and then replies is *five*
180
+ messages, each separated by the orchestrator executing one call. The adapter queues them
181
+ and hands over one per `generate_next_message`; the tool results the orchestrator returns
182
+ are ignored, because Claude Code already saw the equivalent result from its own copy and
183
+ the run is over by then. Their purpose is the trajectory.
184
+
185
+ > **Stamp each message when you hand it over, not when you build it.** τ² sorts the final
186
+ > trajectory by timestamp. A queue built in one burst carries timestamps *older* than the
187
+ > tool results answering it, so the sort files every assistant message ahead of every tool
188
+ > message and the evaluator refuses to replay it. It is intermittent — when the clock does
189
+ > not tick between batches the timestamps tie and the stable sort happens to be right — so
190
+ > it presents as roughly one run in three dying with an infrastructure error.
191
+
192
+ Two smaller things in the same area:
193
+
194
+ - **Give every assistant message a `cost`, even zero.** `get_cost` discards the whole
195
+ conversation's cost if any single one is `None`, so the tool-call messages carry `0.0`
196
+ and the reply carries the run's total.
197
+ - **Translate the token counts.** τ² sums `prompt_tokens` / `completion_tokens`; Claude
198
+ Code reports `input_tokens`, `output_tokens` and two cache figures. Cached reads are real
199
+ prompt tokens and are most of the prompt on a resumed session, so they belong in the
200
+ total or every turn after the first looks nearly free.
201
+
202
+ ### 4.3 Passing the policy
203
+
204
+ The agent constructor receives `domain_policy` as a string. It goes into the session as
205
+ `--append-system-prompt` (or `--append-system-prompt-file` — retail's `policy.md` is long
206
+ enough that a file is the better shape). Do not paste it into the per-turn prompt: it is
207
+ constant across the episode, and repeating it per turn inflates every cost number we are
208
+ trying to measure.
209
+
210
+ ### 4.4 Per-turn invocation
211
+
212
+ ```bash
213
+ # turn 1
214
+ claude -p "$USER_MESSAGE" \
215
+ --append-system-prompt-file "$POLICY_MD" \
216
+ --settings "$BIR_SETTINGS_JSON" \
217
+ --mcp-config "$EPISODE_MCP_JSON" \
218
+ --output-format json \
219
+ --permission-mode dontAsk \
220
+ --allowedTools "mcp__tau2__*" \
221
+ --max-turns 30
222
+
223
+ # turns 2..n
224
+ claude -p "$USER_MESSAGE" --resume "$SESSION_ID" --output-format json ...
225
+ ```
226
+
227
+ `--allowedTools "mcp__tau2__*"` and nothing else. The agent has no business touching the
228
+ filesystem or shell in a customer-service episode, and an allow list that says so turns a
229
+ whole class of confusing failures into a clean permission denial.
230
+
231
+ Check `system/init`'s `mcp_server_errors` on turn 1 of every episode. A `--mcp-config`
232
+ entry that fails validation is skipped *silently*; the episode then runs with no tools and
233
+ scores 0, which reads exactly like a hard task.
234
+
235
+ ---
236
+
237
+ ## 5. BIR wiring
238
+
239
+ ```bash
240
+ export BIR_AUTH_URL=https://your-basein-service
241
+ export BIR_CONTROL_URL=http://127.0.0.1:53411 # not discovery
242
+ export BIR_CONTROL_TOKEN=...
243
+ export BIR_REPLAY_ALLOW_SERVERS=tau2
244
+ BIR_REPLAY=1 bir-hooks 2>&1 | tee -a results/tau2-audit.log
245
+ ```
246
+
247
+ `BIR_CONTROL_URL` / `BIR_CONTROL_TOKEN` are mandatory, not convenient: the README calls
248
+ them the only reliable channel for an SDK session, and every episode turn is one. Discovery
249
+ across a few hundred harness-spawned processes is how an arm silently degrades to Tier 2.
250
+
251
+ `BIR_REPLAY_ALLOW_SERVERS=tau2` is the whole allowance. Direct execution dispatches steps on
252
+ connections that are already open and never reaches the permission system, so the list is
253
+ the blast radius.
254
+
255
+ ### 5.1 The per-prompt run boundary — the structural mismatch
256
+
257
+ This is the one that will shape the results, and it is worth understanding before the first
258
+ run rather than after.
259
+
260
+ `onPrompt` rolls a **new run on every prompt**. In τ²-bench, every user-simulator turn is a
261
+ prompt. So an episode of six turns is **six runs**, not one, and matching happens per turn
262
+ against a library of per-turn scenarios.
263
+
264
+ What follows from that:
265
+
266
+ - **Turn 1 carries the intent** — "I want to cancel my order" — and is where a match should
267
+ fire. Later turns are mostly parameter supply ("it's W5678") and are short, cheap, and
268
+ poor matching material.
269
+ - **A scenario is a turn's tool chain, not an episode's.** Cross-episode reuse therefore
270
+ happens turn-by-turn, and the savings ledger sums over turns.
271
+ - **Per-run usage watermarking matters more here than anywhere.** Six runs share one
272
+ Claude Code transcript; summing the file gives turn 6 the cost of turns 1–6. The design
273
+ doc's `markTranscriptUsage` / `usageSince` delta is not an optimisation in this setting,
274
+ it is the difference between a real number and a fabricated one. Check `measured: true`.
275
+
276
+ If the results show matching that fires reliably on turn 1 and rarely after, that is not a
277
+ bug — it is the honest shape of the product against a multi-turn benchmark, and it should
278
+ be reported that way.
279
+
280
+ ---
281
+
282
+ ## 6. Running it
283
+
284
+ ```bash
285
+ git clone https://github.com/sierra-research/tau2-bench && cd tau2-bench
286
+ uv sync && cp .env.example .env # user-simulator key goes here
287
+
288
+ # smoke: one task, one trial
289
+ tau2 run --domain retail --agent claude_code --user-llm <model> \
290
+ --num-trials 1 --task-ids 0
291
+
292
+ tau2 view # read the trajectory you just produced
293
+
294
+ # baseline arm, recording only (BIR_REPLAY unset)
295
+ tau2 run --domain retail --agent claude_code --user-llm <model> \
296
+ --num-trials 4 --max-concurrency 4
297
+
298
+ # replay arm, library seeded from the passing runs above
299
+ BIR_REPLAY=1 tau2 run --domain retail --agent claude_code --user-llm <model> \
300
+ --num-trials 4 --max-concurrency 4
301
+ ```
302
+
303
+ Pin the `--user-llm` model and version across all arms. It is the environment; changing it
304
+ changes the task, and a shifted baseline invalidates every comparison.
305
+
306
+ ---
307
+
308
+ ## 7. Verification gates
309
+
310
+ Run these on `--task-ids 0`, in order. Each has a distinct failure signature, and they are
311
+ ordered so that the cheapest and most misleading failures are caught first.
312
+
313
+ | # | Gate | Failure looks like | Means |
314
+ |---|---|---|---|
315
+ | 1 | `mcp_servers` in `system/init` lists `tau2` as connected | score 0, no tool calls | config skipped silently |
316
+ | 2 | `bir doctor` green, `metadata.tier == 1` inside a harness-spawned run | recording exists but thin | Tier 2 — hooks not loading |
317
+ | 3 | The proxy recorded the episode's tool calls | empty step stream on a passing task | traffic bypassed bir |
318
+ | 4 | Reward > 0 on a task the reference solves | *every* task exactly 0.0 | the trajectory carries no tool calls (§3.1) |
319
+ | 5 | No duplicated side effects | reward 0, doubled refund in the env | one environment, executed twice (§4) |
320
+ | 6 | Zero retries across a multi-trial run | ~1 run in 3 dies as "infrastructure error" | the timestamp race (§4.2) |
321
+ | 7 | `Avg Cost/Conversation` is a number, not `n/a` | costs silently absent | an assistant message with `cost=None` |
322
+ | 8 | On the replay arm: DB=1 **and** COMMUNICATE=1 | reward 0, step ledger healthy | **§3.3** — replay produced no response text |
323
+
324
+ **Every one of these fails quietly.** That is the point of the list. Three of them we hit
325
+ for real, and none announced itself:
326
+
327
+ - **Gate 4** produced a clean table reading `reward 0.0000, DB match 0/1` on a task the
328
+ agent had solved *perfectly* — its calls matched the reference exactly. It reads like a
329
+ hard task, not a broken harness.
330
+ - **Gate 6** was absorbed by τ²'s own retry logic, so the run "succeeded" while costing 3×
331
+ and mixing infrastructure errors into what would look like task variance across a sweep.
332
+ - A fourth, not a gate but worth knowing: on Windows, npm's `claude.cmd` shim routes
333
+ arguments through `cmd.exe`, which re-parses them. Any user turn containing `&`, `|`,
334
+ `%`, `^` or a newline arrives truncated, and the agent answers *"your message came
335
+ through empty."* Call `claude.exe` directly and refuse to fall back to the shim.
336
+
337
+ The lesson generalises past this benchmark: a harness bug and a hard task produce the same
338
+ number. Read one transcript per configuration before trusting any aggregate.
339
+
340
+ ---
341
+
342
+ ## 8. What to measure
343
+
344
+ Per arm, per domain: reward (mean), pass^k across trials, `$`/episode and `$`/turn from
345
+ `--output-format json`'s `total_cost_usd`, wall-clock per episode, and turn count.
346
+
347
+ Per replayed turn, from the audit log and the execution ledger: outcome
348
+ (`steered_full` / `diverged` / `not_steered` / `failed`), match similarity against
349
+ `BIR_MIN_STEER_SIMILARITY` (default 0.92), direct vs steered mode, `deriveCostUsd` against
350
+ `sessionCostUsd`, and the per-step verdicts.
351
+
352
+ Two τ-specific splits worth reporting that MCPMark cannot give us:
353
+
354
+ - **Reward decomposed.** DB and COMMUNICATE separately, always. An arm where DB holds and
355
+ COMMUNICATE slips is a precise, fixable finding about what replay omits; the product hides
356
+ it completely.
357
+ - **Match rate by turn index.** Turn 1 versus turns 2..n (§5.1). This is the number that
358
+ says whether calculated replay generalises across a *task family* or only across
359
+ restatements of one task, and it is the question τ-bench was chosen to answer.
360
+
361
+ Then the control from [benchmark.md](benchmark.md) §7 Phase 3, which matters more here than
362
+ anywhere: τ-bench task families differ by *entity*, so a matcher that ignores which order
363
+ id the customer named will replay cheerfully against the wrong order. Build the near-miss
364
+ set by swapping entities between same-shape tasks — same words, different customer — and
365
+ report the **false replay rate**. In a retail domain, that number has an obvious real-world
366
+ reading, which is exactly why it should be published whatever it says.
367
+
368
+ ---
369
+
370
+ ## 9. Known limits
371
+
372
+ - **The MCP wrapper is ours.** τ²-bench has no MCP interface, so the transport under test is
373
+ a shim we wrote. A sceptic can question it; the answer is to publish it and to run gate 3
374
+ and gate 5 in public. MCPMark has no such caveat, which is the other reason it is primary.
375
+ - **The user simulator is an LLM**, so the environment is stochastic across trials. pass^k
376
+ here mixes agent variance with simulator variance. Pin the simulator model and report it
377
+ as a limit rather than pretending the variance is all ours.
378
+ - **Six runs per episode** (§5.1) makes every per-run number a per-turn number. Say so in
379
+ the table headers; a `$`/run figure that a reader takes for `$`/task is off by ~6×.
380
+ - **`banking_knowledge` stays out.** Same reasoning as [benchmark.md](benchmark.md) §5 — it
381
+ is retrieval over a document set, not a repeatable tool chain, and there is nothing there
382
+ for a scenario to be.
383
+ - **Voice mode is out of scope.** Full-duplex needs `FullDuplexAgent` and a realtime
384
+ provider; Claude Code is half-duplex and so is this integration.
385
+
386
+ ---
387
+
388
+ ## Sources
389
+
390
+ - τ²-bench — [repo](https://github.com/sierra-research/tau2-bench) · [agent developer guide](https://github.com/sierra-research/tau2-bench/blob/main/src/tau2/agent/README.md) · [orchestrator guide](https://github.com/sierra-research/tau2-bench/blob/main/src/tau2/orchestrator/README.md) · [evaluation docs](https://github.com/sierra-research/tau2-bench/blob/main/docs/evaluation.md)
391
+ - Papers — [τ²-bench (arXiv:2506.07982)](https://arxiv.org/pdf/2506.07982) · [τ-bench (arXiv:2406.12045)](https://arxiv.org/pdf/2406.12045)
392
+ - [Trajectory visualizer](https://taubench.com/trajectory-visualizer/?model=qwen3.5-397b-a17b-think_sierra_2026-03-02&domain=retail&task=0) · [leaderboard](https://taubench.com/leaderboard/)
393
+ - [Claude Code headless mode](https://code.claude.com/docs/en/headless)
394
+ - This repo — [benchmark.md](benchmark.md) · [calculatedReplay.md](calculatedReplay.md) · [calculatedReplayGuide.md](calculatedReplayGuide.md)
package/package.json ADDED
@@ -0,0 +1,56 @@
1
+ {
2
+ "name": "@basein/runner",
3
+ "version": "0.1.0",
4
+ "description": "A recording MCP proxy: sits between any MCP client and its MCP servers, executes each call on the client's behalf, and records the run as a reusable BaseIn scenario.",
5
+ "type": "module",
6
+ "license": "MIT",
7
+ "repository": {
8
+ "type": "git",
9
+ "url": "git+https://github.com/eran0lavi/BaseInstRunnerMCP.git"
10
+ },
11
+ "publishConfig": {
12
+ "access": "public"
13
+ },
14
+ "engines": {
15
+ "node": ">=20"
16
+ },
17
+ "bin": {
18
+ "bir": "dist/bin/bir.js",
19
+ "bir-proxy": "dist/bin/bir-proxy.js",
20
+ "bir-hooks": "dist/bin/bir-hooks.js",
21
+ "bir-scenario": "dist/bin/bir-scenario.js"
22
+ },
23
+ "main": "dist/index.js",
24
+ "types": "dist/index.d.ts",
25
+ "files": [
26
+ "dist",
27
+ "!dist/**/*.map",
28
+ "README.md",
29
+ "LICENSE",
30
+ "docs",
31
+ "!docs/next.md"
32
+ ],
33
+ "scripts": {
34
+ "build": "tsc -p tsconfig.json",
35
+ "typecheck": "tsc -p tsconfig.test.json --noEmit",
36
+ "pretest": "tsc -p tsconfig.test.json",
37
+ "test": "node test/run-tests.mjs",
38
+ "prepublishOnly": "npm run build",
39
+ "test:smoke": "node test/smoke-replay.mjs"
40
+ },
41
+ "keywords": [
42
+ "mcp",
43
+ "proxy",
44
+ "recorder",
45
+ "claude-code",
46
+ "agent",
47
+ "observability",
48
+ "replay",
49
+ "scenario"
50
+ ],
51
+ "dependencies": {},
52
+ "devDependencies": {
53
+ "@types/node": "^20.14.0",
54
+ "typescript": "^5.6.0"
55
+ }
56
+ }