mandala-computer-mcp 0.4.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/README.md +558 -29
  2. package/dist/api.d.ts +59 -1
  3. package/dist/api.d.ts.map +1 -1
  4. package/dist/api.js +404 -33
  5. package/dist/api.js.map +1 -1
  6. package/dist/artifacts.d.ts +63 -0
  7. package/dist/artifacts.d.ts.map +1 -0
  8. package/dist/artifacts.js +81 -0
  9. package/dist/artifacts.js.map +1 -0
  10. package/dist/cli.d.ts +4 -0
  11. package/dist/cli.d.ts.map +1 -1
  12. package/dist/cli.js +52 -21
  13. package/dist/cli.js.map +1 -1
  14. package/dist/credentials.d.ts +33 -0
  15. package/dist/credentials.d.ts.map +1 -0
  16. package/dist/credentials.js +395 -0
  17. package/dist/credentials.js.map +1 -0
  18. package/dist/errors.d.ts +122 -11
  19. package/dist/errors.d.ts.map +1 -1
  20. package/dist/errors.js +237 -41
  21. package/dist/errors.js.map +1 -1
  22. package/dist/executions.d.ts +43 -0
  23. package/dist/executions.d.ts.map +1 -0
  24. package/dist/executions.js +162 -0
  25. package/dist/executions.js.map +1 -0
  26. package/dist/format.d.ts +33 -8
  27. package/dist/format.d.ts.map +1 -1
  28. package/dist/format.js +107 -12
  29. package/dist/format.js.map +1 -1
  30. package/dist/http.d.ts +42 -0
  31. package/dist/http.d.ts.map +1 -1
  32. package/dist/http.js +383 -3
  33. package/dist/http.js.map +1 -1
  34. package/dist/index.d.ts +5 -4
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +3 -3
  37. package/dist/index.js.map +1 -1
  38. package/dist/paths.d.ts +50 -2
  39. package/dist/paths.d.ts.map +1 -1
  40. package/dist/paths.js +65 -5
  41. package/dist/paths.js.map +1 -1
  42. package/dist/results.d.ts +97 -0
  43. package/dist/results.d.ts.map +1 -0
  44. package/dist/results.js +263 -0
  45. package/dist/results.js.map +1 -0
  46. package/dist/secret-errors.d.ts +59 -0
  47. package/dist/secret-errors.d.ts.map +1 -0
  48. package/dist/secret-errors.js +199 -0
  49. package/dist/secret-errors.js.map +1 -0
  50. package/dist/secret-store.d.ts +115 -0
  51. package/dist/secret-store.d.ts.map +1 -0
  52. package/dist/secret-store.js +105 -0
  53. package/dist/secret-store.js.map +1 -0
  54. package/dist/server.d.ts +3 -2
  55. package/dist/server.d.ts.map +1 -1
  56. package/dist/server.js +60 -9
  57. package/dist/server.js.map +1 -1
  58. package/dist/session.d.ts +6 -1
  59. package/dist/session.d.ts.map +1 -1
  60. package/dist/session.js +1 -1
  61. package/dist/session.js.map +1 -1
  62. package/dist/stdio.d.ts +5 -1
  63. package/dist/stdio.d.ts.map +1 -1
  64. package/dist/stdio.js +10 -4
  65. package/dist/stdio.js.map +1 -1
  66. package/dist/tool-filters.d.ts +38 -0
  67. package/dist/tool-filters.d.ts.map +1 -0
  68. package/dist/tool-filters.js +137 -0
  69. package/dist/tool-filters.js.map +1 -0
  70. package/dist/tools/account.d.ts +3 -0
  71. package/dist/tools/account.d.ts.map +1 -0
  72. package/dist/tools/account.js +122 -0
  73. package/dist/tools/account.js.map +1 -0
  74. package/dist/tools/activities.d.ts +3 -0
  75. package/dist/tools/activities.d.ts.map +1 -0
  76. package/dist/tools/activities.js +140 -0
  77. package/dist/tools/activities.js.map +1 -0
  78. package/dist/tools/agent.d.ts.map +1 -1
  79. package/dist/tools/agent.js +21 -9
  80. package/dist/tools/agent.js.map +1 -1
  81. package/dist/tools/artifacts.d.ts +3 -0
  82. package/dist/tools/artifacts.d.ts.map +1 -0
  83. package/dist/tools/artifacts.js +97 -0
  84. package/dist/tools/artifacts.js.map +1 -0
  85. package/dist/tools/chat.d.ts +6 -0
  86. package/dist/tools/chat.d.ts.map +1 -0
  87. package/dist/tools/chat.js +192 -0
  88. package/dist/tools/chat.js.map +1 -0
  89. package/dist/tools/computers.d.ts.map +1 -1
  90. package/dist/tools/computers.js +28 -16
  91. package/dist/tools/computers.js.map +1 -1
  92. package/dist/tools/directory.d.ts +13 -0
  93. package/dist/tools/directory.d.ts.map +1 -0
  94. package/dist/tools/directory.js +88 -0
  95. package/dist/tools/directory.js.map +1 -0
  96. package/dist/tools/executions.d.ts +3 -0
  97. package/dist/tools/executions.d.ts.map +1 -0
  98. package/dist/tools/executions.js +87 -0
  99. package/dist/tools/executions.js.map +1 -0
  100. package/dist/tools/guest.d.ts.map +1 -1
  101. package/dist/tools/guest.js +184 -35
  102. package/dist/tools/guest.js.map +1 -1
  103. package/dist/tools/input.d.ts +2 -0
  104. package/dist/tools/input.d.ts.map +1 -1
  105. package/dist/tools/input.js +31 -5
  106. package/dist/tools/input.js.map +1 -1
  107. package/dist/tools/results.d.ts +30 -0
  108. package/dist/tools/results.d.ts.map +1 -0
  109. package/dist/tools/results.js +106 -0
  110. package/dist/tools/results.js.map +1 -0
  111. package/dist/tools/secrets.d.ts +78 -0
  112. package/dist/tools/secrets.d.ts.map +1 -0
  113. package/dist/tools/secrets.js +448 -0
  114. package/dist/tools/secrets.js.map +1 -0
  115. package/dist/tools/signals.d.ts +3 -0
  116. package/dist/tools/signals.d.ts.map +1 -0
  117. package/dist/tools/signals.js +116 -0
  118. package/dist/tools/signals.js.map +1 -0
  119. package/dist/tools/snapshots.d.ts.map +1 -1
  120. package/dist/tools/snapshots.js +52 -10
  121. package/dist/tools/snapshots.js.map +1 -1
  122. package/dist/tools/ssh.d.ts +3 -0
  123. package/dist/tools/ssh.d.ts.map +1 -0
  124. package/dist/tools/ssh.js +186 -0
  125. package/dist/tools/ssh.js.map +1 -0
  126. package/dist/tools/webhooks.js +1 -1
  127. package/dist/tools/webhooks.js.map +1 -1
  128. package/package.json +1 -1
package/README.md CHANGED
@@ -12,17 +12,54 @@ model looks at the screen and clicks what it sees.
12
12
 
13
13
  ## Install
14
14
 
15
- You need an API key from the dashboard — **Settings → API keys**, a `com_…`
16
- string. It is scoped to your account and it is every computer on it, so treat it
17
- the way you would treat a password.
15
+ MCP requires Node 20.3 or newer. The **TypeScript CLI** login setup requires
16
+ Node 22 or newer (`mandala-computer` on npm). Sign in once, then add the MCP
17
+ server:
18
18
 
19
- Node 20.3 or newer. There is nothing else to install: `npx` fetches the server
20
- the first time a client starts it.
19
+ ```sh
20
+ npm install -g mandala-computer
21
+ mandala login
22
+ claude mcp add mandala -- npx -y mandala-computer-mcp
23
+ ```
24
+
25
+ On Node 20, MCP can use an existing saved profile or an environment key.
26
+ The TypeScript CLI guides you through browser approval and saves an API key in
27
+ `~/.mandala/credentials.json`. MCP only reads that file; it never issues or
28
+ approves a device, writes credentials, or starts login automatically. The
29
+ Python distribution also provides a `mandala` command; use the TypeScript
30
+ CLI for `login`.
31
+
32
+ To select a separate profile and workspace:
33
+
34
+ ```sh
35
+ mandala login --profile Work --workspace Research
36
+ claude mcp add mandala -- npx -y mandala-computer-mcp --profile Work
37
+ ```
38
+
39
+ `--profile` takes precedence over `MANDALA_PROFILE`, then the saved default.
40
+ Names are case-sensitive. An explicit key or nonempty `MANDALA_API_KEY` takes
41
+ precedence over every profile and avoids accessing the credential store.
42
+ Empty explicit keys fail; an empty or whitespace-only environment key is absent.
43
+ You can still use a key from **Settings → API keys** with `MANDALA_API_KEY`.
44
+
45
+ Saved credentials require POSIX protection: a real directory owned by you with
46
+ mode `0700`, and an owned regular file with mode `0600` and one link. Symlinks,
47
+ hardlinks, unsafe permissions, malformed files and unsupported protection are
48
+ refused before any API request. Windows file loading is unsupported; explicit
49
+ or environment keys remain available. Only your home directory's store is read.
50
+
51
+ A saved profile also binds its API base URL. `--base-url` or `MANDALA_BASE_URL`
52
+ must match that stored base after canonicalization, including the entire path
53
+ prefix and port. An explicitly empty base flag is invalid with saved credentials.
54
+ Each stdio session resolves once: restart to pick up another saved key. Revoking
55
+ the key in **Settings → API keys** makes later calls fail; MCP does not switch
56
+ profiles, reread the file, or retry a refused action. Run login explicitly when
57
+ new credentials are needed.
21
58
 
22
59
  **Claude Code** — as a plugin, which installs the server and a skill together:
23
60
 
24
61
  ```sh
25
- export MANDALA_API_KEY=com_… # in the shell Claude Code starts from
62
+ # Run mandala login first, or export MANDALA_API_KEY in this shell.
26
63
  /plugin marketplace add mandalacomputer/mcp
27
64
  /plugin install mandala-computer@mandala
28
65
  ```
@@ -34,9 +71,9 @@ right answer at all, that it costs money until it is suspended or stopped, that
34
71
  which refusals are worth a second try. It is a description of *when and how*,
35
72
  not a second client; once the server is installed it stays out of the way.
36
73
  `MANDALA_MODEL_KEY`, if exported alongside, is passed through and turns on
37
- `run_agent`.
74
+ `run_agent` when the configured filters permit it.
38
75
 
39
- Or the server on its own, with the key inline:
76
+ To use an environment key instead of a saved profile:
40
77
 
41
78
  ```sh
42
79
  claude mcp add mandala -e MANDALA_API_KEY=com_… -- npx -y mandala-computer-mcp
@@ -49,8 +86,7 @@ claude mcp add mandala -e MANDALA_API_KEY=com_… -- npx -y mandala-computer-mcp
49
86
  "mcpServers": {
50
87
  "mandala": {
51
88
  "command": "npx",
52
- "args": ["-y", "mandala-computer-mcp"],
53
- "env": { "MANDALA_API_KEY": "com_…" }
89
+ "args": ["-y", "mandala-computer-mcp"]
54
90
  }
55
91
  }
56
92
  }
@@ -59,8 +95,11 @@ claude mcp add mandala -e MANDALA_API_KEY=com_… -- npx -y mandala-computer-mcp
59
95
  **Cursor, Windsurf and the rest** take the same three fields — `command`,
60
96
  `args`, `env` — in whichever file they keep their MCP servers in.
61
97
 
62
- Nothing is hosted and nothing is operated: your MCP client starts this as a
63
- subprocess, and it talks to `https://app.mandala.computer/api/v1` with your key.
98
+ Your MCP client starts this as a subprocess. It uses the saved profile’s API
99
+ base, or `https://app.mandala.computer/api/v1` by default with an environment key.
100
+ Programmatic local hosts can call `runStdio({ profile: 'Work' })`; the exported
101
+ `StdioConfig` allows a local key or profile. `createServer` still requires an
102
+ explicit API key, and `HttpConfig` has no local credential options.
64
103
 
65
104
  ## Use
66
105
 
@@ -95,6 +134,12 @@ entirely.
95
134
 
96
135
  ## The tools
97
136
 
137
+ This is the unfiltered inventory. The filters below can withhold tools from
138
+ both listing and calling; workflows in this README apply only when the needed
139
+ tools are available. The offline fixture checks exercised operation coverage;
140
+ verification against the live publication is a separate required CI gate.
141
+ Parameter and response-mode support remains a separate contract.
142
+
98
143
  **Choosing a machine** — `list_templates`, `list_sizes`, `list_computers`, `get_computer`,
99
144
  `use_computer`, `wait_for_computer`, `get_desktop_url`
100
145
 
@@ -105,9 +150,26 @@ entirely.
105
150
  **Driving the desktop** — `screenshot`, `click`, `type_text`, `press_key`,
106
151
  `scroll`, `drag`, `move_mouse`, `mouse_button`, `cursor_position`, `wait`
107
152
 
108
- **Inside the guest** — `exec`, `exec_poll`, `exec_kill`, `open_url`,
153
+ **Inside the guest** — `exec`, `exec_poll`, `exec_kill`, `get_execution`,
154
+ `read_execution_output`, `open_url`,
109
155
  `list_windows`, `window_action`, `read_clipboard`, `write_clipboard`,
110
- `read_file`, `write_file`
156
+ `read_file`, `write_file`, `list_directory`
157
+
158
+ `write_file` replaces a file already at the path. With `overwrite: false` it
159
+ creates the file only if nothing is there, and a path that is taken is refused
160
+ without that attempt writing anything (Linux computers only). Incomplete
161
+ contents are never published at the path, but a failure while publishing or
162
+ answering can leave the complete file there, so after any error read the path
163
+ before retrying or overwriting. `read_file` and `write_file` take `no_wake: true`
164
+ to refuse (409) rather than resume a computer that is not running. For
165
+ embedders, the `Api` raises that refusal as `FileExistsError`, and a
166
+ create-only 409 whose reason could not be read as `CreateOnlyConflictError`,
167
+ which says nothing about the path; `isTransient` is false for both.
168
+
169
+ **Retained versions** — `retain_execution_output`, `get_result`, `read_result_output`,
170
+ `delete_result`, `publish_artifact`, `get_artifact`, `read_artifact`, `delete_artifact`
171
+
172
+ **Passive metadata** — `list_activities`, `get_activity`, `get_activity_results`, `read_signals`
111
173
 
112
174
  **Being told rather than asking** — `wait_for_event`, `poll_events`,
113
175
  `wait_for_file_change`
@@ -121,14 +183,131 @@ entirely.
121
183
 
122
184
  **Building one** — `build_template`, `list_builds`, `get_build`, `watch_build`
123
185
 
186
+ **Account quota** — `get_account`
187
+
124
188
  **Spending** — `get_usage`
125
189
 
126
190
  **Being told somewhere else** — `list_webhooks`, `create_webhook`,
127
191
  `get_webhook`, `update_webhook`, `rotate_webhook_secret`, `test_webhook`,
128
192
  `list_webhook_deliveries`, `delete_webhook`
129
193
 
130
- **Delegating** — `run_agent`, registered only when a model key is present:
194
+ **SSH access** — `list_ssh_keys`, `add_ssh_key`, `remove_ssh_key`,
195
+ `get_computer_ssh`, `set_computer_ssh`. Keys belong to the person the API key
196
+ was issued to, not to the account, and are accepted by every computer with SSH
197
+ on, on every account where that person is an owner or member. A
198
+ workspace-scoped key can read keys but not add or remove them.
199
+
200
+ **Secrets** — `list_secrets`, `get_secret`, `create_secret`, `replace_secret`,
201
+ `delete_secret` for the account's secret store, and `get_computer_secrets`,
202
+ `set_computer_secrets` and `create_computer`'s `secrets` for which of them a
203
+ computer receives. A value goes in through `create_secret` or `replace_secret`
204
+ and never comes back out. No route answers one, and a store tool's result
205
+ holds only the decoded documented fields (success) or the status, the `reason`
206
+ word and a sentence of its own (refusal). It never includes the platform's
207
+ response text, and the `Api` keeps none for these routes. The bindings carry
208
+ only secret ids, revisions and names. A computer receives a secret as an environment variable (`env`) or as a
209
+ file under `/run/mandala-secrets/user/files` (`file`). A set replaces the whole
210
+ binding list (`[]` removes every binding) and reaches the guest at the
211
+ computer's next start or restart; send the `version` a read answered to have it
212
+ refused with 409 if the list changed since. A replaced value also reaches a
213
+ running computer: a file binding's file is rewritten in place, and on an image
214
+ that supports it an env binding reaches new shells and `exec` with
215
+ `desktop: true` (programs already running keep the old value until a restart).
216
+ A command that needs a bound variable should use `desktop: true`. Whether a
217
+ plain root `exec` sees it depends on the platform version. `replace_secret`
218
+ and `delete_secret` need the current `revision_id`, and a stale one is a 409.
219
+ `delete_secret` also needs `confirm: true`: a computer still bound to a deleted
220
+ secret cannot start again until that binding is removed.
221
+
222
+ **Delegating** — `run_agent`, `run_agent_chat`, registered only when a model key is present:
131
223
  `MANDALA_MODEL_KEY` on stdio, or the caller's own `X-Model-Key` header over HTTP.
224
+ Both must also survive the configured filters.
225
+
226
+ ### Tool filters
227
+
228
+ Set `MANDALA_READ_ONLY=1` to register only tools whose existing
229
+ `annotations.readOnlyHint` is exactly `true`. It accepts `1`, `true`, `yes`,
230
+ `on` for on and `0`, `false`, `no`, `off`, empty or unset for off, ignoring
231
+ surrounding whitespace and letter case. Any other value fails at startup.
232
+ Safety is not inferred from a tool's name or HTTP method: `read_file` and
233
+ `cursor_position` can wake and bill a computer, so they are withheld. So are
234
+ mixed read/write tools such as `wait_for_computer` and `snapshot_schedule`.
235
+ `screenshot`, `read_clipboard` and `list_windows` retain their read-only hints.
236
+
237
+ Set `MANDALA_TAGS=input,guest` to select a union of named tool groups. Names
238
+ are **lowercase only**, comma-separated, trimmed and deduplicated. Empty or
239
+ unset means no tag filter; empty comma-separated entries are ignored. An
240
+ unknown nonempty tag fails before stdio connects or HTTP starts listening,
241
+ with an error listing all valid tags.
242
+
243
+ | Tag | Tools |
244
+ | --- | --- |
245
+ | `account` | `get_account` |
246
+ | `computers` | `list_computers`, `get_computer`, `use_computer`, `wait_for_computer`, `get_desktop_url`, `list_sizes` |
247
+ | `lifecycle` | `create_computer`, `start_computer`, `stop_computer`, `suspend_computer`, `restart_computer`, `update_computer`, `clone_computer`, `delete_computer`, `move_computer`, `list_moves` |
248
+ | `input` | `screenshot`, `click`, `type_text`, `press_key`, `scroll`, `drag`, `move_mouse`, `mouse_button`, `cursor_position`, `wait` |
249
+ | `guest` | `exec`, `exec_poll`, `exec_kill`, `open_url`, `list_windows`, `window_action`, `read_clipboard`, `write_clipboard` |
250
+ | `files` | `list_directory`, `read_file`, `write_file`, `wait_for_file_change` |
251
+ | `executions` | `get_execution`, `read_execution_output` |
252
+ | `results` | `get_activity_results`, `retain_execution_output`, `get_result`, `read_result_output`, `delete_result` |
253
+ | `artifacts` | `publish_artifact`, `get_artifact`, `read_artifact`, `delete_artifact` |
254
+ | `snapshots` | All snapshot tools listed above, including `get_retention` |
255
+ | `templates` | `list_templates`, all your-own-template tools and all build tools listed above |
256
+ | `events` | `wait_for_event`, `poll_events`, `wait_for_file_change` |
257
+ | `usage` | `get_usage` |
258
+ | `webhooks` | All webhook tools listed above |
259
+ | `ssh` | All SSH tools listed above |
260
+ | `secrets` | `list_secrets`, `get_secret`, `create_secret`, `replace_secret`, `delete_secret`, `get_computer_secrets`, `set_computer_secrets` |
261
+ | `agent` | `run_agent`, `run_agent_chat` |
262
+ | `activities` | `list_activities`, `get_activity`, `get_activity_results` |
263
+ | `signals` | `read_signals` |
264
+
265
+ The selected tags form a union, then intersect with read-only and the existing
266
+ lifecycle and model-key restrictions. `MANDALA_NO_LIFECYCLE=1` still withholds
267
+ its five tools even when their tags are selected; selecting `agent` cannot
268
+ enable either agent tool without a per-session model key. Both agent tools are
269
+ withheld by read-only, but remain available with lifecycle disabled alone.
270
+ `MANDALA_TAGS=files MANDALA_READ_ONLY=1` exposes only `list_directory`.
271
+ An `agent` selection with read-only is a valid empty tool list.
272
+ `activities,results` includes `get_activity_results` once; `signals` is separate
273
+ from the active `events` socket. Filters never grant API privileges: the API
274
+ checks current member/owner and workspace authorization on every request.
275
+
276
+ `use_computer` is not read-only. When it is withheld, pass `computer_id`
277
+ explicitly, or bind a computer with `MANDALA_COMPUTER_ID` at stdio startup.
278
+ HTTP callers must supply their own computer selection. These variables apply
279
+ to both transports and the plugin forwards them. Embedders can pass
280
+ `readOnly: true` and `tags: ['input', 'guest']` in `ServerConfig`; the server
281
+ does not read the environment itself.
282
+
283
+ ### Current account quota
284
+
285
+ `get_account` takes no arguments and reads `GET /api/v1/account` once with the
286
+ caller's account credential. Viewer or stronger access is required. It reports
287
+ account-wide aggregates, including for a workspace-scoped key, without resource
288
+ identities. It needs no selected computer or model key, opens no event stream,
289
+ and remains available with read-only and no-lifecycle filters. Select the
290
+ `account` tag to expose it on its own.
291
+
292
+ The report includes the effective plan, pool ceilings, per-computer maxima,
293
+ Windows capability, current consumption and remaining quota. Configured CPU and
294
+ disk include kept computers regardless of power state; running or reserved RAM
295
+ includes current reservations. CPU, MB, GB and snapshot bytes retain the API's
296
+ units. `get_usage` separately reports historical metered consumption over time.
297
+
298
+ Quota is **advisory**: `observed_at` is an observation time, not a reservation or
299
+ consistency token. Later create, resize, start or snapshot requests can still be
300
+ refused, and existing 402 messages remain unchanged. Snapshot headroom is against
301
+ **indexed stored bytes**; it does not include in-flight capture reservations and
302
+ does not establish that a new capture will fit.
303
+
304
+ `complete.computers` and `complete.snapshots` are independent. An incomplete group
305
+ has `null` for all its consumption and remaining figures, meaning **unknown**,
306
+ while the other group can retain numeric values. Complete zero usage and zero
307
+ remaining quota remain numeric zeros. Verified plan ceilings remain available in
308
+ a partial report, including a no-plan account that still has retained usage.
309
+ Malformed reports and HTTP failures remain errors, never an empty account.
310
+ The tool prints unknown and advisory guidance before the projected public fields.
132
311
 
133
312
  ## Things worth knowing
134
313
 
@@ -357,6 +536,107 @@ foreground comes back as a timeout, with the work still going inside the guest
357
536
  and its output unreadable. With a handle you get the exit code and the output,
358
537
  and `exec_kill` stops it.
359
538
 
539
+ **Stable background reads.** When an accepted `exec` returns `execution_id`,
540
+ use `get_execution` for its last observed `running`, `exited` or `lost` state.
541
+ Only `exited` carries an `ended_at` and signed `exit_code`; `running` does not
542
+ prove the computer is awake, and `lost` establishes neither success nor failure.
543
+ Older replies can omit the ID. A malformed supplied ID leaves the accepted
544
+ command and its valid PID usable, but supplies no stable identity: do not replay
545
+ the command to obtain one. PID polls/kills never reconstruct this association.
546
+
547
+ `read_execution_output` takes that ID and **both** `stdout_offset` and
548
+ `stderr_offset` byte positions. Start each reader at zero, then pass its own
549
+ returned positions. For example:
550
+
551
+ ```json
552
+ {"execution_id":"exec_0123456789abcdef0123456789abcdef","stdout_offset":0,"stderr_offset":0,"limit":4096}
553
+ ```
554
+
555
+ Each stream is bounded to 4,096 bytes by default, at most 16,384. Complete
556
+ lossless UTF-8 appears as `stdout`/`stderr` with its BOM preserved. NUL, binary
557
+ and split UTF-8 chunks remain exact canonical `stdout_b64`/`stderr_b64`; nothing
558
+ is trimmed or replaced. `stdout_more` and `stderr_more` are independent, and
559
+ false means EOF at that moment, not that the command has finished.
560
+
561
+ The separate `diagnostic`/`diagnostic_b64` repeats on every read. At most 4,096
562
+ of its up-to-65,536 available bytes are displayed: inspect
563
+ `diagnostic_available_bytes`, `diagnostic_displayed_bytes`, and
564
+ `diagnostic_display_truncated`. The independent `diagnostic_truncated` flag is
565
+ the platform's capture limitation. Diagnostics never advance either cursor.
566
+ Neither new tool consumes output from another reader or the shared `exec_poll`
567
+ cursor, and both carry read-only, non-destructive, idempotent annotations.
568
+
569
+ These are single requests with cancellation, no automatic resume, retry,
570
+ watcher, capture or command replay. Metadata reads no guest files. Output reads
571
+ perform guest I/O without refreshing activity and are unsuitable for passive
572
+ Activities/history. Files are mutable guest data, not retained artifacts;
573
+ handles can vanish on restart, replacement or cleanup, and observed exits
574
+ expire after ten minutes. An unavailable read is an error, not empty output.
575
+ Every new-tool result is bounded to 256 KiB of serialized data.
576
+
577
+ **Explicit immutable retained versions.** `retain_execution_output` accepts an
578
+ `execution_id` and captures one version of its volatile output with one POST.
579
+ It performs guest I/O, without resuming or replaying the command. Its optional
580
+ `max_bytes_per_stream` defaults to 1 MiB (maximum 4 MiB); `retention_seconds`
581
+ defaults to 86400 (maximum 604800). Diagnostics are separate, up to 64 KiB.
582
+ Each capture creates a version; there is no automatic capture or retry.
583
+
584
+ `get_result` reads finite metadata by `result_id`. `read_result_output` requires
585
+ `result_id`, `stream` (`stdout`, `stderr` or `diagnostic`) and `offset`, with
586
+ `limit` defaulting to 4096 and capped at 16384 bytes. It reads one independent
587
+ page, with exact `offset`, `next_offset`, `eof` and returned `bytes` count.
588
+ Content is lossless `text` (BOM preserved) or canonical `base64` for binary,
589
+ controls and split UTF-8. EOF means the end of this retained prefix, not task
590
+ completion. A page alone does not verify the full manifest hash.
591
+ `delete_result` deletes that version once; a repeated 404 remains unavailable.
592
+
593
+ Existing synchronous `exec` accepts `retain_output: true` or a strict object
594
+ with those same two options. False or absence leaves default behavior alone.
595
+ It cannot be combined with `background: true`. A canonical returned `result_id`
596
+ confirms optional retention; no execution ID is fabricated. Missing, malformed
597
+ or unsupported optional metadata leaves the command outcome unchanged and does
598
+ not authorize replay. Retained-prefix truncation and upstream response
599
+ truncation are distinct. Synchronous results have no diagnostic stream; an
600
+ explicit diagnostic read can return 409.
601
+
602
+ **Nominated file versions.** `publish_artifact` requires an absolute `path`,
603
+ `expected_size` and `expected_sha256` supplied by the caller. For example:
604
+
605
+ ```json
606
+ {"path":"/tmp/empty.txt","expected_size":0,"expected_sha256":"e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"}
607
+ ```
608
+
609
+ Paths preserve legal Unicode, spaces and Linux backslashes; Windows drive and
610
+ UNC paths are also accepted, subject to the platform's current OS proof.
611
+ Publication performs one nominated guest-file capture, without a client-side
612
+ stat, list, read, hash or exec preflight. `max_bytes` defaults to 8 MiB and is at
613
+ most 64 MiB; expected size must fit. Retention has the bounds above. Optional
614
+ `execution_id` records verified **caller selection**, not proof the execution
615
+ created the file. Every publication creates an independent immutable version.
616
+
617
+ `get_artifact` reads metadata. `read_artifact` first reads that metadata and,
618
+ only if the complete size fits `max_bytes` (default 4096, maximum 16384), reads
619
+ the entire retained object and verifies its exact length and SHA-256. Over-cap
620
+ objects return metadata with a refusal before any content request; use an SDK
621
+ whole download with an adequate cap. There are no partial artifacts, Range
622
+ requests, local destinations, filenames, image/HTML previews or guest fallbacks.
623
+ Verified content uses the same lossless `text`/`base64` presentation.
624
+ `delete_artifact` deletes only that stored version, not the guest file; repeat
625
+ 404s remain truthful failures.
626
+
627
+ All eight retained tools resolve the computer selection once. Reads are passive
628
+ but require current authorization and scope availability; retained bytes can be
629
+ unavailable when the host cannot verify access, after expiry or deletion, or
630
+ when storage is unavailable. They never wake a guest or substitute live output.
631
+ New operations have one 90-second budget across headers, bodies, and both
632
+ artifact requests. An MCP client can impose an earlier deadline. Cancellation,
633
+ redirects and malformed responses never trigger a retry; a lost publication
634
+ response means commitment is **unconfirmed**, not undone. Each entire serialized
635
+ retained tool result is bounded to 256 KiB. Reads are marked read-only;
636
+ capture/publication create new versions; deletes are destructive with an
637
+ idempotent deletion effect. Activities result detail remains outside this MCP
638
+ runtime.
639
+
360
640
  Past about **two minutes** it does not even come back as a timeout. A proxy in
361
641
  front of the platform abandons a request that has produced no response for
362
642
  roughly that long and answers 524, which arrives as `GatewayTimeoutError` —
@@ -457,6 +737,85 @@ about storage instead.
457
737
  guest agent answers 409. The platform's own error messages come through
458
738
  unedited, because they are written to be acted on.
459
739
 
740
+ HTTP failures preserve their actual response status. An unsupported method on a
741
+ known path is `MethodNotAllowedError` (405), with the received `Allow` value when
742
+ available. A missing computer, snapshot, route or guest file remains
743
+ `NotFoundError` (404); other guest failures keep their own status and message.
744
+ Neither response causes an automatic retry or method switch.
745
+
746
+ `isTransient(err)` answers "is this worth sending again unchanged". A `503` is
747
+ transient only for a GET or HEAD. Any change answered `503` may or may not have
748
+ happened, because the platform answers a failure after the request was sent the
749
+ same way. So `isTransient` is false for it whatever `reason` it carries, the
750
+ tool's answer says to read the current state first, and `error.method` records
751
+ the method. A `reason` this version does not
752
+ know, such as a new word, is treated as no classification. `reasonKind` returns
753
+ `undefined` for it, and the status decides.
754
+
755
+ The secret store has typed calls on `api.secrets`: `list`, `create`, `get`,
756
+ `replace` and `delete`. Each answer is decoded strictly to the documented
757
+ fields, and none of them carries a value:
758
+
759
+ ```ts
760
+ const { secrets } = await api.secrets.list();
761
+ const s = await api.secrets.create({ name: 'OPENAI_API_KEY', value: key });
762
+ await api.secrets.replace(s.id, { value: next, revisionId: s.revision_id });
763
+ ```
764
+
765
+ Embedders can inspect optional diagnostics on every `APIError`:
766
+
767
+ ```ts
768
+ import { Api, APIError, MethodNotAllowedError } from 'mandala-computer-mcp';
769
+
770
+ const api = new Api(process.env.MANDALA_API_KEY!);
771
+ try {
772
+ await api.json('GET', 'account');
773
+ } catch (error) {
774
+ if (error instanceof APIError) {
775
+ console.error({ status: error.status, requestId: error.requestId,
776
+ reason: error.reason, allow: error.allow,
777
+ wwwAuthenticate: error.wwwAuthenticate });
778
+ if (error instanceof MethodNotAllowedError) {
779
+ // Inspect error.allow before correcting the request method.
780
+ }
781
+ }
782
+ }
783
+ ```
784
+
785
+ `requestId` uses a nonblank `X-Request-ID` header first, then the top-level
786
+ `request_id` body field. It is an opaque diagnostic, never an idempotency key.
787
+ The raw body remains available on `error.body`, including any differing body ID
788
+ and nested chat accounting. `allow` and `wwwAuthenticate` come only from received
789
+ headers. HEAD failures can carry these fields with no body. Older servers,
790
+ intermediaries and connection failures may supply none of them. Existing
791
+ constructor arguments retain their meanings; an optional trailing
792
+ `APIErrorMetadata` object adds these three fields, and `method`.
793
+
794
+ MCP error results include supplied `reason`, `request_id`, `allow`,
795
+ `www_authenticate` and `retry_after_ms` as labelled JSON metadata. Diagnostic
796
+ strings are limited to 128, 256, 512 and 512 characters respectively; an
797
+ oversized field is omitted with a notice, so a shortened Allow is never
798
+ presented as complete. Tool-specific warnings about partial work, retained
799
+ publication, execution reads and explicit template continuation still apply.
800
+ Serialized JSON object or array prefixes are not displayed as error prose.
801
+ Valid scalar error messages retain their wording; embedders still have the
802
+ original `APIError.message` and `APIError.body` for diagnostics. Native agent
803
+ error frames retain their supplied numeric status even without a reason or
804
+ request ID, independently of the successful HTTP stream carrying them.
805
+
806
+ For a 401, `missing` means a platform credential was not supplied; `invalid`
807
+ means the supplied platform credential was not accepted; `revoked` means its
808
+ authority no longer holds. Unknown reasons stay visible. An unclassified 401
809
+ alone does not identify whether the account key or model key was refused,
810
+ including a nested chat failure. A 403 remains a permission or authority
811
+ refusal. Inspect recorded work before another run; no key fallback, login or
812
+ automatic replay is performed. Nested error reasons and in-band stream errors
813
+ do not grant permission to replay a partly completed run.
814
+
815
+ For desktop events, use the exact returned `events_url`, including its desktop
816
+ capability, in a WebSocket client. A REST Bearer key alone is not sufficient;
817
+ the HTTP events response provides JSON guidance rather than another login flow.
818
+
460
819
  **Desktop links are credentials.** `get_desktop_url` returns the watch-only URL
461
820
  by default — the platform drops input on that socket, so it is safe to hand to
462
821
  somebody. `control: true` returns the full-control one, which is root-equivalent
@@ -558,7 +917,9 @@ for whatever is checking that the process is up.
558
917
  bearer token and is used only for their session; there is no store, and nothing
559
918
  outlives a session but a digest of the key — kept so that a later request can be
560
919
  shown to come from the same holder, which means a leaked session id on its own
561
- is not enough to drive somebody else's desktop.
920
+ is not enough to drive somebody else's desktop. HTTP startup and requests never
921
+ read the operator's credential store, `MANDALA_API_KEY` or `MANDALA_PROFILE`.
922
+ `--profile` is local-only and ignored with `--http`.
562
923
 
563
924
  That is also why anyone can run their own: point the same container at the same
564
925
  API and it works, with no secret to provision.
@@ -577,21 +938,89 @@ not list it, and every request is refused with a 403. Set
577
938
  is in force, so a `403` on a deployment that worked before has a line above it
578
939
  naming the fix.
579
940
 
941
+ ### Hosted, with OAuth
942
+
943
+ The endpoint at `https://app.mandala.computer/mcp` is this server run with
944
+ `--http` behind the platform's own proxy, with the platform as its OAuth 2.1
945
+ authorization server. Nobody pastes a key:
946
+
947
+ ```sh
948
+ claude mcp add --transport http mandala https://app.mandala.computer/mcp
949
+ ```
950
+
951
+ Two variables make an HTTP server behave that way, and without them nothing
952
+ changes — self-hosters keep bringing API keys:
953
+
954
+ ```sh
955
+ MANDALA_MCP_RESOURCE_METADATA_URL=https://app.mandala.computer/.well-known/oauth-protected-resource/mcp \
956
+ MANDALA_MCP_SERVICE_SECRET=… \
957
+ MANDALA_ALLOWED_HOSTS=app.mandala.computer \
958
+ npx mandala-computer-mcp --http --host 127.0.0.1 --port 3000
959
+ ```
960
+
961
+ In this mode `X-Forwarded-For` is believed from a loopback peer only, so the
962
+ proxy in front must be on the same machine and must set it to the real client
963
+ address.
964
+
965
+ - **Every `/mcp` request needs a bearer**, initialize and `tools/list`
966
+ included. One without is answered `401` with exactly
967
+ `WWW-Authenticate: Bearer resource_metadata="<that URL>", scope="mcp:tools"`,
968
+ which is how a client finds where to authorize. `/healthz` stays open.
969
+ - **The bearer is passed through unchanged** — an `mcpat_…` access token, or an
970
+ API key, which the platform still accepts. The access token is good for this
971
+ endpoint only. It is not an API key: sent to the API directly, the platform
972
+ refuses it with `401` (`reason: "invalid"`). A person revokes it under
973
+ Settings → Connected apps.
974
+ - **A bearer is checked with the platform before it gets anything.** An
975
+ initialize makes no session until the platform accepts its token — a `2xx`
976
+ from `GET ssh-keys`, which every valid credential gets — so invented tokens
977
+ cannot fill the session pool. A `401` is the challenge above; any other
978
+ answer is `503` and nothing is remembered. Refused initializes are budgeted
979
+ per source address, 20 a minute. Past that, a token already accepted still
980
+ passes and a new one is still checked, but only one every 5 s per address;
981
+ the rest get `429` with `Retry-After`. The budget comes back when its minute
982
+ is up. A POST
983
+ carrying a request is checked again before it is dispatched, with an
984
+ acceptance cached for 60 s under the token's digest, so an expired or
985
+ revoked token is a clean `401` before any stream opens.
986
+ - **A token the platform refuses during a call comes back as that same
987
+ `401`**, not as a tool error, so the client refreshes or authorizes again —
988
+ the answer is held until its first byte for this. The one case that cannot
989
+ work: a token that dies mid-call AFTER the stream has committed (the SDK's
990
+ 15 s keep-alive, a progress notification or a partial result). That call
991
+ ends as a tool error, the session remembers the refusal, and the next
992
+ request on that token gets the `401` without reaching the platform.
993
+ - **A refreshed token starts a new session.** A session is bound to the digest
994
+ of the bearer that opened it, and a different bearer on it is answered
995
+ `404 Unknown session`, the MCP spec's signal to initialize again. An access
996
+ token says nothing this server can check about which grant it came from, so
997
+ rebinding would let any valid credential that learned a session id take over
998
+ its bound computer, buffered events and retained results.
999
+ - `X-Mandala-MCP-Service` carries `MANDALA_MCP_SERVICE_SECRET` on every
1000
+ platform request, only ever to `MANDALA_BASE_URL`, so a token cannot be
1001
+ replayed at the API directly by the app it was issued to. A client's own
1002
+ header of that name is never forwarded.
1003
+
580
1004
  ### Configuration
581
1005
 
582
1006
  | Variable | Meaning |
583
1007
  | --- | --- |
584
- | `MANDALA_API_KEY` | `com_…` from Settings → API keys. Required on stdio; over HTTP each caller sends their own. |
585
- | `MANDALA_BASE_URL` | Defaults to `https://app.mandala.computer/api/v1`. |
1008
+ | `MANDALA_API_KEY` | Optional local API key, taking precedence over saved credentials. Ignored by HTTP; each caller sends their own bearer token. |
1009
+ | `MANDALA_PROFILE` | Local saved profile; `--profile` overrides it. Otherwise the file default is selected. Ignored by HTTP. |
1010
+ | `MANDALA_BASE_URL` | With a saved profile, must match its stored base. Otherwise defaults to `https://app.mandala.computer/api/v1`. |
586
1011
  | `MANDALA_COMPUTER_ID` | Bind a computer at startup, so `use_computer` is not needed. **stdio only** — under `--http` it is ignored rather than bound into every caller's session, since it names a machine on the operator's account. |
587
- | `MANDALA_MODEL_KEY` | An Anthropic key. Enables `run_agent`, which runs the platform's own loop on that key. **stdio only** — under `--http` each caller sends their own as `X-Model-Key`, and this variable is ignored. |
1012
+ | `MANDALA_MODEL_KEY` | An Anthropic key. Enables `run_agent` and `run_agent_chat` when filters permit them, which runs the platform's own loop on that key. **stdio only** — under `--http` each caller sends their own as `X-Model-Key`, and this variable is ignored. |
588
1013
  | `MANDALA_NO_LIFECYCLE` | `1`, `true`, `yes` or `on` withholds `create_computer`, `clone_computer`, `clone_snapshot`, `delete_computer` and `delete_snapshot` — every tool that makes a computer or destroys one. `0`, `false`, `no`, `off` or unset leaves them registered. Any other value is **refused at startup** rather than read as off: a typo here would otherwise leave those tools in place on a server whose operator believes they are gone. The `--no-lifecycle` flag reads the same vocabulary and refuses the same way, except that it has no spelling for *unset*: `--no-lifecycle=` is refused rather than ignored, so a launcher template whose variable did not expand stops instead of quietly leaving the tools registered. |
589
1014
  | `PORT`, `HOST` | For `--http`. Default `3000`, `127.0.0.1`. |
1015
+ | `MANDALA_READ_ONLY` | Keep only tools annotated `readOnlyHint: true`; strict boolean parsing as described under Tool filters. |
1016
+ | `MANDALA_TAGS` | Comma-separated lowercase tool tags; see Tool filters for the inventory and intersection rules. |
590
1017
  | `MANDALA_ALLOWED_HOSTS`, `MANDALA_ALLOWED_ORIGINS` | Comma-separated. Which `Host` and `Origin` values this server answers to. On a loopback bind the host list defaults to the address it was given, so DNS-rebinding protection is on without configuration; set this when serving under a name. |
1018
+ | `MANDALA_MCP_RESOURCE_METADATA_URL` | `--http` only. The OAuth protected-resource metadata URL this server is published under. Set, it answers OAuth clients as described under Hosted, with OAuth; unset, callers bring an API key. |
1019
+ | `MANDALA_MCP_SERVICE_SECRET` | `--http` only. Sent as `X-Mandala-MCP-Service` on every platform request, to `MANDALA_BASE_URL` only. Never logged. |
591
1020
 
592
- Every one of these but the model key has a flag as well, and a flag overrides
1021
+ Every one of these but the model key, the two tool filters and the two OAuth settings has a flag as well, and a flag overrides
593
1022
  the environment: `--http`, `--port`, `--host`, `--base-url`, `--computer`,
594
- `--allowed-hosts`, `--allowed-origins`, `--no-lifecycle`, plus `--help` and
1023
+ `--allowed-hosts`, `--allowed-origins`, `--no-lifecycle`, local `--profile`, plus `--help` and
595
1024
  `--version`. `--key` exists for a caller launching several servers under
596
1025
  different keys, and warns when used, because an argument vector is readable by
597
1026
  `ps`, lands in shell history and is recorded verbatim by any exec audit —
@@ -603,17 +1032,116 @@ it when a stretch of pixel work would otherwise cost the calling model a
603
1032
  screenshot per step — ten clicks stop being ten images. It bills your Anthropic
604
1033
  key, and the platform never stores that key.
605
1034
 
1035
+ ### Passive directories, activity history and signals
1036
+
1037
+ `list_directory` takes an exact absolute guest `path`. Unicode, spaces and
1038
+ punctuation survive query encoding. It requires an already running computer,
1039
+ does not resume it or extend its idle timer, and still contacts the guest with
1040
+ ordinary rate/capacity admission. The result preserves `path`, `entries` with
1041
+ `name`, `type` and optional `size_bytes`, `truncated` and `skipped`. An unavailable
1042
+ entry has no inferred type or zero size. Symlinks are not followed, and a final
1043
+ symlink directory is refused. No file content is read. A partial listing is an
1044
+ unordered subset: at most 512 examined entries/128 KiB, with no continuation
1045
+ token. Narrow the path when names are omitted; do not automatically rescan.
1046
+
1047
+ `list_activities` returns one newest-first history page with a fixed watermark.
1048
+ Pass an opaque `cursor` for continuation, or `changes:true` with a cursor for
1049
+ late final updates to older rows. False/absent `changes` is omitted on the wire;
1050
+ there is no `limit` argument. Preserve `next_cursor`, `changes_cursor`, `gap`
1051
+ and health fields. A gap requires refreshing history. The tools keep no cursor
1052
+ cache and never drain pages automatically. `get_activity` takes an `activity_id`
1053
+ shaped as `act_` plus 32 lowercase hex digits, a request identity rather than an
1054
+ idempotency key. Recorded state, scope, revision, timestamps, execution identity
1055
+ and status remain distinct: accepted background work is not finished execution,
1056
+ and a dispatched error may have had effects. History is selected, retained,
1057
+ best-effort metadata, not all guest work or an agent identity.
1058
+
1059
+ `get_activity_results` returns at most eight newest metadata links, preserving
1060
+ revision, availability, association, byte/truncation metadata and observations.
1061
+ `more:true` has no continuation cursor and does not mean all versions were
1062
+ returned. An unavailable item is not an empty available item. Caller-selected
1063
+ artifact association does not prove creation or command success. This tool
1064
+ never follows a link to content, files, captures, downloads or execution polling.
1065
+
1066
+ `read_signals` reads one passive daemon page, independently of the active event
1067
+ socket. Omit `since` or send an empty string for a head-only baseline with no
1068
+ replay. Optional `limit` is 1–100; omitting it uses the API default 50. Preserve
1069
+ the returned `cursor` even when `events` is empty: filtered-out rows may advance
1070
+ it. Retention is ephemeral; expired/restarted/migrated cursors can return an
1071
+ explicit reset gap. Baselines and gaps do not prove there was no earlier work
1072
+ or that a task succeeded. Unsupported (501) and unavailable (503) responses
1073
+ remain errors, never empty history or new checkpoints. There is no watcher,
1074
+ guest action, retry loop or automatic checkpoint cache in this tool.
1075
+
1076
+ ### JSON chat with your own Anthropic key
1077
+
1078
+ `run_agent_chat` drives the selected computer using the same BYOK Anthropic loop
1079
+ as `run_agent`. It accepts a nonempty array of textual OpenAI-shaped `messages`,
1080
+ including string content or arrays of `{type:"text",text:"..."}` parts. The
1081
+ **last user message** supplies the task; system messages supply standing
1082
+ instructions. Earlier user/assistant conversation is not replayed. Optional
1083
+ `model` is sent unchanged and must name an Anthropic model. `max_steps` is 1–100,
1084
+ default 20. This is computer control, not hosted general-purpose inference or a
1085
+ chat UI; it adds no key storage or model billing service.
1086
+
1087
+ This tool deliberately sends `stream:false` and uses the JSON response so the
1088
+ result preserves the underlying `agent.stop`, step count and token usage.
1089
+ Only explicit `end_turn` consistent with the completion's finish reason can
1090
+ report success. Limits, refusal, missing or conflicting terminal fields remain
1091
+ errors with valid partial results. Nested failures preserve the agent computer,
1092
+ recorded steps and native usage, including separate cache-read and cache-write
1093
+ token counts, alongside OpenAI-shaped aggregate usage. Malformed native detail
1094
+ is explicitly marked incomplete. A failed call may include completed, billed
1095
+ work: inspect it before deciding what remains, and do not automatically replay.
1096
+ Neither agent tool starts the computer or switches endpoints after a failure.
1097
+
1098
+ The caller supplies the model key through `MANDALA_MODEL_KEY` on stdio or their
1099
+ own `X-Model-Key` header on HTTP. The account Authorization remains separate;
1100
+ HTTP never falls back to the operator's model key. Waiting heartbeats count
1101
+ notifications, not completed actions. Long-running clients need a
1102
+ `progressToken` plus `resetTimeoutOnProgress`; otherwise choose a smaller
1103
+ `max_steps`. That bound is neither a time cap nor a spend cap.
1104
+
606
1105
  ## Development
607
1106
 
608
1107
  ```sh
609
- npm install
610
- npm test # vitest, plus the surface check below
1108
+ npm ci
1109
+ npx vitest run # deterministic offline tests, including the synthetic contract
1110
+ npm test # offline tests plus the separate private mirror check
611
1111
  npm run build
612
1112
  npm run lint
613
1113
  ```
614
1114
 
615
- CI runs the suite on Node 20, 22, 24 and 26 — the floor `package.json`
616
- declares and the ceiling a current `npx` will actually use.
1115
+ CI runs the offline suite on Node 20, 22, 24 and 26. A separate Node 22
1116
+ `published-openapi` job runs on pull requests and main pushes:
1117
+
1118
+ ```sh
1119
+ npx vitest run --config vitest.openapi.config.ts
1120
+ ```
1121
+
1122
+ It anonymously fetches the fixed publication at
1123
+ `https://app.mandala.computer/api/docs/openapi.json` once, with a 20-second
1124
+ overall deadline and an 8 MiB body ceiling, then exercises the real MCP tools
1125
+ and requires request evidence for every published `/api/v1` operation. Every
1126
+ exercise variant must succeed and dispatch a request, including later variants
1127
+ of a tool that already dispatched successfully.
1128
+ Non-v1 operations are explicitly excluded and reported. The check uses OpenAPI
1129
+ server inheritance and literal-path precedence, not the local route allowlist.
1130
+ Each path is appended to its effective server base; repeated prefixes are not
1131
+ removed. Fully prefixed paths work with an absent or root server.
1132
+ It fails on a blocked fetch, redirect, invalid document or missing operation;
1133
+ there is no fixture fallback, credential requirement or skip branch.
1134
+
1135
+ The committed OpenAPI fixture is synthetic, assembled from the public MCP route
1136
+ inventory, and is never presented as a downloaded publication. Offline green
1137
+ proves deterministic implementation coverage only. Anonymous CI at
1138
+ [commit 1b04314](https://github.com/mandalacomputer/mcp/commit/1b04314e84f5701e2e15c65cbb1c44a0c14a4948)
1139
+ returned HTTP 200 and matched all 56 operations in the fetched v1 contract using
1140
+ 123 requests from 83 tools. That result applies to that commit and publication;
1141
+ every subsequent head must pass the gate independently. These counts are
1142
+ observations, never fixed thresholds or exemptions. A blocked or red public gate
1143
+ must not be merged. Parameters and response modes remain separate contracts,
1144
+ with documented parameter exceptions and JSON-only chat support.
617
1145
 
618
1146
  ### Where the platform's rules live
619
1147
 
@@ -626,10 +1154,11 @@ change to the platform's route table, not a wider pass-through here.
626
1154
 
627
1155
  The platform allowlists every route `/api/v1` will answer and 404s the rest.
628
1156
  `test/allowlist.ts` mirrors that table, and the tests assert two things: that
629
- every call this server can make lands on an allowlisted route, and that the gap
630
- between the platform's surface and this server's coverage is *exactly* the set
631
- written down in `UNIMPLEMENTED`. A route added upstream becomes a failing test
632
- here rather than a feature nobody noticed.
1157
+ every successful exercised call lands on an allowlisted route, and that no
1158
+ mirrored operation remains unexercised (`UNIMPLEMENTED` is empty). Tool names
1159
+ and exercise entries must match in both directions, and every tool must make
1160
+ an observed HTTP request during its callback. The independent public gate also
1161
+ detects a new published operation when the mirror has not yet been updated.
633
1162
 
634
1163
  `npm run check:surface` goes further and diffs the mirror against the platform's
635
1164
  published surface manifest — a file the platform generates from its own tables