mandala-computer-mcp 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of mandala-computer-mcp might be problematic. Click here for more details.

Files changed (66) hide show
  1. package/README.md +102 -10
  2. package/dist/api.d.ts +19 -6
  3. package/dist/api.d.ts.map +1 -1
  4. package/dist/api.js +213 -31
  5. package/dist/api.js.map +1 -1
  6. package/dist/cli.d.ts.map +1 -1
  7. package/dist/cli.js +113 -8
  8. package/dist/cli.js.map +1 -1
  9. package/dist/errors.d.ts +60 -0
  10. package/dist/errors.d.ts.map +1 -1
  11. package/dist/errors.js +71 -2
  12. package/dist/errors.js.map +1 -1
  13. package/dist/events.d.ts +17 -2
  14. package/dist/events.d.ts.map +1 -1
  15. package/dist/events.js +104 -13
  16. package/dist/events.js.map +1 -1
  17. package/dist/format.d.ts +16 -0
  18. package/dist/format.d.ts.map +1 -1
  19. package/dist/format.js +11 -1
  20. package/dist/format.js.map +1 -1
  21. package/dist/http-body.d.ts +17 -0
  22. package/dist/http-body.d.ts.map +1 -0
  23. package/dist/http-body.js +48 -0
  24. package/dist/http-body.js.map +1 -0
  25. package/dist/http.d.ts.map +1 -1
  26. package/dist/http.js +177 -51
  27. package/dist/http.js.map +1 -1
  28. package/dist/index.d.ts +1 -1
  29. package/dist/index.d.ts.map +1 -1
  30. package/dist/index.js +1 -1
  31. package/dist/index.js.map +1 -1
  32. package/dist/paths.d.ts +22 -13
  33. package/dist/paths.d.ts.map +1 -1
  34. package/dist/paths.js +68 -16
  35. package/dist/paths.js.map +1 -1
  36. package/dist/poll.d.ts +103 -0
  37. package/dist/poll.d.ts.map +1 -0
  38. package/dist/poll.js +129 -0
  39. package/dist/poll.js.map +1 -0
  40. package/dist/server.d.ts +1 -1
  41. package/dist/server.js +1 -1
  42. package/dist/tools/agent.d.ts.map +1 -1
  43. package/dist/tools/agent.js +14 -2
  44. package/dist/tools/agent.js.map +1 -1
  45. package/dist/tools/computers.d.ts.map +1 -1
  46. package/dist/tools/computers.js +228 -49
  47. package/dist/tools/computers.js.map +1 -1
  48. package/dist/tools/events.d.ts.map +1 -1
  49. package/dist/tools/events.js +224 -39
  50. package/dist/tools/events.js.map +1 -1
  51. package/dist/tools/guest.d.ts.map +1 -1
  52. package/dist/tools/guest.js +223 -30
  53. package/dist/tools/guest.js.map +1 -1
  54. package/dist/tools/input.d.ts.map +1 -1
  55. package/dist/tools/input.js +29 -3
  56. package/dist/tools/input.js.map +1 -1
  57. package/dist/tools/snapshots.d.ts.map +1 -1
  58. package/dist/tools/snapshots.js +478 -20
  59. package/dist/tools/snapshots.js.map +1 -1
  60. package/dist/tools/templates.d.ts.map +1 -1
  61. package/dist/tools/templates.js +50 -11
  62. package/dist/tools/templates.js.map +1 -1
  63. package/dist/tools/webhooks.d.ts.map +1 -1
  64. package/dist/tools/webhooks.js +116 -17
  65. package/dist/tools/webhooks.js.map +1 -1
  66. package/package.json +1 -1
@@ -1,13 +1,128 @@
1
1
  import { z } from 'zod';
2
- import { NotFoundError } from '../errors.js';
2
+ import { CancelledError, ConflictError, isTransientForPoll, NotFoundError } from '../errors.js';
3
3
  import { describe, guarded, incompleteWarning, json, refused, said, unwrapComputer, withoutCredentials, } from '../format.js';
4
4
  import * as P from '../paths.js';
5
+ import { heartbeat, POLL_MS, pollDelay, sleep } from '../poll.js';
5
6
  const idArg = {
6
7
  computer_id: z
7
8
  .string()
8
9
  .optional()
9
10
  .describe('Which computer. Defaults to the one selected with use_computer.'),
10
11
  };
12
+ const isRow = (v) => v !== null && typeof v === 'object' && !Array.isArray(v);
13
+ /**
14
+ * The state a capture that has not finished is in, and the ONLY one the poll
15
+ * below waits out.
16
+ *
17
+ * Waiting for `pending` specifically is the bug the platform's own reference
18
+ * warns about: `pending` is where a finished capture lands, but replication can
19
+ * carry it on to `durable` between two polls, so a loop matching the literal
20
+ * string can watch a small snapshot go past and never match. Every state that
21
+ * is not this one is a state the snapshot can be restored, cloned and deleted
22
+ * from.
23
+ */
24
+ const CAPTURING = 'capturing';
25
+ /**
26
+ * The states a capture is OVER in, and the reason this is a list of names
27
+ * rather than "anything but {@link CAPTURING}".
28
+ *
29
+ * It is read in one place: the check on the acceptance body, which asks "is
30
+ * there something to wait for". An answer nobody can classify has to mean YES
31
+ * there — a 202 whose `state` arrives misspelt, renamed, or under another key
32
+ * would otherwise read as a snapshot that had landed, and what came back would
33
+ * be the placeholder: `size_bytes: 0` and an id restore, clone and delete all
34
+ * 404 on, reported as a snapshot that can be acted on. That is the defect
35
+ * OPL-4568 removed, reached through a typo instead of an omission and
36
+ * reinstated by drift this server cannot see. An ABSENT state was already
37
+ * handled; `"capturin"` is every bit as unreadable and was not.
38
+ *
39
+ * The POLL LOOP reads the opposite way round and is right to — a row it cannot
40
+ * classify is not a claim, so it asks again rather than deciding, where the
41
+ * alternative is a wait that ends on a state nobody could read. Both directions
42
+ * are the safe one for where they sit, and they are safe in opposite
43
+ * directions.
44
+ *
45
+ * The three names the platform documents beside `capturing`. `deleting` is
46
+ * among them because a row in it is a row the capture is over for, whatever
47
+ * else is true of it. A platform that invents a fourth costs one listing: the
48
+ * poll finds the row already there, carrying whatever state it really has, and
49
+ * returns it at once. So this list going stale costs a round trip, and the
50
+ * other spelling costs the bug.
51
+ *
52
+ * The TypeScript SDK settled here first, under `acceptedCapture` (OPL-4568,
53
+ * `33c47b3`).
54
+ */
55
+ const LANDED = ['pending', 'durable', 'deleting'];
56
+ /**
57
+ * The state the platform marks a snapshot with once a deletion has detached its
58
+ * dependents and is removing the stored objects.
59
+ *
60
+ * Left out of a bare listing, because a half-deleted snapshot is not one you can
61
+ * restore or clone — so a poll that wants to SEE one has to ask for
62
+ * `include=unfinished`. There is no state that means deleted: the row going is
63
+ * the deletion having finished, and a row that stays in this state is one that
64
+ * stalled.
65
+ */
66
+ const DELETING = 'deleting';
67
+ const why = (err) => (err instanceof Error ? err.message : String(err));
68
+ /**
69
+ * One listing, and what it says about `sid`.
70
+ *
71
+ * Asked WITHOUT `allow_partial`, deliberately: the platform then answers a short
72
+ * inventory with a 503, which arrives here as a failure to ride out rather than
73
+ * as a 200 whose missing rows could be read as an answer. The `incomplete`
74
+ * check below is the second line of that defence, for a deployment that sends
75
+ * the header anyway.
76
+ */
77
+ const snapshotTurn = async (api, sid, unfinished) => {
78
+ let items;
79
+ let incomplete = null;
80
+ try {
81
+ ({ items, incomplete } = await api.listing(P.SNAPSHOTS, {
82
+ query: { include: unfinished ? 'unfinished' : undefined },
83
+ }));
84
+ }
85
+ catch (err) {
86
+ if (err instanceof CancelledError)
87
+ return { kind: 'cancelled', why: why(err) };
88
+ if (!isTransientForPoll(err))
89
+ return { kind: 'broken', why: why(err) };
90
+ return { kind: 'blocked', why: why(err), after: pollDelay(err) };
91
+ }
92
+ if (!Array.isArray(items)) {
93
+ const got = items === undefined ? 'no body at all' : items === null ? 'null' : typeof items;
94
+ return { kind: 'blocked', why: `GET /snapshots answered with ${got}, not a list of snapshots` };
95
+ }
96
+ if (incomplete !== null) {
97
+ return {
98
+ kind: 'blocked',
99
+ why: 'GET /snapshots answered short — a hypervisor did not report, so a row missing from it establishes nothing',
100
+ };
101
+ }
102
+ const row = items.find((r) => isRow(r) && r.id === sid);
103
+ if (!isRow(row)) {
104
+ // An unreadable identity could be the target. Only a list whose rows can
105
+ // all be identified establishes absence; a readable target still wins.
106
+ if (items.some((r) => !isRow(r) || typeof r.id !== 'string' || !r.id.trim())) {
107
+ return {
108
+ kind: 'blocked',
109
+ why: 'GET /snapshots contained rows without readable ids, so a missing snapshot establishes nothing',
110
+ };
111
+ }
112
+ return { kind: 'absent' };
113
+ }
114
+ // A row served from the host cache because its host did not answer. It
115
+ // carries an id and nothing else — its `state` is last-known or absent — so it
116
+ // can confirm neither what a capture is doing nor that a deletion has not
117
+ // finished.
118
+ if (row.unreachable) {
119
+ return {
120
+ kind: 'blocked',
121
+ why: `the hypervisor holding ${sid} did not answer, so its row could not be read`,
122
+ };
123
+ }
124
+ return { kind: 'row', row };
125
+ };
11
126
  /**
12
127
  * The sentence in front of a retention window, for the reason every tool here
13
128
  * leads with one: the model reads the text, and three integers in a JSON blob
@@ -48,7 +163,7 @@ export const registerSnapshots = (server, session, opts) => {
48
163
  : 'The fingerprint binds a purge to the snapshots you were shown, so one that arrived after you looked cannot be swept up in it. This server cannot purge them — it was started with the lifecycle tools withheld — so the count and the size are all this answers.';
49
164
  server.registerTool('list_snapshots', {
50
165
  title: 'List snapshots',
51
- description: 'Every snapshot on this account. `orphaned` means its computer is gone: such a snapshot can still be cloned into a new computer, but cannot be restored, because a restore puts the disk back on a source that no longer exists. Read `state` before acting on a row: a capture still being taken is listed FIRST and is not a snapshot yet — it reads `capturing` and its id begins `cap-`, and restore, clone and delete all fail on one. `pending` is the point at which it can be acted on, and `durable` means it has reached backup storage as well.',
166
+ description: 'Every snapshot on this account. `orphaned` means its computer is gone: such a snapshot can still be cloned into a new computer, but cannot be restored, because a restore puts the disk back on a source that no longer exists. Read `state` ON EVERY ROW before acting, rather than on the newest one — this is one answer per hypervisor concatenated in a fixed host order that has nothing to do with time, so a capture running on one host routinely appears after finished snapshots from another. A row reading `capturing` is not a snapshot yet: the copy is still being taken and restore, clone and delete all fail on it. It carries the id the finished snapshot will keep, so this is also the route to poll after create_snapshot: the row stops reading `capturing` in place rather than being replaced under another id, and a row that vanishes without ever leaving `capturing` is a capture that failed. `pending` is the point at which it can be acted on, and `durable` means it has reached backup storage as well.',
52
167
  inputSchema: {
53
168
  computer_id: z
54
169
  .string()
@@ -57,7 +172,7 @@ export const registerSnapshots = (server, session, opts) => {
57
172
  include_unfinished: z
58
173
  .boolean()
59
174
  .default(false)
60
- .describe('Also return deletions that began and did not finish. They are not usable — their state is "deleting" and nothing can be restored or cloned from one — but they still hold objects and are still billed, so this is the flag to set when the question is about storage rather than about what can be restored.'),
175
+ .describe('Also return deletions that began and did not finish. They are not usable — their state is "deleting" and nothing can be restored or cloned from one — but they still hold objects and are still billed, so this is the flag to set when the question is about storage rather than about what can be restored. It is also the flag to set when following a deletion: without it a snapshot stuck half-deleted is hidden, and a poll watching for the row to go cannot tell that from one that finished.'),
61
176
  allow_partial: z
62
177
  .boolean()
63
178
  .optional()
@@ -108,7 +223,7 @@ export const registerSnapshots = (server, session, opts) => {
108
223
  // The filter keeps the unreachable placeholders, and that is not a
109
224
  // nicety. A partial listing does not merely omit rows — the platform
110
225
  // APPENDS one `{id, unreachable: true}` stub per snapshot it could not
111
- // reach, and publicSnapshot drops `computer_id` from such a row because
226
+ // reach, and the platform drops `computer_id` from such a row because
112
227
  // there is no daemon to have said what it belongs to. Filtering on
113
228
  // equality therefore deletes precisely the markers that say something
114
229
  // is missing, and then reports a count: the confident wrong number
@@ -116,7 +231,7 @@ export const registerSnapshots = (server, session, opts) => {
116
231
  //
117
232
  // They cannot be attributed to a computer, so keeping them over-reports
118
233
  // for this one. That is the trade the platform itself makes and writes
119
- // down in lib/hostroute: an extra unreachable row is visible and is
234
+ // down on the platform side: an extra unreachable row is visible and is
120
235
  // corrected by the next complete answer, while a withheld one makes a
121
236
  // row vanish mid-outage, which is the failure worth preventing.
122
237
  // A filter that was GIVEN and trims to nothing is refused, not dropped.
@@ -161,7 +276,7 @@ export const registerSnapshots = (server, session, opts) => {
161
276
  }));
162
277
  server.registerTool('create_snapshot', {
163
278
  title: 'Snapshot a computer',
164
- description: 'Capture a computer so it can be restored or forked later. A disk snapshot is the filesystem; a memory snapshot also saves the running session, so a fork of it comes up with the same processes and windows already open. Name it after the step it is about — that name is what picks it out of the list later.',
279
+ description: 'Capture a computer so it can be restored or forked later. A disk snapshot is the filesystem; a memory snapshot also saves the running session, so a fork of it comes up with the same processes and windows already open. Name it after the step it is about — that name is what picks it out of the list later. THE CAPTURE OUTLIVES THE REQUEST that starts it: the platform accepts it and copies the disk afterwards, which takes minutes and scales with how much has been written. This waits for the copy to land by default and answers with the finished snapshot; pass wait: false to get the id straight back and poll list_snapshots yourself. Everything that can refuse a capture — no such computer, one already running, a memory snapshot of a computer that is not running, an allowance that will not stretch — is refused by this call, so anything else is a capture that started. A wait reports progress while it runs, so a client that sends a progressToken and sets resetTimeoutOnProgress can hold the request open; a client that cannot should pass wait: false and poll list_snapshots, rather than watch its own default timeout cancel a call while the capture goes on running.',
165
280
  inputSchema: {
166
281
  ...idArg,
167
282
  name: z
@@ -181,22 +296,216 @@ export const registerSnapshots = (server, session, opts) => {
181
296
  .boolean()
182
297
  .default(false)
183
298
  .describe('Include the running session. A memory snapshot is a saved machine, so it only loads back into the shape it came off: resize the computer afterwards and the restore is refused, because the vCPU count and the memory size are part of the state rather than decoration around it. Clone it instead in that case, which restores the disk and boots fresh.'),
299
+ wait: z
300
+ .boolean()
301
+ .default(true)
302
+ .describe('Wait for the copy to finish, and answer with the snapshot rather than with the placeholder. Set it false to get the id back at once — useful when the capture is a side errand and there is other work to do meanwhile — and then poll list_snapshots for that id, which is what this does for you.'),
303
+ timeout_s: z
304
+ .number()
305
+ .int()
306
+ .min(5)
307
+ .max(1800)
308
+ .default(300)
309
+ .describe('How long to wait for the capture before handing back and letting you poll. Ignored when wait is false. Giving up on the wait does not stop the capture.'),
184
310
  },
185
- }, ({ computer_id, name, memory }, extra) => guarded(async () => {
311
+ }, ({ computer_id, name, memory, wait, timeout_s }, extra) => guarded(async () => {
186
312
  const id = session.resolve(computer_id);
187
- const res = await session.api
188
- .with(extra.signal)
189
- .json('POST', P.computerAction(id, 'snapshots'), {
190
- body: P.snapshotBody({ memory, name }),
191
- });
313
+ // One deadline for the whole call, armed before the POST, exactly as
314
+ // move_computer arms its own: timeout_s is a promise about when this
315
+ // comes back, and a POST left on undici's own five-minute header clock
316
+ // could break that promise before the poll ever ran. Nothing is armed
317
+ // when nobody is waiting — a caller who asked for the id and no wait
318
+ // was promised no deadline, so imposing one would cancel a POST that
319
+ // was still perfectly capable of starting a capture.
320
+ const untilDeadline = wait ? AbortSignal.timeout(timeout_s * 1000) : undefined;
321
+ const signal = !untilDeadline
322
+ ? extra.signal
323
+ : extra.signal
324
+ ? AbortSignal.any([extra.signal, untilDeadline])
325
+ : untilDeadline;
326
+ const api = session.api.with(signal);
327
+ // The 202. Its body is a Snapshot row in `state: "capturing"` — a
328
+ // placeholder rather than a snapshot, on which restore, clone and delete
329
+ // all answer 404 until the copy lands (OPL-4562). It is kept whatever
330
+ // happens next, because it is the only description of this capture that
331
+ // does not depend on a later read succeeding.
332
+ let started;
333
+ try {
334
+ started = await api.json('POST', P.computerAction(id, 'snapshots'), {
335
+ body: P.snapshotBody({ memory, name }),
336
+ });
337
+ }
338
+ catch (err) {
339
+ // The deadline, or the caller, arriving while the POST is in flight.
340
+ // The generic sentence for this says the request may have been
341
+ // received and to treat what it would have changed as unknown, which
342
+ // is true and is not enough here: a capture that started is billable,
343
+ // takes minutes, and — because the answer that carried its id is the
344
+ // thing that was lost — cannot be polled for at all. Worse, the
345
+ // obvious next move is wrong. A second capture while one is running is
346
+ // refused 409, so a model that simply retries reads that as a failure
347
+ // on top of a capture that is fine (observed live, OPL-4577).
348
+ if (err instanceof CancelledError) {
349
+ return refused(`${why(err)}\n\nA CAPTURE OF ${id} MAY BE RUNNING. What was lost is the answer that carried ` +
350
+ `its id, so there is nothing here to poll on — look instead: list_snapshots on ${id} shows ` +
351
+ `a row reading "${CAPTURING}" if one started, and that row's id is the one to follow. Do ` +
352
+ `not simply call this again; a second capture while one is running is refused, and the ` +
353
+ `refusal will be about the capture this call may have started.`);
354
+ }
355
+ throw err;
356
+ }
192
357
  // The name read back rather than the one sent, because the interesting
193
358
  // case is the one that was not sent: the platform generates
194
359
  // "<computer> <timestamp>" when `name` is absent, and that generated
195
360
  // name is what a later list_snapshots will show. Saying it here is the
196
361
  // difference between a caller that can find this capture again and one
197
362
  // that has to go looking for it.
198
- const called = typeof res?.name === 'string' && res.name.trim() ? ` as "${res.name}"` : '';
199
- return said(`Snapshotted ${id}${memory ? ' with its memory' : ''}${called}.`, res);
363
+ const called = typeof started?.name === 'string' && started.name.trim() ? ` as "${started.name}"` : '';
364
+ // Two sentences for the two things that can be true, and keeping them
365
+ // apart is the whole of OPL-4568. "Snapshotted vm-1" over a 202 is the
366
+ // defect: it is said before a byte has been copied, and a model reads it
367
+ // as a snapshot it may now restore. So the past tense is reserved for a
368
+ // capture this call watched land, and everything that hands back with
369
+ // the copy still running leads with `startedLine` instead.
370
+ const took = `Snapshotted ${id}${memory ? ' with its memory' : ''}${called}.`;
371
+ const startedLine = `Capture of ${id}${memory ? ' with its memory' : ''} started${called}.`;
372
+ // The id allocated before the copy begins, and the whole reason a poll
373
+ // is possible: it is the SNAPSHOT'S own id and does not change when the
374
+ // capture lands, so the row stops reading `capturing` in place rather
375
+ // than being replaced by something under another id.
376
+ const sid = typeof started?.id === 'string' && started.id.trim() ? started.id.trim() : '';
377
+ const startedState = typeof started?.state === 'string' ? started.state : undefined;
378
+ // How to follow it by hand — said wherever this hands back with the
379
+ // capture still running, because "poll list_snapshots" without the two
380
+ // rules under it is an instruction a model gets wrong in both
381
+ // directions: by waiting for the literal `pending`, and by reading a
382
+ // row that vanished as a row that has not appeared yet.
383
+ const byHand = `Poll list_snapshots for the id ${sid}: the row stops reading "${CAPTURING}" when the capture ` +
384
+ `lands — do not wait for "pending" specifically, since replication can carry it straight on to ` +
385
+ `"durable" — and a row that DISAPPEARS without ever leaving "${CAPTURING}" is a capture that failed.`;
386
+ // A capture accepted under an id nobody was told cannot be polled for,
387
+ // and it cannot be found again either — every later call takes the id.
388
+ // `refused`, for clone_snapshot's reason one route over: the platform
389
+ // did something billable and the caller has no handle on it, which is
390
+ // not a result to report as success.
391
+ //
392
+ // BEFORE the wait: false branch, not after it, and Codex caught that it
393
+ // was the other way round. `wait: false` is the answer whose whole
394
+ // content is an id — handed back without one it reported success and
395
+ // told the caller to poll list_snapshots for "the id ", which is the
396
+ // same nothing dressed as an instruction.
397
+ if (!sid) {
398
+ return refused(`${startedLine} THE CAPTURE IS RUNNING, but the platform sent no snapshot id back, so there ` +
399
+ `is nothing to poll on and the snapshot cannot be named — every call that acts on one takes ` +
400
+ `its id. list_snapshots on ${id} will show it when it lands.`, started);
401
+ }
402
+ if (!wait) {
403
+ return said(`${startedLine} It is NOT a snapshot yet: while its row reads "${CAPTURING}" it is a ` +
404
+ `placeholder, and restore, clone and delete all fail on it. ${byHand}`, started);
405
+ }
406
+ // Already finished when it was answered for — a `201` from a platform
407
+ // that has not taken the 202 yet, or a capture with nothing to copy.
408
+ // Not a state to poll out of, and a loop that did would ask for a row it
409
+ // already holds.
410
+ //
411
+ // An ALLOW-LIST, for the reason {@link LANDED} gives: this asks whether
412
+ // there is anything to wait for, and every unreadable answer — absent,
413
+ // misspelt, renamed, moved — has to mean yes. Only a state this server
414
+ // can read AS a landed one skips the wait.
415
+ if (startedState !== undefined && LANDED.includes(startedState)) {
416
+ return said(`${took} It is ${startedState} — snapshot ${sid}.`, started);
417
+ }
418
+ // The keepalive. Armed after the 202, because everything before it is a
419
+ // single short request and there is nothing to report until there is a
420
+ // capture to report on.
421
+ const beat = heartbeat(extra, server.server);
422
+ await beat(`Capturing ${sid} of ${id} — the copy has started.`);
423
+ let blocked;
424
+ // `untilDeadline &&` rather than `!untilDeadline?.aborted`: the early
425
+ // return above is what guarantees it is armed here, and the optional
426
+ // form would turn a later edit that broke that guarantee into a loop
427
+ // with no deadline at all rather than into a visible mistake.
428
+ while (untilDeadline && !untilDeadline.aborted) {
429
+ // The caller giving up ends the wait. The signal aborts the request in
430
+ // flight, but nothing about an aborted request stops the next
431
+ // iteration from starting one.
432
+ if (extra.signal?.aborted) {
433
+ return refused(`Cancelled while waiting for the capture of ${id}. THE CAPTURE IS STILL RUNNING — nothing ` +
434
+ `was stopped, because a disk copy already under way cannot be called back. ${byHand}`, started);
435
+ }
436
+ const turn = await snapshotTurn(api, sid, false);
437
+ if (extra.signal?.aborted)
438
+ continue;
439
+ // A body stream can fail without either signal firing, so the
440
+ // deadline's own arrival is what tells a real timeout from an undici
441
+ // idle abort. Same shape as the two loops in computers.ts.
442
+ if (turn.kind === 'cancelled') {
443
+ if (untilDeadline?.aborted)
444
+ break;
445
+ blocked = turn.why;
446
+ await beat(`Capturing ${sid} of ${id} — the platform could not be asked: ${turn.why}`);
447
+ await sleep(POLL_MS, signal);
448
+ continue;
449
+ }
450
+ // A poll that failed for a reason worth riding out is weather; one
451
+ // that failed for any other reason is a real failure with a capture
452
+ // still running behind it, and a thrown error's handler has no way to
453
+ // say so. So it is said here rather than rethrown.
454
+ if (turn.kind === 'broken') {
455
+ return refused(`${turn.why}\n\nTHE CAPTURE IS STILL RUNNING — this was the poll failing, not the ` +
456
+ `capture. ${byHand}`, started);
457
+ }
458
+ if (turn.kind === 'blocked') {
459
+ blocked = turn.why;
460
+ await beat(`Capturing ${sid} of ${id} — the platform could not be asked: ${turn.why}`);
461
+ await sleep(turn.after ?? POLL_MS, signal);
462
+ continue;
463
+ }
464
+ // Absence is the ONLY signal a failed capture leaves, which is what
465
+ // makes snapshotTurn's guards load-bearing rather than tidy: a body
466
+ // that is not a list, and a listing the platform had to answer short,
467
+ // both come back `blocked` there rather than as this, because read as
468
+ // "the row is not there" they would announce that somebody's backup
469
+ // failed while it was still being taken.
470
+ if (turn.kind === 'absent') {
471
+ return refused(`THE CAPTURE OF ${id} FAILED and nothing was saved. Snapshot ${sid} is no longer listed and ` +
472
+ `nothing took its place, which is what a capture that starts and then fails leaves behind — ` +
473
+ `this is a failure during the copy, not a wait that ran out. The computer itself is ` +
474
+ `untouched and can be snapshotted again.`, started);
475
+ }
476
+ const row = turn.row;
477
+ const state = typeof row.state === 'string' ? row.state : undefined;
478
+ if (state === CAPTURING) {
479
+ blocked = undefined;
480
+ await beat(`Capturing ${sid} of ${id} — still copying.`);
481
+ await sleep(POLL_MS, signal);
482
+ continue;
483
+ }
484
+ // A row with no readable `state` is the platform failing to describe
485
+ // it, and it must not be read as "not capturing, therefore landed" —
486
+ // that sentence says a placeholder may be restored, which is the
487
+ // defect this whole tool is here to stop. Only a state that SAYS it is
488
+ // no longer capturing ends the wait; anything else rides out, and the
489
+ // deadline reports that the platform could not be asked.
490
+ if (state === undefined) {
491
+ blocked = `the row for ${sid} carried no state, so nothing said whether the capture had landed`;
492
+ await beat(`Capturing ${sid} of ${id} — its row carried no state to read.`);
493
+ await sleep(POLL_MS, signal);
494
+ continue;
495
+ }
496
+ return said(`${took} The capture landed: snapshot ${sid} reads "${state}" and can now be restored, ` +
497
+ `cloned or deleted.`, row);
498
+ }
499
+ // A refusal, for the reason move_computer's is: the wait never reached
500
+ // what it was told to wait for. What it must not do is read as a capture
501
+ // that failed — that is a different answer with a different id in it,
502
+ // and the difference between calling this again and going looking for a
503
+ // snapshot that is on its way.
504
+ return refused(blocked
505
+ ? `${startedLine} Gave up watching after ${timeout_s}s; the platform could not be asked — the ` +
506
+ `last attempt said: ${blocked}. THE CAPTURE IS STILL RUNNING. ${byHand}`
507
+ : `${startedLine} Still capturing after ${timeout_s}s, which a large disk takes. THE CAPTURE IS ` +
508
+ `STILL RUNNING and nothing was changed by giving up on the wait. ${byHand}`, started);
200
509
  }));
201
510
  server.registerTool('restore_snapshot', {
202
511
  title: 'Restore a snapshot',
@@ -246,7 +555,21 @@ export const registerSnapshots = (server, session, opts) => {
246
555
  if (clear)
247
556
  return said('Schedule cleared.', await session.api.with(extra.signal).send('DELETE', path));
248
557
  if (set) {
249
- return said(`Snapshot scheduled for ${String(set.hour).padStart(2, '0')}:${String(set.minute).padStart(2, '0')} ${set.tz}.`, await session.api.with(extra.signal).send('PUT', path, { body: P.scheduleBody(set) }));
558
+ const at = `${String(set.hour).padStart(2, '0')}:${String(set.minute).padStart(2, '0')} ${set.tz}`;
559
+ const res = await session.api
560
+ .with(extra.signal)
561
+ .send('PUT', path, { body: P.scheduleBody(set) });
562
+ // `enabled` is a required field of `set`, and it is the one that
563
+ // decides whether backups happen at all — so the sentence has to read
564
+ // it. "Snapshot scheduled for 03:00 UTC" over `enabled: false` tells a
565
+ // model the opposite of what the call just did, in the line it acts
566
+ // on, and the hour it names is real, which is what makes the wrong
567
+ // reading easy to believe. The hour is kept on a disabled schedule, so
568
+ // it is worth naming: a model that reads "disabled" and nothing else
569
+ // calls this again to find out what it would have run at.
570
+ return said(set.enabled
571
+ ? `Snapshot scheduled for ${at}.`
572
+ : `The nightly snapshot on ${id} is now DISABLED — none will be taken automatically. ${at} is the time it holds, which is when it would resume if you set it again with enabled: true. To remove the schedule rather than turn it off, call this with clear.`, res);
250
573
  }
251
574
  return json(await session.api.with(extra.signal).json('GET', path));
252
575
  }));
@@ -295,17 +618,85 @@ export const registerSnapshots = (server, session, opts) => {
295
618
  }));
296
619
  server.registerTool('delete_snapshot', {
297
620
  title: 'Delete a snapshot',
298
- description: 'Remove a snapshot permanently. Later snapshots in the same chain are unaffected.',
621
+ description: 'Remove a snapshot permanently. Later snapshots in the same chain are unaffected. THE DELETION OUTLIVES THE REQUEST that starts it: the platform accepts it and then detaches the dependent snapshots and removes the stored objects, which takes time that scales with the chain and with how much is stored. This waits for it by default and reports what actually happened; pass wait: false to hand back as soon as it is accepted. A 409 saying the snapshot is ALREADY BEING DELETED is progress rather than a fault — the platform is doing what you asked, and the answer is to watch that one finish, never to go and delete something else. A wait reports progress while it runs, so a client that sends a progressToken and sets resetTimeoutOnProgress can hold the request open; a client that cannot should pass wait: false and poll list_snapshots.',
299
622
  inputSchema: {
300
- snapshot_id: z.string(),
623
+ snapshot_id: z.string().trim(),
301
624
  confirm: z.literal(true).describe('Must be true.'),
625
+ wait: z
626
+ .boolean()
627
+ .default(true)
628
+ .describe('Wait for the snapshot to stop being listed, which is the only thing that means it is gone. Set it false to hand back as soon as the platform accepts the deletion, and then poll list_snapshots yourself.'),
629
+ timeout_s: z
630
+ .number()
631
+ .int()
632
+ .min(5)
633
+ .max(1800)
634
+ .default(300)
635
+ .describe('How long to wait for the deletion before handing back. Ignored when wait is false. Giving up on the wait does not stop the deletion.'),
302
636
  },
303
637
  annotations: { destructiveHint: true, idempotentHint: true },
304
- }, ({ snapshot_id }, extra) => guarded(async () => {
638
+ }, ({ snapshot_id, wait, timeout_s }, extra) => guarded(async () => {
639
+ // The deadline is armed before the DELETE, as create_snapshot arms its
640
+ // own and for the same reason: timeout_s is a promise about when this
641
+ // comes back. Nothing is armed when nobody is waiting.
642
+ const untilDeadline = wait ? AbortSignal.timeout(timeout_s * 1000) : undefined;
643
+ const signal = !untilDeadline
644
+ ? extra.signal
645
+ : extra.signal
646
+ ? AbortSignal.any([extra.signal, untilDeadline])
647
+ : untilDeadline;
648
+ const api = session.api.with(signal);
649
+ // How to follow it by hand, said wherever this hands back with the
650
+ // deletion still running. The polarity is the whole of it and it is the
651
+ // opposite of a capture's: there is no state that means deleted, so the
652
+ // row GOING is the finish, and a row that stays is one that stalled.
653
+ // `include_unfinished` is not a nicety either — the platform marks a
654
+ // half-finished deletion `deleting` and leaves that state out of a bare
655
+ // listing, so without the flag a stalled deletion looks exactly like a
656
+ // finished one.
657
+ const byHand = `Poll list_snapshots with include_unfinished: true and watch for ${snapshot_id} to stop being ` +
658
+ `listed — its absence is the deletion having finished, and there is no state that means deleted. ` +
659
+ `A row that STAYS is one that stalled; the platform retries those itself every fifteen minutes.`;
660
+ let accepted;
305
661
  try {
306
- await session.api.with(extra.signal).send('DELETE', P.snapshot(snapshot_id));
662
+ accepted = await api.send('DELETE', P.snapshot(snapshot_id));
307
663
  }
308
664
  catch (err) {
665
+ // Every refusal is still decided before the 202 — no such snapshot, a
666
+ // capture reading through it, a clone or a migration holding it, a
667
+ // deletion of this id already running — so a 409 here is a statement
668
+ // about state that clears itself, and the model needs to be told
669
+ // which way to read it rather than left to invent a way round it.
670
+ //
671
+ // Matched on the CLASS and never on the sentence: the platform sends
672
+ // no `reason` for these, and keying on prose is the mistake OPL-3724
673
+ // took out of three clients. So both readings are named and the
674
+ // platform's own words are printed with them.
675
+ // The deadline, or the caller, arriving while the DELETE is in
676
+ // flight. The generic cancellation sentence says the request may have
677
+ // been received; what it cannot say is that the work it started
678
+ // OUTLIVES the request, so "cancelled" here reads as a deletion that
679
+ // did not happen while the objects are being removed. Claiming less
680
+ // than happened, which on a destructive call is its own kind of wrong
681
+ // answer (codex review, gpt-5.6-sol).
682
+ //
683
+ // The retry rule is worth saying in the same breath, because it is
684
+ // the one thing that makes this recoverable: repeating the call is
685
+ // safe. A snapshot already being deleted answers 409, and one that is
686
+ // gone answers 404, and this tool has a sentence for each.
687
+ if (err instanceof CancelledError) {
688
+ return refused(`${why(err)}\n\nTHE DELETION OF ${snapshot_id} MAY BE RUNNING: the request may have reached ` +
689
+ `the platform and been accepted, and nothing about the answer being lost calls it back. ` +
690
+ `${byHand} Calling this again is safe either way — a snapshot already being deleted ` +
691
+ `answers a conflict, and one that is gone answers that there is nothing to delete.`);
692
+ }
693
+ if (err instanceof ConflictError) {
694
+ return refused(`${why(err)}\n\nNOTHING WAS DELETED and nothing is broken. If that says the snapshot is ` +
695
+ `already being deleted, the platform is doing what you asked — watch it finish rather ` +
696
+ `than deleting anything else: ${byHand} If it says something is reading through the ` +
697
+ `snapshot — a restore, a clone, a capture chaining onto it — that finishes on its own and ` +
698
+ `the same call works afterwards.`);
699
+ }
309
700
  // A 404 means the snapshot is not there, which is the state this call
310
701
  // was asking for. `idempotentHint` above invites a client to retry a
311
702
  // lost 2xx, and `#fetch` throws on every non-OK — so that invited
@@ -327,7 +718,74 @@ export const registerSnapshots = (server, session, opts) => {
327
718
  'a real one may still be held under the id you meant. list_snapshots says which of the two ' +
328
719
  'this is.');
329
720
  }
330
- return said(`Deleted snapshot ${snapshot_id}.`);
721
+ // The 202 and its body: the snapshot's row as it stood when the deletion
722
+ // was accepted — the row that GOES when the work finishes.
723
+ if (!wait) {
724
+ return said(`Deletion of ${snapshot_id} accepted and RUNNING — it is not gone yet, and this call did not ` +
725
+ `wait to find out. ${byHand}`, accepted);
726
+ }
727
+ const beat = heartbeat(extra, server.server);
728
+ await beat(`Deleting ${snapshot_id} — the platform has accepted it.`);
729
+ let blocked;
730
+ let seen;
731
+ while (untilDeadline && !untilDeadline.aborted) {
732
+ if (extra.signal?.aborted) {
733
+ return refused(`Cancelled while waiting for ${snapshot_id} to be deleted. THE DELETION IS STILL RUNNING — ` +
734
+ `nothing was called back, and objects it has already removed are already gone. ${byHand}`, accepted);
735
+ }
736
+ // `unfinished`, because the state a stalled deletion sits in is the
737
+ // one a bare listing hides. Asking without it would read a snapshot
738
+ // stuck half-deleted as a snapshot successfully deleted, which is the
739
+ // one wrong answer this tool must not give.
740
+ const turn = await snapshotTurn(api, snapshot_id, true);
741
+ if (extra.signal?.aborted)
742
+ continue;
743
+ if (turn.kind === 'cancelled') {
744
+ if (untilDeadline.aborted)
745
+ break;
746
+ blocked = turn.why;
747
+ await beat(`Deleting ${snapshot_id} — the platform could not be asked: ${turn.why}`);
748
+ await sleep(POLL_MS, signal);
749
+ continue;
750
+ }
751
+ if (turn.kind === 'broken') {
752
+ return refused(`${turn.why}\n\nTHE DELETION IS STILL RUNNING — this was the poll failing, not the ` +
753
+ `deletion. ${byHand}`, accepted);
754
+ }
755
+ if (turn.kind === 'blocked') {
756
+ blocked = turn.why;
757
+ await beat(`Deleting ${snapshot_id} — the platform could not be asked: ${turn.why}`);
758
+ await sleep(turn.after ?? POLL_MS, signal);
759
+ continue;
760
+ }
761
+ // The row is gone from a listing that was read whole and that ASKED
762
+ // for the unfinished ones. Both halves are what make this sentence
763
+ // true rather than merely likely.
764
+ if (turn.kind === 'absent')
765
+ return said(`Deleted snapshot ${snapshot_id}.`, accepted);
766
+ blocked = undefined;
767
+ seen = typeof turn.row.state === 'string' ? turn.row.state : undefined;
768
+ await beat(`Deleting ${snapshot_id} — still listed${seen ? `, reading "${seen}"` : ''}.`);
769
+ await sleep(POLL_MS, signal);
770
+ }
771
+ // Still listed. Three different things that can mean, and they are not
772
+ // one sentence: the platform could not be asked, the deletion got part
773
+ // way and stalled, or it never got started on — which is the shape the
774
+ // one conflict that arrives AFTER the 202 leaves behind, a dependent
775
+ // that is itself being deleted and so cannot be detached.
776
+ return refused(blocked
777
+ ? `Gave up watching after ${timeout_s}s; the platform could not be asked whether ${snapshot_id} ` +
778
+ `is gone — the last attempt said: ${blocked}. THE DELETION IS STILL RUNNING. ${byHand}`
779
+ : seen === DELETING
780
+ ? `${snapshot_id} is still listed after ${timeout_s}s and reads "${DELETING}": the deletion ` +
781
+ `started and has not finished. Nothing was undone by giving up on the wait, and the ` +
782
+ `platform retries a stalled deletion itself every fifteen minutes — so this usually needs ` +
783
+ `watching rather than repeating. ${byHand}`
784
+ : `${snapshot_id} is still listed after ${timeout_s}s${seen ? `, reading "${seen}"` : ''} — ` +
785
+ `the deletion has not reached the point of marking it "${DELETING}". That is what a big ` +
786
+ `chain looks like early on, and it is also what the one conflict that arrives after the ` +
787
+ `deletion is accepted looks like: a dependent snapshot that is ITSELF being deleted cannot ` +
788
+ `be detached, so this one waits for that one. Nothing was destroyed either way. ${byHand}`, accepted);
331
789
  }));
332
790
  };
333
791
  //# sourceMappingURL=snapshots.js.map