mandala-computer-mcp 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of mandala-computer-mcp might be problematic. Click here for more details.
- package/README.md +102 -10
- package/dist/api.d.ts +19 -6
- package/dist/api.d.ts.map +1 -1
- package/dist/api.js +213 -31
- package/dist/api.js.map +1 -1
- package/dist/cli.d.ts.map +1 -1
- package/dist/cli.js +113 -8
- package/dist/cli.js.map +1 -1
- package/dist/errors.d.ts +60 -0
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +71 -2
- package/dist/errors.js.map +1 -1
- package/dist/events.d.ts +17 -2
- package/dist/events.d.ts.map +1 -1
- package/dist/events.js +104 -13
- package/dist/events.js.map +1 -1
- package/dist/format.d.ts +16 -0
- package/dist/format.d.ts.map +1 -1
- package/dist/format.js +11 -1
- package/dist/format.js.map +1 -1
- package/dist/http-body.d.ts +17 -0
- package/dist/http-body.d.ts.map +1 -0
- package/dist/http-body.js +48 -0
- package/dist/http-body.js.map +1 -0
- package/dist/http.d.ts.map +1 -1
- package/dist/http.js +177 -51
- package/dist/http.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/paths.d.ts +22 -13
- package/dist/paths.d.ts.map +1 -1
- package/dist/paths.js +68 -16
- package/dist/paths.js.map +1 -1
- package/dist/poll.d.ts +103 -0
- package/dist/poll.d.ts.map +1 -0
- package/dist/poll.js +129 -0
- package/dist/poll.js.map +1 -0
- package/dist/server.d.ts +1 -1
- package/dist/server.js +1 -1
- package/dist/tools/agent.d.ts.map +1 -1
- package/dist/tools/agent.js +14 -2
- package/dist/tools/agent.js.map +1 -1
- package/dist/tools/computers.d.ts.map +1 -1
- package/dist/tools/computers.js +228 -49
- package/dist/tools/computers.js.map +1 -1
- package/dist/tools/events.d.ts.map +1 -1
- package/dist/tools/events.js +224 -39
- package/dist/tools/events.js.map +1 -1
- package/dist/tools/guest.d.ts.map +1 -1
- package/dist/tools/guest.js +223 -30
- package/dist/tools/guest.js.map +1 -1
- package/dist/tools/input.d.ts.map +1 -1
- package/dist/tools/input.js +29 -3
- package/dist/tools/input.js.map +1 -1
- package/dist/tools/snapshots.d.ts.map +1 -1
- package/dist/tools/snapshots.js +478 -20
- package/dist/tools/snapshots.js.map +1 -1
- package/dist/tools/templates.d.ts.map +1 -1
- package/dist/tools/templates.js +50 -11
- package/dist/tools/templates.js.map +1 -1
- package/dist/tools/webhooks.d.ts.map +1 -1
- package/dist/tools/webhooks.js +116 -17
- package/dist/tools/webhooks.js.map +1 -1
- package/package.json +1 -1
package/dist/tools/snapshots.js
CHANGED
|
@@ -1,13 +1,128 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
|
-
import { NotFoundError } from '../errors.js';
|
|
2
|
+
import { CancelledError, ConflictError, isTransientForPoll, NotFoundError } from '../errors.js';
|
|
3
3
|
import { describe, guarded, incompleteWarning, json, refused, said, unwrapComputer, withoutCredentials, } from '../format.js';
|
|
4
4
|
import * as P from '../paths.js';
|
|
5
|
+
import { heartbeat, POLL_MS, pollDelay, sleep } from '../poll.js';
|
|
5
6
|
const idArg = {
|
|
6
7
|
computer_id: z
|
|
7
8
|
.string()
|
|
8
9
|
.optional()
|
|
9
10
|
.describe('Which computer. Defaults to the one selected with use_computer.'),
|
|
10
11
|
};
|
|
12
|
+
const isRow = (v) => v !== null && typeof v === 'object' && !Array.isArray(v);
|
|
13
|
+
/**
|
|
14
|
+
* The state a capture that has not finished is in, and the ONLY one the poll
|
|
15
|
+
* below waits out.
|
|
16
|
+
*
|
|
17
|
+
* Waiting for `pending` specifically is the bug the platform's own reference
|
|
18
|
+
* warns about: `pending` is where a finished capture lands, but replication can
|
|
19
|
+
* carry it on to `durable` between two polls, so a loop matching the literal
|
|
20
|
+
* string can watch a small snapshot go past and never match. Every state that
|
|
21
|
+
* is not this one is a state the snapshot can be restored, cloned and deleted
|
|
22
|
+
* from.
|
|
23
|
+
*/
|
|
24
|
+
const CAPTURING = 'capturing';
|
|
25
|
+
/**
|
|
26
|
+
* The states a capture is OVER in, and the reason this is a list of names
|
|
27
|
+
* rather than "anything but {@link CAPTURING}".
|
|
28
|
+
*
|
|
29
|
+
* It is read in one place: the check on the acceptance body, which asks "is
|
|
30
|
+
* there something to wait for". An answer nobody can classify has to mean YES
|
|
31
|
+
* there — a 202 whose `state` arrives misspelt, renamed, or under another key
|
|
32
|
+
* would otherwise read as a snapshot that had landed, and what came back would
|
|
33
|
+
* be the placeholder: `size_bytes: 0` and an id restore, clone and delete all
|
|
34
|
+
* 404 on, reported as a snapshot that can be acted on. That is the defect
|
|
35
|
+
* OPL-4568 removed, reached through a typo instead of an omission and
|
|
36
|
+
* reinstated by drift this server cannot see. An ABSENT state was already
|
|
37
|
+
* handled; `"capturin"` is every bit as unreadable and was not.
|
|
38
|
+
*
|
|
39
|
+
* The POLL LOOP reads the opposite way round and is right to — a row it cannot
|
|
40
|
+
* classify is not a claim, so it asks again rather than deciding, where the
|
|
41
|
+
* alternative is a wait that ends on a state nobody could read. Both directions
|
|
42
|
+
* are the safe one for where they sit, and they are safe in opposite
|
|
43
|
+
* directions.
|
|
44
|
+
*
|
|
45
|
+
* The three names the platform documents beside `capturing`. `deleting` is
|
|
46
|
+
* among them because a row in it is a row the capture is over for, whatever
|
|
47
|
+
* else is true of it. A platform that invents a fourth costs one listing: the
|
|
48
|
+
* poll finds the row already there, carrying whatever state it really has, and
|
|
49
|
+
* returns it at once. So this list going stale costs a round trip, and the
|
|
50
|
+
* other spelling costs the bug.
|
|
51
|
+
*
|
|
52
|
+
* The TypeScript SDK settled here first, under `acceptedCapture` (OPL-4568,
|
|
53
|
+
* `33c47b3`).
|
|
54
|
+
*/
|
|
55
|
+
const LANDED = ['pending', 'durable', 'deleting'];
|
|
56
|
+
/**
|
|
57
|
+
* The state the platform marks a snapshot with once a deletion has detached its
|
|
58
|
+
* dependents and is removing the stored objects.
|
|
59
|
+
*
|
|
60
|
+
* Left out of a bare listing, because a half-deleted snapshot is not one you can
|
|
61
|
+
* restore or clone — so a poll that wants to SEE one has to ask for
|
|
62
|
+
* `include=unfinished`. There is no state that means deleted: the row going is
|
|
63
|
+
* the deletion having finished, and a row that stays in this state is one that
|
|
64
|
+
* stalled.
|
|
65
|
+
*/
|
|
66
|
+
const DELETING = 'deleting';
|
|
67
|
+
const why = (err) => (err instanceof Error ? err.message : String(err));
|
|
68
|
+
/**
|
|
69
|
+
* One listing, and what it says about `sid`.
|
|
70
|
+
*
|
|
71
|
+
* Asked WITHOUT `allow_partial`, deliberately: the platform then answers a short
|
|
72
|
+
* inventory with a 503, which arrives here as a failure to ride out rather than
|
|
73
|
+
* as a 200 whose missing rows could be read as an answer. The `incomplete`
|
|
74
|
+
* check below is the second line of that defence, for a deployment that sends
|
|
75
|
+
* the header anyway.
|
|
76
|
+
*/
|
|
77
|
+
const snapshotTurn = async (api, sid, unfinished) => {
|
|
78
|
+
let items;
|
|
79
|
+
let incomplete = null;
|
|
80
|
+
try {
|
|
81
|
+
({ items, incomplete } = await api.listing(P.SNAPSHOTS, {
|
|
82
|
+
query: { include: unfinished ? 'unfinished' : undefined },
|
|
83
|
+
}));
|
|
84
|
+
}
|
|
85
|
+
catch (err) {
|
|
86
|
+
if (err instanceof CancelledError)
|
|
87
|
+
return { kind: 'cancelled', why: why(err) };
|
|
88
|
+
if (!isTransientForPoll(err))
|
|
89
|
+
return { kind: 'broken', why: why(err) };
|
|
90
|
+
return { kind: 'blocked', why: why(err), after: pollDelay(err) };
|
|
91
|
+
}
|
|
92
|
+
if (!Array.isArray(items)) {
|
|
93
|
+
const got = items === undefined ? 'no body at all' : items === null ? 'null' : typeof items;
|
|
94
|
+
return { kind: 'blocked', why: `GET /snapshots answered with ${got}, not a list of snapshots` };
|
|
95
|
+
}
|
|
96
|
+
if (incomplete !== null) {
|
|
97
|
+
return {
|
|
98
|
+
kind: 'blocked',
|
|
99
|
+
why: 'GET /snapshots answered short — a hypervisor did not report, so a row missing from it establishes nothing',
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
const row = items.find((r) => isRow(r) && r.id === sid);
|
|
103
|
+
if (!isRow(row)) {
|
|
104
|
+
// An unreadable identity could be the target. Only a list whose rows can
|
|
105
|
+
// all be identified establishes absence; a readable target still wins.
|
|
106
|
+
if (items.some((r) => !isRow(r) || typeof r.id !== 'string' || !r.id.trim())) {
|
|
107
|
+
return {
|
|
108
|
+
kind: 'blocked',
|
|
109
|
+
why: 'GET /snapshots contained rows without readable ids, so a missing snapshot establishes nothing',
|
|
110
|
+
};
|
|
111
|
+
}
|
|
112
|
+
return { kind: 'absent' };
|
|
113
|
+
}
|
|
114
|
+
// A row served from the host cache because its host did not answer. It
|
|
115
|
+
// carries an id and nothing else — its `state` is last-known or absent — so it
|
|
116
|
+
// can confirm neither what a capture is doing nor that a deletion has not
|
|
117
|
+
// finished.
|
|
118
|
+
if (row.unreachable) {
|
|
119
|
+
return {
|
|
120
|
+
kind: 'blocked',
|
|
121
|
+
why: `the hypervisor holding ${sid} did not answer, so its row could not be read`,
|
|
122
|
+
};
|
|
123
|
+
}
|
|
124
|
+
return { kind: 'row', row };
|
|
125
|
+
};
|
|
11
126
|
/**
|
|
12
127
|
* The sentence in front of a retention window, for the reason every tool here
|
|
13
128
|
* leads with one: the model reads the text, and three integers in a JSON blob
|
|
@@ -48,7 +163,7 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
48
163
|
: 'The fingerprint binds a purge to the snapshots you were shown, so one that arrived after you looked cannot be swept up in it. This server cannot purge them — it was started with the lifecycle tools withheld — so the count and the size are all this answers.';
|
|
49
164
|
server.registerTool('list_snapshots', {
|
|
50
165
|
title: 'List snapshots',
|
|
51
|
-
description: 'Every snapshot on this account. `orphaned` means its computer is gone: such a snapshot can still be cloned into a new computer, but cannot be restored, because a restore puts the disk back on a source that no longer exists. Read `state` before acting on a
|
|
166
|
+
description: 'Every snapshot on this account. `orphaned` means its computer is gone: such a snapshot can still be cloned into a new computer, but cannot be restored, because a restore puts the disk back on a source that no longer exists. Read `state` ON EVERY ROW before acting, rather than on the newest one — this is one answer per hypervisor concatenated in a fixed host order that has nothing to do with time, so a capture running on one host routinely appears after finished snapshots from another. A row reading `capturing` is not a snapshot yet: the copy is still being taken and restore, clone and delete all fail on it. It carries the id the finished snapshot will keep, so this is also the route to poll after create_snapshot: the row stops reading `capturing` in place rather than being replaced under another id, and a row that vanishes without ever leaving `capturing` is a capture that failed. `pending` is the point at which it can be acted on, and `durable` means it has reached backup storage as well.',
|
|
52
167
|
inputSchema: {
|
|
53
168
|
computer_id: z
|
|
54
169
|
.string()
|
|
@@ -57,7 +172,7 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
57
172
|
include_unfinished: z
|
|
58
173
|
.boolean()
|
|
59
174
|
.default(false)
|
|
60
|
-
.describe('Also return deletions that began and did not finish. They are not usable — their state is "deleting" and nothing can be restored or cloned from one — but they still hold objects and are still billed, so this is the flag to set when the question is about storage rather than about what can be restored.'),
|
|
175
|
+
.describe('Also return deletions that began and did not finish. They are not usable — their state is "deleting" and nothing can be restored or cloned from one — but they still hold objects and are still billed, so this is the flag to set when the question is about storage rather than about what can be restored. It is also the flag to set when following a deletion: without it a snapshot stuck half-deleted is hidden, and a poll watching for the row to go cannot tell that from one that finished.'),
|
|
61
176
|
allow_partial: z
|
|
62
177
|
.boolean()
|
|
63
178
|
.optional()
|
|
@@ -108,7 +223,7 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
108
223
|
// The filter keeps the unreachable placeholders, and that is not a
|
|
109
224
|
// nicety. A partial listing does not merely omit rows — the platform
|
|
110
225
|
// APPENDS one `{id, unreachable: true}` stub per snapshot it could not
|
|
111
|
-
// reach, and
|
|
226
|
+
// reach, and the platform drops `computer_id` from such a row because
|
|
112
227
|
// there is no daemon to have said what it belongs to. Filtering on
|
|
113
228
|
// equality therefore deletes precisely the markers that say something
|
|
114
229
|
// is missing, and then reports a count: the confident wrong number
|
|
@@ -116,7 +231,7 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
116
231
|
//
|
|
117
232
|
// They cannot be attributed to a computer, so keeping them over-reports
|
|
118
233
|
// for this one. That is the trade the platform itself makes and writes
|
|
119
|
-
// down
|
|
234
|
+
// down on the platform side: an extra unreachable row is visible and is
|
|
120
235
|
// corrected by the next complete answer, while a withheld one makes a
|
|
121
236
|
// row vanish mid-outage, which is the failure worth preventing.
|
|
122
237
|
// A filter that was GIVEN and trims to nothing is refused, not dropped.
|
|
@@ -161,7 +276,7 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
161
276
|
}));
|
|
162
277
|
server.registerTool('create_snapshot', {
|
|
163
278
|
title: 'Snapshot a computer',
|
|
164
|
-
description: 'Capture a computer so it can be restored or forked later. A disk snapshot is the filesystem; a memory snapshot also saves the running session, so a fork of it comes up with the same processes and windows already open. Name it after the step it is about — that name is what picks it out of the list later.',
|
|
279
|
+
description: 'Capture a computer so it can be restored or forked later. A disk snapshot is the filesystem; a memory snapshot also saves the running session, so a fork of it comes up with the same processes and windows already open. Name it after the step it is about — that name is what picks it out of the list later. THE CAPTURE OUTLIVES THE REQUEST that starts it: the platform accepts it and copies the disk afterwards, which takes minutes and scales with how much has been written. This waits for the copy to land by default and answers with the finished snapshot; pass wait: false to get the id straight back and poll list_snapshots yourself. Everything that can refuse a capture — no such computer, one already running, a memory snapshot of a computer that is not running, an allowance that will not stretch — is refused by this call, so anything else is a capture that started. A wait reports progress while it runs, so a client that sends a progressToken and sets resetTimeoutOnProgress can hold the request open; a client that cannot should pass wait: false and poll list_snapshots, rather than watch its own default timeout cancel a call while the capture goes on running.',
|
|
165
280
|
inputSchema: {
|
|
166
281
|
...idArg,
|
|
167
282
|
name: z
|
|
@@ -181,22 +296,216 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
181
296
|
.boolean()
|
|
182
297
|
.default(false)
|
|
183
298
|
.describe('Include the running session. A memory snapshot is a saved machine, so it only loads back into the shape it came off: resize the computer afterwards and the restore is refused, because the vCPU count and the memory size are part of the state rather than decoration around it. Clone it instead in that case, which restores the disk and boots fresh.'),
|
|
299
|
+
wait: z
|
|
300
|
+
.boolean()
|
|
301
|
+
.default(true)
|
|
302
|
+
.describe('Wait for the copy to finish, and answer with the snapshot rather than with the placeholder. Set it false to get the id back at once — useful when the capture is a side errand and there is other work to do meanwhile — and then poll list_snapshots for that id, which is what this does for you.'),
|
|
303
|
+
timeout_s: z
|
|
304
|
+
.number()
|
|
305
|
+
.int()
|
|
306
|
+
.min(5)
|
|
307
|
+
.max(1800)
|
|
308
|
+
.default(300)
|
|
309
|
+
.describe('How long to wait for the capture before handing back and letting you poll. Ignored when wait is false. Giving up on the wait does not stop the capture.'),
|
|
184
310
|
},
|
|
185
|
-
}, ({ computer_id, name, memory }, extra) => guarded(async () => {
|
|
311
|
+
}, ({ computer_id, name, memory, wait, timeout_s }, extra) => guarded(async () => {
|
|
186
312
|
const id = session.resolve(computer_id);
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
313
|
+
// One deadline for the whole call, armed before the POST, exactly as
|
|
314
|
+
// move_computer arms its own: timeout_s is a promise about when this
|
|
315
|
+
// comes back, and a POST left on undici's own five-minute header clock
|
|
316
|
+
// could break that promise before the poll ever ran. Nothing is armed
|
|
317
|
+
// when nobody is waiting — a caller who asked for the id and no wait
|
|
318
|
+
// was promised no deadline, so imposing one would cancel a POST that
|
|
319
|
+
// was still perfectly capable of starting a capture.
|
|
320
|
+
const untilDeadline = wait ? AbortSignal.timeout(timeout_s * 1000) : undefined;
|
|
321
|
+
const signal = !untilDeadline
|
|
322
|
+
? extra.signal
|
|
323
|
+
: extra.signal
|
|
324
|
+
? AbortSignal.any([extra.signal, untilDeadline])
|
|
325
|
+
: untilDeadline;
|
|
326
|
+
const api = session.api.with(signal);
|
|
327
|
+
// The 202. Its body is a Snapshot row in `state: "capturing"` — a
|
|
328
|
+
// placeholder rather than a snapshot, on which restore, clone and delete
|
|
329
|
+
// all answer 404 until the copy lands (OPL-4562). It is kept whatever
|
|
330
|
+
// happens next, because it is the only description of this capture that
|
|
331
|
+
// does not depend on a later read succeeding.
|
|
332
|
+
let started;
|
|
333
|
+
try {
|
|
334
|
+
started = await api.json('POST', P.computerAction(id, 'snapshots'), {
|
|
335
|
+
body: P.snapshotBody({ memory, name }),
|
|
336
|
+
});
|
|
337
|
+
}
|
|
338
|
+
catch (err) {
|
|
339
|
+
// The deadline, or the caller, arriving while the POST is in flight.
|
|
340
|
+
// The generic sentence for this says the request may have been
|
|
341
|
+
// received and to treat what it would have changed as unknown, which
|
|
342
|
+
// is true and is not enough here: a capture that started is billable,
|
|
343
|
+
// takes minutes, and — because the answer that carried its id is the
|
|
344
|
+
// thing that was lost — cannot be polled for at all. Worse, the
|
|
345
|
+
// obvious next move is wrong. A second capture while one is running is
|
|
346
|
+
// refused 409, so a model that simply retries reads that as a failure
|
|
347
|
+
// on top of a capture that is fine (observed live, OPL-4577).
|
|
348
|
+
if (err instanceof CancelledError) {
|
|
349
|
+
return refused(`${why(err)}\n\nA CAPTURE OF ${id} MAY BE RUNNING. What was lost is the answer that carried ` +
|
|
350
|
+
`its id, so there is nothing here to poll on — look instead: list_snapshots on ${id} shows ` +
|
|
351
|
+
`a row reading "${CAPTURING}" if one started, and that row's id is the one to follow. Do ` +
|
|
352
|
+
`not simply call this again; a second capture while one is running is refused, and the ` +
|
|
353
|
+
`refusal will be about the capture this call may have started.`);
|
|
354
|
+
}
|
|
355
|
+
throw err;
|
|
356
|
+
}
|
|
192
357
|
// The name read back rather than the one sent, because the interesting
|
|
193
358
|
// case is the one that was not sent: the platform generates
|
|
194
359
|
// "<computer> <timestamp>" when `name` is absent, and that generated
|
|
195
360
|
// name is what a later list_snapshots will show. Saying it here is the
|
|
196
361
|
// difference between a caller that can find this capture again and one
|
|
197
362
|
// that has to go looking for it.
|
|
198
|
-
const called = typeof
|
|
199
|
-
|
|
363
|
+
const called = typeof started?.name === 'string' && started.name.trim() ? ` as "${started.name}"` : '';
|
|
364
|
+
// Two sentences for the two things that can be true, and keeping them
|
|
365
|
+
// apart is the whole of OPL-4568. "Snapshotted vm-1" over a 202 is the
|
|
366
|
+
// defect: it is said before a byte has been copied, and a model reads it
|
|
367
|
+
// as a snapshot it may now restore. So the past tense is reserved for a
|
|
368
|
+
// capture this call watched land, and everything that hands back with
|
|
369
|
+
// the copy still running leads with `startedLine` instead.
|
|
370
|
+
const took = `Snapshotted ${id}${memory ? ' with its memory' : ''}${called}.`;
|
|
371
|
+
const startedLine = `Capture of ${id}${memory ? ' with its memory' : ''} started${called}.`;
|
|
372
|
+
// The id allocated before the copy begins, and the whole reason a poll
|
|
373
|
+
// is possible: it is the SNAPSHOT'S own id and does not change when the
|
|
374
|
+
// capture lands, so the row stops reading `capturing` in place rather
|
|
375
|
+
// than being replaced by something under another id.
|
|
376
|
+
const sid = typeof started?.id === 'string' && started.id.trim() ? started.id.trim() : '';
|
|
377
|
+
const startedState = typeof started?.state === 'string' ? started.state : undefined;
|
|
378
|
+
// How to follow it by hand — said wherever this hands back with the
|
|
379
|
+
// capture still running, because "poll list_snapshots" without the two
|
|
380
|
+
// rules under it is an instruction a model gets wrong in both
|
|
381
|
+
// directions: by waiting for the literal `pending`, and by reading a
|
|
382
|
+
// row that vanished as a row that has not appeared yet.
|
|
383
|
+
const byHand = `Poll list_snapshots for the id ${sid}: the row stops reading "${CAPTURING}" when the capture ` +
|
|
384
|
+
`lands — do not wait for "pending" specifically, since replication can carry it straight on to ` +
|
|
385
|
+
`"durable" — and a row that DISAPPEARS without ever leaving "${CAPTURING}" is a capture that failed.`;
|
|
386
|
+
// A capture accepted under an id nobody was told cannot be polled for,
|
|
387
|
+
// and it cannot be found again either — every later call takes the id.
|
|
388
|
+
// `refused`, for clone_snapshot's reason one route over: the platform
|
|
389
|
+
// did something billable and the caller has no handle on it, which is
|
|
390
|
+
// not a result to report as success.
|
|
391
|
+
//
|
|
392
|
+
// BEFORE the wait: false branch, not after it, and Codex caught that it
|
|
393
|
+
// was the other way round. `wait: false` is the answer whose whole
|
|
394
|
+
// content is an id — handed back without one it reported success and
|
|
395
|
+
// told the caller to poll list_snapshots for "the id ", which is the
|
|
396
|
+
// same nothing dressed as an instruction.
|
|
397
|
+
if (!sid) {
|
|
398
|
+
return refused(`${startedLine} THE CAPTURE IS RUNNING, but the platform sent no snapshot id back, so there ` +
|
|
399
|
+
`is nothing to poll on and the snapshot cannot be named — every call that acts on one takes ` +
|
|
400
|
+
`its id. list_snapshots on ${id} will show it when it lands.`, started);
|
|
401
|
+
}
|
|
402
|
+
if (!wait) {
|
|
403
|
+
return said(`${startedLine} It is NOT a snapshot yet: while its row reads "${CAPTURING}" it is a ` +
|
|
404
|
+
`placeholder, and restore, clone and delete all fail on it. ${byHand}`, started);
|
|
405
|
+
}
|
|
406
|
+
// Already finished when it was answered for — a `201` from a platform
|
|
407
|
+
// that has not taken the 202 yet, or a capture with nothing to copy.
|
|
408
|
+
// Not a state to poll out of, and a loop that did would ask for a row it
|
|
409
|
+
// already holds.
|
|
410
|
+
//
|
|
411
|
+
// An ALLOW-LIST, for the reason {@link LANDED} gives: this asks whether
|
|
412
|
+
// there is anything to wait for, and every unreadable answer — absent,
|
|
413
|
+
// misspelt, renamed, moved — has to mean yes. Only a state this server
|
|
414
|
+
// can read AS a landed one skips the wait.
|
|
415
|
+
if (startedState !== undefined && LANDED.includes(startedState)) {
|
|
416
|
+
return said(`${took} It is ${startedState} — snapshot ${sid}.`, started);
|
|
417
|
+
}
|
|
418
|
+
// The keepalive. Armed after the 202, because everything before it is a
|
|
419
|
+
// single short request and there is nothing to report until there is a
|
|
420
|
+
// capture to report on.
|
|
421
|
+
const beat = heartbeat(extra, server.server);
|
|
422
|
+
await beat(`Capturing ${sid} of ${id} — the copy has started.`);
|
|
423
|
+
let blocked;
|
|
424
|
+
// `untilDeadline &&` rather than `!untilDeadline?.aborted`: the early
|
|
425
|
+
// return above is what guarantees it is armed here, and the optional
|
|
426
|
+
// form would turn a later edit that broke that guarantee into a loop
|
|
427
|
+
// with no deadline at all rather than into a visible mistake.
|
|
428
|
+
while (untilDeadline && !untilDeadline.aborted) {
|
|
429
|
+
// The caller giving up ends the wait. The signal aborts the request in
|
|
430
|
+
// flight, but nothing about an aborted request stops the next
|
|
431
|
+
// iteration from starting one.
|
|
432
|
+
if (extra.signal?.aborted) {
|
|
433
|
+
return refused(`Cancelled while waiting for the capture of ${id}. THE CAPTURE IS STILL RUNNING — nothing ` +
|
|
434
|
+
`was stopped, because a disk copy already under way cannot be called back. ${byHand}`, started);
|
|
435
|
+
}
|
|
436
|
+
const turn = await snapshotTurn(api, sid, false);
|
|
437
|
+
if (extra.signal?.aborted)
|
|
438
|
+
continue;
|
|
439
|
+
// A body stream can fail without either signal firing, so the
|
|
440
|
+
// deadline's own arrival is what tells a real timeout from an undici
|
|
441
|
+
// idle abort. Same shape as the two loops in computers.ts.
|
|
442
|
+
if (turn.kind === 'cancelled') {
|
|
443
|
+
if (untilDeadline?.aborted)
|
|
444
|
+
break;
|
|
445
|
+
blocked = turn.why;
|
|
446
|
+
await beat(`Capturing ${sid} of ${id} — the platform could not be asked: ${turn.why}`);
|
|
447
|
+
await sleep(POLL_MS, signal);
|
|
448
|
+
continue;
|
|
449
|
+
}
|
|
450
|
+
// A poll that failed for a reason worth riding out is weather; one
|
|
451
|
+
// that failed for any other reason is a real failure with a capture
|
|
452
|
+
// still running behind it, and a thrown error's handler has no way to
|
|
453
|
+
// say so. So it is said here rather than rethrown.
|
|
454
|
+
if (turn.kind === 'broken') {
|
|
455
|
+
return refused(`${turn.why}\n\nTHE CAPTURE IS STILL RUNNING — this was the poll failing, not the ` +
|
|
456
|
+
`capture. ${byHand}`, started);
|
|
457
|
+
}
|
|
458
|
+
if (turn.kind === 'blocked') {
|
|
459
|
+
blocked = turn.why;
|
|
460
|
+
await beat(`Capturing ${sid} of ${id} — the platform could not be asked: ${turn.why}`);
|
|
461
|
+
await sleep(turn.after ?? POLL_MS, signal);
|
|
462
|
+
continue;
|
|
463
|
+
}
|
|
464
|
+
// Absence is the ONLY signal a failed capture leaves, which is what
|
|
465
|
+
// makes snapshotTurn's guards load-bearing rather than tidy: a body
|
|
466
|
+
// that is not a list, and a listing the platform had to answer short,
|
|
467
|
+
// both come back `blocked` there rather than as this, because read as
|
|
468
|
+
// "the row is not there" they would announce that somebody's backup
|
|
469
|
+
// failed while it was still being taken.
|
|
470
|
+
if (turn.kind === 'absent') {
|
|
471
|
+
return refused(`THE CAPTURE OF ${id} FAILED and nothing was saved. Snapshot ${sid} is no longer listed and ` +
|
|
472
|
+
`nothing took its place, which is what a capture that starts and then fails leaves behind — ` +
|
|
473
|
+
`this is a failure during the copy, not a wait that ran out. The computer itself is ` +
|
|
474
|
+
`untouched and can be snapshotted again.`, started);
|
|
475
|
+
}
|
|
476
|
+
const row = turn.row;
|
|
477
|
+
const state = typeof row.state === 'string' ? row.state : undefined;
|
|
478
|
+
if (state === CAPTURING) {
|
|
479
|
+
blocked = undefined;
|
|
480
|
+
await beat(`Capturing ${sid} of ${id} — still copying.`);
|
|
481
|
+
await sleep(POLL_MS, signal);
|
|
482
|
+
continue;
|
|
483
|
+
}
|
|
484
|
+
// A row with no readable `state` is the platform failing to describe
|
|
485
|
+
// it, and it must not be read as "not capturing, therefore landed" —
|
|
486
|
+
// that sentence says a placeholder may be restored, which is the
|
|
487
|
+
// defect this whole tool is here to stop. Only a state that SAYS it is
|
|
488
|
+
// no longer capturing ends the wait; anything else rides out, and the
|
|
489
|
+
// deadline reports that the platform could not be asked.
|
|
490
|
+
if (state === undefined) {
|
|
491
|
+
blocked = `the row for ${sid} carried no state, so nothing said whether the capture had landed`;
|
|
492
|
+
await beat(`Capturing ${sid} of ${id} — its row carried no state to read.`);
|
|
493
|
+
await sleep(POLL_MS, signal);
|
|
494
|
+
continue;
|
|
495
|
+
}
|
|
496
|
+
return said(`${took} The capture landed: snapshot ${sid} reads "${state}" and can now be restored, ` +
|
|
497
|
+
`cloned or deleted.`, row);
|
|
498
|
+
}
|
|
499
|
+
// A refusal, for the reason move_computer's is: the wait never reached
|
|
500
|
+
// what it was told to wait for. What it must not do is read as a capture
|
|
501
|
+
// that failed — that is a different answer with a different id in it,
|
|
502
|
+
// and the difference between calling this again and going looking for a
|
|
503
|
+
// snapshot that is on its way.
|
|
504
|
+
return refused(blocked
|
|
505
|
+
? `${startedLine} Gave up watching after ${timeout_s}s; the platform could not be asked — the ` +
|
|
506
|
+
`last attempt said: ${blocked}. THE CAPTURE IS STILL RUNNING. ${byHand}`
|
|
507
|
+
: `${startedLine} Still capturing after ${timeout_s}s, which a large disk takes. THE CAPTURE IS ` +
|
|
508
|
+
`STILL RUNNING and nothing was changed by giving up on the wait. ${byHand}`, started);
|
|
200
509
|
}));
|
|
201
510
|
server.registerTool('restore_snapshot', {
|
|
202
511
|
title: 'Restore a snapshot',
|
|
@@ -246,7 +555,21 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
246
555
|
if (clear)
|
|
247
556
|
return said('Schedule cleared.', await session.api.with(extra.signal).send('DELETE', path));
|
|
248
557
|
if (set) {
|
|
249
|
-
|
|
558
|
+
const at = `${String(set.hour).padStart(2, '0')}:${String(set.minute).padStart(2, '0')} ${set.tz}`;
|
|
559
|
+
const res = await session.api
|
|
560
|
+
.with(extra.signal)
|
|
561
|
+
.send('PUT', path, { body: P.scheduleBody(set) });
|
|
562
|
+
// `enabled` is a required field of `set`, and it is the one that
|
|
563
|
+
// decides whether backups happen at all — so the sentence has to read
|
|
564
|
+
// it. "Snapshot scheduled for 03:00 UTC" over `enabled: false` tells a
|
|
565
|
+
// model the opposite of what the call just did, in the line it acts
|
|
566
|
+
// on, and the hour it names is real, which is what makes the wrong
|
|
567
|
+
// reading easy to believe. The hour is kept on a disabled schedule, so
|
|
568
|
+
// it is worth naming: a model that reads "disabled" and nothing else
|
|
569
|
+
// calls this again to find out what it would have run at.
|
|
570
|
+
return said(set.enabled
|
|
571
|
+
? `Snapshot scheduled for ${at}.`
|
|
572
|
+
: `The nightly snapshot on ${id} is now DISABLED — none will be taken automatically. ${at} is the time it holds, which is when it would resume if you set it again with enabled: true. To remove the schedule rather than turn it off, call this with clear.`, res);
|
|
250
573
|
}
|
|
251
574
|
return json(await session.api.with(extra.signal).json('GET', path));
|
|
252
575
|
}));
|
|
@@ -295,17 +618,85 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
295
618
|
}));
|
|
296
619
|
server.registerTool('delete_snapshot', {
|
|
297
620
|
title: 'Delete a snapshot',
|
|
298
|
-
description: 'Remove a snapshot permanently. Later snapshots in the same chain are unaffected.',
|
|
621
|
+
description: 'Remove a snapshot permanently. Later snapshots in the same chain are unaffected. THE DELETION OUTLIVES THE REQUEST that starts it: the platform accepts it and then detaches the dependent snapshots and removes the stored objects, which takes time that scales with the chain and with how much is stored. This waits for it by default and reports what actually happened; pass wait: false to hand back as soon as it is accepted. A 409 saying the snapshot is ALREADY BEING DELETED is progress rather than a fault — the platform is doing what you asked, and the answer is to watch that one finish, never to go and delete something else. A wait reports progress while it runs, so a client that sends a progressToken and sets resetTimeoutOnProgress can hold the request open; a client that cannot should pass wait: false and poll list_snapshots.',
|
|
299
622
|
inputSchema: {
|
|
300
|
-
snapshot_id: z.string(),
|
|
623
|
+
snapshot_id: z.string().trim(),
|
|
301
624
|
confirm: z.literal(true).describe('Must be true.'),
|
|
625
|
+
wait: z
|
|
626
|
+
.boolean()
|
|
627
|
+
.default(true)
|
|
628
|
+
.describe('Wait for the snapshot to stop being listed, which is the only thing that means it is gone. Set it false to hand back as soon as the platform accepts the deletion, and then poll list_snapshots yourself.'),
|
|
629
|
+
timeout_s: z
|
|
630
|
+
.number()
|
|
631
|
+
.int()
|
|
632
|
+
.min(5)
|
|
633
|
+
.max(1800)
|
|
634
|
+
.default(300)
|
|
635
|
+
.describe('How long to wait for the deletion before handing back. Ignored when wait is false. Giving up on the wait does not stop the deletion.'),
|
|
302
636
|
},
|
|
303
637
|
annotations: { destructiveHint: true, idempotentHint: true },
|
|
304
|
-
}, ({ snapshot_id }, extra) => guarded(async () => {
|
|
638
|
+
}, ({ snapshot_id, wait, timeout_s }, extra) => guarded(async () => {
|
|
639
|
+
// The deadline is armed before the DELETE, as create_snapshot arms its
|
|
640
|
+
// own and for the same reason: timeout_s is a promise about when this
|
|
641
|
+
// comes back. Nothing is armed when nobody is waiting.
|
|
642
|
+
const untilDeadline = wait ? AbortSignal.timeout(timeout_s * 1000) : undefined;
|
|
643
|
+
const signal = !untilDeadline
|
|
644
|
+
? extra.signal
|
|
645
|
+
: extra.signal
|
|
646
|
+
? AbortSignal.any([extra.signal, untilDeadline])
|
|
647
|
+
: untilDeadline;
|
|
648
|
+
const api = session.api.with(signal);
|
|
649
|
+
// How to follow it by hand, said wherever this hands back with the
|
|
650
|
+
// deletion still running. The polarity is the whole of it and it is the
|
|
651
|
+
// opposite of a capture's: there is no state that means deleted, so the
|
|
652
|
+
// row GOING is the finish, and a row that stays is one that stalled.
|
|
653
|
+
// `include_unfinished` is not a nicety either — the platform marks a
|
|
654
|
+
// half-finished deletion `deleting` and leaves that state out of a bare
|
|
655
|
+
// listing, so without the flag a stalled deletion looks exactly like a
|
|
656
|
+
// finished one.
|
|
657
|
+
const byHand = `Poll list_snapshots with include_unfinished: true and watch for ${snapshot_id} to stop being ` +
|
|
658
|
+
`listed — its absence is the deletion having finished, and there is no state that means deleted. ` +
|
|
659
|
+
`A row that STAYS is one that stalled; the platform retries those itself every fifteen minutes.`;
|
|
660
|
+
let accepted;
|
|
305
661
|
try {
|
|
306
|
-
await
|
|
662
|
+
accepted = await api.send('DELETE', P.snapshot(snapshot_id));
|
|
307
663
|
}
|
|
308
664
|
catch (err) {
|
|
665
|
+
// Every refusal is still decided before the 202 — no such snapshot, a
|
|
666
|
+
// capture reading through it, a clone or a migration holding it, a
|
|
667
|
+
// deletion of this id already running — so a 409 here is a statement
|
|
668
|
+
// about state that clears itself, and the model needs to be told
|
|
669
|
+
// which way to read it rather than left to invent a way round it.
|
|
670
|
+
//
|
|
671
|
+
// Matched on the CLASS and never on the sentence: the platform sends
|
|
672
|
+
// no `reason` for these, and keying on prose is the mistake OPL-3724
|
|
673
|
+
// took out of three clients. So both readings are named and the
|
|
674
|
+
// platform's own words are printed with them.
|
|
675
|
+
// The deadline, or the caller, arriving while the DELETE is in
|
|
676
|
+
// flight. The generic cancellation sentence says the request may have
|
|
677
|
+
// been received; what it cannot say is that the work it started
|
|
678
|
+
// OUTLIVES the request, so "cancelled" here reads as a deletion that
|
|
679
|
+
// did not happen while the objects are being removed. Claiming less
|
|
680
|
+
// than happened, which on a destructive call is its own kind of wrong
|
|
681
|
+
// answer (codex review, gpt-5.6-sol).
|
|
682
|
+
//
|
|
683
|
+
// The retry rule is worth saying in the same breath, because it is
|
|
684
|
+
// the one thing that makes this recoverable: repeating the call is
|
|
685
|
+
// safe. A snapshot already being deleted answers 409, and one that is
|
|
686
|
+
// gone answers 404, and this tool has a sentence for each.
|
|
687
|
+
if (err instanceof CancelledError) {
|
|
688
|
+
return refused(`${why(err)}\n\nTHE DELETION OF ${snapshot_id} MAY BE RUNNING: the request may have reached ` +
|
|
689
|
+
`the platform and been accepted, and nothing about the answer being lost calls it back. ` +
|
|
690
|
+
`${byHand} Calling this again is safe either way — a snapshot already being deleted ` +
|
|
691
|
+
`answers a conflict, and one that is gone answers that there is nothing to delete.`);
|
|
692
|
+
}
|
|
693
|
+
if (err instanceof ConflictError) {
|
|
694
|
+
return refused(`${why(err)}\n\nNOTHING WAS DELETED and nothing is broken. If that says the snapshot is ` +
|
|
695
|
+
`already being deleted, the platform is doing what you asked — watch it finish rather ` +
|
|
696
|
+
`than deleting anything else: ${byHand} If it says something is reading through the ` +
|
|
697
|
+
`snapshot — a restore, a clone, a capture chaining onto it — that finishes on its own and ` +
|
|
698
|
+
`the same call works afterwards.`);
|
|
699
|
+
}
|
|
309
700
|
// A 404 means the snapshot is not there, which is the state this call
|
|
310
701
|
// was asking for. `idempotentHint` above invites a client to retry a
|
|
311
702
|
// lost 2xx, and `#fetch` throws on every non-OK — so that invited
|
|
@@ -327,7 +718,74 @@ export const registerSnapshots = (server, session, opts) => {
|
|
|
327
718
|
'a real one may still be held under the id you meant. list_snapshots says which of the two ' +
|
|
328
719
|
'this is.');
|
|
329
720
|
}
|
|
330
|
-
|
|
721
|
+
// The 202 and its body: the snapshot's row as it stood when the deletion
|
|
722
|
+
// was accepted — the row that GOES when the work finishes.
|
|
723
|
+
if (!wait) {
|
|
724
|
+
return said(`Deletion of ${snapshot_id} accepted and RUNNING — it is not gone yet, and this call did not ` +
|
|
725
|
+
`wait to find out. ${byHand}`, accepted);
|
|
726
|
+
}
|
|
727
|
+
const beat = heartbeat(extra, server.server);
|
|
728
|
+
await beat(`Deleting ${snapshot_id} — the platform has accepted it.`);
|
|
729
|
+
let blocked;
|
|
730
|
+
let seen;
|
|
731
|
+
while (untilDeadline && !untilDeadline.aborted) {
|
|
732
|
+
if (extra.signal?.aborted) {
|
|
733
|
+
return refused(`Cancelled while waiting for ${snapshot_id} to be deleted. THE DELETION IS STILL RUNNING — ` +
|
|
734
|
+
`nothing was called back, and objects it has already removed are already gone. ${byHand}`, accepted);
|
|
735
|
+
}
|
|
736
|
+
// `unfinished`, because the state a stalled deletion sits in is the
|
|
737
|
+
// one a bare listing hides. Asking without it would read a snapshot
|
|
738
|
+
// stuck half-deleted as a snapshot successfully deleted, which is the
|
|
739
|
+
// one wrong answer this tool must not give.
|
|
740
|
+
const turn = await snapshotTurn(api, snapshot_id, true);
|
|
741
|
+
if (extra.signal?.aborted)
|
|
742
|
+
continue;
|
|
743
|
+
if (turn.kind === 'cancelled') {
|
|
744
|
+
if (untilDeadline.aborted)
|
|
745
|
+
break;
|
|
746
|
+
blocked = turn.why;
|
|
747
|
+
await beat(`Deleting ${snapshot_id} — the platform could not be asked: ${turn.why}`);
|
|
748
|
+
await sleep(POLL_MS, signal);
|
|
749
|
+
continue;
|
|
750
|
+
}
|
|
751
|
+
if (turn.kind === 'broken') {
|
|
752
|
+
return refused(`${turn.why}\n\nTHE DELETION IS STILL RUNNING — this was the poll failing, not the ` +
|
|
753
|
+
`deletion. ${byHand}`, accepted);
|
|
754
|
+
}
|
|
755
|
+
if (turn.kind === 'blocked') {
|
|
756
|
+
blocked = turn.why;
|
|
757
|
+
await beat(`Deleting ${snapshot_id} — the platform could not be asked: ${turn.why}`);
|
|
758
|
+
await sleep(turn.after ?? POLL_MS, signal);
|
|
759
|
+
continue;
|
|
760
|
+
}
|
|
761
|
+
// The row is gone from a listing that was read whole and that ASKED
|
|
762
|
+
// for the unfinished ones. Both halves are what make this sentence
|
|
763
|
+
// true rather than merely likely.
|
|
764
|
+
if (turn.kind === 'absent')
|
|
765
|
+
return said(`Deleted snapshot ${snapshot_id}.`, accepted);
|
|
766
|
+
blocked = undefined;
|
|
767
|
+
seen = typeof turn.row.state === 'string' ? turn.row.state : undefined;
|
|
768
|
+
await beat(`Deleting ${snapshot_id} — still listed${seen ? `, reading "${seen}"` : ''}.`);
|
|
769
|
+
await sleep(POLL_MS, signal);
|
|
770
|
+
}
|
|
771
|
+
// Still listed. Three different things that can mean, and they are not
|
|
772
|
+
// one sentence: the platform could not be asked, the deletion got part
|
|
773
|
+
// way and stalled, or it never got started on — which is the shape the
|
|
774
|
+
// one conflict that arrives AFTER the 202 leaves behind, a dependent
|
|
775
|
+
// that is itself being deleted and so cannot be detached.
|
|
776
|
+
return refused(blocked
|
|
777
|
+
? `Gave up watching after ${timeout_s}s; the platform could not be asked whether ${snapshot_id} ` +
|
|
778
|
+
`is gone — the last attempt said: ${blocked}. THE DELETION IS STILL RUNNING. ${byHand}`
|
|
779
|
+
: seen === DELETING
|
|
780
|
+
? `${snapshot_id} is still listed after ${timeout_s}s and reads "${DELETING}": the deletion ` +
|
|
781
|
+
`started and has not finished. Nothing was undone by giving up on the wait, and the ` +
|
|
782
|
+
`platform retries a stalled deletion itself every fifteen minutes — so this usually needs ` +
|
|
783
|
+
`watching rather than repeating. ${byHand}`
|
|
784
|
+
: `${snapshot_id} is still listed after ${timeout_s}s${seen ? `, reading "${seen}"` : ''} — ` +
|
|
785
|
+
`the deletion has not reached the point of marking it "${DELETING}". That is what a big ` +
|
|
786
|
+
`chain looks like early on, and it is also what the one conflict that arrives after the ` +
|
|
787
|
+
`deletion is accepted looks like: a dependent snapshot that is ITSELF being deleted cannot ` +
|
|
788
|
+
`be detached, so this one waits for that one. Nothing was destroyed either way. ${byHand}`, accepted);
|
|
331
789
|
}));
|
|
332
790
|
};
|
|
333
791
|
//# sourceMappingURL=snapshots.js.map
|