pi-durable-subagents 1.0.19 → 1.0.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +63 -0
- package/README.md +115 -12
- package/dist/agent/main/schema.js +25 -0
- package/dist/agent/main/tool.js +23 -24
- package/dist/agent/main.js +37 -16
- package/dist/cli/control.js +43 -20
- package/dist/cli/main.js +54 -17
- package/dist/cli/requests.js +403 -0
- package/dist/cli/restart.js +34 -16
- package/dist/kernel/lifecycle.js +2 -1
- package/dist/orchestrator/engine.js +33 -19
- package/dist/orchestrator/executor/index.js +42 -0
- package/dist/orchestrator/ledger.js +9 -3
- package/dist/orchestrator/restart.js +57 -0
- package/dist/orchestrator/snapshot.js +7 -4
- package/dist/orchestrator/store.js +83 -8
- package/dist/requests.js +87 -0
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,68 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.0.22
|
|
4
|
+
|
|
5
|
+
- The installed CLI runs `run`, `send` and `stop` again: 1.0.21 loaded the
|
|
6
|
+
optional pi peer package from the tool schema and failed with
|
|
7
|
+
`ERR_MODULE_NOT_FOUND` where it does not resolve (pi's own npm prefix,
|
|
8
|
+
`npm i -g`). The schema lives apart now; the import check rejects any pi
|
|
9
|
+
import reachable from the CLI, orchestrator or evaluator, and `pack:smoke`
|
|
10
|
+
drives a run by request id from a package directory that cannot see pi.
|
|
11
|
+
- `describe` reports `lastFence` only for an execution that was cut off; an
|
|
12
|
+
execution that ended its turn, hibernated, or was stopped, timed out or
|
|
13
|
+
over budget, or hibernated on its question (also when recovery finds only its
|
|
14
|
+
`ask` was running) is not an interruption. A `once` call sealed `unknown` and a call
|
|
15
|
+
sealed after repeated losses still name their fence.
|
|
16
|
+
- With `--json`, a `run`/`send`/`stop` refused before submission (unknown
|
|
17
|
+
agent, invalid spec, usage) answers `{request, applied: false, reason,
|
|
18
|
+
spec_digest?}` instead of plain text; exit code 1 as before. A failure after
|
|
19
|
+
submission, and a retry of a recorded run whose agent has since gone, are
|
|
20
|
+
pending (75), never a refusal; other content under a recorded id is a
|
|
21
|
+
`request-conflict` (3) before any agent check.
|
|
22
|
+
- The import check parses with TypeScript, follows the executor, which the
|
|
23
|
+
orchestrator loads through `import(new URL(…))`, and treats only
|
|
24
|
+
`import type` as erased (`import { type X }` still loads the module).
|
|
25
|
+
|
|
26
|
+
## 1.0.21
|
|
27
|
+
|
|
28
|
+
1.0.20 was not published; its changes ship in this release.
|
|
29
|
+
|
|
30
|
+
- Programs can name requests: `run`, `send` and `stop` take `--request <id>`
|
|
31
|
+
(CLI) or `request` (the `subagents` tool). A retry with the same id and the
|
|
32
|
+
same content gets the first outcome and never starts a second workflow or
|
|
33
|
+
follow-up; other content under the id is a `request-conflict` and sends
|
|
34
|
+
nothing. Exit codes: 0 applied, 1 rejected, 3 conflict, 75 not decided yet
|
|
35
|
+
(retry with the same id). `describe --key <id>` (or a wid) reports the
|
|
36
|
+
state, open questions and every call's output in full, what live calls
|
|
37
|
+
wait for and why their last execution was fenced. A pruned workflow leaves
|
|
38
|
+
a tombstone with its final status, request id and digest. See "Driving dsa
|
|
39
|
+
from a program" in the README.
|
|
40
|
+
- A forced restart needs the user's approval in a checkable form: a refused
|
|
41
|
+
`restart` lists the running executions grouped by session (with held
|
|
42
|
+
leases) and a token; `restart --force <token> --reason <text>` fences
|
|
43
|
+
exactly that set and is refused again if it changed. A subagent cannot
|
|
44
|
+
force a restart. The ledger records who forced it and why, and `status`
|
|
45
|
+
shows it for 24 hours.
|
|
46
|
+
- An asker cut off by a restart or crash before its planned hibernation no
|
|
47
|
+
longer loses its question: it hibernates and resumes with the answer
|
|
48
|
+
(also an answer given while dsa restarted), and a `once` step waiting for
|
|
49
|
+
an answer no longer ends as `unknown`.
|
|
50
|
+
- A workflow's done notice names a follow-up still going (queued or running)
|
|
51
|
+
instead of calling it `unknown`.
|
|
52
|
+
- Orchestrator starts are fast again. Each start re-parsed every workflow's
|
|
53
|
+
staged snapshot (tens of megabytes each when it holds a forked parent
|
|
54
|
+
session) before handling any request, so a restart stalled all work for
|
|
55
|
+
over a minute. A verified `pins.json` record per revision now replaces that
|
|
56
|
+
work: on a copy of a home with 117 workflows, recovery takes about 6 s
|
|
57
|
+
instead of 80 s. The first start after the update builds the records once.
|
|
58
|
+
A record that does not verify falls back to the snapshot, so a changed
|
|
59
|
+
pinned file is still reported as a conflict.
|
|
60
|
+
- `restart` no longer reports a failure while the orchestrator is still
|
|
61
|
+
recovering: it says the request is submitted and keeps waiting (up to 10
|
|
62
|
+
minutes), then exits 75 with "still pending — do not resubmit" if no
|
|
63
|
+
orchestrator reached it. After the restart it waits for the orchestrator
|
|
64
|
+
that actually decided it.
|
|
65
|
+
|
|
3
66
|
## 1.0.19
|
|
4
67
|
|
|
5
68
|
- The orchestrator no longer keeps every workflow's pinned origin branch (the
|
package/README.md
CHANGED
|
@@ -57,7 +57,9 @@ the npx cache, so `install-service` refuses to run from there.
|
|
|
57
57
|
| A step is refused, or a dependency fails | The workflow stops that branch cleanly. Nothing is retried in vain. |
|
|
58
58
|
| A provider's usage window runs out (`No available accounts`, usage limit, quota exceeded) | Found at the second refusal in a row, while pi is still retrying. A call in a pool continues **in the same session** on the pool's next model (within pi's next retry or two); new calls skip that provider. After 15 minutes the next call that wants it tries it once; when it answers, new calls and new generations use it again. A call with a single model waits for it instead of failing. Billing errors (402, insufficient balance) still fail at once. |
|
|
59
59
|
| Two subagents would write in the same worktree | Only one runs there at a time. A call that can write (its tools include `edit` or `write`, which pi's default tools do) holds its git worktree's writer lock from its launch until it ends, also while it waits for an answer. Another writer for that worktree waits in order, and status shows `waiting for writer lock: <root> held by <wid>/<key>`. `writer: false` (a call that does not write there), `isolation: "worktree"` and `"writerLock": "off"` opt out. |
|
|
60
|
-
| A subagent waits for an answer for a long time | It releases its model slot and memory, then resumes exactly once when you answer. |
|
|
60
|
+
| A subagent waits for an answer for a long time | It releases its model slot and memory, then resumes exactly once when you answer. The question survives orchestrator restarts (also forced ones) and crashes, including one that hits before the subagent released its slot. |
|
|
61
|
+
| A subagent's work ends (finished, stopped, or cut off) | Every process its tools started ends with that execution, also ones started with `nohup`, `setsid` or `&`: they carry the execution's tag (see the limit below). Anything that must outlive the subagent has to be started by you or the parent session. A command run under `hold` is no exception: a forced restart stops it and its lease is released. |
|
|
62
|
+
| A subagent runs in a worktree you made (`isolation: "none"`, the default, with `cwd`) | Durable Subagents never creates, cleans, moves or deletes that directory or its branch, also not on `prune`; `prune` deletes only its own state and the worktrees it created for `isolation: "worktree"`. |
|
|
61
63
|
|
|
62
64
|
## Use it
|
|
63
65
|
|
|
@@ -238,11 +240,18 @@ pi-durable-subagents start start the orchestrator if work is pendin
|
|
|
238
240
|
pi-durable-subagents resume [wid] continue unfinished or parked work (undoes drain / stop-all)
|
|
239
241
|
pi-durable-subagents drain hold existing workflows: running calls finish, nothing new starts in them
|
|
240
242
|
pi-durable-subagents stop <wid|call>
|
|
243
|
+
pi-durable-subagents run --request <id> --spec <file|-> [--cwd <dir>] [--json] [--wait-ms <n>]
|
|
244
|
+
start a run under a caller-chosen id; safe to retry (see below)
|
|
245
|
+
pi-durable-subagents send --request <id> --to <run-id|wid/key> --kind follow-up|answer|steer|model
|
|
246
|
+
[--call <key>] [--qid <qid> --rev <n>] --message <text|@file> [--model <m>] [--json]
|
|
247
|
+
pi-durable-subagents stop --request <id> <run-id|wid|wid/key|call> [--json]
|
|
248
|
+
pi-durable-subagents describe --key <run-id> | <wid> [--json]
|
|
249
|
+
full read-only state of one run (never starts the orchestrator)
|
|
241
250
|
pi-durable-subagents stop-all pause every existing workflow now; journals stay resumable
|
|
242
251
|
(runs you start afterwards are not held)
|
|
243
252
|
pi-durable-subagents prune [wid] [--older-than <days>]
|
|
244
253
|
delete finished workflows (done, failed, stopped); prints count and bytes freed
|
|
245
|
-
pi-durable-subagents restart [--force] switch to the installed version (see "Updating Durable Subagents")
|
|
254
|
+
pi-durable-subagents restart [--force <token> --reason <text>] switch to the installed version (see "Updating Durable Subagents")
|
|
246
255
|
pi-durable-subagents hold <resource> [--shared] [--max-wait <s>] [--note <text>] -- <command…>
|
|
247
256
|
run one command while holding a resource lease (see below)
|
|
248
257
|
pi-durable-subagents leases [--json] who holds and who waits for each resource
|
|
@@ -255,6 +264,71 @@ The service only runs `start`: it never resumes work you drained or
|
|
|
255
264
|
stopped. Install the CLI globally (`npm i -g pi-durable-subagents`) before
|
|
256
265
|
`install-service`.
|
|
257
266
|
|
|
267
|
+
### Driving dsa from a program
|
|
268
|
+
|
|
269
|
+
A program (a CI job, a script, another agent) should name its requests with
|
|
270
|
+
`--request <id>`: 1–124 characters `[A-Za-z0-9][A-Za-z0-9._:-]*`, unique per
|
|
271
|
+
`DSA_HOME` across `run`, `send` and `stop`. The id is the identity: the first
|
|
272
|
+
submission under an id is decided once and every retry with **the same
|
|
273
|
+
content** gets that first outcome; a retry never starts a second workflow or
|
|
274
|
+
a second follow-up. The run id is also the workflow key: `send --to <run-id>`
|
|
275
|
+
and `describe --key <run-id>` find the workflow without knowing its wid.
|
|
276
|
+
|
|
277
|
+
The content is hashed into `spec_digest` (the request kind, its body and, for
|
|
278
|
+
answers, the question id and revision). Persist the exact spec bytes you
|
|
279
|
+
submit and retry with those bytes: the run's `cwd` is part of the content
|
|
280
|
+
(`spec.cwd`, else `--cwd`, else the current directory, made absolute), so
|
|
281
|
+
retry from the same place or pass it explicitly. The `spec` file is the
|
|
282
|
+
`subagents` tool's run form (`{agent, task, model?, schema?, …}` or
|
|
283
|
+
`{tasks: […], name?, usageBudget?, maxCalls?}`); `context: "fork"` needs a pi
|
|
284
|
+
session and is rejected.
|
|
285
|
+
|
|
286
|
+
| exit | meaning | `--json` reply |
|
|
287
|
+
| --- | --- | --- |
|
|
288
|
+
| 0 | decided and applied (a retry gets the same answer) | run: `{request, wid, created, spec_digest}`; send/stop: `{request, applied, generation?, call?, spec_digest}` |
|
|
289
|
+
| 1 | decided and rejected (`reason`), or refused before submission (usage error, invalid spec, unknown agent: this invocation submitted nothing) | `{request, applied: false, reason, spec_digest?}` (`spec_digest` once the content could be hashed) |
|
|
290
|
+
| 3 | `request-conflict`: the id already names other content; nothing was sent | `{request, error, wid?, spec_digest, state}` (the original's digest and state) |
|
|
291
|
+
| 75 | not decided yet: submitted but undecided in `--wait-ms` (default 60 s), submitted and then a later step failed (`reason` says which), or a lock was busy (`reason: "busy"`); an id already recorded is never refused again (the same content gets its first outcome, its agents are not rechecked; other content is a conflict); retry with the same id | `{request, pending: true, reason?}` |
|
|
292
|
+
|
|
293
|
+
`created` is false when the id had already been decided before the command
|
|
294
|
+
ran. A conflicting id stays conflicting forever, also after `prune`: the
|
|
295
|
+
tombstone keeps the final status, the id and the digest. Any sender's retry
|
|
296
|
+
completes a submission another one recorded but did not publish (it died in
|
|
297
|
+
between), so a retry with the same content always converges.
|
|
298
|
+
|
|
299
|
+
From the `subagents` tool, `request` works the same, with one difference: a
|
|
300
|
+
run's origin session (what `context: "fork"` copies, and where notices go) is
|
|
301
|
+
not part of the digest. A retry of the same id from another session therefore
|
|
302
|
+
gets the first session's workflow, its notices and its forked context. A
|
|
303
|
+
retried answer may leave out `to`, `qid` and `rev`: it addresses the question
|
|
304
|
+
the first attempt answered.
|
|
305
|
+
|
|
306
|
+
`describe` reports one of `absent` (never seen), `pending` (submitted, not
|
|
307
|
+
decided — typically no orchestrator is running; `start` or a retry starts it),
|
|
308
|
+
`rejected` (with `reason`), `running`, `asking` (open questions with their
|
|
309
|
+
full text, `qid`, `rev` and the `to` address to answer), `sealed` (finished:
|
|
310
|
+
`status` plus every call's unclipped `output`, `error` and schema `data`) or
|
|
311
|
+
`pruned` (`pruned: {status, endedAt}`; workflows pruned before 1.0.21 have
|
|
312
|
+
only `endedAt`), with `wid`, `request` and
|
|
313
|
+
`spec_digest`. Live calls also show what they wait for (slot, writer lock,
|
|
314
|
+
lease, exhausted provider). `lastFence: {at, exec, reason}` appears only
|
|
315
|
+
when an execution was cut off: it had not ended its turn when it was fenced,
|
|
316
|
+
did not hibernate on a question (also when recovery finds it cut off while only
|
|
317
|
+
its question's `ask` ran), and was not ended on purpose (stop, timeout,
|
|
318
|
+
budget). Every execution ends with a fence, so an ordinary end is not
|
|
319
|
+
reported; a `once` call sealed `unknown` or a call sealed after repeated losses
|
|
320
|
+
is. The reason is a best-effort reading of the journals: `restart-force` when a forced restart listed the execution,
|
|
321
|
+
`orchestrator-crash` when the orchestrator died uncleanly while it ran, else
|
|
322
|
+
`process-died`.
|
|
323
|
+
|
|
324
|
+
```sh
|
|
325
|
+
pi-durable-subagents run --request build-42 --spec build-42.json --json
|
|
326
|
+
# exit 75: retry later with the same id and bytes
|
|
327
|
+
pi-durable-subagents describe --key build-42 --json
|
|
328
|
+
pi-durable-subagents send --request build-42-a1 --to build-42 --kind answer \
|
|
329
|
+
--qid <qid> --rev <rev> --message "yes"
|
|
330
|
+
```
|
|
331
|
+
|
|
258
332
|
### Housekeeping
|
|
259
333
|
|
|
260
334
|
Journals are never compacted, so state only grows. `prune` removes finished
|
|
@@ -387,23 +461,45 @@ pi-durable-subagents restart # or the subagents tool: action "restart"
|
|
|
387
461
|
```
|
|
388
462
|
|
|
389
463
|
The orchestrator refuses while any execution runs (a subagent process, or a
|
|
390
|
-
gate before a call's seal)
|
|
391
|
-
|
|
464
|
+
gate before a call's seal). The refusal groups executions by session with ages,
|
|
465
|
+
lease annotations and a token for that exact set; no new execution starts while
|
|
466
|
+
it decides, so nothing slips in between. Calls waiting
|
|
392
467
|
for your answer (hibernated), waiting for a provider slot, or held by a drain
|
|
393
468
|
do not block it. Otherwise it exits and its successor starts at once from the
|
|
394
469
|
installed files and resumes every workflow: an asker keeps its question, a
|
|
395
|
-
queued call launches on the new version.
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
470
|
+
queued call launches on the new version.
|
|
471
|
+
|
|
472
|
+
To interrupt those executions deliberately, first show the user the refusal's
|
|
473
|
+
list and obtain their explicit approval, then use its token and a non-empty
|
|
474
|
+
reason (at most 500 characters):
|
|
475
|
+
|
|
476
|
+
```sh
|
|
477
|
+
pi-durable-subagents restart --force <token> --reason "<why>"
|
|
478
|
+
# tool: {action:"restart", force:"<token>", reason:"<why>"}
|
|
479
|
+
```
|
|
480
|
+
|
|
481
|
+
A changed execution set is refused with a fresh list and token. With no live
|
|
482
|
+
executions no token is needed. Bare force cannot fence live executions, and the
|
|
483
|
+
tool rejects `force:true`. Subagents cannot force a restart, even from bash:
|
|
484
|
+
it would fence themselves and other sessions' work. Force fences running
|
|
485
|
+
executions; they resume on the new version from their sessions, like after a
|
|
486
|
+
crash, so a tool call that was running is repeated or reported as interrupted.
|
|
487
|
+
The restart ledger records the reason and initiator; after the next start,
|
|
488
|
+
`status` shows who forced it and why for 24 hours.
|
|
489
|
+
|
|
490
|
+
An orchestrator reads requests only after it has recovered its workflows, so
|
|
491
|
+
a `restart` sent to one that just started waits for it (and says so); if no
|
|
492
|
+
orchestrator reaches the request in time, `restart` exits 75 and the request
|
|
493
|
+
stays pending — it is decided later, so do not send another.
|
|
399
494
|
|
|
400
495
|
To restart only when the machine is quiet, `drain` first (running calls finish
|
|
401
496
|
and nothing new starts in existing workflows), retry `restart` until it is
|
|
402
497
|
accepted, then `resume`. Never kill the orchestrator process: other sessions'
|
|
403
498
|
running calls would be interrupted without a check. An orchestrator from 1.0.17
|
|
404
499
|
or earlier does not know the restart request; `restart` then checks the
|
|
405
|
-
journals itself
|
|
406
|
-
|
|
500
|
+
journals itself with the same token, reason and subagent checks and ends it with
|
|
501
|
+
SIGTERM, which is not atomic: a call launched in between is fenced and resumes.
|
|
502
|
+
The old orchestrator cannot record the new audit fields. A pi session started before the update still
|
|
407
503
|
loads the old extension; start a new one.
|
|
408
504
|
|
|
409
505
|
## What we do not promise
|
|
@@ -419,8 +515,15 @@ loads the old extension; start a new one.
|
|
|
419
515
|
"outcome unknown" notice; other work keeps running.
|
|
420
516
|
- A tool that already ran inside a subagent may run again after a crash, if
|
|
421
517
|
its result never reached the session. Make external side effects
|
|
422
|
-
idempotent, or mark the step `once: true
|
|
423
|
-
|
|
518
|
+
idempotent, or mark the step `once: true`: when an execution is cut off
|
|
519
|
+
while a tool call is running (its result never arrived), the step then
|
|
520
|
+
ends as `unknown` instead of repeating it. Cut off between tool calls, a
|
|
521
|
+
`once` step continues like any other (nothing was left half done). A
|
|
522
|
+
step cut off while its only unfinished tool call is the question it asked
|
|
523
|
+
is not `unknown` either: it keeps waiting and resumes with the answer, also
|
|
524
|
+
an answer given while dsa restarted. If it was cut off while another tool
|
|
525
|
+
call ran beside the question, a `once` step still ends as `unknown`. A follow-up on an `unknown` step continues the
|
|
526
|
+
same session as its next generation.
|
|
424
527
|
- After a crash, the model call that was in flight is paid for again.
|
|
425
528
|
- Process containment uses process tags plus a 1-second tracker. A process
|
|
426
529
|
that clears its tag and leaves the process tree within its first second
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
// The subagents tool's parameter schema. Kept apart from tool.ts because it needs the optional pi peer
|
|
2
|
+
// package, which the CLI (sharing tool.ts) must not import: it runs from the package directory.
|
|
3
|
+
import { Type } from "@earendil-works/pi-ai";
|
|
4
|
+
const stepsDoc = "Call specs {agent, task, model?, cwd?, timeoutMs?, output?, schema?, gate?, isolation?, context?, budget?, once?, tools?, skills?, writer?, key?}; " +
|
|
5
|
+
"each call is addressed as '<wid>/<key>', where key is the step's own unique key or else 'tasks:<i>' / 'chain:<i>'.";
|
|
6
|
+
export const parameters = Type.Object({
|
|
7
|
+
action: Type.Optional(Type.Union(["run", "agents", "send", "stop", "revise", "status", "resume", "drain", "restart"].map(v => Type.Literal(v)))),
|
|
8
|
+
workflow: Type.Optional(Type.String()), source: Type.Optional(Type.String()), args: Type.Optional(Type.Unknown()),
|
|
9
|
+
tasks: Type.Optional(Type.Array(Type.Any(), { description: `Parallel calls. ${stepsDoc}` })),
|
|
10
|
+
chain: Type.Optional(Type.Array(Type.Any(), { description: `Sequential calls ({previous} = previous output). ${stepsDoc}` })),
|
|
11
|
+
agent: Type.Optional(Type.String()), task: Type.Optional(Type.String()), model: Type.Optional(Type.String()),
|
|
12
|
+
cwd: Type.Optional(Type.String({ description: "Run directory (default: this session's). Relative workflow, inputs and call cwd paths resolve against it." })),
|
|
13
|
+
to: Type.Optional(Type.String()), kind: Type.Optional(Type.Union([Type.Literal("steer"), Type.Literal("follow-up"), Type.Literal("answer"), Type.Literal("model")])),
|
|
14
|
+
message: Type.Optional(Type.String()), qid: Type.Optional(Type.String()), rev: Type.Optional(Type.Integer({ minimum: 1 })),
|
|
15
|
+
replaces: Type.Optional(Type.Array(Type.String())), target: Type.Optional(Type.String()), wid: Type.Optional(Type.String()),
|
|
16
|
+
usageBudget: Type.Optional(Type.Object({ tokens: Type.Optional(Type.Number()), costUsd: Type.Optional(Type.Number()) })),
|
|
17
|
+
maxCalls: Type.Optional(Type.Integer({ minimum: 1 })), inputs: Type.Optional(Type.Record(Type.String(), Type.String())),
|
|
18
|
+
name: Type.Optional(Type.String()),
|
|
19
|
+
timeoutMs: Type.Optional(Type.Number({ description: "Per-call limit on active time in milliseconds (a number). Omit unless a hard limit is needed; prefer budgets." })),
|
|
20
|
+
key: Type.Optional(Type.String({ description: "A single agent/task run: the call's key. status with wid: that call's full result." })),
|
|
21
|
+
full: Type.Optional(Type.Boolean({ description: "status: with wid, the complete workflow detail including every output." })),
|
|
22
|
+
force: Type.Optional(Type.Union([Type.String(), Type.Boolean()], { description: "restart: the token shown by a refusal. Show the user the list and obtain explicit approval first; boolean true is refused. Subagents cannot force a restart." })),
|
|
23
|
+
reason: Type.Optional(Type.String({ description: "restart: non-empty reason, at most 500 characters; required with force." })),
|
|
24
|
+
request: Type.Optional(Type.String({ description: "run/send/stop: your own request id (1-124 chars [A-Za-z0-9][A-Za-z0-9._:-]*) making a retry safe: the same id with the same content gets the first outcome; other content is refused (request-conflict)." })),
|
|
25
|
+
}, { additionalProperties: true });
|
package/dist/agent/main/tool.js
CHANGED
|
@@ -1,27 +1,7 @@
|
|
|
1
|
+
import { restartInputError } from "../../orchestrator/restart.js";
|
|
1
2
|
import { resolve } from "node:path";
|
|
2
|
-
import { Type } from "@earendil-works/pi-ai";
|
|
3
3
|
import { validateCallSpec } from "../../compat/spec.js";
|
|
4
4
|
import { compileFanout } from "../../compat/fanout.js";
|
|
5
|
-
const stepsDoc = "Call specs {agent, task, model?, cwd?, timeoutMs?, output?, schema?, gate?, isolation?, context?, budget?, once?, tools?, skills?, writer?, key?}; " +
|
|
6
|
-
"each call is addressed as '<wid>/<key>', where key is the step's own unique key or else 'tasks:<i>' / 'chain:<i>'.";
|
|
7
|
-
export const parameters = Type.Object({
|
|
8
|
-
action: Type.Optional(Type.Union(["run", "agents", "send", "stop", "revise", "status", "resume", "drain", "restart"].map(v => Type.Literal(v)))),
|
|
9
|
-
workflow: Type.Optional(Type.String()), source: Type.Optional(Type.String()), args: Type.Optional(Type.Unknown()),
|
|
10
|
-
tasks: Type.Optional(Type.Array(Type.Any(), { description: `Parallel calls. ${stepsDoc}` })),
|
|
11
|
-
chain: Type.Optional(Type.Array(Type.Any(), { description: `Sequential calls ({previous} = previous output). ${stepsDoc}` })),
|
|
12
|
-
agent: Type.Optional(Type.String()), task: Type.Optional(Type.String()), model: Type.Optional(Type.String()),
|
|
13
|
-
cwd: Type.Optional(Type.String({ description: "Run directory (default: this session's). Relative workflow, inputs and call cwd paths resolve against it." })),
|
|
14
|
-
to: Type.Optional(Type.String()), kind: Type.Optional(Type.Union([Type.Literal("steer"), Type.Literal("follow-up"), Type.Literal("answer"), Type.Literal("model")])),
|
|
15
|
-
message: Type.Optional(Type.String()), qid: Type.Optional(Type.String()), rev: Type.Optional(Type.Integer({ minimum: 1 })),
|
|
16
|
-
replaces: Type.Optional(Type.Array(Type.String())), target: Type.Optional(Type.String()), wid: Type.Optional(Type.String()),
|
|
17
|
-
usageBudget: Type.Optional(Type.Object({ tokens: Type.Optional(Type.Number()), costUsd: Type.Optional(Type.Number()) })),
|
|
18
|
-
maxCalls: Type.Optional(Type.Integer({ minimum: 1 })), inputs: Type.Optional(Type.Record(Type.String(), Type.String())),
|
|
19
|
-
name: Type.Optional(Type.String()),
|
|
20
|
-
timeoutMs: Type.Optional(Type.Number({ description: "Per-call limit on active time in milliseconds (a number). Omit unless a hard limit is needed; prefer budgets." })),
|
|
21
|
-
key: Type.Optional(Type.String({ description: "A single agent/task run: the call's key. status with wid: that call's full result." })),
|
|
22
|
-
full: Type.Optional(Type.Boolean({ description: "status: with wid, the complete workflow detail including every output." })),
|
|
23
|
-
force: Type.Optional(Type.Boolean({ description: "restart: fence running executions instead of refusing (they resume on the new orchestrator)." })),
|
|
24
|
-
}, { additionalProperties: true });
|
|
25
5
|
/** Call fields a tasks/chain run applies to every step that does not set its own. */
|
|
26
6
|
export const stepDefaults = ["model", "timeoutMs", "budget", "isolation", "context", "tools", "skills", "once", "writer"];
|
|
27
7
|
function string(args, name) {
|
|
@@ -39,6 +19,16 @@ function call(value, cwd, where) {
|
|
|
39
19
|
spec.cwd = resolve(cwd, spec.cwd);
|
|
40
20
|
return spec;
|
|
41
21
|
}
|
|
22
|
+
/** v12 §2: Reject unknown explicit call agents before starter or outbox publication; scripts remain call-local. Shared by
|
|
23
|
+
* the tool and the CLI `run --request` (R2). */
|
|
24
|
+
export function checkAgents(body, available) {
|
|
25
|
+
const names = [...(body.call ? [body.call] : []), ...(body.tasks ?? []), ...(body.chain ?? [])].map(call => call.agent);
|
|
26
|
+
if (!names.length)
|
|
27
|
+
return;
|
|
28
|
+
const known = available(), unknown = [...new Set(names.filter(name => !known.includes(name)))];
|
|
29
|
+
if (unknown.length)
|
|
30
|
+
throw new Error(`Unknown agent${unknown.length === 1 ? "" : "s"}: ${unknown.join(", ")}. Available agents: ${known.join(", ") || "(none)"}`);
|
|
31
|
+
}
|
|
42
32
|
/** v12 §2: Infer unambiguous runs and normalize controls into unchanged wire bodies. */
|
|
43
33
|
/** P12: a send naming a model is answered with that model and when it applies — `next-request` (a running call switches
|
|
44
34
|
* at its next provider request), `next-execution` (a call with no live execution launches on it) or `next-generation`
|
|
@@ -55,7 +45,7 @@ export function request(args, cwd) {
|
|
|
55
45
|
if (typeof action !== "string" || !action)
|
|
56
46
|
throw new Error("action is required: run, agents, send, stop, revise, status, resume, drain, restart");
|
|
57
47
|
if (action === "run") {
|
|
58
|
-
const { action: _, workflow, source, tasks, chain, args: inputs, name, usageBudget, maxCalls, inputs: files, by: _by, ...spec } = args;
|
|
48
|
+
const { action: _, workflow, source, tasks, chain, args: inputs, name, usageBudget, maxCalls, inputs: files, by: _by, request: _request, ...spec } = args;
|
|
59
49
|
const choices = [workflow, source, tasks, chain, spec.agent === undefined && spec.task === undefined ? undefined : spec];
|
|
60
50
|
if (choices.filter(v => v !== undefined).length !== 1)
|
|
61
51
|
throw new Error("run requires exactly one of workflow, source, tasks, chain, or agent/task");
|
|
@@ -142,7 +132,16 @@ export function request(args, cwd) {
|
|
|
142
132
|
return { kind: "resume", body: args.wid !== undefined ? { wid: string(args, "wid") } : typeof args.origin === "string" ? { origin: args.origin } : {} };
|
|
143
133
|
if (action === "drain")
|
|
144
134
|
return { kind: "drain", body: {} };
|
|
145
|
-
if (action === "restart")
|
|
146
|
-
|
|
135
|
+
if (action === "restart") {
|
|
136
|
+
if (args.force === true)
|
|
137
|
+
throw new Error('force:true is refused; show the user the running executions from a restart refusal, then use force:"<token>" and reason:"<why>" only with explicit user approval');
|
|
138
|
+
if (args.force !== undefined && args.force !== false && typeof args.force !== "string")
|
|
139
|
+
throw new Error("force must be the token from a refused restart");
|
|
140
|
+
const body = { ...(typeof args.force === "string" ? { token: args.force } : {}), ...(args.reason !== undefined ? { reason: args.reason } : {}) };
|
|
141
|
+
const invalid = restartInputError(body);
|
|
142
|
+
if (invalid)
|
|
143
|
+
throw new Error(invalid);
|
|
144
|
+
return { kind: "restart", body };
|
|
145
|
+
}
|
|
147
146
|
throw new Error(`Unsupported action: ${action}; use run, agents, send, stop, revise, status, resume, drain, or restart`);
|
|
148
147
|
}
|
package/dist/agent/main.js
CHANGED
|
@@ -13,8 +13,11 @@ import { dsaHome, orchInbox, orchLedger, orchLock, outboxRoot } from "../paths.j
|
|
|
13
13
|
import { CT, JT } from "../types.js";
|
|
14
14
|
import { attention, presentText, presented, resolved, unfinishedWorkflow } from "./main/snapshots.js";
|
|
15
15
|
import { isLive, pausedElsewhere, runningOrchestrator, statusBrief, statusCallDetail, statusCompactDetail, statusDetail, statusView, widOfRid } from "../orchestrator/snapshot.js";
|
|
16
|
-
import {
|
|
16
|
+
import { checkAgents, request, sendReceipt } from "./main/tool.js";
|
|
17
|
+
import { parameters } from "./main/schema.js";
|
|
18
|
+
import { findRequest, requestRid, sendIdentified } from "../requests.js";
|
|
17
19
|
import { discoverAgents } from "../compat/agents.js";
|
|
20
|
+
import { restartInputError } from "../orchestrator/restart.js";
|
|
18
21
|
import { currentOrchestrator, legacyRestart, waitExit } from "../cli/restart.js";
|
|
19
22
|
import { packageVersion } from "../version.js";
|
|
20
23
|
let noteSink;
|
|
@@ -40,6 +43,7 @@ function quitPolicy(home) {
|
|
|
40
43
|
return "pause";
|
|
41
44
|
}
|
|
42
45
|
}
|
|
46
|
+
const REQUEST_USE = "request is a string id for run, send or stop (not combined with replaces)";
|
|
43
47
|
export function registerMain(pi, ui) {
|
|
44
48
|
const home = dsaHome();
|
|
45
49
|
let ctx, sender = "", outbox;
|
|
@@ -261,6 +265,8 @@ export function registerMain(pi, ui) {
|
|
|
261
265
|
for (const field of ["wid", "to", "target"])
|
|
262
266
|
if (typeof args[field] === "string")
|
|
263
267
|
args = { ...args, [field]: ridToWid(args[field]) };
|
|
268
|
+
if (args.request !== undefined && (args.action === "status" || args.action === "agents"))
|
|
269
|
+
throw new Error(REQUEST_USE);
|
|
264
270
|
if (args.action === "status") {
|
|
265
271
|
if (typeof args.wid !== "string" || !args.wid)
|
|
266
272
|
return statusBrief(home, { origin: sender });
|
|
@@ -278,34 +284,41 @@ export function registerMain(pi, ui) {
|
|
|
278
284
|
const { target, ...rest } = args;
|
|
279
285
|
args = { ...rest, to: target };
|
|
280
286
|
}
|
|
287
|
+
// A retried answer addresses the question the first attempt resolved (it may be closed by now), like the CLI.
|
|
288
|
+
if (args.action === "send" && args.kind === "answer" && typeof args.request === "string" && (args.qid === undefined || args.rev === undefined)) {
|
|
289
|
+
const prior = (await findRequest(home, requestRid(args.request)))?.request, body = prior?.body;
|
|
290
|
+
if (prior?.kind === "send" && body?.kind === "answer" && typeof body.to === "string" && prior.cond?.qid !== undefined &&
|
|
291
|
+
(args.to === undefined || args.to === body.to) && (args.qid === undefined || args.qid === prior.cond.qid))
|
|
292
|
+
args = { ...args, to: body.to, qid: prior.cond.qid, rev: args.rev ?? prior.cond.rev };
|
|
293
|
+
}
|
|
281
294
|
if (args.action === "send")
|
|
282
295
|
args = completeSend(args);
|
|
283
296
|
// A session resumes its own held work (what its quit paused); the CLI `resume` remains the global one.
|
|
284
297
|
if (args.action === "resume" && args.wid === undefined)
|
|
285
298
|
args = { ...args, origin: sender };
|
|
286
299
|
const normalized = request(args, cwd);
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
const unknown = [...new Set(names.filter(name => !available.includes(name)))];
|
|
294
|
-
if (unknown.length)
|
|
295
|
-
throw new Error(`Unknown agent${unknown.length === 1 ? "" : "s"}: ${unknown.join(", ")}. Available agents: ${available.join(", ") || "(none)"}`);
|
|
296
|
-
}
|
|
297
|
-
}
|
|
300
|
+
// R1: a caller-chosen request id names a run, send or stop; a retry with the same content gets the first outcome.
|
|
301
|
+
if (args.request !== undefined && (typeof args.request !== "string" || !["run", "send", "stop"].includes(normalized.kind) || normalized.replaces?.length))
|
|
302
|
+
throw new Error(REQUEST_USE);
|
|
303
|
+
const rid = typeof args.request === "string" ? requestRid(args.request) : undefined;
|
|
304
|
+
if (normalized.kind === "run")
|
|
305
|
+
checkAgents(normalized.body, () => agentsAt(cwd).map(agent => agent.name));
|
|
298
306
|
// P33: any call of the run may fork the origin context, so the origin branch is always offered for pinning.
|
|
299
307
|
const sessionFile = ctx?.sessionManager.getSessionFile();
|
|
300
308
|
if (normalized.kind === "run" && sessionFile)
|
|
301
309
|
normalized.body.origin = { sessionFile, leafId: ctx.sessionManager.getLeafId() };
|
|
302
310
|
if (normalized.kind === "restart") {
|
|
303
311
|
// Only an orchestrator that decides restarts is sent one (an older one would keep it as an invalid inbox file).
|
|
304
|
-
const
|
|
312
|
+
const body = normalized.body;
|
|
313
|
+
body.initiator = process.env.DSA_CALL ? { call: process.env.DSA_CALL } : { origin: sender };
|
|
314
|
+
const invalid = restartInputError(body, process.env.DSA_EXEC !== undefined);
|
|
315
|
+
if (invalid)
|
|
316
|
+
return { applied: false, reason: invalid };
|
|
317
|
+
const previous = currentOrchestrator(home);
|
|
305
318
|
if (!previous)
|
|
306
319
|
return { applied: true, note: "no orchestrator is running; the next one starts on the installed version when work is submitted" };
|
|
307
320
|
if (!previous.restart) {
|
|
308
|
-
const legacy = legacyRestart(home, previous,
|
|
321
|
+
const legacy = legacyRestart(home, previous, body, { subagent: process.env.DSA_EXEC !== undefined, tool: true });
|
|
309
322
|
if (!legacy.applied)
|
|
310
323
|
return { applied: false, reason: legacy.reason };
|
|
311
324
|
// It does not start its successor; this session does once it has exited (or its next periodic check would).
|
|
@@ -313,7 +326,7 @@ export function registerMain(pi, ui) {
|
|
|
313
326
|
return { applied: true, note: restartNote(previous) };
|
|
314
327
|
}
|
|
315
328
|
}
|
|
316
|
-
const
|
|
329
|
+
const outcome = await serial(async () => {
|
|
317
330
|
if (!outbox || stopped)
|
|
318
331
|
throw new Error("Main session is not active");
|
|
319
332
|
signal?.throwIfAborted();
|
|
@@ -322,8 +335,16 @@ export function registerMain(pi, ui) {
|
|
|
322
335
|
const withdrawn = await outbox.send("orch", "withdraw", { rids: normalized.replaces });
|
|
323
336
|
normalized.cond = { ...normalized.cond, after: withdrawn.rid };
|
|
324
337
|
}
|
|
338
|
+
if (rid)
|
|
339
|
+
return sendIdentified(home, outbox, sender, rid, normalized.kind, normalized.body, normalized.cond);
|
|
325
340
|
return outbox.send("orch", normalized.kind, normalized.body, normalized.cond);
|
|
326
341
|
});
|
|
342
|
+
if ("conflict" in outcome) {
|
|
343
|
+
const created = ledger().find(e => e.type === JT.created && e.rid === rid);
|
|
344
|
+
return { applied: false, reason: "request-conflict", request: args.request, spec_digest: outcome.digest, ...(created ? { wid: created.wid } : {}),
|
|
345
|
+
note: "this request id was used for different content; use a new id" };
|
|
346
|
+
}
|
|
347
|
+
const sent = "digest" in outcome ? outcome.request : outcome;
|
|
327
348
|
// P25: run waits for `created`; control requests wait for their terminal lifecycle record (applied or rejected+reason).
|
|
328
349
|
const deadline = performance.now() + 10_000;
|
|
329
350
|
while (wait || sent.kind === "run") {
|
|
@@ -360,7 +381,7 @@ export function registerMain(pi, ui) {
|
|
|
360
381
|
"run (action optional for exactly one launch form): agent+task; tasks:[call specs] parallel; chain:[call specs] sequential ({previous}); workflow:'./script.js' or source (runs.run(key,spec), runs.all([...]), emit(value), args, runs.input(name)). Optional name, cwd, usageBudget, maxCalls, inputs. With tasks/chain, top-level model, timeoutMs, budget, isolation, context, tools, skills, once are defaults for every step (a step's own value wins); a workflow/source script sets them per runs.run call. timeoutMs is milliseconds of active time (a number); omit it unless a hard limit is needed. Explicit unknown agents are rejected BEFORE creation, with available names; unknown script agents fail only their call.",
|
|
361
382
|
"agents: list names, descriptions, default models and source for this cwd; use these names for run.",
|
|
362
383
|
"send to:'<wid>/<key>' (bare '<wid>' only for a single-call workflow): steer on a running call delivers at the next safe point (receipt in status/UI); a steer to a call waiting on its question interrupts the question and the subagent usually asks again — use answer to answer it; sealed → finished:<status> — use kind 'follow-up'. follow-up continues a sealed call as generation g+1 or queues after a running turn; follow-up model:'provider/id' or a pool name runs that generation on it. answer: give the qid (or just the call, or nothing when one question is open); to and rev are filled in. A question that needs the user's decision goes to the user; if you answer one yourself, tell the user what you chose. model ('provider/id' or a pool name — its first model not used up): a running call switches at its next provider request; an asking, hibernated or queued call launches on it when it runs again; the reply's model/effect (next-request|next-execution|next-generation) says which. status model = model actually used by the last request; switching = requested, not used yet; switchFailed = refused. A provider content refusal (ToS/usage policy) fails the call at once, not retried. Unknown targets list valid addresses. replaces:[rid] supersedes an earlier send.",
|
|
363
|
-
"stop target:<wid|<wid>/<key>> is terminal stopped (usage and partial edits kept); a sealed call → already-sealed:<status>, a finished workflow → terminal:<status>. drain holds existing workflows reversibly (new runs unaffected); resume [wid] releases held workflows. restart (after an update) replaces the orchestrator with the installed version: refused with busy:<running executions> while any runs
|
|
384
|
+
"stop target:<wid|<wid>/<key>> is terminal stopped (usage and partial edits kept); a sealed call → already-sealed:<status>, a finished workflow → terminal:<status>. drain holds existing workflows reversibly (new runs unaffected); resume [wid] releases held workflows. restart (after an update) replaces the orchestrator with the installed version: refused with busy:<running executions> while any runs. Never force without the user's explicit approval: show the user the refusal's list first, then supply force:'<token>' and reason. Subagents cannot force; hibernated askers and queued calls do not block it. Never kill the orchestrator process. Commands that need the machine (benchmarks, timing) take a lease: tell the subagent to run them as `pi-durable-subagents hold machine [--shared] -- <command>` (FIFO; status lists lease holders and waiters). status: without wid, what runs, asks (with its answer address; hibernated:true holds no slot) or failed, writerWait: a call queued for its git worktree's writer lock (one call whose tools include edit/write runs per worktree; spec writer:false or isolation:'worktree' opts out), sharedWorktree names calls sharing observed edit/write roots (reminder), lease: a call holding or waiting for a resource lease, finished workflows one line each, provider slots held/limit, the config in effect and providers whose usage window is used up (avoided until a probe finds them answering again), and the orchestrator version (versionNote when it differs from the loaded one); wid: one workflow, outputs clipped; wid+key: one call's full result; full:true: everything. A run's rid from {submitted:{rid}} works wherever a wid is expected. revise wid + workflow/source/args starts a revision.",
|
|
364
385
|
"Control replies are {applied:true,rid} or {applied:false,reason,rid} when decided; otherwise {submitted:{rid}} after 10s.",
|
|
365
386
|
...(agents ? [`Available agents: ${agents}.`] : []),
|
|
366
387
|
"User sees a summary line above the editor; ↓ on an empty editor (or /subagents) opens the list, Enter watches live OR finished calls (finished transcripts remain on disk) and expands finished workflows. List keys: s steer (paste-capable input), x stop (confirm y), m model, a answer when asked, f follow-up on finished calls; action feedback appears in footer.",
|
package/dist/cli/control.js
CHANGED
|
@@ -10,7 +10,10 @@ import { reduceLifecycle } from "../kernel/lifecycle.js";
|
|
|
10
10
|
import { OsLock } from "../platform/lock.js";
|
|
11
11
|
import { orchInbox, orchLedger, orchLock, outboxRoot } from "../paths.js";
|
|
12
12
|
import { unfinishedWorkflow } from "../agent/main/snapshots.js";
|
|
13
|
+
import { cliInitiator } from "./restart.js";
|
|
14
|
+
import { restartInputError } from "../orchestrator/restart.js";
|
|
13
15
|
import { JT } from "../types.js";
|
|
16
|
+
import { RequestsBusy, sendIdentified } from "../requests.js";
|
|
14
17
|
/** P1: Start the detached orchestrator only after probing its OS lock. */
|
|
15
18
|
export async function startOrchestrator(home, env) {
|
|
16
19
|
const lock = await new OsLock().tryAcquire(orchLock(home));
|
|
@@ -73,10 +76,47 @@ export async function resolution(home, rid, timeoutMs, interval = 100) {
|
|
|
73
76
|
await delay(interval);
|
|
74
77
|
}
|
|
75
78
|
}
|
|
76
|
-
/** P5, P38:
|
|
79
|
+
/** P5, P38: Send one control request through the CLI sender, then start the orchestrator. */
|
|
77
80
|
export async function submit(home, command, target, env = process.env, options = {}) {
|
|
78
81
|
if (command === "stop" && !target)
|
|
79
82
|
throw new Error("stop requires a workflow or call id");
|
|
83
|
+
return withSender(home, async (outbox) => {
|
|
84
|
+
const requests = [];
|
|
85
|
+
if (command === "stop-all") {
|
|
86
|
+
const body = { fence: true };
|
|
87
|
+
requests.push(await outbox.send("orch", "drain", body));
|
|
88
|
+
}
|
|
89
|
+
else if (command === "prune") {
|
|
90
|
+
const body = { ...(target ? { wid: target } : {}), ...(options.olderThanDays !== undefined ? { olderThanDays: options.olderThanDays } : {}) };
|
|
91
|
+
requests.push(await outbox.send("orch", "prune", body));
|
|
92
|
+
}
|
|
93
|
+
else if (command === "restart") {
|
|
94
|
+
const body = { ...options.restart, initiator: cliInitiator(env) };
|
|
95
|
+
const invalid = restartInputError(body, env.DSA_EXEC !== undefined);
|
|
96
|
+
if (invalid)
|
|
97
|
+
throw new Error(invalid);
|
|
98
|
+
requests.push(await outbox.send("orch", "restart", body));
|
|
99
|
+
}
|
|
100
|
+
else
|
|
101
|
+
requests.push(await outbox.send("orch", command, command === "stop" ? { target } : command === "resume" && target ? { wid: target } : {}));
|
|
102
|
+
// Publish first: even a starter failure leaves a recoverable request and no idle-exit race.
|
|
103
|
+
await startOrchestrator(home, env);
|
|
104
|
+
return requests;
|
|
105
|
+
});
|
|
106
|
+
}
|
|
107
|
+
/** R1: Submit a request named by a caller-chosen id through the CLI sender: a retry with the same content republishes
|
|
108
|
+
* (or reuses) the recorded envelope, other content is a conflict and publishes nothing. Starts the orchestrator
|
|
109
|
+
* unless the request conflicts. */
|
|
110
|
+
export async function submitIdentified(home, rid, kind, body, cond, env = process.env, starter = startOrchestrator) {
|
|
111
|
+
return withSender(home, async (outbox, sender) => {
|
|
112
|
+
const result = await sendIdentified(home, outbox, sender, rid, kind, body, cond);
|
|
113
|
+
if ("request" in result)
|
|
114
|
+
await starter(home, env);
|
|
115
|
+
return result;
|
|
116
|
+
});
|
|
117
|
+
}
|
|
118
|
+
/** P5, P38: Serialize the stable CLI sender across processes and recover its durable outbox. */
|
|
119
|
+
async function withSender(home, fn) {
|
|
80
120
|
await mkdir(home, { recursive: true });
|
|
81
121
|
const sender = `cli:${userInfo().username}@${hostname()}`;
|
|
82
122
|
const locker = new OsLock(), deadline = performance.now() + 10_000;
|
|
@@ -86,7 +126,7 @@ export async function submit(home, command, target, env = process.env, options =
|
|
|
86
126
|
lock = await locker.tryAcquire(join(home, `${sender}.lock`));
|
|
87
127
|
}
|
|
88
128
|
if (!lock)
|
|
89
|
-
throw new
|
|
129
|
+
throw new RequestsBusy("CLI sender is busy; retry the command");
|
|
90
130
|
try {
|
|
91
131
|
const outbox = await Outbox.open(outboxRoot(home), sender, () => orchInbox(home));
|
|
92
132
|
try {
|
|
@@ -94,24 +134,7 @@ export async function submit(home, command, target, env = process.env, options =
|
|
|
94
134
|
for (const rid of reduceLifecycle(records).resolved.keys())
|
|
95
135
|
await outbox.markResolved(rid);
|
|
96
136
|
await outbox.republishPending();
|
|
97
|
-
|
|
98
|
-
if (command === "stop-all") {
|
|
99
|
-
const body = { fence: true };
|
|
100
|
-
requests.push(await outbox.send("orch", "drain", body));
|
|
101
|
-
}
|
|
102
|
-
else if (command === "prune") {
|
|
103
|
-
const body = { ...(target ? { wid: target } : {}), ...(options.olderThanDays !== undefined ? { olderThanDays: options.olderThanDays } : {}) };
|
|
104
|
-
requests.push(await outbox.send("orch", "prune", body));
|
|
105
|
-
}
|
|
106
|
-
else if (command === "restart") {
|
|
107
|
-
const body = options.force ? { force: true } : {};
|
|
108
|
-
requests.push(await outbox.send("orch", "restart", body));
|
|
109
|
-
}
|
|
110
|
-
else
|
|
111
|
-
requests.push(await outbox.send("orch", command, command === "stop" ? { target } : command === "resume" && target ? { wid: target } : {}));
|
|
112
|
-
// Publish first: even a starter failure leaves a recoverable request and no idle-exit race.
|
|
113
|
-
await startOrchestrator(home, env);
|
|
114
|
-
return requests;
|
|
137
|
+
return await fn(outbox, sender);
|
|
115
138
|
}
|
|
116
139
|
finally {
|
|
117
140
|
await outbox.close();
|