pi-durable-subagents 1.0.15 → 1.0.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/README.md +3 -3
- package/dist/agent/child.js +2 -0
- package/dist/agent/main/tool.js +1 -1
- package/dist/agent/main.js +1 -1
- package/dist/orchestrator/engine.js +4 -2
- package/dist/orchestrator/executor/index.js +119 -28
- package/dist/orchestrator/snapshot.js +1 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,20 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.0.16
|
|
4
|
+
|
|
5
|
+
- A used-up usage window is found while pi is still retrying: at the second
|
|
6
|
+
quota refusal in a row (`No available accounts`, usage limit, quota
|
|
7
|
+
exceeded), not after pi's retries end. A call from a pool moves to the
|
|
8
|
+
pool's next model that is not used up and has a free slot, in the same
|
|
9
|
+
execution and session, within pi's next retry or two (refused requests use
|
|
10
|
+
no quota); one with a single model waits for its provider. Before, a call kept retrying the used-up provider for as
|
|
11
|
+
long as pi's retry settings allowed (over ten minutes with ten retries).
|
|
12
|
+
- `send kind:"model"` and a follow-up's `model` accept a pool's name: the
|
|
13
|
+
first model of the pool that is not used up (for a running call, also with
|
|
14
|
+
a free slot). The reply names the model picked; a call from that pool stays
|
|
15
|
+
in it, so a later used-up window still moves it on. A follow-up naming a
|
|
16
|
+
pool starts its generation from the pool's order.
|
|
17
|
+
|
|
3
18
|
## 1.0.15
|
|
4
19
|
|
|
5
20
|
- The changes listed under 1.0.14, which was tagged but never published:
|
package/README.md
CHANGED
|
@@ -55,7 +55,7 @@ the npx cache, so `install-service` refuses to run from there.
|
|
|
55
55
|
| You steer a subagent while it is asking you a question | Your message reaches it, in order. Nothing is rejected or lost. |
|
|
56
56
|
| Two steers arrive out of order and the second replaces the first | Only the second one applies. |
|
|
57
57
|
| A step is refused, or a dependency fails | The workflow stops that branch cleanly. Nothing is retried in vain. |
|
|
58
|
-
| A provider's usage window runs out (`No available accounts`, usage limit, quota exceeded) | A call in a pool continues **in the same session** on the pool's next model; new calls skip that provider. After 15 minutes the next call that wants it tries it once; when it answers, new calls and new generations use it again. A call with a single model waits for it instead of failing. Billing errors (402, insufficient balance) still fail at once. |
|
|
58
|
+
| A provider's usage window runs out (`No available accounts`, usage limit, quota exceeded) | Found at the second refusal in a row, while pi is still retrying. A call in a pool continues **in the same session** on the pool's next model (within pi's next retry or two); new calls skip that provider. After 15 minutes the next call that wants it tries it once; when it answers, new calls and new generations use it again. A call with a single model waits for it instead of failing. Billing errors (402, insufficient balance) still fail at once. |
|
|
59
59
|
| Two subagents edit the same worktree | A reminder names both calls; neither is blocked or locked. Only observed `edit`/`write` calls count (bash-only writes are not seen). Calls with `isolation: "worktree"` have their own worktrees. |
|
|
60
60
|
| A subagent waits for an answer for a long time | It releases its model slot and memory, then resumes exactly once when you answer. |
|
|
61
61
|
|
|
@@ -98,9 +98,9 @@ it something. Each verb means one thing, and a refusal says what would work:
|
|
|
98
98
|
|---|---|---|
|
|
99
99
|
| `run` | — | Start one subagent, `tasks` in parallel, a `chain`, or a workflow script. An unknown agent name is refused before anything starts, with the list of agents. |
|
|
100
100
|
| `send steer` | a running subagent | Reaches it at its next safe point. To a finished one: refused, use `follow-up`; To one waiting on its question: it interrupts the question, and the subagent usually asks again; `answer` answers it. |
|
|
101
|
-
| `send follow-up` | a finished subagent | Continues the same session as a new generation (`key@2`). With `model
|
|
101
|
+
| `send follow-up` | a finished subagent | Continues the same session as a new generation (`key@2`). With `model` (a model or a pool's name), that generation runs on it. |
|
|
102
102
|
| `send answer` | an open question | Answers it once. |
|
|
103
|
-
| `send model` | any subagent | A running one switches at its next request; one asking, hibernated or waiting for a slot launches on it when it runs again. |
|
|
103
|
+
| `send model` | any subagent | A running one switches at its next request; one asking, hibernated or waiting for a slot launches on it when it runs again. A pool's name picks its first model that is not used up (and, for a running call, has a free slot); the reply names the model picked, and a call from that pool stays in it. |
|
|
104
104
|
| `stop` | a subagent or a workflow | Final: `stopped`, usage kept, edits left as they are. |
|
|
105
105
|
| `drain` / `resume` | existing workflows | A reversible hold; runs started later are not held. |
|
|
106
106
|
|
package/dist/agent/child.js
CHANGED
|
@@ -98,6 +98,8 @@ export function registerChild(pi) {
|
|
|
98
98
|
return { action: 'defer' };
|
|
99
99
|
if (req.kind === 'model') {
|
|
100
100
|
const body = req.body;
|
|
101
|
+
if (body?.exec !== undefined && body.exec !== exec)
|
|
102
|
+
return { action: 'reject', reason: 'stale-execution' };
|
|
101
103
|
return body && ctx.modelRegistry.find(body.provider, body.model) ? { action: 'apply' } : { action: 'reject', reason: 'unknown-model' };
|
|
102
104
|
}
|
|
103
105
|
if (MESSAGES.includes(req.kind)) {
|
package/dist/agent/main/tool.js
CHANGED
|
@@ -44,7 +44,7 @@ function call(value, cwd, where) {
|
|
|
44
44
|
* (a follow-up's new generation runs on it). From the orchestrator ledger's `send-note`. */
|
|
45
45
|
export function sendReceipt(ledger, rid) {
|
|
46
46
|
const note = ledger.find(e => e.type === "send-note" && e.rid === rid);
|
|
47
|
-
return note ? { model: String(note.model), effect: String(note.effect) } : {};
|
|
47
|
+
return note ? { model: String(note.model), effect: String(note.effect), ...(note.pool ? { pool: String(note.pool) } : {}) } : {};
|
|
48
48
|
}
|
|
49
49
|
export function request(args, cwd) {
|
|
50
50
|
// v12 §2: Infer run only when one launch form is present; never guess a control verb.
|
package/dist/agent/main.js
CHANGED
|
@@ -342,7 +342,7 @@ export function registerMain(pi, ui) {
|
|
|
342
342
|
"Durable asynchronous subagents; run returns {wid} when created (or {submitted:{rid}} while pending). A finished workflow (its notice carries every agent's result) or a question wakes you, so after starting work end your turn: never poll with sleep or repeated status. Crash recovery resumes sessions, not external side effects. Background helper processes (orchestrator, evaluator) exit by themselves about 10 s after all work ends: never kill processes or delete files to 'clean up'. When the user quits pi, this session's running workflows pause (nothing is spent); resume continues them.",
|
|
343
343
|
"run (action optional for exactly one launch form): agent+task; tasks:[call specs] parallel; chain:[call specs] sequential ({previous}); workflow:'./script.js' or source (runs.run(key,spec), runs.all([...]), emit(value), args, runs.input(name)). Optional name, cwd, usageBudget, maxCalls, inputs. With tasks/chain, top-level model, timeoutMs, budget, isolation, context, tools, skills, once are defaults for every step (a step's own value wins); a workflow/source script sets them per runs.run call. timeoutMs is milliseconds of active time (a number); omit it unless a hard limit is needed. Explicit unknown agents are rejected BEFORE creation, with available names; unknown script agents fail only their call.",
|
|
344
344
|
"agents: list names, descriptions, default models and source for this cwd; use these names for run.",
|
|
345
|
-
"send to:'<wid>/<key>' (bare '<wid>' only for a single-call workflow): steer on a running call delivers at the next safe point (receipt in status/UI); a steer to a call waiting on its question interrupts the question and the subagent usually asks again — use answer to answer it; sealed → finished:<status> — use kind 'follow-up'. follow-up continues a sealed call as generation g+1 or queues after a running turn; follow-up model:'provider/id' runs that generation on it. answer: give the qid (or just the call, or nothing when one question is open); to and rev are filled in. A question that needs the user's decision goes to the user; if you answer one yourself, tell the user what you chose. model: a running call switches at its next provider request; an asking, hibernated or queued call launches on it when it runs again; the reply's model/effect (next-request|next-execution|next-generation) says which. status model = model actually used by the last request; switching = requested, not used yet; switchFailed = refused. A provider content refusal (ToS/usage policy) fails the call at once, not retried. Unknown targets list valid addresses. replaces:[rid] supersedes an earlier send.",
|
|
345
|
+
"send to:'<wid>/<key>' (bare '<wid>' only for a single-call workflow): steer on a running call delivers at the next safe point (receipt in status/UI); a steer to a call waiting on its question interrupts the question and the subagent usually asks again — use answer to answer it; sealed → finished:<status> — use kind 'follow-up'. follow-up continues a sealed call as generation g+1 or queues after a running turn; follow-up model:'provider/id' or a pool name runs that generation on it. answer: give the qid (or just the call, or nothing when one question is open); to and rev are filled in. A question that needs the user's decision goes to the user; if you answer one yourself, tell the user what you chose. model ('provider/id' or a pool name — its first model not used up): a running call switches at its next provider request; an asking, hibernated or queued call launches on it when it runs again; the reply's model/effect (next-request|next-execution|next-generation) says which. status model = model actually used by the last request; switching = requested, not used yet; switchFailed = refused. A provider content refusal (ToS/usage policy) fails the call at once, not retried. Unknown targets list valid addresses. replaces:[rid] supersedes an earlier send.",
|
|
346
346
|
"stop target:<wid|<wid>/<key>> is terminal stopped (usage and partial edits kept); a sealed call → already-sealed:<status>, a finished workflow → terminal:<status>. drain holds existing workflows reversibly (new runs unaffected); resume [wid] releases held workflows. status: without wid, what runs, asks (with its answer address; hibernated:true holds no slot) or failed, sharedWorktree names calls sharing observed edit/write roots (reminder only), finished workflows one line each, provider slots held/limit, the config in effect and providers whose usage window is used up (avoided until a probe finds them answering again), and the orchestrator version (versionNote when it differs from the loaded one); wid: one workflow, outputs clipped; wid+key: one call's full result; full:true: everything. A run's rid from {submitted:{rid}} works wherever a wid is expected. revise wid + workflow/source/args starts a revision.",
|
|
347
347
|
"Control replies are {applied:true,rid} or {applied:false,reason,rid} when decided; otherwise {submitted:{rid}} after 10s.",
|
|
348
348
|
...(agents ? [`Available agents: ${agents}.`] : []),
|
|
@@ -331,7 +331,9 @@ export class Engine {
|
|
|
331
331
|
const seal = wf.journal.entries().find(e => e.type === JT.sealed && e.call === from);
|
|
332
332
|
if (seal && send.kind === 'steer')
|
|
333
333
|
return { action: 'reject', reason: `finished:${seal.result.status} — use kind "follow-up" to continue it` };
|
|
334
|
-
|
|
334
|
+
// A pool's name is a model too: the call keeps the pool, and its order and failover apply to the new generation.
|
|
335
|
+
const pools = this.ledgers.config.pools, pool = send.model !== undefined && pools && Object.hasOwn(pools, send.model);
|
|
336
|
+
if (send.kind === 'follow-up' && send.model !== undefined && !pool) {
|
|
335
337
|
try {
|
|
336
338
|
if (!parseModel(send.model).provider)
|
|
337
339
|
throw new Error('missing provider');
|
|
@@ -346,7 +348,7 @@ export class Engine {
|
|
|
346
348
|
const spec = send.model !== undefined ? { ...entry.spec, model: send.model } : entry.spec;
|
|
347
349
|
if (send.model !== undefined)
|
|
348
350
|
await this.note(req.rid, send.model, 'next-generation');
|
|
349
|
-
const opened = await wf.journal.append('generation', { rid: req.rid, key: entry.key, gen, from, spec, revision: wf.revision, opening: { rid: req.rid, kind: send.kind, message: send.message ?? '' }, ...(send.model !== undefined ? { model: send.model } : {}) });
|
|
351
|
+
const opened = await wf.journal.append('generation', { rid: req.rid, key: entry.key, gen, from, spec, revision: wf.revision, opening: { rid: req.rid, kind: send.kind, message: send.message ?? '' }, ...(send.model !== undefined && !pool ? { model: send.model } : {}) });
|
|
350
352
|
this.dispatchGeneration(wf, opened);
|
|
351
353
|
return { action: 'apply' };
|
|
352
354
|
}
|
|
@@ -34,6 +34,8 @@ import { WorktreeIndex, worktreeCalls, worktreeLabel, worktreePair, worktreeRoot
|
|
|
34
34
|
const MEM_RECORD_MS = 30000;
|
|
35
35
|
/** A4, P29: A child admitted within this window may not show in MemAvailable yet; its share is reserved explicitly. */
|
|
36
36
|
const MEM_WARMUP_MS = 30000;
|
|
37
|
+
/** Quota refusals in a row that find a provider's usage window used up while pi is still retrying. */
|
|
38
|
+
const QUOTA_REFUSALS = 2;
|
|
37
39
|
const ignoreMissing = (error) => { if (error.code !== "ENOENT")
|
|
38
40
|
throw error; };
|
|
39
41
|
const callOf = (exec) => exec.slice(0, exec.lastIndexOf("#"));
|
|
@@ -61,7 +63,8 @@ export function requestedModel(journal, call, followUp) {
|
|
|
61
63
|
wanted = m;
|
|
62
64
|
}
|
|
63
65
|
for (const [index, e] of all.entries()) {
|
|
64
|
-
|
|
66
|
+
// A failover's switch is not a request: an execution that ends before applying it leaves the choice to the pool.
|
|
67
|
+
if (e.type !== "forward" || e.dest !== call || e.envelope?.kind !== "model" || e.failover)
|
|
65
68
|
continue;
|
|
66
69
|
const delivered = all.find(r => r.type === "forward-delivered" && r.call === call && r.rid2 === e.rid2);
|
|
67
70
|
if (delivered) {
|
|
@@ -300,9 +303,9 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
300
303
|
}
|
|
301
304
|
/** P7, P27: Record forward-delivered once when a forward's child receipt is first observed; serial sections only. */
|
|
302
305
|
/** The reply to a model send says which model and when it applies (orchestrator ledger `send-note`, once per rid). */
|
|
303
|
-
async function note(rid, model, effect) {
|
|
306
|
+
async function note(rid, model, effect, pool) {
|
|
304
307
|
if (!orch.entries().some(e => e.type === "send-note" && e.rid === rid))
|
|
305
|
-
await orch.append("send-note", { rid, model, effect });
|
|
308
|
+
await orch.append("send-note", { rid, model, effect, ...(pool ? { pool } : {}) });
|
|
306
309
|
}
|
|
307
310
|
async function forwardsDelivered(journal, call, entries) {
|
|
308
311
|
const all = journal.entries();
|
|
@@ -491,7 +494,7 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
491
494
|
const decision = await decide(), models = decision.candidates, pool = decision.pool;
|
|
492
495
|
continuation = decision.continuation;
|
|
493
496
|
for (const model of models) {
|
|
494
|
-
if (!continuation && pool && skipped(pool, model))
|
|
497
|
+
if (!continuation && pool && models.length > 1 && skipped(pool, model))
|
|
495
498
|
continue;
|
|
496
499
|
const provider = model.provider;
|
|
497
500
|
if (unavailable(provider))
|
|
@@ -559,14 +562,19 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
559
562
|
const candidate = recorded && candidates.some(m => m.provider === recorded.provider && m.id === recorded.id);
|
|
560
563
|
// Leave the session's model for the pool's others when its pool skips it after losses, or its provider's usage
|
|
561
564
|
// window is used up; and at a new generation, go back to the pool's order of preference.
|
|
562
|
-
|
|
565
|
+
// A new generation of a pool call starts from the pool also when the session's model is not one of its models
|
|
566
|
+
// (switched outside it, or the follow-up named the pool).
|
|
567
|
+
const skip = pool && (candidate ? previous && ownSegment && skipped(pool, recorded) || unavailable(recorded.provider) || !ownSegment && !!t.continueFrom
|
|
568
|
+
: !ownSegment && !!t.continueFrom);
|
|
563
569
|
// A model the call was asked to use replaces the session's: launched with it, and holding its provider's slot.
|
|
564
570
|
const wanted = requestedModel(t.journal, t.callId, t.model);
|
|
565
571
|
// It outranks the pool's order at a new generation too, also when it names the model the session already has.
|
|
572
|
+
// A requested model of the call's own pool keeps the pool: a used-up window still moves the call on.
|
|
573
|
+
const keep = pool && candidates.some(m => m.provider === wanted?.provider && m.id === wanted?.id) ? pool : undefined;
|
|
566
574
|
if (wanted)
|
|
567
575
|
return recorded && !freshFork && recorded.provider === wanted.provider && recorded.id === wanted.id
|
|
568
|
-
? { candidates: [recorded], continuation: true, pool:
|
|
569
|
-
: { candidates: [wanted], continuation: false, pool:
|
|
576
|
+
? { candidates: [recorded], continuation: true, pool: keep }
|
|
577
|
+
: { candidates: [wanted], continuation: false, pool: keep };
|
|
570
578
|
if (recorded && !freshFork && !skip)
|
|
571
579
|
return { candidates: [recorded], continuation: true, pool: candidate ? pool : undefined };
|
|
572
580
|
return { candidates, continuation: false, pool };
|
|
@@ -634,15 +642,87 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
634
642
|
// While a probe runs, its outcome alone decides: a late refusal of an execution admitted earlier changes nothing.
|
|
635
643
|
if (x && (x.probe ? x.probe !== exec : now < x.nextTry))
|
|
636
644
|
return;
|
|
637
|
-
|
|
645
|
+
// Once per execution and provider: an execution moved on by failover can find a second provider used up too.
|
|
646
|
+
if (orch.entries().some(e => e.type === "provider-exhausted" && e.exec === exec && e.provider === provider))
|
|
638
647
|
return;
|
|
639
648
|
await orch.append("provider-exhausted", { provider, exec, since: x?.since ?? now, nextTry: now + (config.k?.probeMs ?? 900_000), error: error.slice(0, 300) });
|
|
640
649
|
}
|
|
650
|
+
/** Quota refusals in a row per execution, from one provider (pi retries a refused request on its own). */
|
|
651
|
+
const refusals = new Map();
|
|
652
|
+
/** A used-up window shows while pi still retries: the second refusal in a row (the first for a probe) finds the
|
|
653
|
+
* provider used up, and a call launched from a pool switches to the pool's next model at its next request. */
|
|
654
|
+
async function refused(t, exec, provider, error) {
|
|
655
|
+
if (!quotaExhausted(error))
|
|
656
|
+
return;
|
|
657
|
+
const last = refusals.get(exec), count = last?.provider === provider ? last.count + 1 : 1;
|
|
658
|
+
refusals.set(exec, { provider, count });
|
|
659
|
+
const probe = folded().exhausted.get(provider)?.probe === exec;
|
|
660
|
+
if (count < (probe ? 1 : QUOTA_REFUSALS))
|
|
661
|
+
return;
|
|
662
|
+
await serial(async () => {
|
|
663
|
+
if (has(t.journal, JT.fenced, exec) || current(t.journal, t.callId) !== exec)
|
|
664
|
+
return;
|
|
665
|
+
await recordExhausted(provider, exec, error);
|
|
666
|
+
await failover(t, exec, provider);
|
|
667
|
+
});
|
|
668
|
+
wake();
|
|
669
|
+
}
|
|
670
|
+
/** Switch a running execution off a used-up provider: to the first model of its pool on another provider that is
|
|
671
|
+
* neither used up nor full, reserving that slot as a requested switch does. Without one, pi's retries go on and
|
|
672
|
+
* the call waits for the provider once they end. */
|
|
673
|
+
async function failover(t, exec, provider) {
|
|
674
|
+
if (pendingSwitch(exec))
|
|
675
|
+
return;
|
|
676
|
+
const pool = t.journal.entries().findLast(e => e.type === "selected" && e.exec === exec)?.pool, pools = settings().pools;
|
|
677
|
+
if (!pool || !pools?.[pool])
|
|
678
|
+
return;
|
|
679
|
+
let models;
|
|
680
|
+
try {
|
|
681
|
+
models = resolveModel(pool, pools);
|
|
682
|
+
}
|
|
683
|
+
catch {
|
|
684
|
+
return;
|
|
685
|
+
}
|
|
686
|
+
for (const m of models) {
|
|
687
|
+
if (!m.provider || m.provider === provider || unavailable(m.provider) || skipped(pool, m))
|
|
688
|
+
continue;
|
|
689
|
+
const rid = contentHash([exec, "failover", provider]);
|
|
690
|
+
if (t.journal.entries().some(e => e.type === "forward" && e.rid === rid))
|
|
691
|
+
return;
|
|
692
|
+
const probe = folded().exhausted.has(m.provider); // its next try is due (`unavailable` said so): this is its probe
|
|
693
|
+
if (!await reserveSwitch(exec, m.provider, rid))
|
|
694
|
+
continue;
|
|
695
|
+
if (probe)
|
|
696
|
+
await orch.append("provider-probe", { provider: m.provider, exec });
|
|
697
|
+
// Bound to this execution: replayed after it ended, a later execution (which chose its model at launch) refuses it.
|
|
698
|
+
const body = { provider: m.provider, model: m.id, ...(m.thinking ? { thinking: m.thinking } : {}), exec };
|
|
699
|
+
const envelope = { to: t.callId, kind: "model", body }, hash = contentHash(envelope);
|
|
700
|
+
const entry = await t.journal.append("forward", { rid, rid2: forwardRid(rid, t.callId.slice(0, t.callId.indexOf("/")), t.key, hash), dest: t.callId, hash, envelope, failover: provider });
|
|
701
|
+
await replayForward(entry);
|
|
702
|
+
return;
|
|
703
|
+
}
|
|
704
|
+
}
|
|
705
|
+
/** Hold a slot of `provider` for a running execution's switch, unless it holds one; false when it is full. */
|
|
706
|
+
async function reserveSwitch(exec, provider, rid) {
|
|
707
|
+
if (holdings().some(h => h.exec === exec && h.pool === provider))
|
|
708
|
+
return true;
|
|
709
|
+
const target = holdings().filter(h => h.pool === provider);
|
|
710
|
+
if (!capacity({ kind: "provider", holders: target.length, capacity: settings().providers?.[provider]?.slots ?? Infinity }))
|
|
711
|
+
return false;
|
|
712
|
+
let slot = 0;
|
|
713
|
+
while (target.some(h => h.slot === slot))
|
|
714
|
+
slot++;
|
|
715
|
+
await orch.append("hold", { pool: provider, slot, exec, reserved: true, rid });
|
|
716
|
+
return true;
|
|
717
|
+
}
|
|
641
718
|
/** An answer from a used-up provider, requested after it was found used up: available again. */
|
|
642
|
-
async function answered(exec, event) {
|
|
719
|
+
async function answered(t, exec, event) {
|
|
643
720
|
const message = event.message, provider = message?.provider;
|
|
644
|
-
if (message?.role !== "assistant" || !provider
|
|
721
|
+
if (message?.role !== "assistant" || !provider)
|
|
645
722
|
return;
|
|
723
|
+
if (message.stopReason === "error")
|
|
724
|
+
return refused(t, exec, provider, String(message.errorMessage ?? ""));
|
|
725
|
+
refusals.delete(exec);
|
|
646
726
|
await serial(async () => {
|
|
647
727
|
const x = folded().exhausted.get(provider);
|
|
648
728
|
if (x && (x.probe === exec || Number(message.timestamp) > x.since))
|
|
@@ -836,7 +916,7 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
836
916
|
});
|
|
837
917
|
}, recordUsage: values => recordUsage(t, values),
|
|
838
918
|
wrote: path => wrote(t, exec, cwd, path),
|
|
839
|
-
switched: event => switched(exec, journal, event), answered: event => answered(exec, event), pendingSwitch: () => pendingSwitch(exec),
|
|
919
|
+
switched: event => switched(exec, journal, event), answered: event => answered(t, exec, event), pendingSwitch: () => pendingSwitch(exec),
|
|
840
920
|
});
|
|
841
921
|
}
|
|
842
922
|
finally {
|
|
@@ -881,6 +961,9 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
881
961
|
for (const e of inUse.keys())
|
|
882
962
|
if (callOf(e) === ticket.callId)
|
|
883
963
|
inUse.delete(e);
|
|
964
|
+
for (const e of refusals.keys())
|
|
965
|
+
if (callOf(e) === ticket.callId)
|
|
966
|
+
refusals.delete(e);
|
|
884
967
|
forgetSession(callSession(home, ticket.wid, ticket.key, ticket.gen));
|
|
885
968
|
wake();
|
|
886
969
|
}
|
|
@@ -931,6 +1014,28 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
931
1014
|
/** P12: a model request to this call, recorded with the rid given; a reject has no effect. */
|
|
932
1015
|
const requestModel = async (rid, model, hash, cond) => {
|
|
933
1016
|
let body;
|
|
1017
|
+
const exec = current(ctx.journal, dest);
|
|
1018
|
+
// P28: with no live execution (not started yet, between executions, hibernated while asking) the model is
|
|
1019
|
+
// recorded and the next execution launches on it (`requestedModel`); its slot is acquired then, as for any launch.
|
|
1020
|
+
const idle = !exec || has(ctx.journal, JT.fenced, exec) || !has(ctx.journal, "selected", exec);
|
|
1021
|
+
// A pool's name asks for its first model that can take the call now: provider not used up, and (for a running
|
|
1022
|
+
// call) a free slot. The call's own pool stays, so a used-up window later moves it on as before.
|
|
1023
|
+
const pools = settings().pools, pool = pools && Object.hasOwn(pools, model) ? model : undefined;
|
|
1024
|
+
if (pool) {
|
|
1025
|
+
let models;
|
|
1026
|
+
try {
|
|
1027
|
+
models = resolveModel(pool, pools);
|
|
1028
|
+
}
|
|
1029
|
+
catch {
|
|
1030
|
+
return { action: "reject", reason: "unknown-model" };
|
|
1031
|
+
}
|
|
1032
|
+
const free = (p) => idle || holdings().some(h => h.exec === exec && h.pool === p) ||
|
|
1033
|
+
capacity({ kind: "provider", holders: holdings().filter(h => h.pool === p).length, capacity: settings().providers?.[p]?.slots ?? Infinity });
|
|
1034
|
+
const m = models.find(m => m.provider && !unavailable(m.provider) && free(m.provider));
|
|
1035
|
+
if (!m)
|
|
1036
|
+
return { action: "reject", reason: "pool-unavailable" };
|
|
1037
|
+
model = `${m.provider}/${m.id}${m.thinking ? `:${m.thinking}` : ""}`;
|
|
1038
|
+
}
|
|
934
1039
|
try {
|
|
935
1040
|
const m = parseModel(model);
|
|
936
1041
|
if (!m.provider)
|
|
@@ -942,28 +1047,14 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
942
1047
|
}
|
|
943
1048
|
const envelope = { to: dest, kind: "model", body, ...(cond && Object.keys(cond).length ? { cond } : {}) };
|
|
944
1049
|
const rid2 = forwardRid(rid, ctx.widRev, ctx.key, hash);
|
|
945
|
-
const exec = current(ctx.journal, dest), provider = body.provider;
|
|
946
|
-
// P28: with no live execution (not started yet, between executions, hibernated while asking) the model is
|
|
947
|
-
// recorded and the next execution launches on it (`requestedModel`); its slot is acquired then, as for any launch.
|
|
948
1050
|
// Launching (`selected`, not `tracked` yet): the child may start on the old model; ask again in a moment.
|
|
949
|
-
const idle = !exec || has(ctx.journal, JT.fenced, exec) || !has(ctx.journal, "selected", exec);
|
|
950
1051
|
if (!idle && !has(ctx.journal, "tracked", exec))
|
|
951
1052
|
return { action: "reject", reason: "call-starting" };
|
|
952
1053
|
if (!idle && pendingSwitch(exec))
|
|
953
1054
|
return { action: "reject", reason: "switch-pending" };
|
|
954
|
-
if (!idle)
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
const target = holdings().filter(h => h.pool === provider);
|
|
958
|
-
if (!capacity({ kind: "provider", holders: target.length, capacity: settings().providers?.[provider]?.slots ?? Infinity }))
|
|
959
|
-
return { action: "reject", reason: "provider-full" };
|
|
960
|
-
let slot = 0;
|
|
961
|
-
while (target.some(h => h.slot === slot))
|
|
962
|
-
slot++;
|
|
963
|
-
await orch.append("hold", { pool: provider, slot, exec, reserved: true, rid });
|
|
964
|
-
}
|
|
965
|
-
}
|
|
966
|
-
await note(req.rid, model, idle ? "next-execution" : "next-request");
|
|
1055
|
+
if (!idle && !await reserveSwitch(exec, body.provider, rid))
|
|
1056
|
+
return { action: "reject", reason: "provider-full" };
|
|
1057
|
+
await note(req.rid, model, idle ? "next-execution" : "next-request", pool);
|
|
967
1058
|
const entry = await ctx.journal.append("forward", { rid, rid2, dest, hash, envelope });
|
|
968
1059
|
await replayForward(entry);
|
|
969
1060
|
if (idle) {
|
|
@@ -261,7 +261,7 @@ function snapshotReducer(wid, entries) {
|
|
|
261
261
|
const pending = forwarded.filter(pendingMessage).length;
|
|
262
262
|
if (pending)
|
|
263
263
|
c.pending = pending;
|
|
264
|
-
const last = c.phase !== "sealed" && !retired.has(c.callId) ? forwarded.findLast(s => s.kind === "model" && s.reason !== "withdrawn") : undefined;
|
|
264
|
+
const last = c.phase !== "sealed" && !retired.has(c.callId) ? forwarded.findLast(s => s.kind === "model" && s.reason !== "withdrawn" && s.reason !== "stale-execution") : undefined;
|
|
265
265
|
if (last?.reason !== undefined)
|
|
266
266
|
c.switchFailed = `${last.model} (${last.reason})`;
|
|
267
267
|
else if (last && last.state !== "retired" && last.model !== c.model?.replace(/:(off|minimal|low|medium|high|xhigh|max)$/, ""))
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-durable-subagents",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.16",
|
|
4
4
|
"description": "Subagents for pi that never lose work and never do it twice. Crash-safe workflows, automatic recovery, and a live view just like the main agent.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|