pi-durable-subagents 1.0.15 → 1.0.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,20 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.0.16
4
+
5
+ - A used-up usage window is found while pi is still retrying: at the second
6
+ quota refusal in a row (`No available accounts`, usage limit, quota
7
+ exceeded), not after pi's retries end. A call from a pool moves to the
8
+ pool's next model that is not used up and has a free slot, in the same
9
+ execution and session, within pi's next retry or two (refused requests use
10
+ no quota); one with a single model waits for its provider. Before, a call kept retrying the used-up provider for as
11
+ long as pi's retry settings allowed (over ten minutes with ten retries).
12
+ - `send kind:"model"` and a follow-up's `model` accept a pool's name: the
13
+ first model of the pool that is not used up (for a running call, also with
14
+ a free slot). The reply names the model picked; a call from that pool stays
15
+ in it, so a later used-up window still moves it on. A follow-up naming a
16
+ pool starts its generation from the pool's order.
17
+
3
18
  ## 1.0.15
4
19
 
5
20
  - The changes listed under 1.0.14, which was tagged but never published:
package/README.md CHANGED
@@ -55,7 +55,7 @@ the npx cache, so `install-service` refuses to run from there.
55
55
  | You steer a subagent while it is asking you a question | Your message reaches it, in order. Nothing is rejected or lost. |
56
56
  | Two steers arrive out of order and the second replaces the first | Only the second one applies. |
57
57
  | A step is refused, or a dependency fails | The workflow stops that branch cleanly. Nothing is retried in vain. |
58
- | A provider's usage window runs out (`No available accounts`, usage limit, quota exceeded) | A call in a pool continues **in the same session** on the pool's next model; new calls skip that provider. After 15 minutes the next call that wants it tries it once; when it answers, new calls and new generations use it again. A call with a single model waits for it instead of failing. Billing errors (402, insufficient balance) still fail at once. |
58
+ | A provider's usage window runs out (`No available accounts`, usage limit, quota exceeded) | Found at the second refusal in a row, while pi is still retrying. A call in a pool continues **in the same session** on the pool's next model (within pi's next retry or two); new calls skip that provider. After 15 minutes the next call that wants it tries it once; when it answers, new calls and new generations use it again. A call with a single model waits for it instead of failing. Billing errors (402, insufficient balance) still fail at once. |
59
59
  | Two subagents edit the same worktree | A reminder names both calls; neither is blocked or locked. Only observed `edit`/`write` calls count (bash-only writes are not seen). Calls with `isolation: "worktree"` have their own worktrees. |
60
60
  | A subagent waits for an answer for a long time | It releases its model slot and memory, then resumes exactly once when you answer. |
61
61
 
@@ -98,9 +98,9 @@ it something. Each verb means one thing, and a refusal says what would work:
98
98
  |---|---|---|
99
99
  | `run` | — | Start one subagent, `tasks` in parallel, a `chain`, or a workflow script. An unknown agent name is refused before anything starts, with the list of agents. |
100
100
  | `send steer` | a running subagent | Reaches it at its next safe point. To a finished one: refused, use `follow-up`; To one waiting on its question: it interrupts the question, and the subagent usually asks again; `answer` answers it. |
101
- | `send follow-up` | a finished subagent | Continues the same session as a new generation (`key@2`). With `model`, that generation runs on it. |
101
+ | `send follow-up` | a finished subagent | Continues the same session as a new generation (`key@2`). With `model` (a model or a pool's name), that generation runs on it. |
102
102
  | `send answer` | an open question | Answers it once. |
103
- | `send model` | any subagent | A running one switches at its next request; one asking, hibernated or waiting for a slot launches on it when it runs again. |
103
+ | `send model` | any subagent | A running one switches at its next request; one asking, hibernated or waiting for a slot launches on it when it runs again. A pool's name picks its first model that is not used up (and, for a running call, has a free slot); the reply names the model picked, and a call from that pool stays in it. |
104
104
  | `stop` | a subagent or a workflow | Final: `stopped`, usage kept, edits left as they are. |
105
105
  | `drain` / `resume` | existing workflows | A reversible hold; runs started later are not held. |
106
106
 
@@ -98,6 +98,8 @@ export function registerChild(pi) {
98
98
  return { action: 'defer' };
99
99
  if (req.kind === 'model') {
100
100
  const body = req.body;
101
+ if (body?.exec !== undefined && body.exec !== exec)
102
+ return { action: 'reject', reason: 'stale-execution' };
101
103
  return body && ctx.modelRegistry.find(body.provider, body.model) ? { action: 'apply' } : { action: 'reject', reason: 'unknown-model' };
102
104
  }
103
105
  if (MESSAGES.includes(req.kind)) {
@@ -44,7 +44,7 @@ function call(value, cwd, where) {
44
44
  * (a follow-up's new generation runs on it). From the orchestrator ledger's `send-note`. */
45
45
  export function sendReceipt(ledger, rid) {
46
46
  const note = ledger.find(e => e.type === "send-note" && e.rid === rid);
47
- return note ? { model: String(note.model), effect: String(note.effect) } : {};
47
+ return note ? { model: String(note.model), effect: String(note.effect), ...(note.pool ? { pool: String(note.pool) } : {}) } : {};
48
48
  }
49
49
  export function request(args, cwd) {
50
50
  // v12 §2: Infer run only when one launch form is present; never guess a control verb.
@@ -342,7 +342,7 @@ export function registerMain(pi, ui) {
342
342
  "Durable asynchronous subagents; run returns {wid} when created (or {submitted:{rid}} while pending). A finished workflow (its notice carries every agent's result) or a question wakes you, so after starting work end your turn: never poll with sleep or repeated status. Crash recovery resumes sessions, not external side effects. Background helper processes (orchestrator, evaluator) exit by themselves about 10 s after all work ends: never kill processes or delete files to 'clean up'. When the user quits pi, this session's running workflows pause (nothing is spent); resume continues them.",
343
343
  "run (action optional for exactly one launch form): agent+task; tasks:[call specs] parallel; chain:[call specs] sequential ({previous}); workflow:'./script.js' or source (runs.run(key,spec), runs.all([...]), emit(value), args, runs.input(name)). Optional name, cwd, usageBudget, maxCalls, inputs. With tasks/chain, top-level model, timeoutMs, budget, isolation, context, tools, skills, once are defaults for every step (a step's own value wins); a workflow/source script sets them per runs.run call. timeoutMs is milliseconds of active time (a number); omit it unless a hard limit is needed. Explicit unknown agents are rejected BEFORE creation, with available names; unknown script agents fail only their call.",
344
344
  "agents: list names, descriptions, default models and source for this cwd; use these names for run.",
345
- "send to:'<wid>/<key>' (bare '<wid>' only for a single-call workflow): steer on a running call delivers at the next safe point (receipt in status/UI); a steer to a call waiting on its question interrupts the question and the subagent usually asks again — use answer to answer it; sealed → finished:<status> — use kind 'follow-up'. follow-up continues a sealed call as generation g+1 or queues after a running turn; follow-up model:'provider/id' runs that generation on it. answer: give the qid (or just the call, or nothing when one question is open); to and rev are filled in. A question that needs the user's decision goes to the user; if you answer one yourself, tell the user what you chose. model: a running call switches at its next provider request; an asking, hibernated or queued call launches on it when it runs again; the reply's model/effect (next-request|next-execution|next-generation) says which. status model = model actually used by the last request; switching = requested, not used yet; switchFailed = refused. A provider content refusal (ToS/usage policy) fails the call at once, not retried. Unknown targets list valid addresses. replaces:[rid] supersedes an earlier send.",
345
+ "send to:'<wid>/<key>' (bare '<wid>' only for a single-call workflow): steer on a running call delivers at the next safe point (receipt in status/UI); a steer to a call waiting on its question interrupts the question and the subagent usually asks again — use answer to answer it; sealed → finished:<status> — use kind 'follow-up'. follow-up continues a sealed call as generation g+1 or queues after a running turn; follow-up model:'provider/id' or a pool name runs that generation on it. answer: give the qid (or just the call, or nothing when one question is open); to and rev are filled in. A question that needs the user's decision goes to the user; if you answer one yourself, tell the user what you chose. model ('provider/id' or a pool name — its first model not used up): a running call switches at its next provider request; an asking, hibernated or queued call launches on it when it runs again; the reply's model/effect (next-request|next-execution|next-generation) says which. status model = model actually used by the last request; switching = requested, not used yet; switchFailed = refused. A provider content refusal (ToS/usage policy) fails the call at once, not retried. Unknown targets list valid addresses. replaces:[rid] supersedes an earlier send.",
346
346
  "stop target:<wid|<wid>/<key>> is terminal stopped (usage and partial edits kept); a sealed call → already-sealed:<status>, a finished workflow → terminal:<status>. drain holds existing workflows reversibly (new runs unaffected); resume [wid] releases held workflows. status: without wid, what runs, asks (with its answer address; hibernated:true holds no slot) or failed, sharedWorktree names calls sharing observed edit/write roots (reminder only), finished workflows one line each, provider slots held/limit, the config in effect and providers whose usage window is used up (avoided until a probe finds them answering again), and the orchestrator version (versionNote when it differs from the loaded one); wid: one workflow, outputs clipped; wid+key: one call's full result; full:true: everything. A run's rid from {submitted:{rid}} works wherever a wid is expected. revise wid + workflow/source/args starts a revision.",
347
347
  "Control replies are {applied:true,rid} or {applied:false,reason,rid} when decided; otherwise {submitted:{rid}} after 10s.",
348
348
  ...(agents ? [`Available agents: ${agents}.`] : []),
@@ -331,7 +331,9 @@ export class Engine {
331
331
  const seal = wf.journal.entries().find(e => e.type === JT.sealed && e.call === from);
332
332
  if (seal && send.kind === 'steer')
333
333
  return { action: 'reject', reason: `finished:${seal.result.status} — use kind "follow-up" to continue it` };
334
- if (send.kind === 'follow-up' && send.model !== undefined) {
334
+ // A pool's name is a model too: the call keeps the pool, and its order and failover apply to the new generation.
335
+ const pools = this.ledgers.config.pools, pool = send.model !== undefined && pools && Object.hasOwn(pools, send.model);
336
+ if (send.kind === 'follow-up' && send.model !== undefined && !pool) {
335
337
  try {
336
338
  if (!parseModel(send.model).provider)
337
339
  throw new Error('missing provider');
@@ -346,7 +348,7 @@ export class Engine {
346
348
  const spec = send.model !== undefined ? { ...entry.spec, model: send.model } : entry.spec;
347
349
  if (send.model !== undefined)
348
350
  await this.note(req.rid, send.model, 'next-generation');
349
- const opened = await wf.journal.append('generation', { rid: req.rid, key: entry.key, gen, from, spec, revision: wf.revision, opening: { rid: req.rid, kind: send.kind, message: send.message ?? '' }, ...(send.model !== undefined ? { model: send.model } : {}) });
351
+ const opened = await wf.journal.append('generation', { rid: req.rid, key: entry.key, gen, from, spec, revision: wf.revision, opening: { rid: req.rid, kind: send.kind, message: send.message ?? '' }, ...(send.model !== undefined && !pool ? { model: send.model } : {}) });
350
352
  this.dispatchGeneration(wf, opened);
351
353
  return { action: 'apply' };
352
354
  }
@@ -34,6 +34,8 @@ import { WorktreeIndex, worktreeCalls, worktreeLabel, worktreePair, worktreeRoot
34
34
  const MEM_RECORD_MS = 30000;
35
35
  /** A4, P29: A child admitted within this window may not show in MemAvailable yet; its share is reserved explicitly. */
36
36
  const MEM_WARMUP_MS = 30000;
37
+ /** Quota refusals in a row that find a provider's usage window used up while pi is still retrying. */
38
+ const QUOTA_REFUSALS = 2;
37
39
  const ignoreMissing = (error) => { if (error.code !== "ENOENT")
38
40
  throw error; };
39
41
  const callOf = (exec) => exec.slice(0, exec.lastIndexOf("#"));
@@ -61,7 +63,8 @@ export function requestedModel(journal, call, followUp) {
61
63
  wanted = m;
62
64
  }
63
65
  for (const [index, e] of all.entries()) {
64
- if (e.type !== "forward" || e.dest !== call || e.envelope?.kind !== "model")
66
+ // A failover's switch is not a request: an execution that ends before applying it leaves the choice to the pool.
67
+ if (e.type !== "forward" || e.dest !== call || e.envelope?.kind !== "model" || e.failover)
65
68
  continue;
66
69
  const delivered = all.find(r => r.type === "forward-delivered" && r.call === call && r.rid2 === e.rid2);
67
70
  if (delivered) {
@@ -300,9 +303,9 @@ export default function createExecutor(ledgers, options = {}) {
300
303
  }
301
304
  /** P7, P27: Record forward-delivered once when a forward's child receipt is first observed; serial sections only. */
302
305
  /** The reply to a model send says which model and when it applies (orchestrator ledger `send-note`, once per rid). */
303
- async function note(rid, model, effect) {
306
+ async function note(rid, model, effect, pool) {
304
307
  if (!orch.entries().some(e => e.type === "send-note" && e.rid === rid))
305
- await orch.append("send-note", { rid, model, effect });
308
+ await orch.append("send-note", { rid, model, effect, ...(pool ? { pool } : {}) });
306
309
  }
307
310
  async function forwardsDelivered(journal, call, entries) {
308
311
  const all = journal.entries();
@@ -491,7 +494,7 @@ export default function createExecutor(ledgers, options = {}) {
491
494
  const decision = await decide(), models = decision.candidates, pool = decision.pool;
492
495
  continuation = decision.continuation;
493
496
  for (const model of models) {
494
- if (!continuation && pool && skipped(pool, model))
497
+ if (!continuation && pool && models.length > 1 && skipped(pool, model))
495
498
  continue;
496
499
  const provider = model.provider;
497
500
  if (unavailable(provider))
@@ -559,14 +562,19 @@ export default function createExecutor(ledgers, options = {}) {
559
562
  const candidate = recorded && candidates.some(m => m.provider === recorded.provider && m.id === recorded.id);
560
563
  // Leave the session's model for the pool's others when its pool skips it after losses, or its provider's usage
561
564
  // window is used up; and at a new generation, go back to the pool's order of preference.
562
- const skip = pool && candidate && (previous && ownSegment && skipped(pool, recorded) || unavailable(recorded.provider) || !ownSegment && !!t.continueFrom);
565
+ // A new generation of a pool call starts from the pool also when the session's model is not one of its models
566
+ // (switched outside it, or the follow-up named the pool).
567
+ const skip = pool && (candidate ? previous && ownSegment && skipped(pool, recorded) || unavailable(recorded.provider) || !ownSegment && !!t.continueFrom
568
+ : !ownSegment && !!t.continueFrom);
563
569
  // A model the call was asked to use replaces the session's: launched with it, and holding its provider's slot.
564
570
  const wanted = requestedModel(t.journal, t.callId, t.model);
565
571
  // It outranks the pool's order at a new generation too, also when it names the model the session already has.
572
+ // A requested model of the call's own pool keeps the pool: a used-up window still moves the call on.
573
+ const keep = pool && candidates.some(m => m.provider === wanted?.provider && m.id === wanted?.id) ? pool : undefined;
566
574
  if (wanted)
567
575
  return recorded && !freshFork && recorded.provider === wanted.provider && recorded.id === wanted.id
568
- ? { candidates: [recorded], continuation: true, pool: undefined }
569
- : { candidates: [wanted], continuation: false, pool: undefined };
576
+ ? { candidates: [recorded], continuation: true, pool: keep }
577
+ : { candidates: [wanted], continuation: false, pool: keep };
570
578
  if (recorded && !freshFork && !skip)
571
579
  return { candidates: [recorded], continuation: true, pool: candidate ? pool : undefined };
572
580
  return { candidates, continuation: false, pool };
@@ -634,15 +642,87 @@ export default function createExecutor(ledgers, options = {}) {
634
642
  // While a probe runs, its outcome alone decides: a late refusal of an execution admitted earlier changes nothing.
635
643
  if (x && (x.probe ? x.probe !== exec : now < x.nextTry))
636
644
  return;
637
- if (orch.entries().some(e => e.type === "provider-exhausted" && e.exec === exec))
645
+ // Once per execution and provider: an execution moved on by failover can find a second provider used up too.
646
+ if (orch.entries().some(e => e.type === "provider-exhausted" && e.exec === exec && e.provider === provider))
638
647
  return;
639
648
  await orch.append("provider-exhausted", { provider, exec, since: x?.since ?? now, nextTry: now + (config.k?.probeMs ?? 900_000), error: error.slice(0, 300) });
640
649
  }
650
+ /** Quota refusals in a row per execution, from one provider (pi retries a refused request on its own). */
651
+ const refusals = new Map();
652
+ /** A used-up window shows while pi still retries: the second refusal in a row (the first for a probe) finds the
653
+ * provider used up, and a call launched from a pool switches to the pool's next model at its next request. */
654
+ async function refused(t, exec, provider, error) {
655
+ if (!quotaExhausted(error))
656
+ return;
657
+ const last = refusals.get(exec), count = last?.provider === provider ? last.count + 1 : 1;
658
+ refusals.set(exec, { provider, count });
659
+ const probe = folded().exhausted.get(provider)?.probe === exec;
660
+ if (count < (probe ? 1 : QUOTA_REFUSALS))
661
+ return;
662
+ await serial(async () => {
663
+ if (has(t.journal, JT.fenced, exec) || current(t.journal, t.callId) !== exec)
664
+ return;
665
+ await recordExhausted(provider, exec, error);
666
+ await failover(t, exec, provider);
667
+ });
668
+ wake();
669
+ }
670
+ /** Switch a running execution off a used-up provider: to the first model of its pool on another provider that is
671
+ * neither used up nor full, reserving that slot as a requested switch does. Without one, pi's retries go on and
672
+ * the call waits for the provider once they end. */
673
+ async function failover(t, exec, provider) {
674
+ if (pendingSwitch(exec))
675
+ return;
676
+ const pool = t.journal.entries().findLast(e => e.type === "selected" && e.exec === exec)?.pool, pools = settings().pools;
677
+ if (!pool || !pools?.[pool])
678
+ return;
679
+ let models;
680
+ try {
681
+ models = resolveModel(pool, pools);
682
+ }
683
+ catch {
684
+ return;
685
+ }
686
+ for (const m of models) {
687
+ if (!m.provider || m.provider === provider || unavailable(m.provider) || skipped(pool, m))
688
+ continue;
689
+ const rid = contentHash([exec, "failover", provider]);
690
+ if (t.journal.entries().some(e => e.type === "forward" && e.rid === rid))
691
+ return;
692
+ const probe = folded().exhausted.has(m.provider); // its next try is due (`unavailable` said so): this is its probe
693
+ if (!await reserveSwitch(exec, m.provider, rid))
694
+ continue;
695
+ if (probe)
696
+ await orch.append("provider-probe", { provider: m.provider, exec });
697
+ // Bound to this execution: replayed after it ended, a later execution (which chose its model at launch) refuses it.
698
+ const body = { provider: m.provider, model: m.id, ...(m.thinking ? { thinking: m.thinking } : {}), exec };
699
+ const envelope = { to: t.callId, kind: "model", body }, hash = contentHash(envelope);
700
+ const entry = await t.journal.append("forward", { rid, rid2: forwardRid(rid, t.callId.slice(0, t.callId.indexOf("/")), t.key, hash), dest: t.callId, hash, envelope, failover: provider });
701
+ await replayForward(entry);
702
+ return;
703
+ }
704
+ }
705
+ /** Hold a slot of `provider` for a running execution's switch, unless it holds one; false when it is full. */
706
+ async function reserveSwitch(exec, provider, rid) {
707
+ if (holdings().some(h => h.exec === exec && h.pool === provider))
708
+ return true;
709
+ const target = holdings().filter(h => h.pool === provider);
710
+ if (!capacity({ kind: "provider", holders: target.length, capacity: settings().providers?.[provider]?.slots ?? Infinity }))
711
+ return false;
712
+ let slot = 0;
713
+ while (target.some(h => h.slot === slot))
714
+ slot++;
715
+ await orch.append("hold", { pool: provider, slot, exec, reserved: true, rid });
716
+ return true;
717
+ }
641
718
  /** An answer from a used-up provider, requested after it was found used up: available again. */
642
- async function answered(exec, event) {
719
+ async function answered(t, exec, event) {
643
720
  const message = event.message, provider = message?.provider;
644
- if (message?.role !== "assistant" || !provider || message.stopReason === "error")
721
+ if (message?.role !== "assistant" || !provider)
645
722
  return;
723
+ if (message.stopReason === "error")
724
+ return refused(t, exec, provider, String(message.errorMessage ?? ""));
725
+ refusals.delete(exec);
646
726
  await serial(async () => {
647
727
  const x = folded().exhausted.get(provider);
648
728
  if (x && (x.probe === exec || Number(message.timestamp) > x.since))
@@ -836,7 +916,7 @@ export default function createExecutor(ledgers, options = {}) {
836
916
  });
837
917
  }, recordUsage: values => recordUsage(t, values),
838
918
  wrote: path => wrote(t, exec, cwd, path),
839
- switched: event => switched(exec, journal, event), answered: event => answered(exec, event), pendingSwitch: () => pendingSwitch(exec),
919
+ switched: event => switched(exec, journal, event), answered: event => answered(t, exec, event), pendingSwitch: () => pendingSwitch(exec),
840
920
  });
841
921
  }
842
922
  finally {
@@ -881,6 +961,9 @@ export default function createExecutor(ledgers, options = {}) {
881
961
  for (const e of inUse.keys())
882
962
  if (callOf(e) === ticket.callId)
883
963
  inUse.delete(e);
964
+ for (const e of refusals.keys())
965
+ if (callOf(e) === ticket.callId)
966
+ refusals.delete(e);
884
967
  forgetSession(callSession(home, ticket.wid, ticket.key, ticket.gen));
885
968
  wake();
886
969
  }
@@ -931,6 +1014,28 @@ export default function createExecutor(ledgers, options = {}) {
931
1014
  /** P12: a model request to this call, recorded with the rid given; a reject has no effect. */
932
1015
  const requestModel = async (rid, model, hash, cond) => {
933
1016
  let body;
1017
+ const exec = current(ctx.journal, dest);
1018
+ // P28: with no live execution (not started yet, between executions, hibernated while asking) the model is
1019
+ // recorded and the next execution launches on it (`requestedModel`); its slot is acquired then, as for any launch.
1020
+ const idle = !exec || has(ctx.journal, JT.fenced, exec) || !has(ctx.journal, "selected", exec);
1021
+ // A pool's name asks for its first model that can take the call now: provider not used up, and (for a running
1022
+ // call) a free slot. The call's own pool stays, so a used-up window later moves it on as before.
1023
+ const pools = settings().pools, pool = pools && Object.hasOwn(pools, model) ? model : undefined;
1024
+ if (pool) {
1025
+ let models;
1026
+ try {
1027
+ models = resolveModel(pool, pools);
1028
+ }
1029
+ catch {
1030
+ return { action: "reject", reason: "unknown-model" };
1031
+ }
1032
+ const free = (p) => idle || holdings().some(h => h.exec === exec && h.pool === p) ||
1033
+ capacity({ kind: "provider", holders: holdings().filter(h => h.pool === p).length, capacity: settings().providers?.[p]?.slots ?? Infinity });
1034
+ const m = models.find(m => m.provider && !unavailable(m.provider) && free(m.provider));
1035
+ if (!m)
1036
+ return { action: "reject", reason: "pool-unavailable" };
1037
+ model = `${m.provider}/${m.id}${m.thinking ? `:${m.thinking}` : ""}`;
1038
+ }
934
1039
  try {
935
1040
  const m = parseModel(model);
936
1041
  if (!m.provider)
@@ -942,28 +1047,14 @@ export default function createExecutor(ledgers, options = {}) {
942
1047
  }
943
1048
  const envelope = { to: dest, kind: "model", body, ...(cond && Object.keys(cond).length ? { cond } : {}) };
944
1049
  const rid2 = forwardRid(rid, ctx.widRev, ctx.key, hash);
945
- const exec = current(ctx.journal, dest), provider = body.provider;
946
- // P28: with no live execution (not started yet, between executions, hibernated while asking) the model is
947
- // recorded and the next execution launches on it (`requestedModel`); its slot is acquired then, as for any launch.
948
1050
  // Launching (`selected`, not `tracked` yet): the child may start on the old model; ask again in a moment.
949
- const idle = !exec || has(ctx.journal, JT.fenced, exec) || !has(ctx.journal, "selected", exec);
950
1051
  if (!idle && !has(ctx.journal, "tracked", exec))
951
1052
  return { action: "reject", reason: "call-starting" };
952
1053
  if (!idle && pendingSwitch(exec))
953
1054
  return { action: "reject", reason: "switch-pending" };
954
- if (!idle) {
955
- const held = holdings().filter(h => h.exec === exec);
956
- if (!held.some(h => h.pool === provider)) {
957
- const target = holdings().filter(h => h.pool === provider);
958
- if (!capacity({ kind: "provider", holders: target.length, capacity: settings().providers?.[provider]?.slots ?? Infinity }))
959
- return { action: "reject", reason: "provider-full" };
960
- let slot = 0;
961
- while (target.some(h => h.slot === slot))
962
- slot++;
963
- await orch.append("hold", { pool: provider, slot, exec, reserved: true, rid });
964
- }
965
- }
966
- await note(req.rid, model, idle ? "next-execution" : "next-request");
1055
+ if (!idle && !await reserveSwitch(exec, body.provider, rid))
1056
+ return { action: "reject", reason: "provider-full" };
1057
+ await note(req.rid, model, idle ? "next-execution" : "next-request", pool);
967
1058
  const entry = await ctx.journal.append("forward", { rid, rid2, dest, hash, envelope });
968
1059
  await replayForward(entry);
969
1060
  if (idle) {
@@ -261,7 +261,7 @@ function snapshotReducer(wid, entries) {
261
261
  const pending = forwarded.filter(pendingMessage).length;
262
262
  if (pending)
263
263
  c.pending = pending;
264
- const last = c.phase !== "sealed" && !retired.has(c.callId) ? forwarded.findLast(s => s.kind === "model" && s.reason !== "withdrawn") : undefined;
264
+ const last = c.phase !== "sealed" && !retired.has(c.callId) ? forwarded.findLast(s => s.kind === "model" && s.reason !== "withdrawn" && s.reason !== "stale-execution") : undefined;
265
265
  if (last?.reason !== undefined)
266
266
  c.switchFailed = `${last.model} (${last.reason})`;
267
267
  else if (last && last.state !== "retired" && last.model !== c.model?.replace(/:(off|minimal|low|medium|high|xhigh|max)$/, ""))
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-durable-subagents",
3
- "version": "1.0.15",
3
+ "version": "1.0.16",
4
4
  "description": "Subagents for pi that never lose work and never do it twice. Crash-safe workflows, automatic recovery, and a live view just like the main agent.",
5
5
  "type": "module",
6
6
  "license": "MIT",