pi-durable-subagents 1.0.9 → 1.0.11
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +25 -0
- package/README.md +9 -1
- package/dist/agent/main.js +1 -1
- package/dist/cli/main.js +1 -1
- package/dist/orchestrator/config.js +1 -1
- package/dist/orchestrator/executor/effects/gate.js +9 -6
- package/dist/orchestrator/executor/index.js +62 -11
- package/dist/orchestrator/executor/observe.js +3 -1
- package/dist/orchestrator/executor/session.js +28 -5
- package/dist/orchestrator/executor/sweep.js +21 -2
- package/dist/orchestrator/providers.js +21 -0
- package/dist/orchestrator/snapshot.js +6 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,30 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.0.11
|
|
4
|
+
|
|
5
|
+
- A gate whose processes outlive their fence is reported once: when the gate
|
|
6
|
+
and the background sweep both recorded the failure at the same moment, a
|
|
7
|
+
workflow could show the same "processes may still run" attention item twice
|
|
8
|
+
(and the failure, the gate's unknown outcome and the resolution likewise).
|
|
9
|
+
|
|
10
|
+
## 1.0.10
|
|
11
|
+
|
|
12
|
+
- Provider failover for a used-up usage window. An error such as `503 No
|
|
13
|
+
available accounts`, "usage limit" or "quota exceeded" (after pi's own
|
|
14
|
+
retries) marks the provider used up instead of counting as a lost
|
|
15
|
+
execution: a call in a pool continues in the same session on the pool's next
|
|
16
|
+
model, and new calls skip the provider. After `k.probeMs` (15 minutes) the
|
|
17
|
+
next call that wants it is admitted to it alone; when it answers, new calls
|
|
18
|
+
and new generations go back to it. A call with a single model waits for the
|
|
19
|
+
provider instead of failing. `status` lists used-up providers with their
|
|
20
|
+
next try. Billing errors (402, insufficient balance) still fail at once.
|
|
21
|
+
A short request rate limit ("429 … resets in 1 second") is not a used-up
|
|
22
|
+
window and is retried as before. A follow-up naming a model runs on it even
|
|
23
|
+
where the pool would start over.
|
|
24
|
+
- After a call moved to another model by a relaunch, the orchestrator now
|
|
25
|
+
reads the session's model as pi restores it (from the last answer), so a
|
|
26
|
+
later relaunch holds the slot of the provider it actually uses.
|
|
27
|
+
|
|
3
28
|
## 1.0.9
|
|
4
29
|
|
|
5
30
|
- `status` shows the model a call actually uses: the model of its last
|
package/README.md
CHANGED
|
@@ -55,6 +55,7 @@ the npx cache, so `install-service` refuses to run from there.
|
|
|
55
55
|
| You steer a subagent while it is asking you a question | Your message reaches it, in order. Nothing is rejected or lost. |
|
|
56
56
|
| Two steers arrive out of order and the second replaces the first | Only the second one applies. |
|
|
57
57
|
| A step is refused, or a dependency fails | The workflow stops that branch cleanly. Nothing is retried in vain. |
|
|
58
|
+
| A provider's usage window runs out (`No available accounts`, usage limit, quota exceeded) | A call in a pool continues **in the same session** on the pool's next model; new calls skip that provider. After 15 minutes the next call that wants it tries it once; when it answers, new calls and new generations use it again. A call with a single model waits for it instead of failing. Billing errors (402, insufficient balance) still fail at once. |
|
|
58
59
|
| A subagent waits for an answer for a long time | It releases its model slot and memory, then resumes exactly once when you answer. |
|
|
59
60
|
|
|
60
61
|
## Use it
|
|
@@ -272,7 +273,14 @@ State lives in `~/.pi/durable-subagents`; set `DSA_HOME` to move it.
|
|
|
272
273
|
```
|
|
273
274
|
|
|
274
275
|
- **Pools:** a model can name a pool. The first candidate with a free slot is
|
|
275
|
-
used, and a candidate that keeps failing is skipped for 10 minutes.
|
|
276
|
+
used, and a candidate that keeps failing is skipped for 10 minutes. The
|
|
277
|
+
order is the preference: list the provider you want to use first.
|
|
278
|
+
- **A used-up provider** is not sent new calls until its next try, 15 minutes
|
|
279
|
+
after it last refused (`"k": { "probeMs": 900000 }`). Then one call at a
|
|
280
|
+
time goes to it, so finding out costs no extra request. A call that moved to
|
|
281
|
+
another provider stays there for the rest of its generation (switching back
|
|
282
|
+
mid-task would lose the prompt cache); a follow-up starts on the first
|
|
283
|
+
candidate again. `status` lists each used-up provider with its next try.
|
|
276
284
|
- **Provider slots:** never exceeded, including while a model switch is in
|
|
277
285
|
progress.
|
|
278
286
|
- **Memory:** new subagents wait while memory is short. Running ones are
|
package/dist/agent/main.js
CHANGED
|
@@ -326,7 +326,7 @@ export function registerMain(pi, ui) {
|
|
|
326
326
|
"run (action optional for exactly one launch form): agent+task; tasks:[call specs] parallel; chain:[call specs] sequential ({previous}); workflow:'./script.js' or source (runs.run(key,spec), runs.all([...]), emit(value), args, runs.input(name)). Optional name, cwd, usageBudget, maxCalls, inputs. With tasks/chain, top-level model, timeoutMs, budget, isolation, context, tools, skills, once are defaults for every step (a step's own value wins); a workflow/source script sets them per runs.run call. timeoutMs is milliseconds of active time (a number); omit it unless a hard limit is needed. Explicit unknown agents are rejected BEFORE creation, with available names; unknown script agents fail only their call.",
|
|
327
327
|
"agents: list names, descriptions, default models and source for this cwd; use these names for run.",
|
|
328
328
|
"send to:'<wid>/<key>' (bare '<wid>' only for a single-call workflow): steer on a running call delivers at the next safe point (receipt in status/UI); a steer to a call waiting on its question interrupts the question and the subagent usually asks again — use answer to answer it; sealed → finished:<status> — use kind 'follow-up'. follow-up continues a sealed call as generation g+1 or queues after a running turn; follow-up model:'provider/id' runs that generation on it. answer: give the qid (or just the call, or nothing when one question is open); to and rev are filled in. A question that needs the user's decision goes to the user; if you answer one yourself, tell the user what you chose. model: a running call switches at its next provider request; an asking, hibernated or queued call launches on it when it runs again; the reply's model/effect (next-request|next-execution|next-generation) says which. status model = model actually used by the last request; switching = requested, not used yet; switchFailed = refused. A provider content refusal (ToS/usage policy) fails the call at once, not retried. Unknown targets list valid addresses. replaces:[rid] supersedes an earlier send.",
|
|
329
|
-
"stop target:<wid|<wid>/<key>> is terminal stopped (usage and partial edits kept); a sealed call → already-sealed:<status>, a finished workflow → terminal:<status>. drain holds existing workflows reversibly (new runs unaffected); resume [wid] releases held workflows. status: without wid, what runs, asks (with its answer address; hibernated:true holds no slot) or failed, finished workflows one line each, provider slots held/limit
|
|
329
|
+
"stop target:<wid|<wid>/<key>> is terminal stopped (usage and partial edits kept); a sealed call → already-sealed:<status>, a finished workflow → terminal:<status>. drain holds existing workflows reversibly (new runs unaffected); resume [wid] releases held workflows. status: without wid, what runs, asks (with its answer address; hibernated:true holds no slot) or failed, finished workflows one line each, provider slots held/limit, the config in effect and providers whose usage window is used up (avoided until a probe finds them answering again); wid: one workflow, outputs clipped; wid+key: one call's full result; full:true: everything. A run's rid from {submitted:{rid}} works wherever a wid is expected. revise wid + workflow/source/args starts a revision.",
|
|
330
330
|
"Control replies are {applied:true,rid} or {applied:false,reason,rid} when decided; otherwise {submitted:{rid}} after 10s.",
|
|
331
331
|
...(agents ? [`Available agents: ${agents}.`] : []),
|
|
332
332
|
"User sees a summary line above the editor; ↓ on an empty editor (or /subagents) opens the list, Enter watches live OR finished calls (finished transcripts remain on disk) and expands finished workflows. List keys: s steer (paste-capable input), x stop (confirm y), m model, a answer when asked, f follow-up on finished calls; action feedback appears in footer.",
|
package/dist/cli/main.js
CHANGED
|
@@ -71,7 +71,7 @@ export function renderView(view) {
|
|
|
71
71
|
lines.unshift(`${view.paused} (pi-durable-subagents resume)`);
|
|
72
72
|
if (view.olderFinished)
|
|
73
73
|
lines.push(`(+${view.olderFinished} older finished workflows; status <wid> shows one in detail)`);
|
|
74
|
-
const footer = [view.slots?.length ? `slots: ${view.slots.join(", ")}` : "", view.config ? `config: ${view.config}` : "", view.configRejected ? `config.json rejected: ${view.configRejected}` : ""].filter(Boolean);
|
|
74
|
+
const footer = [view.slots?.length ? `slots: ${view.slots.join(", ")}` : "", view.config ? `config: ${view.config}` : "", view.configRejected ? `config.json rejected: ${view.configRejected}` : "", ...(view.exhausted ?? [])].filter(Boolean);
|
|
75
75
|
if (!lines.length)
|
|
76
76
|
lines.push("No workflows");
|
|
77
77
|
return [...lines, ...footer].join("\n");
|
|
@@ -9,7 +9,7 @@ import { join } from "node:path";
|
|
|
9
9
|
import { contentHash } from "../kernel/ids.js";
|
|
10
10
|
/** The keys the orchestrator reads; config.json also holds pi-side settings (ui, onQuit) that it ignores. */
|
|
11
11
|
const KEYS = ["defaultModel", "pools", "providers", "memory", "k"];
|
|
12
|
-
const K = ["lossBound", "checkpointMs", "stallMs", "progressMs", "switchTimeoutMs", "idleExitMs", "trackerMs", "hibernateMs", "spawnBudget"];
|
|
12
|
+
const K = ["lossBound", "checkpointMs", "stallMs", "progressMs", "switchTimeoutMs", "idleExitMs", "trackerMs", "hibernateMs", "spawnBudget", "probeMs"];
|
|
13
13
|
export const configPath = (home) => join(home, "config.json");
|
|
14
14
|
/** The orchestrator's part of a parsed config.json. */
|
|
15
15
|
export function orchestratorSettings(raw) {
|
|
@@ -5,15 +5,17 @@ import { join } from "node:path";
|
|
|
5
5
|
import { publishFile } from "../../../kernel/mailbox.js";
|
|
6
6
|
import { callDir } from "../../../paths.js";
|
|
7
7
|
import { validate } from "../../../agent/child/schema.js";
|
|
8
|
-
import { recordFenceFailure,
|
|
8
|
+
import { fenceAttentionResolved, recordFenceFailure, recordOnce } from "../sweep.js";
|
|
9
9
|
const tracked = (journal, id) => journal.entries().filter(e => e.type === "gate-tracked" && e.id === id).map(e => e.process);
|
|
10
10
|
// Calls parked on an unfenced gate, woken when the sweep records the gate's outcome.
|
|
11
11
|
const parked = new Map();
|
|
12
12
|
/** P30, F1: A gate proven retired after a failed fence gets its unknown outcome once and its attention resolved. */
|
|
13
13
|
export async function gateRetired(journal, id) {
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
14
|
+
await recordOnce(journal, async () => {
|
|
15
|
+
if (!journal.entries().some(e => e.type === "gate" && e.id === id))
|
|
16
|
+
await journal.append("gate", { id, unknown: true });
|
|
17
|
+
await fenceAttentionResolved(journal, id);
|
|
18
|
+
});
|
|
17
19
|
for (const wake of parked.get(id) ?? [])
|
|
18
20
|
wake();
|
|
19
21
|
}
|
|
@@ -52,8 +54,9 @@ async function fenceGate(journal, intent, containment) {
|
|
|
52
54
|
console.error(`durable-subagents: fence of ${id} failed: ${String(error)}`);
|
|
53
55
|
if (/Fence timeout/.test(String(error)))
|
|
54
56
|
await recordFenceFailure(journal, id, String(intent.call), error);
|
|
55
|
-
else
|
|
56
|
-
await journal.
|
|
57
|
+
else
|
|
58
|
+
await recordOnce(journal, async () => { if (!journal.entries().some(e => e.type === "fence-failed" && e.exec === id))
|
|
59
|
+
await journal.append("fence-failed", { exec: id, error: String(error) }); });
|
|
57
60
|
return false;
|
|
58
61
|
}
|
|
59
62
|
}
|
|
@@ -22,7 +22,8 @@ import { buildCallResult } from "../../compat/result.js";
|
|
|
22
22
|
import createEffects from "./effects/index.js";
|
|
23
23
|
import { continueSession } from "./generation.js";
|
|
24
24
|
import { hibernation, openQuestion } from "./hibernate.js";
|
|
25
|
-
import {
|
|
25
|
+
import { foldExhaustion } from "../providers.js";
|
|
26
|
+
import { evidence, fatalProviderError, quotaExhausted, refusedByProvider, forgetSession, readSessionState, receiptId, sessionModel } from "./session.js";
|
|
26
27
|
import { activeTotal } from "./time.js";
|
|
27
28
|
import { observeExecution } from "./observe.js";
|
|
28
29
|
import { availableMemory } from "./memory.js";
|
|
@@ -115,7 +116,7 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
115
116
|
const wake = () => { for (const fn of waiters)
|
|
116
117
|
fn(); waiters.clear(); };
|
|
117
118
|
// F3: fold only orchestrator entries appended since the last fold (holdings, observed switches, K7 skips).
|
|
118
|
-
const ledger = { seen: 0, held: new Map(), observed: new Set(), skips: new Map() };
|
|
119
|
+
const ledger = { seen: 0, held: new Map(), observed: new Set(), skips: new Map(), exhausted: new Map() };
|
|
119
120
|
const folded = () => {
|
|
120
121
|
const entries = orch.entries();
|
|
121
122
|
for (; ledger.seen < entries.length; ledger.seen++) {
|
|
@@ -128,11 +129,17 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
128
129
|
ledger.observed.add(`${e.exec}\n${e.rid}`);
|
|
129
130
|
else if (e.type === "skip")
|
|
130
131
|
ledger.skips.set(`${e.pool}\n${e.model}`, Math.max(Number(e.until), ledger.skips.get(`${e.pool}\n${e.model}`) ?? 0));
|
|
132
|
+
foldExhaustion(ledger.exhausted, e);
|
|
131
133
|
}
|
|
132
134
|
return ledger;
|
|
133
135
|
};
|
|
134
136
|
const holdings = () => [...folded().held.values()];
|
|
135
137
|
const skipped = (pool, model) => (folded().skips.get(`${pool}\n${model.provider}/${model.id}`) ?? 0) > Date.now();
|
|
138
|
+
/** A provider whose usage window is used up admits no call until its next try, and then one probe at a time. */
|
|
139
|
+
const unavailable = (provider) => {
|
|
140
|
+
const x = provider ? folded().exhausted.get(provider) : undefined;
|
|
141
|
+
return !!x && (Date.now() < x.nextTry || x.probe !== undefined);
|
|
142
|
+
};
|
|
136
143
|
async function release(exec) {
|
|
137
144
|
await serial(async () => { for (const h of holdings().filter(e => e.exec === exec))
|
|
138
145
|
await orch.append("release", { pool: h.pool, slot: h.slot, exec }); });
|
|
@@ -443,6 +450,9 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
443
450
|
if (!continuation && pool && skipped(pool, model))
|
|
444
451
|
continue;
|
|
445
452
|
const provider = model.provider;
|
|
453
|
+
if (unavailable(provider))
|
|
454
|
+
continue;
|
|
455
|
+
const probe = provider !== undefined && folded().exhausted.has(provider);
|
|
446
456
|
const holders = holdings().filter(e => e.pool === provider);
|
|
447
457
|
const limit = config.providers?.[provider ?? ""]?.slots ?? Infinity;
|
|
448
458
|
if (!capacity({ kind: "provider", holders: holders.length, capacity: limit }))
|
|
@@ -470,6 +480,10 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
470
480
|
while (holders.some(e => e.slot === slot))
|
|
471
481
|
slot++;
|
|
472
482
|
await orch.append("hold", { pool: provider, slot, exec });
|
|
483
|
+
// After its next try, the first call admitted to a used-up provider is its probe: its first request
|
|
484
|
+
// either goes through (the provider is available again) or is refused, which uses no quota.
|
|
485
|
+
if (probe)
|
|
486
|
+
await orch.append("provider-probe", { provider, exec });
|
|
473
487
|
}
|
|
474
488
|
await a.ticket.journal.append("selected", { exec, model, ...(pool ? { pool } : {}) });
|
|
475
489
|
return model;
|
|
@@ -498,11 +512,16 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
498
512
|
const pool = raw && config.pools?.[raw] ? raw : undefined;
|
|
499
513
|
const candidates = raw ? resolveModel(raw, config.pools) : [{ id: "" }];
|
|
500
514
|
const candidate = recorded && candidates.some(m => m.provider === recorded.provider && m.id === recorded.id);
|
|
501
|
-
|
|
515
|
+
// Leave the session's model for the pool's others when its pool skips it after losses, or its provider's usage
|
|
516
|
+
// window is used up; and at a new generation, go back to the pool's order of preference.
|
|
517
|
+
const skip = pool && candidate && (previous && ownSegment && skipped(pool, recorded) || unavailable(recorded.provider) || !ownSegment && !!t.continueFrom);
|
|
502
518
|
// A model the call was asked to use replaces the session's: launched with it, and holding its provider's slot.
|
|
503
519
|
const wanted = requestedModel(t.journal, t.callId, t.model);
|
|
504
|
-
|
|
505
|
-
|
|
520
|
+
// It outranks the pool's order at a new generation too, also when it names the model the session already has.
|
|
521
|
+
if (wanted)
|
|
522
|
+
return recorded && !freshFork && recorded.provider === wanted.provider && recorded.id === wanted.id
|
|
523
|
+
? { candidates: [recorded], continuation: true, pool: undefined }
|
|
524
|
+
: { candidates: [wanted], continuation: false, pool: undefined };
|
|
506
525
|
if (recorded && !freshFork && !skip)
|
|
507
526
|
return { candidates: [recorded], continuation: true, pool: candidate ? pool : undefined };
|
|
508
527
|
return { candidates, continuation: false, pool };
|
|
@@ -560,6 +579,32 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
560
579
|
});
|
|
561
580
|
wake();
|
|
562
581
|
}
|
|
582
|
+
/** The model an execution last answered with, or was launched on. */
|
|
583
|
+
function modelOf(journal, exec) {
|
|
584
|
+
return journal.entries().findLast(e => (e.type === "selected" || e.type === "model-used") && e.exec === exec)?.model;
|
|
585
|
+
}
|
|
586
|
+
/** Record a used-up provider once per window: again only when its probe (or any call after the next try) is refused. */
|
|
587
|
+
async function recordExhausted(provider, exec, error) {
|
|
588
|
+
const x = folded().exhausted.get(provider), now = Date.now();
|
|
589
|
+
// While a probe runs, its outcome alone decides: a late refusal of an execution admitted earlier changes nothing.
|
|
590
|
+
if (x && (x.probe ? x.probe !== exec : now < x.nextTry))
|
|
591
|
+
return;
|
|
592
|
+
if (orch.entries().some(e => e.type === "provider-exhausted" && e.exec === exec))
|
|
593
|
+
return;
|
|
594
|
+
await orch.append("provider-exhausted", { provider, exec, since: x?.since ?? now, nextTry: now + (config.k?.probeMs ?? 900_000), error: error.slice(0, 300) });
|
|
595
|
+
}
|
|
596
|
+
/** An answer from a used-up provider, requested after it was found used up: available again. */
|
|
597
|
+
async function answered(exec, event) {
|
|
598
|
+
const message = event.message, provider = message?.provider;
|
|
599
|
+
if (message?.role !== "assistant" || !provider || message.stopReason === "error")
|
|
600
|
+
return;
|
|
601
|
+
await serial(async () => {
|
|
602
|
+
const x = folded().exhausted.get(provider);
|
|
603
|
+
if (x && (x.probe === exec || Number(message.timestamp) > x.since))
|
|
604
|
+
await orch.append("provider-available", { provider, exec });
|
|
605
|
+
});
|
|
606
|
+
wake();
|
|
607
|
+
}
|
|
563
608
|
function pendingSwitch(exec) {
|
|
564
609
|
return holdings().find(h => h.exec === exec && h.reserved && !folded().observed.has(`${exec}\n${h.rid}`));
|
|
565
610
|
}
|
|
@@ -639,11 +684,17 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
639
684
|
// A refusal of the content is deterministic: the same request is refused again, so it is reported, not retried.
|
|
640
685
|
if (has(journal, "settled", exec) && !ev.text && ev.error && refusedByProvider(ev.error))
|
|
641
686
|
return finish(journal, t.callId, exec, makeResult("failed", "", `Refused by the provider (not retried): ${ev.error.slice(0, 500)}`));
|
|
642
|
-
|
|
643
|
-
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
687
|
+
// A used-up usage window is no loss: the provider is avoided until a probe finds it accepting requests again,
|
|
688
|
+
// and the call goes on with the pool's next model, or waits for that provider.
|
|
689
|
+
const exhausted = has(journal, "settled", exec) && !ev.text && ev.error && quotaExhausted(ev.error) ? modelOf(journal, exec)?.provider : undefined;
|
|
690
|
+
if (exhausted)
|
|
691
|
+
await serial(() => recordExhausted(exhausted, exec, ev.error));
|
|
692
|
+
else
|
|
693
|
+
await serial(async () => {
|
|
694
|
+
if (!has(journal, "loss", exec))
|
|
695
|
+
await journal.append("loss", { exec });
|
|
696
|
+
await skipLostCandidate(journal, orch, exec);
|
|
697
|
+
});
|
|
647
698
|
const losses = journal.entries().filter(e => e.type === "loss" && String(e.exec).startsWith(`${t.callId}#`)).length;
|
|
648
699
|
if (losses >= (config.k?.lossBound ?? 5))
|
|
649
700
|
return finish(journal, t.callId, exec, makeResult("failed", "", `lost ×${losses}${ev.error ? `; last error: ${ev.error.slice(0, 300)}` : ""}`));
|
|
@@ -739,7 +790,7 @@ export default function createExecutor(ledgers, options = {}) {
|
|
|
739
790
|
}
|
|
740
791
|
});
|
|
741
792
|
}, recordUsage: values => recordUsage(t, values),
|
|
742
|
-
switched: event => switched(exec, journal, event), pendingSwitch: () => pendingSwitch(exec),
|
|
793
|
+
switched: event => switched(exec, journal, event), answered: event => answered(exec, event), pendingSwitch: () => pendingSwitch(exec),
|
|
743
794
|
});
|
|
744
795
|
}
|
|
745
796
|
finally {
|
|
@@ -110,8 +110,10 @@ export async function observeExecution(d) {
|
|
|
110
110
|
await serial(() => t.journal.append("observation", { exec, event: slim }));
|
|
111
111
|
if (event.type === "message_start")
|
|
112
112
|
await d.switched(event);
|
|
113
|
-
if (event.type === "message_end")
|
|
113
|
+
if (event.type === "message_end") {
|
|
114
114
|
await d.recordUsage([{ id: String(slim.id), usage: slim.usage }]);
|
|
115
|
+
await d.answered?.(event);
|
|
116
|
+
}
|
|
115
117
|
}
|
|
116
118
|
await limits();
|
|
117
119
|
await stall();
|
|
@@ -107,9 +107,24 @@ export function evidence(entries, exec) {
|
|
|
107
107
|
const error = last?.stopReason === "error" ? last.errorMessage : undefined;
|
|
108
108
|
return { report, budget, text, error, dangling: [...tools].map(([id, name]) => `${name} (${id})`), usage };
|
|
109
109
|
}
|
|
110
|
-
/** Only explicit
|
|
110
|
+
/** Only explicit payment failures are terminal; rate limits, overload and transport errors still retry, and a used-up
|
|
111
|
+
* usage window (`quotaExhausted`) waits for the provider or moves to another one. */
|
|
111
112
|
export function fatalProviderError(text) {
|
|
112
|
-
return /\b402\b|insufficient[_ ]?(quota|balance|funds)|
|
|
113
|
+
return /\b402\b|insufficient[_ ]?(quota|balance|funds)|billing|credit balance|余额/i.test(text);
|
|
114
|
+
}
|
|
115
|
+
/** A provider's usage window is used up: its requests are refused (and not counted) until the window resets, hours
|
|
116
|
+
* later. Seen as a gateway's `503 No available accounts` once pi's own retries are spent, or a usage-limit message.
|
|
117
|
+
* The provider is then avoided until a probe finds it accepting requests again. */
|
|
118
|
+
export function quotaExhausted(text) {
|
|
119
|
+
if (fatalProviderError(text))
|
|
120
|
+
return false;
|
|
121
|
+
if (/no available accounts?/i.test(text))
|
|
122
|
+
return true;
|
|
123
|
+
// A request rate limit clears in seconds ("rate limit exceeded; resets in 1 second", "quota exceeded for requests
|
|
124
|
+
// per minute"): pi's retries and the lost-execution path handle it; it must not take the provider out for minutes.
|
|
125
|
+
if (/rate.?limit|too many requests|request limit|per (second|minute)|\b[RT]PM\b|resets? in \d+ ?(ms|s|secs?|seconds?|minutes?)\b/i.test(text))
|
|
126
|
+
return false;
|
|
127
|
+
return /usage limit|quota (exceeded|exhausted)|exceeded your (current )?(usage|quota)|limit (reached|exceeded)[^.]*resets?\b|额度/i.test(text);
|
|
113
128
|
}
|
|
114
129
|
/** A refusal of the request's content (terms of service, usage or content policy): the same request is refused again,
|
|
115
130
|
* on this provider and usually on another, so it is reported at once instead of retried as a lost execution. */
|
|
@@ -119,8 +134,16 @@ export function refusedByProvider(text) {
|
|
|
119
134
|
return false;
|
|
120
135
|
return /terms of service|usage polic(y|ies)|acceptable use|content[_ ]?(policy|filter|management policy)|safety (system|filter)|flagged as (unsafe|harmful)/i.test(text);
|
|
121
136
|
}
|
|
122
|
-
/** P13, C8: Restore the effective provider from the
|
|
137
|
+
/** P13, C8: Restore the effective provider as pi does: from the last model change or assistant message. */
|
|
123
138
|
export function sessionModel(entries) {
|
|
124
|
-
|
|
125
|
-
|
|
139
|
+
// pi restores the model of the last model change or assistant message: a relaunch with `--model` on an existing
|
|
140
|
+
// session records no model change, so only the answer tells which model the session went on with.
|
|
141
|
+
const last = entries.findLast(e => e.type === "model_change" && e.provider && e.modelId
|
|
142
|
+
|| e.type === "message" && e.message?.role === "assistant" && !!e.message.provider && !!e.message.model);
|
|
143
|
+
if (!last)
|
|
144
|
+
return undefined;
|
|
145
|
+
if (last.type === "model_change")
|
|
146
|
+
return { provider: last.provider, id: last.modelId };
|
|
147
|
+
const m = last.message;
|
|
148
|
+
return { provider: m.provider, id: m.model };
|
|
126
149
|
}
|
|
@@ -21,8 +21,23 @@ export async function skipLostCandidate(journal, orch, exec) {
|
|
|
21
21
|
if (tail.length === 3 && tail.every(e => e.type === "candidate-loss" && e.model === name))
|
|
22
22
|
await orch.append("skip", { pool: selected.pool, model: name, until: Date.now() + 600000 });
|
|
23
23
|
}
|
|
24
|
+
// The gate path records outside the executor's serial section and the sweep inside it: each check-then-append of a
|
|
25
|
+
// fence record runs in this per-journal section, so two recorders that both find no record still write it once (A5).
|
|
26
|
+
const sections = new WeakMap();
|
|
27
|
+
/** A5: Run a check-then-append of fence records alone on its journal. Not reentrant: never nest two. */
|
|
28
|
+
export function recordOnce(journal, operation) {
|
|
29
|
+
const result = (sections.get(journal) ?? Promise.resolve()).then(operation);
|
|
30
|
+
const tail = result.catch(() => { });
|
|
31
|
+
sections.set(journal, tail);
|
|
32
|
+
void tail.then(() => { if (sections.get(journal) === tail)
|
|
33
|
+
sections.delete(journal); });
|
|
34
|
+
return result;
|
|
35
|
+
}
|
|
24
36
|
/** F1: Record a fence timeout of an execution or gate identity once, with one unknown attention item for the origin. */
|
|
25
|
-
export
|
|
37
|
+
export function recordFenceFailure(journal, id, call, error) {
|
|
38
|
+
return recordOnce(journal, () => fenceFailure(journal, id, call, error));
|
|
39
|
+
}
|
|
40
|
+
async function fenceFailure(journal, id, call, error) {
|
|
26
41
|
if (!journal.entries().some(e => e.type === "fence-failed" && e.exec === id))
|
|
27
42
|
await journal.append("fence-failed", { exec: id, error: String(error) });
|
|
28
43
|
const item = `fence:${id}`;
|
|
@@ -34,7 +49,11 @@ export async function recordFenceFailure(journal, id, call, error) {
|
|
|
34
49
|
await journal.append(JT.attention, { item: { id: item, rev: 1, kind: "unknown", text, wid: call.slice(0, call.lastIndexOf("@", call.indexOf("/"))), call } });
|
|
35
50
|
}
|
|
36
51
|
/** F1: Resolve the fence attention item of an identity that a later fence proved retired. */
|
|
37
|
-
export
|
|
52
|
+
export function resolveFenceAttention(journal, id) {
|
|
53
|
+
return recordOnce(journal, () => fenceAttentionResolved(journal, id));
|
|
54
|
+
}
|
|
55
|
+
/** The same, inside a `recordOnce` section. */
|
|
56
|
+
export async function fenceAttentionResolved(journal, id) {
|
|
38
57
|
const item = journal.entries().find(e => e.type === JT.attention && e.item.id === `fence:${id}`)?.item;
|
|
39
58
|
if (item && !journal.entries().some(e => e.type === JT.attentionResolved && e.id === `fence:${id}` && e.rev === item.rev))
|
|
40
59
|
await journal.append(JT.attentionResolved, { id: `fence:${id}`, rev: item.rev, resolution: "fenced" });
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/** Apply one orchestrator ledger entry to the map of used-up providers. */
|
|
2
|
+
export function foldExhaustion(exhausted, e) {
|
|
3
|
+
const provider = String(e.provider ?? e.pool ?? "");
|
|
4
|
+
if (e.type === "provider-exhausted") {
|
|
5
|
+
// A refusal of the probe ends it; any other execution's (the executor records none while a probe runs) keeps it.
|
|
6
|
+
const probe = exhausted.get(provider)?.probe;
|
|
7
|
+
exhausted.set(provider, { since: Number(e.since), nextTry: Number(e.nextTry), error: String(e.error ?? ""), ...(probe && probe !== e.exec ? { probe } : {}) });
|
|
8
|
+
}
|
|
9
|
+
else if (e.type === "provider-available")
|
|
10
|
+
exhausted.delete(provider);
|
|
11
|
+
else if (e.type === "provider-probe") {
|
|
12
|
+
const x = exhausted.get(provider);
|
|
13
|
+
if (x)
|
|
14
|
+
x.probe = String(e.exec);
|
|
15
|
+
}
|
|
16
|
+
else if (e.type === "release") {
|
|
17
|
+
const x = exhausted.get(provider);
|
|
18
|
+
if (x && x.probe === e.exec)
|
|
19
|
+
delete x.probe;
|
|
20
|
+
}
|
|
21
|
+
}
|
|
@@ -6,6 +6,7 @@ import { compileFanout } from "../compat/fanout.js";
|
|
|
6
6
|
import { readJournalSnapshot } from "../kernel/journal.js";
|
|
7
7
|
import { journalPath, orchLedger, pinnedDir, workflowDir } from "../paths.js";
|
|
8
8
|
import { JT } from "../types.js";
|
|
9
|
+
import { foldExhaustion } from "./providers.js";
|
|
9
10
|
/** A workflow has live work: it runs, or follow-ups opened on it after it finished have not ended yet. */
|
|
10
11
|
export function isLive(wf) {
|
|
11
12
|
return wf.status === "running" || (wf.followUps ?? 0) > 0;
|
|
@@ -469,7 +470,7 @@ const latestCalls = (wf) => [...new Map(wf.calls.map(c => [c.key, c])).values()]
|
|
|
469
470
|
/** Provider slots from the orchestrator ledger: holders per provider (hold/release{pool,slot,exec}) and the limits of
|
|
470
471
|
* the settings in effect (the latest config{hash,config}); config-rejected after it is reported too. */
|
|
471
472
|
export function slotsView(home, now = Date.now()) {
|
|
472
|
-
const held = new Map();
|
|
473
|
+
const held = new Map(), used = new Map();
|
|
473
474
|
let config, rejected;
|
|
474
475
|
for (const e of readJournalSnapshot(orchLedger(home))) {
|
|
475
476
|
if (e.type === "hold")
|
|
@@ -482,7 +483,10 @@ export function slotsView(home, now = Date.now()) {
|
|
|
482
483
|
}
|
|
483
484
|
else if (e.type === "config-rejected")
|
|
484
485
|
rejected = e;
|
|
486
|
+
foldExhaustion(used, e);
|
|
485
487
|
}
|
|
488
|
+
const exhausted = [...used].sort(([a], [b]) => a.localeCompare(b)).map(([p, x]) => `${p} exhausted since ${age(now - x.since)} ago (${clip(x.error, 80)}), ` +
|
|
489
|
+
(x.probe ? `probing with ${x.probe.split("#")[0]}` : x.nextTry > now ? `next try in ${age(x.nextTry - now)}` : "next call probes it"));
|
|
486
490
|
const limits = (config?.config?.providers) ?? {};
|
|
487
491
|
const holders = new Map();
|
|
488
492
|
for (const e of held.values())
|
|
@@ -491,6 +495,7 @@ export function slotsView(home, now = Date.now()) {
|
|
|
491
495
|
const names = [...new Set([...Object.keys(limits), ...holders.keys()])].sort();
|
|
492
496
|
const slots = names.map(p => { const n = holders.get(p) ?? 0, limit = limits[p]?.slots; return typeof limit === "number" ? `${p} ${n}/${limit}` : `${p} ${n} (no limit)`; });
|
|
493
497
|
return { ...(slots.length ? { slots } : {}), ...(config ? { config: `${String(config.hash)} since ${age(now - config.ts)} ago` } : {}),
|
|
498
|
+
...(exhausted.length ? { exhausted } : {}),
|
|
494
499
|
...(rejected ? { configRejected: `${clip(String(rejected.error), 200)} (${age(now - rejected.ts)} ago); ${config ? String(config.hash) : "the start settings"} stay in effect` } : {}) };
|
|
495
500
|
}
|
|
496
501
|
/** Tool status without a wid: what runs, what waits for an answer and what failed, with finished workflows one line each.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-durable-subagents",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.11",
|
|
4
4
|
"description": "Subagents for pi that never lose work and never do it twice. Crash-safe workflows, automatic recovery, and a live view just like the main agent.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|