@avantf/dsh-mission 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +249 -0
- package/cordis.patch.yml +13 -0
- package/lib/client.js +225 -0
- package/lib/dsh-build.json +16 -0
- package/lib/envinit-bootstrap.js +334 -0
- package/lib/index.js +7021 -0
- package/lib/interface-version.json +4 -0
- package/lib/mission-core/capacity.d.ts +100 -0
- package/lib/mission-core/continuation.d.ts +106 -0
- package/lib/mission-core/dispatch.d.ts +165 -0
- package/lib/mission-core/engine.d.ts +331 -0
- package/lib/mission-core/index.d.ts +25 -0
- package/lib/mission-core/liveness.d.ts +99 -0
- package/lib/mission-core/prompt.d.ts +112 -0
- package/lib/mission-core/resources.d.ts +57 -0
- package/lib/mission-core/timing.d.ts +37 -0
- package/lib/mission-core/tree.d.ts +447 -0
- package/lib/mission-core/trouble.d.ts +40 -0
- package/lib/mission-core/types.d.ts +409 -0
- package/lib/mission-core/wellformed.d.ts +98 -0
- package/lib/types/claims.d.ts +3 -0
- package/lib/types/claims.d.ts.map +1 -0
- package/lib/types/client/MissionTreeView.d.ts +247 -0
- package/lib/types/client/MissionTreeView.d.ts.map +1 -0
- package/lib/types/client/api.d.ts +184 -0
- package/lib/types/client/api.d.ts.map +1 -0
- package/lib/types/client/contract.d.ts +209 -0
- package/lib/types/client/contract.d.ts.map +1 -0
- package/lib/types/client/index.d.ts +48 -0
- package/lib/types/client/index.d.ts.map +1 -0
- package/lib/types/client/seat.d.ts +26 -0
- package/lib/types/client/seat.d.ts.map +1 -0
- package/lib/types/client/styles.d.ts +6 -0
- package/lib/types/client/styles.d.ts.map +1 -0
- package/lib/types/coldResume.d.ts +123 -0
- package/lib/types/coldResume.d.ts.map +1 -0
- package/lib/types/domain.d.ts +93 -0
- package/lib/types/domain.d.ts.map +1 -0
- package/lib/types/envinit.d.ts +126 -0
- package/lib/types/envinit.d.ts.map +1 -0
- package/lib/types/executorSession.d.ts +129 -0
- package/lib/types/executorSession.d.ts.map +1 -0
- package/lib/types/faces.d.ts +30 -0
- package/lib/types/faces.d.ts.map +1 -0
- package/lib/types/host.d.ts +752 -0
- package/lib/types/host.d.ts.map +1 -0
- package/lib/types/index.d.ts +48 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/interface_gate.d.ts +51 -0
- package/lib/types/interface_gate.d.ts.map +1 -0
- package/lib/types/log.d.ts +18 -0
- package/lib/types/log.d.ts.map +1 -0
- package/lib/types/projectionCache.d.ts +22 -0
- package/lib/types/projectionCache.d.ts.map +1 -0
- package/lib/types/prompt.d.ts +106 -0
- package/lib/types/prompt.d.ts.map +1 -0
- package/lib/types/source.d.ts +22 -0
- package/lib/types/source.d.ts.map +1 -0
- package/lib/types/store.d.ts +20 -0
- package/lib/types/store.d.ts.map +1 -0
- package/lib/types/timeFormat.d.ts +56 -0
- package/lib/types/timeFormat.d.ts.map +1 -0
- package/lib/types/tools.d.ts +34 -0
- package/lib/types/tools.d.ts.map +1 -0
- package/lib/types/wellformed.d.ts +43 -0
- package/lib/types/wellformed.d.ts.map +1 -0
- package/lib/types/wire.d.ts +178 -0
- package/lib/types/wire.d.ts.map +1 -0
- package/lib/types/workerEvents.d.ts +36 -0
- package/lib/types/workerEvents.d.ts.map +1 -0
- package/lib/types/workerSessions.d.ts +256 -0
- package/lib/types/workerSessions.d.ts.map +1 -0
- package/package.json +145 -0
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The DERIVED half of a node's three timestamps: how long it queued, how long it ran, and how long
|
|
3
|
+
* it existed. The three canonical instants live on {@link NodeRecord} (`createdAt` / `dispatchedAt`
|
|
4
|
+
* / `endedAt`); these helpers are the only arithmetic over them, so a display, a tool answer and a
|
|
5
|
+
* test cannot each round or clamp a duration their own way.
|
|
6
|
+
*
|
|
7
|
+
* Nothing here reads a clock: every quantity is a function of the record alone. A live "how long has
|
|
8
|
+
* this been waiting" is deliberately NOT offered — a duration that grows between renders belongs to
|
|
9
|
+
* the renderer, and inventing one here would make a persisted value look like a clock reading.
|
|
10
|
+
*
|
|
11
|
+
* @module @avantf/mission-core/timing
|
|
12
|
+
*/
|
|
13
|
+
/** The three instants, in the minimal shape every reader of them shares. */
|
|
14
|
+
export interface NodeTiming {
|
|
15
|
+
readonly createdAt: number;
|
|
16
|
+
readonly dispatchedAt: number | null;
|
|
17
|
+
readonly endedAt: number | null;
|
|
18
|
+
}
|
|
19
|
+
/**
|
|
20
|
+
* How long this node QUEUED before its first dispatch: `dispatchedAt − createdAt`. `null` when it
|
|
21
|
+
* has never been dispatched — the honest answer, since the wait is still going and its end is not
|
|
22
|
+
* recorded. Never negative: a clock that went backwards is clamped to 0 rather than shown as a
|
|
23
|
+
* time-travelling queue.
|
|
24
|
+
*/
|
|
25
|
+
export declare function queueMs(node: NodeTiming): number | null;
|
|
26
|
+
/**
|
|
27
|
+
* How long this node's executor ran: `endedAt − dispatchedAt`. `null` while it is still running or
|
|
28
|
+
* queued, and also for the one terminal case that has no execution to measure — a node cancelled
|
|
29
|
+
* before it ever ran (`dispatchedAt === null`). Clamped to 0 for the same reason as {@link queueMs}.
|
|
30
|
+
*/
|
|
31
|
+
export declare function runMs(node: NodeTiming): number | null;
|
|
32
|
+
/**
|
|
33
|
+
* How long this node has existed so far: `endedAt − createdAt` once terminal, and `null` while it is
|
|
34
|
+
* still in play. Unlike {@link queueMs} and {@link runMs} this is defined even for a node that never
|
|
35
|
+
* ran — a cancelled prerequisite still spent wall-clock time in the tree.
|
|
36
|
+
*/
|
|
37
|
+
export declare function totalMs(node: NodeTiming): number | null;
|
|
@@ -0,0 +1,447 @@
|
|
|
1
|
+
import { type ContinuationDelta } from './continuation.js';
|
|
2
|
+
import { type CapacityPolicy, type DispatchPlan } from './dispatch.js';
|
|
3
|
+
import { type WellFormedSource } from './wellformed.js';
|
|
4
|
+
import { type ChildSpec, type DecomposeOutcome, type DispatchView, type MutationResult, type NodeRecord, type TreeRecord } from './types.js';
|
|
5
|
+
export interface TreeState {
|
|
6
|
+
tree: TreeRecord;
|
|
7
|
+
nodes: Map<string, NodeRecord>;
|
|
8
|
+
}
|
|
9
|
+
/** Durable sink for tree records; the store owns batching and durability. */
|
|
10
|
+
export interface TreeStore {
|
|
11
|
+
/** Every persisted tree document, for startup reconciliation. */
|
|
12
|
+
loadAll(): Promise<TreeState[]>;
|
|
13
|
+
/** Persist one tree document. Resolves once the write is durable. */
|
|
14
|
+
put(state: TreeState): Promise<void>;
|
|
15
|
+
remove(rootId: string): Promise<void>;
|
|
16
|
+
}
|
|
17
|
+
/** One persisted spill artifact: where it is, and how to read it back. */
|
|
18
|
+
export interface SpilledText {
|
|
19
|
+
/** Opaque model-facing locator produced by the storage backend. */
|
|
20
|
+
readonly locator: string;
|
|
21
|
+
/** The backend's retrieval guidance, shown next to the locator. */
|
|
22
|
+
readonly hint: string;
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* What asking durable storage about an owner session resolved to. Three states, not a boolean:
|
|
26
|
+
* "cannot tell" and "gone" demand different actions — a gone owner's tree is destroyed, an
|
|
27
|
+
* opaque host must never be allowed to destroy mission (a stray tree is recoverable, destroyed mission
|
|
28
|
+
* is not), yet the operator still has to hear about it.
|
|
29
|
+
*/
|
|
30
|
+
export type OwnerProbe = {
|
|
31
|
+
readonly kind: 'exists';
|
|
32
|
+
} | {
|
|
33
|
+
readonly kind: 'missing';
|
|
34
|
+
} | {
|
|
35
|
+
readonly kind: 'unobservable';
|
|
36
|
+
readonly detail: string;
|
|
37
|
+
};
|
|
38
|
+
/** One tree whose owner did not resolve to `exists`, with the probe that explains why. */
|
|
39
|
+
export interface OrphanedTree {
|
|
40
|
+
readonly tree: TreeRecord;
|
|
41
|
+
readonly probe: OwnerProbe;
|
|
42
|
+
}
|
|
43
|
+
/** Injected environment facts; the core never imports the harness. */
|
|
44
|
+
export interface TreeDeps {
|
|
45
|
+
isAgentLive(sessionId: string): boolean;
|
|
46
|
+
/** Whether a session still EXISTS, asked of durable storage rather than the live registry: an
|
|
47
|
+
* agent is materialized on demand, so right after a restart a live check is false. Ownership is
|
|
48
|
+
* decided by this one. `unobservable` is a THIRD answer, not a synonym for either. */
|
|
49
|
+
probeOwner(sessionId: string): Promise<OwnerProbe>;
|
|
50
|
+
/** Persist an over-long result, or return `null` when no spill backend is configured — the node
|
|
51
|
+
* then keeps the full text inline, because a locator nobody can resolve loses the tail. */
|
|
52
|
+
spill(text: string): Promise<SpilledText | null>;
|
|
53
|
+
/** Clock, injectable for tests. */
|
|
54
|
+
now(): number;
|
|
55
|
+
/** Id generator, injectable for tests. */
|
|
56
|
+
newId(): string;
|
|
57
|
+
/**
|
|
58
|
+
* The family's well-formed repair, injected by the plugin from the LOADED BASE KIT (interface v2,
|
|
59
|
+
* `wellFormedText` / `wellFormedDeep`). Absent means the base could not provide it — base missing,
|
|
60
|
+
* or older than v2 — and the tree uses its own {@link LOCAL_WELL_FORMED} copy instead. This is a
|
|
61
|
+
* degradation, never a refusal: a tree with no injected source still mounts and still repairs.
|
|
62
|
+
*/
|
|
63
|
+
readonly wellFormed?: WellFormedSource;
|
|
64
|
+
}
|
|
65
|
+
export interface CreateRootInput {
|
|
66
|
+
readonly ownerSessionId: string;
|
|
67
|
+
readonly title: string;
|
|
68
|
+
readonly description: string;
|
|
69
|
+
/** The owner's initial analysis; stored as the root's background facts. */
|
|
70
|
+
readonly analysis: readonly string[];
|
|
71
|
+
/** The scope this mission will modify (a directory or file), or `undefined`/blank for none. The
|
|
72
|
+
* root has no parent to inherit from, so this is the only declaration site for a tree's unit. */
|
|
73
|
+
readonly unit?: string | null;
|
|
74
|
+
/** This mission's declared capacity weight (cores-equivalent); `undefined` reads as the default 1.
|
|
75
|
+
* The root has no parent to inherit from. */
|
|
76
|
+
readonly weight?: number;
|
|
77
|
+
/** This mission's declared round-cap relaxation (see {@link NodeRecord.roundMs}); the root has no
|
|
78
|
+
* parent, so this is the only declaration site for a tree's relaxed ceiling. */
|
|
79
|
+
readonly roundMs?: number | null;
|
|
80
|
+
}
|
|
81
|
+
/** Random 8-hex-char id, matching the short ids the tools expose. `crypto` is reached through a
|
|
82
|
+
* typed globals lookup because the core is built without DOM or Node ambient types. */
|
|
83
|
+
export declare function defaultNewId(): string;
|
|
84
|
+
export declare class MissionTree {
|
|
85
|
+
private readonly deps;
|
|
86
|
+
private readonly states;
|
|
87
|
+
private chain;
|
|
88
|
+
private store;
|
|
89
|
+
/** Trees whose in-memory progress moved since the last durable flush. */
|
|
90
|
+
private readonly progressDirty;
|
|
91
|
+
constructor(store: TreeStore | undefined, deps: TreeDeps);
|
|
92
|
+
/**
|
|
93
|
+
* The repair pair every INBOUND write uses: the base kit the plugin injected, or the local
|
|
94
|
+
* degradation copy. ONE accessor, so "which implementation" is decided in one place and every
|
|
95
|
+
* write-path entry below asks the same question.
|
|
96
|
+
*/
|
|
97
|
+
private get wellFormed();
|
|
98
|
+
private requireStore;
|
|
99
|
+
/** Load every persisted tree and reconcile bindings against live agents: a `running` node whose
|
|
100
|
+
* `claimedBy` no longer resolves is reclaimed as `interrupted`, restart and hot reload alike. */
|
|
101
|
+
open(): Promise<void>;
|
|
102
|
+
/**
|
|
103
|
+
* Three outcomes on open, decided by what the durable record says and what this process can see:
|
|
104
|
+
*
|
|
105
|
+
* 1. SURVIVOR — `running` and the worker IS materialized (hot reload): keep the binding, only
|
|
106
|
+
* reset `progressAt`, or a long run looks silent from the moment we open.
|
|
107
|
+
* 2. CONTINUABLE — `running` but the worker is not materialized in THIS process (a restart or a
|
|
108
|
+
* crash): demote to `interrupted`, but FIRST park `claimedBy` in `lastWorkerId`. The binding is
|
|
109
|
+
* the only place that session id was recorded, so dropping it here is what made every restart
|
|
110
|
+
* unable to do anything but start over.
|
|
111
|
+
* 3. NO HANDLE — not `running`: nothing to do, and `lastWorkerId` (if any) is already spent.
|
|
112
|
+
*
|
|
113
|
+
* The survivor branch is deliberately unchanged: a live binding is authoritative, and a stale
|
|
114
|
+
* handle left beside it is only ever consulted by `adoptContinuation`, which requires the node to
|
|
115
|
+
* be dispatchable — a `running` node never is.
|
|
116
|
+
*/
|
|
117
|
+
private reconcileOnOpen;
|
|
118
|
+
/** Trees whose owner session did not resolve to `exists`, each with the probe that says why.
|
|
119
|
+
* Asked of durable storage rather than the live registry, so a restart keeps its trees and only a
|
|
120
|
+
* genuinely deleted session loses them. `missing` and `unobservable` are reported separately: the
|
|
121
|
+
* caller decides what may be destroyed (see `MissionEngine.reconcileOrphans`). */
|
|
122
|
+
orphanedTrees(): Promise<readonly OrphanedTree[]>;
|
|
123
|
+
destroyTree(rootId: string): Promise<void>;
|
|
124
|
+
treeOf(rootId: string): TreeRecord | undefined;
|
|
125
|
+
node(id: string): NodeRecord | undefined;
|
|
126
|
+
/** Claim the right to report one tree's terminal state: true exactly once per transition,
|
|
127
|
+
* durably, so an engine restart cannot produce a second wake. */
|
|
128
|
+
claimReport(rootId: string): Promise<boolean>;
|
|
129
|
+
/** Allow a later terminal transition of the same tree to be reported again. */
|
|
130
|
+
clearReport(rootId: string): Promise<void>;
|
|
131
|
+
trees(): readonly TreeRecord[];
|
|
132
|
+
nodesOf(rootId: string): readonly NodeRecord[];
|
|
133
|
+
/** Ancestors of a node, root → parent. */
|
|
134
|
+
chainOf(node: NodeRecord): readonly NodeRecord[];
|
|
135
|
+
/** The current dispatch view for one node. Built at spawn time, not dispatch time, so the prompt
|
|
136
|
+
* reflects children results that landed between the dispatch decision and the spawn. */
|
|
137
|
+
view(nodeId: string): DispatchView | undefined;
|
|
138
|
+
/** Terminal children of an aggregate, in decomposition order. */
|
|
139
|
+
terminalChildren(node: NodeRecord): readonly NodeRecord[];
|
|
140
|
+
/** What changed on this node since the prompt its bound session was handed (see
|
|
141
|
+
* `@avantf/mission-core/continuation`). Read-only and synchronous, because the wake path consults
|
|
142
|
+
* it before it adopts anything: a refusal that has already consumed an address is a refusal that
|
|
143
|
+
* cannot be undone.
|
|
144
|
+
*
|
|
145
|
+
* `undefined` when the node does not exist; a node that merely has no baseline still answers, with
|
|
146
|
+
* `baselineKnown: false` — "unknown" is a fact the caller renders, not an error. */
|
|
147
|
+
continuationDelta(nodeId: string): ContinuationDelta | undefined;
|
|
148
|
+
/** The next node to dispatch: `ready`/`interrupted`, oldest first across trees, excluding ones
|
|
149
|
+
* whose binding still resolves to a live agent. A vanished worker's node is reclaimed by the
|
|
150
|
+
* engine's sweep before it becomes a candidate again.
|
|
151
|
+
*
|
|
152
|
+
* The selection itself — ordering, parked/backoff rules and the unit-lease skip — lives in
|
|
153
|
+
* `@avantf/mission-core/dispatch`; this is the state access, and the lease side of it is
|
|
154
|
+
* re-checked atomically at every transition into `running` below, so a race that slips past this
|
|
155
|
+
* filter is refused rather than run. The optional `capacity` policy arms the capacity gate; without
|
|
156
|
+
* it this is exactly the pre-capacity contract. */
|
|
157
|
+
nextDispatchable(exclude?: ReadonlySet<string>, capacity?: CapacityPolicy): NodeRecord | undefined;
|
|
158
|
+
/** The full admission plan (see `@avantf/mission-core/dispatch`): the node to dispatch AND every
|
|
159
|
+
* candidate left waiting with its reason. The engine uses the latter half for `waitingFor` and for
|
|
160
|
+
* the aging bookkeeping; tree-level callers that only need the answer use `nextDispatchable`. */
|
|
161
|
+
planDispatch(exclude?: ReadonlySet<string>, capacity?: CapacityPolicy): DispatchPlan;
|
|
162
|
+
/**
|
|
163
|
+
* What the capacity gate currently has in flight: a node counts once it is BOUND (`running` with a
|
|
164
|
+
* `claimedBy`), not once its worker materializes, so a pass cannot over-subscribe the pool by
|
|
165
|
+
* dispatching into a lazily-created child.
|
|
166
|
+
*/
|
|
167
|
+
runningLoad(): {
|
|
168
|
+
count: number;
|
|
169
|
+
weight: number;
|
|
170
|
+
};
|
|
171
|
+
/**
|
|
172
|
+
* The lease check shared by all three transitions INTO `running` (`dispatch`, `adoptParked`,
|
|
173
|
+
* `adoptContinuation`). It has to be re-made here, under the tree lock, and not only in
|
|
174
|
+
* `nextDispatchable`: the wake paths bind from a snapshot taken outside it, and two passes can
|
|
175
|
+
* interleave between "this unit looked free" and "mark it running".
|
|
176
|
+
*
|
|
177
|
+
* `undefined` means the unit is free (or the node declared none — the no-op case that keeps
|
|
178
|
+
* undeclared records byte-for-byte as they behaved before leases existed). Reached through
|
|
179
|
+
* {@link admissionRefusal}, which pairs it with the capacity recheck; the capacity gate has the
|
|
180
|
+
* same snapshot problem and is re-made in the same breath.
|
|
181
|
+
*/
|
|
182
|
+
private unitRefusal;
|
|
183
|
+
/**
|
|
184
|
+
* The capacity counterpart of {@link unitRefusal}, re-made under the tree lock at every transition
|
|
185
|
+
* INTO `running`. `planDispatch` decides from a snapshot taken OUTSIDE the lock, so two passes can
|
|
186
|
+
* each see an empty machine and both bind; the arithmetic therefore has to be re-run here, in the
|
|
187
|
+
* same place the unit lease is, and through the SAME judgement the plan uses
|
|
188
|
+
* ({@link capacityWaitingFor}) so the two can never drift.
|
|
189
|
+
*
|
|
190
|
+
* The policy supplies the machine's capacity, the slot ceiling and (when the host set it) the
|
|
191
|
+
* machine-wide block. Its `runningCount`/`runningWeight` are the CALLER'S SNAPSHOT and are
|
|
192
|
+
* deliberately ignored: trusting them is the very TOCTOU this closes. `undefined` policy means the
|
|
193
|
+
* caller opted out of the gate entirely — the pre-capacity contract, unchanged.
|
|
194
|
+
*
|
|
195
|
+
* The refusal is a DEFERRAL in the strictest sense (`capacity-busy`): the node stays `ready` and
|
|
196
|
+
* NOTHING is charged — no `attempts`, `failures`, `spawnFailures`, no cooldown, no stall. See
|
|
197
|
+
* {@link RefusalCode}.
|
|
198
|
+
*/
|
|
199
|
+
private capacityRefusal;
|
|
200
|
+
/**
|
|
201
|
+
* The two RESOURCE admissions every transition into `running` must re-make under the tree lock:
|
|
202
|
+
* the unit lease first (a scope conflict), then capacity (the machine). Kept in ONE call site per
|
|
203
|
+
* transition so "what has to be rechecked before binding" is named once and the two checks cannot
|
|
204
|
+
* drift or be reordered by accident. Budgets are deliberately NOT here: a spent budget is a
|
|
205
|
+
* FAILURE (`failExhausted`), while both of these are "not yet" and must charge nothing.
|
|
206
|
+
*/
|
|
207
|
+
private admissionRefusal;
|
|
208
|
+
/** Nodes waiting for their parked session to be woken: `ready` with a recorded address, i.e. all
|
|
209
|
+
* children terminal and a session to continue in. The owner's wake gate admits on this. */
|
|
210
|
+
parkedReadyNodes(rootId?: string): readonly NodeRecord[];
|
|
211
|
+
/** Claim a parked node FOR its parked session so it can be woken rather than replaced. Unlike
|
|
212
|
+
* `dispatch` it reserves no new claim id — the identity is the parked session — and consumes
|
|
213
|
+
* `parkedWorker` in the same locked step, so a concurrent pass cannot both adopt and re-dispatch
|
|
214
|
+
* it. The claim must land BEFORE the wake is delivered, because a woken session may submit or
|
|
215
|
+
* decompose immediately and both authorize on `claimedBy`; a failed delivery is undone by a
|
|
216
|
+
* `reclaim(..., 'wake-failed')`. */
|
|
217
|
+
adoptParked(nodeId: string, workerId: string, capacity?: CapacityPolicy): Promise<MutationResult<DispatchView>>;
|
|
218
|
+
/**
|
|
219
|
+
* Claim a node FOR the lost session recorded in `lastWorkerId`, so its next execution CONTINUES
|
|
220
|
+
* that session (a cold wake) rather than starting a fresh executor. The continuation counterpart
|
|
221
|
+
* of {@link adoptParked}, and the same shape: no new claim id is reserved — the identity IS the
|
|
222
|
+
* recorded session — and the handle is consumed in the same locked step.
|
|
223
|
+
*
|
|
224
|
+
* Why the ceilings ARE consulted here, unlike for a parked adoption: a continuation IS a
|
|
225
|
+
* dispatch. It hands work to a model and advances `attempts`, so a node whose failure budget is
|
|
226
|
+
* spent must fail here exactly as `dispatch` would fail it, instead of being resurrected.
|
|
227
|
+
*
|
|
228
|
+
* The handle is CONSUMED (`lastWorkerId = null`) rather than kept: this dispatch either continues
|
|
229
|
+
* that session or — through `reclaim(..., 'wake-failed')` — falls back to a fresh one, and neither
|
|
230
|
+
* outcome may try the same address again. A runtime that refused the resume once will refuse it
|
|
231
|
+
* again; the fallback is the complete answer (`note_mission` is the cross-session hand-off).
|
|
232
|
+
*/
|
|
233
|
+
adoptContinuation(nodeId: string, workerId: string, capacity?: CapacityPolicy): Promise<MutationResult<DispatchView>>;
|
|
234
|
+
/**
|
|
235
|
+
* Stamp the node with what the prompt just handed to the session showed it: the snapshot a later
|
|
236
|
+
* cold wake subtracts. The HOST owns the moment (it calls this once the delivery resolved, at each
|
|
237
|
+
* of its three prompt sites), because only the host knows a prompt was actually read — a dispatch
|
|
238
|
+
* that never produced, or never delivered, a prompt must leave no baseline claiming otherwise.
|
|
239
|
+
*
|
|
240
|
+
* GUARDED on the live binding: between the delivery and this call, a sweep can reclaim the node
|
|
241
|
+
* (or another pass can re-dispatch it), and a baseline that says "this session saw this" must
|
|
242
|
+
* belong to the dispatch that is actually bound. A stamp that loses that race is refused
|
|
243
|
+
* silently: the wake that later reads a stale-or-missing baseline takes the conservative path,
|
|
244
|
+
* while a wrongly stamped one would under-report the drift to a live session.
|
|
245
|
+
*
|
|
246
|
+
* Tolerant rather than a mutation result, like {@link markCorrectionsDelivered}: it is bookkeeping
|
|
247
|
+
* about a prompt that already exists, and the caller can do nothing useful with a refusal. The
|
|
248
|
+
* return value exists so the caller (and tests) can tell a stamp from a lost race.
|
|
249
|
+
*/
|
|
250
|
+
recordDispatchBaseline(nodeId: string, holder: string): Promise<boolean>;
|
|
251
|
+
/**
|
|
252
|
+
* Spend a continuation handle WITHOUT using it, so the node's next dispatch starts fresh. Used
|
|
253
|
+
* when the drift since that session's prompt is material (`isMaterialChange`): the address is not
|
|
254
|
+
* worth spending, and leaving it would make every later pass re-decide the same thing.
|
|
255
|
+
*
|
|
256
|
+
* No status change, no budget, no cooldown — the caller's ordinary `dispatch` follows in the same
|
|
257
|
+
* engine pass. The baseline is left alone: the fresh dispatch stamps its own, which is the whole
|
|
258
|
+
* point of replacing the session.
|
|
259
|
+
*/
|
|
260
|
+
abandonContinuation(nodeId: string, workerId: string): Promise<MutationResult<NodeRecord>>;
|
|
261
|
+
/** Roll up every OPEN tree into the counts the guidance layer renders; closed trees are archived
|
|
262
|
+
* and no longer echo into the owner's prompt. */
|
|
263
|
+
summary(): {
|
|
264
|
+
trees: number;
|
|
265
|
+
ready: number;
|
|
266
|
+
running: number;
|
|
267
|
+
blocked: number;
|
|
268
|
+
done: number;
|
|
269
|
+
failed: number;
|
|
270
|
+
};
|
|
271
|
+
createRoot(input: CreateRootInput): Promise<MutationResult<NodeRecord>>;
|
|
272
|
+
/** Allocate an id not present in the tree (and not already reserved this call). */
|
|
273
|
+
private allocateId;
|
|
274
|
+
/** One fresh node record; the only place node defaults are written. */
|
|
275
|
+
private makeNode;
|
|
276
|
+
/** Mark one node dispatched and bind it to a reserved child session id, in the same locked step
|
|
277
|
+
* as the dispatch decision, so two engine passes cannot both dispatch it. `attempts` increments
|
|
278
|
+
* on every dispatch (the aggregate pass counts as one) and is the generation marker the
|
|
279
|
+
* `note_mission` gate reads; the failed ceiling rides `failures` instead, so a successful aggregate
|
|
280
|
+
* round is never charged against it. */
|
|
281
|
+
dispatch(nodeId: string, claimId: string, capacity?: CapacityPolicy): Promise<MutationResult<DispatchView>>;
|
|
282
|
+
/** Fail a node whose budget ran out and refuse the dispatch, so the owner is told (the engine
|
|
283
|
+
* reports terminal roots) and the node never re-enters the candidate pool. */
|
|
284
|
+
private failExhausted;
|
|
285
|
+
/** Return a dispatched node to the pool; `cause` decides which budget the reclaim charges.
|
|
286
|
+
* `vanished`/`stalled`: a worker ran (or was starting) without a result — charges `failures`,
|
|
287
|
+
* and only `stalled` also increments `stalls`. `spawn-failed`: no worker ever started — charges
|
|
288
|
+
* `spawnFailures` and pushes the next dispatch back by a cooldown. `wake-failed`: an adoption
|
|
289
|
+
* that could not be delivered — back to `ready`, charging neither budget, so the caller's fresh
|
|
290
|
+
* dispatch proceeds. `hung`: the worker was still alive but produced nothing for a whole window
|
|
291
|
+
* (or exceeded the round cap) — like `wake-failed` it charges NEITHER budget and adds no cooldown,
|
|
292
|
+
* because an apparatus outage is not a failed mission and must not spend the budget that ends the
|
|
293
|
+
* node; unlike `wake-failed` it lands in `interrupted`, the ordinary re-queue, and it advances the
|
|
294
|
+
* `hungCount` STREAK (cleared by real output, never by a re-dispatch) so a node that hangs every
|
|
295
|
+
* round eventually reaches the owner instead of re-running forever. `attempts` is never
|
|
296
|
+
* rolled back: it is the `note_mission` generation marker, not a budget.
|
|
297
|
+
*
|
|
298
|
+
* Every arm leaves `running`, so this is one of the paths that RELEASES the node's unit lease
|
|
299
|
+
* (the lease is the set of `running` nodes' units, never a separate table — see `dispatch.ts`).
|
|
300
|
+
*
|
|
301
|
+
* `expectedHolder` is a compare-and-swap for a caller that judged the node from a snapshot taken
|
|
302
|
+
* OUTSIDE this lock — `MissionEngine.reclaimStale` takes one, then awaits `interruptWorker` per
|
|
303
|
+
* node. If a second sweep reclaimed and re-dispatched the node in that window, the binding has
|
|
304
|
+
* moved on and this stale verdict must be refused: acting on it would unbind a LIVE worker (whose
|
|
305
|
+
* `submit_mission` then answers `not-owner`) and charge `failures` a second time for one attempt.
|
|
306
|
+
* `undefined` means "no expectation" — the callers that act on a value they just read use that. */
|
|
307
|
+
reclaim(nodeId: string, cause?: 'vanished' | 'stalled' | 'hung' | 'spawn-failed' | 'wake-failed', expectedHolder?: string | null): Promise<MutationResult<NodeRecord>>;
|
|
308
|
+
/** Clear the spawn-failure counter after a worker started successfully: the budget is for
|
|
309
|
+
* CONSECUTIVE start failures. Memory-only between flushes is fine — a dispatch flush follows. */
|
|
310
|
+
noteSpawnSuccess(nodeId: string): void;
|
|
311
|
+
/** Record that one worker PRODUCED something — model output, a tool call, a tool result — on a
|
|
312
|
+
* hot path from the worker's durable append feed, so it updates memory only. This is the clock the
|
|
313
|
+
* stale check reads: output is also evidence of life, so one write moves `activityAt` too. A
|
|
314
|
+
* timestamp not newer than what we have is ignored, which makes racing a dispatch safe.
|
|
315
|
+
*
|
|
316
|
+
* It is ALSO the clearing point of the hang streak: production is the one event that proves the
|
|
317
|
+
* round was not merely retrying forever, so `hungCount` goes back to 0 here (and only then — a
|
|
318
|
+
* re-dispatch sets `progressAt` without proving anything and deliberately leaves the streak). */
|
|
319
|
+
touchProgress(nodeId: string, at: number): void;
|
|
320
|
+
/** Record that a worker's session emitted SOME event without claiming it produced anything: the
|
|
321
|
+
* transport-layer half of liveness. It moves only `activityAt`, so a worker whose events are all
|
|
322
|
+
* retries and route snapshots is still heard from (never `stalled`) while `progressAt` stays
|
|
323
|
+
* where its last real output left it — exactly the "alive but unproductive" state that
|
|
324
|
+
* `judgeWorker` reclaims as `hung`. */
|
|
325
|
+
touchActivity(nodeId: string, at: number): void;
|
|
326
|
+
/** Persist progress observed since the last flush, coalesced on purpose: writing the whole tree
|
|
327
|
+
* document per worker event would cost more than the mission it guards. The usual lock keeps a
|
|
328
|
+
* stale document from landing after a newer one. */
|
|
329
|
+
flushProgress(): Promise<void>;
|
|
330
|
+
/** Claim the right to tell the owner about one node's stalls: true exactly once per node,
|
|
331
|
+
* durably, the same contract as `claimReport`. */
|
|
332
|
+
claimStallReport(nodeId: string): Promise<boolean>;
|
|
333
|
+
/**
|
|
334
|
+
* Remember the executor session a LAZY lookup resolved for one node, so the panel never pays for
|
|
335
|
+
* the same lookup twice (W18: the node id is clickable on every node, and a historical record has
|
|
336
|
+
* no `executorSessionId` — the resolution reads session logs, which is exactly the cost that must
|
|
337
|
+
* not happen at render time).
|
|
338
|
+
*
|
|
339
|
+
* Write-once: a handle already on the record is authoritative and the write is skipped. A handle
|
|
340
|
+
* that arrived while the lookup was in flight belongs to a NEWER attempt, and overwriting it with
|
|
341
|
+
* a resolved OLDER one would point the panel at the wrong session — the same "only the last
|
|
342
|
+
* attempt" rule every other writer of this field follows.
|
|
343
|
+
*
|
|
344
|
+
* Tolerant rather than a mutation result, like {@link recordDispatchBaseline} and
|
|
345
|
+
* {@link markCorrectionsDelivered}: it is bookkeeping, and the caller can do nothing useful with a
|
|
346
|
+
* refusal. The `boolean` exists so the caller knows whether to announce a change.
|
|
347
|
+
*/
|
|
348
|
+
rememberExecutor(nodeId: string, sessionId: string): Promise<boolean>;
|
|
349
|
+
nodeHeldBy(sessionId: string): NodeRecord | undefined;
|
|
350
|
+
/** Record the holding executor's own analysis, attributed to the `attempts` value it is running
|
|
351
|
+
* under — that attribution is what `decompose` checks, so notes inherited from an earlier round
|
|
352
|
+
* do not authorize a split. Text is appended (duplicates dropped) and kept for every later
|
|
353
|
+
* dispatch. Refused on a terminal node. */
|
|
354
|
+
recordAnalysis(nodeId: string, callerSessionId: string, analysis: string): Promise<MutationResult<NodeRecord>>;
|
|
355
|
+
/** Split a node into children: a first decomposition, or a re-decomposition of an aggregate whose
|
|
356
|
+
* children are all terminal — the pass that read their conclusions and judged the objective
|
|
357
|
+
* unmet. A node with UNFINISHED children is refused. `submitResult` and `decompose` stay mutually
|
|
358
|
+
* exclusive through node state, never prompt discipline. The analysis is written earlier by
|
|
359
|
+
* `recordAnalysis`; this only checks the splitting dispatch wrote its own.
|
|
360
|
+
*
|
|
361
|
+
* Leaving `running` here is also what RELEASES this node's unit lease, and the children created
|
|
362
|
+
* below inherit that unit unless they declare their own — so the same-scope siblings the split
|
|
363
|
+
* just created are serialized against each other by the same state rule (see `dispatch.ts`). */
|
|
364
|
+
decompose(nodeId: string, callerSessionId: string, children: readonly ChildSpec[]): Promise<MutationResult<DecomposeOutcome>>;
|
|
365
|
+
/** Submit a terminal result; rejected while any child is unfinished. An aggregate (every child
|
|
366
|
+
* terminal) DOES submit — that pass reads the conclusions and states the outcome, which is how a
|
|
367
|
+
* decomposed node reaches `done` and the tree converges. */
|
|
368
|
+
submitResult(nodeId: string, callerSessionId: string, result: string): Promise<MutationResult<{
|
|
369
|
+
node: NodeRecord;
|
|
370
|
+
parentReady: boolean;
|
|
371
|
+
}>>;
|
|
372
|
+
/** Record one correction, on the tree owner's authority only. The text lands in the node's
|
|
373
|
+
* `corrections`, NOT in `context`: `context` is the decomposer's "why this mission exists", while a
|
|
374
|
+
* correction is the owner's instruction about mission already handed out — a different author and a
|
|
375
|
+
* different lifetime. (Rendering them together once made every correction invisible to every
|
|
376
|
+
* descendant, because the chain line only carries `context[0]`.) Either way it is a durable block
|
|
377
|
+
* that every later dispatch of this node renders, so it survives a reclaim, a retry and the
|
|
378
|
+
* aggregate round rather than living in one worker's inbox. */
|
|
379
|
+
correct(nodeId: string, callerSessionId: string, text: string): Promise<MutationResult<NodeRecord>>;
|
|
380
|
+
/**
|
|
381
|
+
* Advance the durable delivery watermark of {@link NodeRecord.correctionsDeliveredUpTo}: the
|
|
382
|
+
* first `upTo` corrections have been confirmed READ by the session that holds this node (either
|
|
383
|
+
* a live steer, or a cold-wake prompt that carried them). What it buys is the restart case — a
|
|
384
|
+
* mark kept only in memory is empty exactly when the wake that needs it happens.
|
|
385
|
+
*
|
|
386
|
+
* Deliberately tolerant rather than a mutation result: this is bookkeeping ABOUT a delivery that
|
|
387
|
+
* already happened, so it is monotone (a raced, older report can never pull the mark back) and
|
|
388
|
+
* clamped to the array's own length (a report that outran a concurrent append cannot make a later
|
|
389
|
+
* wake skip a correction nobody read). A no-op returns without writing.
|
|
390
|
+
*/
|
|
391
|
+
markCorrectionsDelivered(nodeId: string, upTo: number): Promise<void>;
|
|
392
|
+
/** Cancel everything BELOW one node, strictly downward: the node itself, its ancestors and its
|
|
393
|
+
* siblings are untouched, and the cancelled nodes become `failed` so the node's aggregate pass
|
|
394
|
+
* becomes dispatchable again. The tree owner is the caller that matters, because a node with
|
|
395
|
+
* unfinished children is `blocked` and holds no claim. */
|
|
396
|
+
cancelSubworks(nodeId: string, callerSessionId: string, onReclaim?: (claimId: string) => void): Promise<MutationResult<readonly NodeRecord[]>>;
|
|
397
|
+
/** Record that the owner read a terminal result; unlocks `finish_mission`. */
|
|
398
|
+
markResultRead(nodeId: string): Promise<MutationResult<NodeRecord>>;
|
|
399
|
+
/** Close a tree out on the owner's authority; refused while results are unread or the root is not
|
|
400
|
+
* terminal. Either terminal root may be closed — a `failed` root must be reported then retired.
|
|
401
|
+
* Closing is archival: nodes, results and readers are untouched. */
|
|
402
|
+
finish(rootId: string, callerSessionId: string): Promise<MutationResult<NodeRecord>>;
|
|
403
|
+
/** Cancel a whole tree: mark every non-terminal node cancelled-as-failed. */
|
|
404
|
+
cancelTree(rootId: string, callerSessionId: string, onReclaim?: (claimId: string) => void): Promise<MutationResult<readonly NodeRecord[]>>;
|
|
405
|
+
/** Delete one whole tree. The unit is the TREE, not the node: its siblings' premises, the
|
|
406
|
+
* aggregate story and the tree's identity all live in the same record, so pruning one node out of
|
|
407
|
+
* a live tree would leave a tree that cannot converge. A live tree is refused (ending one early is
|
|
408
|
+
* `cancel_mission`); `finish_mission` archives instead. */
|
|
409
|
+
deleteTree(rootId: string): Promise<MutationResult<readonly string[]>>;
|
|
410
|
+
/** Nodes a live worker still holds, for cancellation to interrupt. */
|
|
411
|
+
heldByLiveWorkers(rootId?: string): readonly NodeRecord[];
|
|
412
|
+
private locate;
|
|
413
|
+
private replace;
|
|
414
|
+
/** A node's aggregate status: `ready` once every child is terminal. No children at all is also
|
|
415
|
+
* `ready` — an ordinary mission, not a parent stuck waiting for premises that no longer exist. */
|
|
416
|
+
private aggregateStatus;
|
|
417
|
+
private pendingChildren;
|
|
418
|
+
/** Every node that lists `id` as a child; a reused prerequisite has more than one. */
|
|
419
|
+
private parentsOf;
|
|
420
|
+
/** Propagate aggregate readiness up the dependency graph after a node changed, starting from
|
|
421
|
+
* EVERY parent that lists it (dedup reuse can give it several) rather than the `parentId` chain.
|
|
422
|
+
* A branch stops at an unchanged status, since an unchanged node cannot change its parents.
|
|
423
|
+
* Returns whether some parent is now ready, i.e. the engine has an aggregate to dispatch. */
|
|
424
|
+
private recomputeAncestors;
|
|
425
|
+
/** Recompute the given nodes' aggregate statuses and spread any change upward — the shared walk
|
|
426
|
+
* behind `recomputeAncestors` and deletion, for callers that already know what was touched. */
|
|
427
|
+
private propagateFrom;
|
|
428
|
+
/** Dedup inside the decomposer's own neighborhood — its siblings' subtrees plus its own — never
|
|
429
|
+
* the whole tree, and never the node itself or an ancestor, which would close a cycle.
|
|
430
|
+
* Equivalence is the normalized TITLE and DESCRIPTION: a shared title is common ("补充测试" is one
|
|
431
|
+
* many unrelated missions wear), and a false reuse hands a later branch a RESULT answering a
|
|
432
|
+
* different question, so a title match with a different description is created as new mission. */
|
|
433
|
+
private findEquivalent;
|
|
434
|
+
/** How many of these specs would actually ADD a node: a child that reuses an existing
|
|
435
|
+
* prerequisite must not be counted, or the ceiling would fire on the engine's own reuse path.
|
|
436
|
+
* A pre-pass, not a gate in the loop, because a refusal must leave no child behind.
|
|
437
|
+
*
|
|
438
|
+
* It has to answer the SAME question the loop answers, including the one case the loop handles
|
|
439
|
+
* through its `created` list: a spec repeated inside ONE call reuses the node that call is about to
|
|
440
|
+
* create. This used to push a `'pending'` placeholder into the equivalent-scope, and
|
|
441
|
+
* `findEquivalent` resolves ids with `nodes.get(id)` → `undefined`, so the repeat was charged as a
|
|
442
|
+
* second new node and a legal decomposition was refused with a wrong number (`node-limit`). */
|
|
443
|
+
private countNewChildren;
|
|
444
|
+
private flush;
|
|
445
|
+
/** Serialize one mutation against every other mutation. */
|
|
446
|
+
private withLock;
|
|
447
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Trouble vocabulary: one node status in the language the model and the panel read, and the ONE
|
|
3
|
+
* predicate that decides whether a node carries trouble worth reporting.
|
|
4
|
+
*
|
|
5
|
+
* It lives in its own module because neither member is a RENDERING concern, even though both used to
|
|
6
|
+
* sit inside `prompt.ts`:
|
|
7
|
+
*
|
|
8
|
+
* - `statusLabel` is what every refusal the STATE MACHINE writes names a status with
|
|
9
|
+
* (`tree.ts`: "任务 X 处于「执行中」,不能…") — the state machine reaching up into the prompt
|
|
10
|
+
* renderer for a word is a layer inversion, not a dependency it needs.
|
|
11
|
+
* - `isTroubledNode` is the ENGINE's reading of a durable record, shared by four channels (the
|
|
12
|
+
* owner-facing `isTroubled` flag, and the stall / repeated-hang / failed-start heads-ups), none of
|
|
13
|
+
* which is a prompt.
|
|
14
|
+
*
|
|
15
|
+
* `prompt.ts` still owns the text that RENDERS trouble; it imports this module, never the other way
|
|
16
|
+
* round.
|
|
17
|
+
*
|
|
18
|
+
* @module @avantf/mission-core/trouble
|
|
19
|
+
*/
|
|
20
|
+
import { type NodeRecord } from './types.js';
|
|
21
|
+
/** One node status in the language the model and the panel read. */
|
|
22
|
+
export declare function statusLabel(status: string | undefined): string;
|
|
23
|
+
/**
|
|
24
|
+
* Whether ONE node carries trouble worth telling the owner about: four durable counters, each at the
|
|
25
|
+
* ENGINE's own floor — silent reclaims (`stalls`), consecutive hangs (`hungCount`), failed attempts
|
|
26
|
+
* (`failures`), starts that never got a worker (`spawnFailures`).
|
|
27
|
+
*
|
|
28
|
+
* ONE definition, because four channels ask this question: the owner-facing flag (`isTroubled`), and
|
|
29
|
+
* the three heads-ups (`escalateTrouble` for stalls and repeated hangs, and the failed-start one in
|
|
30
|
+
* the host). They used to disagree — the flag counted `spawnFailures` while the stall gate did not —
|
|
31
|
+
* so a mission that could not get a worker started read as 「反复出过问题」 in `list_missions` and the
|
|
32
|
+
* owner was never told why. Splitting the predicate out is what makes "both channels speak one
|
|
33
|
+
* vocabulary" true by construction rather than by comment.
|
|
34
|
+
*
|
|
35
|
+
* `hungCount` is the one member that is a STREAK rather than a history: any real output clears it, so
|
|
36
|
+
* a node that hung three times and then made progress stops reading as troubled. That is deliberate —
|
|
37
|
+
* the flag answers "is this happening to it NOW", and the other three counters (`stalls`, `failures`,
|
|
38
|
+
* `spawnFailures`) are the ones that never clear.
|
|
39
|
+
*/
|
|
40
|
+
export declare function isTroubledNode(node: NodeRecord): boolean;
|