@avantf/dsh-mission 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +249 -0
  3. package/cordis.patch.yml +13 -0
  4. package/lib/client.js +225 -0
  5. package/lib/dsh-build.json +16 -0
  6. package/lib/envinit-bootstrap.js +334 -0
  7. package/lib/index.js +7021 -0
  8. package/lib/interface-version.json +4 -0
  9. package/lib/mission-core/capacity.d.ts +100 -0
  10. package/lib/mission-core/continuation.d.ts +106 -0
  11. package/lib/mission-core/dispatch.d.ts +165 -0
  12. package/lib/mission-core/engine.d.ts +331 -0
  13. package/lib/mission-core/index.d.ts +25 -0
  14. package/lib/mission-core/liveness.d.ts +99 -0
  15. package/lib/mission-core/prompt.d.ts +112 -0
  16. package/lib/mission-core/resources.d.ts +57 -0
  17. package/lib/mission-core/timing.d.ts +37 -0
  18. package/lib/mission-core/tree.d.ts +447 -0
  19. package/lib/mission-core/trouble.d.ts +40 -0
  20. package/lib/mission-core/types.d.ts +409 -0
  21. package/lib/mission-core/wellformed.d.ts +98 -0
  22. package/lib/types/claims.d.ts +3 -0
  23. package/lib/types/claims.d.ts.map +1 -0
  24. package/lib/types/client/MissionTreeView.d.ts +247 -0
  25. package/lib/types/client/MissionTreeView.d.ts.map +1 -0
  26. package/lib/types/client/api.d.ts +184 -0
  27. package/lib/types/client/api.d.ts.map +1 -0
  28. package/lib/types/client/contract.d.ts +209 -0
  29. package/lib/types/client/contract.d.ts.map +1 -0
  30. package/lib/types/client/index.d.ts +48 -0
  31. package/lib/types/client/index.d.ts.map +1 -0
  32. package/lib/types/client/seat.d.ts +26 -0
  33. package/lib/types/client/seat.d.ts.map +1 -0
  34. package/lib/types/client/styles.d.ts +6 -0
  35. package/lib/types/client/styles.d.ts.map +1 -0
  36. package/lib/types/coldResume.d.ts +123 -0
  37. package/lib/types/coldResume.d.ts.map +1 -0
  38. package/lib/types/domain.d.ts +93 -0
  39. package/lib/types/domain.d.ts.map +1 -0
  40. package/lib/types/envinit.d.ts +126 -0
  41. package/lib/types/envinit.d.ts.map +1 -0
  42. package/lib/types/executorSession.d.ts +129 -0
  43. package/lib/types/executorSession.d.ts.map +1 -0
  44. package/lib/types/faces.d.ts +30 -0
  45. package/lib/types/faces.d.ts.map +1 -0
  46. package/lib/types/host.d.ts +752 -0
  47. package/lib/types/host.d.ts.map +1 -0
  48. package/lib/types/index.d.ts +48 -0
  49. package/lib/types/index.d.ts.map +1 -0
  50. package/lib/types/interface_gate.d.ts +51 -0
  51. package/lib/types/interface_gate.d.ts.map +1 -0
  52. package/lib/types/log.d.ts +18 -0
  53. package/lib/types/log.d.ts.map +1 -0
  54. package/lib/types/projectionCache.d.ts +22 -0
  55. package/lib/types/projectionCache.d.ts.map +1 -0
  56. package/lib/types/prompt.d.ts +106 -0
  57. package/lib/types/prompt.d.ts.map +1 -0
  58. package/lib/types/source.d.ts +22 -0
  59. package/lib/types/source.d.ts.map +1 -0
  60. package/lib/types/store.d.ts +20 -0
  61. package/lib/types/store.d.ts.map +1 -0
  62. package/lib/types/timeFormat.d.ts +56 -0
  63. package/lib/types/timeFormat.d.ts.map +1 -0
  64. package/lib/types/tools.d.ts +34 -0
  65. package/lib/types/tools.d.ts.map +1 -0
  66. package/lib/types/wellformed.d.ts +43 -0
  67. package/lib/types/wellformed.d.ts.map +1 -0
  68. package/lib/types/wire.d.ts +178 -0
  69. package/lib/types/wire.d.ts.map +1 -0
  70. package/lib/types/workerEvents.d.ts +36 -0
  71. package/lib/types/workerEvents.d.ts.map +1 -0
  72. package/lib/types/workerSessions.d.ts +256 -0
  73. package/lib/types/workerSessions.d.ts.map +1 -0
  74. package/package.json +145 -0
@@ -0,0 +1,37 @@
1
+ /**
2
+ * The DERIVED half of a node's three timestamps: how long it queued, how long it ran, and how long
3
+ * it existed. The three canonical instants live on {@link NodeRecord} (`createdAt` / `dispatchedAt`
4
+ * / `endedAt`); these helpers are the only arithmetic over them, so a display, a tool answer and a
5
+ * test cannot each round or clamp a duration their own way.
6
+ *
7
+ * Nothing here reads a clock: every quantity is a function of the record alone. A live "how long has
8
+ * this been waiting" is deliberately NOT offered — a duration that grows between renders belongs to
9
+ * the renderer, and inventing one here would make a persisted value look like a clock reading.
10
+ *
11
+ * @module @avantf/mission-core/timing
12
+ */
13
+ /** The three instants, in the minimal shape every reader of them shares. */
14
+ export interface NodeTiming {
15
+ readonly createdAt: number;
16
+ readonly dispatchedAt: number | null;
17
+ readonly endedAt: number | null;
18
+ }
19
+ /**
20
+ * How long this node QUEUED before its first dispatch: `dispatchedAt − createdAt`. `null` when it
21
+ * has never been dispatched — the honest answer, since the wait is still going and its end is not
22
+ * recorded. Never negative: a clock that went backwards is clamped to 0 rather than shown as a
23
+ * time-travelling queue.
24
+ */
25
+ export declare function queueMs(node: NodeTiming): number | null;
26
+ /**
27
+ * How long this node's executor ran: `endedAt − dispatchedAt`. `null` while it is still running or
28
+ * queued, and also for the one terminal case that has no execution to measure — a node cancelled
29
+ * before it ever ran (`dispatchedAt === null`). Clamped to 0 for the same reason as {@link queueMs}.
30
+ */
31
+ export declare function runMs(node: NodeTiming): number | null;
32
+ /**
33
+ * How long this node has existed so far: `endedAt − createdAt` once terminal, and `null` while it is
34
+ * still in play. Unlike {@link queueMs} and {@link runMs} this is defined even for a node that never
35
+ * ran — a cancelled prerequisite still spent wall-clock time in the tree.
36
+ */
37
+ export declare function totalMs(node: NodeTiming): number | null;
@@ -0,0 +1,447 @@
1
+ import { type ContinuationDelta } from './continuation.js';
2
+ import { type CapacityPolicy, type DispatchPlan } from './dispatch.js';
3
+ import { type WellFormedSource } from './wellformed.js';
4
+ import { type ChildSpec, type DecomposeOutcome, type DispatchView, type MutationResult, type NodeRecord, type TreeRecord } from './types.js';
5
+ export interface TreeState {
6
+ tree: TreeRecord;
7
+ nodes: Map<string, NodeRecord>;
8
+ }
9
+ /** Durable sink for tree records; the store owns batching and durability. */
10
+ export interface TreeStore {
11
+ /** Every persisted tree document, for startup reconciliation. */
12
+ loadAll(): Promise<TreeState[]>;
13
+ /** Persist one tree document. Resolves once the write is durable. */
14
+ put(state: TreeState): Promise<void>;
15
+ remove(rootId: string): Promise<void>;
16
+ }
17
+ /** One persisted spill artifact: where it is, and how to read it back. */
18
+ export interface SpilledText {
19
+ /** Opaque model-facing locator produced by the storage backend. */
20
+ readonly locator: string;
21
+ /** The backend's retrieval guidance, shown next to the locator. */
22
+ readonly hint: string;
23
+ }
24
+ /**
25
+ * What asking durable storage about an owner session resolved to. Three states, not a boolean:
26
+ * "cannot tell" and "gone" demand different actions — a gone owner's tree is destroyed, an
27
+ * opaque host must never be allowed to destroy mission (a stray tree is recoverable, destroyed mission
28
+ * is not), yet the operator still has to hear about it.
29
+ */
30
+ export type OwnerProbe = {
31
+ readonly kind: 'exists';
32
+ } | {
33
+ readonly kind: 'missing';
34
+ } | {
35
+ readonly kind: 'unobservable';
36
+ readonly detail: string;
37
+ };
38
+ /** One tree whose owner did not resolve to `exists`, with the probe that explains why. */
39
+ export interface OrphanedTree {
40
+ readonly tree: TreeRecord;
41
+ readonly probe: OwnerProbe;
42
+ }
43
+ /** Injected environment facts; the core never imports the harness. */
44
+ export interface TreeDeps {
45
+ isAgentLive(sessionId: string): boolean;
46
+ /** Whether a session still EXISTS, asked of durable storage rather than the live registry: an
47
+ * agent is materialized on demand, so right after a restart a live check is false. Ownership is
48
+ * decided by this one. `unobservable` is a THIRD answer, not a synonym for either. */
49
+ probeOwner(sessionId: string): Promise<OwnerProbe>;
50
+ /** Persist an over-long result, or return `null` when no spill backend is configured — the node
51
+ * then keeps the full text inline, because a locator nobody can resolve loses the tail. */
52
+ spill(text: string): Promise<SpilledText | null>;
53
+ /** Clock, injectable for tests. */
54
+ now(): number;
55
+ /** Id generator, injectable for tests. */
56
+ newId(): string;
57
+ /**
58
+ * The family's well-formed repair, injected by the plugin from the LOADED BASE KIT (interface v2,
59
+ * `wellFormedText` / `wellFormedDeep`). Absent means the base could not provide it — base missing,
60
+ * or older than v2 — and the tree uses its own {@link LOCAL_WELL_FORMED} copy instead. This is a
61
+ * degradation, never a refusal: a tree with no injected source still mounts and still repairs.
62
+ */
63
+ readonly wellFormed?: WellFormedSource;
64
+ }
65
+ export interface CreateRootInput {
66
+ readonly ownerSessionId: string;
67
+ readonly title: string;
68
+ readonly description: string;
69
+ /** The owner's initial analysis; stored as the root's background facts. */
70
+ readonly analysis: readonly string[];
71
+ /** The scope this mission will modify (a directory or file), or `undefined`/blank for none. The
72
+ * root has no parent to inherit from, so this is the only declaration site for a tree's unit. */
73
+ readonly unit?: string | null;
74
+ /** This mission's declared capacity weight (cores-equivalent); `undefined` reads as the default 1.
75
+ * The root has no parent to inherit from. */
76
+ readonly weight?: number;
77
+ /** This mission's declared round-cap relaxation (see {@link NodeRecord.roundMs}); the root has no
78
+ * parent, so this is the only declaration site for a tree's relaxed ceiling. */
79
+ readonly roundMs?: number | null;
80
+ }
81
+ /** Random 8-hex-char id, matching the short ids the tools expose. `crypto` is reached through a
82
+ * typed globals lookup because the core is built without DOM or Node ambient types. */
83
+ export declare function defaultNewId(): string;
84
+ export declare class MissionTree {
85
+ private readonly deps;
86
+ private readonly states;
87
+ private chain;
88
+ private store;
89
+ /** Trees whose in-memory progress moved since the last durable flush. */
90
+ private readonly progressDirty;
91
+ constructor(store: TreeStore | undefined, deps: TreeDeps);
92
+ /**
93
+ * The repair pair every INBOUND write uses: the base kit the plugin injected, or the local
94
+ * degradation copy. ONE accessor, so "which implementation" is decided in one place and every
95
+ * write-path entry below asks the same question.
96
+ */
97
+ private get wellFormed();
98
+ private requireStore;
99
+ /** Load every persisted tree and reconcile bindings against live agents: a `running` node whose
100
+ * `claimedBy` no longer resolves is reclaimed as `interrupted`, restart and hot reload alike. */
101
+ open(): Promise<void>;
102
+ /**
103
+ * Three outcomes on open, decided by what the durable record says and what this process can see:
104
+ *
105
+ * 1. SURVIVOR — `running` and the worker IS materialized (hot reload): keep the binding, only
106
+ * reset `progressAt`, or a long run looks silent from the moment we open.
107
+ * 2. CONTINUABLE — `running` but the worker is not materialized in THIS process (a restart or a
108
+ * crash): demote to `interrupted`, but FIRST park `claimedBy` in `lastWorkerId`. The binding is
109
+ * the only place that session id was recorded, so dropping it here is what made every restart
110
+ * unable to do anything but start over.
111
+ * 3. NO HANDLE — not `running`: nothing to do, and `lastWorkerId` (if any) is already spent.
112
+ *
113
+ * The survivor branch is deliberately unchanged: a live binding is authoritative, and a stale
114
+ * handle left beside it is only ever consulted by `adoptContinuation`, which requires the node to
115
+ * be dispatchable — a `running` node never is.
116
+ */
117
+ private reconcileOnOpen;
118
+ /** Trees whose owner session did not resolve to `exists`, each with the probe that says why.
119
+ * Asked of durable storage rather than the live registry, so a restart keeps its trees and only a
120
+ * genuinely deleted session loses them. `missing` and `unobservable` are reported separately: the
121
+ * caller decides what may be destroyed (see `MissionEngine.reconcileOrphans`). */
122
+ orphanedTrees(): Promise<readonly OrphanedTree[]>;
123
+ destroyTree(rootId: string): Promise<void>;
124
+ treeOf(rootId: string): TreeRecord | undefined;
125
+ node(id: string): NodeRecord | undefined;
126
+ /** Claim the right to report one tree's terminal state: true exactly once per transition,
127
+ * durably, so an engine restart cannot produce a second wake. */
128
+ claimReport(rootId: string): Promise<boolean>;
129
+ /** Allow a later terminal transition of the same tree to be reported again. */
130
+ clearReport(rootId: string): Promise<void>;
131
+ trees(): readonly TreeRecord[];
132
+ nodesOf(rootId: string): readonly NodeRecord[];
133
+ /** Ancestors of a node, root → parent. */
134
+ chainOf(node: NodeRecord): readonly NodeRecord[];
135
+ /** The current dispatch view for one node. Built at spawn time, not dispatch time, so the prompt
136
+ * reflects children results that landed between the dispatch decision and the spawn. */
137
+ view(nodeId: string): DispatchView | undefined;
138
+ /** Terminal children of an aggregate, in decomposition order. */
139
+ terminalChildren(node: NodeRecord): readonly NodeRecord[];
140
+ /** What changed on this node since the prompt its bound session was handed (see
141
+ * `@avantf/mission-core/continuation`). Read-only and synchronous, because the wake path consults
142
+ * it before it adopts anything: a refusal that has already consumed an address is a refusal that
143
+ * cannot be undone.
144
+ *
145
+ * `undefined` when the node does not exist; a node that merely has no baseline still answers, with
146
+ * `baselineKnown: false` — "unknown" is a fact the caller renders, not an error. */
147
+ continuationDelta(nodeId: string): ContinuationDelta | undefined;
148
+ /** The next node to dispatch: `ready`/`interrupted`, oldest first across trees, excluding ones
149
+ * whose binding still resolves to a live agent. A vanished worker's node is reclaimed by the
150
+ * engine's sweep before it becomes a candidate again.
151
+ *
152
+ * The selection itself — ordering, parked/backoff rules and the unit-lease skip — lives in
153
+ * `@avantf/mission-core/dispatch`; this is the state access, and the lease side of it is
154
+ * re-checked atomically at every transition into `running` below, so a race that slips past this
155
+ * filter is refused rather than run. The optional `capacity` policy arms the capacity gate; without
156
+ * it this is exactly the pre-capacity contract. */
157
+ nextDispatchable(exclude?: ReadonlySet<string>, capacity?: CapacityPolicy): NodeRecord | undefined;
158
+ /** The full admission plan (see `@avantf/mission-core/dispatch`): the node to dispatch AND every
159
+ * candidate left waiting with its reason. The engine uses the latter half for `waitingFor` and for
160
+ * the aging bookkeeping; tree-level callers that only need the answer use `nextDispatchable`. */
161
+ planDispatch(exclude?: ReadonlySet<string>, capacity?: CapacityPolicy): DispatchPlan;
162
+ /**
163
+ * What the capacity gate currently has in flight: a node counts once it is BOUND (`running` with a
164
+ * `claimedBy`), not once its worker materializes, so a pass cannot over-subscribe the pool by
165
+ * dispatching into a lazily-created child.
166
+ */
167
+ runningLoad(): {
168
+ count: number;
169
+ weight: number;
170
+ };
171
+ /**
172
+ * The lease check shared by all three transitions INTO `running` (`dispatch`, `adoptParked`,
173
+ * `adoptContinuation`). It has to be re-made here, under the tree lock, and not only in
174
+ * `nextDispatchable`: the wake paths bind from a snapshot taken outside it, and two passes can
175
+ * interleave between "this unit looked free" and "mark it running".
176
+ *
177
+ * `undefined` means the unit is free (or the node declared none — the no-op case that keeps
178
+ * undeclared records byte-for-byte as they behaved before leases existed). Reached through
179
+ * {@link admissionRefusal}, which pairs it with the capacity recheck; the capacity gate has the
180
+ * same snapshot problem and is re-made in the same breath.
181
+ */
182
+ private unitRefusal;
183
+ /**
184
+ * The capacity counterpart of {@link unitRefusal}, re-made under the tree lock at every transition
185
+ * INTO `running`. `planDispatch` decides from a snapshot taken OUTSIDE the lock, so two passes can
186
+ * each see an empty machine and both bind; the arithmetic therefore has to be re-run here, in the
187
+ * same place the unit lease is, and through the SAME judgement the plan uses
188
+ * ({@link capacityWaitingFor}) so the two can never drift.
189
+ *
190
+ * The policy supplies the machine's capacity, the slot ceiling and (when the host set it) the
191
+ * machine-wide block. Its `runningCount`/`runningWeight` are the CALLER'S SNAPSHOT and are
192
+ * deliberately ignored: trusting them is the very TOCTOU this closes. `undefined` policy means the
193
+ * caller opted out of the gate entirely — the pre-capacity contract, unchanged.
194
+ *
195
+ * The refusal is a DEFERRAL in the strictest sense (`capacity-busy`): the node stays `ready` and
196
+ * NOTHING is charged — no `attempts`, `failures`, `spawnFailures`, no cooldown, no stall. See
197
+ * {@link RefusalCode}.
198
+ */
199
+ private capacityRefusal;
200
+ /**
201
+ * The two RESOURCE admissions every transition into `running` must re-make under the tree lock:
202
+ * the unit lease first (a scope conflict), then capacity (the machine). Kept in ONE call site per
203
+ * transition so "what has to be rechecked before binding" is named once and the two checks cannot
204
+ * drift or be reordered by accident. Budgets are deliberately NOT here: a spent budget is a
205
+ * FAILURE (`failExhausted`), while both of these are "not yet" and must charge nothing.
206
+ */
207
+ private admissionRefusal;
208
+ /** Nodes waiting for their parked session to be woken: `ready` with a recorded address, i.e. all
209
+ * children terminal and a session to continue in. The owner's wake gate admits on this. */
210
+ parkedReadyNodes(rootId?: string): readonly NodeRecord[];
211
+ /** Claim a parked node FOR its parked session so it can be woken rather than replaced. Unlike
212
+ * `dispatch` it reserves no new claim id — the identity is the parked session — and consumes
213
+ * `parkedWorker` in the same locked step, so a concurrent pass cannot both adopt and re-dispatch
214
+ * it. The claim must land BEFORE the wake is delivered, because a woken session may submit or
215
+ * decompose immediately and both authorize on `claimedBy`; a failed delivery is undone by a
216
+ * `reclaim(..., 'wake-failed')`. */
217
+ adoptParked(nodeId: string, workerId: string, capacity?: CapacityPolicy): Promise<MutationResult<DispatchView>>;
218
+ /**
219
+ * Claim a node FOR the lost session recorded in `lastWorkerId`, so its next execution CONTINUES
220
+ * that session (a cold wake) rather than starting a fresh executor. The continuation counterpart
221
+ * of {@link adoptParked}, and the same shape: no new claim id is reserved — the identity IS the
222
+ * recorded session — and the handle is consumed in the same locked step.
223
+ *
224
+ * Why the ceilings ARE consulted here, unlike for a parked adoption: a continuation IS a
225
+ * dispatch. It hands work to a model and advances `attempts`, so a node whose failure budget is
226
+ * spent must fail here exactly as `dispatch` would fail it, instead of being resurrected.
227
+ *
228
+ * The handle is CONSUMED (`lastWorkerId = null`) rather than kept: this dispatch either continues
229
+ * that session or — through `reclaim(..., 'wake-failed')` — falls back to a fresh one, and neither
230
+ * outcome may try the same address again. A runtime that refused the resume once will refuse it
231
+ * again; the fallback is the complete answer (`note_mission` is the cross-session hand-off).
232
+ */
233
+ adoptContinuation(nodeId: string, workerId: string, capacity?: CapacityPolicy): Promise<MutationResult<DispatchView>>;
234
+ /**
235
+ * Stamp the node with what the prompt just handed to the session showed it: the snapshot a later
236
+ * cold wake subtracts. The HOST owns the moment (it calls this once the delivery resolved, at each
237
+ * of its three prompt sites), because only the host knows a prompt was actually read — a dispatch
238
+ * that never produced, or never delivered, a prompt must leave no baseline claiming otherwise.
239
+ *
240
+ * GUARDED on the live binding: between the delivery and this call, a sweep can reclaim the node
241
+ * (or another pass can re-dispatch it), and a baseline that says "this session saw this" must
242
+ * belong to the dispatch that is actually bound. A stamp that loses that race is refused
243
+ * silently: the wake that later reads a stale-or-missing baseline takes the conservative path,
244
+ * while a wrongly stamped one would under-report the drift to a live session.
245
+ *
246
+ * Tolerant rather than a mutation result, like {@link markCorrectionsDelivered}: it is bookkeeping
247
+ * about a prompt that already exists, and the caller can do nothing useful with a refusal. The
248
+ * return value exists so the caller (and tests) can tell a stamp from a lost race.
249
+ */
250
+ recordDispatchBaseline(nodeId: string, holder: string): Promise<boolean>;
251
+ /**
252
+ * Spend a continuation handle WITHOUT using it, so the node's next dispatch starts fresh. Used
253
+ * when the drift since that session's prompt is material (`isMaterialChange`): the address is not
254
+ * worth spending, and leaving it would make every later pass re-decide the same thing.
255
+ *
256
+ * No status change, no budget, no cooldown — the caller's ordinary `dispatch` follows in the same
257
+ * engine pass. The baseline is left alone: the fresh dispatch stamps its own, which is the whole
258
+ * point of replacing the session.
259
+ */
260
+ abandonContinuation(nodeId: string, workerId: string): Promise<MutationResult<NodeRecord>>;
261
+ /** Roll up every OPEN tree into the counts the guidance layer renders; closed trees are archived
262
+ * and no longer echo into the owner's prompt. */
263
+ summary(): {
264
+ trees: number;
265
+ ready: number;
266
+ running: number;
267
+ blocked: number;
268
+ done: number;
269
+ failed: number;
270
+ };
271
+ createRoot(input: CreateRootInput): Promise<MutationResult<NodeRecord>>;
272
+ /** Allocate an id not present in the tree (and not already reserved this call). */
273
+ private allocateId;
274
+ /** One fresh node record; the only place node defaults are written. */
275
+ private makeNode;
276
+ /** Mark one node dispatched and bind it to a reserved child session id, in the same locked step
277
+ * as the dispatch decision, so two engine passes cannot both dispatch it. `attempts` increments
278
+ * on every dispatch (the aggregate pass counts as one) and is the generation marker the
279
+ * `note_mission` gate reads; the failed ceiling rides `failures` instead, so a successful aggregate
280
+ * round is never charged against it. */
281
+ dispatch(nodeId: string, claimId: string, capacity?: CapacityPolicy): Promise<MutationResult<DispatchView>>;
282
+ /** Fail a node whose budget ran out and refuse the dispatch, so the owner is told (the engine
283
+ * reports terminal roots) and the node never re-enters the candidate pool. */
284
+ private failExhausted;
285
+ /** Return a dispatched node to the pool; `cause` decides which budget the reclaim charges.
286
+ * `vanished`/`stalled`: a worker ran (or was starting) without a result — charges `failures`,
287
+ * and only `stalled` also increments `stalls`. `spawn-failed`: no worker ever started — charges
288
+ * `spawnFailures` and pushes the next dispatch back by a cooldown. `wake-failed`: an adoption
289
+ * that could not be delivered — back to `ready`, charging neither budget, so the caller's fresh
290
+ * dispatch proceeds. `hung`: the worker was still alive but produced nothing for a whole window
291
+ * (or exceeded the round cap) — like `wake-failed` it charges NEITHER budget and adds no cooldown,
292
+ * because an apparatus outage is not a failed mission and must not spend the budget that ends the
293
+ * node; unlike `wake-failed` it lands in `interrupted`, the ordinary re-queue, and it advances the
294
+ * `hungCount` STREAK (cleared by real output, never by a re-dispatch) so a node that hangs every
295
+ * round eventually reaches the owner instead of re-running forever. `attempts` is never
296
+ * rolled back: it is the `note_mission` generation marker, not a budget.
297
+ *
298
+ * Every arm leaves `running`, so this is one of the paths that RELEASES the node's unit lease
299
+ * (the lease is the set of `running` nodes' units, never a separate table — see `dispatch.ts`).
300
+ *
301
+ * `expectedHolder` is a compare-and-swap for a caller that judged the node from a snapshot taken
302
+ * OUTSIDE this lock — `MissionEngine.reclaimStale` takes one, then awaits `interruptWorker` per
303
+ * node. If a second sweep reclaimed and re-dispatched the node in that window, the binding has
304
+ * moved on and this stale verdict must be refused: acting on it would unbind a LIVE worker (whose
305
+ * `submit_mission` then answers `not-owner`) and charge `failures` a second time for one attempt.
306
+ * `undefined` means "no expectation" — the callers that act on a value they just read use that. */
307
+ reclaim(nodeId: string, cause?: 'vanished' | 'stalled' | 'hung' | 'spawn-failed' | 'wake-failed', expectedHolder?: string | null): Promise<MutationResult<NodeRecord>>;
308
+ /** Clear the spawn-failure counter after a worker started successfully: the budget is for
309
+ * CONSECUTIVE start failures. Memory-only between flushes is fine — a dispatch flush follows. */
310
+ noteSpawnSuccess(nodeId: string): void;
311
+ /** Record that one worker PRODUCED something — model output, a tool call, a tool result — on a
312
+ * hot path from the worker's durable append feed, so it updates memory only. This is the clock the
313
+ * stale check reads: output is also evidence of life, so one write moves `activityAt` too. A
314
+ * timestamp not newer than what we have is ignored, which makes racing a dispatch safe.
315
+ *
316
+ * It is ALSO the clearing point of the hang streak: production is the one event that proves the
317
+ * round was not merely retrying forever, so `hungCount` goes back to 0 here (and only then — a
318
+ * re-dispatch sets `progressAt` without proving anything and deliberately leaves the streak). */
319
+ touchProgress(nodeId: string, at: number): void;
320
+ /** Record that a worker's session emitted SOME event without claiming it produced anything: the
321
+ * transport-layer half of liveness. It moves only `activityAt`, so a worker whose events are all
322
+ * retries and route snapshots is still heard from (never `stalled`) while `progressAt` stays
323
+ * where its last real output left it — exactly the "alive but unproductive" state that
324
+ * `judgeWorker` reclaims as `hung`. */
325
+ touchActivity(nodeId: string, at: number): void;
326
+ /** Persist progress observed since the last flush, coalesced on purpose: writing the whole tree
327
+ * document per worker event would cost more than the mission it guards. The usual lock keeps a
328
+ * stale document from landing after a newer one. */
329
+ flushProgress(): Promise<void>;
330
+ /** Claim the right to tell the owner about one node's stalls: true exactly once per node,
331
+ * durably, the same contract as `claimReport`. */
332
+ claimStallReport(nodeId: string): Promise<boolean>;
333
+ /**
334
+ * Remember the executor session a LAZY lookup resolved for one node, so the panel never pays for
335
+ * the same lookup twice (W18: the node id is clickable on every node, and a historical record has
336
+ * no `executorSessionId` — the resolution reads session logs, which is exactly the cost that must
337
+ * not happen at render time).
338
+ *
339
+ * Write-once: a handle already on the record is authoritative and the write is skipped. A handle
340
+ * that arrived while the lookup was in flight belongs to a NEWER attempt, and overwriting it with
341
+ * a resolved OLDER one would point the panel at the wrong session — the same "only the last
342
+ * attempt" rule every other writer of this field follows.
343
+ *
344
+ * Tolerant rather than a mutation result, like {@link recordDispatchBaseline} and
345
+ * {@link markCorrectionsDelivered}: it is bookkeeping, and the caller can do nothing useful with a
346
+ * refusal. The `boolean` exists so the caller knows whether to announce a change.
347
+ */
348
+ rememberExecutor(nodeId: string, sessionId: string): Promise<boolean>;
349
+ nodeHeldBy(sessionId: string): NodeRecord | undefined;
350
+ /** Record the holding executor's own analysis, attributed to the `attempts` value it is running
351
+ * under — that attribution is what `decompose` checks, so notes inherited from an earlier round
352
+ * do not authorize a split. Text is appended (duplicates dropped) and kept for every later
353
+ * dispatch. Refused on a terminal node. */
354
+ recordAnalysis(nodeId: string, callerSessionId: string, analysis: string): Promise<MutationResult<NodeRecord>>;
355
+ /** Split a node into children: a first decomposition, or a re-decomposition of an aggregate whose
356
+ * children are all terminal — the pass that read their conclusions and judged the objective
357
+ * unmet. A node with UNFINISHED children is refused. `submitResult` and `decompose` stay mutually
358
+ * exclusive through node state, never prompt discipline. The analysis is written earlier by
359
+ * `recordAnalysis`; this only checks the splitting dispatch wrote its own.
360
+ *
361
+ * Leaving `running` here is also what RELEASES this node's unit lease, and the children created
362
+ * below inherit that unit unless they declare their own — so the same-scope siblings the split
363
+ * just created are serialized against each other by the same state rule (see `dispatch.ts`). */
364
+ decompose(nodeId: string, callerSessionId: string, children: readonly ChildSpec[]): Promise<MutationResult<DecomposeOutcome>>;
365
+ /** Submit a terminal result; rejected while any child is unfinished. An aggregate (every child
366
+ * terminal) DOES submit — that pass reads the conclusions and states the outcome, which is how a
367
+ * decomposed node reaches `done` and the tree converges. */
368
+ submitResult(nodeId: string, callerSessionId: string, result: string): Promise<MutationResult<{
369
+ node: NodeRecord;
370
+ parentReady: boolean;
371
+ }>>;
372
+ /** Record one correction, on the tree owner's authority only. The text lands in the node's
373
+ * `corrections`, NOT in `context`: `context` is the decomposer's "why this mission exists", while a
374
+ * correction is the owner's instruction about mission already handed out — a different author and a
375
+ * different lifetime. (Rendering them together once made every correction invisible to every
376
+ * descendant, because the chain line only carries `context[0]`.) Either way it is a durable block
377
+ * that every later dispatch of this node renders, so it survives a reclaim, a retry and the
378
+ * aggregate round rather than living in one worker's inbox. */
379
+ correct(nodeId: string, callerSessionId: string, text: string): Promise<MutationResult<NodeRecord>>;
380
+ /**
381
+ * Advance the durable delivery watermark of {@link NodeRecord.correctionsDeliveredUpTo}: the
382
+ * first `upTo` corrections have been confirmed READ by the session that holds this node (either
383
+ * a live steer, or a cold-wake prompt that carried them). What it buys is the restart case — a
384
+ * mark kept only in memory is empty exactly when the wake that needs it happens.
385
+ *
386
+ * Deliberately tolerant rather than a mutation result: this is bookkeeping ABOUT a delivery that
387
+ * already happened, so it is monotone (a raced, older report can never pull the mark back) and
388
+ * clamped to the array's own length (a report that outran a concurrent append cannot make a later
389
+ * wake skip a correction nobody read). A no-op returns without writing.
390
+ */
391
+ markCorrectionsDelivered(nodeId: string, upTo: number): Promise<void>;
392
+ /** Cancel everything BELOW one node, strictly downward: the node itself, its ancestors and its
393
+ * siblings are untouched, and the cancelled nodes become `failed` so the node's aggregate pass
394
+ * becomes dispatchable again. The tree owner is the caller that matters, because a node with
395
+ * unfinished children is `blocked` and holds no claim. */
396
+ cancelSubworks(nodeId: string, callerSessionId: string, onReclaim?: (claimId: string) => void): Promise<MutationResult<readonly NodeRecord[]>>;
397
+ /** Record that the owner read a terminal result; unlocks `finish_mission`. */
398
+ markResultRead(nodeId: string): Promise<MutationResult<NodeRecord>>;
399
+ /** Close a tree out on the owner's authority; refused while results are unread or the root is not
400
+ * terminal. Either terminal root may be closed — a `failed` root must be reported then retired.
401
+ * Closing is archival: nodes, results and readers are untouched. */
402
+ finish(rootId: string, callerSessionId: string): Promise<MutationResult<NodeRecord>>;
403
+ /** Cancel a whole tree: mark every non-terminal node cancelled-as-failed. */
404
+ cancelTree(rootId: string, callerSessionId: string, onReclaim?: (claimId: string) => void): Promise<MutationResult<readonly NodeRecord[]>>;
405
+ /** Delete one whole tree. The unit is the TREE, not the node: its siblings' premises, the
406
+ * aggregate story and the tree's identity all live in the same record, so pruning one node out of
407
+ * a live tree would leave a tree that cannot converge. A live tree is refused (ending one early is
408
+ * `cancel_mission`); `finish_mission` archives instead. */
409
+ deleteTree(rootId: string): Promise<MutationResult<readonly string[]>>;
410
+ /** Nodes a live worker still holds, for cancellation to interrupt. */
411
+ heldByLiveWorkers(rootId?: string): readonly NodeRecord[];
412
+ private locate;
413
+ private replace;
414
+ /** A node's aggregate status: `ready` once every child is terminal. No children at all is also
415
+ * `ready` — an ordinary mission, not a parent stuck waiting for premises that no longer exist. */
416
+ private aggregateStatus;
417
+ private pendingChildren;
418
+ /** Every node that lists `id` as a child; a reused prerequisite has more than one. */
419
+ private parentsOf;
420
+ /** Propagate aggregate readiness up the dependency graph after a node changed, starting from
421
+ * EVERY parent that lists it (dedup reuse can give it several) rather than the `parentId` chain.
422
+ * A branch stops at an unchanged status, since an unchanged node cannot change its parents.
423
+ * Returns whether some parent is now ready, i.e. the engine has an aggregate to dispatch. */
424
+ private recomputeAncestors;
425
+ /** Recompute the given nodes' aggregate statuses and spread any change upward — the shared walk
426
+ * behind `recomputeAncestors` and deletion, for callers that already know what was touched. */
427
+ private propagateFrom;
428
+ /** Dedup inside the decomposer's own neighborhood — its siblings' subtrees plus its own — never
429
+ * the whole tree, and never the node itself or an ancestor, which would close a cycle.
430
+ * Equivalence is the normalized TITLE and DESCRIPTION: a shared title is common ("补充测试" is one
431
+ * many unrelated missions wear), and a false reuse hands a later branch a RESULT answering a
432
+ * different question, so a title match with a different description is created as new mission. */
433
+ private findEquivalent;
434
+ /** How many of these specs would actually ADD a node: a child that reuses an existing
435
+ * prerequisite must not be counted, or the ceiling would fire on the engine's own reuse path.
436
+ * A pre-pass, not a gate in the loop, because a refusal must leave no child behind.
437
+ *
438
+ * It has to answer the SAME question the loop answers, including the one case the loop handles
439
+ * through its `created` list: a spec repeated inside ONE call reuses the node that call is about to
440
+ * create. This used to push a `'pending'` placeholder into the equivalent-scope, and
441
+ * `findEquivalent` resolves ids with `nodes.get(id)` → `undefined`, so the repeat was charged as a
442
+ * second new node and a legal decomposition was refused with a wrong number (`node-limit`). */
443
+ private countNewChildren;
444
+ private flush;
445
+ /** Serialize one mutation against every other mutation. */
446
+ private withLock;
447
+ }
@@ -0,0 +1,40 @@
1
+ /**
2
+ * Trouble vocabulary: one node status in the language the model and the panel read, and the ONE
3
+ * predicate that decides whether a node carries trouble worth reporting.
4
+ *
5
+ * It lives in its own module because neither member is a RENDERING concern, even though both used to
6
+ * sit inside `prompt.ts`:
7
+ *
8
+ * - `statusLabel` is what every refusal the STATE MACHINE writes names a status with
9
+ * (`tree.ts`: "任务 X 处于「执行中」,不能…") — the state machine reaching up into the prompt
10
+ * renderer for a word is a layer inversion, not a dependency it needs.
11
+ * - `isTroubledNode` is the ENGINE's reading of a durable record, shared by four channels (the
12
+ * owner-facing `isTroubled` flag, and the stall / repeated-hang / failed-start heads-ups), none of
13
+ * which is a prompt.
14
+ *
15
+ * `prompt.ts` still owns the text that RENDERS trouble; it imports this module, never the other way
16
+ * round.
17
+ *
18
+ * @module @avantf/mission-core/trouble
19
+ */
20
+ import { type NodeRecord } from './types.js';
21
+ /** One node status in the language the model and the panel read. */
22
+ export declare function statusLabel(status: string | undefined): string;
23
+ /**
24
+ * Whether ONE node carries trouble worth telling the owner about: four durable counters, each at the
25
+ * ENGINE's own floor — silent reclaims (`stalls`), consecutive hangs (`hungCount`), failed attempts
26
+ * (`failures`), starts that never got a worker (`spawnFailures`).
27
+ *
28
+ * ONE definition, because four channels ask this question: the owner-facing flag (`isTroubled`), and
29
+ * the three heads-ups (`escalateTrouble` for stalls and repeated hangs, and the failed-start one in
30
+ * the host). They used to disagree — the flag counted `spawnFailures` while the stall gate did not —
31
+ * so a mission that could not get a worker started read as 「反复出过问题」 in `list_missions` and the
32
+ * owner was never told why. Splitting the predicate out is what makes "both channels speak one
33
+ * vocabulary" true by construction rather than by comment.
34
+ *
35
+ * `hungCount` is the one member that is a STREAK rather than a history: any real output clears it, so
36
+ * a node that hung three times and then made progress stops reading as troubled. That is deliberate —
37
+ * the flag answers "is this happening to it NOW", and the other three counters (`stalls`, `failures`,
38
+ * `spawnFailures`) are the ones that never clear.
39
+ */
40
+ export declare function isTroubledNode(node: NodeRecord): boolean;