@avantf/dsh-mission 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +249 -0
  3. package/cordis.patch.yml +13 -0
  4. package/lib/client.js +225 -0
  5. package/lib/dsh-build.json +16 -0
  6. package/lib/envinit-bootstrap.js +334 -0
  7. package/lib/index.js +7021 -0
  8. package/lib/interface-version.json +4 -0
  9. package/lib/mission-core/capacity.d.ts +100 -0
  10. package/lib/mission-core/continuation.d.ts +106 -0
  11. package/lib/mission-core/dispatch.d.ts +165 -0
  12. package/lib/mission-core/engine.d.ts +331 -0
  13. package/lib/mission-core/index.d.ts +25 -0
  14. package/lib/mission-core/liveness.d.ts +99 -0
  15. package/lib/mission-core/prompt.d.ts +112 -0
  16. package/lib/mission-core/resources.d.ts +57 -0
  17. package/lib/mission-core/timing.d.ts +37 -0
  18. package/lib/mission-core/tree.d.ts +447 -0
  19. package/lib/mission-core/trouble.d.ts +40 -0
  20. package/lib/mission-core/types.d.ts +409 -0
  21. package/lib/mission-core/wellformed.d.ts +98 -0
  22. package/lib/types/claims.d.ts +3 -0
  23. package/lib/types/claims.d.ts.map +1 -0
  24. package/lib/types/client/MissionTreeView.d.ts +247 -0
  25. package/lib/types/client/MissionTreeView.d.ts.map +1 -0
  26. package/lib/types/client/api.d.ts +184 -0
  27. package/lib/types/client/api.d.ts.map +1 -0
  28. package/lib/types/client/contract.d.ts +209 -0
  29. package/lib/types/client/contract.d.ts.map +1 -0
  30. package/lib/types/client/index.d.ts +48 -0
  31. package/lib/types/client/index.d.ts.map +1 -0
  32. package/lib/types/client/seat.d.ts +26 -0
  33. package/lib/types/client/seat.d.ts.map +1 -0
  34. package/lib/types/client/styles.d.ts +6 -0
  35. package/lib/types/client/styles.d.ts.map +1 -0
  36. package/lib/types/coldResume.d.ts +123 -0
  37. package/lib/types/coldResume.d.ts.map +1 -0
  38. package/lib/types/domain.d.ts +93 -0
  39. package/lib/types/domain.d.ts.map +1 -0
  40. package/lib/types/envinit.d.ts +126 -0
  41. package/lib/types/envinit.d.ts.map +1 -0
  42. package/lib/types/executorSession.d.ts +129 -0
  43. package/lib/types/executorSession.d.ts.map +1 -0
  44. package/lib/types/faces.d.ts +30 -0
  45. package/lib/types/faces.d.ts.map +1 -0
  46. package/lib/types/host.d.ts +752 -0
  47. package/lib/types/host.d.ts.map +1 -0
  48. package/lib/types/index.d.ts +48 -0
  49. package/lib/types/index.d.ts.map +1 -0
  50. package/lib/types/interface_gate.d.ts +51 -0
  51. package/lib/types/interface_gate.d.ts.map +1 -0
  52. package/lib/types/log.d.ts +18 -0
  53. package/lib/types/log.d.ts.map +1 -0
  54. package/lib/types/projectionCache.d.ts +22 -0
  55. package/lib/types/projectionCache.d.ts.map +1 -0
  56. package/lib/types/prompt.d.ts +106 -0
  57. package/lib/types/prompt.d.ts.map +1 -0
  58. package/lib/types/source.d.ts +22 -0
  59. package/lib/types/source.d.ts.map +1 -0
  60. package/lib/types/store.d.ts +20 -0
  61. package/lib/types/store.d.ts.map +1 -0
  62. package/lib/types/timeFormat.d.ts +56 -0
  63. package/lib/types/timeFormat.d.ts.map +1 -0
  64. package/lib/types/tools.d.ts +34 -0
  65. package/lib/types/tools.d.ts.map +1 -0
  66. package/lib/types/wellformed.d.ts +43 -0
  67. package/lib/types/wellformed.d.ts.map +1 -0
  68. package/lib/types/wire.d.ts +178 -0
  69. package/lib/types/wire.d.ts.map +1 -0
  70. package/lib/types/workerEvents.d.ts +36 -0
  71. package/lib/types/workerEvents.d.ts.map +1 -0
  72. package/lib/types/workerSessions.d.ts +256 -0
  73. package/lib/types/workerSessions.d.ts.map +1 -0
  74. package/package.json +145 -0
@@ -0,0 +1,331 @@
1
+ /**
2
+ * The dispatch loop: pure host code that scans the tree, dispatches mission units and reclaims nodes whose worker vanished; it never calls a model.
3
+ * @module @avantf/mission-core/engine
4
+ */
5
+ import { MissionTree } from './tree.js';
6
+ import { type LivenessBound } from './liveness.js';
7
+ import type { CapacityPolicy } from './dispatch.js';
8
+ import type { ResourceProbe } from './resources.js';
9
+ import { type NodeRecord, type WaitingFor } from './types.js';
10
+ /** Resume a worker for one node; the engine awaits only the reservation, not the worker. */
11
+ export interface StartWorkerInput {
12
+ readonly node: NodeRecord;
13
+ /** The session id reserved for this dispatch, already bound on the node. */
14
+ readonly claimId: string;
15
+ /**
16
+ * How long the CAPACITY gate deferred this node before this dispatch picked it up, in ms. Read
17
+ * from the engine's own aging clock at the moment the candidate was selected — the same
18
+ * `capacityWaits` bookkeeping `waitingFor` is built from — so the prompt's "it queued for N" can
19
+ * never disagree with the panel's queue marker. `0`/absent means the node never waited for
20
+ * capacity, and the prompt then says nothing about a queue.
21
+ */
22
+ readonly waitedMs?: number;
23
+ }
24
+ /** One node the host should try to CONTINUE in the session it was interrupted in (a cold wake). */
25
+ export interface ResumeWorkerInput {
26
+ readonly node: NodeRecord;
27
+ /** The session id recorded in `lastWorkerId`: the address to deliver to, and the identity the
28
+ * adoption binds if the delivery is accepted. */
29
+ readonly workerId: string;
30
+ /**
31
+ * The capacity policy this candidate was selected under. The host hands it straight to
32
+ * `adoptContinuation`, so the adoption re-runs the plan's own judgement against the live load
33
+ * under the tree lock exactly as `dispatch` does — a continuation is a dispatch and must not bind
34
+ * past a machine that filled up between the plan and the adoption. It is always supplied by
35
+ * `MissionEngine.pass`; the parallel parked wake builds the same object through
36
+ * `MissionEngine.admissionPolicy()`, so both wake paths gate identically instead of one of them
37
+ * opting out — which is what let a wake oversubscribe the machine.
38
+ */
39
+ readonly capacity?: CapacityPolicy;
40
+ /** The capacity wait this candidate served before being selected; see {@link StartWorkerInput}. */
41
+ readonly waitedMs?: number;
42
+ }
43
+ /**
44
+ * What a continuation attempt resolved to. Three answers, not a boolean, because "nothing to do" is
45
+ * not "it failed": when another delivery already owns the session, starting a fresh executor on top
46
+ * of it is precisely the double run this guard exists to prevent.
47
+ *
48
+ * - `resumed` — the node is bound to `workerId` and the prompt was accepted; the caller counts one
49
+ * dispatch and starts nobody.
50
+ * - `failed` — the continuation is not going to happen: either the delivery was refused and the
51
+ * host has already undone its own adoption (`reclaim(..., 'wake-failed')`, which charges no
52
+ * budget), or the host declined it BEFORE adopting because the node drifted too far from what that
53
+ * session last read (a material change; see `continuation`). Either way the caller takes the
54
+ * ordinary fresh path with a newly reserved claim, which is the complete answer: a new executor
55
+ * reads every correction and every note.
56
+ * - `skip` — the node must be left exactly as it is: a wake for that session is in flight, the owner
57
+ * is not materialized, or the node's state moved under us. Never a reason to spawn.
58
+ */
59
+ export type ResumeOutcome = 'resumed' | 'failed' | 'skip';
60
+ export interface EngineHooks {
61
+ /** Reserve a child session id without creating it, which closes the spawn/bind window: the node is bound before anything is materialized. Pair with {@link releaseClaimId}: a claim reserved for a dispatch that is then refused was never bound and must be handed back, or it stays "live" as a ghost forever. */
62
+ reserveClaimId(): string;
63
+ /** Hand back a claim reserved by {@link reserveClaimId} whose dispatch did not happen; nothing was materialized for it. */
64
+ releaseClaimId(claimId: string): void;
65
+ /** Materialize the worker for a dispatched node and deliver its prompt; the prompt is built HERE from `tree.view(node.id)`, because a sibling result may have landed since the dispatch decision. */
66
+ startWorker(input: StartWorkerInput): Promise<void>;
67
+ /**
68
+ * Try to CONTINUE a node in the session it was interrupted in, instead of starting a fresh one
69
+ * (a cold wake). The host owns this because only it can reach `ctx.subagents`, and the delivery
70
+ * must be sent by the child's live direct parent (`authorizeLineage`).
71
+ *
72
+ * Called BEFORE a claim is reserved, so a continuation that lands consumes no claim id and a
73
+ * fallback reserves one exactly as any other dispatch does. See {@link ResumeOutcome} for the
74
+ * contract, including the rule that `skip` must never be answered with a spawn.
75
+ *
76
+ * Priority when several addresses apply — documented because it is a decision, not an accident:
77
+ * a parked session wins (it is an ALIVE continuation waiting to be woken), then this cold
78
+ * continuation, then a fresh spawn (`nextDispatchable` already excludes parked nodes).
79
+ */
80
+ resumeWorker?(input: ResumeWorkerInput): Promise<ResumeOutcome>;
81
+ /** Stop a worker believed to be stuck. */
82
+ interruptWorker(sessionId: string): Promise<void>;
83
+ /** Deliver a one-line signal to the tree owner; it carries no content — the guidance layer does. */
84
+ notifyOwner(rootId: string, reason: string): void;
85
+ /** Heads-up that one node keeps going silent, so the owner can decide whether to intervene — deliberately separate from {@link notifyOwner}, which means "a tree reached a terminal state". */
86
+ notifyStalled?(info: StallReport): void;
87
+ /**
88
+ * Diagnostic sink for a `hung` reclaim: the worker was still alive but had produced nothing for a
89
+ * whole window (or exceeded the round cap), so the engine interrupted it and re-queued the node
90
+ * WITHOUT charging any budget. Nothing here is a decision for the owner — an apparatus outage is
91
+ * not a mission failure — so this exists to be LOGGED, not to wake anybody.
92
+ */
93
+ notifyHung?(info: HungReport): void;
94
+ /**
95
+ * A dispatch was DEFERRED by the capacity gate (never refused). Rate-limited by the engine, one
96
+ * line per node per minute, so a queue waiting on a heavy mission does not flood the log while
97
+ * still remaining diagnosable. The host formats and logs it.
98
+ */
99
+ notifyDeferred?(info: DispatchDeferral): void;
100
+ /**
101
+ * Report nodes waiting for a parked session to be woken; the engine cannot wake anything itself, because a wake is a delivery through `ctx.subagents` that only the host owns and must be authorized by the child's live direct parent.
102
+ * Its whole job is to say a convergence pass is due on an idle session, as a BATCH: an owner that was offline can materialize with several parked nodes at once, and one wake per node would be a wake storm.
103
+ * The host performs the actual wake inside the owner's next turn; a host that cannot wake leaves the address on the node, and this reports again on the next pass.
104
+ */
105
+ notifyParkedReady?(nodes: readonly NodeRecord[]): void;
106
+ reportDispatchFailure?(nodeId: string, error: unknown): void;
107
+ trace?(message: string): void;
108
+ }
109
+ export interface StallReport {
110
+ readonly rootId: string;
111
+ readonly nodeId: string;
112
+ readonly title: string;
113
+ /** Dispatches so far, NOT a budget: successful rounds — including aggregate/convergence passes — raise it too. */
114
+ readonly attempts: number;
115
+ /** Failed executions so far, including the reclaim being reported; this is the counter the `failed` ceiling bounds. */
116
+ readonly failures: number;
117
+ /** Reclaims for silence so far, including the one being reported. */
118
+ readonly stalls: number;
119
+ readonly silentMs: number;
120
+ /**
121
+ * Which trouble this is. `silence` (the default, and what an older caller sees) is a stall; `hung`
122
+ * is the repeated-hang heads-up, which shares the channel, the {@link isTroubledNode} floor and the
123
+ * one-message-per-node marker precisely so the owner reads ONE vocabulary rather than two.
124
+ */
125
+ readonly cause?: 'silence' | 'hung';
126
+ /** Consecutive hangs at report time; present only for `cause: 'hung'`. */
127
+ readonly hungs?: number;
128
+ }
129
+ /** What a `hung` reclaim looked like, for the engine's diagnostic log. This per-reclaim report is
130
+ * NOT an owner-facing message by itself — one hung round is an apparatus event, not a decision —
131
+ * but a STREAK of them is (`isTroubledNode` reads `hungCount`), and that escalation travels through
132
+ * {@link StallReport} with `cause: 'hung'`. */
133
+ export interface HungReport {
134
+ readonly rootId: string;
135
+ readonly nodeId: string;
136
+ readonly title: string;
137
+ readonly attempts: number;
138
+ readonly failures: number;
139
+ readonly stalls: number;
140
+ /** Which bound fired: the `staleMs` output window, or the `roundMs` ceiling. */
141
+ readonly bound: Extract<LivenessBound, 'output' | 'round'>;
142
+ /** ms since the dispatch that opened this round. */
143
+ readonly ranMs: number;
144
+ /** ms since the worker last produced anything. */
145
+ readonly idleMs: number;
146
+ /** Consecutive hangs after this reclaim, including it — the streak `isTroubledNode` reads. */
147
+ readonly hungCount: number;
148
+ }
149
+ /** Why a dispatch is waiting, for the engine's rate-limited diagnostic log. This is NOT an
150
+ * `isTroubled` signal: ordinary queuing is not "keeps going wrong". */
151
+ export interface DispatchDeferral {
152
+ readonly nodeId: string;
153
+ readonly rootId: string;
154
+ readonly title: string;
155
+ readonly waitingFor: WaitingFor;
156
+ /** Weight of the deferred node (`capacity` reason). */
157
+ readonly needed: number;
158
+ readonly capacity: number;
159
+ readonly running: number;
160
+ /** How long the capacity gate has been deferring this node; `0` for a non-capacity reason. */
161
+ readonly waitedMs: number;
162
+ /** True when this node has aged into a reservation (no new admissions until it fits). */
163
+ readonly reserved: boolean;
164
+ }
165
+ export interface EngineOptions {
166
+ /**
167
+ * Backstop on the NUMBER of running units: with `capacity` it makes up the two-part admission
168
+ * rule. Capacity is the MASTER gate (how much work the machine can carry); this only stops a flood
169
+ * of weight-1 missions from opening more sessions than anyone wants. Both must admit a candidate.
170
+ */
171
+ readonly maxConcurrent: number;
172
+ /**
173
+ * Master gate: how many capacity units (cores-equivalent, the same unit as `NodeRecord.weight`)
174
+ * may run at once. Required, because the core must not guess at the hardware — the host derives it
175
+ * (configured → `os.availableParallelism()` → `os.cpus().length` → 4, minus one reserved core) and
176
+ * injects the number, which is also what makes every scheduling decision deterministic in tests.
177
+ */
178
+ readonly capacity: number;
179
+ /**
180
+ * How long a node may be repeatedly deferred by the capacity gate before it RESERVES the machine
181
+ * (no new admissions until it fits). See `capacity.ts` for the value and its rationale.
182
+ */
183
+ readonly capacityWaitMs: number;
184
+ /**
185
+ * Free-memory floor, in bytes, enforced through {@link ResourceProbe.memoryBudget}: below it the
186
+ * gate DEFERS dispatch (never refuses). `0` disables the gate; the probe itself is optional, and a
187
+ * probe that does not implement `memoryBudget` leaves this gate inactive (the v1 degradation path).
188
+ */
189
+ readonly minFreeMemoryBytes: number;
190
+ /** Machine reader, injected by the host. Absent means "no resource signal": the memory floor is
191
+ * then the only gate this could arm, and with no probe it is inactive. */
192
+ readonly probe?: ResourceProbe;
193
+ /** How long a worker may produce NOTHING before it is considered stuck (see `liveness.ts`). */
194
+ readonly staleMs: number;
195
+ /** Wall-clock ceiling on one dispatch: past this the node is reclaimed as `hung` no matter how
196
+ * many events refreshed its timestamps. The backstop against a transport that retries forever. */
197
+ readonly roundMs: number;
198
+ /** Clock for the stale check and the capacity aging window, injectable so tests can move time
199
+ * instead of waiting for it; production leaves it at `Date.now`. */
200
+ readonly now?: () => number;
201
+ }
202
+ /**
203
+ * Advisory parallelism for a core that has no Node API of its own (the core builds without Node
204
+ * types, so the real machine reading is the HOST's job — see `resource.ts` and the plugin's probe).
205
+ *
206
+ * The lookup order is deliberate: `navigator.hardwareConcurrency` when a DOM-shaped global is
207
+ * present, then the family's old fallback of 4. An earlier build tried
208
+ * `process.availableParallelism?.()` here and that was DEAD CODE: Node has never exposed that member
209
+ * on `process` — the real API is `os.availableParallelism()`, which the host now calls. Core callers
210
+ * that need the machine number must inject `EngineOptions.capacity`; this function only exists so
211
+ * `DEFAULT_ENGINE_OPTIONS` (and any non-Node embedder) has a sane advisory value.
212
+ */
213
+ export declare function detectConcurrency(): number;
214
+ export declare const DEFAULT_ENGINE_OPTIONS: EngineOptions;
215
+ export declare class MissionEngine {
216
+ private readonly tree;
217
+ private readonly hooks;
218
+ private readonly options;
219
+ private pumping;
220
+ private pumpRequested;
221
+ /** Node id → when the capacity gate first deferred it. In-memory on purpose: a restart loses the
222
+ * aging clock, which re-arms it (one extra wait window) — never a correctness loss. */
223
+ private readonly capacityWaits;
224
+ /** The live `waitingFor` projection, refreshed at the end of every pass: node id → reason, or
225
+ * absent for a node nothing is holding back. */
226
+ private readonly waiting;
227
+ /** Last time each node's deferral was logged, for the rate limit. */
228
+ private readonly deferredLogged;
229
+ /** Nodes whose RESERVATION transition has already been logged. A reservation overrides the rate
230
+ * limit: it changes what the engine admits, so it must not be swallowed by a recent deferral line. */
231
+ private readonly reservedLogged;
232
+ constructor(tree: MissionTree, hooks: EngineHooks, options?: EngineOptions);
233
+ /** Why one node has not been dispatched, or `null` when nothing is holding it back. The projection
234
+ * is read by the host for `mission_result` and the panel; it is live state, never persisted. */
235
+ waitingFor(nodeId: string): WaitingFor | null;
236
+ /**
237
+ * The admission policy THIS moment would gate on, from the engine's own single source: the same
238
+ * `capacityPolicy(memoryFloor())` the dispatch loop builds, carrying the engine's own aging clock
239
+ * (`capacityWaits`) and the live machine-wide memory block. Built, never stored: the caller gets a
240
+ * snapshot for one decision, exactly like a pass's policy object.
241
+ *
242
+ * It exists for the WAKE paths. `adoptParked` / `adoptContinuation` re-run the gate under the tree
243
+ * lock, and the rule is that a wake must be judged by the very policy the dispatch loop would use —
244
+ * deriving a second one in the host (its own capacity number, its own clock) is how the two drift.
245
+ * The continuation wake already receives the pass's policy through `resumeWorker`; this is how the
246
+ * parked wake, which has no plan snapshot of its own, gets the same one.
247
+ */
248
+ admissionPolicy(): CapacityPolicy;
249
+ /** The capacity the engine is actually gating on, for diagnostics and tests. */
250
+ capacity(): {
251
+ capacity: number;
252
+ maxConcurrent: number;
253
+ runningCount: number;
254
+ runningWeight: number;
255
+ };
256
+ private now;
257
+ /** Run one dispatch pass; `dispatch()` decides and marks in the same await, so a second pass cannot see the same node as available and no batch bookkeeping is needed. */
258
+ pump(): Promise<number>;
259
+ sweep(): Promise<{
260
+ reclaimed: number;
261
+ dispatched: number;
262
+ }>;
263
+ /**
264
+ * Reclaim nodes whose worker is gone or stuck; liveness is the primary signal, and the windows
265
+ * only cover "alive but not useful".
266
+ *
267
+ * The judgement is `judgeWorker` (see `liveness.ts`): silence measured from the last event at all
268
+ * is `stalled` and charges the failure budget exactly as before; a worker that keeps being heard
269
+ * from while producing NOTHING for `staleMs`, or that exceeds the round cap, is `hung` and charges
270
+ * nothing. Both interrupt the live binding before reclaiming, and both go through the same
271
+ * compare-and-swap `MissionTree.reclaim`, so a verdict judged against a snapshot taken outside
272
+ * the tree lock cannot strip a fresh binding or charge one attempt twice.
273
+ */
274
+ reclaimStale(): Promise<number>;
275
+ /**
276
+ * The ONE owner-facing "this node keeps going wrong" path, shared by every cause. Three rules make
277
+ * it one vocabulary rather than two: the gate is {@link isTroubledNode} (the same predicate the
278
+ * `isTroubled` flag reads), the report rides the `notifyStalled` channel (a heads-up, never a
279
+ * question), and the durable `claimStallReport` marker keeps it to one message per node whichever
280
+ * way the node is failing. The engine already recovered on its own in every case, so waiting for a
281
+ * repeat is what keeps a permanently flaky node from turning a sweep into a wake storm.
282
+ */
283
+ private escalateTrouble;
284
+ /** Tell the host a node was reclaimed as `hung`, so every hung round is diagnosable — a provider
285
+ * that hangs (or only retries) every dispatch must never be invisible, which is exactly how W8
286
+ * stayed unnoticed for 7.5 hours. This per-reclaim line is NOT the owner escalation: one hung
287
+ * round is an apparatus event the engine recovered from, and only a STREAK crosses
288
+ * {@link escalateTrouble}'s `isTroubledNode` floor. */
289
+ private reportHung;
290
+ /** Destroy trees whose owner session no longer exists: the owner is the authority over the tree, and a hot reload or host restart leaves it resolvable, so nothing is destroyed there. A tree whose owner merely cannot be OBSERVED is left alone and reported instead — the host being unable to answer is not evidence the owner is gone, and a destroyed tree cannot be recovered. */
291
+ reconcileOrphans(): Promise<readonly string[]>;
292
+ /**
293
+ * The machine-wide gates that do not depend on WHICH node is being considered: the free-memory
294
+ * floor. Returns the `waitingFor` every candidate carries while it is set, or `undefined` when the
295
+ * gate is open.
296
+ *
297
+ * A probe that does not implement `memoryBudget` leaves the gate inactive (v1: no platform
298
+ * adapter). A probe that implements it and answers `null` means "cannot confirm the floor", and
299
+ * that is a DEFER, not "plenty" — see `resource.ts`: no signal never relaxes scheduling.
300
+ */
301
+ private memoryFloor;
302
+ /** The capacity arithmetic for this moment, built from the tree's live load. */
303
+ private capacityPolicy;
304
+ private pass;
305
+ /**
306
+ * Publish the live `waitingFor` projection for the nodes a pass left behind, and drop entries for
307
+ * nodes that are no longer waiting (dispatched, terminal, or blocked by something outside
308
+ * admission). The map is the ONE place the host reads waiting state from, so it is rebuilt to
309
+ * exactly the deferred set rather than accumulated across passes.
310
+ */
311
+ private publishWaiting;
312
+ /**
313
+ * The rate-limited deferral log: one line per node per minute while the capacity gate holds it
314
+ * back. Deliberately not an owner wake and not an `isTroubled` fact — normal queuing is not a
315
+ * problem the owner can act on.
316
+ *
317
+ * Two reasons count as "waiting for capacity" here: `capacity` itself (this node does not fit) and
318
+ * `aging` (ANOTHER node has reserved the machine, so this one is held back while capacity may have
319
+ * room). The second is the one the aging mechanism exists to make visible — leaving it out is how
320
+ * the reservation changed what the engine admits with no line anywhere saying so.
321
+ */
322
+ private reportDeferrals;
323
+ /** Report the parked-ready batch to the host after the dispatch loop; the wake itself is the host's job — see {@link EngineHooks.notifyParkedReady}. */
324
+ private reportParkedReady;
325
+ /**
326
+ * Wake the owner about every terminal tree, exactly once each; dedup lives on the durable tree record (`claimReport`), because an in-memory set is empty after a restart and would re-report every tree that had ever reached a terminal state.
327
+ */
328
+ private reportTerminalRoots;
329
+ /** Forget the report marker for a tree, so a re-rooted tree can report again. */
330
+ forgetReport(rootId: string): void;
331
+ }
@@ -0,0 +1,25 @@
1
+ /**
2
+ * @avantf/mission-core — the mission-tree model, dispatch loop and prompt construction; harness environment facts arrive as injected dependencies.
3
+ * @module @avantf/mission-core
4
+ */
5
+ export * from './types.js';
6
+ export { MissionTree, defaultNewId } from './tree.js';
7
+ export type { TreeState, TreeStore, TreeDeps, SpilledText, CreateRootInput, OwnerProbe, OrphanedTree, } from './tree.js';
8
+ export { MissionEngine, DEFAULT_ENGINE_OPTIONS, detectConcurrency } from './engine.js';
9
+ export type { DispatchDeferral, EngineHooks, EngineOptions, HungReport, ResumeOutcome, ResumeWorkerInput, StartWorkerInput, StallReport, } from './engine.js';
10
+ export { CAPACITY_CEILING, CAPACITY_FALLBACK_PARALLELISM, DEFAULT_CAPACITY_WAIT_MS, DEFAULT_MIN_FREE_MEMORY_BYTES, DEFAULT_WEIGHT, MAX_WEIGHT, MIN_CAPACITY_WAIT_MS, MIN_WEIGHT, RESERVED_CORES, clampCapacity, deriveCapacity, normalizeWeight, } from './capacity.js';
11
+ export type { CapacityInput, CapacityReading, CapacitySource } from './capacity.js';
12
+ export type { ResourceProbe } from './resources.js';
13
+ export { effectiveRoundMs, heardAt, judgeWorker, MAX_DECLARED_ROUND_MS, normalizeRoundMs, producedAt, resolveChildRoundMs, storedTime } from './liveness.js';
14
+ export type { LivenessBound, LivenessVerdict, LivenessWindows, ReclaimCause } from './liveness.js';
15
+ export { byCreatedAtThenId, capacityWaitingFor, heldUnits, normalizeUnit, planDispatch, resolveChildUnit, resolveChildWeight, selectNextDispatchable, spawnBackoffMs, unitHolder, } from './dispatch.js';
16
+ export type { CapacityLoad, CapacityPolicy, DeferredCandidate, DispatchPlan, DispatchPolicy, DispatchScope, } from './dispatch.js';
17
+ export { computeContinuationDelta, isMaterialChange, nodeFingerprint } from './continuation.js';
18
+ export type { ContinuationDelta } from './continuation.js';
19
+ export { queueMs, runMs, totalMs } from './timing.js';
20
+ export type { NodeTiming } from './timing.js';
21
+ export { buildWorkerPrompt, buildProgressLine, isTroubled, PROMPT_LIMITS, spillPointer, waitedLabel } from './prompt.js';
22
+ export type { WorkerPromptOptions } from './prompt.js';
23
+ export { isTroubledNode, statusLabel } from './trouble.js';
24
+ export { LOCAL_WELL_FORMED, wellFormedDeep, wellFormedText } from './wellformed.js';
25
+ export type { WellFormedSource } from './wellformed.js';
@@ -0,0 +1,99 @@
1
+ /**
2
+ * The liveness verdict for ONE running node: is its worker still making progress, or has the round
3
+ * become something the engine must take back?
4
+ *
5
+ * Two clocks, because "no events" and "no output" are different failures — and conflating them is
6
+ * what let a transport-layer hang sit `running` for 7.5 hours (2026-10-02):
7
+ *
8
+ * - `progressAt` — the last PRODUCTION: the model's committed output (`assistant/message`), a tool
9
+ * it asked for (`tool/call`), or a tool that finished (`tool/result`). The plugin classifies the
10
+ * host's durable event feed (see its `workerEvents` module) and only those refresh it.
11
+ * - `activityAt` — the last durable event of ANY kind, including transport-layer noise: provider
12
+ * retry attempts (`assistant/attempt`) and the log-only per-request route snapshots
13
+ * (`request/header`, `request/context`). Noise proves the session object is alive; it does not
14
+ * prove the mission moved.
15
+ *
16
+ * Three bounds come out of that, in this order:
17
+ *
18
+ * - **silence** (`now - max(activityAt, progressAt, claimedAt) > staleMs`) → `stalled`. Nothing was
19
+ * heard at all. This is the original verdict with its original budget charge, and it is checked
20
+ * first so a node the old build would have called silent is still called silent.
21
+ * - **round** (`now - claimedAt > roundMs`) → `hung`. A wall-clock ceiling on the dispatch itself,
22
+ * which no timestamp can veto. It fires before the output bound on purpose: when a productive
23
+ * worker reaches it, "the round hit its ceiling" is the accurate description, not "no output".
24
+ * - **output** (activity fresh but `now - (progressAt || claimedAt) > staleMs`) → `hung`. The W8
25
+ * shape exactly: the worker keeps being heard from while producing nothing.
26
+ *
27
+ * Both `hung` bounds are reclaimed WITHOUT charging the failure budget: a stuck provider is not a
28
+ * failed mission, and letting retries eat the budget would turn a recoverable node `failed`. What
29
+ * keeps that leniency from becoming an unbounded loop is `hungCount` — the engine's own consecutive
30
+ * hang counter, cleared by real output — which at `CAPACITY.maxHungsBeforeReport` makes the node
31
+ * visible to the owner through the same trouble channel `stalled` uses (see `MissionTree.reclaim`
32
+ * and `MissionEngine.reclaimStale`).
33
+ *
34
+ * A record written before `activityAt` existed has no value there, and its `progressAt` was
35
+ * refreshed by ANY event. {@link heardAt}/{@link producedAt} then fall back to `progressAt` for
36
+ * both clocks, so such a node is judged exactly as the previous build judged it — plus the round
37
+ * cap, which is the one bound that still catches the old records.
38
+ *
39
+ * @module @avantf/mission-core/liveness
40
+ */
41
+ import type { NodeRecord } from './types.js';
42
+ /** Why a running node's worker must be taken back. `stalled` charges the failure budget; `hung` does not. */
43
+ export type ReclaimCause = 'stalled' | 'hung';
44
+ /** Which bound produced a `hung` (or `stalled`) verdict. */
45
+ export type LivenessBound = 'silence' | 'output' | 'round';
46
+ export interface LivenessVerdict {
47
+ readonly cause: ReclaimCause;
48
+ readonly bound: LivenessBound;
49
+ /** ms since the worker was last heard from at all. */
50
+ readonly silentMs: number;
51
+ /** ms since the worker last PRODUCED something. */
52
+ readonly idleMs: number;
53
+ /** ms since the dispatch that opened this round. */
54
+ readonly ranMs: number;
55
+ }
56
+ export interface LivenessWindows {
57
+ readonly staleMs: number;
58
+ readonly roundMs: number;
59
+ }
60
+ /**
61
+ * Hard ceiling on a DECLARED round cap: 24 hours. The declaration exists to relax the cap for
62
+ * genuinely heavy work, not to opt out of it — a single round longer than a day is indistinguishable
63
+ * from a hang, and the mission can always be re-entered by the ordinary retry. An operator's own
64
+ * configured `roundMs` is NOT clamped by this; only a node's declaration is.
65
+ */
66
+ export declare const MAX_DECLARED_ROUND_MS: number;
67
+ /**
68
+ * Read a node's declared round-cap relaxation into shape: a finite number > 0, or `null` for
69
+ * "declared nothing". Missing, dirty and non-positive values all read as `null`, the same direction
70
+ * as a record written before the field existed — a declaration that cannot be believed must not
71
+ * change when a round is taken back. Anything past {@link MAX_DECLARED_ROUND_MS} falls to it.
72
+ */
73
+ export declare function normalizeRoundMs(raw: unknown): number | null;
74
+ /**
75
+ * A decomposed child's round-cap relaxation. `undefined` — the spec said nothing — does NOT inherit
76
+ * the parent's declaration: a parent that needs hours says nothing about one child, and the same
77
+ * reasoning already governs `weight` (`resolveChildWeight`).
78
+ */
79
+ export declare function resolveChildRoundMs(declared: number | null | undefined): number | null;
80
+ /**
81
+ * The round cap actually applied to one node: the engine's configured ceiling, RELAXED by the node's
82
+ * own declaration and never shortened by it. Relaxation-only is the whole rule — the cap is the
83
+ * backstop against a transport that retries forever, so a mission may ask for more room but may not
84
+ * opt out of the backstop, and a smaller declaration (a countdown a model could otherwise set for
85
+ * itself) is a no-op.
86
+ */
87
+ export declare function effectiveRoundMs(node: NodeRecord, configured: number): number;
88
+ /** A timestamp as stored. A record written before the field existed has `undefined` at runtime even
89
+ * though the type says `number`, and `Math.max(undefined, …)` is `NaN` — which would compare false
90
+ * against every window and keep a dead node running forever. */
91
+ export declare function storedTime(value: number | undefined): number;
92
+ /** When the worker was last heard from at all; the dispatch that opened the round is the floor, so
93
+ * a node that never reported anything is "silent since it started" rather than "silent since 0". */
94
+ export declare function heardAt(node: NodeRecord): number;
95
+ /** When the worker last PRODUCED something. Falls back to the dispatch: a node that has produced
96
+ * nothing yet has been unproductive since it was bound, which is the reading `hung` needs. */
97
+ export declare function producedAt(node: NodeRecord): number;
98
+ /** The verdict for one running node, or `undefined` when it is still doing fine. */
99
+ export declare function judgeWorker(node: NodeRecord, at: number, windows: LivenessWindows): LivenessVerdict | undefined;
@@ -0,0 +1,112 @@
1
+ /**
2
+ * Worker prompt construction: one template, branching on the node's state for the trailing section; the tail is derived from the node at dispatch time, exists only in this dispatch's input and never enters a conversation.
3
+ * @module @avantf/mission-core/prompt
4
+ */
5
+ import { type DispatchView, type NodeRecord } from './types.js';
6
+ import type { ContinuationDelta } from './continuation.js';
7
+ import { type WellFormedSource } from './wellformed.js';
8
+ /**
9
+ * Whether a mission should read as TROUBLED to its owner: some unfinished node trips {@link isTroubledNode}.
10
+ * The wording has to say this is HISTORY, because a flag that never clears would invite the owner to steer or cancel a mission that is running fine, and both actions are destructive.
11
+ * Deliberately NOT here: which mission it was, how often, and how much budget is left — those belong to the engine's retry policy, and the owner's two actions (`adjust_mission`, `cancel_mission`) take the whole mission, not a node inside it.
12
+ */
13
+ export declare function isTroubled(nodes: readonly NodeRecord[]): boolean;
14
+ /**
15
+ * The prompt's size guardrails. The tail is rewritten on every dispatch and everything here is paid
16
+ * for in model tokens, so each of these bounds one axis that could grow without limit:
17
+ *
18
+ * - {@link chainLine} rendered EVERY correction of EVERY ancestor (a correction is written on the
19
+ * root, so one long-lived root steered repeatedly made every descendant carry all of them);
20
+ * - {@link analysisSection} rendered every note an executor had ever appended;
21
+ * - {@link childrenBlock} rendered every terminal child's conclusion. Per-child size is already
22
+ * bounded by the engine (`CAPACITY.maxInlineResultChars`, longer results are spilled with a
23
+ * pointer), but the COUNT is not: `children` accumulates across re-decompositions, and 50
24
+ * children × the 2 KB inline cap was a measured ~100 KB (~50k tokens) aggregate prompt.
25
+ *
26
+ * What the model still sees after a bound bites is stated IN the prompt at each site, so a judge can
27
+ * tell "this child said nothing" from "this child's text was summarized here".
28
+ */
29
+ export declare const PROMPT_LIMITS: {
30
+ /** Corrections rendered per mission-chain ancestor (the NEWEST ones; the rest are counted). */
31
+ readonly chainCorrections: 2;
32
+ readonly chainCorrectionChars: 200;
33
+ /** Corrections rendered verbatim on the node being executed (the newest ones; the rest counted). */
34
+ readonly currentCorrections: 10;
35
+ /** `note_mission` notes rendered (the newest ones; the rest counted). */
36
+ readonly analysisNotes: 8;
37
+ readonly analysisNoteChars: 400;
38
+ /** Terminal children whose conclusion is rendered in full — the newest ones. Older children still
39
+ * list id, title, status and the head of their conclusion. The count is what accumulates across
40
+ * re-decompositions, so this is the bound that stops a 50-child aggregate from growing without end. */
41
+ readonly childrenBody: 10;
42
+ /** Safety clip on one child's conclusion; equal to the engine's inline cap, so a well-formed record
43
+ * is never actually cut (an older/foreign record could exceed it). */
44
+ readonly childBodyChars: 2000;
45
+ /** How much of an older child's conclusion survives in the compact form. */
46
+ readonly childExcerptChars: 120;
47
+ };
48
+ /**
49
+ * Where a spilled full result lives, with the backend's own retrieval guidance; one definition for both renderers — the worker prompt and `mission_result` — because a locator without its hint is a pointer nobody can follow.
50
+ */
51
+ export declare function spillPointer(node: NodeRecord): string;
52
+ /** How one dispatch renders the node it is about; the default is the fresh-executor reading. */
53
+ export interface WorkerPromptOptions {
54
+ /** This dispatch CONTINUES the session that was interrupted: the hand-off notice says which
55
+ * execution this is and points at the resumed session's own history. */
56
+ readonly resumed?: boolean;
57
+ /** Corrections to render on the CURRENT node. Defaults to every recorded correction, because a
58
+ * fresh executor has read none. A continuation wake passes
59
+ * `node.corrections.slice(node.correctionsDeliveredUpTo)`: a correction already delivered to that
60
+ * same session must not be argued to it twice. The chain's ancestor lines keep their own text —
61
+ * they are a different node's corrections, and none of them is repeated by this override. */
62
+ readonly corrections?: readonly string[];
63
+ /** What changed on this node since the session's own last prompt, for the delta section. Passed
64
+ * by a CONTINUATION only: a fresh executor has never executed this mission, so "since you last
65
+ * executed" would be a lie, and it reads every correction, note and child conclusion anyway. */
66
+ readonly delta?: ContinuationDelta;
67
+ /** How long the CAPACITY gate deferred this node before the dispatch that built this prompt
68
+ * picked it up, in ms. Passed by the engine (through `StartWorkerInput`/`ResumeWorkerInput`) from
69
+ * its own aging clock; `0`/absent means "never waited", and then no queue line is rendered at all.
70
+ * It exists because an executor with no such fact reports "I did not wait" — a real answer from a
71
+ * run that had in fact queued for two minutes behind a full machine. */
72
+ readonly capacityWaitedMs?: number;
73
+ }
74
+ /**
75
+ * How a capacity wait reads in the prompt. A stopwatch would be false precision: the number is one
76
+ * reading of the engine's aging clock, so it is rounded and says 约. Seconds below a minute (a node
77
+ * can be skipped for a few seconds and that is still "it queued"), minutes above it.
78
+ */
79
+ export declare function waitedLabel(ms: number): string;
80
+ /**
81
+ * Build the complete prompt for one dispatch; the mission chain carries only titles and one-line context, so its size is bounded by the depth limit, and the full `description`/`context` is included for the current node only.
82
+ * The tail branches on the CHILDREN in the view, not on the node's status: the prompt is built after `dispatch()` has already marked the node `running`, so a status test can never see the aggregate case, and a `failed` node is never dispatched at all.
83
+ *
84
+ * ── the OUTBOUND well-formed boundary ──
85
+ * Everything this function writes lands in the executor's context, and the tree it reads may hold a
86
+ * lone surrogate that an OLDER build persisted (before the inbound funnel existed) or that a foreign
87
+ * writer put there. So the whole view (node, chain, children) and the whole options record are
88
+ * recursively repaired BEFORE a single line is assembled: `wellFormedDeep` leaves numbers, booleans
89
+ * and `null` untouched, so nothing but lone surrogates can change. `wellFormed` is the plugin's
90
+ * loaded base kit when it has one and {@link LOCAL_WELL_FORMED} otherwise.
91
+ *
92
+ * @param view - the node/chain/children this dispatch renders.
93
+ * @param options - how this dispatch differs from a fresh executor's.
94
+ * @param wellFormed - the repair pair; the plugin injects the loaded base's.
95
+ */
96
+ export declare function buildWorkerPrompt(view: DispatchView, options?: WorkerPromptOptions, wellFormed?: WellFormedSource): string;
97
+ /**
98
+ * The compact per-tree progress line the guidance layer carries; every clause is anchored so it stays true when read late — the counts are a statement about a moment, not about "now".
99
+ * Granularity is deliberate: the owner is told WHETHER mission is still running and WHETHER any of it has a history of trouble, never how the mission is distributed across states, because no owner action depends on that distribution, so a per-state breakdown would only invite it to reason about a layer it cannot touch.
100
+ * Trouble is the one fact that changes what it should do, and it is counted in WORKS, not in nodes, so the size of anything stays inside the engine.
101
+ *
102
+ * The SAME outbound rule as {@link buildWorkerPrompt}: this line is assembled into the owner's
103
+ * context, so titles are repaired first (see the note there). The plugin injects the loaded base's
104
+ * repair pair; the core's local copy is the degradation path.
105
+ */
106
+ export declare function buildProgressLine(input: {
107
+ readonly roots: readonly NodeRecord[];
108
+ /** How many missions have not finished yet. */
109
+ readonly ongoing: number;
110
+ /** Whether any unfinished mission carries a history of stalls or failures at the engine's floors. */
111
+ readonly troubled: boolean;
112
+ }, wellFormed?: WellFormedSource): string;
@@ -0,0 +1,57 @@
1
+ /**
2
+ * `ResourceProbe` — the port through which the engine reads the machine, and the one place the
3
+ * family's `null`-is-not-zero rule is stated. Node-free: the core declares the interface and never
4
+ * implements it; the plugin/host supplies a probe built on `node:os` (see `host.ts`), which is also
5
+ * the ONLY seam allowed to contain platform branches.
6
+ *
7
+ * ## `null` is a first-class answer: "no signal on this platform"
8
+ *
9
+ * **"No signal" ≠ "idle".** A `null` must never be read as evidence that resources are free, so it
10
+ * may only ever make scheduling MORE conservative — never more permissive. Concretely:
11
+ *
12
+ * - `memoryBudget()` returning `null` while the free-memory floor is armed means "cannot confirm the
13
+ * floor", and the gate DEFERS dispatch. It is not "plenty of memory".
14
+ * - `pressure()` returning `null` means "cannot tell"; no caller may treat it as `0` to widen
15
+ * concurrency or lower a weight. v1 has no pressure-driven widening at all, so the honest reading
16
+ * is "no adjustment".
17
+ * - `workerUsage()` returning `null` means "this process's CPU is unknown"; it is not `0`, and v2's
18
+ * correction loop must treat it as "do not correct from this sample".
19
+ *
20
+ * The distinction between an OPTIONAL member that is absent and one that returns `null` is
21
+ * deliberate: absent = this port has no implementation for that signal at all (the v1 degradation
22
+ * path — no platform adapter), while `null` = implemented but the platform will not say. The former
23
+ * leaves the corresponding gate inactive; the latter keeps the gate conservative.
24
+ *
25
+ * ## v1 scope
26
+ *
27
+ * Only the interface, the wiring, and a host implementation over the UNIFIED Node API
28
+ * (`os.availableParallelism()`, `os.totalmem()`, `process.constrainedMemory()`, `os.freemem()`).
29
+ * Platform adapters (POSIX/Windows `pressure`, per-worker process attribution) and measured
30
+ * correction of weights are v2; every unimplemented branch answers `null`.
31
+ *
32
+ * @module @avantf/mission-core/resources
33
+ */
34
+ export interface ResourceProbe {
35
+ /**
36
+ * How much independent work this machine will ACTUALLY run in parallel — i.e.
37
+ * `os.availableParallelism()`, which honours cgroup CPU quotas and CPU affinity, rather than the
38
+ * raw core count. Required, because every host can answer it: Node falls back to `os.cpus()`.
39
+ */
40
+ parallelism(): number;
41
+ /**
42
+ * A LOWER BOUND, in bytes, on the memory this process may still use, or `null` when the platform
43
+ * will not say. "Lower bound" is the whole contract: the value may understate what is available,
44
+ * which is the safe direction for a floor gate, and it includes the cgroup cap when one exists
45
+ * (a cgroup limit below `os.totalmem()` is the real budget). Never an exact figure, never "free
46
+ * memory" in the leaving-room sense.
47
+ */
48
+ memoryBudget?(): number | null;
49
+ /**
50
+ * A coarse 0..1 pressure reading (sustained CPU contention / memory pressure), or `null` when the
51
+ * platform cannot produce one. v1 never widens from it; the member exists so v2's adapters and the
52
+ * scheduler can be written against a stable signature.
53
+ */
54
+ pressure(): number | null;
55
+ /** CPU usage of one worker process, or `null` when this platform cannot attribute it (v2). */
56
+ workerUsage?(pid: number): number | null;
57
+ }