@avantf/dsh-mission 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +249 -0
- package/cordis.patch.yml +13 -0
- package/lib/client.js +225 -0
- package/lib/dsh-build.json +16 -0
- package/lib/envinit-bootstrap.js +334 -0
- package/lib/index.js +7021 -0
- package/lib/interface-version.json +4 -0
- package/lib/mission-core/capacity.d.ts +100 -0
- package/lib/mission-core/continuation.d.ts +106 -0
- package/lib/mission-core/dispatch.d.ts +165 -0
- package/lib/mission-core/engine.d.ts +331 -0
- package/lib/mission-core/index.d.ts +25 -0
- package/lib/mission-core/liveness.d.ts +99 -0
- package/lib/mission-core/prompt.d.ts +112 -0
- package/lib/mission-core/resources.d.ts +57 -0
- package/lib/mission-core/timing.d.ts +37 -0
- package/lib/mission-core/tree.d.ts +447 -0
- package/lib/mission-core/trouble.d.ts +40 -0
- package/lib/mission-core/types.d.ts +409 -0
- package/lib/mission-core/wellformed.d.ts +98 -0
- package/lib/types/claims.d.ts +3 -0
- package/lib/types/claims.d.ts.map +1 -0
- package/lib/types/client/MissionTreeView.d.ts +247 -0
- package/lib/types/client/MissionTreeView.d.ts.map +1 -0
- package/lib/types/client/api.d.ts +184 -0
- package/lib/types/client/api.d.ts.map +1 -0
- package/lib/types/client/contract.d.ts +209 -0
- package/lib/types/client/contract.d.ts.map +1 -0
- package/lib/types/client/index.d.ts +48 -0
- package/lib/types/client/index.d.ts.map +1 -0
- package/lib/types/client/seat.d.ts +26 -0
- package/lib/types/client/seat.d.ts.map +1 -0
- package/lib/types/client/styles.d.ts +6 -0
- package/lib/types/client/styles.d.ts.map +1 -0
- package/lib/types/coldResume.d.ts +123 -0
- package/lib/types/coldResume.d.ts.map +1 -0
- package/lib/types/domain.d.ts +93 -0
- package/lib/types/domain.d.ts.map +1 -0
- package/lib/types/envinit.d.ts +126 -0
- package/lib/types/envinit.d.ts.map +1 -0
- package/lib/types/executorSession.d.ts +129 -0
- package/lib/types/executorSession.d.ts.map +1 -0
- package/lib/types/faces.d.ts +30 -0
- package/lib/types/faces.d.ts.map +1 -0
- package/lib/types/host.d.ts +752 -0
- package/lib/types/host.d.ts.map +1 -0
- package/lib/types/index.d.ts +48 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/interface_gate.d.ts +51 -0
- package/lib/types/interface_gate.d.ts.map +1 -0
- package/lib/types/log.d.ts +18 -0
- package/lib/types/log.d.ts.map +1 -0
- package/lib/types/projectionCache.d.ts +22 -0
- package/lib/types/projectionCache.d.ts.map +1 -0
- package/lib/types/prompt.d.ts +106 -0
- package/lib/types/prompt.d.ts.map +1 -0
- package/lib/types/source.d.ts +22 -0
- package/lib/types/source.d.ts.map +1 -0
- package/lib/types/store.d.ts +20 -0
- package/lib/types/store.d.ts.map +1 -0
- package/lib/types/timeFormat.d.ts +56 -0
- package/lib/types/timeFormat.d.ts.map +1 -0
- package/lib/types/tools.d.ts +34 -0
- package/lib/types/tools.d.ts.map +1 -0
- package/lib/types/wellformed.d.ts +43 -0
- package/lib/types/wellformed.d.ts.map +1 -0
- package/lib/types/wire.d.ts +178 -0
- package/lib/types/wire.d.ts.map +1 -0
- package/lib/types/workerEvents.d.ts +36 -0
- package/lib/types/workerEvents.d.ts.map +1 -0
- package/lib/types/workerSessions.d.ts +256 -0
- package/lib/types/workerSessions.d.ts.map +1 -0
- package/package.json +145 -0
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The dispatch loop: pure host code that scans the tree, dispatches mission units and reclaims nodes whose worker vanished; it never calls a model.
|
|
3
|
+
* @module @avantf/mission-core/engine
|
|
4
|
+
*/
|
|
5
|
+
import { MissionTree } from './tree.js';
|
|
6
|
+
import { type LivenessBound } from './liveness.js';
|
|
7
|
+
import type { CapacityPolicy } from './dispatch.js';
|
|
8
|
+
import type { ResourceProbe } from './resources.js';
|
|
9
|
+
import { type NodeRecord, type WaitingFor } from './types.js';
|
|
10
|
+
/** Resume a worker for one node; the engine awaits only the reservation, not the worker. */
|
|
11
|
+
export interface StartWorkerInput {
|
|
12
|
+
readonly node: NodeRecord;
|
|
13
|
+
/** The session id reserved for this dispatch, already bound on the node. */
|
|
14
|
+
readonly claimId: string;
|
|
15
|
+
/**
|
|
16
|
+
* How long the CAPACITY gate deferred this node before this dispatch picked it up, in ms. Read
|
|
17
|
+
* from the engine's own aging clock at the moment the candidate was selected — the same
|
|
18
|
+
* `capacityWaits` bookkeeping `waitingFor` is built from — so the prompt's "it queued for N" can
|
|
19
|
+
* never disagree with the panel's queue marker. `0`/absent means the node never waited for
|
|
20
|
+
* capacity, and the prompt then says nothing about a queue.
|
|
21
|
+
*/
|
|
22
|
+
readonly waitedMs?: number;
|
|
23
|
+
}
|
|
24
|
+
/** One node the host should try to CONTINUE in the session it was interrupted in (a cold wake). */
|
|
25
|
+
export interface ResumeWorkerInput {
|
|
26
|
+
readonly node: NodeRecord;
|
|
27
|
+
/** The session id recorded in `lastWorkerId`: the address to deliver to, and the identity the
|
|
28
|
+
* adoption binds if the delivery is accepted. */
|
|
29
|
+
readonly workerId: string;
|
|
30
|
+
/**
|
|
31
|
+
* The capacity policy this candidate was selected under. The host hands it straight to
|
|
32
|
+
* `adoptContinuation`, so the adoption re-runs the plan's own judgement against the live load
|
|
33
|
+
* under the tree lock exactly as `dispatch` does — a continuation is a dispatch and must not bind
|
|
34
|
+
* past a machine that filled up between the plan and the adoption. It is always supplied by
|
|
35
|
+
* `MissionEngine.pass`; the parallel parked wake builds the same object through
|
|
36
|
+
* `MissionEngine.admissionPolicy()`, so both wake paths gate identically instead of one of them
|
|
37
|
+
* opting out — which is what let a wake oversubscribe the machine.
|
|
38
|
+
*/
|
|
39
|
+
readonly capacity?: CapacityPolicy;
|
|
40
|
+
/** The capacity wait this candidate served before being selected; see {@link StartWorkerInput}. */
|
|
41
|
+
readonly waitedMs?: number;
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* What a continuation attempt resolved to. Three answers, not a boolean, because "nothing to do" is
|
|
45
|
+
* not "it failed": when another delivery already owns the session, starting a fresh executor on top
|
|
46
|
+
* of it is precisely the double run this guard exists to prevent.
|
|
47
|
+
*
|
|
48
|
+
* - `resumed` — the node is bound to `workerId` and the prompt was accepted; the caller counts one
|
|
49
|
+
* dispatch and starts nobody.
|
|
50
|
+
* - `failed` — the continuation is not going to happen: either the delivery was refused and the
|
|
51
|
+
* host has already undone its own adoption (`reclaim(..., 'wake-failed')`, which charges no
|
|
52
|
+
* budget), or the host declined it BEFORE adopting because the node drifted too far from what that
|
|
53
|
+
* session last read (a material change; see `continuation`). Either way the caller takes the
|
|
54
|
+
* ordinary fresh path with a newly reserved claim, which is the complete answer: a new executor
|
|
55
|
+
* reads every correction and every note.
|
|
56
|
+
* - `skip` — the node must be left exactly as it is: a wake for that session is in flight, the owner
|
|
57
|
+
* is not materialized, or the node's state moved under us. Never a reason to spawn.
|
|
58
|
+
*/
|
|
59
|
+
export type ResumeOutcome = 'resumed' | 'failed' | 'skip';
|
|
60
|
+
export interface EngineHooks {
|
|
61
|
+
/** Reserve a child session id without creating it, which closes the spawn/bind window: the node is bound before anything is materialized. Pair with {@link releaseClaimId}: a claim reserved for a dispatch that is then refused was never bound and must be handed back, or it stays "live" as a ghost forever. */
|
|
62
|
+
reserveClaimId(): string;
|
|
63
|
+
/** Hand back a claim reserved by {@link reserveClaimId} whose dispatch did not happen; nothing was materialized for it. */
|
|
64
|
+
releaseClaimId(claimId: string): void;
|
|
65
|
+
/** Materialize the worker for a dispatched node and deliver its prompt; the prompt is built HERE from `tree.view(node.id)`, because a sibling result may have landed since the dispatch decision. */
|
|
66
|
+
startWorker(input: StartWorkerInput): Promise<void>;
|
|
67
|
+
/**
|
|
68
|
+
* Try to CONTINUE a node in the session it was interrupted in, instead of starting a fresh one
|
|
69
|
+
* (a cold wake). The host owns this because only it can reach `ctx.subagents`, and the delivery
|
|
70
|
+
* must be sent by the child's live direct parent (`authorizeLineage`).
|
|
71
|
+
*
|
|
72
|
+
* Called BEFORE a claim is reserved, so a continuation that lands consumes no claim id and a
|
|
73
|
+
* fallback reserves one exactly as any other dispatch does. See {@link ResumeOutcome} for the
|
|
74
|
+
* contract, including the rule that `skip` must never be answered with a spawn.
|
|
75
|
+
*
|
|
76
|
+
* Priority when several addresses apply — documented because it is a decision, not an accident:
|
|
77
|
+
* a parked session wins (it is an ALIVE continuation waiting to be woken), then this cold
|
|
78
|
+
* continuation, then a fresh spawn (`nextDispatchable` already excludes parked nodes).
|
|
79
|
+
*/
|
|
80
|
+
resumeWorker?(input: ResumeWorkerInput): Promise<ResumeOutcome>;
|
|
81
|
+
/** Stop a worker believed to be stuck. */
|
|
82
|
+
interruptWorker(sessionId: string): Promise<void>;
|
|
83
|
+
/** Deliver a one-line signal to the tree owner; it carries no content — the guidance layer does. */
|
|
84
|
+
notifyOwner(rootId: string, reason: string): void;
|
|
85
|
+
/** Heads-up that one node keeps going silent, so the owner can decide whether to intervene — deliberately separate from {@link notifyOwner}, which means "a tree reached a terminal state". */
|
|
86
|
+
notifyStalled?(info: StallReport): void;
|
|
87
|
+
/**
|
|
88
|
+
* Diagnostic sink for a `hung` reclaim: the worker was still alive but had produced nothing for a
|
|
89
|
+
* whole window (or exceeded the round cap), so the engine interrupted it and re-queued the node
|
|
90
|
+
* WITHOUT charging any budget. Nothing here is a decision for the owner — an apparatus outage is
|
|
91
|
+
* not a mission failure — so this exists to be LOGGED, not to wake anybody.
|
|
92
|
+
*/
|
|
93
|
+
notifyHung?(info: HungReport): void;
|
|
94
|
+
/**
|
|
95
|
+
* A dispatch was DEFERRED by the capacity gate (never refused). Rate-limited by the engine, one
|
|
96
|
+
* line per node per minute, so a queue waiting on a heavy mission does not flood the log while
|
|
97
|
+
* still remaining diagnosable. The host formats and logs it.
|
|
98
|
+
*/
|
|
99
|
+
notifyDeferred?(info: DispatchDeferral): void;
|
|
100
|
+
/**
|
|
101
|
+
* Report nodes waiting for a parked session to be woken; the engine cannot wake anything itself, because a wake is a delivery through `ctx.subagents` that only the host owns and must be authorized by the child's live direct parent.
|
|
102
|
+
* Its whole job is to say a convergence pass is due on an idle session, as a BATCH: an owner that was offline can materialize with several parked nodes at once, and one wake per node would be a wake storm.
|
|
103
|
+
* The host performs the actual wake inside the owner's next turn; a host that cannot wake leaves the address on the node, and this reports again on the next pass.
|
|
104
|
+
*/
|
|
105
|
+
notifyParkedReady?(nodes: readonly NodeRecord[]): void;
|
|
106
|
+
reportDispatchFailure?(nodeId: string, error: unknown): void;
|
|
107
|
+
trace?(message: string): void;
|
|
108
|
+
}
|
|
109
|
+
export interface StallReport {
|
|
110
|
+
readonly rootId: string;
|
|
111
|
+
readonly nodeId: string;
|
|
112
|
+
readonly title: string;
|
|
113
|
+
/** Dispatches so far, NOT a budget: successful rounds — including aggregate/convergence passes — raise it too. */
|
|
114
|
+
readonly attempts: number;
|
|
115
|
+
/** Failed executions so far, including the reclaim being reported; this is the counter the `failed` ceiling bounds. */
|
|
116
|
+
readonly failures: number;
|
|
117
|
+
/** Reclaims for silence so far, including the one being reported. */
|
|
118
|
+
readonly stalls: number;
|
|
119
|
+
readonly silentMs: number;
|
|
120
|
+
/**
|
|
121
|
+
* Which trouble this is. `silence` (the default, and what an older caller sees) is a stall; `hung`
|
|
122
|
+
* is the repeated-hang heads-up, which shares the channel, the {@link isTroubledNode} floor and the
|
|
123
|
+
* one-message-per-node marker precisely so the owner reads ONE vocabulary rather than two.
|
|
124
|
+
*/
|
|
125
|
+
readonly cause?: 'silence' | 'hung';
|
|
126
|
+
/** Consecutive hangs at report time; present only for `cause: 'hung'`. */
|
|
127
|
+
readonly hungs?: number;
|
|
128
|
+
}
|
|
129
|
+
/** What a `hung` reclaim looked like, for the engine's diagnostic log. This per-reclaim report is
|
|
130
|
+
* NOT an owner-facing message by itself — one hung round is an apparatus event, not a decision —
|
|
131
|
+
* but a STREAK of them is (`isTroubledNode` reads `hungCount`), and that escalation travels through
|
|
132
|
+
* {@link StallReport} with `cause: 'hung'`. */
|
|
133
|
+
export interface HungReport {
|
|
134
|
+
readonly rootId: string;
|
|
135
|
+
readonly nodeId: string;
|
|
136
|
+
readonly title: string;
|
|
137
|
+
readonly attempts: number;
|
|
138
|
+
readonly failures: number;
|
|
139
|
+
readonly stalls: number;
|
|
140
|
+
/** Which bound fired: the `staleMs` output window, or the `roundMs` ceiling. */
|
|
141
|
+
readonly bound: Extract<LivenessBound, 'output' | 'round'>;
|
|
142
|
+
/** ms since the dispatch that opened this round. */
|
|
143
|
+
readonly ranMs: number;
|
|
144
|
+
/** ms since the worker last produced anything. */
|
|
145
|
+
readonly idleMs: number;
|
|
146
|
+
/** Consecutive hangs after this reclaim, including it — the streak `isTroubledNode` reads. */
|
|
147
|
+
readonly hungCount: number;
|
|
148
|
+
}
|
|
149
|
+
/** Why a dispatch is waiting, for the engine's rate-limited diagnostic log. This is NOT an
|
|
150
|
+
* `isTroubled` signal: ordinary queuing is not "keeps going wrong". */
|
|
151
|
+
export interface DispatchDeferral {
|
|
152
|
+
readonly nodeId: string;
|
|
153
|
+
readonly rootId: string;
|
|
154
|
+
readonly title: string;
|
|
155
|
+
readonly waitingFor: WaitingFor;
|
|
156
|
+
/** Weight of the deferred node (`capacity` reason). */
|
|
157
|
+
readonly needed: number;
|
|
158
|
+
readonly capacity: number;
|
|
159
|
+
readonly running: number;
|
|
160
|
+
/** How long the capacity gate has been deferring this node; `0` for a non-capacity reason. */
|
|
161
|
+
readonly waitedMs: number;
|
|
162
|
+
/** True when this node has aged into a reservation (no new admissions until it fits). */
|
|
163
|
+
readonly reserved: boolean;
|
|
164
|
+
}
|
|
165
|
+
export interface EngineOptions {
|
|
166
|
+
/**
|
|
167
|
+
* Backstop on the NUMBER of running units: with `capacity` it makes up the two-part admission
|
|
168
|
+
* rule. Capacity is the MASTER gate (how much work the machine can carry); this only stops a flood
|
|
169
|
+
* of weight-1 missions from opening more sessions than anyone wants. Both must admit a candidate.
|
|
170
|
+
*/
|
|
171
|
+
readonly maxConcurrent: number;
|
|
172
|
+
/**
|
|
173
|
+
* Master gate: how many capacity units (cores-equivalent, the same unit as `NodeRecord.weight`)
|
|
174
|
+
* may run at once. Required, because the core must not guess at the hardware — the host derives it
|
|
175
|
+
* (configured → `os.availableParallelism()` → `os.cpus().length` → 4, minus one reserved core) and
|
|
176
|
+
* injects the number, which is also what makes every scheduling decision deterministic in tests.
|
|
177
|
+
*/
|
|
178
|
+
readonly capacity: number;
|
|
179
|
+
/**
|
|
180
|
+
* How long a node may be repeatedly deferred by the capacity gate before it RESERVES the machine
|
|
181
|
+
* (no new admissions until it fits). See `capacity.ts` for the value and its rationale.
|
|
182
|
+
*/
|
|
183
|
+
readonly capacityWaitMs: number;
|
|
184
|
+
/**
|
|
185
|
+
* Free-memory floor, in bytes, enforced through {@link ResourceProbe.memoryBudget}: below it the
|
|
186
|
+
* gate DEFERS dispatch (never refuses). `0` disables the gate; the probe itself is optional, and a
|
|
187
|
+
* probe that does not implement `memoryBudget` leaves this gate inactive (the v1 degradation path).
|
|
188
|
+
*/
|
|
189
|
+
readonly minFreeMemoryBytes: number;
|
|
190
|
+
/** Machine reader, injected by the host. Absent means "no resource signal": the memory floor is
|
|
191
|
+
* then the only gate this could arm, and with no probe it is inactive. */
|
|
192
|
+
readonly probe?: ResourceProbe;
|
|
193
|
+
/** How long a worker may produce NOTHING before it is considered stuck (see `liveness.ts`). */
|
|
194
|
+
readonly staleMs: number;
|
|
195
|
+
/** Wall-clock ceiling on one dispatch: past this the node is reclaimed as `hung` no matter how
|
|
196
|
+
* many events refreshed its timestamps. The backstop against a transport that retries forever. */
|
|
197
|
+
readonly roundMs: number;
|
|
198
|
+
/** Clock for the stale check and the capacity aging window, injectable so tests can move time
|
|
199
|
+
* instead of waiting for it; production leaves it at `Date.now`. */
|
|
200
|
+
readonly now?: () => number;
|
|
201
|
+
}
|
|
202
|
+
/**
|
|
203
|
+
* Advisory parallelism for a core that has no Node API of its own (the core builds without Node
|
|
204
|
+
* types, so the real machine reading is the HOST's job — see `resource.ts` and the plugin's probe).
|
|
205
|
+
*
|
|
206
|
+
* The lookup order is deliberate: `navigator.hardwareConcurrency` when a DOM-shaped global is
|
|
207
|
+
* present, then the family's old fallback of 4. An earlier build tried
|
|
208
|
+
* `process.availableParallelism?.()` here and that was DEAD CODE: Node has never exposed that member
|
|
209
|
+
* on `process` — the real API is `os.availableParallelism()`, which the host now calls. Core callers
|
|
210
|
+
* that need the machine number must inject `EngineOptions.capacity`; this function only exists so
|
|
211
|
+
* `DEFAULT_ENGINE_OPTIONS` (and any non-Node embedder) has a sane advisory value.
|
|
212
|
+
*/
|
|
213
|
+
export declare function detectConcurrency(): number;
|
|
214
|
+
export declare const DEFAULT_ENGINE_OPTIONS: EngineOptions;
|
|
215
|
+
export declare class MissionEngine {
|
|
216
|
+
private readonly tree;
|
|
217
|
+
private readonly hooks;
|
|
218
|
+
private readonly options;
|
|
219
|
+
private pumping;
|
|
220
|
+
private pumpRequested;
|
|
221
|
+
/** Node id → when the capacity gate first deferred it. In-memory on purpose: a restart loses the
|
|
222
|
+
* aging clock, which re-arms it (one extra wait window) — never a correctness loss. */
|
|
223
|
+
private readonly capacityWaits;
|
|
224
|
+
/** The live `waitingFor` projection, refreshed at the end of every pass: node id → reason, or
|
|
225
|
+
* absent for a node nothing is holding back. */
|
|
226
|
+
private readonly waiting;
|
|
227
|
+
/** Last time each node's deferral was logged, for the rate limit. */
|
|
228
|
+
private readonly deferredLogged;
|
|
229
|
+
/** Nodes whose RESERVATION transition has already been logged. A reservation overrides the rate
|
|
230
|
+
* limit: it changes what the engine admits, so it must not be swallowed by a recent deferral line. */
|
|
231
|
+
private readonly reservedLogged;
|
|
232
|
+
constructor(tree: MissionTree, hooks: EngineHooks, options?: EngineOptions);
|
|
233
|
+
/** Why one node has not been dispatched, or `null` when nothing is holding it back. The projection
|
|
234
|
+
* is read by the host for `mission_result` and the panel; it is live state, never persisted. */
|
|
235
|
+
waitingFor(nodeId: string): WaitingFor | null;
|
|
236
|
+
/**
|
|
237
|
+
* The admission policy THIS moment would gate on, from the engine's own single source: the same
|
|
238
|
+
* `capacityPolicy(memoryFloor())` the dispatch loop builds, carrying the engine's own aging clock
|
|
239
|
+
* (`capacityWaits`) and the live machine-wide memory block. Built, never stored: the caller gets a
|
|
240
|
+
* snapshot for one decision, exactly like a pass's policy object.
|
|
241
|
+
*
|
|
242
|
+
* It exists for the WAKE paths. `adoptParked` / `adoptContinuation` re-run the gate under the tree
|
|
243
|
+
* lock, and the rule is that a wake must be judged by the very policy the dispatch loop would use —
|
|
244
|
+
* deriving a second one in the host (its own capacity number, its own clock) is how the two drift.
|
|
245
|
+
* The continuation wake already receives the pass's policy through `resumeWorker`; this is how the
|
|
246
|
+
* parked wake, which has no plan snapshot of its own, gets the same one.
|
|
247
|
+
*/
|
|
248
|
+
admissionPolicy(): CapacityPolicy;
|
|
249
|
+
/** The capacity the engine is actually gating on, for diagnostics and tests. */
|
|
250
|
+
capacity(): {
|
|
251
|
+
capacity: number;
|
|
252
|
+
maxConcurrent: number;
|
|
253
|
+
runningCount: number;
|
|
254
|
+
runningWeight: number;
|
|
255
|
+
};
|
|
256
|
+
private now;
|
|
257
|
+
/** Run one dispatch pass; `dispatch()` decides and marks in the same await, so a second pass cannot see the same node as available and no batch bookkeeping is needed. */
|
|
258
|
+
pump(): Promise<number>;
|
|
259
|
+
sweep(): Promise<{
|
|
260
|
+
reclaimed: number;
|
|
261
|
+
dispatched: number;
|
|
262
|
+
}>;
|
|
263
|
+
/**
|
|
264
|
+
* Reclaim nodes whose worker is gone or stuck; liveness is the primary signal, and the windows
|
|
265
|
+
* only cover "alive but not useful".
|
|
266
|
+
*
|
|
267
|
+
* The judgement is `judgeWorker` (see `liveness.ts`): silence measured from the last event at all
|
|
268
|
+
* is `stalled` and charges the failure budget exactly as before; a worker that keeps being heard
|
|
269
|
+
* from while producing NOTHING for `staleMs`, or that exceeds the round cap, is `hung` and charges
|
|
270
|
+
* nothing. Both interrupt the live binding before reclaiming, and both go through the same
|
|
271
|
+
* compare-and-swap `MissionTree.reclaim`, so a verdict judged against a snapshot taken outside
|
|
272
|
+
* the tree lock cannot strip a fresh binding or charge one attempt twice.
|
|
273
|
+
*/
|
|
274
|
+
reclaimStale(): Promise<number>;
|
|
275
|
+
/**
|
|
276
|
+
* The ONE owner-facing "this node keeps going wrong" path, shared by every cause. Three rules make
|
|
277
|
+
* it one vocabulary rather than two: the gate is {@link isTroubledNode} (the same predicate the
|
|
278
|
+
* `isTroubled` flag reads), the report rides the `notifyStalled` channel (a heads-up, never a
|
|
279
|
+
* question), and the durable `claimStallReport` marker keeps it to one message per node whichever
|
|
280
|
+
* way the node is failing. The engine already recovered on its own in every case, so waiting for a
|
|
281
|
+
* repeat is what keeps a permanently flaky node from turning a sweep into a wake storm.
|
|
282
|
+
*/
|
|
283
|
+
private escalateTrouble;
|
|
284
|
+
/** Tell the host a node was reclaimed as `hung`, so every hung round is diagnosable — a provider
|
|
285
|
+
* that hangs (or only retries) every dispatch must never be invisible, which is exactly how W8
|
|
286
|
+
* stayed unnoticed for 7.5 hours. This per-reclaim line is NOT the owner escalation: one hung
|
|
287
|
+
* round is an apparatus event the engine recovered from, and only a STREAK crosses
|
|
288
|
+
* {@link escalateTrouble}'s `isTroubledNode` floor. */
|
|
289
|
+
private reportHung;
|
|
290
|
+
/** Destroy trees whose owner session no longer exists: the owner is the authority over the tree, and a hot reload or host restart leaves it resolvable, so nothing is destroyed there. A tree whose owner merely cannot be OBSERVED is left alone and reported instead — the host being unable to answer is not evidence the owner is gone, and a destroyed tree cannot be recovered. */
|
|
291
|
+
reconcileOrphans(): Promise<readonly string[]>;
|
|
292
|
+
/**
|
|
293
|
+
* The machine-wide gates that do not depend on WHICH node is being considered: the free-memory
|
|
294
|
+
* floor. Returns the `waitingFor` every candidate carries while it is set, or `undefined` when the
|
|
295
|
+
* gate is open.
|
|
296
|
+
*
|
|
297
|
+
* A probe that does not implement `memoryBudget` leaves the gate inactive (v1: no platform
|
|
298
|
+
* adapter). A probe that implements it and answers `null` means "cannot confirm the floor", and
|
|
299
|
+
* that is a DEFER, not "plenty" — see `resource.ts`: no signal never relaxes scheduling.
|
|
300
|
+
*/
|
|
301
|
+
private memoryFloor;
|
|
302
|
+
/** The capacity arithmetic for this moment, built from the tree's live load. */
|
|
303
|
+
private capacityPolicy;
|
|
304
|
+
private pass;
|
|
305
|
+
/**
|
|
306
|
+
* Publish the live `waitingFor` projection for the nodes a pass left behind, and drop entries for
|
|
307
|
+
* nodes that are no longer waiting (dispatched, terminal, or blocked by something outside
|
|
308
|
+
* admission). The map is the ONE place the host reads waiting state from, so it is rebuilt to
|
|
309
|
+
* exactly the deferred set rather than accumulated across passes.
|
|
310
|
+
*/
|
|
311
|
+
private publishWaiting;
|
|
312
|
+
/**
|
|
313
|
+
* The rate-limited deferral log: one line per node per minute while the capacity gate holds it
|
|
314
|
+
* back. Deliberately not an owner wake and not an `isTroubled` fact — normal queuing is not a
|
|
315
|
+
* problem the owner can act on.
|
|
316
|
+
*
|
|
317
|
+
* Two reasons count as "waiting for capacity" here: `capacity` itself (this node does not fit) and
|
|
318
|
+
* `aging` (ANOTHER node has reserved the machine, so this one is held back while capacity may have
|
|
319
|
+
* room). The second is the one the aging mechanism exists to make visible — leaving it out is how
|
|
320
|
+
* the reservation changed what the engine admits with no line anywhere saying so.
|
|
321
|
+
*/
|
|
322
|
+
private reportDeferrals;
|
|
323
|
+
/** Report the parked-ready batch to the host after the dispatch loop; the wake itself is the host's job — see {@link EngineHooks.notifyParkedReady}. */
|
|
324
|
+
private reportParkedReady;
|
|
325
|
+
/**
|
|
326
|
+
* Wake the owner about every terminal tree, exactly once each; dedup lives on the durable tree record (`claimReport`), because an in-memory set is empty after a restart and would re-report every tree that had ever reached a terminal state.
|
|
327
|
+
*/
|
|
328
|
+
private reportTerminalRoots;
|
|
329
|
+
/** Forget the report marker for a tree, so a re-rooted tree can report again. */
|
|
330
|
+
forgetReport(rootId: string): void;
|
|
331
|
+
}
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @avantf/mission-core — the mission-tree model, dispatch loop and prompt construction; harness environment facts arrive as injected dependencies.
|
|
3
|
+
* @module @avantf/mission-core
|
|
4
|
+
*/
|
|
5
|
+
export * from './types.js';
|
|
6
|
+
export { MissionTree, defaultNewId } from './tree.js';
|
|
7
|
+
export type { TreeState, TreeStore, TreeDeps, SpilledText, CreateRootInput, OwnerProbe, OrphanedTree, } from './tree.js';
|
|
8
|
+
export { MissionEngine, DEFAULT_ENGINE_OPTIONS, detectConcurrency } from './engine.js';
|
|
9
|
+
export type { DispatchDeferral, EngineHooks, EngineOptions, HungReport, ResumeOutcome, ResumeWorkerInput, StartWorkerInput, StallReport, } from './engine.js';
|
|
10
|
+
export { CAPACITY_CEILING, CAPACITY_FALLBACK_PARALLELISM, DEFAULT_CAPACITY_WAIT_MS, DEFAULT_MIN_FREE_MEMORY_BYTES, DEFAULT_WEIGHT, MAX_WEIGHT, MIN_CAPACITY_WAIT_MS, MIN_WEIGHT, RESERVED_CORES, clampCapacity, deriveCapacity, normalizeWeight, } from './capacity.js';
|
|
11
|
+
export type { CapacityInput, CapacityReading, CapacitySource } from './capacity.js';
|
|
12
|
+
export type { ResourceProbe } from './resources.js';
|
|
13
|
+
export { effectiveRoundMs, heardAt, judgeWorker, MAX_DECLARED_ROUND_MS, normalizeRoundMs, producedAt, resolveChildRoundMs, storedTime } from './liveness.js';
|
|
14
|
+
export type { LivenessBound, LivenessVerdict, LivenessWindows, ReclaimCause } from './liveness.js';
|
|
15
|
+
export { byCreatedAtThenId, capacityWaitingFor, heldUnits, normalizeUnit, planDispatch, resolveChildUnit, resolveChildWeight, selectNextDispatchable, spawnBackoffMs, unitHolder, } from './dispatch.js';
|
|
16
|
+
export type { CapacityLoad, CapacityPolicy, DeferredCandidate, DispatchPlan, DispatchPolicy, DispatchScope, } from './dispatch.js';
|
|
17
|
+
export { computeContinuationDelta, isMaterialChange, nodeFingerprint } from './continuation.js';
|
|
18
|
+
export type { ContinuationDelta } from './continuation.js';
|
|
19
|
+
export { queueMs, runMs, totalMs } from './timing.js';
|
|
20
|
+
export type { NodeTiming } from './timing.js';
|
|
21
|
+
export { buildWorkerPrompt, buildProgressLine, isTroubled, PROMPT_LIMITS, spillPointer, waitedLabel } from './prompt.js';
|
|
22
|
+
export type { WorkerPromptOptions } from './prompt.js';
|
|
23
|
+
export { isTroubledNode, statusLabel } from './trouble.js';
|
|
24
|
+
export { LOCAL_WELL_FORMED, wellFormedDeep, wellFormedText } from './wellformed.js';
|
|
25
|
+
export type { WellFormedSource } from './wellformed.js';
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The liveness verdict for ONE running node: is its worker still making progress, or has the round
|
|
3
|
+
* become something the engine must take back?
|
|
4
|
+
*
|
|
5
|
+
* Two clocks, because "no events" and "no output" are different failures — and conflating them is
|
|
6
|
+
* what let a transport-layer hang sit `running` for 7.5 hours (2026-10-02):
|
|
7
|
+
*
|
|
8
|
+
* - `progressAt` — the last PRODUCTION: the model's committed output (`assistant/message`), a tool
|
|
9
|
+
* it asked for (`tool/call`), or a tool that finished (`tool/result`). The plugin classifies the
|
|
10
|
+
* host's durable event feed (see its `workerEvents` module) and only those refresh it.
|
|
11
|
+
* - `activityAt` — the last durable event of ANY kind, including transport-layer noise: provider
|
|
12
|
+
* retry attempts (`assistant/attempt`) and the log-only per-request route snapshots
|
|
13
|
+
* (`request/header`, `request/context`). Noise proves the session object is alive; it does not
|
|
14
|
+
* prove the mission moved.
|
|
15
|
+
*
|
|
16
|
+
* Three bounds come out of that, in this order:
|
|
17
|
+
*
|
|
18
|
+
* - **silence** (`now - max(activityAt, progressAt, claimedAt) > staleMs`) → `stalled`. Nothing was
|
|
19
|
+
* heard at all. This is the original verdict with its original budget charge, and it is checked
|
|
20
|
+
* first so a node the old build would have called silent is still called silent.
|
|
21
|
+
* - **round** (`now - claimedAt > roundMs`) → `hung`. A wall-clock ceiling on the dispatch itself,
|
|
22
|
+
* which no timestamp can veto. It fires before the output bound on purpose: when a productive
|
|
23
|
+
* worker reaches it, "the round hit its ceiling" is the accurate description, not "no output".
|
|
24
|
+
* - **output** (activity fresh but `now - (progressAt || claimedAt) > staleMs`) → `hung`. The W8
|
|
25
|
+
* shape exactly: the worker keeps being heard from while producing nothing.
|
|
26
|
+
*
|
|
27
|
+
* Both `hung` bounds are reclaimed WITHOUT charging the failure budget: a stuck provider is not a
|
|
28
|
+
* failed mission, and letting retries eat the budget would turn a recoverable node `failed`. What
|
|
29
|
+
* keeps that leniency from becoming an unbounded loop is `hungCount` — the engine's own consecutive
|
|
30
|
+
* hang counter, cleared by real output — which at `CAPACITY.maxHungsBeforeReport` makes the node
|
|
31
|
+
* visible to the owner through the same trouble channel `stalled` uses (see `MissionTree.reclaim`
|
|
32
|
+
* and `MissionEngine.reclaimStale`).
|
|
33
|
+
*
|
|
34
|
+
* A record written before `activityAt` existed has no value there, and its `progressAt` was
|
|
35
|
+
* refreshed by ANY event. {@link heardAt}/{@link producedAt} then fall back to `progressAt` for
|
|
36
|
+
* both clocks, so such a node is judged exactly as the previous build judged it — plus the round
|
|
37
|
+
* cap, which is the one bound that still catches the old records.
|
|
38
|
+
*
|
|
39
|
+
* @module @avantf/mission-core/liveness
|
|
40
|
+
*/
|
|
41
|
+
import type { NodeRecord } from './types.js';
|
|
42
|
+
/** Why a running node's worker must be taken back. `stalled` charges the failure budget; `hung` does not. */
|
|
43
|
+
export type ReclaimCause = 'stalled' | 'hung';
|
|
44
|
+
/** Which bound produced a `hung` (or `stalled`) verdict. */
|
|
45
|
+
export type LivenessBound = 'silence' | 'output' | 'round';
|
|
46
|
+
export interface LivenessVerdict {
|
|
47
|
+
readonly cause: ReclaimCause;
|
|
48
|
+
readonly bound: LivenessBound;
|
|
49
|
+
/** ms since the worker was last heard from at all. */
|
|
50
|
+
readonly silentMs: number;
|
|
51
|
+
/** ms since the worker last PRODUCED something. */
|
|
52
|
+
readonly idleMs: number;
|
|
53
|
+
/** ms since the dispatch that opened this round. */
|
|
54
|
+
readonly ranMs: number;
|
|
55
|
+
}
|
|
56
|
+
export interface LivenessWindows {
|
|
57
|
+
readonly staleMs: number;
|
|
58
|
+
readonly roundMs: number;
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Hard ceiling on a DECLARED round cap: 24 hours. The declaration exists to relax the cap for
|
|
62
|
+
* genuinely heavy work, not to opt out of it — a single round longer than a day is indistinguishable
|
|
63
|
+
* from a hang, and the mission can always be re-entered by the ordinary retry. An operator's own
|
|
64
|
+
* configured `roundMs` is NOT clamped by this; only a node's declaration is.
|
|
65
|
+
*/
|
|
66
|
+
export declare const MAX_DECLARED_ROUND_MS: number;
|
|
67
|
+
/**
|
|
68
|
+
* Read a node's declared round-cap relaxation into shape: a finite number > 0, or `null` for
|
|
69
|
+
* "declared nothing". Missing, dirty and non-positive values all read as `null`, the same direction
|
|
70
|
+
* as a record written before the field existed — a declaration that cannot be believed must not
|
|
71
|
+
* change when a round is taken back. Anything past {@link MAX_DECLARED_ROUND_MS} falls to it.
|
|
72
|
+
*/
|
|
73
|
+
export declare function normalizeRoundMs(raw: unknown): number | null;
|
|
74
|
+
/**
|
|
75
|
+
* A decomposed child's round-cap relaxation. `undefined` — the spec said nothing — does NOT inherit
|
|
76
|
+
* the parent's declaration: a parent that needs hours says nothing about one child, and the same
|
|
77
|
+
* reasoning already governs `weight` (`resolveChildWeight`).
|
|
78
|
+
*/
|
|
79
|
+
export declare function resolveChildRoundMs(declared: number | null | undefined): number | null;
|
|
80
|
+
/**
|
|
81
|
+
* The round cap actually applied to one node: the engine's configured ceiling, RELAXED by the node's
|
|
82
|
+
* own declaration and never shortened by it. Relaxation-only is the whole rule — the cap is the
|
|
83
|
+
* backstop against a transport that retries forever, so a mission may ask for more room but may not
|
|
84
|
+
* opt out of the backstop, and a smaller declaration (a countdown a model could otherwise set for
|
|
85
|
+
* itself) is a no-op.
|
|
86
|
+
*/
|
|
87
|
+
export declare function effectiveRoundMs(node: NodeRecord, configured: number): number;
|
|
88
|
+
/** A timestamp as stored. A record written before the field existed has `undefined` at runtime even
|
|
89
|
+
* though the type says `number`, and `Math.max(undefined, …)` is `NaN` — which would compare false
|
|
90
|
+
* against every window and keep a dead node running forever. */
|
|
91
|
+
export declare function storedTime(value: number | undefined): number;
|
|
92
|
+
/** When the worker was last heard from at all; the dispatch that opened the round is the floor, so
|
|
93
|
+
* a node that never reported anything is "silent since it started" rather than "silent since 0". */
|
|
94
|
+
export declare function heardAt(node: NodeRecord): number;
|
|
95
|
+
/** When the worker last PRODUCED something. Falls back to the dispatch: a node that has produced
|
|
96
|
+
* nothing yet has been unproductive since it was bound, which is the reading `hung` needs. */
|
|
97
|
+
export declare function producedAt(node: NodeRecord): number;
|
|
98
|
+
/** The verdict for one running node, or `undefined` when it is still doing fine. */
|
|
99
|
+
export declare function judgeWorker(node: NodeRecord, at: number, windows: LivenessWindows): LivenessVerdict | undefined;
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Worker prompt construction: one template, branching on the node's state for the trailing section; the tail is derived from the node at dispatch time, exists only in this dispatch's input and never enters a conversation.
|
|
3
|
+
* @module @avantf/mission-core/prompt
|
|
4
|
+
*/
|
|
5
|
+
import { type DispatchView, type NodeRecord } from './types.js';
|
|
6
|
+
import type { ContinuationDelta } from './continuation.js';
|
|
7
|
+
import { type WellFormedSource } from './wellformed.js';
|
|
8
|
+
/**
|
|
9
|
+
* Whether a mission should read as TROUBLED to its owner: some unfinished node trips {@link isTroubledNode}.
|
|
10
|
+
* The wording has to say this is HISTORY, because a flag that never clears would invite the owner to steer or cancel a mission that is running fine, and both actions are destructive.
|
|
11
|
+
* Deliberately NOT here: which mission it was, how often, and how much budget is left — those belong to the engine's retry policy, and the owner's two actions (`adjust_mission`, `cancel_mission`) take the whole mission, not a node inside it.
|
|
12
|
+
*/
|
|
13
|
+
export declare function isTroubled(nodes: readonly NodeRecord[]): boolean;
|
|
14
|
+
/**
|
|
15
|
+
* The prompt's size guardrails. The tail is rewritten on every dispatch and everything here is paid
|
|
16
|
+
* for in model tokens, so each of these bounds one axis that could grow without limit:
|
|
17
|
+
*
|
|
18
|
+
* - {@link chainLine} rendered EVERY correction of EVERY ancestor (a correction is written on the
|
|
19
|
+
* root, so one long-lived root steered repeatedly made every descendant carry all of them);
|
|
20
|
+
* - {@link analysisSection} rendered every note an executor had ever appended;
|
|
21
|
+
* - {@link childrenBlock} rendered every terminal child's conclusion. Per-child size is already
|
|
22
|
+
* bounded by the engine (`CAPACITY.maxInlineResultChars`, longer results are spilled with a
|
|
23
|
+
* pointer), but the COUNT is not: `children` accumulates across re-decompositions, and 50
|
|
24
|
+
* children × the 2 KB inline cap was a measured ~100 KB (~50k tokens) aggregate prompt.
|
|
25
|
+
*
|
|
26
|
+
* What the model still sees after a bound bites is stated IN the prompt at each site, so a judge can
|
|
27
|
+
* tell "this child said nothing" from "this child's text was summarized here".
|
|
28
|
+
*/
|
|
29
|
+
export declare const PROMPT_LIMITS: {
|
|
30
|
+
/** Corrections rendered per mission-chain ancestor (the NEWEST ones; the rest are counted). */
|
|
31
|
+
readonly chainCorrections: 2;
|
|
32
|
+
readonly chainCorrectionChars: 200;
|
|
33
|
+
/** Corrections rendered verbatim on the node being executed (the newest ones; the rest counted). */
|
|
34
|
+
readonly currentCorrections: 10;
|
|
35
|
+
/** `note_mission` notes rendered (the newest ones; the rest counted). */
|
|
36
|
+
readonly analysisNotes: 8;
|
|
37
|
+
readonly analysisNoteChars: 400;
|
|
38
|
+
/** Terminal children whose conclusion is rendered in full — the newest ones. Older children still
|
|
39
|
+
* list id, title, status and the head of their conclusion. The count is what accumulates across
|
|
40
|
+
* re-decompositions, so this is the bound that stops a 50-child aggregate from growing without end. */
|
|
41
|
+
readonly childrenBody: 10;
|
|
42
|
+
/** Safety clip on one child's conclusion; equal to the engine's inline cap, so a well-formed record
|
|
43
|
+
* is never actually cut (an older/foreign record could exceed it). */
|
|
44
|
+
readonly childBodyChars: 2000;
|
|
45
|
+
/** How much of an older child's conclusion survives in the compact form. */
|
|
46
|
+
readonly childExcerptChars: 120;
|
|
47
|
+
};
|
|
48
|
+
/**
|
|
49
|
+
* Where a spilled full result lives, with the backend's own retrieval guidance; one definition for both renderers — the worker prompt and `mission_result` — because a locator without its hint is a pointer nobody can follow.
|
|
50
|
+
*/
|
|
51
|
+
export declare function spillPointer(node: NodeRecord): string;
|
|
52
|
+
/** How one dispatch renders the node it is about; the default is the fresh-executor reading. */
|
|
53
|
+
export interface WorkerPromptOptions {
|
|
54
|
+
/** This dispatch CONTINUES the session that was interrupted: the hand-off notice says which
|
|
55
|
+
* execution this is and points at the resumed session's own history. */
|
|
56
|
+
readonly resumed?: boolean;
|
|
57
|
+
/** Corrections to render on the CURRENT node. Defaults to every recorded correction, because a
|
|
58
|
+
* fresh executor has read none. A continuation wake passes
|
|
59
|
+
* `node.corrections.slice(node.correctionsDeliveredUpTo)`: a correction already delivered to that
|
|
60
|
+
* same session must not be argued to it twice. The chain's ancestor lines keep their own text —
|
|
61
|
+
* they are a different node's corrections, and none of them is repeated by this override. */
|
|
62
|
+
readonly corrections?: readonly string[];
|
|
63
|
+
/** What changed on this node since the session's own last prompt, for the delta section. Passed
|
|
64
|
+
* by a CONTINUATION only: a fresh executor has never executed this mission, so "since you last
|
|
65
|
+
* executed" would be a lie, and it reads every correction, note and child conclusion anyway. */
|
|
66
|
+
readonly delta?: ContinuationDelta;
|
|
67
|
+
/** How long the CAPACITY gate deferred this node before the dispatch that built this prompt
|
|
68
|
+
* picked it up, in ms. Passed by the engine (through `StartWorkerInput`/`ResumeWorkerInput`) from
|
|
69
|
+
* its own aging clock; `0`/absent means "never waited", and then no queue line is rendered at all.
|
|
70
|
+
* It exists because an executor with no such fact reports "I did not wait" — a real answer from a
|
|
71
|
+
* run that had in fact queued for two minutes behind a full machine. */
|
|
72
|
+
readonly capacityWaitedMs?: number;
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* How a capacity wait reads in the prompt. A stopwatch would be false precision: the number is one
|
|
76
|
+
* reading of the engine's aging clock, so it is rounded and says 约. Seconds below a minute (a node
|
|
77
|
+
* can be skipped for a few seconds and that is still "it queued"), minutes above it.
|
|
78
|
+
*/
|
|
79
|
+
export declare function waitedLabel(ms: number): string;
|
|
80
|
+
/**
|
|
81
|
+
* Build the complete prompt for one dispatch; the mission chain carries only titles and one-line context, so its size is bounded by the depth limit, and the full `description`/`context` is included for the current node only.
|
|
82
|
+
* The tail branches on the CHILDREN in the view, not on the node's status: the prompt is built after `dispatch()` has already marked the node `running`, so a status test can never see the aggregate case, and a `failed` node is never dispatched at all.
|
|
83
|
+
*
|
|
84
|
+
* ── the OUTBOUND well-formed boundary ──
|
|
85
|
+
* Everything this function writes lands in the executor's context, and the tree it reads may hold a
|
|
86
|
+
* lone surrogate that an OLDER build persisted (before the inbound funnel existed) or that a foreign
|
|
87
|
+
* writer put there. So the whole view (node, chain, children) and the whole options record are
|
|
88
|
+
* recursively repaired BEFORE a single line is assembled: `wellFormedDeep` leaves numbers, booleans
|
|
89
|
+
* and `null` untouched, so nothing but lone surrogates can change. `wellFormed` is the plugin's
|
|
90
|
+
* loaded base kit when it has one and {@link LOCAL_WELL_FORMED} otherwise.
|
|
91
|
+
*
|
|
92
|
+
* @param view - the node/chain/children this dispatch renders.
|
|
93
|
+
* @param options - how this dispatch differs from a fresh executor's.
|
|
94
|
+
* @param wellFormed - the repair pair; the plugin injects the loaded base's.
|
|
95
|
+
*/
|
|
96
|
+
export declare function buildWorkerPrompt(view: DispatchView, options?: WorkerPromptOptions, wellFormed?: WellFormedSource): string;
|
|
97
|
+
/**
|
|
98
|
+
* The compact per-tree progress line the guidance layer carries; every clause is anchored so it stays true when read late — the counts are a statement about a moment, not about "now".
|
|
99
|
+
* Granularity is deliberate: the owner is told WHETHER mission is still running and WHETHER any of it has a history of trouble, never how the mission is distributed across states, because no owner action depends on that distribution, so a per-state breakdown would only invite it to reason about a layer it cannot touch.
|
|
100
|
+
* Trouble is the one fact that changes what it should do, and it is counted in WORKS, not in nodes, so the size of anything stays inside the engine.
|
|
101
|
+
*
|
|
102
|
+
* The SAME outbound rule as {@link buildWorkerPrompt}: this line is assembled into the owner's
|
|
103
|
+
* context, so titles are repaired first (see the note there). The plugin injects the loaded base's
|
|
104
|
+
* repair pair; the core's local copy is the degradation path.
|
|
105
|
+
*/
|
|
106
|
+
export declare function buildProgressLine(input: {
|
|
107
|
+
readonly roots: readonly NodeRecord[];
|
|
108
|
+
/** How many missions have not finished yet. */
|
|
109
|
+
readonly ongoing: number;
|
|
110
|
+
/** Whether any unfinished mission carries a history of stalls or failures at the engine's floors. */
|
|
111
|
+
readonly troubled: boolean;
|
|
112
|
+
}, wellFormed?: WellFormedSource): string;
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `ResourceProbe` — the port through which the engine reads the machine, and the one place the
|
|
3
|
+
* family's `null`-is-not-zero rule is stated. Node-free: the core declares the interface and never
|
|
4
|
+
* implements it; the plugin/host supplies a probe built on `node:os` (see `host.ts`), which is also
|
|
5
|
+
* the ONLY seam allowed to contain platform branches.
|
|
6
|
+
*
|
|
7
|
+
* ## `null` is a first-class answer: "no signal on this platform"
|
|
8
|
+
*
|
|
9
|
+
* **"No signal" ≠ "idle".** A `null` must never be read as evidence that resources are free, so it
|
|
10
|
+
* may only ever make scheduling MORE conservative — never more permissive. Concretely:
|
|
11
|
+
*
|
|
12
|
+
* - `memoryBudget()` returning `null` while the free-memory floor is armed means "cannot confirm the
|
|
13
|
+
* floor", and the gate DEFERS dispatch. It is not "plenty of memory".
|
|
14
|
+
* - `pressure()` returning `null` means "cannot tell"; no caller may treat it as `0` to widen
|
|
15
|
+
* concurrency or lower a weight. v1 has no pressure-driven widening at all, so the honest reading
|
|
16
|
+
* is "no adjustment".
|
|
17
|
+
* - `workerUsage()` returning `null` means "this process's CPU is unknown"; it is not `0`, and v2's
|
|
18
|
+
* correction loop must treat it as "do not correct from this sample".
|
|
19
|
+
*
|
|
20
|
+
* The distinction between an OPTIONAL member that is absent and one that returns `null` is
|
|
21
|
+
* deliberate: absent = this port has no implementation for that signal at all (the v1 degradation
|
|
22
|
+
* path — no platform adapter), while `null` = implemented but the platform will not say. The former
|
|
23
|
+
* leaves the corresponding gate inactive; the latter keeps the gate conservative.
|
|
24
|
+
*
|
|
25
|
+
* ## v1 scope
|
|
26
|
+
*
|
|
27
|
+
* Only the interface, the wiring, and a host implementation over the UNIFIED Node API
|
|
28
|
+
* (`os.availableParallelism()`, `os.totalmem()`, `process.constrainedMemory()`, `os.freemem()`).
|
|
29
|
+
* Platform adapters (POSIX/Windows `pressure`, per-worker process attribution) and measured
|
|
30
|
+
* correction of weights are v2; every unimplemented branch answers `null`.
|
|
31
|
+
*
|
|
32
|
+
* @module @avantf/mission-core/resources
|
|
33
|
+
*/
|
|
34
|
+
export interface ResourceProbe {
|
|
35
|
+
/**
|
|
36
|
+
* How much independent work this machine will ACTUALLY run in parallel — i.e.
|
|
37
|
+
* `os.availableParallelism()`, which honours cgroup CPU quotas and CPU affinity, rather than the
|
|
38
|
+
* raw core count. Required, because every host can answer it: Node falls back to `os.cpus()`.
|
|
39
|
+
*/
|
|
40
|
+
parallelism(): number;
|
|
41
|
+
/**
|
|
42
|
+
* A LOWER BOUND, in bytes, on the memory this process may still use, or `null` when the platform
|
|
43
|
+
* will not say. "Lower bound" is the whole contract: the value may understate what is available,
|
|
44
|
+
* which is the safe direction for a floor gate, and it includes the cgroup cap when one exists
|
|
45
|
+
* (a cgroup limit below `os.totalmem()` is the real budget). Never an exact figure, never "free
|
|
46
|
+
* memory" in the leaving-room sense.
|
|
47
|
+
*/
|
|
48
|
+
memoryBudget?(): number | null;
|
|
49
|
+
/**
|
|
50
|
+
* A coarse 0..1 pressure reading (sustained CPU contention / memory pressure), or `null` when the
|
|
51
|
+
* platform cannot produce one. v1 never widens from it; the member exists so v2's adapters and the
|
|
52
|
+
* scheduler can be written against a stable signature.
|
|
53
|
+
*/
|
|
54
|
+
pressure(): number | null;
|
|
55
|
+
/** CPU usage of one worker process, or `null` when this platform cannot attribute it (v2). */
|
|
56
|
+
workerUsage?(pid: number): number | null;
|
|
57
|
+
}
|