@memberjunction/task-graph 6.1.0-edge.1 → 6.1.0-edge.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. package/LICENSE +180 -4
  2. package/README.md +214 -0
  3. package/dist/TaskClaimStore.d.ts +387 -4
  4. package/dist/TaskClaimStore.d.ts.map +1 -1
  5. package/dist/TaskClaimStore.js +605 -20
  6. package/dist/TaskClaimStore.js.map +1 -1
  7. package/dist/TaskGraphDispatcher.d.ts +668 -5
  8. package/dist/TaskGraphDispatcher.d.ts.map +1 -1
  9. package/dist/TaskGraphDispatcher.js +2942 -127
  10. package/dist/TaskGraphDispatcher.js.map +1 -1
  11. package/dist/TaskGraphService.d.ts +364 -5
  12. package/dist/TaskGraphService.d.ts.map +1 -1
  13. package/dist/TaskGraphService.js +1039 -43
  14. package/dist/TaskGraphService.js.map +1 -1
  15. package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -1
  16. package/dist/TaskGraphSubmitterImpl.js +5 -0
  17. package/dist/TaskGraphSubmitterImpl.js.map +1 -1
  18. package/dist/TaskLoopExecutor.d.ts +62 -0
  19. package/dist/TaskLoopExecutor.d.ts.map +1 -0
  20. package/dist/TaskLoopExecutor.js +248 -0
  21. package/dist/TaskLoopExecutor.js.map +1 -0
  22. package/dist/WorkflowSpecSync.d.ts +28 -2
  23. package/dist/WorkflowSpecSync.d.ts.map +1 -1
  24. package/dist/WorkflowSpecSync.js +83 -2
  25. package/dist/WorkflowSpecSync.js.map +1 -1
  26. package/dist/condition-gate.d.ts +128 -0
  27. package/dist/condition-gate.d.ts.map +1 -0
  28. package/dist/condition-gate.js +257 -0
  29. package/dist/condition-gate.js.map +1 -0
  30. package/dist/debug-state.d.ts +102 -0
  31. package/dist/debug-state.d.ts.map +1 -0
  32. package/dist/debug-state.js +135 -0
  33. package/dist/debug-state.js.map +1 -0
  34. package/dist/index.d.ts +7 -0
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +7 -0
  37. package/dist/index.js.map +1 -1
  38. package/dist/operations/TaskGraphDebugOperations.d.ts +99 -0
  39. package/dist/operations/TaskGraphDebugOperations.d.ts.map +1 -0
  40. package/dist/operations/TaskGraphDebugOperations.js +310 -0
  41. package/dist/operations/TaskGraphDebugOperations.js.map +1 -0
  42. package/dist/operations/TaskGraphOperations.d.ts +20 -2
  43. package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
  44. package/dist/operations/TaskGraphOperations.js +51 -8
  45. package/dist/operations/TaskGraphOperations.js.map +1 -1
  46. package/dist/operations/WorkflowDraftOperation.d.ts +37 -0
  47. package/dist/operations/WorkflowDraftOperation.d.ts.map +1 -0
  48. package/dist/operations/WorkflowDraftOperation.js +141 -0
  49. package/dist/operations/WorkflowDraftOperation.js.map +1 -0
  50. package/dist/settlement-rescue.d.ts +85 -0
  51. package/dist/settlement-rescue.d.ts.map +1 -0
  52. package/dist/settlement-rescue.js +119 -0
  53. package/dist/settlement-rescue.js.map +1 -0
  54. package/dist/task-graph-kick.d.ts +3 -0
  55. package/dist/task-graph-kick.d.ts.map +1 -0
  56. package/dist/task-graph-kick.js +17 -0
  57. package/dist/task-graph-kick.js.map +1 -0
  58. package/dist/task-predicates.d.ts +77 -0
  59. package/dist/task-predicates.d.ts.map +1 -0
  60. package/dist/task-predicates.js +75 -0
  61. package/dist/task-predicates.js.map +1 -0
  62. package/dist/types.d.ts +224 -1
  63. package/dist/types.d.ts.map +1 -1
  64. package/dist/types.js.map +1 -1
  65. package/package.json +12 -8
@@ -1,6 +1,6 @@
1
1
  import { UserInfo } from '@memberjunction/core';
2
2
  import { IShutdownable } from '@memberjunction/global';
3
- import { ProviderFactory, TaskAgentRunner, TaskGraphDispatcherConfig, type TaskContinuationDeliverer, type TaskGraphObserver } from './types.js';
3
+ import { ProviderFactory, TaskActionRunner, TaskAgentRunner, TaskPromptRunner, TaskGraphDispatcherConfig, type TaskContinuationDeliverer, type TaskGraphObserver } from './types.js';
4
4
  export declare class TaskGraphDispatcher implements IShutdownable {
5
5
  private readonly providerFactory;
6
6
  private readonly agentRunner;
@@ -16,18 +16,105 @@ export declare class TaskGraphDispatcher implements IShutdownable {
16
16
  * announces nothing.
17
17
  */
18
18
  private readonly observer?;
19
+ /**
20
+ * Optional. Absent means this host cannot run action nodes; they stay Pending and visible
21
+ * rather than being failed, because "nobody here can run this" is not "this ran and broke".
22
+ */
23
+ private readonly actionRunner?;
24
+ /**
25
+ * Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
26
+ * rather than being failed, for the same reason action nodes do.
27
+ */
28
+ private readonly promptRunner?;
19
29
  private readonly config;
20
30
  private readonly claims;
31
+ /**
32
+ * Edges already reported as unevaluable, so the report is once per transition and not once per
33
+ * poll. Per-instance and in-memory by design: a restart re-reports, which is the right amount of
34
+ * noise for a condition that is still broken after a restart.
35
+ */
36
+ private readonly reportedUnevaluableConditions;
37
+ /** Resolved once it EXISTS; null while it does not, so a fresh install is not cached blind. */
38
+ private cachedWorkflowTaskTypeID;
39
+ /**
40
+ * Graphs this instance is still trying to settle, with how many passes it has spent trying.
41
+ *
42
+ * **The sweep's window is on `__mj_UpdatedAt`, and a failing pass writes nothing** — the terminal
43
+ * write returns rowcount 0 because the row is already terminal, the layout pass touches only
44
+ * children, a refused CAS writes nothing at all. So a graph that fails to settle stops advancing
45
+ * its own timestamp and, after 24h of futile retries, ages out of the steady-state window while
46
+ * the process is up. The doc comment claimed the bound was "on abandonment, not age"; for this
47
+ * case it was on age, and R2-2's deferral made the case ordinary rather than exotic.
48
+ *
49
+ * In memory rather than a touch column because the alternative is a write on every failed
50
+ * attempt — more load exactly when something is already wrong — and because a restart is covered
51
+ * by the wide startup sweep, which is the durable backstop this leans on.
52
+ */
53
+ private readonly retryingSettlement;
54
+ /**
55
+ * Graphs whose settled-branch ANNOUNCEMENTS have already been made by this process.
56
+ *
57
+ * Re-entry is the point of the rescue, but only the parts that failed should repeat. Layout and
58
+ * the `GraphSettled` frame are idempotent facts about a finished graph, so a graph stuck in
59
+ * retry was re-persisting geometry and re-emitting the same frame every poll — for the whole
60
+ * 24h window, for as long as it kept failing.
61
+ */
62
+ private readonly announcedSettlements;
63
+ /** Graphs already reported as settled-but-undeliverable by this instance. */
64
+ private readonly reportedUndeliverable;
65
+ /** Live claim heartbeats by task ID, so the drain can silence the ones it gives up waiting for. */
66
+ private readonly heartbeats;
67
+ /** Latched once the drain has given up waiting, so a late arrival does not re-register. */
68
+ private heartbeatsPurged;
21
69
  private readonly conditionEvaluator;
22
70
  private running;
23
71
  private pollTimer;
24
72
  private reconcileTimer;
73
+ private unregisterKick;
25
74
  /** Tasks this instance is currently executing — bounds concurrency and drives heartbeats. */
26
75
  private readonly inFlight;
27
76
  /** Guards against a slow poll overlapping the next tick. */
28
77
  private polling;
78
+ /**
79
+ * Timer-driven passes currently running — poll and reconcile alike.
80
+ *
81
+ * A counter rather than a boolean because the two timers overlap by design, and `Stop()` has to
82
+ * wait for BOTH. Neither pass is held by anything else: they are launched `void`-ed from
83
+ * `setInterval`, so without this they are unobservable from the outside and a stopped dispatcher
84
+ * keeps writing.
85
+ */
86
+ private activePasses;
29
87
  /** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
30
88
  private readonly ownerByParentID;
89
+ /** Monotonic pass counter for `PassCompleted` frames, so a viewer can order and gap-detect ticks. */
90
+ private passCounter;
91
+ /**
92
+ * Debug state per graph, cached for ONE pass. `pollOnce` clears it at entry, so within a pass
93
+ * the claim filter and the propagation loop read the same state (loading it twice could see a
94
+ * pause land between them and gate half a pass), and across passes a control verb written by any
95
+ * instance is picked up within one poll interval.
96
+ */
97
+ private readonly debugStateByGraph;
98
+ /**
99
+ * The pause state last announced per graph, so `GraphPaused`/`GraphResumed` are emitted on the
100
+ * TRANSITION rather than every pass — the verbs write durable state, not events, and it is this
101
+ * instance's job to notice the change and say so exactly once.
102
+ */
103
+ private readonly announcedPaused;
104
+ /**
105
+ * The verdict last emitted per gating edge. `GateDecision` frames announce CHANGES: edges are
106
+ * re-resolved every pass, and an unconditional emission would repeat every few seconds for as
107
+ * long as the graph lives — the frame-topic version of the log flood
108
+ * `logUnevaluableConditionOnce` exists to prevent.
109
+ *
110
+ * Nested by graph so a settled run's entries go in one delete. Flat per-edge maps in a process
111
+ * that runs for weeks are a slow leak with no upper bound but the table's size.
112
+ */
113
+ private readonly emittedGateVerdicts;
114
+ /** Last `NodeProgress` emission per task, for rate limiting chatty runners. Nested by graph. */
115
+ private readonly nodeProgressLastEmit;
116
+ /** Minimum interval between `NodeProgress` frames for one task. */
117
+ private static readonly NODE_PROGRESS_MIN_INTERVAL_MS;
31
118
  constructor(providerFactory: ProviderFactory, agentRunner: TaskAgentRunner, contextUser: UserInfo, config: Partial<TaskGraphDispatcherConfig> & Pick<TaskGraphDispatcherConfig, 'InstanceID'>,
32
119
  /**
33
120
  * Optional. Absent means a host that cannot post messages or start agent turns — a worker,
@@ -39,7 +126,17 @@ export declare class TaskGraphDispatcher implements IShutdownable {
39
126
  * Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
40
127
  * announces nothing.
41
128
  */
42
- observer?: TaskGraphObserver);
129
+ observer?: TaskGraphObserver,
130
+ /**
131
+ * Optional. Absent means this host cannot run action nodes; they stay Pending and visible
132
+ * rather than being failed, because "nobody here can run this" is not "this ran and broke".
133
+ */
134
+ actionRunner?: TaskActionRunner,
135
+ /**
136
+ * Optional. Absent means this host cannot run prompt nodes; they stay Pending and visible
137
+ * rather than being failed, for the same reason action nodes do.
138
+ */
139
+ promptRunner?: TaskPromptRunner);
43
140
  /**
44
141
  * Announce something that happened, and never let the announcement matter.
45
142
  *
@@ -48,6 +145,15 @@ export declare class TaskGraphDispatcher implements IShutdownable {
48
145
  * careful means one place enforces it.
49
146
  */
50
147
  private emit;
148
+ /**
149
+ * A progress sink for one task's runner, rate-limited into `NodeProgress` frames.
150
+ *
151
+ * Rate-limited HERE rather than asking every runner to be polite, for the same reason `emit`
152
+ * swallows observer throws in one place: a chatty runner (an agent streaming token-level
153
+ * updates) must not be able to flood the topic, and the limit belongs to the announcement, not
154
+ * the work. A 100% report always passes — the terminal update is the one a viewer must not lose.
155
+ */
156
+ private nodeProgressEmitter;
51
157
  /**
52
158
  * Who a graph belongs to, memoized for the process's lifetime.
53
159
  *
@@ -61,6 +167,33 @@ export declare class TaskGraphDispatcher implements IShutdownable {
61
167
  * Skipped entirely when nobody is observing — the lookup exists only to address frames.
62
168
  */
63
169
  private resolveOwner;
170
+ /**
171
+ * A graph's debug state, cached for the current pass.
172
+ *
173
+ * Read fresh (BypassCache) because the state is written by direct `JSON_MODIFY` statements that
174
+ * fire no cache invalidation — the same reason every row read in this class bypasses the cache.
175
+ */
176
+ private readDebugState;
177
+ /**
178
+ * Reads the debug bag for a set of graphs in ONE query, priming the per-pass cache.
179
+ *
180
+ * Batched because both loops in a pass — propagation and claiming — walk every active graph, so
181
+ * a per-graph read made observability cost scale with the number of live workflows on a path
182
+ * that is meant to be flat. One `ID IN (…)` per pass costs the same whether a server is running
183
+ * one workflow or fifty.
184
+ *
185
+ * Already-cached graphs are skipped, so calling this from both loops is free the second time.
186
+ */
187
+ private primeDebugStates;
188
+ /**
189
+ * Announces a graph's pause-state TRANSITION, once, whichever instance notices first in its own
190
+ * frame stream.
191
+ *
192
+ * Per-instance dedup rather than a CAS: frames are advisory commentary, and a viewer receiving
193
+ * the transition from two instances is a duplicate line, not a duplicate execution — the price
194
+ * of a cross-instance guard here would be a write on every pass for a purely cosmetic guarantee.
195
+ */
196
+ private announcePauseTransition;
64
197
  /**
65
198
  * Begins dispatching.
66
199
  *
@@ -70,7 +203,23 @@ export declare class TaskGraphDispatcher implements IShutdownable {
70
203
  */
71
204
  Start(): Promise<void>;
72
205
  /**
73
- * Stops accepting new work and waits for in-flight tasks to finish.
206
+ * Stops accepting new work and waits for everything already started to finish.
207
+ *
208
+ * **"Everything" includes the timer passes, and that is the fix.** This waited only on
209
+ * `inFlight` — the task executions — while a poll pass is a `void`-ed promise nothing held. So
210
+ * `Stop()` returned while a pass was mid-flight, and that pass went on to settle graphs, emit
211
+ * lifecycle frames and CLAIM NEW TASKS afterwards. Three consequences, all of them quiet:
212
+ *
213
+ * - a `GraphSettled` frame arrived after every subscriber had gone, so the settlement was
214
+ * invisible to exactly the viewer watching for it;
215
+ * - a process shutting down claimed work it was about to abandon, leaving claims to expire —
216
+ * the orphaned-claim state reconciliation exists to clean up, manufactured by the shutdown;
217
+ * - the host reused the connection the moment `Stop()` resolved, and the still-running pass's
218
+ * statements collided with it (`Requests can only be made in the LoggedIn state`).
219
+ *
220
+ * A pass is bookkeeping for work that already happened, so it is DRAINED rather than cancelled:
221
+ * abandoning one halfway is the crash window the unsettled sweep exists to rescue, and choosing
222
+ * to open it on every clean shutdown would be perverse.
74
223
  *
75
224
  * Deliberately does NOT release claims on the way out: an abandoned claim expires on its own,
76
225
  * and releasing eagerly would hand a still-running task to another instance mid-execution.
@@ -89,6 +238,23 @@ export declare class TaskGraphDispatcher implements IShutdownable {
89
238
  * that shape indicates tampering or a bug and Record Changes already carries the audit trail.
90
239
  */
91
240
  Reconcile(): Promise<void>;
241
+ /**
242
+ * Announces expired claims the sweep just released, so a viewer watching the graph sees "the
243
+ * step's worker vanished and the engine requeued it" as it happens.
244
+ *
245
+ * Best-effort by contract: the release already succeeded and is the durable truth; a frame that
246
+ * cannot be addressed (row unloadable, no parent) is dropped, never retried.
247
+ */
248
+ private announceReclaims;
249
+ /** The reconciliation body. Wrapped by {@link Reconcile} so `Stop()` can drain it. */
250
+ private reconcileOnce;
251
+ /**
252
+ * Run a pass now instead of waiting for the next poll tick.
253
+ *
254
+ * Submit calls this (via {@link KickTaskGraphDispatchers}) so a just-written graph is claimed
255
+ * in milliseconds rather than up to {@link TaskGraphDispatcherConfig.PollIntervalSeconds}.
256
+ */
257
+ Kick(): void;
92
258
  /**
93
259
  * One dispatch pass: find claimable work, claim what fits under the concurrency cap, execute.
94
260
  *
@@ -110,6 +276,105 @@ export declare class TaskGraphDispatcher implements IShutdownable {
110
276
  * graph is wedged — are delegated to the pure algorithms, unchanged from Phase 1.
111
277
  */
112
278
  private propagateAndRollup;
279
+ /**
280
+ * Credits a finished graph's spending back to the agent run that submitted it.
281
+ *
282
+ * **Why this cannot happen during the run.** `BaseAgent` totals a run by walking its steps in
283
+ * memory at finalization — but a submitting run *ends at submission*. Submit-and-detach is the
284
+ * point: the run returns as soon as the graph is durable, and the graph executes afterwards,
285
+ * possibly minutes later on a different instance. At the moment the run computes its totals the
286
+ * spending has not happened yet, so there is nothing to count. The only place the number can be
287
+ * known is here, when the graph settles.
288
+ *
289
+ * **Why the `…Rollup` columns and not the plain ones.** `AIAgentRun` has carried six `…Rollup`
290
+ * columns since v3 that nothing has ever written — they exist for exactly this distinction:
291
+ *
292
+ * - `TotalCost` — what the run itself spent. For a Flow agent that is genuinely near zero: it
293
+ * compiled a graph and handed it off. This value is already final and is never rewritten here,
294
+ * so nothing that reads it today changes meaning, and no guardrail that already evaluated
295
+ * against it is retroactively falsified.
296
+ * - `TotalCostRollup` — the run plus everything it caused. Provisional until the graph settles,
297
+ * which is now.
298
+ *
299
+ * **The tree is the authority; these columns are its settlement-time cache.** The total is a SUM
300
+ * over `GetAgentRunTree`, not arithmetic of its own. The previous version walked the graph's
301
+ * child tasks and added each one's agent run, which was wrong in two ways that no test could
302
+ * see: a `Prompt` task has no agent run at all, so every prompt step's spend was simply missing;
303
+ * and it read each nested run's `…Rollup ?? …Total`, mixing a descendant-inclusive number with an
304
+ * own-spend one and depending on whether that nested graph happened to have settled yet. The
305
+ * tree already models every one of those cases — it reaches prompt runs through
306
+ * `Configuration.runtime.promptRunID`, and it descends into nested runs and their graphs
307
+ * structurally — so summing it cannot disagree with what the run viewer shows, because it IS
308
+ * what the run viewer shows.
309
+ *
310
+ * **This refuses rather than guesses.** A tree that failed to load, hit the depth cap, or does
311
+ * not contain the settling graph would still produce a number — a lower bound. Writing one would
312
+ * put an authoritative-looking total in a column every cost surface reads. Each of those cases
313
+ * logs and leaves the column alone, so `?? TotalCost` keeps its honest meaning: not settled.
314
+ *
315
+ * A graph with no submitting run (a scheduled job, a remote-operation caller) simply has nobody
316
+ * to credit — its own Task rows still carry the truth, and this returns quietly.
317
+ */
318
+ private rollUpCostToSubmittingRun;
319
+ /**
320
+ * Clears a rollup that can no longer be trusted, and says why.
321
+ *
322
+ * **Why clear rather than leave.** The four `…Rollup` columns are a cache of the run tree, and
323
+ * every reader treats a value there as the total. When the tree cannot be summed, any value
324
+ * already in the column was computed from an EARLIER settlement — it excludes the graph that
325
+ * just finished, so it is not merely incomplete, it is a wrong total presented as a right one.
326
+ * `?? TotalCost` protects a reader from null, not from stale.
327
+ *
328
+ * Nulling restores the invariant this whole design rests on: **when the column is present, it
329
+ * equals the tree.** Absent means not settled, which is exactly what a reader should conclude.
330
+ * A run with no rollup yet is untouched — there is nothing stale to clear, and writing nulls
331
+ * over nulls would churn Record Changes for nothing.
332
+ */
333
+ private clearStaleRollup;
334
+ /**
335
+ * Whether the settling graph is actually reachable from the submitting run's tree.
336
+ *
337
+ * Matched on the graph's parent Task id, which is the node the `TaskGraph` member of the query
338
+ * emits. A run that submitted a graph but recorded no `parentTaskID` produces a tree that stops
339
+ * at the run — structurally indistinguishable, at the SUM, from a run that never dispatched
340
+ * anything. This is the check that tells those two apart.
341
+ */
342
+ private treeContainsGraph;
343
+ /**
344
+ * Ends a graph early because a prompt said the work is finished.
345
+ *
346
+ * **Why `Skipped` and not `Cancelled`.** Nothing went wrong and nobody intervened — the workflow
347
+ * reached its own conclusion before running every drawn step, which is exactly what a reasoning
348
+ * step is for. `Cancelled` would tell a reader someone stopped it; `Skipped` says these routes
349
+ * were not taken, which is true and already the vocabulary the fork machinery uses.
350
+ *
351
+ * The message is written to the parent so the graph carries its own answer, rather than the
352
+ * answer living only on the step that produced it.
353
+ */
354
+ private endGraphEarly;
355
+ /**
356
+ * How deep the continuation chain already is, read from the graph's parent metadata.
357
+ *
358
+ * A run started by a graph inherits that graph's depth **plus one**. Without this every spawned
359
+ * run begins at zero, so a self-referencing flow — one that dispatches a graph containing itself
360
+ * — recurses without bound while the cap it should be hitting compares against a permanent zero.
361
+ */
362
+ private graphContext;
363
+ /**
364
+ * Which failures the workflow drew a way out of.
365
+ *
366
+ * A Failed task with a **satisfied outgoing edge** is a handled failure: its author drew a
367
+ * recovery route and that route is now live. Downstream work should be released along it, and the
368
+ * parent should not roll up Failed because of a step the workflow explicitly planned around.
369
+ *
370
+ * Scoped to `failureSemantics: 'edges'` on purpose. Under `'block'` — every agent-emitted graph —
371
+ * a failure is terminal for its dependents whatever edges exist, because nobody drew those edges
372
+ * as a recovery path; they are ordinary sequencing, and treating them as recovery would let a
373
+ * graph sail past a failure it never anticipated.
374
+ */
375
+ private computeHandledFailures;
376
+ /** The graph's child tasks, with the fields the rollup needs. */
377
+ private loadChildTasks;
113
378
  /**
114
379
  * Runs the graph's continuation exactly once, now that it has settled.
115
380
  *
@@ -121,8 +386,8 @@ export declare class TaskGraphDispatcher implements IShutdownable {
121
386
  * user-visible noise and, for `reinvoke`, an unbounded agent-run loop. Given one of the two has
122
387
  * to be chosen, the quiet failure is the safe one.
123
388
  *
124
- * The marker is written with a compare-and-swap read-back, so two instances reconciling the same
125
- * completed graph produce one winner rather than two.
389
+ * The marker is claimed with a real compare-and-swap (one guarded UPDATE, rowcount as verdict),
390
+ * so two instances reconciling the same completed graph produce one winner rather than two.
126
391
  */
127
392
  private deliverContinuation;
128
393
  /** Reads the parent's durable continuation metadata through the shared parser. */
@@ -133,6 +398,12 @@ export declare class TaskGraphDispatcher implements IShutdownable {
133
398
  * `MJ: Tasks` stays user-writable (D20), so a plain "read, decide, write" is not enough — the
134
399
  * read-back is what makes a lost race observable instead of producing a duplicate delivery.
135
400
  */
401
+ /**
402
+ * True when this graph finished so long ago that announcing it would surprise rather than inform.
403
+ *
404
+ * Measured from the parent's completion, not from when we noticed: the point is how stale the
405
+ * NEWS is to whoever would receive it.
406
+ */
136
407
  private claimContinuation;
137
408
  /** One line describing how the graph ended, for the completion log and message delivery. */
138
409
  private buildContinuationSummary;
@@ -161,6 +432,40 @@ export declare class TaskGraphDispatcher implements IShutdownable {
161
432
  * invalidate. Left cached, a completed task keeps reading as `In Progress` and the graph never
162
433
  * rolls up: submitted work simply never settles.
163
434
  */
435
+ /**
436
+ * Settles graphs that reached terminal without completing their post-settlement sequence.
437
+ *
438
+ * Runs the ordinary propagation path, which is safe to re-enter by construction: the terminal
439
+ * write is guarded on not-already-terminal, the cost rollup assigns rather than accumulates, run
440
+ * settlement is guarded on `Paused`, and delivery is guarded by the continuation CAS. A revisit
441
+ * therefore corrects whatever is missing and does nothing where nothing is.
442
+ *
443
+ * @param windowHours how far back to look — wide once at startup, narrow in steady state
444
+ */
445
+ private sweepUnsettledGraphs;
446
+ /**
447
+ * The `AI Workflow` task type, resolved once per process.
448
+ *
449
+ * `MJ: Tasks` is a GENERAL-PURPOSE entity — conversations and user to-dos live there too — so an
450
+ * unscoped sweep treats every root task hierarchy as a workflow: rolling up and overwriting the
451
+ * status of somebody's to-do list, raising agent requests against plain tasks, and (once the
452
+ * continuation CAS exists) injecting marker keys into a user's own `InputPayload`.
453
+ *
454
+ * `Submit` has always stamped this type on the parent and every child (`ensureTaskType`, which
455
+ * runs before the persist transaction), so the discriminator D3 called for already exists on
456
+ * every dispatcher-owned row. Verified against the live database: every parent graph carries it.
457
+ *
458
+ * Null when the type row does not exist yet — no graph has ever been submitted — in which case
459
+ * there is nothing for the dispatcher to find and the sweep returns empty rather than unscoped.
460
+ *
461
+ * **A miss is never cached**, and that is not a micro-optimisation. `TaskGraphService.Submit`
462
+ * creates the row on first use, so on a fresh install the ordinary sequence is: dispatcher
463
+ * starts, looks, finds nothing — then somebody submits the first workflow. Caching that first
464
+ * `null` would blind this process to every graph until it was restarted, with each poll reporting
465
+ * a clean, empty sweep. The row is created once and never removed, so the retry costs one
466
+ * `MaxRows: 1` lookup per poll for exactly as long as there is genuinely nothing to dispatch.
467
+ */
468
+ private workflowTaskTypeID;
164
469
  private findActiveGraphIDs;
165
470
  /**
166
471
  * Tasks eligible to claim right now, across all active graphs.
@@ -170,8 +475,132 @@ export declare class TaskGraphDispatcher implements IShutdownable {
170
475
  * the same rule, free to drift from the one the in-run executor uses.
171
476
  */
172
477
  private findClaimableTasks;
478
+ private clearSkipBreakpoint;
479
+ /**
480
+ * Drops the per-graph frame-dedup state for a graph that has settled.
481
+ *
482
+ * `ownerByParentID` is deliberately NOT purged here: it is the delivery key for the
483
+ * `GraphSettled` frame emitted moments earlier and for any rescue-sweep pass that revisits the
484
+ * graph, it is one small string per graph, and ownership never changes — the cost of keeping it
485
+ * is bounded and the cost of losing it is a re-query on a path that is meant to be cheap.
486
+ */
487
+ private forgetGraphObservability;
488
+ /**
489
+ * Whether THIS host can act on a task right now — the runner-availability question, asked
490
+ * without acting on it.
491
+ *
492
+ * Mirrors the checks in the claim loop so a step allowance is spent only when something will
493
+ * actually move. A human step counts as actionable: stepping onto one legitimately produces a
494
+ * notification rather than a claim.
495
+ */
496
+ private canActOn;
497
+ /**
498
+ * Says why a step press released nothing, instead of leaving the allowance spent and the
499
+ * console silent.
500
+ *
501
+ * The allowance is deliberately NOT consumed on this path — the operator's intent stands, and
502
+ * the step will release as soon as the named work becomes actionable (a runner arrives, an
503
+ * in-flight task finishes). Announced once per pass rather than logged only, because the person
504
+ * waiting is looking at the console, not the server log.
505
+ */
506
+ private reportStepReleasedNothing;
507
+ /**
508
+ * Marks a human task as notified, so the request is raised exactly once.
509
+ *
510
+ * Written even when delivery threw. Retrying on every poll is a worse failure than one missed
511
+ * notification: the task stays visible in the inbox either way, whereas a notification storm is
512
+ * not self-correcting.
513
+ */
514
+ private markHumanTaskNotified;
515
+ /**
516
+ * Raises the `MJ: AI Agent Requests` row a person answers to release this step.
517
+ *
518
+ * **Why that entity rather than something new.** It already models everything a workflow's human
519
+ * step needs — who is being asked, what for, a typed response schema, priority, expiry, and an
520
+ * inbox surface people already use. A second HITL substrate beside it would split the inbox in
521
+ * two and leave one of them without expiry or permissions.
522
+ *
523
+ * **What it deliberately does NOT set is `ResumingAgentRunID`.** A request normally suspends an
524
+ * agent run and resumes it. A workflow needs none of that: the graph OUTLIVES the run that
525
+ * submitted it, so nothing is suspended — the task sits Pending, every other branch keeps
526
+ * running, and answering settles the task. That column staying null is meaningful, not missing.
527
+ */
528
+ private raiseHumanRequest;
529
+ /**
530
+ * The agent that owns this task's workflow — who the request is asked on behalf of.
531
+ *
532
+ * Reads the graph's parent row, falling back to the run that submitted it. A human step has no
533
+ * agent of its own by design: `AgentID` names what EXECUTES a step, and a person is not an agent.
534
+ */
535
+ private owningAgentOf;
536
+ /**
537
+ * Every still-open request for a task, oldest first.
538
+ *
539
+ * Plural, and ordered, for one reason each. Ordered, because the oldest row is the one every
540
+ * instance must agree is "the" request — it is the one the assignee most likely already saw,
541
+ * and the one `withdrawDuplicateRequests` keeps; unordered, two instances could each decide a
542
+ * different duplicate was the keeper and withdraw each other's. Plural, because a caller that
543
+ * only ever sees the first cannot notice there are two, which is how the duplicate below
544
+ * survived: every reader of this took `[0]` and moved on.
545
+ */
546
+ private findOpenRequests;
547
+ /**
548
+ * Settles a human task from the request a person answered.
549
+ *
550
+ * Runs on the poll rather than on a save hook, because the answer can arrive through any surface
551
+ * — the inbox, the API, a conversation — and only the dispatcher knows how to release the rest
552
+ * of the graph afterwards.
553
+ *
554
+ * **`ResponseData` becomes the task's output.** That is what makes a human step useful rather
555
+ * than a gate: a downstream edge can branch on what the person actually said, typed by the
556
+ * request's own ResponseSchema. A step that only recorded "approved" would force every decision
557
+ * back into a separate action.
558
+ */
559
+ private settleAnsweredHumanTasks;
560
+ /**
561
+ * Reconciles the requests behind human steps that are waiting on somebody.
562
+ *
563
+ * Two things can be wrong with a waiting step, and both are silent. It can have NO open request
564
+ * — the cancel case below — or it can have MORE than one, which the raise cannot fix because it
565
+ * never runs again for a notified task. Both are corrected here, on the only sweep that visits
566
+ * these tasks every pass.
567
+ *
568
+ * **Re-opening a human step whose request was CANCELLED.**
569
+ *
570
+ * `answeredRequestFor` deliberately excludes `Canceled`, because cancelling withdraws the ASK
571
+ * rather than deciding the step — the task is supposed to keep waiting "for whatever replaces
572
+ * it". Nothing replaced it. `raiseHumanRequest` refuses to raise twice (the notified marker on
573
+ * `ClaimedBy` is what stops the notification storm), so a cancelled request left the task Pending
574
+ * with no open request and no path to acquiring one: a workflow waiting forever on a question
575
+ * nobody is being asked.
576
+ *
577
+ * Clearing the marker is the whole fix — the next poll sees an un-notified Pending human task
578
+ * and raises a fresh request, which is exactly the replacement the design assumed. Bounded by
579
+ * human action: it takes another person cancelling again to come back here.
580
+ */
581
+ private reconcileWaitingHumanTasks;
582
+ /** The answered (or expired) request for a task, if any. */
583
+ private answeredRequestFor;
584
+ /**
585
+ * Expires requests whose deadline has passed.
586
+ *
587
+ * A deadline that nothing enforces is a comment. Without this an `ExpiresAt` in the past leaves
588
+ * the request `Requested` forever and the workflow waiting on it just as long.
589
+ */
590
+ private expireOverdueRequests;
591
+ /** The agent run that submitted this task's graph, for provenance on the request. */
592
+ private submittingRunOf;
173
593
  /** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
174
594
  private loadGraphState;
595
+ /**
596
+ * Reports an unevaluable condition ONCE per edge, not once per poll.
597
+ *
598
+ * Eligibility is recomputed every cycle, so an unqualified LogError here would repeat every few
599
+ * seconds for as long as the graph is held — which buries the one line that matters under
600
+ * thousands of copies of itself. Keyed by edge id plus the failure text, so a condition that
601
+ * starts failing differently is reported again.
602
+ */
603
+ private logUnevaluableConditionOnce;
175
604
  /**
176
605
  * Decides whether a conditional dependency edge is live.
177
606
  *
@@ -180,7 +609,241 @@ export declare class TaskGraphDispatcher implements IShutdownable {
180
609
  * an unevaluable condition keeps the edge for the reason stated at the call site.
181
610
  */
182
611
  private evaluateEdgeCondition;
612
+ /**
613
+ * An exclusive edge's condition as a three-way outcome.
614
+ *
615
+ * `ResolveExclusiveGroups` needs to tell "false" from "could not be evaluated": the first loses
616
+ * the branch, the second holds the whole group. The generic keep/drop path cannot express that
617
+ * difference, which is why exclusive edges take this route instead.
618
+ */
619
+ private evaluateExclusiveCondition;
620
+ /**
621
+ * Announces gate verdicts that CHANGED since this instance last looked.
622
+ *
623
+ * Fire-and-forget by design: `loadGraphState` is synchronous graph assembly, and the owner
624
+ * lookup the frame needs is async — so the emission floats behind rather than making state
625
+ * loading wait on observability. Frames are commentary, never a step of the work.
626
+ */
627
+ private emitGateDecisions;
183
628
  /** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
184
629
  private loadDependencyOutputs;
630
+ /**
631
+ * Runs one task's body, whatever kind of step it is.
632
+ *
633
+ * **Routing is on `StepType`, not on which key happens to be set.** A loop step carries the same
634
+ * `ActionID` or `AgentID` as an ordinary step — that key is what the loop *repeats* — so the old
635
+ * `task.ActionID ? action : agent` test would have run a loop exactly once and called it done.
636
+ * `StepType` is the only field that distinguishes them.
637
+ *
638
+ * Every branch is normalized to one shape so the recording path above stays single: an action has
639
+ * no agent run to point at, because its forensics live in `ActionExecutionLog` instead.
640
+ */
641
+ private runTaskBody;
642
+ /**
643
+ * Runs a loop step: its body once per iteration, with the item and index in scope.
644
+ *
645
+ * The loop's own `Configuration` supplies the definition; the row's `ActionID` / `AgentID`
646
+ * supplies what to repeat. Per-iteration inputs are resolved fresh each pass — the bindings are
647
+ * merged into the payload before the mapping is applied, which is how a body can reference the
648
+ * current item at all.
649
+ */
650
+ private runLoopTask;
651
+ /**
652
+ * Folds one pass's result into the loop's running payload.
653
+ *
654
+ * **With a body mapping**, the pass's declared outputs are filed where the author said to put
655
+ * them — including `name[]`, which appends, so a ForEach can collect one entry per item. That is
656
+ * the whole point of a loop over a collection, and it is only expressible per pass.
657
+ *
658
+ * **Without one**, the raw result is deep-merged, which is the pre-existing behaviour and the
659
+ * right default for a `While` that converges on a value: each pass refines what the condition
660
+ * reads. It is the wrong default for a ForEach that collects — hence the mapping.
661
+ *
662
+ * An unmapped output is reported per pass rather than swallowed, for the same reason
663
+ * {@link applyStepOutputMapping} reports it: a mapping that names something the body never
664
+ * returned means the pass did work that went nowhere, while everything reports success.
665
+ */
666
+ private foldIterationOutput;
667
+ /**
668
+ * Files a step's result into the payload it hands downstream.
669
+ *
670
+ * **This is what makes a branch condition possible.** A workflow that branches on
671
+ * `payload.stockPrice` has that value only because this step mapped `CurrentPrice -> stockPrice`.
672
+ * Without it the condition reads `undefined` — merely falsy — so the workflow takes the other
673
+ * branch, finishes, and reports success with nothing to indicate anything went wrong.
674
+ *
675
+ * The incoming payload is carried through as well as the update, so a value written three steps
676
+ * back is still readable here. Returning only this step's own output is what used to limit a
677
+ * condition's view to its immediate predecessor.
678
+ */
679
+ private applyStepOutputMapping;
680
+ /**
681
+ * Runs an Agent step, telling the runner where in the graph it sits.
682
+ *
683
+ * Depth and provenance are read together because they come from the same row: the graph's parent
684
+ * task knows both how many continuation hops led here and which run submitted it.
685
+ */
686
+ private runAgentNode;
687
+ /**
688
+ * Leaves exactly one open request standing for a task, withdrawing any others.
689
+ *
690
+ * The oldest wins — it is the one whose notification the assignee most likely already saw.
691
+ *
692
+ * **Called from every path that reads a task's open requests**, not only from the raise. That is
693
+ * deliberate and it is the half R3-5 first got wrong: a task is notified exactly once, and
694
+ * `notifyHumanTaskReady` returns at the marker forever after, so a duplicate minted after that
695
+ * pass — by an instance that crashed between its insert and its de-dup, or by any build older
696
+ * than this one — was never looked at again by the only code that could have collapsed it. The
697
+ * waiting-task sweep is what actually reaches those.
698
+ *
699
+ * @param open the task's open requests, oldest first
700
+ * @param keepIfSole the row this caller is responsible for, named only for the log
701
+ */
702
+ private withdrawDuplicateRequests;
703
+ /**
704
+ * Closes the still-open asks raised for tasks that will never be answered.
705
+ *
706
+ * `Canceled` rather than `Expired`: nobody ran out of time, the ask was withdrawn — and the two
707
+ * mean different things downstream, since an expired human step is treated as a FAILURE that a
708
+ * give-up edge can route around, which would be a lie about a step the workflow decided it no
709
+ * longer needed.
710
+ *
711
+ * Failures are logged and never propagated. The graph's outcome is already decided; refusing to
712
+ * finish over an inbox row would trade a stale notification for a stalled workflow.
713
+ */
714
+ private withdrawOpenRequests;
715
+ /**
716
+ * Says once, per graph, that this instance settled work it cannot announce.
717
+ *
718
+ * Once because the sweep re-offers the graph every poll for the rest of its window, and a line
719
+ * per poll would bury the thing it is trying to report — which is a DEPLOYMENT fact, not a graph
720
+ * fact: if no instance anywhere carries a deliverer, these settlements never reach anyone.
721
+ */
722
+ private reportUndeliverableOnce;
723
+ /**
724
+ * Keeps a graph in this instance's sweep regardless of what its row timestamp says.
725
+ *
726
+ * Bounded, and the bound is about noise rather than surrender: past the cap the graph has failed
727
+ * on every attempt for minutes, so another identical attempt will not fix it, and continuing
728
+ * costs a full graph load per poll forever. It is reported once and left to the startup sweep.
729
+ */
730
+ private keepRetryingSettlement;
731
+ /**
732
+ * A graph's durable metadata bag, for the questions a pass asks of it.
733
+ *
734
+ * Defaults on any failure to read it, and the defaults are the safe directions: `'block'` means
735
+ * a failed step decides nothing, so a graph whose metadata we cannot read stalls visibly instead
736
+ * of resolving forks on the say-so of a failure; and no early-finish declaration means nothing
737
+ * is removed from the claim filter on the strength of a read that did not work.
738
+ */
739
+ private readParentMetadataFor;
740
+ /**
741
+ * Whether the submitting run is in a state where this pass's writes to it will mean anything.
742
+ *
743
+ * **Read-only on purpose.** The settled branch's write order — layout, frame, cost, lifecycle,
744
+ * delivery — is load-bearing and documented at each step; this asks the question those writes
745
+ * depend on without joining them. What it prevents is a pass that goes through the motions and
746
+ * then claims the delivery marker, making itself the last pass ever to look at the graph.
747
+ *
748
+ * Three answers, and the middle one is the bug:
749
+ *
750
+ * - **no run** — a scheduled or remote-triggered graph has nobody waiting. Proceed.
751
+ * - **still `Running`** — `finalizeAgentRun` has not parked it yet. The graph beat its own
752
+ * submitter to the finish line, which is ordinary for a fast graph and lasts milliseconds.
753
+ * Defer: one poll later the run is parked and everything lands.
754
+ * - **anything else** — `Paused` (settle it), or already `Completed`/`Failed`/`Cancelled` for
755
+ * its own reasons (leave it; the lifecycle write's own guard declines). Proceed.
756
+ *
757
+ * **The deferral is bounded**, because "not parked yet" and "the submitting process died before
758
+ * it could park" look identical from here. Waiting forever on the second would lose the outcome
759
+ * of work that actually completed — strictly worse than announcing it late — so past the grace
760
+ * period this proceeds and says why. The run itself stays `Running`, which is visibly wrong and
761
+ * belongs to whatever reconciles abandoned runs, not to the graph that finished correctly.
762
+ */
763
+ private submittingRunReadiness;
764
+ /**
765
+ * Completes the agent run that parked on this graph.
766
+ *
767
+ * **This is the other half of submit-and-detach.** A run that dispatches a graph does not
768
+ * complete at submission — it ends `Paused`, because reporting `Completed` above a workflow
769
+ * where nothing has happened yet is a claim the row cannot support. The run's lifecycle is
770
+ * finished HERE, when the graph it was waiting on actually settles, which is the first moment
771
+ * the answer exists.
772
+ *
773
+ * Doing it from the dispatcher rather than by awaiting in the agent is what keeps the properties
774
+ * that made detach right in the first place: a graph containing a human approval can park for
775
+ * days without holding a conversation turn open, and a graph reclaimed by another instance after
776
+ * a crash still settles its submitting run, because the settling happens wherever the graph
777
+ * finishes rather than wherever it started.
778
+ *
779
+ * **Only a parked run is touched.** A run that is already `Completed`, `Failed` or `Cancelled`
780
+ * reached that state for its own reasons — a second graph settling later, a run the user
781
+ * cancelled, a run that failed after submitting — and overwriting it would rewrite history from
782
+ * the outside. The `Paused` predicate is the whole guard.
783
+ *
784
+ * @param graphStatus the parent rollup's status: what the workflow as a whole did
785
+ */
786
+ private settleSubmittingRun;
787
+ /**
788
+ * Gives every step that lacks one a position, once the graph has finished.
789
+ *
790
+ * **Why the run stores geometry at all.** A `TaskGraphSpec` is a logical structure with no
791
+ * layout field, so a graph an agent emitted has no opinion about where its boxes go. Every
792
+ * viewer was therefore laying it out for itself at render time — and a viewer that failed to
793
+ * (because the canvas measures nodes it has not drawn yet) fell back to every node at the
794
+ * origin, piled on one another, with the zoom-to-fit that follows fitting a one-node bounding
795
+ * box. Settling it once, server-side, means the agent-run canvas, the Workflows runs tab and
796
+ * anything built later all draw the same picture, and none of them has to compute it.
797
+ *
798
+ * **An authored position is never overwritten.** A workflow compiled from a Flow agent carries
799
+ * the arrangement someone dragged into place; replacing it with an algorithm's guess would
800
+ * discard a deliberate act. Only steps with no geometry get one, so a partially-arranged graph
801
+ * keeps what it has.
802
+ *
803
+ * Failure here is logged and swallowed: this is presentation. A graph whose work completed must
804
+ * not be reported as failed because its picture could not be saved.
805
+ */
806
+ private persistComputedLayout;
807
+ /**
808
+ * The earliest moment any step in the graph began, or null when none has.
809
+ *
810
+ * Null is a real answer — a graph whose tasks are all still Pending has not started — and is
811
+ * deliberately not collapsed to "now", which would date the graph from whenever this pass
812
+ * happened to run.
813
+ */
814
+ private earliestStart;
815
+ /**
816
+ * The step's Configuration with this run's artefacts folded in, or `undefined` to leave it be.
817
+ *
818
+ * **Merged into the authored bag, never written over it.** The Configuration column holds the
819
+ * step's definition — its loop body, its mappings, its policy, the position someone dragged it
820
+ * to. Writing a fresh object containing only `runtime` would erase all of that the first time a
821
+ * prompt step completed, which is the kind of loss that surfaces much later as a workflow that
822
+ * mysteriously stopped mapping its output.
823
+ *
824
+ * Returns `undefined` when there is nothing to record, so the guarded write omits the column
825
+ * rather than rewriting it with what it already held.
826
+ */
827
+ private configurationWithRuntime;
828
+ /**
829
+ * Reads a step's Configuration bag, tolerating a row whose JSON cannot be parsed.
830
+ *
831
+ * Unparseable configuration is logged rather than thrown: the step has already RUN by the time
832
+ * this is called, and refusing to record its outcome because its definition is malformed would
833
+ * discard the result of real work and leave the task claimed until the claim lapsed.
834
+ */
835
+ private parseConfiguration;
836
+ /**
837
+ * The payload a step sees: everything its prerequisites produced, plus its own declared input.
838
+ *
839
+ * **Why the outputs are merged rather than kept per-task.** A flow carried ONE payload that
840
+ * accumulated as it went, so a condition on the edge into step C could read a value step A wrote.
841
+ * Handing each task only its immediate predecessor's output would silently narrow that: the
842
+ * condition reads `undefined`, which is falsy, and the workflow quietly takes a different route
843
+ * than the flow it was compiled from. Merging in dependency order restores the accumulation.
844
+ *
845
+ * Later prerequisites win on a key collision, matching a flow's own last-write-wins behaviour.
846
+ */
847
+ private mergedPayload;
185
848
  }
186
849
  //# sourceMappingURL=TaskGraphDispatcher.d.ts.map