@memberjunction/task-graph 0.0.0 → 6.1.0-edge.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. package/LICENSE +7 -0
  2. package/dist/DispatcherConditionEvaluator.d.ts +10 -0
  3. package/dist/DispatcherConditionEvaluator.d.ts.map +1 -0
  4. package/dist/DispatcherConditionEvaluator.js +27 -0
  5. package/dist/DispatcherConditionEvaluator.js.map +1 -0
  6. package/dist/TaskClaimStore.d.ts +114 -0
  7. package/dist/TaskClaimStore.d.ts.map +1 -0
  8. package/dist/TaskClaimStore.js +218 -0
  9. package/dist/TaskClaimStore.js.map +1 -0
  10. package/dist/TaskGraphDispatcher.d.ts +186 -0
  11. package/dist/TaskGraphDispatcher.d.ts.map +1 -0
  12. package/dist/TaskGraphDispatcher.js +735 -0
  13. package/dist/TaskGraphDispatcher.js.map +1 -0
  14. package/dist/TaskGraphService.d.ts +138 -0
  15. package/dist/TaskGraphService.d.ts.map +1 -0
  16. package/dist/TaskGraphService.js +314 -0
  17. package/dist/TaskGraphService.js.map +1 -0
  18. package/dist/TaskGraphSubmitterImpl.d.ts +7 -0
  19. package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -0
  20. package/dist/TaskGraphSubmitterImpl.js +52 -0
  21. package/dist/TaskGraphSubmitterImpl.js.map +1 -0
  22. package/dist/WorkflowSpecSync.d.ts +171 -0
  23. package/dist/WorkflowSpecSync.d.ts.map +1 -0
  24. package/dist/WorkflowSpecSync.js +393 -0
  25. package/dist/WorkflowSpecSync.js.map +1 -0
  26. package/dist/index.d.ts +18 -0
  27. package/dist/index.d.ts.map +1 -0
  28. package/dist/index.js +18 -0
  29. package/dist/index.js.map +1 -0
  30. package/dist/operations/TaskGraphOperations.d.ts +39 -0
  31. package/dist/operations/TaskGraphOperations.d.ts.map +1 -0
  32. package/dist/operations/TaskGraphOperations.js +164 -0
  33. package/dist/operations/TaskGraphOperations.js.map +1 -0
  34. package/dist/operations/WorkflowOperations.d.ts +22 -0
  35. package/dist/operations/WorkflowOperations.d.ts.map +1 -0
  36. package/dist/operations/WorkflowOperations.js +99 -0
  37. package/dist/operations/WorkflowOperations.js.map +1 -0
  38. package/dist/types.d.ts +214 -0
  39. package/dist/types.d.ts.map +1 -0
  40. package/dist/types.js +9 -0
  41. package/dist/types.js.map +1 -0
  42. package/package.json +33 -8
  43. package/README.md +0 -45
@@ -0,0 +1,735 @@
1
+ /**
2
+ * @fileoverview Durable, host-agnostic execution of submitted task graphs.
3
+ *
4
+ * The dispatcher is what makes task graphs survive things the old client-driven path could not: a
5
+ * page reload, the submitting agent run ending, a server restart, or a second server instance
6
+ * running the same table. It polls for claimable work, claims atomically, executes with a fresh
7
+ * provider per task, and reconciles orphaned state on a timer.
8
+ *
9
+ * **What it deliberately does not do.** It does not decide graph semantics. Eligibility, failure
10
+ * propagation, parent rollup and stall detection all come from the pure algorithms in
11
+ * `@memberjunction/ai-core-plus` — the same functions the in-run executor consumes. That is
12
+ * the whole reason those were factored out dependency-free: the in-run executor and the durable
13
+ * executor cannot drift apart if neither owns the rules.
14
+ *
15
+ * **Host-agnostic by construction.** Provider minting and agent execution arrive as injected
16
+ * dependencies (`ProviderFactory`, `TaskAgentRunner`), so this package never imports MJServer. The
17
+ * dependency runs MJServer -> task-graph, never the reverse.
18
+ *
19
+ * @module @memberjunction/task-graph
20
+ */
21
+ import { ComputeEligibleTasks, ComputeParentRollup, ComputeTasksToBlock, IsGraphStalled, } from '@memberjunction/ai-core-plus';
22
+ import { LogError, LogStatus, RunView } from '@memberjunction/core';
23
+ import { ShutdownRegistry } from '@memberjunction/global';
24
+ import { TaskClaimStore } from './TaskClaimStore.js';
25
+ import { DispatcherConditionEvaluator } from './DispatcherConditionEvaluator.js';
26
+ import { NotificationEngine } from '@memberjunction/notifications';
27
+ /** Metadata-seeded notification type for human tasks (metadata/notifications/.task-assignment-type.json). */
28
+ const HUMAN_TASK_NOTIFICATION_TYPE = 'Task Assignment';
29
+ /**
30
+ * Written to a human task's `ClaimedBy` once its assignee has been told it is ready.
31
+ *
32
+ * A human task has no executor, so the claim column is otherwise unused — which makes it the natural
33
+ * place to record a fact that must survive a restart. Reconciliation already exempts human tasks
34
+ * from reclamation, so this value is never mistaken for a live claim.
35
+ */
36
+ const HUMAN_TASK_NOTIFIED_MARKER = '__human-notified__';
37
+ import { IsReinvokeCapReached, MAX_REINVOKE_DEPTH, ParseTaskGraphParentMetadata } from './TaskGraphService.js';
38
+ import { DEFAULT_DISPATCHER_CONFIG, } from './types.js';
39
+ export class TaskGraphDispatcher {
40
+ constructor(providerFactory, agentRunner, contextUser, config,
41
+ /**
42
+ * Optional. Absent means a host that cannot post messages or start agent turns — a worker,
43
+ * a test. The dispatcher still records and logs every completion, so a graph's outcome is
44
+ * never lost; it simply is not announced.
45
+ */
46
+ continuationDeliverer,
47
+ /**
48
+ * Optional. Absent means nobody is watching — the dispatcher behaves identically, it just
49
+ * announces nothing.
50
+ */
51
+ observer) {
52
+ this.providerFactory = providerFactory;
53
+ this.agentRunner = agentRunner;
54
+ this.contextUser = contextUser;
55
+ this.continuationDeliverer = continuationDeliverer;
56
+ this.observer = observer;
57
+ this.running = false;
58
+ this.pollTimer = null;
59
+ this.reconcileTimer = null;
60
+ /** Tasks this instance is currently executing — bounds concurrency and drives heartbeats. */
61
+ this.inFlight = new Set();
62
+ /** Guards against a slow poll overlapping the next tick. */
63
+ this.polling = false;
64
+ /** Graph → owning user, from the parent's durable metadata. Ownership never changes, so this never goes stale. */
65
+ this.ownerByParentID = new Map();
66
+ /** Name shown in the shutdown drain log. */
67
+ this.ShutdownName = 'TaskGraphDispatcher';
68
+ this.config = { ...DEFAULT_DISPATCHER_CONFIG, ...config };
69
+ this.claims = new TaskClaimStore(this.config.InstanceID, this.config.ClaimTTLSeconds);
70
+ this.conditionEvaluator = new DispatcherConditionEvaluator();
71
+ }
72
+ /**
73
+ * Announce something that happened, and never let the announcement matter.
74
+ *
75
+ * A frame is commentary on work, never a step of it, so an observer that throws must not be able
76
+ * to fail a task or stall a graph. Swallowing here rather than asking every implementation to be
77
+ * careful means one place enforces it.
78
+ */
79
+ emit(frame) {
80
+ if (!this.observer)
81
+ return;
82
+ try {
83
+ this.observer.OnFrame(frame);
84
+ }
85
+ catch (e) {
86
+ LogError(`[TaskGraphDispatcher] Observer threw on ${frame.Kind} (ignored): ${e instanceof Error ? e.message : String(e)}`);
87
+ }
88
+ }
89
+ /**
90
+ * Who a graph belongs to, memoized for the process's lifetime.
91
+ *
92
+ * Read from the parent's durable metadata rather than a column, because `Task.UserID` means
93
+ * "the person this task is waiting on" — setting it on a parent would make every graph look
94
+ * like a human task. Memoized because frames are emitted per step: without the cache, watching
95
+ * a run would cost one query per event, and observability that scales with work is the thing a
96
+ * push mechanism exists to avoid. Ownership never changes for a given graph, so the cache can
97
+ * never go stale.
98
+ *
99
+ * Skipped entirely when nobody is observing — the lookup exists only to address frames.
100
+ */
101
+ async resolveOwner(provider, parentTaskID) {
102
+ if (!this.observer)
103
+ return null;
104
+ const cached = this.ownerByParentID.get(parentTaskID);
105
+ if (cached !== undefined)
106
+ return cached;
107
+ let owner = null;
108
+ try {
109
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
110
+ if (await parent.Load(parentTaskID)) {
111
+ owner = this.readParentMetadata(parent).submittedByUserID ?? null;
112
+ }
113
+ }
114
+ catch (e) {
115
+ LogError(`[TaskGraphDispatcher] Could not resolve owner for graph ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
116
+ }
117
+ this.ownerByParentID.set(parentTaskID, owner);
118
+ return owner;
119
+ }
120
+ /**
121
+ * Begins dispatching.
122
+ *
123
+ * Runs reconciliation FIRST, before accepting any new work. On a restart this instance may be
124
+ * looking at tasks its own previous incarnation claimed and never released — reclaiming those
125
+ * up front is what turns a crash from "work stranded forever" into "work resumes".
126
+ */
127
+ async Start() {
128
+ if (this.running)
129
+ return;
130
+ this.running = true;
131
+ // Self-register rather than make each host remember to stop us. A dispatcher that keeps
132
+ // polling through a graceful shutdown would claim work the process is about to abandon,
133
+ // which is exactly the orphaned-claim state reconciliation exists to clean up.
134
+ ShutdownRegistry.Instance.Register(this);
135
+ LogStatus(`[TaskGraphDispatcher] Starting as instance '${this.config.InstanceID}'.`);
136
+ await this.Reconcile();
137
+ this.pollTimer = setInterval(() => { void this.pollOnce(); }, this.config.PollIntervalSeconds * 1000);
138
+ this.reconcileTimer = setInterval(() => { void this.Reconcile(); }, this.config.ReconciliationIntervalSeconds * 1000);
139
+ }
140
+ /**
141
+ * Stops accepting new work and waits for in-flight tasks to finish.
142
+ *
143
+ * Deliberately does NOT release claims on the way out: an abandoned claim expires on its own,
144
+ * and releasing eagerly would hand a still-running task to another instance mid-execution.
145
+ * Letting the TTL do it is the safer failure mode.
146
+ */
147
+ async Stop() {
148
+ this.running = false;
149
+ if (this.pollTimer) {
150
+ clearInterval(this.pollTimer);
151
+ this.pollTimer = null;
152
+ }
153
+ if (this.reconcileTimer) {
154
+ clearInterval(this.reconcileTimer);
155
+ this.reconcileTimer = null;
156
+ }
157
+ const deadline = Date.now() + 30_000;
158
+ while (this.inFlight.size > 0 && Date.now() < deadline) {
159
+ await new Promise((r) => setTimeout(r, 250));
160
+ }
161
+ if (this.inFlight.size > 0) {
162
+ LogError(`[TaskGraphDispatcher] Stopped with ${this.inFlight.size} task(s) still in flight; their claims will expire.`);
163
+ }
164
+ LogStatus(`[TaskGraphDispatcher] Stopped.`);
165
+ }
166
+ /** {@link IShutdownable} — idempotent by way of `Stop`'s `running` guard. */
167
+ async Shutdown() {
168
+ await this.Stop();
169
+ }
170
+ /**
171
+ * Reclaims expired claims and reports anomalies.
172
+ *
173
+ * Also enforces the two schema promises that previously had no enforcer anywhere: agent tasks
174
+ * left `In Progress` with no claim are surfaced loudly rather than silently corrected, since
175
+ * that shape indicates tampering or a bug and Record Changes already carries the audit trail.
176
+ */
177
+ async Reconcile() {
178
+ let provider = null;
179
+ try {
180
+ provider = await this.providerFactory.CreateProvider();
181
+ const released = await this.claims.ReleaseExpiredClaims(provider, this.contextUser);
182
+ const orphaned = await this.claims.FindOrphanedInProgress(provider, this.contextUser);
183
+ if (released.length > 0 || orphaned.length > 0) {
184
+ LogStatus(`[TaskGraphDispatcher] Reconciliation: ${released.length} expired claim(s) released, ` +
185
+ `${orphaned.length} orphaned task(s) reported.`);
186
+ }
187
+ }
188
+ catch (e) {
189
+ LogError(`[TaskGraphDispatcher] Reconciliation failed: ${e instanceof Error ? e.message : String(e)}`);
190
+ }
191
+ }
192
+ /**
193
+ * One dispatch pass: find claimable work, claim what fits under the concurrency cap, execute.
194
+ *
195
+ * Overlap-guarded — a pass that runs long simply skips the next tick rather than stacking, which
196
+ * would otherwise let a slow database multiply in-flight work past the cap.
197
+ */
198
+ async pollOnce() {
199
+ if (!this.running || this.polling)
200
+ return;
201
+ const capacity = this.config.MaxConcurrentTasks - this.inFlight.size;
202
+ if (capacity <= 0)
203
+ return;
204
+ this.polling = true;
205
+ try {
206
+ const provider = await this.providerFactory.CreateProvider();
207
+ // Settle graphs before picking new work, so a failure earlier in this pass stops its
208
+ // branch immediately rather than after another wave has already launched.
209
+ await this.propagateAndRollup(provider);
210
+ const candidates = await this.findClaimableTasks(provider, capacity);
211
+ for (const task of candidates) {
212
+ if (this.inFlight.size >= this.config.MaxConcurrentTasks)
213
+ break;
214
+ if (!(await this.claims.TryClaim(provider, task.ID, this.contextUser))) {
215
+ // Another instance won the race, or the task is no longer Pending. Normal.
216
+ continue;
217
+ }
218
+ this.inFlight.add(task.ID);
219
+ // Intentionally not awaited — the poll loop must keep dispatching while this runs.
220
+ void this.executeClaimed(task.ID).finally(() => this.inFlight.delete(task.ID));
221
+ }
222
+ }
223
+ catch (e) {
224
+ LogError(`[TaskGraphDispatcher] Poll failed: ${e instanceof Error ? e.message : String(e)}`);
225
+ }
226
+ finally {
227
+ this.polling = false;
228
+ }
229
+ }
230
+ /**
231
+ * Executes one claimed task on its own provider, heartbeating until it settles.
232
+ *
233
+ * A fresh provider per task is the point of `ProviderFactory`: parallel tasks must not share a
234
+ * transaction scope or entity instances, or one task's work becomes visible inside another's.
235
+ */
236
+ async executeClaimed(taskID) {
237
+ let heartbeat = null;
238
+ try {
239
+ const provider = await this.providerFactory.CreateProvider();
240
+ const task = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
241
+ if (!(await task.Load(taskID))) {
242
+ LogError(`[TaskGraphDispatcher] Claimed task ${taskID} could not be loaded.`);
243
+ return;
244
+ }
245
+ heartbeat = setInterval(() => {
246
+ void this.claims.Heartbeat(provider, taskID, this.contextUser).then((ok) => {
247
+ if (!ok) {
248
+ // Lost ownership — reconciliation reclaimed it, or a human intervened.
249
+ LogError(`[TaskGraphDispatcher] Lost claim on task ${taskID} while executing; another instance may take it over.`);
250
+ }
251
+ });
252
+ }, this.config.HeartbeatIntervalSeconds * 1000);
253
+ // Emitted after the claim is held, not before: a frame saying "started" for work another
254
+ // instance actually took would be a lie a viewer cannot detect.
255
+ const graphID = task.ParentID ?? taskID;
256
+ const ownerUserID = await this.resolveOwner(provider, graphID);
257
+ this.emit({ Kind: 'TaskStarted', ParentTaskID: graphID, OwnerUserID: ownerUserID, TaskID: taskID, TaskName: task.Name, Status: 'In Progress' });
258
+ const dependencyOutputs = await this.loadDependencyOutputs(provider, taskID);
259
+ let inputPayload = null;
260
+ if (task.InputPayload) {
261
+ try {
262
+ inputPayload = JSON.parse(task.InputPayload);
263
+ }
264
+ catch (e) {
265
+ LogError(`[TaskGraphDispatcher] Task ${taskID} has malformed InputPayload: ${e}`);
266
+ }
267
+ }
268
+ const result = await this.agentRunner.RunAgentForTask({
269
+ TaskID: taskID,
270
+ AgentID: task.AgentID,
271
+ InputPayload: inputPayload,
272
+ DependencyOutputs: dependencyOutputs,
273
+ Provider: provider,
274
+ ContextUser: this.contextUser,
275
+ });
276
+ const recorded = await this.claims.CompleteClaimed(provider, taskID, {
277
+ Status: result.Success ? 'Complete' : 'Failed',
278
+ OutputPayload: result.Output != null ? JSON.stringify(result.Output) : null,
279
+ ErrorMessage: result.ErrorMessage ?? null,
280
+ AgentRunID: result.AgentRunID ?? null,
281
+ }, this.contextUser);
282
+ if (!recorded) {
283
+ // The guarded write refused: the row changed underneath us (cancelled, reassigned,
284
+ // or reclaimed). Deferring to whoever owns it now is correct — overwriting would
285
+ // undo a newer, deliberate decision.
286
+ LogError(`[TaskGraphDispatcher] Could not record outcome for ${taskID}; the task is no longer owned by this instance.`);
287
+ }
288
+ else {
289
+ // Only announced when the guarded write actually landed. Announcing an outcome we
290
+ // failed to persist would show a viewer a completion the database never recorded.
291
+ this.emit({
292
+ Kind: result.Success ? 'TaskCompleted' : 'TaskFailed',
293
+ ParentTaskID: graphID,
294
+ OwnerUserID: ownerUserID,
295
+ TaskID: taskID,
296
+ TaskName: task.Name,
297
+ Status: result.Success ? 'Complete' : 'Failed',
298
+ ErrorMessage: result.Success ? undefined : (result.ErrorMessage ?? undefined),
299
+ });
300
+ }
301
+ }
302
+ catch (e) {
303
+ LogError(`[TaskGraphDispatcher] Execution failed for ${taskID}: ${e instanceof Error ? e.message : String(e)}`);
304
+ try {
305
+ const provider = await this.providerFactory.CreateProvider();
306
+ await this.claims.CompleteClaimed(provider, taskID, { Status: 'Failed', ErrorMessage: e instanceof Error ? e.message : String(e) }, this.contextUser);
307
+ }
308
+ catch { /* already logged; nothing further to do */ }
309
+ }
310
+ finally {
311
+ if (heartbeat)
312
+ clearInterval(heartbeat);
313
+ }
314
+ }
315
+ /**
316
+ * Applies failure propagation and parent rollup across every graph with active work.
317
+ *
318
+ * All four decisions — what is eligible, what must block, what the parent status is, whether the
319
+ * graph is wedged — are delegated to the pure algorithms, unchanged from Phase 1.
320
+ */
321
+ async propagateAndRollup(provider) {
322
+ for (const parentID of await this.findActiveGraphIDs(provider)) {
323
+ const graph = await this.loadGraphState(provider, parentID);
324
+ if (graph.nodes.length === 0)
325
+ continue;
326
+ const toBlock = new Set([...ComputeTasksToBlock(graph.nodes, graph.edges), ...graph.unreachableTaskIDs]);
327
+ for (const taskID of toBlock) {
328
+ const entity = graph.entityById.get(taskID);
329
+ if (!entity)
330
+ continue;
331
+ entity.Status = 'Blocked';
332
+ if (await entity.Save()) {
333
+ LogStatus(`[TaskGraphDispatcher] Blocked '${entity.Name}' (${taskID}) — a dependency can never be satisfied.`);
334
+ // Worth announcing on its own: a blocked step is the one outcome a viewer would
335
+ // otherwise see as a task that simply never starts.
336
+ this.emit({
337
+ Kind: 'TaskBlocked',
338
+ ParentTaskID: parentID,
339
+ OwnerUserID: await this.resolveOwner(provider, parentID),
340
+ TaskID: taskID,
341
+ TaskName: entity.Name,
342
+ Status: 'Blocked',
343
+ });
344
+ }
345
+ }
346
+ if (IsGraphStalled(graph.nodes, graph.edges)) {
347
+ LogError(`[TaskGraphDispatcher] Graph ${parentID} is stalled: pending work with no satisfiable path.`);
348
+ }
349
+ const fresh = await this.loadGraphState(provider, parentID);
350
+ // ComputeParentRollup treats an empty child set as Complete-and-terminal, which is right
351
+ // for a graph that genuinely has no children and catastrophic for one whose reload came
352
+ // back empty transiently — it would mark live work finished and fire its continuation.
353
+ // The outer guard covered the first load only.
354
+ if (fresh.nodes.length === 0)
355
+ continue;
356
+ const rollup = ComputeParentRollup(fresh.nodes);
357
+ const parent = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
358
+ if (!(await parent.Load(parentID)))
359
+ continue;
360
+ if (parent.Status !== rollup.status || parent.PercentComplete !== rollup.percentComplete) {
361
+ parent.Status = rollup.status;
362
+ parent.PercentComplete = rollup.percentComplete;
363
+ if (rollup.isTerminal)
364
+ parent.CompletedAt = new Date();
365
+ await parent.Save();
366
+ }
367
+ if (rollup.isTerminal) {
368
+ // Emitted before the continuation is delivered, and outside its once-only guard: a
369
+ // viewer watching the run should learn it finished whether or not this instance is
370
+ // the one that wins the delivery CAS.
371
+ this.emit({
372
+ Kind: 'GraphSettled',
373
+ ParentTaskID: parentID,
374
+ OwnerUserID: await this.resolveOwner(provider, parentID),
375
+ Status: rollup.status,
376
+ CompletedCount: fresh.nodes.filter((n) => n.status === 'Complete').length,
377
+ TotalCount: fresh.nodes.length,
378
+ });
379
+ await this.deliverContinuation(provider, parent, fresh);
380
+ }
381
+ }
382
+ }
383
+ /**
384
+ * Runs the graph's continuation exactly once, now that it has settled.
385
+ *
386
+ * **Why the delivery marker is written before the side effect.** Delivery is at-least-once by
387
+ * nature: the process can die between "the graph is done" and "the user has been told". Marking
388
+ * first and acting second means the worst case is a *missed* notification that shows up in the
389
+ * task record as delivered — recoverable, visible, and inspectable. Marking after would make the
390
+ * worst case a *repeated* notification on every reconciliation sweep, forever, which is both
391
+ * user-visible noise and, for `reinvoke`, an unbounded agent-run loop. Given one of the two has
392
+ * to be chosen, the quiet failure is the safe one.
393
+ *
394
+ * The marker is written with a compare-and-swap read-back, so two instances reconciling the same
395
+ * completed graph produce one winner rather than two.
396
+ */
397
+ async deliverContinuation(provider, parent, graph) {
398
+ const meta = this.readParentMetadata(parent);
399
+ if (meta.continuationDeliveredAt)
400
+ return;
401
+ // At the cap, downgrade rather than refuse: the results still reach the user, the chain just
402
+ // stops growing. Refusing outright would lose the outcome of work that actually completed.
403
+ const mode = IsReinvokeCapReached(meta) ? 'message' : meta.continuation;
404
+ if (mode !== 'none' && IsReinvokeCapReached(meta) && meta.continuation === 'reinvoke') {
405
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} hit the reinvoke cap (${MAX_REINVOKE_DEPTH}); ` +
406
+ `delivering results as a message instead of starting another turn.`);
407
+ }
408
+ if (!(await this.claimContinuation(provider, parent.ID, meta)))
409
+ return;
410
+ if (mode === 'none')
411
+ return;
412
+ const summary = this.buildContinuationSummary(parent, graph);
413
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID} finished — ${summary}`);
414
+ if (!this.continuationDeliverer)
415
+ return;
416
+ const params = {
417
+ ParentTaskID: parent.ID,
418
+ WorkflowName: parent.Name,
419
+ ConversationDetailID: parent.ConversationDetailID ?? null,
420
+ SubmittedByAgentRunID: meta.submittedByAgentRunID,
421
+ ReinvokeDepth: meta.reinvokeDepth,
422
+ Tasks: [...graph.entityById.values()].map((t) => ({
423
+ TaskID: t.ID,
424
+ Name: t.Name,
425
+ Status: t.Status,
426
+ // A reference, not the payload. Inlining every task's output would swamp the
427
+ // continuation turn's context; the agent pulls what it needs by task ID.
428
+ Summary: t.OutputPayload ? `output available (${t.OutputPayload.length} chars)` : undefined,
429
+ ErrorMessage: t.ErrorMessage ?? undefined,
430
+ })),
431
+ Summary: summary,
432
+ };
433
+ try {
434
+ // Reinvoke degrades to a message when the host cannot start agent turns. Degrading is
435
+ // right rather than throwing: the work genuinely ran, and the user losing the results
436
+ // because nobody could start a follow-up turn would be the worse outcome.
437
+ if (mode === 'reinvoke' && this.continuationDeliverer.Reinvoke) {
438
+ await this.continuationDeliverer.Reinvoke(params);
439
+ }
440
+ else {
441
+ if (mode === 'reinvoke') {
442
+ LogStatus(`[TaskGraphDispatcher] Graph ${parent.ID}: host cannot reinvoke; delivering as a message.`);
443
+ }
444
+ await this.continuationDeliverer.PostMessage(params);
445
+ }
446
+ }
447
+ catch (e) {
448
+ // Already marked delivered, so this will not retry. That is the deliberate trade stated
449
+ // on the marker: a missed notification visible in the record beats one repeated forever.
450
+ LogError(`[TaskGraphDispatcher] Continuation delivery failed for ${parent.ID}: ${e instanceof Error ? e.message : String(e)}`);
451
+ }
452
+ }
453
+ /** Reads the parent's durable continuation metadata through the shared parser. */
454
+ readParentMetadata(parent) {
455
+ return ParseTaskGraphParentMetadata(parent.InputPayload);
456
+ }
457
+ /**
458
+ * Stamps the delivery marker and confirms this instance won the race.
459
+ *
460
+ * `MJ: Tasks` stays user-writable (D20), so a plain "read, decide, write" is not enough — the
461
+ * read-back is what makes a lost race observable instead of producing a duplicate delivery.
462
+ */
463
+ async claimContinuation(provider, parentID, meta) {
464
+ const row = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
465
+ if (!(await row.Load(parentID)))
466
+ return false;
467
+ const current = this.readParentMetadata(row);
468
+ if (current.continuationDeliveredAt)
469
+ return false; // a peer got there first
470
+ row.InputPayload = JSON.stringify({ ...meta, continuationDeliveredAt: new Date().toISOString() });
471
+ if (!(await row.Save())) {
472
+ LogError(`[TaskGraphDispatcher] Could not mark continuation delivered for ${parentID}; skipping to avoid a duplicate.`);
473
+ return false;
474
+ }
475
+ return true;
476
+ }
477
+ /** One line describing how the graph ended, for the completion log and message delivery. */
478
+ buildContinuationSummary(parent, graph) {
479
+ const counts = new Map();
480
+ for (const node of graph.nodes)
481
+ counts.set(node.status, (counts.get(node.status) ?? 0) + 1);
482
+ const breakdown = [...counts.entries()].map(([status, n]) => `${n} ${status}`).join(', ');
483
+ return `"${parent.Name}": ${graph.nodes.length} task(s) — ${breakdown}.`;
484
+ }
485
+ /**
486
+ * Tells the assignee that a human task is ready, exactly once.
487
+ *
488
+ * **Once** matters more than it looks: eligibility is recomputed on every poll, so a task parked
489
+ * on a person for three days would otherwise re-notify every five seconds until they acted. The
490
+ * marker is the task's own `ClaimedBy` — a human task has no executor to claim it, so the column
491
+ * is free, and reusing it means the "already notified" fact is as durable and as crash-safe as
492
+ * every other piece of graph state. A restart cannot resend.
493
+ *
494
+ * Best-effort by design. A notification that fails to send must not stop the graph or the poll
495
+ * loop; the task is still visible in the Tasks UI, so the work is discoverable even when the
496
+ * nudge does not arrive.
497
+ */
498
+ async notifyHumanTaskReady(task, provider) {
499
+ if (task.ClaimedBy === HUMAN_TASK_NOTIFIED_MARKER)
500
+ return;
501
+ if (!task.UserID)
502
+ return; // unassigned human task — nobody to tell
503
+ try {
504
+ await NotificationEngine.Instance.Config(false, this.contextUser);
505
+ await NotificationEngine.Instance.SendNotification({
506
+ userId: task.UserID,
507
+ typeNameOrId: HUMAN_TASK_NOTIFICATION_TYPE,
508
+ title: `Action needed: ${task.Name}`,
509
+ message: task.Description || 'A workflow is waiting on you to complete this task.',
510
+ resourceConfiguration: { type: 'Task', taskId: task.ID, parentTaskId: task.ParentID ?? '' },
511
+ }, this.contextUser);
512
+ }
513
+ catch (e) {
514
+ LogError(`[TaskGraphDispatcher] Could not notify ${task.UserID} about task ${task.ID}: ${e instanceof Error ? e.message : String(e)}`);
515
+ }
516
+ // Marked even when delivery threw. Retrying a notification on every five-second poll is a
517
+ // worse failure than one that was missed: the task remains visible in the Tasks UI either
518
+ // way, whereas a notification storm is not self-correcting.
519
+ task.ClaimedBy = HUMAN_TASK_NOTIFIED_MARKER;
520
+ if (!(await task.Save())) {
521
+ LogError(`[TaskGraphDispatcher] Could not mark task ${task.ID} as notified; it may notify again.`);
522
+ }
523
+ // Emitted once, alongside the marker, so a viewer sees the graph stop on a person rather
524
+ // than appearing to stall for no reason.
525
+ this.emit({
526
+ Kind: 'TaskAwaitingHuman',
527
+ ParentTaskID: task.ParentID ?? task.ID,
528
+ OwnerUserID: await this.resolveOwner(provider, task.ParentID ?? task.ID),
529
+ TaskID: task.ID,
530
+ TaskName: task.Name,
531
+ Status: task.Status,
532
+ AssignedUserID: task.UserID,
533
+ });
534
+ }
535
+ /**
536
+ * Parent tasks that still have work to do.
537
+ *
538
+ * `BypassCache` for the reason the caching guide names explicitly: **the claim protocol mutates
539
+ * these rows through direct SQL**, because the CAS guarantee IS the database's atomicity and a
540
+ * `BaseEntity.Save()` cannot express a guarded UPDATE. Direct DML fires no invalidation event,
541
+ * so a cached read of this query is stale the instant any task is claimed or completed — and
542
+ * the dispatcher would then be reading its own work queue through a cache its own writes never
543
+ * invalidate. Left cached, a completed task keeps reading as `In Progress` and the graph never
544
+ * rolls up: submitted work simply never settles.
545
+ */
546
+ async findActiveGraphIDs(provider) {
547
+ const rv = RunView.FromMetadataProvider(provider);
548
+ // TWO queries, because "has work left to do" and "needs attention" are not the same set.
549
+ //
550
+ // Selecting only graphs with non-terminal CHILDREN looks right and is subtly fatal: the
551
+ // moment the last child completes, the graph leaves that set — so the pass that would have
552
+ // rolled the parent up never sees it. A graph whose tasks all succeed therefore stays
553
+ // In Progress forever and its continuation never fires. (A graph that FAILS happened to
554
+ // survive this, because blocking its dependents left them non-terminal for one more pass —
555
+ // which is why the bug hid behind a passing failure-path test.)
556
+ //
557
+ // The second query closes it: a parent that is itself non-terminal still needs looking at,
558
+ // whatever its children are doing.
559
+ const [withPendingWork, unsettledParents] = await rv.RunViews([
560
+ {
561
+ EntityName: 'MJ: Tasks',
562
+ ExtraFilter: `ParentID IS NOT NULL AND Status IN ('Pending','In Progress')`,
563
+ Fields: ['ParentID'],
564
+ ResultType: 'simple',
565
+ BypassCache: true,
566
+ },
567
+ {
568
+ EntityName: 'MJ: Tasks',
569
+ ExtraFilter: `ParentID IS NULL AND Status IN ('Pending','In Progress')`,
570
+ Fields: ['ID'],
571
+ ResultType: 'simple',
572
+ BypassCache: true,
573
+ },
574
+ ], this.contextUser);
575
+ const ids = new Set();
576
+ for (const r of (withPendingWork?.Results ?? [])) {
577
+ if (r.ParentID)
578
+ ids.add(r.ParentID);
579
+ }
580
+ // Childless tasks match the second query too; propagateAndRollup skips anything with no
581
+ // nodes, so they cost one empty load and nothing else.
582
+ for (const r of (unsettledParents?.Results ?? [])) {
583
+ if (r.ID)
584
+ ids.add(r.ID);
585
+ }
586
+ return [...ids];
587
+ }
588
+ /**
589
+ * Tasks eligible to claim right now, across all active graphs.
590
+ *
591
+ * Eligibility is decided by the pure algorithm rather than by SQL: expressing "all prerequisites
592
+ * complete" as a query is possible but would be a second, independently-maintained definition of
593
+ * the same rule, free to drift from the one the in-run executor uses.
594
+ */
595
+ async findClaimableTasks(provider, limit) {
596
+ const claimable = [];
597
+ for (const parentID of await this.findActiveGraphIDs(provider)) {
598
+ if (claimable.length >= limit)
599
+ break;
600
+ const graph = await this.loadGraphState(provider, parentID);
601
+ for (const node of ComputeEligibleTasks(graph.nodes, graph.edges)) {
602
+ const entity = graph.entityById.get(node.id);
603
+ if (!entity)
604
+ continue;
605
+ // Human tasks are never dispatched — a person completes them. But "eligible" is the
606
+ // moment that person can finally act, and nothing else in the system knows it has
607
+ // arrived: the task sat Pending behind prerequisites, and no save touched it when
608
+ // they cleared. Without a notification here a workflow simply stops, waiting on
609
+ // someone who was never told. That silent stall is the failure mode this exists to
610
+ // prevent, so it happens on the eligibility check rather than at submission.
611
+ if (!entity.AgentID) {
612
+ await this.notifyHumanTaskReady(entity, provider);
613
+ continue;
614
+ }
615
+ if (this.inFlight.has(entity.ID))
616
+ continue;
617
+ claimable.push(entity);
618
+ if (claimable.length >= limit)
619
+ break;
620
+ }
621
+ }
622
+ return claimable;
623
+ }
624
+ /** Loads a graph's children and edges in the shapes both the algorithms and mutation need. */
625
+ async loadGraphState(provider, parentTaskID) {
626
+ const rv = RunView.FromMetadataProvider(provider);
627
+ // BypassCache throughout: task status is written by the claim protocol's direct SQL, which
628
+ // fires no cache invalidation. See findActiveGraphIDs.
629
+ const childrenResult = await rv.RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
630
+ const children = (childrenResult.Success ? childrenResult.Results : []) ?? [];
631
+ if (children.length === 0)
632
+ return { nodes: [], edges: [], entityById: new Map(), unreachableTaskIDs: new Set() };
633
+ const idList = children.map((c) => `'${c.ID}'`).join(',');
634
+ const depsResult = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID IN (${idList})`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
635
+ const deps = (depsResult.Success ? depsResult.Results : []) ?? [];
636
+ const entityById = new Map(children.map((c) => [c.ID, c]));
637
+ // Conditional edges are resolved HERE, before eligibility runs, by dropping edges whose
638
+ // condition does not hold. Expressing it as edge removal rather than as a second rule inside
639
+ // the eligibility algorithm is what keeps one definition of "ready": a task with no live
640
+ // incoming edges is ready for exactly the same reason a task with no edges at all is.
641
+ //
642
+ // An edge whose condition cannot be evaluated is KEPT, which is the opposite of the flow
643
+ // executor's choice and deliberately so. There, a broken condition means an edge is not
644
+ // followed and the graph moves on. Here it would mean a prerequisite silently disappears and
645
+ // the dependent task runs early — turning a typo into out-of-order execution. Keeping the
646
+ // edge instead stalls the graph, which the stall detector already reports loudly.
647
+ const liveEdges = [];
648
+ // A definitely-false edge must not merely disappear. Removing a task's only prerequisite
649
+ // makes it eligible in the very next wave — so "this branch was not taken" would execute the
650
+ // branch, potentially before the node that gated it. The dependent is recorded as
651
+ // unreachable instead, and blocked before anything can claim it.
652
+ const droppedInto = new Set();
653
+ const stillReachable = new Set();
654
+ for (const d of deps) {
655
+ if (d.Condition?.trim()) {
656
+ const outcome = this.evaluateEdgeCondition(d, entityById);
657
+ if (outcome === 'drop') {
658
+ droppedInto.add(d.TaskID);
659
+ continue;
660
+ }
661
+ }
662
+ stillReachable.add(d.TaskID);
663
+ liveEdges.push({
664
+ taskId: d.TaskID,
665
+ dependsOnTaskId: d.DependsOnTaskID,
666
+ dependencyType: d.DependencyType,
667
+ });
668
+ }
669
+ // Only unreachable when EVERY route in was cut. A node still holding a live edge is simply
670
+ // waiting on it, and a node reached by an alternate branch is genuinely reachable.
671
+ const unreachableTaskIDs = new Set([...droppedInto].filter((id) => !stillReachable.has(id)));
672
+ return {
673
+ nodes: children.map((c) => ({ id: c.ID, status: c.Status })),
674
+ edges: liveEdges,
675
+ entityById,
676
+ unreachableTaskIDs,
677
+ };
678
+ }
679
+ /**
680
+ * Decides whether a conditional dependency edge is live.
681
+ *
682
+ * The condition sees the upstream task's outcome — its status and parsed output — which is the
683
+ * only information a runtime graph has to branch on. Returns `'drop'` only on a definite false;
684
+ * an unevaluable condition keeps the edge for the reason stated at the call site.
685
+ */
686
+ evaluateEdgeCondition(dep, entityById) {
687
+ const upstream = entityById.get(dep.DependsOnTaskID);
688
+ if (!upstream)
689
+ return 'keep';
690
+ let output = null;
691
+ if (upstream.OutputPayload) {
692
+ try {
693
+ output = JSON.parse(upstream.OutputPayload);
694
+ }
695
+ catch { /* a malformed payload is not grounds to drop a prerequisite */ }
696
+ }
697
+ const result = this.conditionEvaluator.Evaluate(dep.Condition, {
698
+ status: upstream.Status,
699
+ succeeded: upstream.Status === 'Complete',
700
+ failed: upstream.Status === 'Failed',
701
+ output,
702
+ errorMessage: upstream.ErrorMessage ?? null,
703
+ });
704
+ if (!result.Success) {
705
+ LogError(`[TaskGraphDispatcher] Dependency ${dep.ID} has an unevaluable condition ` +
706
+ `(${result.ErrorMessage}); keeping the edge so the graph stalls visibly rather than ` +
707
+ `running ${dep.TaskID} out of order.`);
708
+ return 'keep';
709
+ }
710
+ return result.Value ? 'keep' : 'drop';
711
+ }
712
+ /** Parsed `OutputPayload` of each completed dependency, keyed by that task's ID. */
713
+ async loadDependencyOutputs(provider, taskID) {
714
+ const outputs = new Map();
715
+ const rv = RunView.FromMetadataProvider(provider);
716
+ // BypassCache: an upstream task's OutputPayload is written on the completion path, so a
717
+ // cached read here can hand a dependent task the previous run's output — or none at all.
718
+ const deps = await rv.RunView({ EntityName: 'MJ: Task Dependencies', ExtraFilter: `TaskID='${taskID}'`, ResultType: 'entity_object', BypassCache: true }, this.contextUser);
719
+ for (const dep of (deps.Success ? deps.Results : []) ?? []) {
720
+ const upstream = await provider.GetEntityObject('MJ: Tasks', this.contextUser);
721
+ if (!(await upstream.Load(dep.DependsOnTaskID)))
722
+ continue;
723
+ if (!upstream.OutputPayload)
724
+ continue;
725
+ try {
726
+ outputs.set(dep.DependsOnTaskID, JSON.parse(upstream.OutputPayload));
727
+ }
728
+ catch (e) {
729
+ LogError(`[TaskGraphDispatcher] Task ${dep.DependsOnTaskID} has malformed OutputPayload: ${e}`);
730
+ }
731
+ }
732
+ return outputs;
733
+ }
734
+ }
735
+ //# sourceMappingURL=TaskGraphDispatcher.js.map