@memberjunction/task-graph 6.1.0-edge.2 → 6.1.0-edge.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/LICENSE +180 -4
  2. package/README.md +30 -1
  3. package/dist/TaskClaimStore.d.ts +376 -4
  4. package/dist/TaskClaimStore.d.ts.map +1 -1
  5. package/dist/TaskClaimStore.js +600 -20
  6. package/dist/TaskClaimStore.js.map +1 -1
  7. package/dist/TaskGraphDispatcher.d.ts +326 -25
  8. package/dist/TaskGraphDispatcher.d.ts.map +1 -1
  9. package/dist/TaskGraphDispatcher.js +1604 -286
  10. package/dist/TaskGraphDispatcher.js.map +1 -1
  11. package/dist/TaskGraphService.d.ts +255 -5
  12. package/dist/TaskGraphService.d.ts.map +1 -1
  13. package/dist/TaskGraphService.js +583 -26
  14. package/dist/TaskGraphService.js.map +1 -1
  15. package/dist/TaskGraphSubmitterImpl.d.ts.map +1 -1
  16. package/dist/TaskGraphSubmitterImpl.js +5 -0
  17. package/dist/TaskGraphSubmitterImpl.js.map +1 -1
  18. package/dist/condition-gate.d.ts +128 -0
  19. package/dist/condition-gate.d.ts.map +1 -0
  20. package/dist/condition-gate.js +257 -0
  21. package/dist/condition-gate.js.map +1 -0
  22. package/dist/debug-state.d.ts +102 -0
  23. package/dist/debug-state.d.ts.map +1 -0
  24. package/dist/debug-state.js +135 -0
  25. package/dist/debug-state.js.map +1 -0
  26. package/dist/index.d.ts +6 -0
  27. package/dist/index.d.ts.map +1 -1
  28. package/dist/index.js +6 -0
  29. package/dist/index.js.map +1 -1
  30. package/dist/operations/TaskGraphDebugOperations.d.ts +99 -0
  31. package/dist/operations/TaskGraphDebugOperations.d.ts.map +1 -0
  32. package/dist/operations/TaskGraphDebugOperations.js +310 -0
  33. package/dist/operations/TaskGraphDebugOperations.js.map +1 -0
  34. package/dist/operations/TaskGraphOperations.d.ts +20 -2
  35. package/dist/operations/TaskGraphOperations.d.ts.map +1 -1
  36. package/dist/operations/TaskGraphOperations.js +47 -8
  37. package/dist/operations/TaskGraphOperations.js.map +1 -1
  38. package/dist/settlement-rescue.d.ts +85 -0
  39. package/dist/settlement-rescue.d.ts.map +1 -0
  40. package/dist/settlement-rescue.js +119 -0
  41. package/dist/settlement-rescue.js.map +1 -0
  42. package/dist/task-graph-kick.d.ts +3 -0
  43. package/dist/task-graph-kick.d.ts.map +1 -0
  44. package/dist/task-graph-kick.js +17 -0
  45. package/dist/task-graph-kick.js.map +1 -0
  46. package/dist/task-predicates.d.ts +77 -0
  47. package/dist/task-predicates.d.ts.map +1 -0
  48. package/dist/task-predicates.js +75 -0
  49. package/dist/task-predicates.js.map +1 -0
  50. package/dist/types.d.ts +110 -1
  51. package/dist/types.d.ts.map +1 -1
  52. package/dist/types.js.map +1 -1
  53. package/package.json +12 -11
@@ -16,8 +16,26 @@
16
16
  * @module @memberjunction/task-graph
17
17
  */
18
18
  import { LogError, LogStatus, RunInEntityTransaction, RunView, } from '@memberjunction/core';
19
- import { FormatValidationErrors, NormalizeDependency, RankGraphNodes, ValidateTaskGraphSpec, ConfigOf, } from '@memberjunction/ai-core-plus';
19
+ import { FormatValidationErrors, NormalizeDependency, RankGraphNodes, SanitizeInvocationEnvelope, ValidateTaskGraphSpec, ConfigOf, } from '@memberjunction/ai-core-plus';
20
20
  import { UUIDsEqual } from '@memberjunction/global';
21
+ import { TaskClaimStore } from './TaskClaimStore.js';
22
+ import { ParseTaskGraphDebugState } from './debug-state.js';
23
+ import { KickTaskGraphDispatchers } from './task-graph-kick.js';
24
+ /**
25
+ * Normalizes a caller-supplied reinvoke depth to a safe cap seed.
26
+ *
27
+ * The runaway-loop cap is a signed comparison over a persisted-verbatim seed, and the remote Submit
28
+ * operation is the one seam where the seed is caller-supplied rather than computed by the engine —
29
+ * a negative value would BUY hops (`-1000` turns a 5-hop cap into 1005). Clamped rather than
30
+ * refused so a stale client sending zero-adjacent noise keeps working, and no accepted value can
31
+ * ever weaken the cap. Exported pure because a boundary rule that cannot be tested directly is a
32
+ * boundary nobody will notice moving.
33
+ */
34
+ export function ClampReinvokeDepth(value) {
35
+ return typeof value === 'number' && Number.isFinite(value)
36
+ ? Math.max(0, Math.floor(value))
37
+ : undefined;
38
+ }
21
39
  /**
22
40
  * Continuation chains are bounded separately from graph nesting.
23
41
  *
@@ -65,18 +83,58 @@ export function ParseTaskGraphParentMetadata(raw) {
65
83
  : 'message',
66
84
  failureSemantics: parsed.failureSemantics === 'edges' ? 'edges' : 'block',
67
85
  reinvokeDepth: Number.isFinite(parsed.reinvokeDepth) ? Number(parsed.reinvokeDepth) : 0,
86
+ // Guarded like the others: this is read to explain a settlement after the fact, and an
87
+ // arbitrary string arriving from a hand edit should read as "unknown", not be echoed.
88
+ continuationDeliveredAs: DELIVERY_OUTCOMES.has(parsed.continuationDeliveredAs)
89
+ ? parsed.continuationDeliveredAs
90
+ : undefined,
68
91
  };
69
92
  }
70
93
  catch {
71
94
  return { ...DEFAULT_PARENT_METADATA };
72
95
  }
73
96
  }
97
+ /**
98
+ * The JSON bag `persistParent` writes. Exported so start-paused is testable without a database —
99
+ * Pause-after-submit races the first dispatcher poll, so this bag is the only place `paused: true`
100
+ * is guaranteed to land before anyone claims.
101
+ */
102
+ export function BuildTaskGraphParentInputPayload(args) {
103
+ return {
104
+ continuation: args.continuation,
105
+ reinvokeDepth: args.reinvokeDepth,
106
+ failureSemantics: args.failureSemantics,
107
+ submittedByAgentRunID: args.submittedByAgentRunID,
108
+ submittedByUserID: args.submittedByUserID,
109
+ ...(args.invocation
110
+ ? { invocation: { data: args.invocation.data, context: args.invocation.context } }
111
+ : {}),
112
+ ...(args.startPaused
113
+ ? {
114
+ debug: {
115
+ paused: true,
116
+ pausedReason: 'user',
117
+ pausedBy: args.submittedByUserID,
118
+ },
119
+ }
120
+ : {}),
121
+ };
122
+ }
123
+ /**
124
+ * How a settlement's announcement ended — the values `TryClaimContinuation` may record.
125
+ *
126
+ * `expired` means found too late to announce; `cancelled` means there was deliberately nobody left
127
+ * to announce to, because the run that submitted the graph was cancelled. Both are settlements that
128
+ * completed WITHOUT an announcement, and keeping them distinct is the difference between "we missed
129
+ * it" and "we chose not to".
130
+ */
131
+ const DELIVERY_OUTCOMES = new Set(['delivered', 'expired', 'cancelled']);
74
132
  /** True when a continuation chain has gone as far as it may. */
75
133
  export function IsReinvokeCapReached(meta) {
76
134
  return meta.reinvokeDepth >= MAX_REINVOKE_DEPTH;
77
135
  }
78
136
  /** Name of the task type used for agent-orchestrated graphs. */
79
- const TASK_TYPE_NAME = 'AI Workflow';
137
+ export const TASK_TYPE_NAME = 'AI Workflow';
80
138
  /**
81
139
  * The node kinds a `Task` row can actually represent, and therefore the ones the dispatcher can run.
82
140
  *
@@ -253,6 +311,29 @@ export function FindCrossUserAssignments(spec, submitterUserID) {
253
311
  `only ask the person who started it. Remove assignToUserID from those steps.`);
254
312
  }
255
313
  export class TaskGraphService {
314
+ constructor() {
315
+ /**
316
+ * Guarded single-statement writes, shared with the dispatcher.
317
+ *
318
+ * The instance id is descriptive only — this service never CLAIMS anything, it only issues
319
+ * guarded transitions whose predicates are about the row's own status rather than about who
320
+ * holds it.
321
+ */
322
+ this.claims = new TaskClaimStore('task-graph-service', 0);
323
+ // ────────────────────────────────────────────────────────────────────────
324
+ // debug / runner control plane
325
+ //
326
+ // Every verb here is durable, declarative state the dispatcher's claim filter consults on its
327
+ // next pass — never a call into a running dispatcher. That is what makes the controls work
328
+ // across instances and restarts, and what bounds their latency to one poll interval. See
329
+ // `debug-state.ts` for the model.
330
+ // ────────────────────────────────────────────────────────────────────────
331
+ /**
332
+ * Store for the guarded JSON_MODIFY writes. The instance identity and TTL are claim-protocol
333
+ * concerns this class never exercises — the debug writes are instance-free.
334
+ */
335
+ this.debugWrites = new TaskClaimStore('task-graph-service', 0);
336
+ }
256
337
  /**
257
338
  * Validates and persists a task graph, returning as soon as it is durable.
258
339
  *
@@ -353,6 +434,7 @@ export class TaskGraphService {
353
434
  });
354
435
  LogStatus(`[TaskGraphService] Submitted "${spec.workflowName}": parent ${parentTaskID}, ${taskIDMap.size} task(s). ` +
355
436
  `Awaiting dispatcher pickup.`);
437
+ KickTaskGraphDispatchers();
356
438
  return { Success: true, ParentTaskID: parentTaskID, TaskIDMap: taskIDMap };
357
439
  }
358
440
  catch (e) {
@@ -362,23 +444,60 @@ export class TaskGraphService {
362
444
  }
363
445
  }
364
446
  /**
365
- * Cancels a graph and everything in it that has not already settled.
447
+ * Cancels a graph, everything in it that has not already settled, and everything it started.
366
448
  *
367
449
  * Cancels children first: a parent marked `Cancelled` while children are still `Pending` would
368
450
  * leave the dispatcher free to pick those children up, which is the opposite of what the caller
369
451
  * asked for.
452
+ *
453
+ * **The verdict is the outcome, not the attempt** (R2-9). This returned `true` unconditionally
454
+ * while logging each child that failed to cancel — so one failed save left that child `Pending`,
455
+ * told the caller cancellation had succeeded, and let the dispatcher run the child afterwards.
456
+ * The graph could then settle `Complete` and ANNOUNCE ITS COMPLETION into the conversation of a
457
+ * workflow the user had cancelled. A partial cancel now says so and names what survived; the
458
+ * graph stays active, so retrying is meaningful rather than cosmetic.
370
459
  */
371
460
  async Cancel(parentTaskID, context) {
461
+ return this.cancelWithDepth(parentTaskID, context, 0, new Set());
462
+ }
463
+ /**
464
+ * `Cancel`, carrying the recursion state the public entry point does not expose.
465
+ *
466
+ * **The depth cap was dead code** (R3-10): `Cancel` passed a literal 0, and the recursion
467
+ * re-entered through `this.Cancel`, which restarted at 0 — so the check could never fire and
468
+ * the "bounded by the reinvoke depth cap" promise was false. A hand-edited `AgentRunID` cycle
469
+ * recursed to stack overflow mid-cancel.
470
+ *
471
+ * The visited set is cheap armour on top: the cap bounds how DEEP a legitimate chain goes, and
472
+ * a cycle is not deep, it is circular. Arithmetic alone would eventually stop it; a visited set
473
+ * stops it immediately and covers linkage shapes the arithmetic does not anticipate.
474
+ */
475
+ async cancelWithDepth(parentTaskID, context, depth, visited) {
476
+ if (visited.has(parentTaskID)) {
477
+ return { Success: true, Cancelled: true, UncancelledTaskNames: [] };
478
+ }
479
+ visited.add(parentTaskID);
372
480
  try {
373
481
  const children = await this.loadChildren(parentTaskID, context);
482
+ const uncancelled = [];
483
+ const settledMeanwhile = [];
374
484
  for (const child of children) {
375
485
  // Terminal work is left alone — cancelling a completed task would rewrite history.
376
- if (['Complete', 'Failed', 'Cancelled'].includes(child.Status))
486
+ // The in-memory test is a cheap pre-filter; the one that MATTERS is in the statement
487
+ // (R3-9), because a child can settle between this snapshot and its own write, and
488
+ // the full-row save this replaces overwrote that outcome wholesale.
489
+ if (['Complete', 'Failed', 'Cancelled', 'Skipped'].includes(child.Status))
377
490
  continue;
378
- child.Status = 'Cancelled';
379
- if (!(await child.Save())) {
380
- LogError(`[TaskGraphService] Failed to cancel task ${child.ID}: ${child.LatestResult?.CompleteMessage ?? 'unknown error'}`);
381
- }
491
+ if (await this.claims.TryCancelTask(context.Provider, child.ID, context.ContextUser))
492
+ continue;
493
+ // Rowcount 0 means it reached a terminal status while we were cancelling its
494
+ // siblings. Its outcome is real and stays; the verdict says so rather than pretending
495
+ // the cancel was total.
496
+ settledMeanwhile.push(child.Name);
497
+ }
498
+ if (settledMeanwhile.length > 0) {
499
+ LogStatus(`[TaskGraphService] ${settledMeanwhile.length} task(s) settled while the cancel ran ` +
500
+ `(${settledMeanwhile.join(', ')}); their outcomes are kept.`);
382
501
  }
383
502
  // Withdraw the questions too. A human step that was waiting has an open
384
503
  // `MJ: AI Agent Requests` row, and cancelling only the Task left that row `Requested`
@@ -387,17 +506,97 @@ export class TaskGraphService {
387
506
  // already Cancelled. Nothing else ever closes these — the dispatcher only expires rows
388
507
  // that carry a deadline, and most do not.
389
508
  await this.cancelOpenRequests(children.map((c) => c.ID), context);
390
- const parent = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
391
- if (!(await parent.Load(parentTaskID)))
392
- return false;
393
- parent.Status = 'Cancelled';
394
- parent.CompletedAt = new Date();
395
- return await parent.Save();
509
+ // THE PARENT IS LEFT TO THE DISPATCHER, DELIBERATELY.
510
+ //
511
+ // Writing it terminal here skipped the settle path entirely — no cost rollup, no run
512
+ // settlement, no notification — so the submitting agent run stayed `Paused` forever.
513
+ // Worse, it was NONDETERMINISTIC: if a dispatcher poll happened to land between the
514
+ // child cancels above and the parent write, the graph settled through the normal path
515
+ // and the run WAS failed and messaged. Cancel behaved differently run to run depending
516
+ // on timing.
517
+ //
518
+ // With the children cancelled, `ComputeParentRollup` reaches `Cancelled` on its own and
519
+ // the ordinary settle sequence runs — rollup, run settlement, continuation — exactly as
520
+ // it does for a graph that finished by itself. Less code, one path, and a deterministic
521
+ // outcome.
522
+ //
523
+ // The parent stays non-terminal until then, so the sweep still sees it as active work.
524
+ // WHAT THIS WORKFLOW STARTED IS ALSO CANCELLED (R2-9).
525
+ //
526
+ // A graph's step can be an agent that submits a graph of its own, and those sub-graphs
527
+ // persist as ROOTS — linked back only through the child task's `AgentRunID`. So
528
+ // cancelling a workflow left its descendants running, and on settlement one of them can
529
+ // REINVOKE the cancelled workflow's own agent for a fresh billed turn: the user stopped
530
+ // a workflow and it started itself again.
531
+ //
532
+ // Bounded by the reinvoke depth cap, which is what bounds the chain in the first place.
533
+ const nested = await this.cancelNestedGraphs(children, context, depth, visited);
534
+ uncancelled.push(...nested);
535
+ if (uncancelled.length > 0) {
536
+ return {
537
+ Success: false,
538
+ Cancelled: false,
539
+ UncancelledTaskNames: uncancelled,
540
+ ErrorMessage: `Cancelled what it could, but ${uncancelled.length} task(s) could not be cancelled ` +
541
+ `(${uncancelled.join(', ')}). The workflow is still active — retry the cancel.`,
542
+ };
543
+ }
544
+ return { Success: true, Cancelled: true, UncancelledTaskNames: [] };
396
545
  }
397
546
  catch (e) {
398
- LogError(`[TaskGraphService] Cancel failed for ${parentTaskID}: ${e instanceof Error ? e.message : String(e)}`);
399
- return false;
547
+ const message = e instanceof Error ? e.message : String(e);
548
+ LogError(`[TaskGraphService] Cancel failed for ${parentTaskID}: ${message}`);
549
+ return { Success: false, Cancelled: false, UncancelledTaskNames: [], ErrorMessage: message };
550
+ }
551
+ }
552
+ /**
553
+ * Cancels the graphs that this graph's own steps submitted, one level at a time.
554
+ *
555
+ * The linkage is `child task → AgentRunID → the graphs that run submitted`, which is exactly how
556
+ * the continuation chain finds its way back up; walking it downward is the same relation read the
557
+ * other way. Depth-capped by the same constant that caps reinvocation, so a self-referencing
558
+ * workflow cannot make cancellation recurse further than it could have spawned.
559
+ *
560
+ * @returns names of tasks in descendant graphs that could not be cancelled
561
+ */
562
+ async cancelNestedGraphs(children, context, depth, visited) {
563
+ if (depth >= MAX_REINVOKE_DEPTH) {
564
+ LogError(`[TaskGraphService] Nested cancel stopped at depth ${depth}; a deeper sub-graph chain ` +
565
+ `than the reinvoke cap allows may still be running.`);
566
+ return [];
400
567
  }
568
+ const runIDs = [...new Set(children.map((c) => c.AgentRunID).filter((id) => !!id))];
569
+ if (runIDs.length === 0)
570
+ return [];
571
+ const rv = RunView.FromMetadataProvider(context.Provider);
572
+ const inList = runIDs.map((id) => `'${id}'`).join(',');
573
+ // TYPE-SCOPED, like every other graph-mutating walk over this table (R3-10). `MJ: Tasks` is
574
+ // general-purpose, and without the predicate any non-workflow root hierarchy that happens to
575
+ // carry a cancelled run's ID gets `Cancelled` written over its children and its requests
576
+ // withdrawn — the user-writable-table threat the claim store's guards exist to defend
577
+ // against. The comment here already claimed this scoping; the query did not have it.
578
+ const typeID = await this.findTaskTypeID(context);
579
+ if (!typeID)
580
+ return [];
581
+ const subGraphs = await rv.RunView({
582
+ EntityName: 'MJ: Tasks',
583
+ ExtraFilter: `TypeID='${typeID}' AND ParentID IS NULL AND AgentRunID IN (${inList})`,
584
+ Fields: ['ID'],
585
+ ResultType: 'simple',
586
+ BypassCache: true,
587
+ }, context.ContextUser);
588
+ if (!subGraphs.Success) {
589
+ LogError(`[TaskGraphService] Could not look for sub-graphs while cancelling: ${subGraphs.ErrorMessage}`);
590
+ return [];
591
+ }
592
+ const failures = [];
593
+ for (const row of subGraphs.Results ?? []) {
594
+ // Through the depth-carrying overload, so the cap actually engages.
595
+ const result = await this.cancelWithDepth(row.ID, context, depth + 1, visited);
596
+ if (!result.Success)
597
+ failures.push(...result.UncancelledTaskNames);
598
+ }
599
+ return failures;
401
600
  }
402
601
  /**
403
602
  * Closes the still-open requests raised for a set of tasks.
@@ -450,7 +649,7 @@ export class TaskGraphService {
450
649
  * leaving them blocked would make the retry pointless, as the graph still could not progress
451
650
  * past this node.
452
651
  */
453
- async Retry(taskID, context) {
652
+ async Retry(taskID, context, inputPayload) {
454
653
  try {
455
654
  const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
456
655
  if (!(await task.Load(taskID)))
@@ -459,6 +658,27 @@ export class TaskGraphService {
459
658
  LogError(`[TaskGraphService] Cannot retry task ${taskID}: status is ${task.Status}, expected Failed.`);
460
659
  return false;
461
660
  }
661
+ // An edited input rides the retry: the operator saw WHY it failed and is re-running the
662
+ // step with a corrected brief. Applies to this run only — the graph's spec is long gone.
663
+ //
664
+ // Written through the GUARDED statement rather than onto the in-memory row, so the edit
665
+ // cannot ride along on the full-row save below. The window here is narrower than
666
+ // `UpdateTaskInput`'s (the pre-state is `Failed`, so a concurrent claim is not the
667
+ // hazard — a concurrent human retry is), but the shape is the same and it costs one
668
+ // statement to not have it. The rest of this method's full-row save predates this PR
669
+ // and is Round 3's to purge; the new write does not add to it.
670
+ if (inputPayload !== undefined) {
671
+ const typeID = await this.ensureTaskType(context);
672
+ const json = typeof inputPayload === 'string' ? inputPayload : JSON.stringify(inputPayload);
673
+ const wrote = await this.debugWrites.TryUpdateInputPayload(context.Provider, taskID, json, 'Failed', typeID, context.ContextUser);
674
+ if (!wrote) {
675
+ LogError(`[TaskGraphService] Could not apply the edited input to task ${taskID}; retry refused rather than re-running the old brief.`);
676
+ return false;
677
+ }
678
+ // Keep the in-memory row in step with what was just written, so the save below does
679
+ // not put the old input back.
680
+ task.InputPayload = json;
681
+ }
462
682
  task.Status = 'Pending';
463
683
  task.ErrorMessage = null;
464
684
  task.StartedAt = null;
@@ -484,6 +704,266 @@ export class TaskGraphService {
484
704
  return false;
485
705
  }
486
706
  }
707
+ /** Result shape shared by the control verbs: what happened, and the state that now holds. */
708
+ controlResult(success, debug, errorMessage) {
709
+ return { Success: success, Debug: debug, ErrorMessage: errorMessage };
710
+ }
711
+ /**
712
+ * Loads a graph parent and proves it IS a workflow graph before any debug write.
713
+ *
714
+ * Read with `BypassCache` for the same reason the dispatcher reads rows that way: the debug bag
715
+ * is written by direct `JSON_MODIFY` statements that fire no cache invalidation, so a cached
716
+ * read here could merge new state over a stale copy and silently resurrect a cleared flag.
717
+ */
718
+ async loadWorkflowParent(parentTaskID, context) {
719
+ const typeID = await this.ensureTaskType(context);
720
+ const rows = await RunView.FromMetadataProvider(context.Provider).RunView({
721
+ EntityName: 'MJ: Tasks',
722
+ ExtraFilter: `ID='${parentTaskID.replace(/'/g, "''")}'`,
723
+ Fields: ['ID', 'TypeID', 'InputPayload', 'Status'],
724
+ ResultType: 'simple',
725
+ BypassCache: true,
726
+ }, context.ContextUser);
727
+ const row = rows.Success ? rows.Results?.[0] : undefined;
728
+ if (!row)
729
+ return null;
730
+ if (!UUIDsEqual(row.TypeID, typeID))
731
+ return null;
732
+ return { typeID, inputPayload: row.InputPayload, status: row.Status };
733
+ }
734
+ /**
735
+ * Writes the debug-bag fields a verb OWNS, and reports the state that results.
736
+ *
737
+ * Field-scoped on purpose — see {@link TaskClaimStore.TryWriteDebugFields}. A verb declares the
738
+ * paths it is responsible for; everything else in the bag is left exactly as the database has
739
+ * it, so a concurrent step-consume, breakpoint edit, or override cannot be undone by a verb that
740
+ * was not talking about them.
741
+ *
742
+ * The returned state is this instance's best view (read + the fields just written) and is
743
+ * advisory — the same posture the console takes toward frames.
744
+ */
745
+ async writeDebugFields(parentTaskID, context, build) {
746
+ const parent = await this.loadWorkflowParent(parentTaskID, context);
747
+ if (!parent)
748
+ return this.controlResult(false, undefined, 'Not a workflow graph this control plane can act on.');
749
+ const { Fields, Next } = build(ParseTaskGraphDebugState(parent.inputPayload));
750
+ const ok = await this.debugWrites.TryWriteDebugFields(context.Provider, parentTaskID, Fields, parent.typeID, context.ContextUser);
751
+ if (!ok)
752
+ return this.controlResult(false, undefined, 'The debug state could not be written; see the server log.');
753
+ return this.controlResult(true, Next);
754
+ }
755
+ /**
756
+ * Pauses a graph: nothing new is claimed until it is resumed. In-flight steps finish naturally
757
+ * and their completions land — a pause gates claiming and never touches a live claim, which is
758
+ * why there is no "what happens to the claim" question to answer.
759
+ */
760
+ async PauseGraph(parentTaskID, context, pausedByUserID) {
761
+ const pausedBy = pausedByUserID ?? context.ContextUser?.ID ?? null;
762
+ return this.writeDebugFields(parentTaskID, context, (current) => ({
763
+ // Pause owns the pause fields AND the step allowance: an allowance armed a moment ago is
764
+ // for a run the operator has now stopped, so clearing it is the verb's meaning rather
765
+ // than a side effect. Breakpoints and overrides are untouched — they outlive a pause.
766
+ Fields: [
767
+ TaskClaimStore.DebugField('$.debug.paused', { Kind: 'bool', Value: true }),
768
+ TaskClaimStore.DebugField('$.debug.pausedReason', { Kind: 'string', Value: 'user' }),
769
+ TaskClaimStore.DebugField('$.debug.pausedBy', pausedBy ? { Kind: 'string', Value: pausedBy } : { Kind: 'null' }),
770
+ TaskClaimStore.DebugField('$.debug.pausedAtTaskID', { Kind: 'null' }),
771
+ TaskClaimStore.DebugField('$.debug.step', { Kind: 'null' }),
772
+ TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' }),
773
+ ],
774
+ Next: { ...current, paused: true, pausedBy, pausedReason: 'user', pausedAtTaskID: null, step: undefined, skipBreakpointTaskID: undefined },
775
+ }));
776
+ }
777
+ /** Resumes a paused graph. Breakpoints and edge overrides survive — only the pause clears. */
778
+ async ResumeGraph(parentTaskID, context) {
779
+ return this.writeDebugFields(parentTaskID, context, (current) => {
780
+ const next = { ...current };
781
+ // Continue from a breakpoint must run the stopped task. Stamping it here so the next
782
+ // poll does not re-hit the same still-eligible breakpoint without claiming.
783
+ const skip = current.pausedReason === 'breakpoint' ? current.pausedAtTaskID : null;
784
+ delete next.paused;
785
+ delete next.pausedBy;
786
+ delete next.pausedReason;
787
+ delete next.pausedAtTaskID;
788
+ delete next.step;
789
+ if (skip)
790
+ next.skipBreakpointTaskID = skip;
791
+ else
792
+ delete next.skipBreakpointTaskID;
793
+ return {
794
+ Fields: [
795
+ TaskClaimStore.DebugField('$.debug.paused', { Kind: 'null' }),
796
+ TaskClaimStore.DebugField('$.debug.pausedReason', { Kind: 'null' }),
797
+ TaskClaimStore.DebugField('$.debug.pausedBy', { Kind: 'null' }),
798
+ TaskClaimStore.DebugField('$.debug.pausedAtTaskID', { Kind: 'null' }),
799
+ TaskClaimStore.DebugField('$.debug.step', { Kind: 'null' }),
800
+ skip
801
+ ? TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'string', Value: skip })
802
+ : TaskClaimStore.DebugField('$.debug.skipBreakpointTaskID', { Kind: 'null' }),
803
+ ],
804
+ Next: next,
805
+ };
806
+ });
807
+ }
808
+ /**
809
+ * Arms a one-shot step allowance on a paused graph: `'one'` releases the next eligible task,
810
+ * `'wave'` releases the current frontier, a task ID releases exactly that task. The dispatcher
811
+ * consumes the allowance CAS-style, so two instances stepping the same graph release work once.
812
+ */
813
+ async StepGraph(parentTaskID, target, context) {
814
+ const parent = await this.loadWorkflowParent(parentTaskID, context);
815
+ if (!parent)
816
+ return this.controlResult(false, undefined, 'Not a workflow graph this control plane can act on.');
817
+ const current = ParseTaskGraphDebugState(parent.inputPayload);
818
+ if (!current.paused) {
819
+ return this.controlResult(false, current, 'Step only applies to a paused workflow — pause it first.');
820
+ }
821
+ return this.writeDebugFields(parentTaskID, context, (state) => ({
822
+ Fields: [TaskClaimStore.DebugField('$.debug.step', { Kind: 'string', Value: target })],
823
+ Next: { ...state, step: target },
824
+ }));
825
+ }
826
+ /**
827
+ * Replaces the graph's breakpoint set. Every ID must name a child of this graph — a breakpoint
828
+ * on a task in some other graph would gate nothing and silently lie to the person who set it.
829
+ */
830
+ async SetBreakpoints(parentTaskID, taskIDs, context) {
831
+ const children = await this.loadChildren(parentTaskID, context);
832
+ const childIDs = new Set(children.map((c) => c.ID.toLowerCase()));
833
+ const foreign = taskIDs.filter((id) => !childIDs.has(id.toLowerCase()));
834
+ if (foreign.length > 0) {
835
+ return this.controlResult(false, undefined, `Not steps of this workflow: ${foreign.join(', ')}`);
836
+ }
837
+ return this.writeDebugFields(parentTaskID, context, (current) => {
838
+ const next = { ...current };
839
+ if (taskIDs.length > 0)
840
+ next.breakpoints = [...taskIDs];
841
+ else
842
+ delete next.breakpoints;
843
+ return {
844
+ Fields: [
845
+ TaskClaimStore.DebugField('$.debug.breakpoints', taskIDs.length > 0 ? { Kind: 'json', Value: JSON.stringify(taskIDs) } : { Kind: 'null' }),
846
+ ],
847
+ Next: next,
848
+ };
849
+ });
850
+ }
851
+ /**
852
+ * Overrides one edge's condition verdict — the operator's answer for a path the engine cannot
853
+ * decide (a held graph) or decided wrongly (a broken guard). `'false'` reads as "branch not
854
+ * taken" and cascades skips; `'true'` opens the gate; `null` removes the override.
855
+ */
856
+ async SetEdgeOverride(parentTaskID, edgeID, verdict, context) {
857
+ // Prove the edge belongs to this graph before writing anything about it.
858
+ const edge = await context.Provider.GetEntityObject('MJ: Task Dependencies', context.ContextUser);
859
+ if (!(await edge.Load(edgeID)))
860
+ return this.controlResult(false, undefined, 'No such path.');
861
+ const target = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
862
+ if (!(await target.Load(edge.TaskID)) || !UUIDsEqual(target.ParentID ?? '', parentTaskID)) {
863
+ return this.controlResult(false, undefined, 'That path does not belong to this workflow.');
864
+ }
865
+ // AN UNCONDITIONAL PATH CANNOT BE OVERRIDDEN — refused here so the two dialects agree.
866
+ //
867
+ // The engine reads overrides at different depths: an ordinary edge only consults one when
868
+ // it HAS a condition, while the exclusive evaluator consults it before its no-condition
869
+ // early return. Left open, the same override would force an unconditional exclusive edge to
870
+ // lose while doing nothing at all to an unconditional ordinary one — the operator's answer
871
+ // meaning two different things depending on a property of the graph they cannot see.
872
+ // Refusing is the conservative reading, and it costs nothing: "don't take this branch" is
873
+ // already expressible, precisely, as SkipTask on the step itself.
874
+ if (verdict !== null && !edge.Condition?.trim()) {
875
+ return this.controlResult(false, undefined, 'That path has no condition to answer — it is always taken. To stop the branch, skip its step instead.');
876
+ }
877
+ return this.writeDebugFields(parentTaskID, context, (current) => {
878
+ const overrides = { ...(current.edgeOverrides ?? {}) };
879
+ if (verdict === null)
880
+ delete overrides[edgeID];
881
+ else
882
+ overrides[edgeID] = verdict;
883
+ const next = { ...current };
884
+ if (Object.keys(overrides).length > 0)
885
+ next.edgeOverrides = overrides;
886
+ else
887
+ delete next.edgeOverrides;
888
+ return {
889
+ // Scoped to THIS edge's key, not the whole map: two operators answering two
890
+ // different held paths at once must not overwrite each other's answer.
891
+ Fields: [
892
+ TaskClaimStore.DebugField(`$.debug.edgeOverrides."${edgeID}"`, verdict === null ? { Kind: 'null' } : { Kind: 'string', Value: verdict }),
893
+ ],
894
+ Next: next,
895
+ };
896
+ });
897
+ }
898
+ /**
899
+ * Declares a Pending step not-taken. Downstream dependents proceed — `Skipped` satisfies a
900
+ * prerequisite — and any open human request for the step is withdrawn so nobody keeps seeing an
901
+ * ask for work the operator decided against.
902
+ */
903
+ async SkipTask(taskID, context) {
904
+ const typeID = await this.ensureTaskType(context);
905
+ const ok = await this.debugWrites.TrySkipPending(context.Provider, taskID, typeID, context.ContextUser);
906
+ if (!ok) {
907
+ return { Success: false, ErrorMessage: 'Only a step that has not started can be skipped.' };
908
+ }
909
+ await this.cancelOpenRequests([taskID], context);
910
+ return { Success: true };
911
+ }
912
+ /**
913
+ * Marks a step Complete with an operator-supplied output.
914
+ *
915
+ * Human steps are refused here on purpose: they already have a first-class completion path
916
+ * (`TaskGraph.CompleteTask`) with the assignee/elevation check, and this verb must not become
917
+ * the door that bypasses it.
918
+ */
919
+ async ForceCompleteTask(taskID, outputPayload, context) {
920
+ const task = await context.Provider.GetEntityObject('MJ: Tasks', context.ContextUser);
921
+ if (!(await task.Load(taskID)))
922
+ return { Success: false, ErrorMessage: 'No such step.' };
923
+ if (task.UserID) {
924
+ return { Success: false, ErrorMessage: 'A human step completes through its assignee — use CompleteTask.' };
925
+ }
926
+ const json = outputPayload == null
927
+ ? null
928
+ : typeof outputPayload === 'string' ? outputPayload : JSON.stringify(outputPayload);
929
+ const typeID = await this.ensureTaskType(context);
930
+ const ok = await this.debugWrites.TryForceComplete(context.Provider, taskID, json, typeID, context.ContextUser);
931
+ if (!ok) {
932
+ return {
933
+ Success: false,
934
+ ErrorMessage: 'The step is running with a live claim, or already finished. Cancel it or wait for the claim to lapse.',
935
+ };
936
+ }
937
+ return { Success: true };
938
+ }
939
+ /**
940
+ * Replaces a Pending step's input — the "edit the brief before stepping" move at a breakpoint.
941
+ * Applies to this run only; the step must not have started.
942
+ *
943
+ * **A guarded statement, not load-check-save.** The in-memory `Status === 'Pending'` check plus
944
+ * `task.Save()` is an unconditional full-row UPDATE: a task claimed in the window between the
945
+ * load and the save has its claim columns reverted to the pre-claim snapshot *while its body
946
+ * runs*, and a second instance then claims it again — the step executes twice. See
947
+ * {@link TaskClaimStore.TryUpdateInputPayload}, which makes the check and the write one atomic
948
+ * operation whose rowcount is the answer.
949
+ */
950
+ async UpdateTaskInput(taskID, inputPayload, context) {
951
+ try {
952
+ const typeID = await this.ensureTaskType(context);
953
+ const json = typeof inputPayload === 'string' ? inputPayload : JSON.stringify(inputPayload);
954
+ const ok = await this.debugWrites.TryUpdateInputPayload(context.Provider, taskID, json, 'Pending', typeID, context.ContextUser);
955
+ if (!ok) {
956
+ return {
957
+ Success: false,
958
+ ErrorMessage: 'Only a workflow step that has not started can have its input edited.',
959
+ };
960
+ }
961
+ return { Success: true };
962
+ }
963
+ catch (e) {
964
+ return { Success: false, ErrorMessage: e instanceof Error ? e.message : String(e) };
965
+ }
966
+ }
487
967
  // ────────────────────────────────────────────────────────────────────────
488
968
  // internals
489
969
  // ────────────────────────────────────────────────────────────────────────
@@ -556,20 +1036,63 @@ export class TaskGraphService {
556
1036
  }
557
1037
  return { Success: true, Map: found };
558
1038
  }
559
- /** Finds or creates the task type used for orchestrated graphs. */
1039
+ /**
1040
+ * Finds or creates the task type used for orchestrated graphs — exactly one of it, ever.
1041
+ *
1042
+ * **This resolves the engine's discriminator, not a label** (R2-7). Round 1 scoped every sweep
1043
+ * arm and both payload-writing guards to this type, so a second row sharing the name lets
1044
+ * different processes bind different IDs — and a graph stamped with the other one is invisible
1045
+ * to the sweep, never claimed, never settled, its submitting run `Paused` forever, with no
1046
+ * error anywhere.
1047
+ *
1048
+ * Race-safe by INSERT-then-reselect rather than by checking harder. Two concurrent first-ever
1049
+ * submissions both read "not there" and both insert; the unique index added in this round makes
1050
+ * the loser's insert fail, and the loser then re-reads and finds the winner's row. Checking
1051
+ * first is what created the window, so the fix cannot be a better check.
1052
+ */
560
1053
  async ensureTaskType(context) {
561
- const existing = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: Task Types', ExtraFilter: `Name='${TASK_TYPE_NAME}'`, Fields: ['ID'], ResultType: 'simple', MaxRows: 1 }, context.ContextUser);
562
- const found = existing.Results?.[0]?.ID;
1054
+ const found = await this.findTaskTypeID(context);
563
1055
  if (found)
564
1056
  return found;
565
1057
  const tt = await context.Provider.GetEntityObject('MJ: Task Types', context.ContextUser);
566
1058
  tt.NewRecord();
567
1059
  tt.Name = TASK_TYPE_NAME;
568
1060
  tt.Description = 'Tasks created by agent-orchestrated workflows.';
569
- if (!(await tt.Save())) {
570
- throw new Error(`Could not create task type: ${tt.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1061
+ if (await tt.Save())
1062
+ return tt.ID;
1063
+ // The insert lost. Almost certainly to the unique index and another process that got there
1064
+ // first — so re-read before treating it as a failure. Any other cause falls through to the
1065
+ // throw below with its own message intact.
1066
+ const winner = await this.findTaskTypeID(context);
1067
+ if (winner)
1068
+ return winner;
1069
+ throw new Error(`Could not create task type: ${tt.LatestResult?.CompleteMessage ?? 'unknown error'}`);
1070
+ }
1071
+ /**
1072
+ * The `AI Workflow` task type's ID, resolved deterministically.
1073
+ *
1074
+ * `ORDER BY` is not decoration: two rows sharing the name come back in whatever order the engine
1075
+ * chooses, so an unordered `MaxRows: 1` lets two processes bind different IDs from the same
1076
+ * data. The index this round adds makes duplicates impossible going forward; the ordering makes
1077
+ * the resolution deterministic on a database that still has some, and the warning makes the
1078
+ * situation visible rather than merely survivable.
1079
+ */
1080
+ async findTaskTypeID(context) {
1081
+ const existing = await RunView.FromMetadataProvider(context.Provider).RunView({
1082
+ EntityName: 'MJ: Task Types',
1083
+ ExtraFilter: `Name='${TASK_TYPE_NAME}'`,
1084
+ Fields: ['ID'],
1085
+ OrderBy: '__mj_CreatedAt ASC, ID ASC',
1086
+ ResultType: 'simple',
1087
+ MaxRows: 2,
1088
+ }, context.ContextUser);
1089
+ const rows = existing.Results ?? [];
1090
+ if (rows.length > 1) {
1091
+ LogError(`[TaskGraphService] More than one '${TASK_TYPE_NAME}' task type exists. Binding the oldest ` +
1092
+ `(${rows[0].ID}), but graphs stamped with the other are invisible to the dispatcher's ` +
1093
+ `sweep and will never settle. Merge them.`);
571
1094
  }
572
- return tt.ID;
1095
+ return rows[0]?.ID ?? null;
573
1096
  }
574
1097
  /** Writes the parent task that represents the graph as a whole. */
575
1098
  async persistParent(spec, taskTypeID, context) {
@@ -591,13 +1114,32 @@ export class TaskGraphService {
591
1114
  // dispatcher memory because the dispatcher that finishes a graph is frequently not the
592
1115
  // process that accepted it — a restart, a second instance, or simply a long-running graph
593
1116
  // all break that assumption. Anything the completion path needs has to be durable too.
594
- parent.InputPayload = JSON.stringify({
1117
+ // Done before the write, and reported: a value that silently left the envelope becomes a
1118
+ // condition reading absent-data and taking the other branch, with nothing saying why.
1119
+ const sanitized = SanitizeInvocationEnvelope(context.Invocation);
1120
+ if (sanitized.DroppedPaths.length > 0) {
1121
+ LogStatus(`[TaskGraphService] Invocation envelope for '${spec.workflowName}' dropped ` +
1122
+ `${sanitized.DroppedPaths.length} non-persistable value(s): ` +
1123
+ `${sanitized.DroppedPaths.join(', ')}. Conditions referencing them will read as ` +
1124
+ `absent data. Pass plain JSON values for anything a condition needs.`);
1125
+ }
1126
+ // Only written when the caller supplied one, so a graph with no invocation envelope
1127
+ // carries no key at all rather than a misleading empty object. SANITIZED first: the
1128
+ // agent's `context` is documented as possibly a class instance holding connections and
1129
+ // credentials, and carrying it verbatim threw `Converting circular structure to JSON`
1130
+ // on any context with a socket in it — killing the run at submit time — while a context
1131
+ // that happened to serialize would have written its credentials to this row.
1132
+ parent.InputPayload = JSON.stringify(BuildTaskGraphParentInputPayload({
595
1133
  continuation: spec.continuation ?? 'message',
596
1134
  reinvokeDepth: context.ReinvokeDepth ?? 0,
597
1135
  failureSemantics: spec.failureSemantics ?? 'block',
598
1136
  submittedByAgentRunID: context.AgentRunID ?? null,
599
1137
  submittedByUserID: context.ContextUser?.ID ?? null,
600
- });
1138
+ invocation: sanitized.Envelope
1139
+ ? { data: sanitized.Envelope.Data, context: sanitized.Envelope.Context }
1140
+ : null,
1141
+ startPaused: context.Debug?.paused === true,
1142
+ }));
601
1143
  if (!(await parent.Save())) {
602
1144
  throw new Error(`Could not create parent task: ${parent.LatestResult?.CompleteMessage ?? 'unknown error'}`);
603
1145
  }
@@ -746,7 +1288,22 @@ export class TaskGraphService {
746
1288
  }
747
1289
  }
748
1290
  async loadChildren(parentTaskID, context) {
749
- const result = await RunView.FromMetadataProvider(context.Provider).RunView({ EntityName: 'MJ: Tasks', ExtraFilter: `ParentID='${parentTaskID}'`, ResultType: 'entity_object' }, context.ContextUser);
1291
+ // `BypassCache` for the reason the dispatcher documents at every one of its reads (C4): task
1292
+ // status is written by the claim protocol's direct SQL, which fires no cache invalidation.
1293
+ // A cached read here returns PRE-EXECUTION state, and both callers act on status — Cancel's
1294
+ // "leave terminal work alone" guard would pass for a task that has since completed, and
1295
+ // write `Cancelled` over a `Complete` row. That is precisely the history-rewriting the guard
1296
+ // exists to prevent, performed by the guard itself.
1297
+ // Type-scoped for the same reason the sub-graph walk is (R3-10): these rows are about to be
1298
+ // written, and `MJ: Tasks` holds conversation tasks and users' own to-dos too.
1299
+ const typeID = await this.findTaskTypeID(context);
1300
+ const ofType = typeID ? `TypeID='${typeID}' AND ` : '';
1301
+ const result = await RunView.FromMetadataProvider(context.Provider).RunView({
1302
+ EntityName: 'MJ: Tasks',
1303
+ ExtraFilter: `${ofType}ParentID='${parentTaskID}'`,
1304
+ ResultType: 'entity_object',
1305
+ BypassCache: true,
1306
+ }, context.ContextUser);
750
1307
  return (result.Success ? result.Results : []) ?? [];
751
1308
  }
752
1309
  }