relay-flow 0.3.8-alpha → 0.3.10-alpha

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/README.md +27 -19
  2. package/cmd/relay-flow/commands_test.go +440 -21
  3. package/cmd/relay-flow/main.go +203 -37
  4. package/cmd/relay-flow/observability_test.go +11 -0
  5. package/cmd/relay-flow/onboarding.go +1 -1
  6. package/cmd/relay-flow/onboarding_test.go +1 -1
  7. package/cmd/relay-flow/scenario_test.go +26 -10
  8. package/cmd/relay-flow/serve.go +27 -1
  9. package/internal/execution/goworkflows/activities.go +11 -0
  10. package/internal/execution/goworkflows/cancellation_test.go +50 -0
  11. package/internal/execution/goworkflows/engine.go +278 -9
  12. package/internal/execution/goworkflows/fakes_test.go +30 -1
  13. package/internal/execution/goworkflows/interpreter.go +18 -1
  14. package/internal/execution/goworkflows/ownership_test.go +259 -0
  15. package/internal/execution/goworkflows/projection.go +8 -0
  16. package/internal/execution/goworkflows/recovery_test.go +563 -1
  17. package/internal/execution/projection/detail_test.go +38 -0
  18. package/internal/execution/projection/projection.go +80 -15
  19. package/internal/execution/temporal/interpreter.go +2 -1
  20. package/internal/execution/temporal/recovery.go +68 -36
  21. package/internal/harness/harness.go +1 -0
  22. package/internal/harness/opencode/opencode.go +5 -0
  23. package/internal/harness/opencode/opencode_test.go +17 -1
  24. package/internal/harness/opencode/repo_setup.go +1 -1
  25. package/internal/harness/pi/pi.go +6 -0
  26. package/internal/harness/pi/task_env_test.go +16 -0
  27. package/internal/recover/recover.go +30 -1
  28. package/internal/repo/binding_test.go +63 -0
  29. package/internal/repo/repo.go +47 -6
  30. package/internal/router/router.go +39 -5
  31. package/internal/router/router_test.go +67 -1
  32. package/internal/run/manager.go +90 -3
  33. package/internal/run/run_manager_test.go +74 -0
  34. package/internal/server/api_test.go +13 -0
  35. package/internal/server/client.go +7 -0
  36. package/internal/server/fixture_test.go +5 -0
  37. package/internal/server/server.go +32 -3
  38. package/internal/task/beads/beads.go +201 -3
  39. package/internal/task/beads/beads_test.go +172 -0
  40. package/internal/task/jira/effects_test.go +86 -0
  41. package/internal/task/jira/filters_test.go +103 -0
  42. package/internal/task/jira/helpers_test.go +11 -2
  43. package/internal/task/jira/jira.go +228 -4
  44. package/internal/task/jira/normalize.go +30 -15
  45. package/internal/task/jira/rest/client.go +17 -2
  46. package/internal/task/jira/transition_defaults_test.go +3 -1
  47. package/internal/task/task.go +66 -0
  48. package/internal/workflow/report.go +3 -0
  49. package/internal/workflow/report_test.go +16 -0
  50. package/internal/workflow/workflow.go +24 -0
  51. package/internal/workflow/workflow_test.go +58 -1
  52. package/package.json +1 -1
@@ -27,8 +27,10 @@ import (
27
27
  "github.com/rajpopat27/relay-flow/internal/harness"
28
28
  "github.com/rajpopat27/relay-flow/internal/identity"
29
29
  "github.com/rajpopat27/relay-flow/internal/repo"
30
+ "github.com/rajpopat27/relay-flow/internal/retry"
30
31
  "github.com/rajpopat27/relay-flow/internal/run"
31
32
  "github.com/rajpopat27/relay-flow/internal/runner"
33
+ "github.com/rajpopat27/relay-flow/internal/task"
32
34
  "github.com/rajpopat27/relay-flow/internal/workflow"
33
35
  )
34
36
 
@@ -61,6 +63,7 @@ type Engine struct {
61
63
  runtime run.RuntimePolicy
62
64
 
63
65
  mu sync.RWMutex
66
+ cancelMu sync.Mutex // serializes cancellation history check + request
64
67
  snapshots map[run.ID]*workflow.Workflow // in-memory cache; history is authoritative
65
68
 
66
69
  workerCtx context.Context
@@ -177,6 +180,11 @@ func (e *Engine) Start(ctx context.Context) error {
177
180
  if err := e.actWorker.Start(e.workerCtx); err != nil {
178
181
  return fmt.Errorf("start activity worker: %w", err)
179
182
  }
183
+ // A cancellation request can outlive the engine instance that accepted it.
184
+ // Reconcile those projections before normal pollers start; a missing
185
+ // workflow instance is a confirmed terminal execution boundary, while
186
+ // other lookup failures remain retryable and are logged below.
187
+ e.reconcileCancelingRuns(ctx)
180
188
  // Startup retention sweep (pre-poller window): remove old terminal
181
189
  // projection rows and their engine histories; nonterminal runs stay.
182
190
  cutoff := time.Now().Add(-e.retention)
@@ -202,6 +210,7 @@ func (e *Engine) registerActivities() error {
202
210
  a.EnsureEnvironment,
203
211
  a.SetEnvironmentStatus,
204
212
  a.LoadNodeRuntime,
213
+ a.LoadCancellationReason,
205
214
  a.EnsureNodeRuntime,
206
215
  a.CloseTerminals,
207
216
  a.CleanupRun,
@@ -462,26 +471,286 @@ func (e *Engine) workflowOf(ctx context.Context, id run.ID) (*workflow.Workflow,
462
471
  return nil, fmt.Errorf("no workflow snapshot in history for run %s", id)
463
472
  }
464
473
 
465
- // CancelRun cancels the workflow instance; cleanup runs on a disconnected
466
- // workflow context and cannot interrupt an already-running activity.
474
+ // CancelRun requests cancellation of the durable workflow. The projection
475
+ // transition is compare-and-set: a concurrent completion wins if it reaches a
476
+ // terminal state first, while a persisted canceling state is the durable
477
+ // request that startup reconciliation retries.
467
478
  func (e *Engine) CancelRun(ctx context.Context, id run.ID, reason string) error {
468
- if _, err := e.runs.get(ctx, id); err != nil {
479
+ r, err := e.runs.beginCancellation(ctx, id, reason)
480
+ if err != nil {
469
481
  return fmt.Errorf("resolve run %s: %w", id, err)
470
482
  }
471
- if err := e.runs.updateState(ctx, id, run.StateCanceling, reason, nil); err != nil {
483
+ if r.State == run.StateCompleted || r.State == run.StateCanceled {
484
+ return nil
485
+ }
486
+ if r.State != run.StateCanceling {
487
+ return fmt.Errorf("cancel %s: state changed to %s", id, r.State)
488
+ }
489
+ // The reason persisted by the first successful CAS is authoritative for
490
+ // every later cancellation request.
491
+ return e.reconcileCancellation(ctx, r, r.LastError)
492
+ }
493
+
494
+ // isMissingWorkflowInstance identifies only confirmed absence of the active
495
+ // go-workflows execution. Database and context failures remain retryable.
496
+ func isMissingWorkflowInstance(err error) bool {
497
+ return errors.Is(err, sql.ErrNoRows) || errors.Is(err, backend.ErrInstanceNotFound)
498
+ }
499
+
500
+ // detachedContext preserves a caller deadline while detaching cancellation
501
+ // from the caller. Startup cleanup must not be able to outlive its bounded
502
+ // reconciliation window.
503
+ func detachedContext(ctx context.Context) (context.Context, context.CancelFunc) {
504
+ base := context.WithoutCancel(ctx)
505
+ if deadline, ok := ctx.Deadline(); ok {
506
+ return context.WithDeadline(base, deadline)
507
+ }
508
+ return base, func() {}
509
+ }
510
+
511
+ // lookupInstance returns the active execution when present. If only a finished
512
+ // row remains, finished is true; a missing row is the only absence outcome.
513
+ func (e *Engine) lookupInstance(ctx context.Context, id run.ID) (*goworkflow.Instance, bool, error) {
514
+ var execID string
515
+ var state int
516
+ err := e.db.QueryRowContext(ctx, `
517
+ SELECT execution_id, state FROM instances WHERE id = ?
518
+ ORDER BY CASE WHEN state = 0 THEN 0 ELSE 1 END, rowid DESC LIMIT 1`, string(id)).Scan(&execID, &state)
519
+ if err != nil {
520
+ return nil, false, fmt.Errorf("workflow instance %s not found: %w", id, err)
521
+ }
522
+ return &goworkflow.Instance{InstanceID: string(id), ExecutionID: execID}, state != 0, nil
523
+ }
524
+
525
+ // reconcileCancellation retries the durable cancellation request while an
526
+ // execution is active and reconciles finished/missing executions separately.
527
+ func (e *Engine) reconcileCancellation(ctx context.Context, r run.Run, reason string) error {
528
+ inst, finished, err := e.lookupInstance(ctx, r.ID)
529
+ if err != nil {
530
+ if isMissingWorkflowInstance(err) {
531
+ return e.finalizeMissingCancellation(ctx, r, reason)
532
+ }
533
+ return fmt.Errorf("resolve workflow instance: %w", err)
534
+ }
535
+ if finished {
536
+ return e.reconcileFinishedCancellation(ctx, r, inst, reason)
537
+ }
538
+ if e.client == nil {
539
+ return errors.New("workflow engine is not started")
540
+ }
541
+ if err := e.requestCancellation(ctx, inst); err != nil {
542
+ if isMissingWorkflowInstance(err) {
543
+ return e.finalizeMissingCancellation(ctx, r, reason)
544
+ }
472
545
  return err
473
546
  }
474
- inst, err := e.instance(ctx, id)
547
+ return nil
548
+ }
549
+
550
+ // requestCancellation checks durable history and appends at most one
551
+ // cancellation event for an execution. The mutex closes the check/request
552
+ // race between concurrent operator calls in this process; the relay lock
553
+ // prevents another relay-flow server from owning the same database.
554
+ func (e *Engine) requestCancellation(ctx context.Context, inst *goworkflow.Instance) error {
555
+ e.cancelMu.Lock()
556
+ defer e.cancelMu.Unlock()
557
+ var requested int
558
+ if err := e.db.QueryRowContext(ctx, `
559
+ SELECT EXISTS(
560
+ SELECT 1 FROM history
561
+ WHERE instance_id = ? AND execution_id = ? AND event_type = ?
562
+ UNION ALL
563
+ SELECT 1 FROM pending_events
564
+ WHERE instance_id = ? AND execution_id = ? AND event_type = ?
565
+ )`,
566
+ inst.InstanceID, inst.ExecutionID, history.EventType_WorkflowExecutionCanceled,
567
+ inst.InstanceID, inst.ExecutionID, history.EventType_WorkflowExecutionCanceled,
568
+ ).Scan(&requested); err != nil {
569
+ return fmt.Errorf("inspect cancellation history: %w", err)
570
+ }
571
+ if requested != 0 {
572
+ return nil
573
+ }
574
+ return e.client.CancelWorkflowInstance(ctx, inst)
575
+ }
576
+
577
+ // finalizeMissingCancellation performs the same roll-forward cleanup as the
578
+ // workflow's disconnected cancellation path when the workflow instance has
579
+ // genuinely disappeared. It intentionally does not recreate an execution.
580
+ func (e *Engine) finalizeMissingCancellation(ctx context.Context, r run.Run, reason string) error {
581
+ cleanupCtx, cleanupCancel := detachedContext(ctx)
582
+ defer cleanupCancel()
583
+ current, err := e.runs.get(cleanupCtx, r.ID)
584
+ if err != nil {
585
+ return fmt.Errorf("reconcile canceled run %s projection: %w", r.ID, err)
586
+ }
587
+ if current.State == run.StateCompleted || current.State == run.StateCanceled {
588
+ return nil
589
+ }
590
+ if current.State != run.StateCanceling {
591
+ return fmt.Errorf("reconcile canceled run %s: state changed to %s", r.ID, current.State)
592
+ }
593
+ if inst, finished, lookupErr := e.lookupInstance(cleanupCtx, r.ID); lookupErr == nil {
594
+ if finished {
595
+ return e.reconcileFinishedCancellation(cleanupCtx, current, inst, reason)
596
+ }
597
+ return fmt.Errorf("reconcile canceled run %s: workflow instance reappeared", r.ID)
598
+ } else if !isMissingWorkflowInstance(lookupErr) {
599
+ return fmt.Errorf("reconcile canceled run %s instance lookup: %w", r.ID, lookupErr)
600
+ }
601
+ return e.finalizeCancellationEffects(cleanupCtx, current, reason)
602
+ }
603
+
604
+ // finalizeCancellationEffects is shared by missing and already-canceled
605
+ // engine executions. The final projection transition is conditional so a
606
+ // concurrent terminal projection cannot be overwritten.
607
+ func (e *Engine) finalizeCancellationEffects(ctx context.Context, r run.Run, reason string) error {
608
+ if r.State == run.StateCompleted || r.State == run.StateCanceled {
609
+ return nil
610
+ }
611
+ if r.State != run.StateCanceling {
612
+ return fmt.Errorf("finalize canceled run %s: state changed to %s", r.ID, r.State)
613
+ }
614
+ repoInfo, ok := e.activities.Repos.Get(r.Repo)
615
+ if !ok {
616
+ return fmt.Errorf("reconcile canceled run %s: repo %q is no longer registered", r.ID, r.Repo)
617
+ }
618
+ work := run.Work{
619
+ RunID: r.ID,
620
+ LogicalID: r.LogicalID,
621
+ AttemptID: r.AttemptID,
622
+ Repo: r.Repo,
623
+ Workflow: r.Workflow,
624
+ Parent: r.Ticket,
625
+ Runtime: e.runtime,
626
+ }
627
+ if reason == "" {
628
+ reason = "operator requested cancellation"
629
+ }
630
+ if err := e.activities.FinalizeNodeRuntimes(ctx, work, repoInfo.Path, e.runtime); err != nil {
631
+ return fmt.Errorf("finalize canceled run %s terminals: %w", r.ID, err)
632
+ }
633
+ markerID := r.LogicalID
634
+ if markerID == "" {
635
+ markerID = r.ID
636
+ }
637
+ if err := e.activities.Comment(ctx, r.Repo, run.CommentWork{
638
+ RunID: r.ID,
639
+ Item: task.Target{Parent: r.Ticket},
640
+ Body: "Run canceled: " + reason,
641
+ Marker: run.CancellationMarker(markerID),
642
+ }); err != nil {
643
+ return fmt.Errorf("finalize canceled run %s comment: %w", r.ID, err)
644
+ }
645
+ finished := time.Now().UTC()
646
+ updated, err := e.runs.updateStateIf(ctx, r.ID, run.StateCanceling, run.StateCanceled, "", &finished)
475
647
  if err != nil {
476
- return fmt.Errorf("cancel %s: %w", id, err)
648
+ return fmt.Errorf("finalize canceled run %s projection: %w", r.ID, err)
477
649
  }
478
- if err := e.client.CancelWorkflowInstance(ctx, inst); err != nil {
479
- return fmt.Errorf("cancel %s: %w", id, err)
650
+ if !updated {
651
+ latest, getErr := e.runs.get(ctx, r.ID)
652
+ if getErr == nil && (latest.State == run.StateCanceled || latest.State == run.StateCompleted) {
653
+ return nil
654
+ }
655
+ if getErr != nil {
656
+ return fmt.Errorf("finalize canceled run %s projection after race: %w", r.ID, getErr)
657
+ }
658
+ return fmt.Errorf("finalize canceled run %s: state changed to %s", r.ID, latest.State)
480
659
  }
660
+ slog.Info("run canceled", "ticket", r.Ticket.Key, "runID", string(r.ID),
661
+ "repo", r.Repo, "workflow", r.Workflow, "state", run.StateCanceled)
481
662
  return nil
482
663
  }
483
664
 
484
- // instance resolves the current execution for the durable run ID.
665
+ // reconcileFinishedCancellation reads the terminal engine event before
666
+ // deciding whether a canceling projection should become completed or canceled.
667
+ func (e *Engine) reconcileFinishedCancellation(ctx context.Context, r run.Run, inst *goworkflow.Instance, reason string) error {
668
+ if e.backend == nil {
669
+ return errors.New("workflow backend is not started")
670
+ }
671
+ events, err := e.backend.GetWorkflowInstanceHistory(ctx, inst, nil)
672
+ if err != nil {
673
+ return fmt.Errorf("read finished workflow history: %w", err)
674
+ }
675
+ var finishedEvent *history.Event
676
+ canceledBeforeFinish := false
677
+ for _, event := range events {
678
+ switch event.Type {
679
+ case history.EventType_WorkflowExecutionCanceled:
680
+ // TicketWorkflow records this cancellation request first; after
681
+ // cancelCleanup returns, go-workflows may append a normal
682
+ // WorkflowExecutionFinished event. The earlier cancellation event
683
+ // remains authoritative for relay-flow semantics.
684
+ if finishedEvent == nil {
685
+ canceledBeforeFinish = true
686
+ }
687
+ case history.EventType_WorkflowExecutionFinished:
688
+ if finishedEvent == nil {
689
+ finishedEvent = event
690
+ }
691
+ case history.EventType_WorkflowExecutionTerminated:
692
+ return fmt.Errorf("workflow %s terminated without a cancellation or completion result", r.ID)
693
+ }
694
+ }
695
+ if canceledBeforeFinish {
696
+ return e.finalizeCancellationEffects(ctx, r, reason)
697
+ }
698
+ if finishedEvent == nil {
699
+ return fmt.Errorf("finished workflow %s has no terminal history event", r.ID)
700
+ }
701
+ {
702
+ finished := finishedEvent.Timestamp
703
+ updated, err := e.runs.updateStateIf(ctx, r.ID, run.StateCanceling, run.StateCompleted, "", &finished)
704
+ if err != nil {
705
+ return fmt.Errorf("reconcile completed run %s projection: %w", r.ID, err)
706
+ }
707
+ if !updated {
708
+ latest, getErr := e.runs.get(ctx, r.ID)
709
+ if getErr == nil && (latest.State == run.StateCompleted || latest.State == run.StateCanceled) {
710
+ return nil
711
+ }
712
+ if getErr != nil {
713
+ return fmt.Errorf("reconcile completed run %s projection after race: %w", r.ID, getErr)
714
+ }
715
+ return fmt.Errorf("reconcile completed run %s: state changed to %s", r.ID, latest.State)
716
+ }
717
+ return nil
718
+ }
719
+ }
720
+
721
+ // reconcileCancelingRuns retries the durable cancellation request for every
722
+ // stale canceling projection. An active instance is not skipped: its cancel
723
+ // event is re-submitted until the workflow accepts it.
724
+ func (e *Engine) reconcileCancelingRuns(ctx context.Context) {
725
+ active := true
726
+ runs, err := e.runs.list(ctx, run.Filter{Active: &active})
727
+ if err != nil {
728
+ slog.Warn("reconcile canceling runs unavailable", "error", err)
729
+ return
730
+ }
731
+ for _, r := range runs {
732
+ if r.State != run.StateCanceling {
733
+ continue
734
+ }
735
+ if err := e.retryStartupCancellation(ctx, r); err != nil {
736
+ slog.Warn("reconcile canceling run failed", "runID", r.ID, "error", err)
737
+ }
738
+ }
739
+ }
740
+
741
+ // retryStartupCancellation gives a durable canceling projection a bounded
742
+ // startup retry window. Normal polling deliberately does not recreate or
743
+ // advance canceling runs, so an active execution must receive its cancel
744
+ // event here or remain visibly retryable for the next restart.
745
+ func (e *Engine) retryStartupCancellation(ctx context.Context, r run.Run) error {
746
+ retryCtx, cancel := context.WithTimeout(context.WithoutCancel(ctx), 5*time.Second)
747
+ defer cancel()
748
+ return retry.Do(retryCtx, retry.DefaultBackoffPolicy, func() error {
749
+ return e.reconcileCancellation(retryCtx, r, r.LastError)
750
+ })
751
+ }
752
+
753
+ // instance resolves the current active execution for the durable run ID.
485
754
  func (e *Engine) instance(ctx context.Context, id run.ID) (*goworkflow.Instance, error) {
486
755
  var execID string
487
756
  err := e.db.QueryRowContext(ctx,
@@ -73,6 +73,7 @@ type fakeTaskSystem struct {
73
73
  labels map[string][]string // mailbox key -> labels
74
74
  specs []task.MailboxSpec
75
75
  comments []recordedComment
76
+ hasCommentN int
76
77
  resets []string
77
78
  renderText func(task.TextKind, task.TextData) (string, error)
78
79
 
@@ -115,6 +116,18 @@ func (s *fakeTaskSystem) CompileFilter(config.RawValues) (func(task.Ticket) bool
115
116
  return func(task.Ticket) bool { return true }, nil
116
117
  }
117
118
 
119
+ func (s *fakeTaskSystem) CompileOwnershipFilter(config.RawValues) (func(task.Ticket) bool, error) {
120
+ return func(task.Ticket) bool { return true }, nil
121
+ }
122
+
123
+ func (s *fakeTaskSystem) ValidateOwnership(context.Context, task.TicketRef, string, config.RawValues) error {
124
+ return nil
125
+ }
126
+
127
+ func (s *fakeTaskSystem) ClaimIfOwned(ctx context.Context, ref task.TicketRef, workflowName string, _ config.RawValues) error {
128
+ return s.Claim(ctx, ref, workflowName)
129
+ }
130
+
118
131
  func (s *fakeTaskSystem) Claim(_ context.Context, ref task.TicketRef, wf string) error {
119
132
  s.log.add("claim:" + ref.Key + ":" + wf)
120
133
  return nil
@@ -208,6 +221,7 @@ func (s *fakeTaskSystem) CompleteMailbox(_ context.Context, mb task.Mailbox) err
208
221
 
209
222
  func (s *fakeTaskSystem) HasComment(_ context.Context, target task.Target, marker string) (bool, error) {
210
223
  s.mu.Lock()
224
+ s.hasCommentN++
211
225
  defer s.mu.Unlock()
212
226
  for _, c := range s.comments {
213
227
  if c.Marker == marker {
@@ -226,7 +240,16 @@ func (s *fakeTaskSystem) Comment(_ context.Context, target task.Target, body, ma
226
240
  if target.Mailbox != nil {
227
241
  key = target.Mailbox.Key
228
242
  }
229
- if s.failComments {
243
+ s.mu.Lock()
244
+ for _, comment := range s.comments {
245
+ if comment.Key == key && comment.Marker == marker {
246
+ s.mu.Unlock()
247
+ return nil
248
+ }
249
+ }
250
+ fail := s.failComments
251
+ s.mu.Unlock()
252
+ if fail {
230
253
  s.log.add("commentFail:" + key)
231
254
  return errTransient
232
255
  }
@@ -258,6 +281,12 @@ func (s *fakeTaskSystem) ResetForRecovery(_ context.Context, parent task.TicketR
258
281
  return nil
259
282
  }
260
283
 
284
+ func (s *fakeTaskSystem) hasCommentCount() int {
285
+ s.mu.Lock()
286
+ defer s.mu.Unlock()
287
+ return s.hasCommentN
288
+ }
289
+
261
290
  func (s *fakeTaskSystem) commentBodies(key string) []recordedComment {
262
291
  s.mu.Lock()
263
292
  defer s.mu.Unlock()
@@ -75,7 +75,23 @@ func (a *Activities) TicketWorkflow(ctx goworkflow.Context, start run.Start) err
75
75
  WorkflowTaskConfig: start.Workflow.TaskConfig,
76
76
  Runtime: start.Runtime,
77
77
  }
78
- return a.cancelCleanup(ctx, work, start.RepoPath, "canceled")
78
+ // Cancellation events do not carry relay-flow's operator reason. Read
79
+ // the reason persisted by CancelRun through the durable disconnected
80
+ // retry path so cleanup cannot write a permanently wrong comment when
81
+ // the projection is temporarily unavailable.
82
+ dctx := goworkflow.NewDisconnectedContext(ctx)
83
+ reason, reasonErr := retryLoop(dctx, start.ID, a, work, "",
84
+ func(ctx2 goworkflow.Context) goworkflow.Future[string] {
85
+ return goworkflow.ExecuteActivity[string](ctx2, noNativeRetries,
86
+ a.LoadCancellationReason, start.ID)
87
+ })
88
+ if reasonErr != nil {
89
+ return reasonErr
90
+ }
91
+ if reason == "" {
92
+ reason = "canceled"
93
+ }
94
+ return a.cancelCleanup(ctx, work, start.RepoPath, reason)
79
95
  }
80
96
  return err
81
97
  }
@@ -276,6 +292,7 @@ func (a *Activities) runGraph(ctx goworkflow.Context, start run.Start) error {
276
292
  Ticket: start.Ticket.Key,
277
293
  Node: current,
278
294
  NodeType: node.Type,
295
+ AutoReject: node.AutoReject,
279
296
  Agent: node.Agent,
280
297
  Title: title,
281
298
  NudgePrompt: node.NudgePrompt,
@@ -0,0 +1,259 @@
1
+ package goworkflows_test
2
+
3
+ import (
4
+ "context"
5
+ "errors"
6
+ "sync"
7
+ "testing"
8
+ "time"
9
+
10
+ "github.com/rajpopat27/relay-flow/internal/config"
11
+ "github.com/rajpopat27/relay-flow/internal/execution/goworkflows"
12
+ "github.com/rajpopat27/relay-flow/internal/identity"
13
+ recoverpkg "github.com/rajpopat27/relay-flow/internal/recover"
14
+ "github.com/rajpopat27/relay-flow/internal/router"
15
+ "github.com/rajpopat27/relay-flow/internal/run"
16
+ "github.com/rajpopat27/relay-flow/internal/task"
17
+ "github.com/rajpopat27/relay-flow/internal/workflow"
18
+ )
19
+
20
+ // ownershipCompositionState models the provider-owned labels and current
21
+ // assignee shared by two independent relay-flow databases. The adapter fake
22
+ // persists a durable wf-owner marker when the first server claims the ticket.
23
+ type ownershipCompositionState struct {
24
+ mu sync.Mutex
25
+ assignee string
26
+ labels []string
27
+ }
28
+
29
+ type ownershipCompositionTaskSystem struct {
30
+ *fakeTaskSystem
31
+ owner string
32
+ state *ownershipCompositionState
33
+ }
34
+
35
+ func newOwnershipCompositionTaskSystem(owner string, state *ownershipCompositionState, log *eventLog) *ownershipCompositionTaskSystem {
36
+ return &ownershipCompositionTaskSystem{
37
+ fakeTaskSystem: newFakeTaskSystem(log), owner: owner, state: state,
38
+ }
39
+ }
40
+
41
+ func (s *ownershipCompositionTaskSystem) Poll(context.Context) ([]task.Ticket, error) {
42
+ s.state.mu.Lock()
43
+ defer s.state.mu.Unlock()
44
+ claims := make([]string, 0, 1)
45
+ for _, label := range s.state.labels {
46
+ if len(label) > len("wf:") && label[:len("wf:")] == "wf:" {
47
+ claims = append(claims, label)
48
+ }
49
+ }
50
+ return []task.Ticket{{
51
+ ID: "1", Key: "PAY-101", Title: "ownership composition",
52
+ WorkflowClaims: claims,
53
+ Fields: map[string]any{
54
+ "assignee": s.state.assignee,
55
+ "labels": append([]string(nil), s.state.labels...),
56
+ },
57
+ }}, nil
58
+ }
59
+
60
+ func (s *ownershipCompositionTaskSystem) CompileFilter(config.RawValues) (func(task.Ticket) bool, error) {
61
+ return func(ticket task.Ticket) bool {
62
+ assignee, _ := ticket.Fields["assignee"].(string)
63
+ return assignee == s.owner
64
+ }, nil
65
+ }
66
+
67
+ func (s *ownershipCompositionTaskSystem) CompileOwnershipFilter(config.RawValues) (func(task.Ticket) bool, error) {
68
+ return func(ticket task.Ticket) bool {
69
+ if len(ticket.WorkflowClaims) != 1 {
70
+ return false
71
+ }
72
+ assignee, _ := ticket.Fields["assignee"].(string)
73
+ if assignee != s.owner {
74
+ return false
75
+ }
76
+ want := "wf-owner:" + ticket.WorkflowClaims[0][len("wf:"):] + ":" + s.owner
77
+ labels, _ := ticket.Fields["labels"].([]string)
78
+ for _, label := range labels {
79
+ if label == want {
80
+ return true
81
+ }
82
+ }
83
+ return false
84
+ }, nil
85
+ }
86
+
87
+ func (s *ownershipCompositionTaskSystem) ValidateOwnership(context.Context, task.TicketRef, string, config.RawValues) error {
88
+ s.state.mu.Lock()
89
+ defer s.state.mu.Unlock()
90
+ if s.hasOwnerMarkerLocked("ownershipFlow") {
91
+ return nil
92
+ }
93
+ return &task.OwnershipMismatchError{Ticket: "PAY-101", Workflow: "ownershipFlow"}
94
+ }
95
+
96
+ func (s *ownershipCompositionTaskSystem) BackfillClaimOwner(context.Context, task.TicketRef, string, config.RawValues) error {
97
+ s.state.mu.Lock()
98
+ defer s.state.mu.Unlock()
99
+ if !containsLabel(s.state.labels, "wf:ownershipFlow") {
100
+ return &task.OwnershipMismatchError{Ticket: "PAY-101", Workflow: "ownershipFlow"}
101
+ }
102
+ marker := "wf-owner:ownershipFlow:" + s.owner
103
+ if !containsLabel(s.state.labels, marker) {
104
+ s.state.labels = append(s.state.labels, marker)
105
+ }
106
+ return nil
107
+ }
108
+
109
+ func (s *ownershipCompositionTaskSystem) ClaimIfOwned(ctx context.Context, ticket task.TicketRef, workflowName string, _ config.RawValues) error {
110
+ s.state.mu.Lock()
111
+ defer s.state.mu.Unlock()
112
+ if s.state.assignee != s.owner {
113
+ return &task.OwnershipMismatchError{Ticket: ticket.Key, Workflow: workflowName}
114
+ }
115
+ ownerMarker := "wf-owner:" + workflowName + ":" + s.owner
116
+ claim := "wf:" + workflowName
117
+ for _, label := range s.state.labels {
118
+ if label == ownerMarker && containsLabel(s.state.labels, claim) {
119
+ return nil
120
+ }
121
+ if len(label) > len("wf:") && label[:len("wf:")] == "wf:" {
122
+ return &task.OwnershipMismatchError{Ticket: ticket.Key, Workflow: workflowName}
123
+ }
124
+ }
125
+ s.state.labels = append(s.state.labels, ownerMarker, claim)
126
+ return nil
127
+ }
128
+
129
+ func (s *ownershipCompositionTaskSystem) hasOwnerMarkerLocked(workflowName string) bool {
130
+ return containsLabel(s.state.labels, "wf-owner:"+workflowName+":"+s.owner)
131
+ }
132
+
133
+ func containsLabel(labels []string, want string) bool {
134
+ for _, label := range labels {
135
+ if label == want {
136
+ return true
137
+ }
138
+ }
139
+ return false
140
+ }
141
+
142
+ func TestExplicitLegacyBackfillPreservesClaimAndEnablesRoutingAndRecovery(t *testing.T) {
143
+ state := &ownershipCompositionState{assignee: "alice", labels: []string{"wf:ownershipFlow"}}
144
+ log := newEventLog()
145
+ sys := newOwnershipCompositionTaskSystem("alice", state, log)
146
+ reg := repoRegistryWith("payments", sys)
147
+ wf := linearWorkflow(false)
148
+ wf.Name = "ownershipFlow"
149
+ wf.TaskConfig = config.RawValues{"filters": map[string]any{"assignees": []any{"currentUser()"}}}
150
+ if err := reg.BindWorkflows([]*workflow.Workflow{&wf}); err != nil {
151
+ t.Fatal(err)
152
+ }
153
+ workflowReg := &workflow.Registry{}
154
+ workflowReg.Replace(&wf)
155
+ fr := newFakeRunner(log)
156
+ engine := newEngine(t, goworkflows.Dependencies{Repos: reg, Runner: fr, Harness: newFakeHarness(log)})
157
+ manager := &run.RunManager{Executor: engine, Runs: engine, Repos: reg, Workflows: workflowReg}
158
+ if err := manager.BackfillClaimOwner(context.Background(), "payments", "PAY-101", "ownershipFlow"); err != nil {
159
+ t.Fatalf("explicit backfill failed: %v", err)
160
+ }
161
+ state.mu.Lock()
162
+ labels := append([]string(nil), state.labels...)
163
+ state.mu.Unlock()
164
+ if !containsLabel(labels, "wf:ownershipFlow") || !containsLabel(labels, "wf-owner:ownershipFlow:alice") || len(labels) != 2 {
165
+ t.Fatalf("backfill labels = %v, want preserved claim plus one provenance marker", labels)
166
+ }
167
+ rp, _ := reg.Get("payments")
168
+ ticket, _ := sys.Poll(context.Background())
169
+ if _, err := router.ResolveWorkflow(rp, ticket[0]); err != nil {
170
+ t.Fatalf("backfilled claim did not route normally: %v", err)
171
+ }
172
+ specsFor := func(system task.System, work run.Work, w *workflow.Workflow) ([]task.MailboxSpec, error) {
173
+ return goworkflows.RenderMailboxSpecs(system, work, w)
174
+ }
175
+ if err := recoverpkg.FromTaskSystem(context.Background(), reg, fr, manager, specsFor); err != nil {
176
+ t.Fatalf("backfilled claim did not recover: %v", err)
177
+ }
178
+ if runs, err := engine.ListRuns(context.Background(), run.Filter{Repo: "payments", Workflow: "ownershipFlow", Ticket: "PAY-101"}); err != nil || len(runs) != 1 {
179
+ t.Fatalf("recovered runs = %v, %v; want one run", runs, err)
180
+ }
181
+ }
182
+
183
+ func TestIndependentRunDatabasesCannotTakeOverReassignedClaim(t *testing.T) {
184
+ state := &ownershipCompositionState{assignee: "alice"}
185
+ logA, logB := newEventLog(), newEventLog()
186
+ sysA := newOwnershipCompositionTaskSystem("alice", state, logA)
187
+ sysB := newOwnershipCompositionTaskSystem("bob", state, logB)
188
+ regA := repoRegistryWith("payments", sysA)
189
+ regB := repoRegistryWith("payments", sysB)
190
+ wf := linearWorkflow(false)
191
+ wf.Name = "ownershipFlow"
192
+ wf.TaskConfig = config.RawValues{"filters": map[string]any{
193
+ "assignees": []any{"currentUser()"},
194
+ }}
195
+ if err := regA.BindWorkflows([]*workflow.Workflow{&wf}); err != nil {
196
+ t.Fatal(err)
197
+ }
198
+ if err := regB.BindWorkflows([]*workflow.Workflow{&wf}); err != nil {
199
+ t.Fatal(err)
200
+ }
201
+
202
+ engineA := newEngine(t, goworkflows.Dependencies{
203
+ Repos: regA, Runner: newFakeRunner(logA), Harness: newFakeHarness(logA),
204
+ })
205
+ engineB := newEngine(t, goworkflows.Dependencies{
206
+ Repos: regB, Runner: newFakeRunner(logB), Harness: newFakeHarness(logB),
207
+ })
208
+ managerA := &run.RunManager{Executor: engineA, Runs: engineA}
209
+ managerB := &run.RunManager{Executor: engineB, Runs: engineB}
210
+
211
+ ticketA, err := sysA.Poll(context.Background())
212
+ if err != nil {
213
+ t.Fatal(err)
214
+ }
215
+ rpA, _ := regA.Get("payments")
216
+ resolvedA, err := router.ResolveWorkflow(rpA, ticketA[0])
217
+ if err != nil {
218
+ t.Fatalf("owner A route failed: %v", err)
219
+ }
220
+ if err := managerA.EnsureRun(context.Background(), rpA, resolvedA, ticketA[0]); err != nil {
221
+ t.Fatalf("owner A EnsureRun failed: %v", err)
222
+ }
223
+ waitFor(t, 10*time.Second, func() bool {
224
+ current, err := engineA.GetRun(context.Background(), identity.NewRunID("payments", "ownershipFlow", "PAY-101"))
225
+ return err == nil && current.State != run.StateCompleted && current.State != run.StateCanceled
226
+ })
227
+
228
+ state.mu.Lock()
229
+ state.assignee = "bob"
230
+ state.mu.Unlock()
231
+ ticketB, err := sysB.Poll(context.Background())
232
+ if err != nil {
233
+ t.Fatal(err)
234
+ }
235
+ rpB, _ := regB.Get("payments")
236
+ if _, err := router.ResolveWorkflow(rpB, ticketB[0]); !errors.Is(err, router.ErrClaimOwnerMismatch) {
237
+ t.Fatalf("owner B route error = %v, want claim-owner-mismatch", err)
238
+ }
239
+ if err := managerB.EnsureRun(context.Background(), rpB, &wf, ticketB[0]); !errors.Is(err, task.ErrOwnershipMismatch) {
240
+ t.Fatalf("owner B EnsureRun error = %v, want ownership mismatch", err)
241
+ }
242
+ if runs, err := engineB.ListRuns(context.Background(), run.Filter{Repo: "payments", Workflow: "ownershipFlow", Ticket: "PAY-101"}); err != nil || len(runs) != 0 {
243
+ t.Fatalf("owner B database runs = %v, %v; want none", runs, err)
244
+ }
245
+
246
+ // Reassignment is a current-owner mismatch for both servers. The
247
+ // original durable run remains in database A, but neither server calls
248
+ // EnsureRun to reconcile or create execution state after the mismatch.
249
+ ticketAAfterReassignment, _ := sysA.Poll(context.Background())
250
+ if _, err := router.ResolveWorkflow(rpA, ticketAAfterReassignment[0]); !errors.Is(err, router.ErrClaimOwnerMismatch) {
251
+ t.Fatalf("owner A route after reassignment = %v, want claim-owner-mismatch", err)
252
+ }
253
+ if err := managerA.EnsureRun(context.Background(), rpA, &wf, ticketAAfterReassignment[0]); !errors.Is(err, task.ErrOwnershipMismatch) {
254
+ t.Fatalf("owner A EnsureRun after reassignment = %v, want ownership mismatch", err)
255
+ }
256
+ if runs, err := engineA.ListRuns(context.Background(), run.Filter{Repo: "payments", Workflow: "ownershipFlow", Ticket: "PAY-101"}); err != nil || len(runs) != 1 {
257
+ t.Fatalf("owner A database runs after reassignment = %v, %v; want one existing run", runs, err)
258
+ }
259
+ }