by-framework 0.2.2.dev10__py3-none-any.whl → 0.2.2.dev12__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
by_framework/__main__.py CHANGED
@@ -12,6 +12,17 @@ import importlib
12
12
  from .worker.app import run_worker
13
13
 
14
14
 
15
+ def _parse_cluster_nodes(value: str):
16
+ """Parse a comma-separated "host:port,host:port" string, mirroring
17
+ RedisConfig.from_env()'s REDIS_CLUSTER_HOST/REDIS_CLUSTER_NODES parsing
18
+ so the CLI and env-var paths accept the same format."""
19
+ nodes = []
20
+ for node in value.split(","):
21
+ node_host, node_port = node.rsplit(":", 1)
22
+ nodes.append((node_host, int(node_port)))
23
+ return nodes
24
+
25
+
15
26
  def parse_args():
16
27
  """Parse command line arguments for the CLI runner."""
17
28
  parser = argparse.ArgumentParser(description="By-Framework CLI Runner")
@@ -25,12 +36,55 @@ def parse_args():
25
36
  ),
26
37
  )
27
38
  parser.add_argument(
28
- "--redis-host", type=str, default="localhost", help="Redis server hostname"
39
+ "--redis-host",
40
+ type=str,
41
+ default=None,
42
+ help="Redis server hostname (default: REDIS_HOST env var, then 'localhost')",
43
+ )
44
+ parser.add_argument(
45
+ "--redis-port",
46
+ type=int,
47
+ default=None,
48
+ help="Redis server port (default: REDIS_PORT env var, then 6379)",
49
+ )
50
+ parser.add_argument(
51
+ "--redis-db",
52
+ type=int,
53
+ default=None,
54
+ help="Redis database number (default: REDIS_DATABASE env var, then 0)",
29
55
  )
30
56
  parser.add_argument(
31
- "--redis-port", type=int, default=6379, help="Redis server port"
57
+ "--redis-password",
58
+ type=str,
59
+ default=None,
60
+ help="Redis password (default: REDIS_PASSWORD env var)",
61
+ )
62
+ parser.add_argument(
63
+ "--redis-username",
64
+ type=str,
65
+ default=None,
66
+ help="Redis username, for ACL-enabled Redis (default: REDIS_USERNAME env var)",
67
+ )
68
+ parser.add_argument(
69
+ "--redis-mode",
70
+ type=str,
71
+ choices=["standalone", "cluster"],
72
+ default=None,
73
+ help=(
74
+ "Redis deployment mode (default: REDIS_MODE env var; implied by "
75
+ "--redis-cluster-nodes/REDIS_CLUSTER_HOST alone if neither is set)"
76
+ ),
77
+ )
78
+ parser.add_argument(
79
+ "--redis-cluster-nodes",
80
+ type=str,
81
+ default=None,
82
+ help=(
83
+ "Comma-separated Redis Cluster seed nodes, 'host:port,host:port' "
84
+ "(default: REDIS_CLUSTER_HOST/REDIS_CLUSTER_NODES env var). "
85
+ "Passing this alone implies --redis-mode cluster."
86
+ ),
32
87
  )
33
- parser.add_argument("--redis-db", type=int, default=0, help="Redis database number")
34
88
  parser.add_argument(
35
89
  "--worker-id", type=str, default="worker-1", help="Unique worker identifier"
36
90
  )
@@ -47,6 +101,17 @@ def parse_args():
47
101
  parser.add_argument(
48
102
  "--redis-max-connections", type=int, help="Max Redis connections allowed"
49
103
  )
104
+ parser.add_argument(
105
+ "--health-port",
106
+ type=int,
107
+ default=None,
108
+ help=(
109
+ "Port for the local /readyz readiness endpoint (Docker/Kubernetes "
110
+ "health checks). Opt-in only - omit to leave it disabled, no "
111
+ "default port is provided. Also settable via BYAI_WORKER_HEALTH_PORT. "
112
+ "See docs/architecture/worker-readiness-endpoint.md."
113
+ ),
114
+ )
50
115
 
51
116
  return parser.parse_args()
52
117
 
@@ -68,17 +133,31 @@ def main():
68
133
 
69
134
  print(f"Starting worker class: {worker_class.__name__} (ID: {args.worker_id})")
70
135
 
71
- # Delegate to the common launcher
136
+ redis_cluster_nodes = (
137
+ _parse_cluster_nodes(args.redis_cluster_nodes)
138
+ if args.redis_cluster_nodes
139
+ else None
140
+ )
141
+
142
+ # Delegate to the common launcher. Every redis_* arg defaults to None
143
+ # here (not a literal like "localhost"), same as run_worker()'s own
144
+ # defaults, so an unset CLI flag actually falls through to the
145
+ # corresponding REDIS_* env var instead of silently shadowing it.
72
146
  run_worker(
73
147
  worker_class=worker_class,
74
148
  worker_id=args.worker_id,
75
149
  redis_host=args.redis_host,
76
150
  redis_port=args.redis_port,
77
151
  redis_db=args.redis_db,
152
+ redis_password=args.redis_password,
153
+ redis_username=args.redis_username,
154
+ redis_mode=args.redis_mode,
155
+ redis_cluster_nodes=redis_cluster_nodes,
78
156
  workspace_dir=args.workspace,
79
157
  max_concurrency=args.max_concurrency,
80
158
  fetch_count=args.fetch_count,
81
159
  redis_max_connections=args.redis_max_connections,
160
+ health_port=args.health_port,
82
161
  )
83
162
 
84
163
 
by_framework/admin/cli.py CHANGED
@@ -50,7 +50,6 @@ app.add_typer(metrics_app, name="metrics", help="Metrics and observability")
50
50
 
51
51
  console = Console()
52
52
  err_console = Console(stderr=True)
53
- _DEFAULT_REDIS_URL = "redis://localhost:6379/0"
54
53
  _redis_url: Optional[str] = None
55
54
 
56
55
 
@@ -85,10 +84,12 @@ def _get_redis(redis_url: Optional[str] = None):
85
84
  configured_url = redis_url if redis_url is not None else _redis_url
86
85
  if configured_url:
87
86
  return init_redis_from_url(configured_url)
88
- config = RedisConfig.from_env()
89
- if config.mode == "cluster":
90
- return init_redis(config=config)
91
- return init_redis_from_url(_DEFAULT_REDIS_URL)
87
+ # Always resolve through RedisConfig.from_env() - for both standalone and
88
+ # cluster mode - so REDIS_HOST/PORT/PASSWORD/USERNAME/DATABASE are honored
89
+ # the same way run_worker()'s config resolution honors them. Previously
90
+ # only cluster mode read env_config; standalone mode silently ignored it
91
+ # and connected to a hardcoded localhost:6379/0.
92
+ return init_redis(config=RedisConfig.from_env())
92
93
 
93
94
 
94
95
  # --------------------------------------------------------------------------- #
@@ -14,6 +14,7 @@ from typing import (TYPE_CHECKING, Any, AsyncIterator, Dict, List, Optional, Pro
14
14
 
15
15
  from by_framework.common.constants import (
16
16
  CANCEL_MESSAGE_ID_PREFIX,
17
+ CLIENT_SOURCE_AGENT_TYPE,
17
18
  EXECUTION_ID_PREFIX,
18
19
  MESSAGE_ID_PREFIX,
19
20
  RedisKeys,
@@ -864,7 +865,7 @@ class GatewayClient:
864
865
  trace_id=trace_id,
865
866
  target_agent_type=params["target_agent_type"],
866
867
  parent_message_id=params["parent_message_id"] or "",
867
- source_agent_type="client",
868
+ source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
868
869
  route_policy=route_policy,
869
870
  route_status=availability.status,
870
871
  stream_name=availability.stream_name or "",
@@ -968,7 +969,7 @@ class GatewayClient:
968
969
  "session_id": params["session_id"],
969
970
  "trace_id": trace_id,
970
971
  "parent_message_id": params["parent_message_id"] or "",
971
- "source_agent_type": "client",
972
+ "source_agent_type": CLIENT_SOURCE_AGENT_TYPE,
972
973
  "target_agent_type": params["target_agent_type"],
973
974
  "stream_name": route.stream_name,
974
975
  "status": "QUEUED",
@@ -1117,7 +1118,7 @@ class GatewayClient:
1117
1118
  message_id=message_id,
1118
1119
  parent_message_id=parent_message_id,
1119
1120
  worker_id=target_worker_id,
1120
- source_agent_type="client",
1121
+ source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
1121
1122
  target_agent_type=target_agent_type,
1122
1123
  route_policy=route_policy,
1123
1124
  route_status=route_status,
@@ -43,6 +43,8 @@ class RedisConfig:
43
43
  """
44
44
  password = os.environ.get("REDIS_PASSWORD", "")
45
45
  username = os.environ.get("REDIS_USERNAME") or None
46
+ host = os.environ.get("REDIS_HOST") or "localhost"
47
+ port_str = os.environ.get("REDIS_PORT")
46
48
  max_connections = os.environ.get("REDIS_MAX_CONNECTIONS")
47
49
  cluster_host_str = os.environ.get("REDIS_CLUSTER_HOST")
48
50
  cluster_nodes_str = cluster_host_str or os.environ.get("REDIS_CLUSTER_NODES")
@@ -61,8 +63,8 @@ class RedisConfig:
61
63
  if db_str is not None:
62
64
  logger.warning("REDIS_DB is deprecated, use REDIS_DATABASE instead")
63
65
  return cls(
64
- host=os.environ.get("REDIS_HOST", "localhost"),
65
- port=int(os.environ.get("REDIS_PORT", "6379")),
66
+ host=host,
67
+ port=int(port_str) if port_str else 6379,
66
68
  db=int(db_str) if db_str is not None else 0,
67
69
  password=password,
68
70
  username=username,
@@ -297,6 +297,100 @@ class RedisKeys:
297
297
  v2_suffix=f"task_group:{{{group_id}}}:results",
298
298
  )
299
299
 
300
+ @classmethod
301
+ def wait_index(cls, shard: int) -> str:
302
+ """ZSET index of suspended callers waiting for a sub-task reply.
303
+
304
+ member = encoded wait-index member (see core/wait_index.py),
305
+ score = deadline in epoch milliseconds. Sharded so sweepers can
306
+ claim disjoint slices without a global lock; the shard is derived
307
+ from session_id (see wait_index_shard()).
308
+
309
+ Cross-entity index (spans every session), so deliberately left
310
+ untagged relative to the per-session keys it points at — same rule
311
+ as trace_index_session/admin_workers.
312
+ """
313
+ return cls._versioned(
314
+ v1_key=f"byai_gateway:wait:index:{shard}",
315
+ v2_suffix=f"wait:index:{shard}",
316
+ )
317
+
318
+ @classmethod
319
+ def wait_sweep_lock(cls, shard: int) -> str:
320
+ """Short-lived claim on one wait_index() shard, held while sweeping it.
321
+
322
+ Ownership is advisory: it only keeps two sweepers from doing the same
323
+ triage at the same moment. Losing it (expiry, a partitioned worker)
324
+ cannot corrupt anything, because every action a sweep takes is
325
+ idempotent — a duplicate synthesized reply is caught by the same
326
+ wait-index gate that catches a duplicate real one. That is why the
327
+ shards need no leader election: a dead worker's claim simply expires
328
+ and another worker picks the shard up on its next cycle.
329
+
330
+ Cross-entity like the shard it guards, so deliberately untagged.
331
+ """
332
+ return cls._versioned(
333
+ v1_key=f"byai_gateway:wait:sweep_lock:{shard}",
334
+ v2_suffix=f"wait:sweep_lock:{shard}",
335
+ )
336
+
337
+ @classmethod
338
+ def wait_consumed(cls, session_id: str, member_digest: str) -> str:
339
+ """Short-lived marker: "this wait-index entry was already resolved".
340
+
341
+ Written by the idempotency gate right after it wins the ZREM for a
342
+ member, and read when a later ZREM for the same member returns 0.
343
+ It is the *only* thing that distinguishes the two meanings of that
344
+ 0 — "someone already consumed this wait" (drop the duplicate) from
345
+ "this wait was never registered" (a pre-upgrade or expired entry,
346
+ which must be let through). Without it, every rolling upgrade would
347
+ silently drop in-flight replies.
348
+
349
+ Per-session entity, so hash-tagged with the session in v2.
350
+ """
351
+ return cls._versioned(
352
+ v1_key=f"byai_gateway:wait:consumed:{session_id}:{member_digest}",
353
+ v2_suffix=f"wait:consumed:{{{session_id}}}:{member_digest}",
354
+ )
355
+
356
+ @classmethod
357
+ def wait_renew_origin(cls, session_id: str, member_digest: str) -> str:
358
+ """The deadline a wait's renewal budget is measured from.
359
+
360
+ Written once (SET NX) by the first sweep that finds the entry due, so
361
+ it holds the wait's *original* deadline even after renewals have
362
+ overwritten the ZSET score. Without it a renewal budget cannot exist
363
+ at all: every sweep would re-measure from the score it just pushed
364
+ out, and a callee whose worker is alive but making no progress would
365
+ be renewed forever.
366
+
367
+ Sweeper-private: nothing on the reply path reads or writes it, so it
368
+ is deliberately NOT part of the wait-index member (which must stay
369
+ rebuildable from a reply alone — see core/wait_index.py). Expiring is
370
+ safe by design: losing it only restarts the budget from the current
371
+ deadline, so the TTL is sized well above any plausible budget.
372
+
373
+ Per-session entity, so hash-tagged with the session in v2.
374
+ """
375
+ return cls._versioned(
376
+ v1_key=f"byai_gateway:wait:renew_origin:{session_id}:{member_digest}",
377
+ v2_suffix=f"wait:renew_origin:{{{session_id}}}:{member_digest}",
378
+ )
379
+
380
+ @classmethod
381
+ def harness_state(cls, execution_id: str) -> str:
382
+ """Serialized in-flight native-agent-harness loop state.
383
+
384
+ Keyed purely by execution_id — the same identity the registry
385
+ already reattaches on RESUME — so any worker instance that picks up
386
+ the eventual ResumeCommand can rehydrate the loop, not only the one
387
+ that started it.
388
+ """
389
+ return cls._versioned(
390
+ v1_key=f"byai_gateway:harness_state:{execution_id}",
391
+ v2_suffix=f"harness_state:{{{execution_id}}}",
392
+ )
393
+
300
394
  # --- Registry ---
301
395
  @classmethod
302
396
  def known_workers(cls) -> str:
@@ -517,8 +611,32 @@ class RedisKeys:
517
611
  MESSAGE_ID_PREFIX = "msg-"
518
612
  EXECUTION_ID_PREFIX = "exec-"
519
613
  TASK_GROUP_ID_PREFIX = "tg-"
614
+ # A single call_agent (non-group) dispatch stores its result in the same
615
+ # task_group_results Hash a real group uses, under a group id derived from
616
+ # the sub-task's own message_id — i.e. a group of size 1. Keeps one result
617
+ # storage/recovery path instead of two.
618
+ TASK_GROUP_SINGLE_ID_PREFIX = "tg-single-"
520
619
  CANCEL_MESSAGE_ID_PREFIX = "msg-cancel-"
521
620
 
621
+ # Sentinel GatewayClient writes as an execution record's source_agent_type for
622
+ # a dispatch it made itself (client/client.py's initialize_execution and
623
+ # record_failed_route_decision). It is NOT an agent type: nothing declares it,
624
+ # so nothing consumes RedisKeys.ctrl_stream(CLIENT_SOURCE_AGENT_TYPE).
625
+ #
626
+ # Load-bearing wherever a resumed execution recovers its caller from its own
627
+ # record instead of from the resume header (GatewayWorker._resolve_reply_command
628
+ # / GatewayProcessor._resolve_reply_header): a root execution's record carries
629
+ # this, and treating it as a caller both posts the result to a stream no one
630
+ # reads and suppresses the end-of-stream event the session data plane owes the
631
+ # user — the visible half of the bug being prevented.
632
+ CLIENT_SOURCE_AGENT_TYPE = "client"
633
+
634
+
635
+ def single_call_task_group_id(child_message_id: str) -> str:
636
+ """Group id under which a single (non-group) call_agent result is stored."""
637
+ return f"{TASK_GROUP_SINGLE_ID_PREFIX}{child_message_id}"
638
+
639
+
522
640
  # --- Redis Hash Field Prefixes ---
523
641
  # Field prefixes in Session Registry Hash
524
642
  EXEC_FIELD_PREFIX = "exec:"
@@ -529,6 +647,10 @@ MSG_MAP_PREFIX = "msg_map:"
529
647
  TASK_GROUP_FIELD_TOTAL = "total"
530
648
  TASK_GROUP_FIELD_COMPLETED = "completed"
531
649
  TASK_GROUP_FIELD_SOURCE_AGENT = "source_agent_type"
650
+ # Set once dispatch fails partway through a batch; any reply that arrives
651
+ # for an aborted group is discarded instead of resuming the (already
652
+ # terminated) caller execution.
653
+ TASK_GROUP_FIELD_ABORTED = "aborted"
532
654
 
533
655
 
534
656
  # --- Timing and Sleep Constants ---
@@ -538,12 +660,133 @@ CONTROL_LOOP_SLEEP_SECONDS = 0.01
538
660
  WAIT_FOR_TASKS_TIMEOUT_SECONDS = 5.0
539
661
  # Task group Key TTL (seconds), default 1 day
540
662
  TASK_GROUP_TTL_SECONDS = 86400
663
+ # Native agent harness loop-state Key TTL (seconds), default 1 day
664
+ HARNESS_STATE_TTL_SECONDS = 86400
541
665
  # First retry wait time (seconds)
542
666
  FIRST_RETRY_WAIT_SECONDS = 1.0
543
667
  # Maximum retry count
544
668
  MAX_RETRY_COUNT = 3
545
669
 
546
670
 
671
+ # --- Suspended-caller liveness (wait index) ---
672
+ # Number of RedisKeys.wait_index() shards. Fixed: changing it re-maps every
673
+ # session to a different shard, so in-flight entries would be swept by no
674
+ # one. Treat as a cross-SDK protocol constant, not a tunable.
675
+ WAIT_INDEX_SHARDS = 16
676
+ # Default deadline for a call_agent(wait_for_reply=True) reply (1 hour).
677
+ # Machine waiting on machine.
678
+ DEFAULT_REPLY_TIMEOUT_MS = 3_600_000
679
+ # Default deadline for an ask_user reply. Machine waiting on a human, so it
680
+ # is deliberately decoupled from DEFAULT_REPLY_TIMEOUT_MS and aligned with
681
+ # the session TTL (which is in seconds) instead.
682
+ DEFAULT_ASK_USER_TIMEOUT_MS = RedisKeys.DEFAULT_SESSION_TTL * 1000
683
+ # How often a worker's sweeper scans the shards it owns (seconds).
684
+ WAIT_SWEEP_INTERVAL_SECONDS = 30
685
+ # TTL of a RedisKeys.wait_sweep_lock() claim. Must comfortably exceed one
686
+ # shard's sweep so the owner doesn't lose the shard mid-pass, and stay short
687
+ # enough that a crashed sweeper's shards are picked up again quickly.
688
+ WAIT_SWEEP_LOCK_TTL_SECONDS = 60
689
+ # Most due entries one sweep resolves per shard per cycle. Bounds the work
690
+ # (and the Redis traffic) of a single pass after an outage leaves a large
691
+ # backlog; the remainder is simply picked up next cycle, since entries stay
692
+ # in the index until a reply clears them.
693
+ WAIT_SWEEP_BATCH_LIMIT = 200
694
+ # Fixed extension applied when a sweep finds the callee still making
695
+ # progress. Deliberately a constant rather than the original timeout: the
696
+ # wait-index member must stay reconstructible from a reply alone, so it
697
+ # cannot carry the caller's original timeout.
698
+ WAIT_RENEW_INCREMENT_MS = 300_000
699
+ # Hard ceiling on renewals, as a multiple of the caller's own timeout: a wait
700
+ # may be renewed until `registered_at + N * timeout`, after which the callee
701
+ # is declared CHILD_TIMEOUT even though its worker is still alive.
702
+ #
703
+ # Without a ceiling the "worker lease alive -> renew" rule is unconditional,
704
+ # so a callee that is running but making no progress (a hung LLM call, a
705
+ # deadlock) suspends its caller forever — the one failure mode the deadline
706
+ # was supposed to bound. N is deliberately expressed against the caller's
707
+ # timeout rather than a renewal count, so a caller that asked for 10 minutes
708
+ # is not held to the same absolute budget as one that asked for four hours,
709
+ # and so retuning WAIT_RENEW_INCREMENT_MS cannot silently change the bound.
710
+ #
711
+ # Why 3: N must exceed 1 (N == 1 is "never renew", which kills every callee
712
+ # that is merely slow); N == 2 leaves a single extra window, so one
713
+ # under-estimated timeout is enough to kill healthy work; N == 3 means the
714
+ # caller's own estimate has to be off by 200% before that happens, while
715
+ # still bounding the default case at 3 hours — two orders of magnitude below
716
+ # DEFAULT_SESSION_TTL, which matters because once the session data expires
717
+ # there is no execution record left to compensate against and the wait is
718
+ # simply dropped. Override per deployment via
719
+ # BY_FRAMEWORK_WAIT_RENEW_MAX_MULTIPLE.
720
+ WAIT_RENEW_MAX_MULTIPLE = 3
721
+ # TTL of RedisKeys.wait_renew_origin(). Must comfortably exceed the largest
722
+ # budget in use (N * timeout), or the budget silently restarts mid-wait.
723
+ WAIT_RENEW_ORIGIN_TTL_SECONDS = TASK_GROUP_TTL_SECONDS
724
+ # How long RedisKeys.wait_consumed() remembers that a wait was already
725
+ # resolved, i.e. how far apart two copies of the same reply may be and still
726
+ # be recognized as duplicates.
727
+ #
728
+ # Sized off DEFAULT_SESSION_TTL, which is the lifetime of the session
729
+ # registry — and the session registry is what keeps a *wait entry* relevant.
730
+ # A marker that expires while entries of that session are still live leaves
731
+ # two holes, and the second is the dangerous one:
732
+ #
733
+ # 1. A repeated ask_user answer (a human may take days; the ask_user
734
+ # deadline is DEFAULT_SESSION_TTL itself) is no longer recognized as a
735
+ # duplicate and wakes the caller a second time.
736
+ # 2. Worse: a stale duplicate sub-agent reply, having lost the marker that
737
+ # would stop it at its own candidate, falls through to the ask_user
738
+ # candidate for the same caller and claims a wait that is still live —
739
+ # after which the real answer is dropped as "already consumed".
740
+ #
741
+ # Both close once the marker outlives every wait it may have to arbitrate,
742
+ # i.e. the session TTL. Erring long costs a handful of idle 1-byte keys with
743
+ # the same lifetime as the session registry they belong to; erring short
744
+ # costs a lost user answer.
745
+ WAIT_CONSUMED_TTL_SECONDS = RedisKeys.DEFAULT_SESSION_TTL
746
+ # How often a sweeper prunes entries that are provably beyond use (see
747
+ # WAIT_PRUNE_AFTER_SECONDS). Deliberately far coarser than
748
+ # WAIT_SWEEP_INTERVAL_SECONDS: this is garbage collection on a multi-day
749
+ # horizon, and it is the only work a sweeper does when compensation is off.
750
+ WAIT_PRUNE_INTERVAL_SECONDS = 3600
751
+ # How far in the past a wait entry's score must lie before pruning it.
752
+ #
753
+ # Every writer of an entry sets its score to its own `now` plus a
754
+ # non-negative offset (registration adds the caller's timeout, a renewal adds
755
+ # WAIT_RENEW_INCREMENT_MS), and only ever does so while the caller's
756
+ # execution record exists. So `now - score > this` proves the entry was last
757
+ # touched more than a session TTL ago, hence that the session registry the
758
+ # sweep would interrogate has expired and no triage is possible any more:
759
+ # the entry can only ever produce "caller missing". Pruning it is therefore
760
+ # not a decision, which is why it needs no opt-in.
761
+ #
762
+ # The margin over DEFAULT_SESSION_TTL is what makes that strict rather than
763
+ # coincident: DEFAULT_ASK_USER_TIMEOUT_MS *equals* the session TTL, so a
764
+ # threshold trimmed to the session TTL exactly would land on the boundary of
765
+ # a live ask_user wait and lose to any clock skew between the worker that
766
+ # registered the entry and the one sweeping it. A day is far beyond plausible
767
+ # skew, and being late costs one ZSET member per unresolved call for one
768
+ # extra day — the asymmetry says err long.
769
+ WAIT_PRUNE_AFTER_SECONDS = RedisKeys.DEFAULT_SESSION_TTL + 86400
770
+
771
+
772
+ class LivenessErrorCode:
773
+ """error_code values carried by synthesized/recovered resume replies.
774
+
775
+ Cross-SDK wire contract — Python/TS/Java must emit the same strings;
776
+ callers match on them. Append only, never rename.
777
+ """
778
+
779
+ # The callee's worker lease expired while its execution was non-terminal.
780
+ CHILD_WORKER_LOST = "CHILD_WORKER_LOST"
781
+ # The callee was alive but produced no reply before the deadline.
782
+ CHILD_TIMEOUT = "CHILD_TIMEOUT"
783
+ # The dispatch was never picked up by any worker.
784
+ CHILD_NEVER_STARTED = "CHILD_NEVER_STARTED"
785
+ # The callee finished and its result was persisted, but the reply
786
+ # message was lost; the result was recovered from storage.
787
+ REPLY_LOST_RECOVERED = "REPLY_LOST_RECOVERED"
788
+
789
+
547
790
  # --- Filesystem Constants ---
548
791
  DEFAULT_WORKSPACE_DIR = "/workspace"
549
792
 
@@ -21,6 +21,10 @@ class EventType(str, Enum):
21
21
  TASK_CREATE: Task creation event
22
22
  STEP_COMPLETE: Step completion event
23
23
  TASK_STOP: Task stop event
24
+ ORPHANED_REPLY: A reply that arrived for an already-resolved wait and
25
+ was therefore dropped by the idempotency gate. Diagnostic only —
26
+ the sub-agent did real work whose result nobody will consume, so
27
+ it must not vanish silently.
24
28
  """
25
29
 
26
30
  ANSWER_DELTA = "answerDelta"
@@ -32,3 +36,4 @@ class EventType(str, Enum):
32
36
  TASK_CREATE = "taskCreate"
33
37
  STEP_COMPLETE = "stepComplete"
34
38
  TASK_STOP = "taskStop"
39
+ ORPHANED_REPLY = "orphanedReply"
@@ -239,6 +239,33 @@ async def check_worker_online(
239
239
  return is_legacy or last_seen > 0
240
240
 
241
241
 
242
+ async def acquire_scoped_lock(
243
+ redis: Redis,
244
+ key: str,
245
+ token: str,
246
+ ttl_seconds: int,
247
+ ) -> bool:
248
+ """Claim `key` for `token` if nobody holds it (Redlock acquire half).
249
+
250
+ The stored value must stay a cjson-decodable object carrying a "token"
251
+ field: `_REFRESH_LOCK_SCRIPT` / `_RELEASE_LOCK_SCRIPT` parse it that way,
252
+ and a bare token string would decode as unparseable legacy data, making
253
+ the holder unable to release its own lock.
254
+ """
255
+ stored = await redis.set(
256
+ key,
257
+ json.dumps({"token": token}, separators=(",", ":")),
258
+ nx=True,
259
+ ex=ttl_seconds,
260
+ )
261
+ return bool(stored)
262
+
263
+
264
+ async def release_scoped_lock(redis: Redis, key: str, token: str) -> bool:
265
+ """Release a lock taken with acquire_scoped_lock(), if still owned."""
266
+ return bool(await redis.eval(_RELEASE_LOCK_SCRIPT, 1, key, token or ""))
267
+
268
+
242
269
  async def check_agent_type_online(
243
270
  redis: Redis,
244
271
  agent_type: str,
@@ -1225,6 +1252,10 @@ class WorkerRegistry:
1225
1252
  "active": self._get_int_hash_value(raw_counts, "active_count"),
1226
1253
  "queued": self._get_int_hash_value(raw_counts, "queued_count"),
1227
1254
  "running": self._get_int_hash_value(raw_counts, "running_count"),
1255
+ "waiting_agent": self._get_int_hash_value(
1256
+ raw_counts, "waiting_agent_count"
1257
+ ),
1258
+ "waiting_user": self._get_int_hash_value(raw_counts, "waiting_user_count"),
1228
1259
  "cancelling": self._get_int_hash_value(raw_counts, "cancelling_count"),
1229
1260
  "completed": self._get_int_hash_value(raw_counts, "completed_count"),
1230
1261
  "failed": self._get_int_hash_value(raw_counts, "failed_count"),
@@ -1245,6 +1276,10 @@ class WorkerRegistry:
1245
1276
  status_counts = {
1246
1277
  "QUEUED": counts["queued"],
1247
1278
  "RUNNING": counts["running"],
1279
+ # Suspended callers persist as WAITING_* rather than QUEUED; these
1280
+ # rows keep them visible (zero-valued entries are filtered below).
1281
+ "WAITING_AGENT": counts["waiting_agent"],
1282
+ "WAITING_USER": counts["waiting_user"],
1248
1283
  "CANCELLING": counts["cancelling"],
1249
1284
  "COMPLETED": counts["completed"],
1250
1285
  "FAILED": counts["failed"],