by-framework 0.2.2.dev11__py3-none-any.whl → 0.2.2.dev13__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -14,6 +14,7 @@ from typing import (TYPE_CHECKING, Any, AsyncIterator, Dict, List, Optional, Pro
14
14
 
15
15
  from by_framework.common.constants import (
16
16
  CANCEL_MESSAGE_ID_PREFIX,
17
+ CLIENT_SOURCE_AGENT_TYPE,
17
18
  EXECUTION_ID_PREFIX,
18
19
  MESSAGE_ID_PREFIX,
19
20
  RedisKeys,
@@ -864,7 +865,7 @@ class GatewayClient:
864
865
  trace_id=trace_id,
865
866
  target_agent_type=params["target_agent_type"],
866
867
  parent_message_id=params["parent_message_id"] or "",
867
- source_agent_type="client",
868
+ source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
868
869
  route_policy=route_policy,
869
870
  route_status=availability.status,
870
871
  stream_name=availability.stream_name or "",
@@ -968,7 +969,7 @@ class GatewayClient:
968
969
  "session_id": params["session_id"],
969
970
  "trace_id": trace_id,
970
971
  "parent_message_id": params["parent_message_id"] or "",
971
- "source_agent_type": "client",
972
+ "source_agent_type": CLIENT_SOURCE_AGENT_TYPE,
972
973
  "target_agent_type": params["target_agent_type"],
973
974
  "stream_name": route.stream_name,
974
975
  "status": "QUEUED",
@@ -1117,7 +1118,7 @@ class GatewayClient:
1117
1118
  message_id=message_id,
1118
1119
  parent_message_id=parent_message_id,
1119
1120
  worker_id=target_worker_id,
1120
- source_agent_type="client",
1121
+ source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
1121
1122
  target_agent_type=target_agent_type,
1122
1123
  route_policy=route_policy,
1123
1124
  route_status=route_status,
@@ -297,6 +297,100 @@ class RedisKeys:
297
297
  v2_suffix=f"task_group:{{{group_id}}}:results",
298
298
  )
299
299
 
300
+ @classmethod
301
+ def wait_index(cls, shard: int) -> str:
302
+ """ZSET index of suspended callers waiting for a sub-task reply.
303
+
304
+ member = encoded wait-index member (see core/wait_index.py),
305
+ score = deadline in epoch milliseconds. Sharded so sweepers can
306
+ claim disjoint slices without a global lock; the shard is derived
307
+ from session_id (see wait_index_shard()).
308
+
309
+ Cross-entity index (spans every session), so deliberately left
310
+ untagged relative to the per-session keys it points at — same rule
311
+ as trace_index_session/admin_workers.
312
+ """
313
+ return cls._versioned(
314
+ v1_key=f"byai_gateway:wait:index:{shard}",
315
+ v2_suffix=f"wait:index:{shard}",
316
+ )
317
+
318
+ @classmethod
319
+ def wait_sweep_lock(cls, shard: int) -> str:
320
+ """Short-lived claim on one wait_index() shard, held while sweeping it.
321
+
322
+ Ownership is advisory: it only keeps two sweepers from doing the same
323
+ triage at the same moment. Losing it (expiry, a partitioned worker)
324
+ cannot corrupt anything, because every action a sweep takes is
325
+ idempotent — a duplicate synthesized reply is caught by the same
326
+ wait-index gate that catches a duplicate real one. That is why the
327
+ shards need no leader election: a dead worker's claim simply expires
328
+ and another worker picks the shard up on its next cycle.
329
+
330
+ Cross-entity like the shard it guards, so deliberately untagged.
331
+ """
332
+ return cls._versioned(
333
+ v1_key=f"byai_gateway:wait:sweep_lock:{shard}",
334
+ v2_suffix=f"wait:sweep_lock:{shard}",
335
+ )
336
+
337
+ @classmethod
338
+ def wait_consumed(cls, session_id: str, member_digest: str) -> str:
339
+ """Short-lived marker: "this wait-index entry was already resolved".
340
+
341
+ Written by the idempotency gate right after it wins the ZREM for a
342
+ member, and read when a later ZREM for the same member returns 0.
343
+ It is the *only* thing that distinguishes the two meanings of that
344
+ 0 — "someone already consumed this wait" (drop the duplicate) from
345
+ "this wait was never registered" (a pre-upgrade or expired entry,
346
+ which must be let through). Without it, every rolling upgrade would
347
+ silently drop in-flight replies.
348
+
349
+ Per-session entity, so hash-tagged with the session in v2.
350
+ """
351
+ return cls._versioned(
352
+ v1_key=f"byai_gateway:wait:consumed:{session_id}:{member_digest}",
353
+ v2_suffix=f"wait:consumed:{{{session_id}}}:{member_digest}",
354
+ )
355
+
356
+ @classmethod
357
+ def wait_renew_origin(cls, session_id: str, member_digest: str) -> str:
358
+ """The deadline a wait's renewal budget is measured from.
359
+
360
+ Written once (SET NX) by the first sweep that finds the entry due, so
361
+ it holds the wait's *original* deadline even after renewals have
362
+ overwritten the ZSET score. Without it a renewal budget cannot exist
363
+ at all: every sweep would re-measure from the score it just pushed
364
+ out, and a callee whose worker is alive but making no progress would
365
+ be renewed forever.
366
+
367
+ Sweeper-private: nothing on the reply path reads or writes it, so it
368
+ is deliberately NOT part of the wait-index member (which must stay
369
+ rebuildable from a reply alone — see core/wait_index.py). Expiring is
370
+ safe by design: losing it only restarts the budget from the current
371
+ deadline, so the TTL is sized well above any plausible budget.
372
+
373
+ Per-session entity, so hash-tagged with the session in v2.
374
+ """
375
+ return cls._versioned(
376
+ v1_key=f"byai_gateway:wait:renew_origin:{session_id}:{member_digest}",
377
+ v2_suffix=f"wait:renew_origin:{{{session_id}}}:{member_digest}",
378
+ )
379
+
380
+ @classmethod
381
+ def harness_state(cls, execution_id: str) -> str:
382
+ """Serialized in-flight native-agent-harness loop state.
383
+
384
+ Keyed purely by execution_id — the same identity the registry
385
+ already reattaches on RESUME — so any worker instance that picks up
386
+ the eventual ResumeCommand can rehydrate the loop, not only the one
387
+ that started it.
388
+ """
389
+ return cls._versioned(
390
+ v1_key=f"byai_gateway:harness_state:{execution_id}",
391
+ v2_suffix=f"harness_state:{{{execution_id}}}",
392
+ )
393
+
300
394
  # --- Registry ---
301
395
  @classmethod
302
396
  def known_workers(cls) -> str:
@@ -517,8 +611,32 @@ class RedisKeys:
517
611
  MESSAGE_ID_PREFIX = "msg-"
518
612
  EXECUTION_ID_PREFIX = "exec-"
519
613
  TASK_GROUP_ID_PREFIX = "tg-"
614
+ # A single call_agent (non-group) dispatch stores its result in the same
615
+ # task_group_results Hash a real group uses, under a group id derived from
616
+ # the sub-task's own message_id — i.e. a group of size 1. Keeps one result
617
+ # storage/recovery path instead of two.
618
+ TASK_GROUP_SINGLE_ID_PREFIX = "tg-single-"
520
619
  CANCEL_MESSAGE_ID_PREFIX = "msg-cancel-"
521
620
 
621
+ # Sentinel GatewayClient writes as an execution record's source_agent_type for
622
+ # a dispatch it made itself (client/client.py's initialize_execution and
623
+ # record_failed_route_decision). It is NOT an agent type: nothing declares it,
624
+ # so nothing consumes RedisKeys.ctrl_stream(CLIENT_SOURCE_AGENT_TYPE).
625
+ #
626
+ # Load-bearing wherever a resumed execution recovers its caller from its own
627
+ # record instead of from the resume header (GatewayWorker._resolve_reply_command
628
+ # / GatewayProcessor._resolve_reply_header): a root execution's record carries
629
+ # this, and treating it as a caller both posts the result to a stream no one
630
+ # reads and suppresses the end-of-stream event the session data plane owes the
631
+ # user — the visible half of the bug being prevented.
632
+ CLIENT_SOURCE_AGENT_TYPE = "client"
633
+
634
+
635
+ def single_call_task_group_id(child_message_id: str) -> str:
636
+ """Group id under which a single (non-group) call_agent result is stored."""
637
+ return f"{TASK_GROUP_SINGLE_ID_PREFIX}{child_message_id}"
638
+
639
+
522
640
  # --- Redis Hash Field Prefixes ---
523
641
  # Field prefixes in Session Registry Hash
524
642
  EXEC_FIELD_PREFIX = "exec:"
@@ -529,6 +647,10 @@ MSG_MAP_PREFIX = "msg_map:"
529
647
  TASK_GROUP_FIELD_TOTAL = "total"
530
648
  TASK_GROUP_FIELD_COMPLETED = "completed"
531
649
  TASK_GROUP_FIELD_SOURCE_AGENT = "source_agent_type"
650
+ # Set once dispatch fails partway through a batch; any reply that arrives
651
+ # for an aborted group is discarded instead of resuming the (already
652
+ # terminated) caller execution.
653
+ TASK_GROUP_FIELD_ABORTED = "aborted"
532
654
 
533
655
 
534
656
  # --- Timing and Sleep Constants ---
@@ -538,12 +660,133 @@ CONTROL_LOOP_SLEEP_SECONDS = 0.01
538
660
  WAIT_FOR_TASKS_TIMEOUT_SECONDS = 5.0
539
661
  # Task group Key TTL (seconds), default 1 day
540
662
  TASK_GROUP_TTL_SECONDS = 86400
663
+ # Native agent harness loop-state Key TTL (seconds), default 1 day
664
+ HARNESS_STATE_TTL_SECONDS = 86400
541
665
  # First retry wait time (seconds)
542
666
  FIRST_RETRY_WAIT_SECONDS = 1.0
543
667
  # Maximum retry count
544
668
  MAX_RETRY_COUNT = 3
545
669
 
546
670
 
671
+ # --- Suspended-caller liveness (wait index) ---
672
+ # Number of RedisKeys.wait_index() shards. Fixed: changing it re-maps every
673
+ # session to a different shard, so in-flight entries would be swept by no
674
+ # one. Treat as a cross-SDK protocol constant, not a tunable.
675
+ WAIT_INDEX_SHARDS = 16
676
+ # Default deadline for a call_agent(wait_for_reply=True) reply (1 hour).
677
+ # Machine waiting on machine.
678
+ DEFAULT_REPLY_TIMEOUT_MS = 3_600_000
679
+ # Default deadline for an ask_user reply. Machine waiting on a human, so it
680
+ # is deliberately decoupled from DEFAULT_REPLY_TIMEOUT_MS and aligned with
681
+ # the session TTL (which is in seconds) instead.
682
+ DEFAULT_ASK_USER_TIMEOUT_MS = RedisKeys.DEFAULT_SESSION_TTL * 1000
683
+ # How often a worker's sweeper scans the shards it owns (seconds).
684
+ WAIT_SWEEP_INTERVAL_SECONDS = 30
685
+ # TTL of a RedisKeys.wait_sweep_lock() claim. Must comfortably exceed one
686
+ # shard's sweep so the owner doesn't lose the shard mid-pass, and stay short
687
+ # enough that a crashed sweeper's shards are picked up again quickly.
688
+ WAIT_SWEEP_LOCK_TTL_SECONDS = 60
689
+ # Most due entries one sweep resolves per shard per cycle. Bounds the work
690
+ # (and the Redis traffic) of a single pass after an outage leaves a large
691
+ # backlog; the remainder is simply picked up next cycle, since entries stay
692
+ # in the index until a reply clears them.
693
+ WAIT_SWEEP_BATCH_LIMIT = 200
694
+ # Fixed extension applied when a sweep finds the callee still making
695
+ # progress. Deliberately a constant rather than the original timeout: the
696
+ # wait-index member must stay reconstructible from a reply alone, so it
697
+ # cannot carry the caller's original timeout.
698
+ WAIT_RENEW_INCREMENT_MS = 300_000
699
+ # Hard ceiling on renewals, as a multiple of the caller's own timeout: a wait
700
+ # may be renewed until `registered_at + N * timeout`, after which the callee
701
+ # is declared CHILD_TIMEOUT even though its worker is still alive.
702
+ #
703
+ # Without a ceiling the "worker lease alive -> renew" rule is unconditional,
704
+ # so a callee that is running but making no progress (a hung LLM call, a
705
+ # deadlock) suspends its caller forever — the one failure mode the deadline
706
+ # was supposed to bound. N is deliberately expressed against the caller's
707
+ # timeout rather than a renewal count, so a caller that asked for 10 minutes
708
+ # is not held to the same absolute budget as one that asked for four hours,
709
+ # and so retuning WAIT_RENEW_INCREMENT_MS cannot silently change the bound.
710
+ #
711
+ # Why 3: N must exceed 1 (N == 1 is "never renew", which kills every callee
712
+ # that is merely slow); N == 2 leaves a single extra window, so one
713
+ # under-estimated timeout is enough to kill healthy work; N == 3 means the
714
+ # caller's own estimate has to be off by 200% before that happens, while
715
+ # still bounding the default case at 3 hours — two orders of magnitude below
716
+ # DEFAULT_SESSION_TTL, which matters because once the session data expires
717
+ # there is no execution record left to compensate against and the wait is
718
+ # simply dropped. Override per deployment via
719
+ # BY_FRAMEWORK_WAIT_RENEW_MAX_MULTIPLE.
720
+ WAIT_RENEW_MAX_MULTIPLE = 3
721
+ # TTL of RedisKeys.wait_renew_origin(). Must comfortably exceed the largest
722
+ # budget in use (N * timeout), or the budget silently restarts mid-wait.
723
+ WAIT_RENEW_ORIGIN_TTL_SECONDS = TASK_GROUP_TTL_SECONDS
724
+ # How long RedisKeys.wait_consumed() remembers that a wait was already
725
+ # resolved, i.e. how far apart two copies of the same reply may be and still
726
+ # be recognized as duplicates.
727
+ #
728
+ # Sized off DEFAULT_SESSION_TTL, which is the lifetime of the session
729
+ # registry — and the session registry is what keeps a *wait entry* relevant.
730
+ # A marker that expires while entries of that session are still live leaves
731
+ # two holes, and the second is the dangerous one:
732
+ #
733
+ # 1. A repeated ask_user answer (a human may take days; the ask_user
734
+ # deadline is DEFAULT_SESSION_TTL itself) is no longer recognized as a
735
+ # duplicate and wakes the caller a second time.
736
+ # 2. Worse: a stale duplicate sub-agent reply, having lost the marker that
737
+ # would stop it at its own candidate, falls through to the ask_user
738
+ # candidate for the same caller and claims a wait that is still live —
739
+ # after which the real answer is dropped as "already consumed".
740
+ #
741
+ # Both close once the marker outlives every wait it may have to arbitrate,
742
+ # i.e. the session TTL. Erring long costs a handful of idle 1-byte keys with
743
+ # the same lifetime as the session registry they belong to; erring short
744
+ # costs a lost user answer.
745
+ WAIT_CONSUMED_TTL_SECONDS = RedisKeys.DEFAULT_SESSION_TTL
746
+ # How often a sweeper prunes entries that are provably beyond use (see
747
+ # WAIT_PRUNE_AFTER_SECONDS). Deliberately far coarser than
748
+ # WAIT_SWEEP_INTERVAL_SECONDS: this is garbage collection on a multi-day
749
+ # horizon, and it is the only work a sweeper does when compensation is off.
750
+ WAIT_PRUNE_INTERVAL_SECONDS = 3600
751
+ # How far in the past a wait entry's score must lie before pruning it.
752
+ #
753
+ # Every writer of an entry sets its score to its own `now` plus a
754
+ # non-negative offset (registration adds the caller's timeout, a renewal adds
755
+ # WAIT_RENEW_INCREMENT_MS), and only ever does so while the caller's
756
+ # execution record exists. So `now - score > this` proves the entry was last
757
+ # touched more than a session TTL ago, hence that the session registry the
758
+ # sweep would interrogate has expired and no triage is possible any more:
759
+ # the entry can only ever produce "caller missing". Pruning it is therefore
760
+ # not a decision, which is why it needs no opt-in.
761
+ #
762
+ # The margin over DEFAULT_SESSION_TTL is what makes that strict rather than
763
+ # coincident: DEFAULT_ASK_USER_TIMEOUT_MS *equals* the session TTL, so a
764
+ # threshold trimmed to the session TTL exactly would land on the boundary of
765
+ # a live ask_user wait and lose to any clock skew between the worker that
766
+ # registered the entry and the one sweeping it. A day is far beyond plausible
767
+ # skew, and being late costs one ZSET member per unresolved call for one
768
+ # extra day — the asymmetry says err long.
769
+ WAIT_PRUNE_AFTER_SECONDS = RedisKeys.DEFAULT_SESSION_TTL + 86400
770
+
771
+
772
+ class LivenessErrorCode:
773
+ """error_code values carried by synthesized/recovered resume replies.
774
+
775
+ Cross-SDK wire contract — Python/TS/Java must emit the same strings;
776
+ callers match on them. Append only, never rename.
777
+ """
778
+
779
+ # The callee's worker lease expired while its execution was non-terminal.
780
+ CHILD_WORKER_LOST = "CHILD_WORKER_LOST"
781
+ # The callee was alive but produced no reply before the deadline.
782
+ CHILD_TIMEOUT = "CHILD_TIMEOUT"
783
+ # The dispatch was never picked up by any worker.
784
+ CHILD_NEVER_STARTED = "CHILD_NEVER_STARTED"
785
+ # The callee finished and its result was persisted, but the reply
786
+ # message was lost; the result was recovered from storage.
787
+ REPLY_LOST_RECOVERED = "REPLY_LOST_RECOVERED"
788
+
789
+
547
790
  # --- Filesystem Constants ---
548
791
  DEFAULT_WORKSPACE_DIR = "/workspace"
549
792
 
@@ -21,6 +21,10 @@ class EventType(str, Enum):
21
21
  TASK_CREATE: Task creation event
22
22
  STEP_COMPLETE: Step completion event
23
23
  TASK_STOP: Task stop event
24
+ ORPHANED_REPLY: A reply that arrived for an already-resolved wait and
25
+ was therefore dropped by the idempotency gate. Diagnostic only —
26
+ the sub-agent did real work whose result nobody will consume, so
27
+ it must not vanish silently.
24
28
  """
25
29
 
26
30
  ANSWER_DELTA = "answerDelta"
@@ -32,3 +36,4 @@ class EventType(str, Enum):
32
36
  TASK_CREATE = "taskCreate"
33
37
  STEP_COMPLETE = "stepComplete"
34
38
  TASK_STOP = "taskStop"
39
+ ORPHANED_REPLY = "orphanedReply"
@@ -239,6 +239,33 @@ async def check_worker_online(
239
239
  return is_legacy or last_seen > 0
240
240
 
241
241
 
242
+ async def acquire_scoped_lock(
243
+ redis: Redis,
244
+ key: str,
245
+ token: str,
246
+ ttl_seconds: int,
247
+ ) -> bool:
248
+ """Claim `key` for `token` if nobody holds it (Redlock acquire half).
249
+
250
+ The stored value must stay a cjson-decodable object carrying a "token"
251
+ field: `_REFRESH_LOCK_SCRIPT` / `_RELEASE_LOCK_SCRIPT` parse it that way,
252
+ and a bare token string would decode as unparseable legacy data, making
253
+ the holder unable to release its own lock.
254
+ """
255
+ stored = await redis.set(
256
+ key,
257
+ json.dumps({"token": token}, separators=(",", ":")),
258
+ nx=True,
259
+ ex=ttl_seconds,
260
+ )
261
+ return bool(stored)
262
+
263
+
264
+ async def release_scoped_lock(redis: Redis, key: str, token: str) -> bool:
265
+ """Release a lock taken with acquire_scoped_lock(), if still owned."""
266
+ return bool(await redis.eval(_RELEASE_LOCK_SCRIPT, 1, key, token or ""))
267
+
268
+
242
269
  async def check_agent_type_online(
243
270
  redis: Redis,
244
271
  agent_type: str,
@@ -1225,6 +1252,10 @@ class WorkerRegistry:
1225
1252
  "active": self._get_int_hash_value(raw_counts, "active_count"),
1226
1253
  "queued": self._get_int_hash_value(raw_counts, "queued_count"),
1227
1254
  "running": self._get_int_hash_value(raw_counts, "running_count"),
1255
+ "waiting_agent": self._get_int_hash_value(
1256
+ raw_counts, "waiting_agent_count"
1257
+ ),
1258
+ "waiting_user": self._get_int_hash_value(raw_counts, "waiting_user_count"),
1228
1259
  "cancelling": self._get_int_hash_value(raw_counts, "cancelling_count"),
1229
1260
  "completed": self._get_int_hash_value(raw_counts, "completed_count"),
1230
1261
  "failed": self._get_int_hash_value(raw_counts, "failed_count"),
@@ -1245,6 +1276,10 @@ class WorkerRegistry:
1245
1276
  status_counts = {
1246
1277
  "QUEUED": counts["queued"],
1247
1278
  "RUNNING": counts["running"],
1279
+ # Suspended callers persist as WAITING_* rather than QUEUED; these
1280
+ # rows keep them visible (zero-valued entries are filtered below).
1281
+ "WAITING_AGENT": counts["waiting_agent"],
1282
+ "WAITING_USER": counts["waiting_user"],
1248
1283
  "CANCELLING": counts["cancelling"],
1249
1284
  "COMPLETED": counts["completed"],
1250
1285
  "FAILED": counts["failed"],
@@ -0,0 +1,209 @@
1
+ """Idempotency gate for replies that resume a suspended caller.
2
+
3
+ A suspended caller is woken by exactly one ``ResumeCommand``. Once a sweep
4
+ can *synthesize* that reply (a callee whose worker died will never send
5
+ one), two copies can exist for the same wait: the synthesized one and the
6
+ real one that shows up late. Waking the caller twice re-runs a finished
7
+ execution and, in a Task Group, pushes ``completed`` past ``total`` and
8
+ aggregates a second time.
9
+
10
+ The gate is the single place that decides which copy wins. It runs at the
11
+ one point every reply passes through — right after a reply is parsed,
12
+ before any registry lookup and before Task Group join accounting — and
13
+ claims the caller's wait-index entry with a ``ZREM``. Exactly one claimant
14
+ can win, because ``ZREM`` is atomic.
15
+
16
+ The hard part is what ``ZREM`` returning 0 means, since it conflates two
17
+ opposite situations:
18
+
19
+ * the entry existed and someone else already claimed it — a true duplicate,
20
+ drop it;
21
+ * the entry never existed — the reply belongs to a dispatch made before
22
+ this version shipped, or to a wait whose entry expired. Dropping it
23
+ silently loses a real reply, and during a rolling upgrade *every*
24
+ in-flight reply looks like this.
25
+
26
+ A short-lived "consumed" marker written by the winner separates them: a 0
27
+ with a marker is a duplicate, a 0 without one is unregistered and must be
28
+ let through. When in doubt the gate lets the message through — a spurious
29
+ extra wake-up is recoverable, a dropped reply is permanent silence. The
30
+ same rule makes the gate fail *open*: any Redis error here allows the
31
+ message.
32
+ """
33
+
34
+ from typing import Any, NamedTuple
35
+
36
+ from by_framework.common.constants import WAIT_CONSUMED_TTL_SECONDS, RedisKeys
37
+ from by_framework.common.logger import logger
38
+ from by_framework.core.protocol.commands import GatewayCommand
39
+ from by_framework.core.wait_index import (
40
+ encode_member,
41
+ member_digest,
42
+ member_from_resume,
43
+ wait_index_key,
44
+ )
45
+
46
+ # Why a reply was allowed through / dropped. Carried on the decision so the
47
+ # caller can log it and put it on the orphaned_reply event.
48
+ ALLOW_CLAIMED = "claimed"
49
+ ALLOW_UNREGISTERED = "unregistered"
50
+ ALLOW_GATE_ERROR = "gate_error"
51
+ DENY_ALREADY_CONSUMED = "already_consumed"
52
+
53
+
54
+ class WaitGateDecision(NamedTuple):
55
+ """Outcome of the gate.
56
+
57
+ Attributes:
58
+ allow: Whether the reply may be processed. False only when the wait
59
+ it targets is provably already resolved.
60
+ reason: One of the ALLOW_*/DENY_* constants above.
61
+ member: The wait-index member the decision was made about ("" when
62
+ no candidate matched).
63
+ """
64
+
65
+ allow: bool
66
+ reason: str
67
+ member: str = ""
68
+
69
+
70
+ def consumed_marker_key(session_id: str, member: str) -> str:
71
+ """Redis key of the "already consumed" marker for one wait-index member.
72
+
73
+ Part of the cross-SDK contract, since any SDK's worker may gate another
74
+ SDK's reply — see ``wait_index.member_digest`` for why the member is
75
+ hashed rather than embedded.
76
+ """
77
+ return RedisKeys.wait_consumed(session_id, member_digest(member))
78
+
79
+
80
+ def candidate_members(command: GatewayCommand) -> list[str]:
81
+ """Wait-index members this reply could be clearing, most-specific first.
82
+
83
+ Normally there is exactly one: the member rebuilt from the reply's own
84
+ header. The second candidate covers ``ask_user``, which registers with
85
+ an empty ``child_message_id`` because it has no sub-task — while the
86
+ matching reply comes from a client that is free to put anything in
87
+ ``header.parent_message_id`` (existing callers put the caller's own
88
+ parent there, not an empty string), so it cannot be rebuilt exactly.
89
+
90
+ Order matters, and so does the fact that each candidate is fully
91
+ resolved (claim, then check its marker) before the next is tried: a
92
+ duplicate sub-agent reply must be caught by *its own* marker rather than
93
+ fall through and clear a live ask_user wait that happens to belong to
94
+ the same caller.
95
+
96
+ A reply carrying a ``task_group_id`` is a sub-agent reply by
97
+ construction, so the ask_user variant is not even considered for it.
98
+ """
99
+ header = command.header
100
+ members = [member_from_resume(command)]
101
+ if not header.task_group_id:
102
+ ask_user_member = encode_member(
103
+ session_id=header.session_id,
104
+ parent_message_id=header.message_id,
105
+ child_message_id="",
106
+ task_group_id="",
107
+ )
108
+ if ask_user_member not in members:
109
+ members.append(ask_user_member)
110
+ return members
111
+
112
+
113
+ async def _mark_consumed(redis: Any, session_id: str, member: str) -> None:
114
+ """Record that this wait was resolved, so a late twin can be recognized.
115
+
116
+ Fail-soft: losing the marker only means a much later duplicate would be
117
+ allowed through (one extra wake-up), which is the direction this whole
118
+ module errs in anyway.
119
+ """
120
+ try:
121
+ await redis.set(
122
+ consumed_marker_key(session_id, member),
123
+ "1",
124
+ ex=WAIT_CONSUMED_TTL_SECONDS,
125
+ )
126
+ except Exception as error: # pylint: disable=broad-exception-caught
127
+ logger.warning(
128
+ "Failed to mark wait entry consumed (session=%s): %s", session_id, error
129
+ )
130
+
131
+
132
+ async def consume_wait_entry(redis: Any, command: GatewayCommand) -> WaitGateDecision:
133
+ """Claim the wait a reply resolves; report whether it may be processed.
134
+
135
+ Call once per ``ResumeCommand``, before the execution lookup and before
136
+ Task Group join accounting.
137
+ """
138
+ session_id = command.header.session_id
139
+ try:
140
+ index_key = wait_index_key(session_id)
141
+ for member in candidate_members(command):
142
+ removed = int(await redis.zrem(index_key, member) or 0)
143
+ if removed > 0:
144
+ await _mark_consumed(redis, session_id, member)
145
+ return WaitGateDecision(True, ALLOW_CLAIMED, member)
146
+ if await redis.exists(consumed_marker_key(session_id, member)):
147
+ return WaitGateDecision(False, DENY_ALREADY_CONSUMED, member)
148
+ # No entry, no marker: nobody ever registered this wait (a dispatch
149
+ # from before this version, or an entry that outlived its index).
150
+ # Unknown is not the same as duplicate — let it through.
151
+ return WaitGateDecision(True, ALLOW_UNREGISTERED)
152
+ except Exception as error: # pylint: disable=broad-exception-caught
153
+ # Fail open. A gate that drops messages when Redis hiccups is worse
154
+ # than the duplicate it was built to prevent.
155
+ logger.warning(
156
+ "Wait-index gate unavailable for session=%s, allowing reply: %s",
157
+ session_id,
158
+ error,
159
+ )
160
+ return WaitGateDecision(True, ALLOW_GATE_ERROR)
161
+
162
+
163
+ async def emit_orphaned_reply(
164
+ redis: Any,
165
+ command: GatewayCommand,
166
+ *,
167
+ worker_id: str = "",
168
+ reason: str = DENY_ALREADY_CONSUMED,
169
+ ) -> None:
170
+ """Announce on the session data stream that a reply was dropped.
171
+
172
+ A dropped reply is not noise: the sub-agent ran, produced a result, and
173
+ may have had side effects that nobody will now account for. Emitting it
174
+ on the existing data plane keeps it visible without inventing a second
175
+ reporting mechanism.
176
+
177
+ Fail-soft by construction — the drop/allow decision has already been
178
+ made, and reporting it must never change or block it.
179
+ """
180
+ header = command.header
181
+ try:
182
+ from by_framework.common.emitter import GatewayDataEmitter
183
+ from by_framework.core.protocol.event_type import EventType
184
+
185
+ await GatewayDataEmitter(redis_client=redis).emit_event(
186
+ session_id=header.session_id,
187
+ trace_id=header.trace_id,
188
+ event_type=EventType.ORPHANED_REPLY.value,
189
+ source_agent_type=header.source_agent_type,
190
+ message_id=header.message_id,
191
+ parent_message_id=header.parent_message_id,
192
+ data={
193
+ "reason": reason,
194
+ # The suspended caller this reply was addressed to...
195
+ "caller_message_id": header.message_id,
196
+ # ...and the sub-task that produced it (empty for ask_user).
197
+ "child_message_id": header.parent_message_id,
198
+ "task_group_id": header.task_group_id,
199
+ "status": str(getattr(command, "status", "") or ""),
200
+ "worker_id": worker_id,
201
+ },
202
+ )
203
+ except Exception as error: # pylint: disable=broad-exception-caught
204
+ logger.warning(
205
+ "Failed to emit orphaned_reply event (session=%s, message_id=%s): %s",
206
+ header.session_id,
207
+ header.message_id,
208
+ error,
209
+ )