by-framework 0.2.2.dev11__py3-none-any.whl → 0.2.2.dev13__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- by_framework/client/client.py +4 -3
- by_framework/common/constants.py +243 -0
- by_framework/core/protocol/event_type.py +5 -0
- by_framework/core/registry.py +35 -0
- by_framework/core/wait_gate.py +209 -0
- by_framework/core/wait_index.py +165 -0
- by_framework/core/wait_reply.py +167 -0
- by_framework/core/wait_sweeper.py +1126 -0
- by_framework/metrics/snapshot.py +34 -3
- by_framework/worker/context.py +400 -108
- by_framework/worker/processor.py +113 -9
- by_framework/worker/runner.py +73 -0
- by_framework/worker/worker.py +239 -13
- {by_framework-0.2.2.dev11.dist-info → by_framework-0.2.2.dev13.dist-info}/METADATA +3 -3
- {by_framework-0.2.2.dev11.dist-info → by_framework-0.2.2.dev13.dist-info}/RECORD +18 -14
- {by_framework-0.2.2.dev11.dist-info → by_framework-0.2.2.dev13.dist-info}/WHEEL +1 -1
- {by_framework-0.2.2.dev11.dist-info → by_framework-0.2.2.dev13.dist-info}/entry_points.txt +0 -0
- {by_framework-0.2.2.dev11.dist-info → by_framework-0.2.2.dev13.dist-info}/licenses/LICENSE +0 -0
by_framework/client/client.py
CHANGED
|
@@ -14,6 +14,7 @@ from typing import (TYPE_CHECKING, Any, AsyncIterator, Dict, List, Optional, Pro
|
|
|
14
14
|
|
|
15
15
|
from by_framework.common.constants import (
|
|
16
16
|
CANCEL_MESSAGE_ID_PREFIX,
|
|
17
|
+
CLIENT_SOURCE_AGENT_TYPE,
|
|
17
18
|
EXECUTION_ID_PREFIX,
|
|
18
19
|
MESSAGE_ID_PREFIX,
|
|
19
20
|
RedisKeys,
|
|
@@ -864,7 +865,7 @@ class GatewayClient:
|
|
|
864
865
|
trace_id=trace_id,
|
|
865
866
|
target_agent_type=params["target_agent_type"],
|
|
866
867
|
parent_message_id=params["parent_message_id"] or "",
|
|
867
|
-
source_agent_type=
|
|
868
|
+
source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
|
|
868
869
|
route_policy=route_policy,
|
|
869
870
|
route_status=availability.status,
|
|
870
871
|
stream_name=availability.stream_name or "",
|
|
@@ -968,7 +969,7 @@ class GatewayClient:
|
|
|
968
969
|
"session_id": params["session_id"],
|
|
969
970
|
"trace_id": trace_id,
|
|
970
971
|
"parent_message_id": params["parent_message_id"] or "",
|
|
971
|
-
"source_agent_type":
|
|
972
|
+
"source_agent_type": CLIENT_SOURCE_AGENT_TYPE,
|
|
972
973
|
"target_agent_type": params["target_agent_type"],
|
|
973
974
|
"stream_name": route.stream_name,
|
|
974
975
|
"status": "QUEUED",
|
|
@@ -1117,7 +1118,7 @@ class GatewayClient:
|
|
|
1117
1118
|
message_id=message_id,
|
|
1118
1119
|
parent_message_id=parent_message_id,
|
|
1119
1120
|
worker_id=target_worker_id,
|
|
1120
|
-
source_agent_type=
|
|
1121
|
+
source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
|
|
1121
1122
|
target_agent_type=target_agent_type,
|
|
1122
1123
|
route_policy=route_policy,
|
|
1123
1124
|
route_status=route_status,
|
by_framework/common/constants.py
CHANGED
|
@@ -297,6 +297,100 @@ class RedisKeys:
|
|
|
297
297
|
v2_suffix=f"task_group:{{{group_id}}}:results",
|
|
298
298
|
)
|
|
299
299
|
|
|
300
|
+
@classmethod
|
|
301
|
+
def wait_index(cls, shard: int) -> str:
|
|
302
|
+
"""ZSET index of suspended callers waiting for a sub-task reply.
|
|
303
|
+
|
|
304
|
+
member = encoded wait-index member (see core/wait_index.py),
|
|
305
|
+
score = deadline in epoch milliseconds. Sharded so sweepers can
|
|
306
|
+
claim disjoint slices without a global lock; the shard is derived
|
|
307
|
+
from session_id (see wait_index_shard()).
|
|
308
|
+
|
|
309
|
+
Cross-entity index (spans every session), so deliberately left
|
|
310
|
+
untagged relative to the per-session keys it points at — same rule
|
|
311
|
+
as trace_index_session/admin_workers.
|
|
312
|
+
"""
|
|
313
|
+
return cls._versioned(
|
|
314
|
+
v1_key=f"byai_gateway:wait:index:{shard}",
|
|
315
|
+
v2_suffix=f"wait:index:{shard}",
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
@classmethod
|
|
319
|
+
def wait_sweep_lock(cls, shard: int) -> str:
|
|
320
|
+
"""Short-lived claim on one wait_index() shard, held while sweeping it.
|
|
321
|
+
|
|
322
|
+
Ownership is advisory: it only keeps two sweepers from doing the same
|
|
323
|
+
triage at the same moment. Losing it (expiry, a partitioned worker)
|
|
324
|
+
cannot corrupt anything, because every action a sweep takes is
|
|
325
|
+
idempotent — a duplicate synthesized reply is caught by the same
|
|
326
|
+
wait-index gate that catches a duplicate real one. That is why the
|
|
327
|
+
shards need no leader election: a dead worker's claim simply expires
|
|
328
|
+
and another worker picks the shard up on its next cycle.
|
|
329
|
+
|
|
330
|
+
Cross-entity like the shard it guards, so deliberately untagged.
|
|
331
|
+
"""
|
|
332
|
+
return cls._versioned(
|
|
333
|
+
v1_key=f"byai_gateway:wait:sweep_lock:{shard}",
|
|
334
|
+
v2_suffix=f"wait:sweep_lock:{shard}",
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
@classmethod
|
|
338
|
+
def wait_consumed(cls, session_id: str, member_digest: str) -> str:
|
|
339
|
+
"""Short-lived marker: "this wait-index entry was already resolved".
|
|
340
|
+
|
|
341
|
+
Written by the idempotency gate right after it wins the ZREM for a
|
|
342
|
+
member, and read when a later ZREM for the same member returns 0.
|
|
343
|
+
It is the *only* thing that distinguishes the two meanings of that
|
|
344
|
+
0 — "someone already consumed this wait" (drop the duplicate) from
|
|
345
|
+
"this wait was never registered" (a pre-upgrade or expired entry,
|
|
346
|
+
which must be let through). Without it, every rolling upgrade would
|
|
347
|
+
silently drop in-flight replies.
|
|
348
|
+
|
|
349
|
+
Per-session entity, so hash-tagged with the session in v2.
|
|
350
|
+
"""
|
|
351
|
+
return cls._versioned(
|
|
352
|
+
v1_key=f"byai_gateway:wait:consumed:{session_id}:{member_digest}",
|
|
353
|
+
v2_suffix=f"wait:consumed:{{{session_id}}}:{member_digest}",
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
@classmethod
|
|
357
|
+
def wait_renew_origin(cls, session_id: str, member_digest: str) -> str:
|
|
358
|
+
"""The deadline a wait's renewal budget is measured from.
|
|
359
|
+
|
|
360
|
+
Written once (SET NX) by the first sweep that finds the entry due, so
|
|
361
|
+
it holds the wait's *original* deadline even after renewals have
|
|
362
|
+
overwritten the ZSET score. Without it a renewal budget cannot exist
|
|
363
|
+
at all: every sweep would re-measure from the score it just pushed
|
|
364
|
+
out, and a callee whose worker is alive but making no progress would
|
|
365
|
+
be renewed forever.
|
|
366
|
+
|
|
367
|
+
Sweeper-private: nothing on the reply path reads or writes it, so it
|
|
368
|
+
is deliberately NOT part of the wait-index member (which must stay
|
|
369
|
+
rebuildable from a reply alone — see core/wait_index.py). Expiring is
|
|
370
|
+
safe by design: losing it only restarts the budget from the current
|
|
371
|
+
deadline, so the TTL is sized well above any plausible budget.
|
|
372
|
+
|
|
373
|
+
Per-session entity, so hash-tagged with the session in v2.
|
|
374
|
+
"""
|
|
375
|
+
return cls._versioned(
|
|
376
|
+
v1_key=f"byai_gateway:wait:renew_origin:{session_id}:{member_digest}",
|
|
377
|
+
v2_suffix=f"wait:renew_origin:{{{session_id}}}:{member_digest}",
|
|
378
|
+
)
|
|
379
|
+
|
|
380
|
+
@classmethod
|
|
381
|
+
def harness_state(cls, execution_id: str) -> str:
|
|
382
|
+
"""Serialized in-flight native-agent-harness loop state.
|
|
383
|
+
|
|
384
|
+
Keyed purely by execution_id — the same identity the registry
|
|
385
|
+
already reattaches on RESUME — so any worker instance that picks up
|
|
386
|
+
the eventual ResumeCommand can rehydrate the loop, not only the one
|
|
387
|
+
that started it.
|
|
388
|
+
"""
|
|
389
|
+
return cls._versioned(
|
|
390
|
+
v1_key=f"byai_gateway:harness_state:{execution_id}",
|
|
391
|
+
v2_suffix=f"harness_state:{{{execution_id}}}",
|
|
392
|
+
)
|
|
393
|
+
|
|
300
394
|
# --- Registry ---
|
|
301
395
|
@classmethod
|
|
302
396
|
def known_workers(cls) -> str:
|
|
@@ -517,8 +611,32 @@ class RedisKeys:
|
|
|
517
611
|
MESSAGE_ID_PREFIX = "msg-"
|
|
518
612
|
EXECUTION_ID_PREFIX = "exec-"
|
|
519
613
|
TASK_GROUP_ID_PREFIX = "tg-"
|
|
614
|
+
# A single call_agent (non-group) dispatch stores its result in the same
|
|
615
|
+
# task_group_results Hash a real group uses, under a group id derived from
|
|
616
|
+
# the sub-task's own message_id — i.e. a group of size 1. Keeps one result
|
|
617
|
+
# storage/recovery path instead of two.
|
|
618
|
+
TASK_GROUP_SINGLE_ID_PREFIX = "tg-single-"
|
|
520
619
|
CANCEL_MESSAGE_ID_PREFIX = "msg-cancel-"
|
|
521
620
|
|
|
621
|
+
# Sentinel GatewayClient writes as an execution record's source_agent_type for
|
|
622
|
+
# a dispatch it made itself (client/client.py's initialize_execution and
|
|
623
|
+
# record_failed_route_decision). It is NOT an agent type: nothing declares it,
|
|
624
|
+
# so nothing consumes RedisKeys.ctrl_stream(CLIENT_SOURCE_AGENT_TYPE).
|
|
625
|
+
#
|
|
626
|
+
# Load-bearing wherever a resumed execution recovers its caller from its own
|
|
627
|
+
# record instead of from the resume header (GatewayWorker._resolve_reply_command
|
|
628
|
+
# / GatewayProcessor._resolve_reply_header): a root execution's record carries
|
|
629
|
+
# this, and treating it as a caller both posts the result to a stream no one
|
|
630
|
+
# reads and suppresses the end-of-stream event the session data plane owes the
|
|
631
|
+
# user — the visible half of the bug being prevented.
|
|
632
|
+
CLIENT_SOURCE_AGENT_TYPE = "client"
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def single_call_task_group_id(child_message_id: str) -> str:
|
|
636
|
+
"""Group id under which a single (non-group) call_agent result is stored."""
|
|
637
|
+
return f"{TASK_GROUP_SINGLE_ID_PREFIX}{child_message_id}"
|
|
638
|
+
|
|
639
|
+
|
|
522
640
|
# --- Redis Hash Field Prefixes ---
|
|
523
641
|
# Field prefixes in Session Registry Hash
|
|
524
642
|
EXEC_FIELD_PREFIX = "exec:"
|
|
@@ -529,6 +647,10 @@ MSG_MAP_PREFIX = "msg_map:"
|
|
|
529
647
|
TASK_GROUP_FIELD_TOTAL = "total"
|
|
530
648
|
TASK_GROUP_FIELD_COMPLETED = "completed"
|
|
531
649
|
TASK_GROUP_FIELD_SOURCE_AGENT = "source_agent_type"
|
|
650
|
+
# Set once dispatch fails partway through a batch; any reply that arrives
|
|
651
|
+
# for an aborted group is discarded instead of resuming the (already
|
|
652
|
+
# terminated) caller execution.
|
|
653
|
+
TASK_GROUP_FIELD_ABORTED = "aborted"
|
|
532
654
|
|
|
533
655
|
|
|
534
656
|
# --- Timing and Sleep Constants ---
|
|
@@ -538,12 +660,133 @@ CONTROL_LOOP_SLEEP_SECONDS = 0.01
|
|
|
538
660
|
WAIT_FOR_TASKS_TIMEOUT_SECONDS = 5.0
|
|
539
661
|
# Task group Key TTL (seconds), default 1 day
|
|
540
662
|
TASK_GROUP_TTL_SECONDS = 86400
|
|
663
|
+
# Native agent harness loop-state Key TTL (seconds), default 1 day
|
|
664
|
+
HARNESS_STATE_TTL_SECONDS = 86400
|
|
541
665
|
# First retry wait time (seconds)
|
|
542
666
|
FIRST_RETRY_WAIT_SECONDS = 1.0
|
|
543
667
|
# Maximum retry count
|
|
544
668
|
MAX_RETRY_COUNT = 3
|
|
545
669
|
|
|
546
670
|
|
|
671
|
+
# --- Suspended-caller liveness (wait index) ---
|
|
672
|
+
# Number of RedisKeys.wait_index() shards. Fixed: changing it re-maps every
|
|
673
|
+
# session to a different shard, so in-flight entries would be swept by no
|
|
674
|
+
# one. Treat as a cross-SDK protocol constant, not a tunable.
|
|
675
|
+
WAIT_INDEX_SHARDS = 16
|
|
676
|
+
# Default deadline for a call_agent(wait_for_reply=True) reply (1 hour).
|
|
677
|
+
# Machine waiting on machine.
|
|
678
|
+
DEFAULT_REPLY_TIMEOUT_MS = 3_600_000
|
|
679
|
+
# Default deadline for an ask_user reply. Machine waiting on a human, so it
|
|
680
|
+
# is deliberately decoupled from DEFAULT_REPLY_TIMEOUT_MS and aligned with
|
|
681
|
+
# the session TTL (which is in seconds) instead.
|
|
682
|
+
DEFAULT_ASK_USER_TIMEOUT_MS = RedisKeys.DEFAULT_SESSION_TTL * 1000
|
|
683
|
+
# How often a worker's sweeper scans the shards it owns (seconds).
|
|
684
|
+
WAIT_SWEEP_INTERVAL_SECONDS = 30
|
|
685
|
+
# TTL of a RedisKeys.wait_sweep_lock() claim. Must comfortably exceed one
|
|
686
|
+
# shard's sweep so the owner doesn't lose the shard mid-pass, and stay short
|
|
687
|
+
# enough that a crashed sweeper's shards are picked up again quickly.
|
|
688
|
+
WAIT_SWEEP_LOCK_TTL_SECONDS = 60
|
|
689
|
+
# Most due entries one sweep resolves per shard per cycle. Bounds the work
|
|
690
|
+
# (and the Redis traffic) of a single pass after an outage leaves a large
|
|
691
|
+
# backlog; the remainder is simply picked up next cycle, since entries stay
|
|
692
|
+
# in the index until a reply clears them.
|
|
693
|
+
WAIT_SWEEP_BATCH_LIMIT = 200
|
|
694
|
+
# Fixed extension applied when a sweep finds the callee still making
|
|
695
|
+
# progress. Deliberately a constant rather than the original timeout: the
|
|
696
|
+
# wait-index member must stay reconstructible from a reply alone, so it
|
|
697
|
+
# cannot carry the caller's original timeout.
|
|
698
|
+
WAIT_RENEW_INCREMENT_MS = 300_000
|
|
699
|
+
# Hard ceiling on renewals, as a multiple of the caller's own timeout: a wait
|
|
700
|
+
# may be renewed until `registered_at + N * timeout`, after which the callee
|
|
701
|
+
# is declared CHILD_TIMEOUT even though its worker is still alive.
|
|
702
|
+
#
|
|
703
|
+
# Without a ceiling the "worker lease alive -> renew" rule is unconditional,
|
|
704
|
+
# so a callee that is running but making no progress (a hung LLM call, a
|
|
705
|
+
# deadlock) suspends its caller forever — the one failure mode the deadline
|
|
706
|
+
# was supposed to bound. N is deliberately expressed against the caller's
|
|
707
|
+
# timeout rather than a renewal count, so a caller that asked for 10 minutes
|
|
708
|
+
# is not held to the same absolute budget as one that asked for four hours,
|
|
709
|
+
# and so retuning WAIT_RENEW_INCREMENT_MS cannot silently change the bound.
|
|
710
|
+
#
|
|
711
|
+
# Why 3: N must exceed 1 (N == 1 is "never renew", which kills every callee
|
|
712
|
+
# that is merely slow); N == 2 leaves a single extra window, so one
|
|
713
|
+
# under-estimated timeout is enough to kill healthy work; N == 3 means the
|
|
714
|
+
# caller's own estimate has to be off by 200% before that happens, while
|
|
715
|
+
# still bounding the default case at 3 hours — two orders of magnitude below
|
|
716
|
+
# DEFAULT_SESSION_TTL, which matters because once the session data expires
|
|
717
|
+
# there is no execution record left to compensate against and the wait is
|
|
718
|
+
# simply dropped. Override per deployment via
|
|
719
|
+
# BY_FRAMEWORK_WAIT_RENEW_MAX_MULTIPLE.
|
|
720
|
+
WAIT_RENEW_MAX_MULTIPLE = 3
|
|
721
|
+
# TTL of RedisKeys.wait_renew_origin(). Must comfortably exceed the largest
|
|
722
|
+
# budget in use (N * timeout), or the budget silently restarts mid-wait.
|
|
723
|
+
WAIT_RENEW_ORIGIN_TTL_SECONDS = TASK_GROUP_TTL_SECONDS
|
|
724
|
+
# How long RedisKeys.wait_consumed() remembers that a wait was already
|
|
725
|
+
# resolved, i.e. how far apart two copies of the same reply may be and still
|
|
726
|
+
# be recognized as duplicates.
|
|
727
|
+
#
|
|
728
|
+
# Sized off DEFAULT_SESSION_TTL, which is the lifetime of the session
|
|
729
|
+
# registry — and the session registry is what keeps a *wait entry* relevant.
|
|
730
|
+
# A marker that expires while entries of that session are still live leaves
|
|
731
|
+
# two holes, and the second is the dangerous one:
|
|
732
|
+
#
|
|
733
|
+
# 1. A repeated ask_user answer (a human may take days; the ask_user
|
|
734
|
+
# deadline is DEFAULT_SESSION_TTL itself) is no longer recognized as a
|
|
735
|
+
# duplicate and wakes the caller a second time.
|
|
736
|
+
# 2. Worse: a stale duplicate sub-agent reply, having lost the marker that
|
|
737
|
+
# would stop it at its own candidate, falls through to the ask_user
|
|
738
|
+
# candidate for the same caller and claims a wait that is still live —
|
|
739
|
+
# after which the real answer is dropped as "already consumed".
|
|
740
|
+
#
|
|
741
|
+
# Both close once the marker outlives every wait it may have to arbitrate,
|
|
742
|
+
# i.e. the session TTL. Erring long costs a handful of idle 1-byte keys with
|
|
743
|
+
# the same lifetime as the session registry they belong to; erring short
|
|
744
|
+
# costs a lost user answer.
|
|
745
|
+
WAIT_CONSUMED_TTL_SECONDS = RedisKeys.DEFAULT_SESSION_TTL
|
|
746
|
+
# How often a sweeper prunes entries that are provably beyond use (see
|
|
747
|
+
# WAIT_PRUNE_AFTER_SECONDS). Deliberately far coarser than
|
|
748
|
+
# WAIT_SWEEP_INTERVAL_SECONDS: this is garbage collection on a multi-day
|
|
749
|
+
# horizon, and it is the only work a sweeper does when compensation is off.
|
|
750
|
+
WAIT_PRUNE_INTERVAL_SECONDS = 3600
|
|
751
|
+
# How far in the past a wait entry's score must lie before pruning it.
|
|
752
|
+
#
|
|
753
|
+
# Every writer of an entry sets its score to its own `now` plus a
|
|
754
|
+
# non-negative offset (registration adds the caller's timeout, a renewal adds
|
|
755
|
+
# WAIT_RENEW_INCREMENT_MS), and only ever does so while the caller's
|
|
756
|
+
# execution record exists. So `now - score > this` proves the entry was last
|
|
757
|
+
# touched more than a session TTL ago, hence that the session registry the
|
|
758
|
+
# sweep would interrogate has expired and no triage is possible any more:
|
|
759
|
+
# the entry can only ever produce "caller missing". Pruning it is therefore
|
|
760
|
+
# not a decision, which is why it needs no opt-in.
|
|
761
|
+
#
|
|
762
|
+
# The margin over DEFAULT_SESSION_TTL is what makes that strict rather than
|
|
763
|
+
# coincident: DEFAULT_ASK_USER_TIMEOUT_MS *equals* the session TTL, so a
|
|
764
|
+
# threshold trimmed to the session TTL exactly would land on the boundary of
|
|
765
|
+
# a live ask_user wait and lose to any clock skew between the worker that
|
|
766
|
+
# registered the entry and the one sweeping it. A day is far beyond plausible
|
|
767
|
+
# skew, and being late costs one ZSET member per unresolved call for one
|
|
768
|
+
# extra day — the asymmetry says err long.
|
|
769
|
+
WAIT_PRUNE_AFTER_SECONDS = RedisKeys.DEFAULT_SESSION_TTL + 86400
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
class LivenessErrorCode:
|
|
773
|
+
"""error_code values carried by synthesized/recovered resume replies.
|
|
774
|
+
|
|
775
|
+
Cross-SDK wire contract — Python/TS/Java must emit the same strings;
|
|
776
|
+
callers match on them. Append only, never rename.
|
|
777
|
+
"""
|
|
778
|
+
|
|
779
|
+
# The callee's worker lease expired while its execution was non-terminal.
|
|
780
|
+
CHILD_WORKER_LOST = "CHILD_WORKER_LOST"
|
|
781
|
+
# The callee was alive but produced no reply before the deadline.
|
|
782
|
+
CHILD_TIMEOUT = "CHILD_TIMEOUT"
|
|
783
|
+
# The dispatch was never picked up by any worker.
|
|
784
|
+
CHILD_NEVER_STARTED = "CHILD_NEVER_STARTED"
|
|
785
|
+
# The callee finished and its result was persisted, but the reply
|
|
786
|
+
# message was lost; the result was recovered from storage.
|
|
787
|
+
REPLY_LOST_RECOVERED = "REPLY_LOST_RECOVERED"
|
|
788
|
+
|
|
789
|
+
|
|
547
790
|
# --- Filesystem Constants ---
|
|
548
791
|
DEFAULT_WORKSPACE_DIR = "/workspace"
|
|
549
792
|
|
|
@@ -21,6 +21,10 @@ class EventType(str, Enum):
|
|
|
21
21
|
TASK_CREATE: Task creation event
|
|
22
22
|
STEP_COMPLETE: Step completion event
|
|
23
23
|
TASK_STOP: Task stop event
|
|
24
|
+
ORPHANED_REPLY: A reply that arrived for an already-resolved wait and
|
|
25
|
+
was therefore dropped by the idempotency gate. Diagnostic only —
|
|
26
|
+
the sub-agent did real work whose result nobody will consume, so
|
|
27
|
+
it must not vanish silently.
|
|
24
28
|
"""
|
|
25
29
|
|
|
26
30
|
ANSWER_DELTA = "answerDelta"
|
|
@@ -32,3 +36,4 @@ class EventType(str, Enum):
|
|
|
32
36
|
TASK_CREATE = "taskCreate"
|
|
33
37
|
STEP_COMPLETE = "stepComplete"
|
|
34
38
|
TASK_STOP = "taskStop"
|
|
39
|
+
ORPHANED_REPLY = "orphanedReply"
|
by_framework/core/registry.py
CHANGED
|
@@ -239,6 +239,33 @@ async def check_worker_online(
|
|
|
239
239
|
return is_legacy or last_seen > 0
|
|
240
240
|
|
|
241
241
|
|
|
242
|
+
async def acquire_scoped_lock(
|
|
243
|
+
redis: Redis,
|
|
244
|
+
key: str,
|
|
245
|
+
token: str,
|
|
246
|
+
ttl_seconds: int,
|
|
247
|
+
) -> bool:
|
|
248
|
+
"""Claim `key` for `token` if nobody holds it (Redlock acquire half).
|
|
249
|
+
|
|
250
|
+
The stored value must stay a cjson-decodable object carrying a "token"
|
|
251
|
+
field: `_REFRESH_LOCK_SCRIPT` / `_RELEASE_LOCK_SCRIPT` parse it that way,
|
|
252
|
+
and a bare token string would decode as unparseable legacy data, making
|
|
253
|
+
the holder unable to release its own lock.
|
|
254
|
+
"""
|
|
255
|
+
stored = await redis.set(
|
|
256
|
+
key,
|
|
257
|
+
json.dumps({"token": token}, separators=(",", ":")),
|
|
258
|
+
nx=True,
|
|
259
|
+
ex=ttl_seconds,
|
|
260
|
+
)
|
|
261
|
+
return bool(stored)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
async def release_scoped_lock(redis: Redis, key: str, token: str) -> bool:
|
|
265
|
+
"""Release a lock taken with acquire_scoped_lock(), if still owned."""
|
|
266
|
+
return bool(await redis.eval(_RELEASE_LOCK_SCRIPT, 1, key, token or ""))
|
|
267
|
+
|
|
268
|
+
|
|
242
269
|
async def check_agent_type_online(
|
|
243
270
|
redis: Redis,
|
|
244
271
|
agent_type: str,
|
|
@@ -1225,6 +1252,10 @@ class WorkerRegistry:
|
|
|
1225
1252
|
"active": self._get_int_hash_value(raw_counts, "active_count"),
|
|
1226
1253
|
"queued": self._get_int_hash_value(raw_counts, "queued_count"),
|
|
1227
1254
|
"running": self._get_int_hash_value(raw_counts, "running_count"),
|
|
1255
|
+
"waiting_agent": self._get_int_hash_value(
|
|
1256
|
+
raw_counts, "waiting_agent_count"
|
|
1257
|
+
),
|
|
1258
|
+
"waiting_user": self._get_int_hash_value(raw_counts, "waiting_user_count"),
|
|
1228
1259
|
"cancelling": self._get_int_hash_value(raw_counts, "cancelling_count"),
|
|
1229
1260
|
"completed": self._get_int_hash_value(raw_counts, "completed_count"),
|
|
1230
1261
|
"failed": self._get_int_hash_value(raw_counts, "failed_count"),
|
|
@@ -1245,6 +1276,10 @@ class WorkerRegistry:
|
|
|
1245
1276
|
status_counts = {
|
|
1246
1277
|
"QUEUED": counts["queued"],
|
|
1247
1278
|
"RUNNING": counts["running"],
|
|
1279
|
+
# Suspended callers persist as WAITING_* rather than QUEUED; these
|
|
1280
|
+
# rows keep them visible (zero-valued entries are filtered below).
|
|
1281
|
+
"WAITING_AGENT": counts["waiting_agent"],
|
|
1282
|
+
"WAITING_USER": counts["waiting_user"],
|
|
1248
1283
|
"CANCELLING": counts["cancelling"],
|
|
1249
1284
|
"COMPLETED": counts["completed"],
|
|
1250
1285
|
"FAILED": counts["failed"],
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Idempotency gate for replies that resume a suspended caller.
|
|
2
|
+
|
|
3
|
+
A suspended caller is woken by exactly one ``ResumeCommand``. Once a sweep
|
|
4
|
+
can *synthesize* that reply (a callee whose worker died will never send
|
|
5
|
+
one), two copies can exist for the same wait: the synthesized one and the
|
|
6
|
+
real one that shows up late. Waking the caller twice re-runs a finished
|
|
7
|
+
execution and, in a Task Group, pushes ``completed`` past ``total`` and
|
|
8
|
+
aggregates a second time.
|
|
9
|
+
|
|
10
|
+
The gate is the single place that decides which copy wins. It runs at the
|
|
11
|
+
one point every reply passes through — right after a reply is parsed,
|
|
12
|
+
before any registry lookup and before Task Group join accounting — and
|
|
13
|
+
claims the caller's wait-index entry with a ``ZREM``. Exactly one claimant
|
|
14
|
+
can win, because ``ZREM`` is atomic.
|
|
15
|
+
|
|
16
|
+
The hard part is what ``ZREM`` returning 0 means, since it conflates two
|
|
17
|
+
opposite situations:
|
|
18
|
+
|
|
19
|
+
* the entry existed and someone else already claimed it — a true duplicate,
|
|
20
|
+
drop it;
|
|
21
|
+
* the entry never existed — the reply belongs to a dispatch made before
|
|
22
|
+
this version shipped, or to a wait whose entry expired. Dropping it
|
|
23
|
+
silently loses a real reply, and during a rolling upgrade *every*
|
|
24
|
+
in-flight reply looks like this.
|
|
25
|
+
|
|
26
|
+
A short-lived "consumed" marker written by the winner separates them: a 0
|
|
27
|
+
with a marker is a duplicate, a 0 without one is unregistered and must be
|
|
28
|
+
let through. When in doubt the gate lets the message through — a spurious
|
|
29
|
+
extra wake-up is recoverable, a dropped reply is permanent silence. The
|
|
30
|
+
same rule makes the gate fail *open*: any Redis error here allows the
|
|
31
|
+
message.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from typing import Any, NamedTuple
|
|
35
|
+
|
|
36
|
+
from by_framework.common.constants import WAIT_CONSUMED_TTL_SECONDS, RedisKeys
|
|
37
|
+
from by_framework.common.logger import logger
|
|
38
|
+
from by_framework.core.protocol.commands import GatewayCommand
|
|
39
|
+
from by_framework.core.wait_index import (
|
|
40
|
+
encode_member,
|
|
41
|
+
member_digest,
|
|
42
|
+
member_from_resume,
|
|
43
|
+
wait_index_key,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
# Why a reply was allowed through / dropped. Carried on the decision so the
|
|
47
|
+
# caller can log it and put it on the orphaned_reply event.
|
|
48
|
+
ALLOW_CLAIMED = "claimed"
|
|
49
|
+
ALLOW_UNREGISTERED = "unregistered"
|
|
50
|
+
ALLOW_GATE_ERROR = "gate_error"
|
|
51
|
+
DENY_ALREADY_CONSUMED = "already_consumed"
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class WaitGateDecision(NamedTuple):
|
|
55
|
+
"""Outcome of the gate.
|
|
56
|
+
|
|
57
|
+
Attributes:
|
|
58
|
+
allow: Whether the reply may be processed. False only when the wait
|
|
59
|
+
it targets is provably already resolved.
|
|
60
|
+
reason: One of the ALLOW_*/DENY_* constants above.
|
|
61
|
+
member: The wait-index member the decision was made about ("" when
|
|
62
|
+
no candidate matched).
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
allow: bool
|
|
66
|
+
reason: str
|
|
67
|
+
member: str = ""
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def consumed_marker_key(session_id: str, member: str) -> str:
|
|
71
|
+
"""Redis key of the "already consumed" marker for one wait-index member.
|
|
72
|
+
|
|
73
|
+
Part of the cross-SDK contract, since any SDK's worker may gate another
|
|
74
|
+
SDK's reply — see ``wait_index.member_digest`` for why the member is
|
|
75
|
+
hashed rather than embedded.
|
|
76
|
+
"""
|
|
77
|
+
return RedisKeys.wait_consumed(session_id, member_digest(member))
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def candidate_members(command: GatewayCommand) -> list[str]:
|
|
81
|
+
"""Wait-index members this reply could be clearing, most-specific first.
|
|
82
|
+
|
|
83
|
+
Normally there is exactly one: the member rebuilt from the reply's own
|
|
84
|
+
header. The second candidate covers ``ask_user``, which registers with
|
|
85
|
+
an empty ``child_message_id`` because it has no sub-task — while the
|
|
86
|
+
matching reply comes from a client that is free to put anything in
|
|
87
|
+
``header.parent_message_id`` (existing callers put the caller's own
|
|
88
|
+
parent there, not an empty string), so it cannot be rebuilt exactly.
|
|
89
|
+
|
|
90
|
+
Order matters, and so does the fact that each candidate is fully
|
|
91
|
+
resolved (claim, then check its marker) before the next is tried: a
|
|
92
|
+
duplicate sub-agent reply must be caught by *its own* marker rather than
|
|
93
|
+
fall through and clear a live ask_user wait that happens to belong to
|
|
94
|
+
the same caller.
|
|
95
|
+
|
|
96
|
+
A reply carrying a ``task_group_id`` is a sub-agent reply by
|
|
97
|
+
construction, so the ask_user variant is not even considered for it.
|
|
98
|
+
"""
|
|
99
|
+
header = command.header
|
|
100
|
+
members = [member_from_resume(command)]
|
|
101
|
+
if not header.task_group_id:
|
|
102
|
+
ask_user_member = encode_member(
|
|
103
|
+
session_id=header.session_id,
|
|
104
|
+
parent_message_id=header.message_id,
|
|
105
|
+
child_message_id="",
|
|
106
|
+
task_group_id="",
|
|
107
|
+
)
|
|
108
|
+
if ask_user_member not in members:
|
|
109
|
+
members.append(ask_user_member)
|
|
110
|
+
return members
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
async def _mark_consumed(redis: Any, session_id: str, member: str) -> None:
|
|
114
|
+
"""Record that this wait was resolved, so a late twin can be recognized.
|
|
115
|
+
|
|
116
|
+
Fail-soft: losing the marker only means a much later duplicate would be
|
|
117
|
+
allowed through (one extra wake-up), which is the direction this whole
|
|
118
|
+
module errs in anyway.
|
|
119
|
+
"""
|
|
120
|
+
try:
|
|
121
|
+
await redis.set(
|
|
122
|
+
consumed_marker_key(session_id, member),
|
|
123
|
+
"1",
|
|
124
|
+
ex=WAIT_CONSUMED_TTL_SECONDS,
|
|
125
|
+
)
|
|
126
|
+
except Exception as error: # pylint: disable=broad-exception-caught
|
|
127
|
+
logger.warning(
|
|
128
|
+
"Failed to mark wait entry consumed (session=%s): %s", session_id, error
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
async def consume_wait_entry(redis: Any, command: GatewayCommand) -> WaitGateDecision:
|
|
133
|
+
"""Claim the wait a reply resolves; report whether it may be processed.
|
|
134
|
+
|
|
135
|
+
Call once per ``ResumeCommand``, before the execution lookup and before
|
|
136
|
+
Task Group join accounting.
|
|
137
|
+
"""
|
|
138
|
+
session_id = command.header.session_id
|
|
139
|
+
try:
|
|
140
|
+
index_key = wait_index_key(session_id)
|
|
141
|
+
for member in candidate_members(command):
|
|
142
|
+
removed = int(await redis.zrem(index_key, member) or 0)
|
|
143
|
+
if removed > 0:
|
|
144
|
+
await _mark_consumed(redis, session_id, member)
|
|
145
|
+
return WaitGateDecision(True, ALLOW_CLAIMED, member)
|
|
146
|
+
if await redis.exists(consumed_marker_key(session_id, member)):
|
|
147
|
+
return WaitGateDecision(False, DENY_ALREADY_CONSUMED, member)
|
|
148
|
+
# No entry, no marker: nobody ever registered this wait (a dispatch
|
|
149
|
+
# from before this version, or an entry that outlived its index).
|
|
150
|
+
# Unknown is not the same as duplicate — let it through.
|
|
151
|
+
return WaitGateDecision(True, ALLOW_UNREGISTERED)
|
|
152
|
+
except Exception as error: # pylint: disable=broad-exception-caught
|
|
153
|
+
# Fail open. A gate that drops messages when Redis hiccups is worse
|
|
154
|
+
# than the duplicate it was built to prevent.
|
|
155
|
+
logger.warning(
|
|
156
|
+
"Wait-index gate unavailable for session=%s, allowing reply: %s",
|
|
157
|
+
session_id,
|
|
158
|
+
error,
|
|
159
|
+
)
|
|
160
|
+
return WaitGateDecision(True, ALLOW_GATE_ERROR)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
async def emit_orphaned_reply(
|
|
164
|
+
redis: Any,
|
|
165
|
+
command: GatewayCommand,
|
|
166
|
+
*,
|
|
167
|
+
worker_id: str = "",
|
|
168
|
+
reason: str = DENY_ALREADY_CONSUMED,
|
|
169
|
+
) -> None:
|
|
170
|
+
"""Announce on the session data stream that a reply was dropped.
|
|
171
|
+
|
|
172
|
+
A dropped reply is not noise: the sub-agent ran, produced a result, and
|
|
173
|
+
may have had side effects that nobody will now account for. Emitting it
|
|
174
|
+
on the existing data plane keeps it visible without inventing a second
|
|
175
|
+
reporting mechanism.
|
|
176
|
+
|
|
177
|
+
Fail-soft by construction — the drop/allow decision has already been
|
|
178
|
+
made, and reporting it must never change or block it.
|
|
179
|
+
"""
|
|
180
|
+
header = command.header
|
|
181
|
+
try:
|
|
182
|
+
from by_framework.common.emitter import GatewayDataEmitter
|
|
183
|
+
from by_framework.core.protocol.event_type import EventType
|
|
184
|
+
|
|
185
|
+
await GatewayDataEmitter(redis_client=redis).emit_event(
|
|
186
|
+
session_id=header.session_id,
|
|
187
|
+
trace_id=header.trace_id,
|
|
188
|
+
event_type=EventType.ORPHANED_REPLY.value,
|
|
189
|
+
source_agent_type=header.source_agent_type,
|
|
190
|
+
message_id=header.message_id,
|
|
191
|
+
parent_message_id=header.parent_message_id,
|
|
192
|
+
data={
|
|
193
|
+
"reason": reason,
|
|
194
|
+
# The suspended caller this reply was addressed to...
|
|
195
|
+
"caller_message_id": header.message_id,
|
|
196
|
+
# ...and the sub-task that produced it (empty for ask_user).
|
|
197
|
+
"child_message_id": header.parent_message_id,
|
|
198
|
+
"task_group_id": header.task_group_id,
|
|
199
|
+
"status": str(getattr(command, "status", "") or ""),
|
|
200
|
+
"worker_id": worker_id,
|
|
201
|
+
},
|
|
202
|
+
)
|
|
203
|
+
except Exception as error: # pylint: disable=broad-exception-caught
|
|
204
|
+
logger.warning(
|
|
205
|
+
"Failed to emit orphaned_reply event (session=%s, message_id=%s): %s",
|
|
206
|
+
header.session_id,
|
|
207
|
+
header.message_id,
|
|
208
|
+
error,
|
|
209
|
+
)
|