by-framework 0.2.2.dev10__py3-none-any.whl → 0.2.2.dev12__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- by_framework/__main__.py +83 -4
- by_framework/admin/cli.py +6 -5
- by_framework/client/client.py +4 -3
- by_framework/common/config.py +4 -2
- by_framework/common/constants.py +243 -0
- by_framework/core/protocol/event_type.py +5 -0
- by_framework/core/registry.py +35 -0
- by_framework/core/wait_gate.py +209 -0
- by_framework/core/wait_index.py +165 -0
- by_framework/core/wait_reply.py +167 -0
- by_framework/core/wait_sweeper.py +1126 -0
- by_framework/metrics/snapshot.py +34 -3
- by_framework/worker/app.py +68 -24
- by_framework/worker/context.py +395 -108
- by_framework/worker/health_server.py +169 -0
- by_framework/worker/processor.py +113 -9
- by_framework/worker/runner.py +106 -0
- by_framework/worker/worker.py +230 -13
- {by_framework-0.2.2.dev10.dist-info → by_framework-0.2.2.dev12.dist-info}/METADATA +28 -5
- {by_framework-0.2.2.dev10.dist-info → by_framework-0.2.2.dev12.dist-info}/RECORD +23 -18
- {by_framework-0.2.2.dev10.dist-info → by_framework-0.2.2.dev12.dist-info}/WHEEL +1 -1
- {by_framework-0.2.2.dev10.dist-info → by_framework-0.2.2.dev12.dist-info}/entry_points.txt +0 -0
- {by_framework-0.2.2.dev10.dist-info → by_framework-0.2.2.dev12.dist-info}/licenses/LICENSE +0 -0
by_framework/__main__.py
CHANGED
|
@@ -12,6 +12,17 @@ import importlib
|
|
|
12
12
|
from .worker.app import run_worker
|
|
13
13
|
|
|
14
14
|
|
|
15
|
+
def _parse_cluster_nodes(value: str):
|
|
16
|
+
"""Parse a comma-separated "host:port,host:port" string, mirroring
|
|
17
|
+
RedisConfig.from_env()'s REDIS_CLUSTER_HOST/REDIS_CLUSTER_NODES parsing
|
|
18
|
+
so the CLI and env-var paths accept the same format."""
|
|
19
|
+
nodes = []
|
|
20
|
+
for node in value.split(","):
|
|
21
|
+
node_host, node_port = node.rsplit(":", 1)
|
|
22
|
+
nodes.append((node_host, int(node_port)))
|
|
23
|
+
return nodes
|
|
24
|
+
|
|
25
|
+
|
|
15
26
|
def parse_args():
|
|
16
27
|
"""Parse command line arguments for the CLI runner."""
|
|
17
28
|
parser = argparse.ArgumentParser(description="By-Framework CLI Runner")
|
|
@@ -25,12 +36,55 @@ def parse_args():
|
|
|
25
36
|
),
|
|
26
37
|
)
|
|
27
38
|
parser.add_argument(
|
|
28
|
-
"--redis-host",
|
|
39
|
+
"--redis-host",
|
|
40
|
+
type=str,
|
|
41
|
+
default=None,
|
|
42
|
+
help="Redis server hostname (default: REDIS_HOST env var, then 'localhost')",
|
|
43
|
+
)
|
|
44
|
+
parser.add_argument(
|
|
45
|
+
"--redis-port",
|
|
46
|
+
type=int,
|
|
47
|
+
default=None,
|
|
48
|
+
help="Redis server port (default: REDIS_PORT env var, then 6379)",
|
|
49
|
+
)
|
|
50
|
+
parser.add_argument(
|
|
51
|
+
"--redis-db",
|
|
52
|
+
type=int,
|
|
53
|
+
default=None,
|
|
54
|
+
help="Redis database number (default: REDIS_DATABASE env var, then 0)",
|
|
29
55
|
)
|
|
30
56
|
parser.add_argument(
|
|
31
|
-
"--redis-
|
|
57
|
+
"--redis-password",
|
|
58
|
+
type=str,
|
|
59
|
+
default=None,
|
|
60
|
+
help="Redis password (default: REDIS_PASSWORD env var)",
|
|
61
|
+
)
|
|
62
|
+
parser.add_argument(
|
|
63
|
+
"--redis-username",
|
|
64
|
+
type=str,
|
|
65
|
+
default=None,
|
|
66
|
+
help="Redis username, for ACL-enabled Redis (default: REDIS_USERNAME env var)",
|
|
67
|
+
)
|
|
68
|
+
parser.add_argument(
|
|
69
|
+
"--redis-mode",
|
|
70
|
+
type=str,
|
|
71
|
+
choices=["standalone", "cluster"],
|
|
72
|
+
default=None,
|
|
73
|
+
help=(
|
|
74
|
+
"Redis deployment mode (default: REDIS_MODE env var; implied by "
|
|
75
|
+
"--redis-cluster-nodes/REDIS_CLUSTER_HOST alone if neither is set)"
|
|
76
|
+
),
|
|
77
|
+
)
|
|
78
|
+
parser.add_argument(
|
|
79
|
+
"--redis-cluster-nodes",
|
|
80
|
+
type=str,
|
|
81
|
+
default=None,
|
|
82
|
+
help=(
|
|
83
|
+
"Comma-separated Redis Cluster seed nodes, 'host:port,host:port' "
|
|
84
|
+
"(default: REDIS_CLUSTER_HOST/REDIS_CLUSTER_NODES env var). "
|
|
85
|
+
"Passing this alone implies --redis-mode cluster."
|
|
86
|
+
),
|
|
32
87
|
)
|
|
33
|
-
parser.add_argument("--redis-db", type=int, default=0, help="Redis database number")
|
|
34
88
|
parser.add_argument(
|
|
35
89
|
"--worker-id", type=str, default="worker-1", help="Unique worker identifier"
|
|
36
90
|
)
|
|
@@ -47,6 +101,17 @@ def parse_args():
|
|
|
47
101
|
parser.add_argument(
|
|
48
102
|
"--redis-max-connections", type=int, help="Max Redis connections allowed"
|
|
49
103
|
)
|
|
104
|
+
parser.add_argument(
|
|
105
|
+
"--health-port",
|
|
106
|
+
type=int,
|
|
107
|
+
default=None,
|
|
108
|
+
help=(
|
|
109
|
+
"Port for the local /readyz readiness endpoint (Docker/Kubernetes "
|
|
110
|
+
"health checks). Opt-in only - omit to leave it disabled, no "
|
|
111
|
+
"default port is provided. Also settable via BYAI_WORKER_HEALTH_PORT. "
|
|
112
|
+
"See docs/architecture/worker-readiness-endpoint.md."
|
|
113
|
+
),
|
|
114
|
+
)
|
|
50
115
|
|
|
51
116
|
return parser.parse_args()
|
|
52
117
|
|
|
@@ -68,17 +133,31 @@ def main():
|
|
|
68
133
|
|
|
69
134
|
print(f"Starting worker class: {worker_class.__name__} (ID: {args.worker_id})")
|
|
70
135
|
|
|
71
|
-
|
|
136
|
+
redis_cluster_nodes = (
|
|
137
|
+
_parse_cluster_nodes(args.redis_cluster_nodes)
|
|
138
|
+
if args.redis_cluster_nodes
|
|
139
|
+
else None
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
# Delegate to the common launcher. Every redis_* arg defaults to None
|
|
143
|
+
# here (not a literal like "localhost"), same as run_worker()'s own
|
|
144
|
+
# defaults, so an unset CLI flag actually falls through to the
|
|
145
|
+
# corresponding REDIS_* env var instead of silently shadowing it.
|
|
72
146
|
run_worker(
|
|
73
147
|
worker_class=worker_class,
|
|
74
148
|
worker_id=args.worker_id,
|
|
75
149
|
redis_host=args.redis_host,
|
|
76
150
|
redis_port=args.redis_port,
|
|
77
151
|
redis_db=args.redis_db,
|
|
152
|
+
redis_password=args.redis_password,
|
|
153
|
+
redis_username=args.redis_username,
|
|
154
|
+
redis_mode=args.redis_mode,
|
|
155
|
+
redis_cluster_nodes=redis_cluster_nodes,
|
|
78
156
|
workspace_dir=args.workspace,
|
|
79
157
|
max_concurrency=args.max_concurrency,
|
|
80
158
|
fetch_count=args.fetch_count,
|
|
81
159
|
redis_max_connections=args.redis_max_connections,
|
|
160
|
+
health_port=args.health_port,
|
|
82
161
|
)
|
|
83
162
|
|
|
84
163
|
|
by_framework/admin/cli.py
CHANGED
|
@@ -50,7 +50,6 @@ app.add_typer(metrics_app, name="metrics", help="Metrics and observability")
|
|
|
50
50
|
|
|
51
51
|
console = Console()
|
|
52
52
|
err_console = Console(stderr=True)
|
|
53
|
-
_DEFAULT_REDIS_URL = "redis://localhost:6379/0"
|
|
54
53
|
_redis_url: Optional[str] = None
|
|
55
54
|
|
|
56
55
|
|
|
@@ -85,10 +84,12 @@ def _get_redis(redis_url: Optional[str] = None):
|
|
|
85
84
|
configured_url = redis_url if redis_url is not None else _redis_url
|
|
86
85
|
if configured_url:
|
|
87
86
|
return init_redis_from_url(configured_url)
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
87
|
+
# Always resolve through RedisConfig.from_env() - for both standalone and
|
|
88
|
+
# cluster mode - so REDIS_HOST/PORT/PASSWORD/USERNAME/DATABASE are honored
|
|
89
|
+
# the same way run_worker()'s config resolution honors them. Previously
|
|
90
|
+
# only cluster mode read env_config; standalone mode silently ignored it
|
|
91
|
+
# and connected to a hardcoded localhost:6379/0.
|
|
92
|
+
return init_redis(config=RedisConfig.from_env())
|
|
92
93
|
|
|
93
94
|
|
|
94
95
|
# --------------------------------------------------------------------------- #
|
by_framework/client/client.py
CHANGED
|
@@ -14,6 +14,7 @@ from typing import (TYPE_CHECKING, Any, AsyncIterator, Dict, List, Optional, Pro
|
|
|
14
14
|
|
|
15
15
|
from by_framework.common.constants import (
|
|
16
16
|
CANCEL_MESSAGE_ID_PREFIX,
|
|
17
|
+
CLIENT_SOURCE_AGENT_TYPE,
|
|
17
18
|
EXECUTION_ID_PREFIX,
|
|
18
19
|
MESSAGE_ID_PREFIX,
|
|
19
20
|
RedisKeys,
|
|
@@ -864,7 +865,7 @@ class GatewayClient:
|
|
|
864
865
|
trace_id=trace_id,
|
|
865
866
|
target_agent_type=params["target_agent_type"],
|
|
866
867
|
parent_message_id=params["parent_message_id"] or "",
|
|
867
|
-
source_agent_type=
|
|
868
|
+
source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
|
|
868
869
|
route_policy=route_policy,
|
|
869
870
|
route_status=availability.status,
|
|
870
871
|
stream_name=availability.stream_name or "",
|
|
@@ -968,7 +969,7 @@ class GatewayClient:
|
|
|
968
969
|
"session_id": params["session_id"],
|
|
969
970
|
"trace_id": trace_id,
|
|
970
971
|
"parent_message_id": params["parent_message_id"] or "",
|
|
971
|
-
"source_agent_type":
|
|
972
|
+
"source_agent_type": CLIENT_SOURCE_AGENT_TYPE,
|
|
972
973
|
"target_agent_type": params["target_agent_type"],
|
|
973
974
|
"stream_name": route.stream_name,
|
|
974
975
|
"status": "QUEUED",
|
|
@@ -1117,7 +1118,7 @@ class GatewayClient:
|
|
|
1117
1118
|
message_id=message_id,
|
|
1118
1119
|
parent_message_id=parent_message_id,
|
|
1119
1120
|
worker_id=target_worker_id,
|
|
1120
|
-
source_agent_type=
|
|
1121
|
+
source_agent_type=CLIENT_SOURCE_AGENT_TYPE,
|
|
1121
1122
|
target_agent_type=target_agent_type,
|
|
1122
1123
|
route_policy=route_policy,
|
|
1123
1124
|
route_status=route_status,
|
by_framework/common/config.py
CHANGED
|
@@ -43,6 +43,8 @@ class RedisConfig:
|
|
|
43
43
|
"""
|
|
44
44
|
password = os.environ.get("REDIS_PASSWORD", "")
|
|
45
45
|
username = os.environ.get("REDIS_USERNAME") or None
|
|
46
|
+
host = os.environ.get("REDIS_HOST") or "localhost"
|
|
47
|
+
port_str = os.environ.get("REDIS_PORT")
|
|
46
48
|
max_connections = os.environ.get("REDIS_MAX_CONNECTIONS")
|
|
47
49
|
cluster_host_str = os.environ.get("REDIS_CLUSTER_HOST")
|
|
48
50
|
cluster_nodes_str = cluster_host_str or os.environ.get("REDIS_CLUSTER_NODES")
|
|
@@ -61,8 +63,8 @@ class RedisConfig:
|
|
|
61
63
|
if db_str is not None:
|
|
62
64
|
logger.warning("REDIS_DB is deprecated, use REDIS_DATABASE instead")
|
|
63
65
|
return cls(
|
|
64
|
-
host=
|
|
65
|
-
port=int(
|
|
66
|
+
host=host,
|
|
67
|
+
port=int(port_str) if port_str else 6379,
|
|
66
68
|
db=int(db_str) if db_str is not None else 0,
|
|
67
69
|
password=password,
|
|
68
70
|
username=username,
|
by_framework/common/constants.py
CHANGED
|
@@ -297,6 +297,100 @@ class RedisKeys:
|
|
|
297
297
|
v2_suffix=f"task_group:{{{group_id}}}:results",
|
|
298
298
|
)
|
|
299
299
|
|
|
300
|
+
@classmethod
|
|
301
|
+
def wait_index(cls, shard: int) -> str:
|
|
302
|
+
"""ZSET index of suspended callers waiting for a sub-task reply.
|
|
303
|
+
|
|
304
|
+
member = encoded wait-index member (see core/wait_index.py),
|
|
305
|
+
score = deadline in epoch milliseconds. Sharded so sweepers can
|
|
306
|
+
claim disjoint slices without a global lock; the shard is derived
|
|
307
|
+
from session_id (see wait_index_shard()).
|
|
308
|
+
|
|
309
|
+
Cross-entity index (spans every session), so deliberately left
|
|
310
|
+
untagged relative to the per-session keys it points at — same rule
|
|
311
|
+
as trace_index_session/admin_workers.
|
|
312
|
+
"""
|
|
313
|
+
return cls._versioned(
|
|
314
|
+
v1_key=f"byai_gateway:wait:index:{shard}",
|
|
315
|
+
v2_suffix=f"wait:index:{shard}",
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
@classmethod
|
|
319
|
+
def wait_sweep_lock(cls, shard: int) -> str:
|
|
320
|
+
"""Short-lived claim on one wait_index() shard, held while sweeping it.
|
|
321
|
+
|
|
322
|
+
Ownership is advisory: it only keeps two sweepers from doing the same
|
|
323
|
+
triage at the same moment. Losing it (expiry, a partitioned worker)
|
|
324
|
+
cannot corrupt anything, because every action a sweep takes is
|
|
325
|
+
idempotent — a duplicate synthesized reply is caught by the same
|
|
326
|
+
wait-index gate that catches a duplicate real one. That is why the
|
|
327
|
+
shards need no leader election: a dead worker's claim simply expires
|
|
328
|
+
and another worker picks the shard up on its next cycle.
|
|
329
|
+
|
|
330
|
+
Cross-entity like the shard it guards, so deliberately untagged.
|
|
331
|
+
"""
|
|
332
|
+
return cls._versioned(
|
|
333
|
+
v1_key=f"byai_gateway:wait:sweep_lock:{shard}",
|
|
334
|
+
v2_suffix=f"wait:sweep_lock:{shard}",
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
@classmethod
|
|
338
|
+
def wait_consumed(cls, session_id: str, member_digest: str) -> str:
|
|
339
|
+
"""Short-lived marker: "this wait-index entry was already resolved".
|
|
340
|
+
|
|
341
|
+
Written by the idempotency gate right after it wins the ZREM for a
|
|
342
|
+
member, and read when a later ZREM for the same member returns 0.
|
|
343
|
+
It is the *only* thing that distinguishes the two meanings of that
|
|
344
|
+
0 — "someone already consumed this wait" (drop the duplicate) from
|
|
345
|
+
"this wait was never registered" (a pre-upgrade or expired entry,
|
|
346
|
+
which must be let through). Without it, every rolling upgrade would
|
|
347
|
+
silently drop in-flight replies.
|
|
348
|
+
|
|
349
|
+
Per-session entity, so hash-tagged with the session in v2.
|
|
350
|
+
"""
|
|
351
|
+
return cls._versioned(
|
|
352
|
+
v1_key=f"byai_gateway:wait:consumed:{session_id}:{member_digest}",
|
|
353
|
+
v2_suffix=f"wait:consumed:{{{session_id}}}:{member_digest}",
|
|
354
|
+
)
|
|
355
|
+
|
|
356
|
+
@classmethod
|
|
357
|
+
def wait_renew_origin(cls, session_id: str, member_digest: str) -> str:
|
|
358
|
+
"""The deadline a wait's renewal budget is measured from.
|
|
359
|
+
|
|
360
|
+
Written once (SET NX) by the first sweep that finds the entry due, so
|
|
361
|
+
it holds the wait's *original* deadline even after renewals have
|
|
362
|
+
overwritten the ZSET score. Without it a renewal budget cannot exist
|
|
363
|
+
at all: every sweep would re-measure from the score it just pushed
|
|
364
|
+
out, and a callee whose worker is alive but making no progress would
|
|
365
|
+
be renewed forever.
|
|
366
|
+
|
|
367
|
+
Sweeper-private: nothing on the reply path reads or writes it, so it
|
|
368
|
+
is deliberately NOT part of the wait-index member (which must stay
|
|
369
|
+
rebuildable from a reply alone — see core/wait_index.py). Expiring is
|
|
370
|
+
safe by design: losing it only restarts the budget from the current
|
|
371
|
+
deadline, so the TTL is sized well above any plausible budget.
|
|
372
|
+
|
|
373
|
+
Per-session entity, so hash-tagged with the session in v2.
|
|
374
|
+
"""
|
|
375
|
+
return cls._versioned(
|
|
376
|
+
v1_key=f"byai_gateway:wait:renew_origin:{session_id}:{member_digest}",
|
|
377
|
+
v2_suffix=f"wait:renew_origin:{{{session_id}}}:{member_digest}",
|
|
378
|
+
)
|
|
379
|
+
|
|
380
|
+
@classmethod
|
|
381
|
+
def harness_state(cls, execution_id: str) -> str:
|
|
382
|
+
"""Serialized in-flight native-agent-harness loop state.
|
|
383
|
+
|
|
384
|
+
Keyed purely by execution_id — the same identity the registry
|
|
385
|
+
already reattaches on RESUME — so any worker instance that picks up
|
|
386
|
+
the eventual ResumeCommand can rehydrate the loop, not only the one
|
|
387
|
+
that started it.
|
|
388
|
+
"""
|
|
389
|
+
return cls._versioned(
|
|
390
|
+
v1_key=f"byai_gateway:harness_state:{execution_id}",
|
|
391
|
+
v2_suffix=f"harness_state:{{{execution_id}}}",
|
|
392
|
+
)
|
|
393
|
+
|
|
300
394
|
# --- Registry ---
|
|
301
395
|
@classmethod
|
|
302
396
|
def known_workers(cls) -> str:
|
|
@@ -517,8 +611,32 @@ class RedisKeys:
|
|
|
517
611
|
MESSAGE_ID_PREFIX = "msg-"
|
|
518
612
|
EXECUTION_ID_PREFIX = "exec-"
|
|
519
613
|
TASK_GROUP_ID_PREFIX = "tg-"
|
|
614
|
+
# A single call_agent (non-group) dispatch stores its result in the same
|
|
615
|
+
# task_group_results Hash a real group uses, under a group id derived from
|
|
616
|
+
# the sub-task's own message_id — i.e. a group of size 1. Keeps one result
|
|
617
|
+
# storage/recovery path instead of two.
|
|
618
|
+
TASK_GROUP_SINGLE_ID_PREFIX = "tg-single-"
|
|
520
619
|
CANCEL_MESSAGE_ID_PREFIX = "msg-cancel-"
|
|
521
620
|
|
|
621
|
+
# Sentinel GatewayClient writes as an execution record's source_agent_type for
|
|
622
|
+
# a dispatch it made itself (client/client.py's initialize_execution and
|
|
623
|
+
# record_failed_route_decision). It is NOT an agent type: nothing declares it,
|
|
624
|
+
# so nothing consumes RedisKeys.ctrl_stream(CLIENT_SOURCE_AGENT_TYPE).
|
|
625
|
+
#
|
|
626
|
+
# Load-bearing wherever a resumed execution recovers its caller from its own
|
|
627
|
+
# record instead of from the resume header (GatewayWorker._resolve_reply_command
|
|
628
|
+
# / GatewayProcessor._resolve_reply_header): a root execution's record carries
|
|
629
|
+
# this, and treating it as a caller both posts the result to a stream no one
|
|
630
|
+
# reads and suppresses the end-of-stream event the session data plane owes the
|
|
631
|
+
# user — the visible half of the bug being prevented.
|
|
632
|
+
CLIENT_SOURCE_AGENT_TYPE = "client"
|
|
633
|
+
|
|
634
|
+
|
|
635
|
+
def single_call_task_group_id(child_message_id: str) -> str:
|
|
636
|
+
"""Group id under which a single (non-group) call_agent result is stored."""
|
|
637
|
+
return f"{TASK_GROUP_SINGLE_ID_PREFIX}{child_message_id}"
|
|
638
|
+
|
|
639
|
+
|
|
522
640
|
# --- Redis Hash Field Prefixes ---
|
|
523
641
|
# Field prefixes in Session Registry Hash
|
|
524
642
|
EXEC_FIELD_PREFIX = "exec:"
|
|
@@ -529,6 +647,10 @@ MSG_MAP_PREFIX = "msg_map:"
|
|
|
529
647
|
TASK_GROUP_FIELD_TOTAL = "total"
|
|
530
648
|
TASK_GROUP_FIELD_COMPLETED = "completed"
|
|
531
649
|
TASK_GROUP_FIELD_SOURCE_AGENT = "source_agent_type"
|
|
650
|
+
# Set once dispatch fails partway through a batch; any reply that arrives
|
|
651
|
+
# for an aborted group is discarded instead of resuming the (already
|
|
652
|
+
# terminated) caller execution.
|
|
653
|
+
TASK_GROUP_FIELD_ABORTED = "aborted"
|
|
532
654
|
|
|
533
655
|
|
|
534
656
|
# --- Timing and Sleep Constants ---
|
|
@@ -538,12 +660,133 @@ CONTROL_LOOP_SLEEP_SECONDS = 0.01
|
|
|
538
660
|
WAIT_FOR_TASKS_TIMEOUT_SECONDS = 5.0
|
|
539
661
|
# Task group Key TTL (seconds), default 1 day
|
|
540
662
|
TASK_GROUP_TTL_SECONDS = 86400
|
|
663
|
+
# Native agent harness loop-state Key TTL (seconds), default 1 day
|
|
664
|
+
HARNESS_STATE_TTL_SECONDS = 86400
|
|
541
665
|
# First retry wait time (seconds)
|
|
542
666
|
FIRST_RETRY_WAIT_SECONDS = 1.0
|
|
543
667
|
# Maximum retry count
|
|
544
668
|
MAX_RETRY_COUNT = 3
|
|
545
669
|
|
|
546
670
|
|
|
671
|
+
# --- Suspended-caller liveness (wait index) ---
|
|
672
|
+
# Number of RedisKeys.wait_index() shards. Fixed: changing it re-maps every
|
|
673
|
+
# session to a different shard, so in-flight entries would be swept by no
|
|
674
|
+
# one. Treat as a cross-SDK protocol constant, not a tunable.
|
|
675
|
+
WAIT_INDEX_SHARDS = 16
|
|
676
|
+
# Default deadline for a call_agent(wait_for_reply=True) reply (1 hour).
|
|
677
|
+
# Machine waiting on machine.
|
|
678
|
+
DEFAULT_REPLY_TIMEOUT_MS = 3_600_000
|
|
679
|
+
# Default deadline for an ask_user reply. Machine waiting on a human, so it
|
|
680
|
+
# is deliberately decoupled from DEFAULT_REPLY_TIMEOUT_MS and aligned with
|
|
681
|
+
# the session TTL (which is in seconds) instead.
|
|
682
|
+
DEFAULT_ASK_USER_TIMEOUT_MS = RedisKeys.DEFAULT_SESSION_TTL * 1000
|
|
683
|
+
# How often a worker's sweeper scans the shards it owns (seconds).
|
|
684
|
+
WAIT_SWEEP_INTERVAL_SECONDS = 30
|
|
685
|
+
# TTL of a RedisKeys.wait_sweep_lock() claim. Must comfortably exceed one
|
|
686
|
+
# shard's sweep so the owner doesn't lose the shard mid-pass, and stay short
|
|
687
|
+
# enough that a crashed sweeper's shards are picked up again quickly.
|
|
688
|
+
WAIT_SWEEP_LOCK_TTL_SECONDS = 60
|
|
689
|
+
# Most due entries one sweep resolves per shard per cycle. Bounds the work
|
|
690
|
+
# (and the Redis traffic) of a single pass after an outage leaves a large
|
|
691
|
+
# backlog; the remainder is simply picked up next cycle, since entries stay
|
|
692
|
+
# in the index until a reply clears them.
|
|
693
|
+
WAIT_SWEEP_BATCH_LIMIT = 200
|
|
694
|
+
# Fixed extension applied when a sweep finds the callee still making
|
|
695
|
+
# progress. Deliberately a constant rather than the original timeout: the
|
|
696
|
+
# wait-index member must stay reconstructible from a reply alone, so it
|
|
697
|
+
# cannot carry the caller's original timeout.
|
|
698
|
+
WAIT_RENEW_INCREMENT_MS = 300_000
|
|
699
|
+
# Hard ceiling on renewals, as a multiple of the caller's own timeout: a wait
|
|
700
|
+
# may be renewed until `registered_at + N * timeout`, after which the callee
|
|
701
|
+
# is declared CHILD_TIMEOUT even though its worker is still alive.
|
|
702
|
+
#
|
|
703
|
+
# Without a ceiling the "worker lease alive -> renew" rule is unconditional,
|
|
704
|
+
# so a callee that is running but making no progress (a hung LLM call, a
|
|
705
|
+
# deadlock) suspends its caller forever — the one failure mode the deadline
|
|
706
|
+
# was supposed to bound. N is deliberately expressed against the caller's
|
|
707
|
+
# timeout rather than a renewal count, so a caller that asked for 10 minutes
|
|
708
|
+
# is not held to the same absolute budget as one that asked for four hours,
|
|
709
|
+
# and so retuning WAIT_RENEW_INCREMENT_MS cannot silently change the bound.
|
|
710
|
+
#
|
|
711
|
+
# Why 3: N must exceed 1 (N == 1 is "never renew", which kills every callee
|
|
712
|
+
# that is merely slow); N == 2 leaves a single extra window, so one
|
|
713
|
+
# under-estimated timeout is enough to kill healthy work; N == 3 means the
|
|
714
|
+
# caller's own estimate has to be off by 200% before that happens, while
|
|
715
|
+
# still bounding the default case at 3 hours — two orders of magnitude below
|
|
716
|
+
# DEFAULT_SESSION_TTL, which matters because once the session data expires
|
|
717
|
+
# there is no execution record left to compensate against and the wait is
|
|
718
|
+
# simply dropped. Override per deployment via
|
|
719
|
+
# BY_FRAMEWORK_WAIT_RENEW_MAX_MULTIPLE.
|
|
720
|
+
WAIT_RENEW_MAX_MULTIPLE = 3
|
|
721
|
+
# TTL of RedisKeys.wait_renew_origin(). Must comfortably exceed the largest
|
|
722
|
+
# budget in use (N * timeout), or the budget silently restarts mid-wait.
|
|
723
|
+
WAIT_RENEW_ORIGIN_TTL_SECONDS = TASK_GROUP_TTL_SECONDS
|
|
724
|
+
# How long RedisKeys.wait_consumed() remembers that a wait was already
|
|
725
|
+
# resolved, i.e. how far apart two copies of the same reply may be and still
|
|
726
|
+
# be recognized as duplicates.
|
|
727
|
+
#
|
|
728
|
+
# Sized off DEFAULT_SESSION_TTL, which is the lifetime of the session
|
|
729
|
+
# registry — and the session registry is what keeps a *wait entry* relevant.
|
|
730
|
+
# A marker that expires while entries of that session are still live leaves
|
|
731
|
+
# two holes, and the second is the dangerous one:
|
|
732
|
+
#
|
|
733
|
+
# 1. A repeated ask_user answer (a human may take days; the ask_user
|
|
734
|
+
# deadline is DEFAULT_SESSION_TTL itself) is no longer recognized as a
|
|
735
|
+
# duplicate and wakes the caller a second time.
|
|
736
|
+
# 2. Worse: a stale duplicate sub-agent reply, having lost the marker that
|
|
737
|
+
# would stop it at its own candidate, falls through to the ask_user
|
|
738
|
+
# candidate for the same caller and claims a wait that is still live —
|
|
739
|
+
# after which the real answer is dropped as "already consumed".
|
|
740
|
+
#
|
|
741
|
+
# Both close once the marker outlives every wait it may have to arbitrate,
|
|
742
|
+
# i.e. the session TTL. Erring long costs a handful of idle 1-byte keys with
|
|
743
|
+
# the same lifetime as the session registry they belong to; erring short
|
|
744
|
+
# costs a lost user answer.
|
|
745
|
+
WAIT_CONSUMED_TTL_SECONDS = RedisKeys.DEFAULT_SESSION_TTL
|
|
746
|
+
# How often a sweeper prunes entries that are provably beyond use (see
|
|
747
|
+
# WAIT_PRUNE_AFTER_SECONDS). Deliberately far coarser than
|
|
748
|
+
# WAIT_SWEEP_INTERVAL_SECONDS: this is garbage collection on a multi-day
|
|
749
|
+
# horizon, and it is the only work a sweeper does when compensation is off.
|
|
750
|
+
WAIT_PRUNE_INTERVAL_SECONDS = 3600
|
|
751
|
+
# How far in the past a wait entry's score must lie before pruning it.
|
|
752
|
+
#
|
|
753
|
+
# Every writer of an entry sets its score to its own `now` plus a
|
|
754
|
+
# non-negative offset (registration adds the caller's timeout, a renewal adds
|
|
755
|
+
# WAIT_RENEW_INCREMENT_MS), and only ever does so while the caller's
|
|
756
|
+
# execution record exists. So `now - score > this` proves the entry was last
|
|
757
|
+
# touched more than a session TTL ago, hence that the session registry the
|
|
758
|
+
# sweep would interrogate has expired and no triage is possible any more:
|
|
759
|
+
# the entry can only ever produce "caller missing". Pruning it is therefore
|
|
760
|
+
# not a decision, which is why it needs no opt-in.
|
|
761
|
+
#
|
|
762
|
+
# The margin over DEFAULT_SESSION_TTL is what makes that strict rather than
|
|
763
|
+
# coincident: DEFAULT_ASK_USER_TIMEOUT_MS *equals* the session TTL, so a
|
|
764
|
+
# threshold trimmed to the session TTL exactly would land on the boundary of
|
|
765
|
+
# a live ask_user wait and lose to any clock skew between the worker that
|
|
766
|
+
# registered the entry and the one sweeping it. A day is far beyond plausible
|
|
767
|
+
# skew, and being late costs one ZSET member per unresolved call for one
|
|
768
|
+
# extra day — the asymmetry says err long.
|
|
769
|
+
WAIT_PRUNE_AFTER_SECONDS = RedisKeys.DEFAULT_SESSION_TTL + 86400
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
class LivenessErrorCode:
|
|
773
|
+
"""error_code values carried by synthesized/recovered resume replies.
|
|
774
|
+
|
|
775
|
+
Cross-SDK wire contract — Python/TS/Java must emit the same strings;
|
|
776
|
+
callers match on them. Append only, never rename.
|
|
777
|
+
"""
|
|
778
|
+
|
|
779
|
+
# The callee's worker lease expired while its execution was non-terminal.
|
|
780
|
+
CHILD_WORKER_LOST = "CHILD_WORKER_LOST"
|
|
781
|
+
# The callee was alive but produced no reply before the deadline.
|
|
782
|
+
CHILD_TIMEOUT = "CHILD_TIMEOUT"
|
|
783
|
+
# The dispatch was never picked up by any worker.
|
|
784
|
+
CHILD_NEVER_STARTED = "CHILD_NEVER_STARTED"
|
|
785
|
+
# The callee finished and its result was persisted, but the reply
|
|
786
|
+
# message was lost; the result was recovered from storage.
|
|
787
|
+
REPLY_LOST_RECOVERED = "REPLY_LOST_RECOVERED"
|
|
788
|
+
|
|
789
|
+
|
|
547
790
|
# --- Filesystem Constants ---
|
|
548
791
|
DEFAULT_WORKSPACE_DIR = "/workspace"
|
|
549
792
|
|
|
@@ -21,6 +21,10 @@ class EventType(str, Enum):
|
|
|
21
21
|
TASK_CREATE: Task creation event
|
|
22
22
|
STEP_COMPLETE: Step completion event
|
|
23
23
|
TASK_STOP: Task stop event
|
|
24
|
+
ORPHANED_REPLY: A reply that arrived for an already-resolved wait and
|
|
25
|
+
was therefore dropped by the idempotency gate. Diagnostic only —
|
|
26
|
+
the sub-agent did real work whose result nobody will consume, so
|
|
27
|
+
it must not vanish silently.
|
|
24
28
|
"""
|
|
25
29
|
|
|
26
30
|
ANSWER_DELTA = "answerDelta"
|
|
@@ -32,3 +36,4 @@ class EventType(str, Enum):
|
|
|
32
36
|
TASK_CREATE = "taskCreate"
|
|
33
37
|
STEP_COMPLETE = "stepComplete"
|
|
34
38
|
TASK_STOP = "taskStop"
|
|
39
|
+
ORPHANED_REPLY = "orphanedReply"
|
by_framework/core/registry.py
CHANGED
|
@@ -239,6 +239,33 @@ async def check_worker_online(
|
|
|
239
239
|
return is_legacy or last_seen > 0
|
|
240
240
|
|
|
241
241
|
|
|
242
|
+
async def acquire_scoped_lock(
|
|
243
|
+
redis: Redis,
|
|
244
|
+
key: str,
|
|
245
|
+
token: str,
|
|
246
|
+
ttl_seconds: int,
|
|
247
|
+
) -> bool:
|
|
248
|
+
"""Claim `key` for `token` if nobody holds it (Redlock acquire half).
|
|
249
|
+
|
|
250
|
+
The stored value must stay a cjson-decodable object carrying a "token"
|
|
251
|
+
field: `_REFRESH_LOCK_SCRIPT` / `_RELEASE_LOCK_SCRIPT` parse it that way,
|
|
252
|
+
and a bare token string would decode as unparseable legacy data, making
|
|
253
|
+
the holder unable to release its own lock.
|
|
254
|
+
"""
|
|
255
|
+
stored = await redis.set(
|
|
256
|
+
key,
|
|
257
|
+
json.dumps({"token": token}, separators=(",", ":")),
|
|
258
|
+
nx=True,
|
|
259
|
+
ex=ttl_seconds,
|
|
260
|
+
)
|
|
261
|
+
return bool(stored)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
async def release_scoped_lock(redis: Redis, key: str, token: str) -> bool:
|
|
265
|
+
"""Release a lock taken with acquire_scoped_lock(), if still owned."""
|
|
266
|
+
return bool(await redis.eval(_RELEASE_LOCK_SCRIPT, 1, key, token or ""))
|
|
267
|
+
|
|
268
|
+
|
|
242
269
|
async def check_agent_type_online(
|
|
243
270
|
redis: Redis,
|
|
244
271
|
agent_type: str,
|
|
@@ -1225,6 +1252,10 @@ class WorkerRegistry:
|
|
|
1225
1252
|
"active": self._get_int_hash_value(raw_counts, "active_count"),
|
|
1226
1253
|
"queued": self._get_int_hash_value(raw_counts, "queued_count"),
|
|
1227
1254
|
"running": self._get_int_hash_value(raw_counts, "running_count"),
|
|
1255
|
+
"waiting_agent": self._get_int_hash_value(
|
|
1256
|
+
raw_counts, "waiting_agent_count"
|
|
1257
|
+
),
|
|
1258
|
+
"waiting_user": self._get_int_hash_value(raw_counts, "waiting_user_count"),
|
|
1228
1259
|
"cancelling": self._get_int_hash_value(raw_counts, "cancelling_count"),
|
|
1229
1260
|
"completed": self._get_int_hash_value(raw_counts, "completed_count"),
|
|
1230
1261
|
"failed": self._get_int_hash_value(raw_counts, "failed_count"),
|
|
@@ -1245,6 +1276,10 @@ class WorkerRegistry:
|
|
|
1245
1276
|
status_counts = {
|
|
1246
1277
|
"QUEUED": counts["queued"],
|
|
1247
1278
|
"RUNNING": counts["running"],
|
|
1279
|
+
# Suspended callers persist as WAITING_* rather than QUEUED; these
|
|
1280
|
+
# rows keep them visible (zero-valued entries are filtered below).
|
|
1281
|
+
"WAITING_AGENT": counts["waiting_agent"],
|
|
1282
|
+
"WAITING_USER": counts["waiting_user"],
|
|
1248
1283
|
"CANCELLING": counts["cancelling"],
|
|
1249
1284
|
"COMPLETED": counts["completed"],
|
|
1250
1285
|
"FAILED": counts["failed"],
|