by-framework 0.2.2.dev9__py3-none-any.whl → 0.2.2.dev11__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- by_framework/__init__.py +2 -0
- by_framework/__main__.py +83 -4
- by_framework/admin/cli.py +6 -5
- by_framework/client/client.py +124 -3
- by_framework/common/config.py +4 -2
- by_framework/common/emitter.py +16 -0
- by_framework/core/__init__.py +2 -0
- by_framework/core/protocol/__init__.py +2 -0
- by_framework/core/protocol/content_type.py +1 -0
- by_framework/core/protocol/responses.py +11 -0
- by_framework/core/registry.py +2 -1
- by_framework/worker/app.py +68 -24
- by_framework/worker/health_server.py +169 -0
- by_framework/worker/runner.py +43 -0
- {by_framework-0.2.2.dev9.dist-info → by_framework-0.2.2.dev11.dist-info}/METADATA +26 -3
- {by_framework-0.2.2.dev9.dist-info → by_framework-0.2.2.dev11.dist-info}/RECORD +19 -18
- {by_framework-0.2.2.dev9.dist-info → by_framework-0.2.2.dev11.dist-info}/WHEEL +0 -0
- {by_framework-0.2.2.dev9.dist-info → by_framework-0.2.2.dev11.dist-info}/entry_points.txt +0 -0
- {by_framework-0.2.2.dev9.dist-info → by_framework-0.2.2.dev11.dist-info}/licenses/LICENSE +0 -0
by_framework/__init__.py
CHANGED
|
@@ -8,6 +8,7 @@ from `GatewayWorker` and running `run_worker`.
|
|
|
8
8
|
from .admin import WorkerManager
|
|
9
9
|
from .client.byai_client import ByaiGatewayClient
|
|
10
10
|
from .client.client import (
|
|
11
|
+
CancelSessionResponse,
|
|
11
12
|
CancelTaskResponse,
|
|
12
13
|
DataStreamEntry,
|
|
13
14
|
GatewayClient,
|
|
@@ -148,6 +149,7 @@ __all__ = [
|
|
|
148
149
|
"DataStreamEntry",
|
|
149
150
|
"SendMessageResponse",
|
|
150
151
|
"CancelTaskResponse",
|
|
152
|
+
"CancelSessionResponse",
|
|
151
153
|
"run_worker",
|
|
152
154
|
"logger",
|
|
153
155
|
"setup_logging",
|
by_framework/__main__.py
CHANGED
|
@@ -12,6 +12,17 @@ import importlib
|
|
|
12
12
|
from .worker.app import run_worker
|
|
13
13
|
|
|
14
14
|
|
|
15
|
+
def _parse_cluster_nodes(value: str):
|
|
16
|
+
"""Parse a comma-separated "host:port,host:port" string, mirroring
|
|
17
|
+
RedisConfig.from_env()'s REDIS_CLUSTER_HOST/REDIS_CLUSTER_NODES parsing
|
|
18
|
+
so the CLI and env-var paths accept the same format."""
|
|
19
|
+
nodes = []
|
|
20
|
+
for node in value.split(","):
|
|
21
|
+
node_host, node_port = node.rsplit(":", 1)
|
|
22
|
+
nodes.append((node_host, int(node_port)))
|
|
23
|
+
return nodes
|
|
24
|
+
|
|
25
|
+
|
|
15
26
|
def parse_args():
|
|
16
27
|
"""Parse command line arguments for the CLI runner."""
|
|
17
28
|
parser = argparse.ArgumentParser(description="By-Framework CLI Runner")
|
|
@@ -25,12 +36,55 @@ def parse_args():
|
|
|
25
36
|
),
|
|
26
37
|
)
|
|
27
38
|
parser.add_argument(
|
|
28
|
-
"--redis-host",
|
|
39
|
+
"--redis-host",
|
|
40
|
+
type=str,
|
|
41
|
+
default=None,
|
|
42
|
+
help="Redis server hostname (default: REDIS_HOST env var, then 'localhost')",
|
|
43
|
+
)
|
|
44
|
+
parser.add_argument(
|
|
45
|
+
"--redis-port",
|
|
46
|
+
type=int,
|
|
47
|
+
default=None,
|
|
48
|
+
help="Redis server port (default: REDIS_PORT env var, then 6379)",
|
|
49
|
+
)
|
|
50
|
+
parser.add_argument(
|
|
51
|
+
"--redis-db",
|
|
52
|
+
type=int,
|
|
53
|
+
default=None,
|
|
54
|
+
help="Redis database number (default: REDIS_DATABASE env var, then 0)",
|
|
29
55
|
)
|
|
30
56
|
parser.add_argument(
|
|
31
|
-
"--redis-
|
|
57
|
+
"--redis-password",
|
|
58
|
+
type=str,
|
|
59
|
+
default=None,
|
|
60
|
+
help="Redis password (default: REDIS_PASSWORD env var)",
|
|
61
|
+
)
|
|
62
|
+
parser.add_argument(
|
|
63
|
+
"--redis-username",
|
|
64
|
+
type=str,
|
|
65
|
+
default=None,
|
|
66
|
+
help="Redis username, for ACL-enabled Redis (default: REDIS_USERNAME env var)",
|
|
67
|
+
)
|
|
68
|
+
parser.add_argument(
|
|
69
|
+
"--redis-mode",
|
|
70
|
+
type=str,
|
|
71
|
+
choices=["standalone", "cluster"],
|
|
72
|
+
default=None,
|
|
73
|
+
help=(
|
|
74
|
+
"Redis deployment mode (default: REDIS_MODE env var; implied by "
|
|
75
|
+
"--redis-cluster-nodes/REDIS_CLUSTER_HOST alone if neither is set)"
|
|
76
|
+
),
|
|
77
|
+
)
|
|
78
|
+
parser.add_argument(
|
|
79
|
+
"--redis-cluster-nodes",
|
|
80
|
+
type=str,
|
|
81
|
+
default=None,
|
|
82
|
+
help=(
|
|
83
|
+
"Comma-separated Redis Cluster seed nodes, 'host:port,host:port' "
|
|
84
|
+
"(default: REDIS_CLUSTER_HOST/REDIS_CLUSTER_NODES env var). "
|
|
85
|
+
"Passing this alone implies --redis-mode cluster."
|
|
86
|
+
),
|
|
32
87
|
)
|
|
33
|
-
parser.add_argument("--redis-db", type=int, default=0, help="Redis database number")
|
|
34
88
|
parser.add_argument(
|
|
35
89
|
"--worker-id", type=str, default="worker-1", help="Unique worker identifier"
|
|
36
90
|
)
|
|
@@ -47,6 +101,17 @@ def parse_args():
|
|
|
47
101
|
parser.add_argument(
|
|
48
102
|
"--redis-max-connections", type=int, help="Max Redis connections allowed"
|
|
49
103
|
)
|
|
104
|
+
parser.add_argument(
|
|
105
|
+
"--health-port",
|
|
106
|
+
type=int,
|
|
107
|
+
default=None,
|
|
108
|
+
help=(
|
|
109
|
+
"Port for the local /readyz readiness endpoint (Docker/Kubernetes "
|
|
110
|
+
"health checks). Opt-in only - omit to leave it disabled, no "
|
|
111
|
+
"default port is provided. Also settable via BYAI_WORKER_HEALTH_PORT. "
|
|
112
|
+
"See docs/architecture/worker-readiness-endpoint.md."
|
|
113
|
+
),
|
|
114
|
+
)
|
|
50
115
|
|
|
51
116
|
return parser.parse_args()
|
|
52
117
|
|
|
@@ -68,17 +133,31 @@ def main():
|
|
|
68
133
|
|
|
69
134
|
print(f"Starting worker class: {worker_class.__name__} (ID: {args.worker_id})")
|
|
70
135
|
|
|
71
|
-
|
|
136
|
+
redis_cluster_nodes = (
|
|
137
|
+
_parse_cluster_nodes(args.redis_cluster_nodes)
|
|
138
|
+
if args.redis_cluster_nodes
|
|
139
|
+
else None
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
# Delegate to the common launcher. Every redis_* arg defaults to None
|
|
143
|
+
# here (not a literal like "localhost"), same as run_worker()'s own
|
|
144
|
+
# defaults, so an unset CLI flag actually falls through to the
|
|
145
|
+
# corresponding REDIS_* env var instead of silently shadowing it.
|
|
72
146
|
run_worker(
|
|
73
147
|
worker_class=worker_class,
|
|
74
148
|
worker_id=args.worker_id,
|
|
75
149
|
redis_host=args.redis_host,
|
|
76
150
|
redis_port=args.redis_port,
|
|
77
151
|
redis_db=args.redis_db,
|
|
152
|
+
redis_password=args.redis_password,
|
|
153
|
+
redis_username=args.redis_username,
|
|
154
|
+
redis_mode=args.redis_mode,
|
|
155
|
+
redis_cluster_nodes=redis_cluster_nodes,
|
|
78
156
|
workspace_dir=args.workspace,
|
|
79
157
|
max_concurrency=args.max_concurrency,
|
|
80
158
|
fetch_count=args.fetch_count,
|
|
81
159
|
redis_max_connections=args.redis_max_connections,
|
|
160
|
+
health_port=args.health_port,
|
|
82
161
|
)
|
|
83
162
|
|
|
84
163
|
|
by_framework/admin/cli.py
CHANGED
|
@@ -50,7 +50,6 @@ app.add_typer(metrics_app, name="metrics", help="Metrics and observability")
|
|
|
50
50
|
|
|
51
51
|
console = Console()
|
|
52
52
|
err_console = Console(stderr=True)
|
|
53
|
-
_DEFAULT_REDIS_URL = "redis://localhost:6379/0"
|
|
54
53
|
_redis_url: Optional[str] = None
|
|
55
54
|
|
|
56
55
|
|
|
@@ -85,10 +84,12 @@ def _get_redis(redis_url: Optional[str] = None):
|
|
|
85
84
|
configured_url = redis_url if redis_url is not None else _redis_url
|
|
86
85
|
if configured_url:
|
|
87
86
|
return init_redis_from_url(configured_url)
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
87
|
+
# Always resolve through RedisConfig.from_env() - for both standalone and
|
|
88
|
+
# cluster mode - so REDIS_HOST/PORT/PASSWORD/USERNAME/DATABASE are honored
|
|
89
|
+
# the same way run_worker()'s config resolution honors them. Previously
|
|
90
|
+
# only cluster mode read env_config; standalone mode silently ignored it
|
|
91
|
+
# and connected to a hardcoded localhost:6379/0.
|
|
92
|
+
return init_redis(config=RedisConfig.from_env())
|
|
92
93
|
|
|
93
94
|
|
|
94
95
|
# --------------------------------------------------------------------------- #
|
by_framework/client/client.py
CHANGED
|
@@ -37,6 +37,7 @@ from by_framework.core.protocol.commands import (
|
|
|
37
37
|
from by_framework.core.protocol.data_message import DataMessage
|
|
38
38
|
from by_framework.core.protocol.message_header import MessageHeader
|
|
39
39
|
from by_framework.core.protocol.responses import (
|
|
40
|
+
CancelSessionResponse,
|
|
40
41
|
CancelTaskResponse,
|
|
41
42
|
ExecutionStatus,
|
|
42
43
|
SendMessageResponse,
|
|
@@ -581,6 +582,99 @@ class GatewayClient:
|
|
|
581
582
|
cancelled_count=len(to_cancel),
|
|
582
583
|
)
|
|
583
584
|
|
|
585
|
+
async def cancel_session(
|
|
586
|
+
self,
|
|
587
|
+
session_id: str,
|
|
588
|
+
reason: str = "",
|
|
589
|
+
requested_by: str = "client",
|
|
590
|
+
cancel_mode: str = CancelMode.GRACEFUL,
|
|
591
|
+
) -> CancelSessionResponse:
|
|
592
|
+
"""Cancel every active execution registered under a session.
|
|
593
|
+
|
|
594
|
+
Unlike cancel_task, this does not require a starting message_id:
|
|
595
|
+
it cancels the whole session's active executions in one call.
|
|
596
|
+
"""
|
|
597
|
+
if self.registry is None:
|
|
598
|
+
raise ValueError("GatewayClient requires a WorkerRegistry to cancel tasks")
|
|
599
|
+
|
|
600
|
+
all_executions = await self.registry.get_all_session_executions(session_id)
|
|
601
|
+
|
|
602
|
+
if not all_executions:
|
|
603
|
+
return CancelSessionResponse(
|
|
604
|
+
success=False,
|
|
605
|
+
session_id=session_id,
|
|
606
|
+
status=ExecutionStatus.NOT_FOUND,
|
|
607
|
+
timestamp=int(time.time() * 1000),
|
|
608
|
+
cancelled_count=0,
|
|
609
|
+
already_finished_count=0,
|
|
610
|
+
)
|
|
611
|
+
|
|
612
|
+
terminal_states = {"COMPLETED", "FAILED", "CANCELLED"}
|
|
613
|
+
to_cancel = [
|
|
614
|
+
node
|
|
615
|
+
for node in all_executions
|
|
616
|
+
if node.get("status", "") not in terminal_states
|
|
617
|
+
]
|
|
618
|
+
terminal_nodes = [
|
|
619
|
+
node for node in all_executions if node.get("status", "") in terminal_states
|
|
620
|
+
]
|
|
621
|
+
|
|
622
|
+
# Flag terminal executions so a still-running descendant can't wake
|
|
623
|
+
# them back up via callback, without changing their status.
|
|
624
|
+
for node in terminal_nodes:
|
|
625
|
+
await self.registry.mark_cancel_requested(
|
|
626
|
+
node.get("execution_id", ""), session_id, reason
|
|
627
|
+
)
|
|
628
|
+
|
|
629
|
+
if not to_cancel:
|
|
630
|
+
return CancelSessionResponse(
|
|
631
|
+
success=False,
|
|
632
|
+
session_id=session_id,
|
|
633
|
+
status=ExecutionStatus.ALREADY_FINISHED,
|
|
634
|
+
timestamp=int(time.time() * 1000),
|
|
635
|
+
cancelled_count=0,
|
|
636
|
+
already_finished_count=len(terminal_nodes),
|
|
637
|
+
)
|
|
638
|
+
|
|
639
|
+
for node in to_cancel:
|
|
640
|
+
node_execution_id = node.get("execution_id", "")
|
|
641
|
+
node_worker_id = node.get("worker_id", "")
|
|
642
|
+
node_message_id = node.get("message_id", "")
|
|
643
|
+
|
|
644
|
+
await self.registry.mark_execution_cancelling(
|
|
645
|
+
node_execution_id, session_id, reason
|
|
646
|
+
)
|
|
647
|
+
|
|
648
|
+
if node_worker_id:
|
|
649
|
+
cancel_command = CancelTaskCommand(
|
|
650
|
+
header=MessageHeader(
|
|
651
|
+
message_id=f"{CANCEL_MESSAGE_ID_PREFIX}{uuid.uuid4().hex[:8]}",
|
|
652
|
+
session_id=session_id,
|
|
653
|
+
trace_id=node.get("trace_id") or uuid.uuid4().hex,
|
|
654
|
+
target_agent_type=node.get("target_agent_type", ""),
|
|
655
|
+
parent_message_id=node_message_id,
|
|
656
|
+
),
|
|
657
|
+
target_message_id=node_message_id,
|
|
658
|
+
target_execution_id=node_execution_id,
|
|
659
|
+
target_worker_id=node_worker_id,
|
|
660
|
+
reason=reason,
|
|
661
|
+
requested_by=requested_by,
|
|
662
|
+
cancel_mode=cancel_mode,
|
|
663
|
+
)
|
|
664
|
+
await self.redis.xadd(
|
|
665
|
+
RedisKeys.worker_ctrl_stream(node_worker_id),
|
|
666
|
+
cancel_command.to_redis_payload(),
|
|
667
|
+
)
|
|
668
|
+
|
|
669
|
+
return CancelSessionResponse(
|
|
670
|
+
success=True,
|
|
671
|
+
session_id=session_id,
|
|
672
|
+
status=ExecutionStatus.CANCEL_REQUESTED,
|
|
673
|
+
timestamp=int(time.time() * 1000),
|
|
674
|
+
cancelled_count=len(to_cancel),
|
|
675
|
+
already_finished_count=len(terminal_nodes),
|
|
676
|
+
)
|
|
677
|
+
|
|
584
678
|
async def send_message(
|
|
585
679
|
self,
|
|
586
680
|
target_agent_type: str,
|
|
@@ -691,7 +785,28 @@ class GatewayClient:
|
|
|
691
785
|
content=params["content"],
|
|
692
786
|
extra_payload=params["extra_payload"],
|
|
693
787
|
)
|
|
694
|
-
|
|
788
|
+
|
|
789
|
+
# A RESUME reuses the message_id of the original AskAgentCommand so the
|
|
790
|
+
# worker can look the suspended execution back up. Reuse its
|
|
791
|
+
# execution_id too, and skip re-initializing the registry record for
|
|
792
|
+
# it below -- initialize_execution() would otherwise overwrite the
|
|
793
|
+
# message_id -> execution_id mapping and detach this resume from the
|
|
794
|
+
# execution it's meant to continue.
|
|
795
|
+
resumed_execution = None
|
|
796
|
+
if (
|
|
797
|
+
params["action_type"] == ActionType.RESUME.value
|
|
798
|
+
and self.registry
|
|
799
|
+
and hasattr(self.registry, "get_execution_by_message_id")
|
|
800
|
+
):
|
|
801
|
+
resumed_execution = await self.registry.get_execution_by_message_id(
|
|
802
|
+
message_id, session_id=params["session_id"]
|
|
803
|
+
)
|
|
804
|
+
|
|
805
|
+
execution_id = (
|
|
806
|
+
resumed_execution["execution_id"]
|
|
807
|
+
if resumed_execution
|
|
808
|
+
else f"{EXECUTION_ID_PREFIX}{uuid.uuid4().hex[:8]}"
|
|
809
|
+
)
|
|
695
810
|
|
|
696
811
|
# 3. Resolve route and optionally probe agent type/liveness
|
|
697
812
|
should_dispatch_control = True
|
|
@@ -837,8 +952,14 @@ class GatewayClient:
|
|
|
837
952
|
error_code=ExecutionStatus.ERR_AGENT_TYPE_UNAVAILABLE,
|
|
838
953
|
)
|
|
839
954
|
|
|
840
|
-
# Initialize execution tracking
|
|
841
|
-
|
|
955
|
+
# Initialize execution tracking. Skipped for a RESUME that reattached
|
|
956
|
+
# to an existing execution -- that execution is already tracked, and
|
|
957
|
+
# re-initializing it here would overwrite its message_id mapping.
|
|
958
|
+
if (
|
|
959
|
+
not resumed_execution
|
|
960
|
+
and self.registry
|
|
961
|
+
and hasattr(self.registry, "initialize_execution")
|
|
962
|
+
):
|
|
842
963
|
try:
|
|
843
964
|
await self.registry.initialize_execution(
|
|
844
965
|
{
|
by_framework/common/config.py
CHANGED
|
@@ -43,6 +43,8 @@ class RedisConfig:
|
|
|
43
43
|
"""
|
|
44
44
|
password = os.environ.get("REDIS_PASSWORD", "")
|
|
45
45
|
username = os.environ.get("REDIS_USERNAME") or None
|
|
46
|
+
host = os.environ.get("REDIS_HOST") or "localhost"
|
|
47
|
+
port_str = os.environ.get("REDIS_PORT")
|
|
46
48
|
max_connections = os.environ.get("REDIS_MAX_CONNECTIONS")
|
|
47
49
|
cluster_host_str = os.environ.get("REDIS_CLUSTER_HOST")
|
|
48
50
|
cluster_nodes_str = cluster_host_str or os.environ.get("REDIS_CLUSTER_NODES")
|
|
@@ -61,8 +63,8 @@ class RedisConfig:
|
|
|
61
63
|
if db_str is not None:
|
|
62
64
|
logger.warning("REDIS_DB is deprecated, use REDIS_DATABASE instead")
|
|
63
65
|
return cls(
|
|
64
|
-
host=
|
|
65
|
-
port=int(
|
|
66
|
+
host=host,
|
|
67
|
+
port=int(port_str) if port_str else 6379,
|
|
66
68
|
db=int(db_str) if db_str is not None else 0,
|
|
67
69
|
password=password,
|
|
68
70
|
username=username,
|
by_framework/common/emitter.py
CHANGED
|
@@ -107,6 +107,22 @@ class DefaultSseLayoutBuilder(DataLayoutBuilder):
|
|
|
107
107
|
parent_order_id: Optional[str] = None,
|
|
108
108
|
) -> Dict[str, Any]:
|
|
109
109
|
"""Build default BaiYing-compatible user input form payload."""
|
|
110
|
+
try:
|
|
111
|
+
parsed_prompt = json.loads(prompt)
|
|
112
|
+
except (json.JSONDecodeError, TypeError):
|
|
113
|
+
parsed_prompt = None
|
|
114
|
+
|
|
115
|
+
# Structured question prompts are already in the format expected by clients.
|
|
116
|
+
if isinstance(parsed_prompt, dict) and "questions" in parsed_prompt:
|
|
117
|
+
return self.build(
|
|
118
|
+
content=prompt,
|
|
119
|
+
role="assistant",
|
|
120
|
+
content_type=SseReasonMessageType.ask_user_question.value,
|
|
121
|
+
source_agent_type=source_agent_type,
|
|
122
|
+
order_id=order_id,
|
|
123
|
+
parent_order_id=parent_order_id,
|
|
124
|
+
)
|
|
125
|
+
|
|
110
126
|
input_form = {
|
|
111
127
|
"formStatus": 0,
|
|
112
128
|
"pluginMachineFields": [
|
by_framework/core/__init__.py
CHANGED
|
@@ -26,6 +26,7 @@ from .protocol import (
|
|
|
26
26
|
BaiYingMessage,
|
|
27
27
|
BaiYingMessageRole,
|
|
28
28
|
BaseCommand,
|
|
29
|
+
CancelSessionResponse,
|
|
29
30
|
CancelTaskCommand,
|
|
30
31
|
CancelTaskResponse,
|
|
31
32
|
DataMessage,
|
|
@@ -53,6 +54,7 @@ from .workspace import WorkspaceManager
|
|
|
53
54
|
__all__ = [
|
|
54
55
|
"SendMessageResponse",
|
|
55
56
|
"CancelTaskResponse",
|
|
57
|
+
"CancelSessionResponse",
|
|
56
58
|
"BaiYingMessage",
|
|
57
59
|
"BaiYingMessageRole",
|
|
58
60
|
"MessageContent",
|
|
@@ -65,6 +65,7 @@ from .message import (
|
|
|
65
65
|
)
|
|
66
66
|
from .message_header import MessageHeader
|
|
67
67
|
from .responses import (
|
|
68
|
+
CancelSessionResponse,
|
|
68
69
|
CancelTaskResponse,
|
|
69
70
|
CancelTaskResponseDict,
|
|
70
71
|
SendMessageResponse,
|
|
@@ -81,6 +82,7 @@ from .results import (
|
|
|
81
82
|
__all__ = [
|
|
82
83
|
"SendMessageResponse",
|
|
83
84
|
"CancelTaskResponse",
|
|
85
|
+
"CancelSessionResponse",
|
|
84
86
|
"SendMessageResponseDict",
|
|
85
87
|
"CancelTaskResponseDict",
|
|
86
88
|
"AgentTaskResult",
|
|
@@ -33,6 +33,7 @@ class SseReasonMessageType(str, Enum):
|
|
|
33
33
|
think_code_result = "3007" # thinking process code execution result
|
|
34
34
|
task_finished = "3009" # task finished
|
|
35
35
|
task_user_input = "3013" # user input
|
|
36
|
+
ask_user_question = "3014" # ask user question
|
|
36
37
|
task_create_file = "3010" # create file
|
|
37
38
|
task_title = "3011" # task title
|
|
38
39
|
agent_card = "2015" # agent card
|
|
@@ -92,3 +92,14 @@ class CancelTaskResponse:
|
|
|
92
92
|
timestamp: int
|
|
93
93
|
error: str = ""
|
|
94
94
|
cancelled_count: int = 0
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@dataclass(frozen=True)
|
|
98
|
+
class CancelSessionResponse:
|
|
99
|
+
success: bool
|
|
100
|
+
session_id: str
|
|
101
|
+
status: str
|
|
102
|
+
timestamp: int
|
|
103
|
+
cancelled_count: int = 0
|
|
104
|
+
already_finished_count: int = 0
|
|
105
|
+
error: str = ""
|
by_framework/core/registry.py
CHANGED
|
@@ -1011,7 +1011,8 @@ class WorkerRegistry:
|
|
|
1011
1011
|
old_execution = dict(current)
|
|
1012
1012
|
current["status"] = status
|
|
1013
1013
|
now = int(time.time() * 1000)
|
|
1014
|
-
|
|
1014
|
+
if is_terminal_state(status):
|
|
1015
|
+
current["finished_at"] = now
|
|
1015
1016
|
current["updated_at"] = now
|
|
1016
1017
|
if completion:
|
|
1017
1018
|
for key, value in completion.items():
|
by_framework/worker/app.py
CHANGED
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
import asyncio
|
|
4
4
|
import inspect
|
|
5
5
|
import pkgutil
|
|
6
|
+
import signal
|
|
6
7
|
from dataclasses import replace
|
|
7
8
|
from importlib import import_module
|
|
8
9
|
from types import ModuleType
|
|
@@ -108,6 +109,7 @@ async def _run_worker_async(
|
|
|
108
109
|
layout_builder: Optional[DataLayoutBuilder] = None,
|
|
109
110
|
redis_mode: Optional[Literal["standalone", "cluster"]] = None,
|
|
110
111
|
redis_cluster_nodes: Optional[List[Tuple[str, int]]] = None,
|
|
112
|
+
health_port: Optional[int] = None,
|
|
111
113
|
**worker_kwargs,
|
|
112
114
|
):
|
|
113
115
|
"""Async worker runner initialization."""
|
|
@@ -243,6 +245,7 @@ async def _run_worker_async(
|
|
|
243
245
|
group_name=consumer_group,
|
|
244
246
|
max_concurrency=max_concurrency,
|
|
245
247
|
fetch_count=fetch_count,
|
|
248
|
+
health_port=health_port,
|
|
246
249
|
)
|
|
247
250
|
|
|
248
251
|
try:
|
|
@@ -253,6 +256,36 @@ async def _run_worker_async(
|
|
|
253
256
|
await close_redis()
|
|
254
257
|
|
|
255
258
|
|
|
259
|
+
async def _run_with_graceful_shutdown(coro: Awaitable) -> None:
|
|
260
|
+
"""Drive `coro` to completion, cancelling it on SIGTERM/SIGINT so the
|
|
261
|
+
caller's own try/finally (WorkerRunner._shutdown's drain-in-flight-tasks
|
|
262
|
+
sequence) runs. Without this, only Ctrl+C (SIGINT, raised by the
|
|
263
|
+
interpreter as KeyboardInterrupt) triggered graceful shutdown -
|
|
264
|
+
`docker stop`/Kubernetes pod termination send SIGTERM by default, which
|
|
265
|
+
Python's default handler turns into an immediate, non-graceful exit."""
|
|
266
|
+
loop = asyncio.get_running_loop()
|
|
267
|
+
task = asyncio.ensure_future(coro)
|
|
268
|
+
|
|
269
|
+
def _request_shutdown(sig: signal.Signals) -> None:
|
|
270
|
+
logger.info("Received %s, shutting down gracefully...", sig.name)
|
|
271
|
+
task.cancel()
|
|
272
|
+
|
|
273
|
+
for sig in (signal.SIGTERM, signal.SIGINT):
|
|
274
|
+
try:
|
|
275
|
+
loop.add_signal_handler(sig, _request_shutdown, sig)
|
|
276
|
+
except NotImplementedError:
|
|
277
|
+
# e.g. Windows' default event loop - falls back to the
|
|
278
|
+
# KeyboardInterrupt handling in run_worker() for SIGINT only.
|
|
279
|
+
logger.debug(
|
|
280
|
+
"Signal handler for %s not supported on this platform", sig.name
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
try:
|
|
284
|
+
await task
|
|
285
|
+
except asyncio.CancelledError:
|
|
286
|
+
pass
|
|
287
|
+
|
|
288
|
+
|
|
256
289
|
def run_worker(
|
|
257
290
|
worker_class: Type[GatewayWorker],
|
|
258
291
|
worker_id: str = "worker-1",
|
|
@@ -278,6 +311,7 @@ def run_worker(
|
|
|
278
311
|
layout_builder: Optional[DataLayoutBuilder] = None,
|
|
279
312
|
redis_mode: Optional[Literal["standalone", "cluster"]] = None,
|
|
280
313
|
redis_cluster_nodes: Optional[List[Tuple[str, int]]] = None,
|
|
314
|
+
health_port: Optional[int] = None,
|
|
281
315
|
**worker_kwargs,
|
|
282
316
|
):
|
|
283
317
|
"""
|
|
@@ -323,33 +357,43 @@ def run_worker(
|
|
|
323
357
|
max_concurrency = int(os.environ.get("BYAI_WORKER_CONCURRENCY", 50))
|
|
324
358
|
if fetch_count is None:
|
|
325
359
|
fetch_count = int(os.environ.get("BYAI_WORKER_FETCH_COUNT", 10))
|
|
360
|
+
if health_port is None:
|
|
361
|
+
# Opt-in only, unlike the two above - no default port number, stays
|
|
362
|
+
# None (disabled) unless explicitly set. See
|
|
363
|
+
# docs/architecture/worker-readiness-endpoint.md Decision 4.
|
|
364
|
+
env_health_port = os.environ.get("BYAI_WORKER_HEALTH_PORT")
|
|
365
|
+
if env_health_port:
|
|
366
|
+
health_port = int(env_health_port)
|
|
326
367
|
|
|
327
368
|
try:
|
|
328
369
|
asyncio.run(
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
370
|
+
_run_with_graceful_shutdown(
|
|
371
|
+
_run_worker_async(
|
|
372
|
+
worker_class,
|
|
373
|
+
worker_id,
|
|
374
|
+
redis_host,
|
|
375
|
+
redis_port,
|
|
376
|
+
redis_db,
|
|
377
|
+
redis_password,
|
|
378
|
+
redis_username,
|
|
379
|
+
workspace_dir,
|
|
380
|
+
consumer_group,
|
|
381
|
+
max_concurrency=max_concurrency,
|
|
382
|
+
fetch_count=fetch_count,
|
|
383
|
+
redis_max_connections=redis_max_connections,
|
|
384
|
+
plugin_list=plugin_list,
|
|
385
|
+
plugin_configurator=plugin_configurator,
|
|
386
|
+
plugin_hook_timeout_seconds=plugin_hook_timeout_seconds,
|
|
387
|
+
plugin_log_hook_stats_on_shutdown=plugin_log_hook_stats_on_shutdown,
|
|
388
|
+
history_backend=history_backend,
|
|
389
|
+
plugin_dir=plugin_dir,
|
|
390
|
+
storage=storage,
|
|
391
|
+
layout_builder=layout_builder,
|
|
392
|
+
redis_mode=redis_mode,
|
|
393
|
+
redis_cluster_nodes=redis_cluster_nodes,
|
|
394
|
+
health_port=health_port,
|
|
395
|
+
**worker_kwargs,
|
|
396
|
+
)
|
|
353
397
|
)
|
|
354
398
|
)
|
|
355
399
|
except KeyboardInterrupt:
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Worker readiness HTTP endpoint for Docker/Kubernetes health checks.
|
|
2
|
+
|
|
3
|
+
See docs/architecture/worker-readiness-endpoint.md for the full design
|
|
4
|
+
record (why readiness-only, why a dedicated thread, why /readyz, the
|
|
5
|
+
reason-priority order, and the hard rule against ever wiring this to a
|
|
6
|
+
liveness probe).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import threading
|
|
11
|
+
import time
|
|
12
|
+
from http import HTTPStatus
|
|
13
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
14
|
+
from typing import Callable, Optional
|
|
15
|
+
from urllib.parse import urlparse
|
|
16
|
+
|
|
17
|
+
from by_framework.common.logger import logger
|
|
18
|
+
|
|
19
|
+
READYZ_PATH = "/readyz"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _make_readyz_handler(
|
|
23
|
+
compute_state: Callable[[], dict],
|
|
24
|
+
) -> type[BaseHTTPRequestHandler]:
|
|
25
|
+
"""Build a request handler bound to the given state computation."""
|
|
26
|
+
|
|
27
|
+
class ReadyzHandler(BaseHTTPRequestHandler):
|
|
28
|
+
"""Serves /readyz; every other path is 404."""
|
|
29
|
+
|
|
30
|
+
server_version = "ByFrameworkWorkerReadiness/0.1"
|
|
31
|
+
|
|
32
|
+
def do_GET(self) -> None: # pylint: disable=invalid-name
|
|
33
|
+
if urlparse(self.path).path != READYZ_PATH:
|
|
34
|
+
self.send_response(HTTPStatus.NOT_FOUND)
|
|
35
|
+
self.end_headers()
|
|
36
|
+
return
|
|
37
|
+
|
|
38
|
+
state = compute_state()
|
|
39
|
+
status = HTTPStatus.OK if state["ready"] else HTTPStatus.SERVICE_UNAVAILABLE
|
|
40
|
+
body = json.dumps(state).encode("utf-8")
|
|
41
|
+
self.send_response(status)
|
|
42
|
+
self.send_header("Content-Type", "application/json; charset=utf-8")
|
|
43
|
+
self.send_header("Content-Length", str(len(body)))
|
|
44
|
+
self.end_headers()
|
|
45
|
+
self.wfile.write(body)
|
|
46
|
+
|
|
47
|
+
def log_message(self, format: str, *args) -> None: # pylint: disable=redefined-builtin
|
|
48
|
+
# The stdlib default logs every request to stderr - a probe
|
|
49
|
+
# hitting this every few seconds would spam Worker logs.
|
|
50
|
+
pass
|
|
51
|
+
|
|
52
|
+
return ReadyzHandler
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class WorkerHealthServer:
|
|
56
|
+
"""Runs /readyz on a dedicated thread, never the Worker's main event loop.
|
|
57
|
+
|
|
58
|
+
Mirrors WorkerHeartbeat's "dedicated thread" pattern for the same
|
|
59
|
+
reason: a busy consume loop must not make readiness checks
|
|
60
|
+
unreachable. Reads Worker state via plain callables passed in by the
|
|
61
|
+
caller (WorkerRunner) - this class has no knowledge of WorkerRunner
|
|
62
|
+
itself, so it can be started standalone against fake state in tests.
|
|
63
|
+
"""
|
|
64
|
+
|
|
65
|
+
def __init__(
|
|
66
|
+
self,
|
|
67
|
+
worker_id: str,
|
|
68
|
+
port: int,
|
|
69
|
+
has_started: Callable[[], bool],
|
|
70
|
+
is_draining: Callable[[], bool],
|
|
71
|
+
admin_lifecycle: Callable[[], str],
|
|
72
|
+
consumer_healthy: Callable[[], bool],
|
|
73
|
+
host: str = "0.0.0.0",
|
|
74
|
+
):
|
|
75
|
+
self.worker_id = worker_id
|
|
76
|
+
self.host = host
|
|
77
|
+
self.port = port
|
|
78
|
+
self._has_started = has_started
|
|
79
|
+
self._is_draining = is_draining
|
|
80
|
+
self._admin_lifecycle = admin_lifecycle
|
|
81
|
+
self._consumer_healthy = consumer_healthy
|
|
82
|
+
self._server: Optional[ThreadingHTTPServer] = None
|
|
83
|
+
self._thread: Optional[threading.Thread] = None
|
|
84
|
+
self._started_at_monotonic = 0.0
|
|
85
|
+
|
|
86
|
+
@staticmethod
|
|
87
|
+
def _compute_reason(
|
|
88
|
+
has_started: bool,
|
|
89
|
+
is_draining: bool,
|
|
90
|
+
admin_lifecycle: str,
|
|
91
|
+
consumer_healthy: bool,
|
|
92
|
+
) -> str:
|
|
93
|
+
# Priority order, first match wins - see
|
|
94
|
+
# docs/architecture/worker-readiness-endpoint.md.
|
|
95
|
+
if not has_started:
|
|
96
|
+
return "starting"
|
|
97
|
+
if is_draining:
|
|
98
|
+
return "draining"
|
|
99
|
+
if admin_lifecycle == "evicted":
|
|
100
|
+
return "evicted"
|
|
101
|
+
if admin_lifecycle == "suspended":
|
|
102
|
+
return "suspended"
|
|
103
|
+
if not consumer_healthy:
|
|
104
|
+
return "consumer_stalled"
|
|
105
|
+
return "serving"
|
|
106
|
+
|
|
107
|
+
def _compute_state(self) -> dict:
|
|
108
|
+
# Read each accessor once - _compute_reason() takes the values
|
|
109
|
+
# rather than re-deriving them, so a request never calls into
|
|
110
|
+
# WorkerRunner's state twice for the same field.
|
|
111
|
+
admin_lifecycle = self._admin_lifecycle()
|
|
112
|
+
consumer_healthy = self._consumer_healthy()
|
|
113
|
+
reason = self._compute_reason(
|
|
114
|
+
has_started=self._has_started(),
|
|
115
|
+
is_draining=self._is_draining(),
|
|
116
|
+
admin_lifecycle=admin_lifecycle,
|
|
117
|
+
consumer_healthy=consumer_healthy,
|
|
118
|
+
)
|
|
119
|
+
uptime_ms = 0
|
|
120
|
+
if self._started_at_monotonic:
|
|
121
|
+
uptime_ms = int((time.monotonic() - self._started_at_monotonic) * 1000)
|
|
122
|
+
return {
|
|
123
|
+
"ready": reason == "serving",
|
|
124
|
+
"reason": reason,
|
|
125
|
+
"worker_id": self.worker_id,
|
|
126
|
+
"admin_lifecycle": admin_lifecycle,
|
|
127
|
+
"consumer_healthy": consumer_healthy,
|
|
128
|
+
"uptime_ms": uptime_ms,
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
@property
|
|
132
|
+
def is_running(self) -> bool:
|
|
133
|
+
"""Whether the server has actually bound and is serving - distinct
|
|
134
|
+
from the WorkerHealthServer object merely having been constructed."""
|
|
135
|
+
return self._server is not None
|
|
136
|
+
|
|
137
|
+
def start(self) -> None:
|
|
138
|
+
"""Bind and start serving /readyz on a dedicated thread. No-op if
|
|
139
|
+
already started."""
|
|
140
|
+
if self._server is not None:
|
|
141
|
+
return
|
|
142
|
+
self._started_at_monotonic = time.monotonic()
|
|
143
|
+
handler = _make_readyz_handler(self._compute_state)
|
|
144
|
+
self._server = ThreadingHTTPServer((self.host, self.port), handler)
|
|
145
|
+
self.port = self._server.server_address[1]
|
|
146
|
+
self._thread = threading.Thread(
|
|
147
|
+
target=self._server.serve_forever,
|
|
148
|
+
daemon=True,
|
|
149
|
+
name=f"health-server-{self.worker_id}",
|
|
150
|
+
)
|
|
151
|
+
self._thread.start()
|
|
152
|
+
logger.info(
|
|
153
|
+
"[%s] Readiness endpoint listening on %s:%d%s",
|
|
154
|
+
self.worker_id,
|
|
155
|
+
self.host,
|
|
156
|
+
self.port,
|
|
157
|
+
READYZ_PATH,
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
def stop(self) -> None:
|
|
161
|
+
"""Stop serving and release the port. Safe to call more than once."""
|
|
162
|
+
if self._server is None:
|
|
163
|
+
return
|
|
164
|
+
self._server.shutdown()
|
|
165
|
+
self._server.server_close()
|
|
166
|
+
if self._thread is not None:
|
|
167
|
+
self._thread.join(timeout=5)
|
|
168
|
+
self._server = None
|
|
169
|
+
self._thread = None
|
by_framework/worker/runner.py
CHANGED
|
@@ -41,6 +41,7 @@ from by_framework.trace.trace_schema import TraceRecord
|
|
|
41
41
|
from by_framework.trace.trace_writer import TraceWriteClient
|
|
42
42
|
from by_framework.util.generate_message_id import generate_message_id
|
|
43
43
|
from by_framework.worker.context import current_worker_id_var
|
|
44
|
+
from by_framework.worker.health_server import WorkerHealthServer
|
|
44
45
|
from by_framework.worker.worker import GatewayWorker
|
|
45
46
|
|
|
46
47
|
from ._control_handling import (
|
|
@@ -69,6 +70,7 @@ class WorkerRunner:
|
|
|
69
70
|
max_concurrency: int = 50,
|
|
70
71
|
fetch_count: int = 10,
|
|
71
72
|
span_recorder: Optional[SpanRecorder] = None,
|
|
73
|
+
health_port: Optional[int] = None,
|
|
72
74
|
):
|
|
73
75
|
if (
|
|
74
76
|
worker is None
|
|
@@ -117,6 +119,22 @@ class WorkerRunner:
|
|
|
117
119
|
float(self.worker.heartbeat_lease_ttl_seconds) * 2.0,
|
|
118
120
|
stream_block_seconds * 3.0,
|
|
119
121
|
)
|
|
122
|
+
# Set as the first step of _shutdown() - see
|
|
123
|
+
# docs/architecture/worker-readiness-endpoint.md Decision 6.
|
|
124
|
+
self._draining: bool = False
|
|
125
|
+
# Opt-in only - see docs/architecture/worker-readiness-endpoint.md.
|
|
126
|
+
# None (the default) means no port opens and nothing here changes
|
|
127
|
+
# behavior for existing deployments.
|
|
128
|
+
self._health_server: Optional[WorkerHealthServer] = None
|
|
129
|
+
if health_port is not None:
|
|
130
|
+
self._health_server = WorkerHealthServer(
|
|
131
|
+
worker_id=self.worker.worker_id,
|
|
132
|
+
port=health_port,
|
|
133
|
+
has_started=lambda: self._consumer_last_tick_monotonic > 0,
|
|
134
|
+
is_draining=lambda: self._draining,
|
|
135
|
+
admin_lifecycle=lambda: self._admin_lifecycle,
|
|
136
|
+
consumer_healthy=self._is_consumer_healthy,
|
|
137
|
+
)
|
|
120
138
|
|
|
121
139
|
@property
|
|
122
140
|
def _terminal_execution_states(self) -> frozenset[str]:
|
|
@@ -505,6 +523,16 @@ class WorkerRunner:
|
|
|
505
523
|
header.message_id, session_id=header.session_id
|
|
506
524
|
)
|
|
507
525
|
|
|
526
|
+
if isinstance(command, ResumeCommand) and existing_execution is None:
|
|
527
|
+
logger.warning(
|
|
528
|
+
"[%s] ResumeCommand did not resolve to an existing execution "
|
|
529
|
+
"(message_id=%s, session_id=%s); starting a new, disconnected "
|
|
530
|
+
"execution instead of continuing the suspended one.",
|
|
531
|
+
self.worker.worker_id,
|
|
532
|
+
header.message_id,
|
|
533
|
+
header.session_id,
|
|
534
|
+
)
|
|
535
|
+
|
|
508
536
|
# Skip terminal state replays
|
|
509
537
|
if (
|
|
510
538
|
existing_execution
|
|
@@ -906,6 +934,12 @@ class WorkerRunner:
|
|
|
906
934
|
"""Start the worker runner main loop."""
|
|
907
935
|
self._running_tasks = set()
|
|
908
936
|
try:
|
|
937
|
+
# As early as possible, so a probe hitting the port during
|
|
938
|
+
# startup gets an honest "starting" 503 instead of connection-
|
|
939
|
+
# refused - see docs/architecture/worker-readiness-endpoint.md.
|
|
940
|
+
if self._health_server is not None:
|
|
941
|
+
self._health_server.start()
|
|
942
|
+
|
|
909
943
|
if hasattr(self.worker.registry, "claim_worker_id"):
|
|
910
944
|
self._lock_token = await self._claim_worker_id_with_retry()
|
|
911
945
|
|
|
@@ -974,6 +1008,10 @@ class WorkerRunner:
|
|
|
974
1008
|
|
|
975
1009
|
async def _shutdown(self):
|
|
976
1010
|
"""Graceful shutdown sequence."""
|
|
1011
|
+
# First step, deliberately - readiness must flip the instant
|
|
1012
|
+
# shutdown begins, not once the drain below finishes. See
|
|
1013
|
+
# docs/architecture/worker-readiness-endpoint.md Decision 6.
|
|
1014
|
+
self._draining = True
|
|
977
1015
|
if self._metrics_collector_task:
|
|
978
1016
|
self._metrics_collector_task.cancel()
|
|
979
1017
|
await asyncio.gather(self._metrics_collector_task, return_exceptions=True)
|
|
@@ -1019,3 +1057,8 @@ class WorkerRunner:
|
|
|
1019
1057
|
if getattr(plugin_registry, "log_hook_stats_on_shutdown", True):
|
|
1020
1058
|
plugin_registry.log_hook_stats()
|
|
1021
1059
|
await plugin_registry.on_worker_shutdown(self.worker)
|
|
1060
|
+
|
|
1061
|
+
# Stopped last, deliberately - the readiness endpoint should keep
|
|
1062
|
+
# responding for the entire drain, not disappear before it.
|
|
1063
|
+
if self._health_server is not None:
|
|
1064
|
+
self._health_server.stop()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: by-framework
|
|
3
|
-
Version: 0.2.2.
|
|
3
|
+
Version: 0.2.2.dev11
|
|
4
4
|
Summary: 分布式 Agent 调度框架
|
|
5
5
|
License-File: LICENSE
|
|
6
6
|
Requires-Python: >=3.12
|
|
@@ -250,6 +250,13 @@ async def main():
|
|
|
250
250
|
asyncio.run(main())
|
|
251
251
|
```
|
|
252
252
|
|
|
253
|
+
A runnable version of this pair lives at
|
|
254
|
+
[`examples/echo_worker.py`](examples/echo_worker.py) (Worker) and
|
|
255
|
+
[`examples/send_and_verify.py`](examples/send_and_verify.py) (client — blocks
|
|
256
|
+
until it sees the echoed reply) — the same pair `deploy/`'s CI smoke test
|
|
257
|
+
drives end to end, see
|
|
258
|
+
[`docs/architecture/production-deployment.md`](docs/architecture/production-deployment.md).
|
|
259
|
+
|
|
253
260
|
---
|
|
254
261
|
|
|
255
262
|
## Core Concepts
|
|
@@ -602,7 +609,7 @@ Tests are organized by module under `tests/`:
|
|
|
602
609
|
|
|
603
610
|
```bash
|
|
604
611
|
# 1. Start Redis
|
|
605
|
-
docker run -d --name by-redis -p 6379:6379
|
|
612
|
+
docker run -d --name by-redis -p 6379:6379 redis:7-alpine
|
|
606
613
|
|
|
607
614
|
# 2. Start a Worker
|
|
608
615
|
python -m by_framework \
|
|
@@ -621,11 +628,27 @@ python -m by_framework --worker-class my_agent.MyAgent --worker-id worker-02 &
|
|
|
621
628
|
python -m by_framework --worker-class my_agent.MyAgent --worker-id worker-03 &
|
|
622
629
|
```
|
|
623
630
|
|
|
631
|
+
`&` is fine for a quick local check but isn't real process supervision — for
|
|
632
|
+
an actual deployment, use the reference Dockerfile, Compose file, and
|
|
633
|
+
Kubernetes Deployment under [`deploy/`](deploy/):
|
|
634
|
+
|
|
635
|
+
```bash
|
|
636
|
+
# Docker Compose — N supervised replicas, distinct worker ids, restart policy
|
|
637
|
+
docker compose -f deploy/docker-compose.yml up --build --scale worker=3
|
|
638
|
+
|
|
639
|
+
# Kubernetes — same idea as a Deployment
|
|
640
|
+
kubectl apply -f deploy/kubernetes/worker-deployment.yaml
|
|
641
|
+
```
|
|
642
|
+
|
|
643
|
+
The CLI also exposes `--redis-password`, `--redis-username`, `--redis-mode`,
|
|
644
|
+
and `--redis-cluster-nodes` (all mirrored by the `REDIS_*` env vars used in
|
|
645
|
+
the Compose/Kubernetes examples above) — see `python -m by_framework --help`.
|
|
646
|
+
|
|
624
647
|
### Reliability
|
|
625
648
|
|
|
626
649
|
- **Message persistence:** Messages are stored in Redis Streams until explicitly acknowledged (`XACK`). Unacknowledged messages are redelivered on Worker restart.
|
|
627
650
|
- **Durable config:** Agent config snapshots are persisted to Redis, so a restarted Worker recovers the last-known plugin configuration.
|
|
628
|
-
- **
|
|
651
|
+
- **Graceful shutdown:** `WorkerRunner` drains in-flight tasks before shutting down, acknowledging completed work. This is wired to both `SIGINT` (Ctrl+C) and `SIGTERM` — the signal `docker stop`/Kubernetes pod termination actually send — so it fires under real container orchestration, not just an interactive terminal. Give it enough `stop_grace_period`/`terminationGracePeriodSeconds` to cover your longest task (see the `deploy/` examples), since a hard `SIGKILL` after the grace period skips the drain entirely.
|
|
629
652
|
- **Separate data path:** Data output goes to session-scoped streams independently of control, so backend consumers are decoupled from Worker scaling.
|
|
630
653
|
|
|
631
654
|
### Logging
|
|
@@ -1,23 +1,23 @@
|
|
|
1
|
-
by_framework/__init__.py,sha256=
|
|
2
|
-
by_framework/__main__.py,sha256=
|
|
1
|
+
by_framework/__init__.py,sha256=aNq4eeO-5WfFC86JJlfDgCvI11vk90ZWtL3yaREcJl4,5356
|
|
2
|
+
by_framework/__main__.py,sha256=P2smZtTse__MOhvJXBo4YZG1Rs2AAjX4Xk5aGrZgnAY,5237
|
|
3
3
|
by_framework/admin/__init__.py,sha256=MKGQBtKk8YdnMrZ5gkzEemMl4iL8OfeFhEV6HdWR2XI,138
|
|
4
|
-
by_framework/admin/cli.py,sha256=
|
|
4
|
+
by_framework/admin/cli.py,sha256=YW4tQxb0kBheRNQR53zOkoAlYHWP4bgycuZkaJIU3WA,11917
|
|
5
5
|
by_framework/admin/worker_manager.py,sha256=ihXRMlFJAR74T2JSh3xTNCDjzAElWd0-_yVDAZwSTJU,5924
|
|
6
6
|
by_framework/client/__init__.py,sha256=kGP7uoXIlTJuiJ2Cgl1Q5wqYjnMA4SZqoa7Q7BhQ5JQ,272
|
|
7
7
|
by_framework/client/byai_client.py,sha256=bB1NyKLvVW82bXefD_3zymqZ1vk2O925CIbxOho9Qww,2426
|
|
8
|
-
by_framework/client/client.py,sha256=
|
|
8
|
+
by_framework/client/client.py,sha256=hT_IRR51jsu3MIPFbZDmmL9xQHtNUFJCADBep9iFyQI,45157
|
|
9
9
|
by_framework/common/__init__.py,sha256=EkfIiVdmb8SCpi7OOUm6Hu84B7Wzekb-TG2hbvD_R1w,1685
|
|
10
|
-
by_framework/common/config.py,sha256=
|
|
10
|
+
by_framework/common/config.py,sha256=mTbCn3YspY8k00b4URSbi07MNC3Cfdj7RPtDLrXrJV8,5107
|
|
11
11
|
by_framework/common/constants.py,sha256=xUp6Nvx31EH7EzkxDNBzXD0GIE9FpPnPmuVbL-BAAoI,22089
|
|
12
|
-
by_framework/common/emitter.py,sha256=
|
|
12
|
+
by_framework/common/emitter.py,sha256=bobdekSsA3vnMu8TMHonkjfgK-awW_p11dgcTU_3mQw,11940
|
|
13
13
|
by_framework/common/exceptions.py,sha256=NKQC7ueLZ4-gVTRz3SFRZNoPHsayu6jxi3-YlW0n_jg,962
|
|
14
14
|
by_framework/common/logger.py,sha256=DX6S7U5CLcCQIUQ3IWqeeDnSC7QGuHKYaEhiKvKrBpk,5823
|
|
15
15
|
by_framework/common/redis_client.py,sha256=JfYE-QutI9G27dbehzvLw0LziTM1gHmG4tdVjxd0IoA,4668
|
|
16
|
-
by_framework/core/__init__.py,sha256=
|
|
16
|
+
by_framework/core/__init__.py,sha256=cGG93S1gzd1N64JaTCVD7JLV0ZzbF92Q9C6Sog0yDSE,2126
|
|
17
17
|
by_framework/core/availability.py,sha256=ZME5S002eNzNpzaW_QuWw8AmUN1JelUxwCt2U3PPAeQ,17624
|
|
18
18
|
by_framework/core/delivery_gate.py,sha256=yijlCOw8zEou1HCyQzwRHDJZtLBsUnXtQ3x1FZsc4yU,2012
|
|
19
19
|
by_framework/core/discovery.py,sha256=12mAnPzJph0zqlfJwu8k-IN5AKc7dpC0sXToBGLU2Fw,13116
|
|
20
|
-
by_framework/core/registry.py,sha256=
|
|
20
|
+
by_framework/core/registry.py,sha256=bpthWbuVrUIBO1Wtz5F3A2s2GAon9zX_6FR9zn7fxAw,52906
|
|
21
21
|
by_framework/core/wakeup_controller.py,sha256=0v3Fr69XzIKO0fesYUPvQvsGLYt0f648Xq7XEM_jk6c,5545
|
|
22
22
|
by_framework/core/workspace.py,sha256=R9mxghDh_INtPOIXHilBamAIue8jZMrlQoQRI02KSxw,3980
|
|
23
23
|
by_framework/core/extensions/__init__.py,sha256=MvPh029FXvkR1Lk4bZxiq9SfzUQ57paUfoD17-whvn4,1032
|
|
@@ -26,7 +26,7 @@ by_framework/core/extensions/agent_config_audit.py,sha256=Ax67XF2dg_oqz8GvCQ5vsu
|
|
|
26
26
|
by_framework/core/extensions/plugin.py,sha256=rmV4wjf_x-v8E55s9_bDQPyvsqZT4ebZ2fnScbmUGkg,9236
|
|
27
27
|
by_framework/core/extensions/registry.py,sha256=s2UYEOvNzZ3Q8CH1vPBCUw3hhOfQTJ6GKTJ2aXPxVCY,25314
|
|
28
28
|
by_framework/core/extensions/trace_provider.py,sha256=ShM-aJWfenHD006m5sj3M3fi5zZOiTaQ9MZyKV9yv2o,582
|
|
29
|
-
by_framework/core/protocol/__init__.py,sha256=
|
|
29
|
+
by_framework/core/protocol/__init__.py,sha256=wxpD8UMYIYmpXtwZZvT9FQZA6hSjHyoEUZ2kFw1_McA,3444
|
|
30
30
|
by_framework/core/protocol/action_type.py,sha256=5uhwzpaf_PwYMPharJHoX_oc4Cf1dZruCpJ8eyUdwXQ,1112
|
|
31
31
|
by_framework/core/protocol/agent_state.py,sha256=CtqwisNWuPy91FEPXjQrR3r0jLeOrUF1huu23ODnEPo,1919
|
|
32
32
|
by_framework/core/protocol/byai_codec.py,sha256=BqCCYS_uEXOpAHogrNl7-3Tce9HBxMKRpO0ccQN80SA,3185
|
|
@@ -34,14 +34,14 @@ by_framework/core/protocol/byai_command.py,sha256=FJjU4H9b7S5Q59OZqDpVu3JBln7_P9
|
|
|
34
34
|
by_framework/core/protocol/byai_types.py,sha256=szP6yONsMwtErZZUOOgLRnwqIJnPYKM7qcqrgwkEyIc,187
|
|
35
35
|
by_framework/core/protocol/commands.py,sha256=5MFP0EjElbPBAOUE_D0gErUJsHD06EuWCkmqyLsRX8U,11144
|
|
36
36
|
by_framework/core/protocol/content_codec.py,sha256=E8uHhRU6jlH16MQ3UmA8jPexm71l3HKxpquF9xJk04o,643
|
|
37
|
-
by_framework/core/protocol/content_type.py,sha256=
|
|
37
|
+
by_framework/core/protocol/content_type.py,sha256=fagsdHPIoPyWwmgYfb3JzP52olxDDzpGbmQ9q3bmQxE,1269
|
|
38
38
|
by_framework/core/protocol/data_message.py,sha256=UEK9W1lIt_6DLdu0o_CW5YoQJrsEjRUHHEtH2syRO-A,1290
|
|
39
39
|
by_framework/core/protocol/data_shapes.py,sha256=nA6OnfwXRJB1lmwcME3K5TEn_IwmShmBOlYSmOF7y3k,2245
|
|
40
40
|
by_framework/core/protocol/event_type.py,sha256=no7rh1Mx_mhyqqPzEM4aK-W8JB8NwzyOdWN6R_CqaPw,1106
|
|
41
41
|
by_framework/core/protocol/events.py,sha256=SVUXWWVIgxC9VuKjPE1qnvWJgKhWqY48pOHKD0HLcRg,1594
|
|
42
42
|
by_framework/core/protocol/message.py,sha256=sBLm374jsCNa5UPCBJK7xgxJBw3CjrNsoEAnPSp3rAc,2374
|
|
43
43
|
by_framework/core/protocol/message_header.py,sha256=udZkcUv-NEz2j-vpxGglrHWaIJy6k3ZJ9M5ajtL3WQo,3140
|
|
44
|
-
by_framework/core/protocol/responses.py,sha256=
|
|
44
|
+
by_framework/core/protocol/responses.py,sha256=Q2xNFL1zPAGR0OpAB4uxu6iHHKUMIsBjc7yren4c0UA,2589
|
|
45
45
|
by_framework/core/protocol/results.py,sha256=xRr0FhTyuck0ANSOD8yuozzJoVOHFp0G2iSK9V69Hec,5297
|
|
46
46
|
by_framework/core/runtime/__init__.py,sha256=mYbe9D2BvZ3eYTjvU64CZyHLuMoRUU7DGwAIQXf7KR4,991
|
|
47
47
|
by_framework/core/runtime/agent_config_manager.py,sha256=QpBkBXqWHEgqx8VXaQnRWJsG1TbHNJMHbLBMv0tQqJ8,8301
|
|
@@ -82,18 +82,19 @@ by_framework/worker/__init__.py,sha256=GBkUWYnHc2j3_Wx2AC2P49DzFl0CSrl0SKhdRP5CT
|
|
|
82
82
|
by_framework/worker/_control_handling.py,sha256=0bL1Alx8_qaeXQV32Xv4ARSd9q77ySh9LMd7KtZZXu8,8111
|
|
83
83
|
by_framework/worker/_execution_tracking.py,sha256=AWahfDKTXGrx_JWSCNisjGKkGAQruu9vSlXisof0BYQ,2493
|
|
84
84
|
by_framework/worker/_message_processing.py,sha256=rgKzJTgPVDFp_HUfMcqJBwNdTCEfspuRAo5IM9VKsxs,2622
|
|
85
|
-
by_framework/worker/app.py,sha256=
|
|
85
|
+
by_framework/worker/app.py,sha256=UHaQZzalqfPmHvfq-XL7vlPOJ8mTm5ULIU4JpeXA85g,16319
|
|
86
86
|
by_framework/worker/byai_context.py,sha256=DvKw-X8ePZr5c5osdapvzTvH4G23u0J80U-fgMLtge4,2388
|
|
87
87
|
by_framework/worker/byai_worker.py,sha256=0NoRqzuTG_MPMxVbZ1ueeOrqI0H8-Adxh7dM3vckjJo,1757
|
|
88
88
|
by_framework/worker/context.py,sha256=XTNeTAhXvzS2OMDRCDKVD0s3uQscE1rFo9eImUw773w,44747
|
|
89
|
+
by_framework/worker/health_server.py,sha256=4Jw3ltAD7Ow4dPHhwTUU_4JjKz-iDCNG7RJzD2VzPlM,6090
|
|
89
90
|
by_framework/worker/heartbeat.py,sha256=trHDS8FUALsA6jyzDPa087hpr8_2vY8ZUFn0ld1kI2o,15465
|
|
90
91
|
by_framework/worker/processor.py,sha256=D5GcV1T21z4mwJnrZFDRNsxdkhW2WZ3TgpygSKKJUfg,8008
|
|
91
|
-
by_framework/worker/runner.py,sha256=
|
|
92
|
+
by_framework/worker/runner.py,sha256=32T5BeCWalSxpp_eTH1vesF_kLs0HLIx0zm48ywCqT4,45954
|
|
92
93
|
by_framework/worker/worker.py,sha256=_RmNkvHEN5kDPlqbHtVQhkhgiKx37xjnCxNkf_tvASM,35441
|
|
93
94
|
by_framework/worker/sandbox/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
94
95
|
by_framework/worker/sandbox/hook_sandbox.py,sha256=XXSdH5YGX814LvV03vkq-zn5DVIBPYYt2NhRv1lT1Bk,2329
|
|
95
|
-
by_framework-0.2.2.
|
|
96
|
-
by_framework-0.2.2.
|
|
97
|
-
by_framework-0.2.2.
|
|
98
|
-
by_framework-0.2.2.
|
|
99
|
-
by_framework-0.2.2.
|
|
96
|
+
by_framework-0.2.2.dev11.dist-info/METADATA,sha256=pr4rQABuXWzDZP2yc50J0qPN3qyuPxTR3tMJ8Xcu2Uc,28018
|
|
97
|
+
by_framework-0.2.2.dev11.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
98
|
+
by_framework-0.2.2.dev11.dist-info/entry_points.txt,sha256=5t6kgMjkR0UgJTpOsbuQDo0fOf7zeXjQW9DhkxfAGFc,56
|
|
99
|
+
by_framework-0.2.2.dev11.dist-info/licenses/LICENSE,sha256=xx0jnfkXJvxRnG63LTGOxlggYnIysveWIZ6H3PNdCrQ,11357
|
|
100
|
+
by_framework-0.2.2.dev11.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|