microvm-ctl 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/PKG-INFO +11 -4
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/README.md +10 -3
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/__init__.py +6 -2
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/cli.py +308 -14
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/endpoint.py +4 -3
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/fleet.py +60 -2
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/hooks/server.py +8 -4
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/images.py +33 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/integrations/__init__.py +5 -2
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/integrations/durable.py +121 -2
- microvm_ctl-0.3.0/microvm/integrations/stepfunctions.py +353 -0
- microvm_ctl-0.3.0/microvm/lease.py +303 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/monitor.py +93 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/playground/server.py +234 -19
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/playground/static/index.html +74 -7
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/PKG-INFO +11 -4
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/SOURCES.txt +5 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/pyproject.toml +1 -1
- microvm_ctl-0.3.0/tests/test_cli_scale.py +378 -0
- microvm_ctl-0.3.0/tests/test_integrations_durable_scale.py +232 -0
- microvm_ctl-0.3.0/tests/test_integrations_scale.py +218 -0
- microvm_ctl-0.3.0/tests/test_monitor_jobs.py +110 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_playground.py +126 -0
- microvm_ctl-0.3.0/tests/test_scale.py +331 -0
- microvm_ctl-0.2.0/microvm/integrations/stepfunctions.py +0 -210
- microvm_ctl-0.2.0/microvm/lease.py +0 -133
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/LICENSE +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/bootstrap.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/client.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/config.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/data/lambda-microvms/2025-09-09/endpoint-rule-set-1.json.gz +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/data/lambda-microvms/2025-09-09/paginators-1.json +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/data/lambda-microvms/2025-09-09/service-2.json.gz +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/hooks/__init__.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/playground/__init__.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/throttle.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/dependency_links.txt +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/entry_points.txt +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/requires.txt +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/top_level.txt +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/setup.cfg +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_config.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_cost_model.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_endpoint.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_fleet.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_hooks.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_hooks_lease.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_hooks_validate.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_integrations.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_integrations_durable.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_lease.py +0 -0
- {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_throttle.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: microvm-ctl
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Control and execution plane for AWS Lambda MicroVMs: build images, launch Firecracker microVMs, scale fleets inside your quotas, call into them securely, and watch it all live.
|
|
5
5
|
Author: Vivek Raja P S
|
|
6
6
|
License: Apache-2.0
|
|
@@ -69,6 +69,7 @@ Lambda MicroVMs exposes the primitive under Lambda itself: a Firecracker VM with
|
|
|
69
69
|
| In-VM hook runtime | zero-dependency server for `/ready`, `/validate`, `/run`, `/resume`, `/suspend`, `/terminate` plus your own routes | `microvm/hooks/server.py` |
|
|
70
70
|
| Observability and cost | live fleet table, CloudWatch tail, a cost model that prices a session shape before you commit to it | `microvm/monitor.py` |
|
|
71
71
|
| Account bootstrap | artifact bucket plus separate build and execution roles | `microvm/bootstrap.py` |
|
|
72
|
+
| Leases at scale | `LeasePolicy` ceilings, `plan` with the honest concurrency and worst-case cost, `lease_many`, `mvm lease asl --map`, `lease_map`, `mvm watch --image` | `microvm/lease.py`, `microvm/integrations/` |
|
|
72
73
|
| Leases | `Lease`, `LeasePolicy`, `FleetManager.lease`: one task per VM, completed by the VM through a task token, callback id, HTTP, SQS, or EventBridge | `microvm/lease.py` |
|
|
73
74
|
| Orchestrator integrations | generated Step Functions state machine and IAM (`mvm lease asl`, `mvm lease policy`), `lease_microvm` for Lambda durable functions | `microvm/integrations/` |
|
|
74
75
|
|
|
@@ -170,9 +171,9 @@ mvm playground # http://127.0.0.1:8765
|
|
|
170
171
|
- [The playground](docs/playground.md): the local web app over the whole SDK, and how to host it.
|
|
171
172
|
- [Integrations](docs/integrations.md): the lease contract for handing a VM to Step Functions, Lambda durable functions, or your own orchestrator, and what the plane should own.
|
|
172
173
|
|
|
173
|
-
## See it working:
|
|
174
|
+
## See it working: thirteen examples and the article series
|
|
174
175
|
|
|
175
|
-
The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds
|
|
176
|
+
The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds thirteen production-shaped examples, each a Dockerfile plus a single-file app, deployed and recorded on the live service:
|
|
176
177
|
|
|
177
178
|
| Example | What it shows |
|
|
178
179
|
|---|---|
|
|
@@ -184,12 +185,18 @@ The fastest way to understand the plane is to read the apps built on it. The com
|
|
|
184
185
|
| [ci-runner](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/ci-runner) | clone, test, report, terminate; `--max-duration` as the runaway cap |
|
|
185
186
|
| [pdf-service](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/pdf-service) | untrusted HTML rendered in the VM; idle policy sleeps it between bursts |
|
|
186
187
|
| [multi-tenant-agents](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/multi-tenant-agents) | one VM per tenant, identity via `runHookPayload`, `run_payload_factory` on a `Fleet` |
|
|
188
|
+
| [handoff-agent](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/handoff-agent) | the one image every orchestrator leases: `on_lease`, phases, progress, heartbeats, typed failures, in-VM parallel steps |
|
|
189
|
+
| [stepfunctions-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/stepfunctions-handoff) | `runMicrovm.waitForTaskToken` machines from `mvm lease asl`, single lease and a governed Map fan-out |
|
|
190
|
+
| [durable-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/durable-handoff) | a Lambda durable function leases a VM with `lease_with_relaunch` and fans out with `lease_map` |
|
|
191
|
+
| [generic-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/generic-handoff) | any controller: the same lease over SQS, EventBridge, or an HTTP collector |
|
|
192
|
+
| [circuit-breaker](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/circuit-breaker) | running-memory metric, alarm, and `Fleet.drain` for every image: the account-wide stop behind `LeasePolicy` |
|
|
187
193
|
|
|
188
|
-
The
|
|
194
|
+
The four-part article series **Building on AWS Lambda MicroVMs** on the AWS Builder Center walks through them:
|
|
189
195
|
|
|
190
196
|
1. [Control and scale AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JIDTpz0ZgatSBv24drra3gEod9/control-and-scale-aws-lambda-microvms-with-microvm-ctl): this plane and its measurements.
|
|
191
197
|
2. [Seven workloads Lambda could never run, until MicroVMs](https://builder.aws.com/content/3JJ2oNWY9EsZzivMMx044cSlrFQ/seven-workloads-lambda-could-never-run-until-microvms): the first seven examples through build, run, cost, and gotchas.
|
|
192
198
|
3. [A kernel for every customer: scaling AI agents to 1,000 tenants on AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JJ7tASPSSUUTrcpWnWtxOuu8g3/a-kernel-for-every-customer-scaling-ai-agents-to-1000-tenants-on-aws-lambda-microvms-with-microvm-ctl): the multi-tenant finale with the decision guide.
|
|
199
|
+
4. Hand a task to a MicroVM from anywhere (part 4, [source in awesome-microvm](https://github.com/Vivek0712/awesome-microvm/blob/main/blog/03-handoff.md), Builder Center link to follow): the lease contract, Step Functions, durable functions, your own controller, and leases at scale.
|
|
193
200
|
|
|
194
201
|
The article sources and a long-form deep dive per example live under [awesome-microvm/blog](https://github.com/Vivek0712/awesome-microvm/tree/main/blog).
|
|
195
202
|
|
|
@@ -33,6 +33,7 @@ Lambda MicroVMs exposes the primitive under Lambda itself: a Firecracker VM with
|
|
|
33
33
|
| In-VM hook runtime | zero-dependency server for `/ready`, `/validate`, `/run`, `/resume`, `/suspend`, `/terminate` plus your own routes | `microvm/hooks/server.py` |
|
|
34
34
|
| Observability and cost | live fleet table, CloudWatch tail, a cost model that prices a session shape before you commit to it | `microvm/monitor.py` |
|
|
35
35
|
| Account bootstrap | artifact bucket plus separate build and execution roles | `microvm/bootstrap.py` |
|
|
36
|
+
| Leases at scale | `LeasePolicy` ceilings, `plan` with the honest concurrency and worst-case cost, `lease_many`, `mvm lease asl --map`, `lease_map`, `mvm watch --image` | `microvm/lease.py`, `microvm/integrations/` |
|
|
36
37
|
| Leases | `Lease`, `LeasePolicy`, `FleetManager.lease`: one task per VM, completed by the VM through a task token, callback id, HTTP, SQS, or EventBridge | `microvm/lease.py` |
|
|
37
38
|
| Orchestrator integrations | generated Step Functions state machine and IAM (`mvm lease asl`, `mvm lease policy`), `lease_microvm` for Lambda durable functions | `microvm/integrations/` |
|
|
38
39
|
|
|
@@ -134,9 +135,9 @@ mvm playground # http://127.0.0.1:8765
|
|
|
134
135
|
- [The playground](docs/playground.md): the local web app over the whole SDK, and how to host it.
|
|
135
136
|
- [Integrations](docs/integrations.md): the lease contract for handing a VM to Step Functions, Lambda durable functions, or your own orchestrator, and what the plane should own.
|
|
136
137
|
|
|
137
|
-
## See it working:
|
|
138
|
+
## See it working: thirteen examples and the article series
|
|
138
139
|
|
|
139
|
-
The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds
|
|
140
|
+
The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds thirteen production-shaped examples, each a Dockerfile plus a single-file app, deployed and recorded on the live service:
|
|
140
141
|
|
|
141
142
|
| Example | What it shows |
|
|
142
143
|
|---|---|
|
|
@@ -148,12 +149,18 @@ The fastest way to understand the plane is to read the apps built on it. The com
|
|
|
148
149
|
| [ci-runner](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/ci-runner) | clone, test, report, terminate; `--max-duration` as the runaway cap |
|
|
149
150
|
| [pdf-service](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/pdf-service) | untrusted HTML rendered in the VM; idle policy sleeps it between bursts |
|
|
150
151
|
| [multi-tenant-agents](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/multi-tenant-agents) | one VM per tenant, identity via `runHookPayload`, `run_payload_factory` on a `Fleet` |
|
|
152
|
+
| [handoff-agent](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/handoff-agent) | the one image every orchestrator leases: `on_lease`, phases, progress, heartbeats, typed failures, in-VM parallel steps |
|
|
153
|
+
| [stepfunctions-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/stepfunctions-handoff) | `runMicrovm.waitForTaskToken` machines from `mvm lease asl`, single lease and a governed Map fan-out |
|
|
154
|
+
| [durable-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/durable-handoff) | a Lambda durable function leases a VM with `lease_with_relaunch` and fans out with `lease_map` |
|
|
155
|
+
| [generic-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/generic-handoff) | any controller: the same lease over SQS, EventBridge, or an HTTP collector |
|
|
156
|
+
| [circuit-breaker](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/circuit-breaker) | running-memory metric, alarm, and `Fleet.drain` for every image: the account-wide stop behind `LeasePolicy` |
|
|
151
157
|
|
|
152
|
-
The
|
|
158
|
+
The four-part article series **Building on AWS Lambda MicroVMs** on the AWS Builder Center walks through them:
|
|
153
159
|
|
|
154
160
|
1. [Control and scale AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JIDTpz0ZgatSBv24drra3gEod9/control-and-scale-aws-lambda-microvms-with-microvm-ctl): this plane and its measurements.
|
|
155
161
|
2. [Seven workloads Lambda could never run, until MicroVMs](https://builder.aws.com/content/3JJ2oNWY9EsZzivMMx044cSlrFQ/seven-workloads-lambda-could-never-run-until-microvms): the first seven examples through build, run, cost, and gotchas.
|
|
156
162
|
3. [A kernel for every customer: scaling AI agents to 1,000 tenants on AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JJ7tASPSSUUTrcpWnWtxOuu8g3/a-kernel-for-every-customer-scaling-ai-agents-to-1000-tenants-on-aws-lambda-microvms-with-microvm-ctl): the multi-tenant finale with the decision guide.
|
|
163
|
+
4. Hand a task to a MicroVM from anywhere (part 4, [source in awesome-microvm](https://github.com/Vivek0712/awesome-microvm/blob/main/blog/03-handoff.md), Builder Center link to follow): the lease contract, Step Functions, durable functions, your own controller, and leases at scale.
|
|
157
164
|
|
|
158
165
|
The article sources and a long-form deep dive per example live under [awesome-microvm/blog](https://github.com/Vivek0712/awesome-microvm/tree/main/blog).
|
|
159
166
|
|
|
@@ -9,10 +9,10 @@ from microvm.config import PlaneConfig
|
|
|
9
9
|
from microvm.endpoint import EndpointClient, EndpointError
|
|
10
10
|
from microvm.fleet import Fleet, FleetManager
|
|
11
11
|
from microvm.images import ImageBuilder, ImageBuildError
|
|
12
|
-
from microvm.lease import Lease, LeasePolicy
|
|
12
|
+
from microvm.lease import FanoutLimit, Lease, LeasePlan, LeasePlanRejected, LeasePolicy, plan_fanout
|
|
13
13
|
from microvm.monitor import CostModel, FleetMonitor
|
|
14
14
|
|
|
15
|
-
__version__ = "0.
|
|
15
|
+
__version__ = "0.3.0"
|
|
16
16
|
|
|
17
17
|
__all__ = [
|
|
18
18
|
"microvm_client",
|
|
@@ -24,6 +24,10 @@ __all__ = [
|
|
|
24
24
|
"FleetManager",
|
|
25
25
|
"Lease",
|
|
26
26
|
"LeasePolicy",
|
|
27
|
+
"LeasePlan",
|
|
28
|
+
"LeasePlanRejected",
|
|
29
|
+
"FanoutLimit",
|
|
30
|
+
"plan_fanout",
|
|
27
31
|
"EndpointClient",
|
|
28
32
|
"EndpointError",
|
|
29
33
|
"FleetMonitor",
|
|
@@ -11,7 +11,8 @@
|
|
|
11
11
|
mvm drain IMAGE terminate the whole fleet
|
|
12
12
|
mvm call ID /path [-X POST -d '{}'] authenticated request into the VM
|
|
13
13
|
mvm status ID | watch ID job telemetry from the hook runtime (/status, /events)
|
|
14
|
-
mvm
|
|
14
|
+
mvm watch --image NAME one live table for every leased VM of an image
|
|
15
|
+
mvm lease asl|policy|run|plan Step Functions ASL, IAM statements, leases, fan-out sizing
|
|
15
16
|
mvm top [--image NAME] [--watch] live fleet dashboard
|
|
16
17
|
mvm logs IMAGE CloudWatch tail
|
|
17
18
|
mvm cost [--memory-gb 2 ...] session economics
|
|
@@ -95,6 +96,26 @@ def cmd_quotas(args):
|
|
|
95
96
|
if not applied:
|
|
96
97
|
console.print("[dim]Service Quotas did not answer (missing servicequotas:GetServiceQuota?); "
|
|
97
98
|
"showing published defaults.[/]")
|
|
99
|
+
for line in fanout_tier_lines(mem):
|
|
100
|
+
console.print(line)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
FANOUT_TIERS_GB = (0.5, 1, 2, 4, 8)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def fanout_tier_lines(memory_quota_gb) -> list[str]:
|
|
107
|
+
"""One line per image size tier: how many can be in flight at once under the applied
|
|
108
|
+
memory quota, the same arithmetic `mvm lease plan` uses."""
|
|
109
|
+
from microvm.lease import DEFAULT_FANOUT_LIMIT, LeasePolicy, fanout_limit
|
|
110
|
+
if memory_quota_gb is None:
|
|
111
|
+
return [f"[dim]memory quota unknown: fan-outs default to {DEFAULT_FANOUT_LIMIT} at once "
|
|
112
|
+
f"(MVM_LEASE_MAX_CONCURRENCY overrides)[/]"]
|
|
113
|
+
policy = LeasePolicy()
|
|
114
|
+
lines = []
|
|
115
|
+
for tier in FANOUT_TIERS_GB:
|
|
116
|
+
lim = fanout_limit(int(tier * 1024), policy, memory_quota_gb=memory_quota_gb, launch_rate=1.0)
|
|
117
|
+
lines.append(f"{tier:g} GB images: [bold]{lim.limit}[/] at once")
|
|
118
|
+
return lines
|
|
98
119
|
|
|
99
120
|
|
|
100
121
|
# ---------------------------------------------------------------- images
|
|
@@ -303,7 +324,88 @@ def cmd_status(args):
|
|
|
303
324
|
console.print(_log_line(entry) if isinstance(entry, dict) else str(entry))
|
|
304
325
|
|
|
305
326
|
|
|
327
|
+
def _answered(row: dict) -> bool:
|
|
328
|
+
"""job_status gives a member that did not answer /status only microvm_id and error."""
|
|
329
|
+
return "phase" in row
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _row_state(row: dict) -> str:
|
|
333
|
+
if not _answered(row):
|
|
334
|
+
return "[red]unreachable[/]"
|
|
335
|
+
if row.get("done"):
|
|
336
|
+
return "[red]failed[/]" if row.get("error") else "[bold green]done[/]"
|
|
337
|
+
if row.get("lost"):
|
|
338
|
+
return "[yellow]lost[/]"
|
|
339
|
+
return "[cyan]running[/]"
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def _fleet_job_view(rows: list[dict], image: str):
|
|
343
|
+
"""The `mvm watch --image` body: one row per RUNNING member plus the aggregate footer."""
|
|
344
|
+
from rich.console import Group
|
|
345
|
+
t = Table(title=f"fleet job: {image}", header_style="bold magenta")
|
|
346
|
+
for c in ("microvm id", "lease id", "phase", "progress", "elapsed", "state"):
|
|
347
|
+
t.add_column(c)
|
|
348
|
+
for r in rows:
|
|
349
|
+
prog = r.get("progress")
|
|
350
|
+
if isinstance(prog, dict):
|
|
351
|
+
done, total = prog.get("done"), prog.get("total")
|
|
352
|
+
prog = f"{done}/{total}" if total else (str(done) if done is not None else "-")
|
|
353
|
+
elapsed = r.get("elapsed_s")
|
|
354
|
+
t.add_row(
|
|
355
|
+
str(r.get("microvm_id", "-")), str(r.get("lease_id") or "-"),
|
|
356
|
+
str(r.get("phase") or "-") if _answered(r) else f"[dim]{r.get('error') or '-'}[/]",
|
|
357
|
+
str(prog if prog is not None else "-"),
|
|
358
|
+
f"{elapsed:.0f} s" if isinstance(elapsed, (int, float)) else "-", _row_state(r),
|
|
359
|
+
)
|
|
360
|
+
return Group(t, fleet_job_footer(rows))
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def fleet_job_footer(rows: list[dict]) -> str:
|
|
364
|
+
"""'done D/N, running R, lost L, slowest <id> <phase> <elapsed>' over job_status rows."""
|
|
365
|
+
answered = [r for r in rows if _answered(r)]
|
|
366
|
+
done = sum(1 for r in answered if r.get("done"))
|
|
367
|
+
lost = sum(1 for r in answered if r.get("lost") and not r.get("done"))
|
|
368
|
+
running = len(answered) - done - lost
|
|
369
|
+
text = f"done {done}/{len(rows)}, running {running}, lost {lost}"
|
|
370
|
+
if len(answered) < len(rows):
|
|
371
|
+
text += f", unreachable {len(rows) - len(answered)}"
|
|
372
|
+
timed = [r for r in answered if isinstance(r.get("elapsed_s"), (int, float))]
|
|
373
|
+
active = [r for r in timed if not r.get("done")] or timed
|
|
374
|
+
if active:
|
|
375
|
+
slow = max(active, key=lambda r: r["elapsed_s"])
|
|
376
|
+
text += f", slowest {slow.get('microvm_id')} {slow.get('phase') or '-'} {slow['elapsed_s']:.0f} s"
|
|
377
|
+
return text
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def fleet_job_finished(rows: list[dict], seen_members: bool) -> bool:
|
|
381
|
+
"""Stop when every member is done, or when members were seen earlier and are now gone."""
|
|
382
|
+
if rows:
|
|
383
|
+
return all(r.get("done") for r in rows)
|
|
384
|
+
return seen_members
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def cmd_watch_image(args):
|
|
388
|
+
from rich.live import Live
|
|
389
|
+
mon = FleetMonitor(_cfg(args))
|
|
390
|
+
kw = {"port": args.port} if args.port else {}
|
|
391
|
+
rows = mon.job_status(args.image, **kw)
|
|
392
|
+
seen = bool(rows)
|
|
393
|
+
deadline = time.time() + args.timeout
|
|
394
|
+
with Live(_fleet_job_view(rows, args.image), refresh_per_second=2, console=console) as live:
|
|
395
|
+
while not fleet_job_finished(rows, seen) and time.time() < deadline:
|
|
396
|
+
time.sleep(args.interval)
|
|
397
|
+
rows = mon.job_status(args.image, **kw)
|
|
398
|
+
seen = seen or bool(rows)
|
|
399
|
+
live.update(_fleet_job_view(rows, args.image))
|
|
400
|
+
# the live view already ends on the footer; a trailing newline keeps the shell prompt off it
|
|
401
|
+
console.print()
|
|
402
|
+
|
|
403
|
+
|
|
306
404
|
def cmd_watch(args):
|
|
405
|
+
if args.image:
|
|
406
|
+
return cmd_watch_image(args)
|
|
407
|
+
if not args.id:
|
|
408
|
+
sys.exit("mvm watch: pass a microVM ID or --image NAME")
|
|
307
409
|
from rich.live import Live
|
|
308
410
|
client = EndpointClient(_cfg(args), args.id, ports=[args.port] if args.port else None)
|
|
309
411
|
snap = client.status()
|
|
@@ -318,9 +420,41 @@ def cmd_watch(args):
|
|
|
318
420
|
|
|
319
421
|
|
|
320
422
|
# ---------------------------------------------------------------- leases
|
|
423
|
+
POLICY_FLAGS = (
|
|
424
|
+
("budget", "budget_s"), ("heartbeat", "heartbeat_timeout_s"), ("slack", "slack_s"),
|
|
425
|
+
("max_concurrency", "max_concurrency"), ("max_vm_seconds", "max_vm_seconds"),
|
|
426
|
+
("approval_usd", "approval_usd"),
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
|
|
321
430
|
def _lease_policy(args):
|
|
431
|
+
"""LeasePolicy from the MVM_LEASE_* environment, then any flag that was given."""
|
|
322
432
|
from microvm.lease import LeasePolicy
|
|
323
|
-
|
|
433
|
+
policy = LeasePolicy.from_env()
|
|
434
|
+
for flag, field in POLICY_FLAGS:
|
|
435
|
+
value = getattr(args, flag, None)
|
|
436
|
+
if value is not None:
|
|
437
|
+
setattr(policy, field, value)
|
|
438
|
+
return policy
|
|
439
|
+
|
|
440
|
+
|
|
441
|
+
def _baseline_mib(args, cfg) -> int:
|
|
442
|
+
"""--baseline-mib when given, else the image version's minimumMemoryInMiB."""
|
|
443
|
+
given = getattr(args, "baseline_mib", None)
|
|
444
|
+
if given:
|
|
445
|
+
return given
|
|
446
|
+
return ImageBuilder(cfg).baseline_mib(args.image, getattr(args, "version", None))
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _fill_template(obj, i: int):
|
|
450
|
+
"""Replace `{i}` in every string value of a JSON-like object with the shard index."""
|
|
451
|
+
if isinstance(obj, str):
|
|
452
|
+
return obj.replace("{i}", str(i))
|
|
453
|
+
if isinstance(obj, list):
|
|
454
|
+
return [_fill_template(v, i) for v in obj]
|
|
455
|
+
if isinstance(obj, dict):
|
|
456
|
+
return {k: _fill_template(v, i) for k, v in obj.items()}
|
|
457
|
+
return obj
|
|
324
458
|
|
|
325
459
|
|
|
326
460
|
def _image_arn_offline(image: str, region: str, role_arn: str | None, profile) -> str:
|
|
@@ -341,14 +475,40 @@ def cmd_lease_asl(args):
|
|
|
341
475
|
role = args.execution_role or cfg.execution_role_arn
|
|
342
476
|
if not role:
|
|
343
477
|
sys.exit("mvm lease asl: --execution-role ARN (or MVM_EXECUTION_ROLE_ARN) is required")
|
|
478
|
+
kw = {}
|
|
479
|
+
if args.map:
|
|
480
|
+
from microvm.integrations.stepfunctions import FanoutSpec
|
|
481
|
+
max_concurrency = args.max_concurrency
|
|
482
|
+
if max_concurrency is None:
|
|
483
|
+
max_concurrency = _computed_fanout_limit(args, cfg)
|
|
484
|
+
kw["fanout"] = FanoutSpec(
|
|
485
|
+
items_expr=args.items_expr, max_concurrency=max_concurrency,
|
|
486
|
+
approval_topic_arn=args.approval_topic, approve_above_shards=args.approve_above_shards,
|
|
487
|
+
)
|
|
344
488
|
asl = lease_state_machine(
|
|
345
489
|
image_arn=_image_arn_offline(args.image, cfg.region, role, cfg.profile),
|
|
346
490
|
execution_role_arn=role, policy=_lease_policy(args), region=cfg.region,
|
|
347
|
-
name=args.name, task_expr=args.task_expr, heartbeat_s=args.heartbeat_every,
|
|
491
|
+
name=args.name, task_expr=args.task_expr, heartbeat_s=args.heartbeat_every, **kw,
|
|
348
492
|
)
|
|
349
493
|
print(json.dumps(asl, indent=2))
|
|
350
494
|
|
|
351
495
|
|
|
496
|
+
def _computed_fanout_limit(args, cfg) -> int:
|
|
497
|
+
"""Map MaxConcurrency when --max-concurrency is omitted: the plane's fan-out limit for the
|
|
498
|
+
image's baseline, with the reason on stderr so the generated ASL is explainable."""
|
|
499
|
+
from microvm.lease import DEFAULT_FANOUT_LIMIT
|
|
500
|
+
try:
|
|
501
|
+
baseline = _baseline_mib(args, cfg)
|
|
502
|
+
limit = FleetManager(cfg).fanout_limit(baseline, _lease_policy(args))
|
|
503
|
+
except Exception as e: # no credentials, unknown image: still emit a usable machine
|
|
504
|
+
print(f"mvm lease asl: could not compute the fan-out limit ({e}); "
|
|
505
|
+
f"using MaxConcurrency {DEFAULT_FANOUT_LIMIT} (default)", file=sys.stderr)
|
|
506
|
+
return DEFAULT_FANOUT_LIMIT
|
|
507
|
+
print(f"mvm lease asl: MaxConcurrency {limit.limit} ({limit.reason}; "
|
|
508
|
+
f"{baseline} MiB baseline)", file=sys.stderr)
|
|
509
|
+
return limit.limit
|
|
510
|
+
|
|
511
|
+
|
|
352
512
|
def cmd_lease_policy(args):
|
|
353
513
|
from microvm.integrations.stepfunctions import iam_statements, orchestrator_statements, policy_document
|
|
354
514
|
cfg = _cfg(args)
|
|
@@ -363,6 +523,8 @@ def cmd_lease_policy(args):
|
|
|
363
523
|
def cmd_lease_run(args):
|
|
364
524
|
from microvm.lease import Lease
|
|
365
525
|
cfg = _cfg(args)
|
|
526
|
+
if args.shards is not None:
|
|
527
|
+
return cmd_lease_run_many(args, cfg)
|
|
366
528
|
if args.kind != "none" and not args.token:
|
|
367
529
|
sys.exit(f"mvm lease run: --token is required for kind {args.kind}")
|
|
368
530
|
try:
|
|
@@ -383,6 +545,95 @@ def cmd_lease_run(args):
|
|
|
383
545
|
console.print(f" now {_state(vm.state)}; follow it with: mvm watch {vm.microvm_id}")
|
|
384
546
|
|
|
385
547
|
|
|
548
|
+
def cmd_lease_run_many(args, cfg):
|
|
549
|
+
"""`mvm lease run IMAGE --shards N`: plan, refuse or launch every shard through lease_many."""
|
|
550
|
+
from microvm.lease import Lease, LeasePlanRejected
|
|
551
|
+
if args.shards < 1:
|
|
552
|
+
sys.exit(f"mvm lease run: --shards must be at least 1, got {args.shards}")
|
|
553
|
+
kind = args.kind if (args.kind != "none" or not args.token_template) else "none"
|
|
554
|
+
if args.token_template and kind == "none":
|
|
555
|
+
sys.exit("mvm lease run: --token-template needs --kind (sfn, durable, http, sqs, eventbridge)")
|
|
556
|
+
if kind != "none" and not args.token_template:
|
|
557
|
+
sys.exit(f"mvm lease run: --token-template with {{i}} is required for kind {kind} and --shards")
|
|
558
|
+
try:
|
|
559
|
+
template = json.loads(args.task_template) if args.task_template else (
|
|
560
|
+
json.loads(args.task) if args.task else {})
|
|
561
|
+
except ValueError as e:
|
|
562
|
+
sys.exit(f"mvm lease run: --task-template is not valid JSON: {e}")
|
|
563
|
+
if not isinstance(template, dict):
|
|
564
|
+
sys.exit("mvm lease run: --task-template must be a JSON object")
|
|
565
|
+
leases, tasks = [], []
|
|
566
|
+
for i in range(args.shards):
|
|
567
|
+
token = args.token_template.replace("{i}", str(i)) if args.token_template else ""
|
|
568
|
+
lease_id = args.id.replace("{i}", str(i)) if args.id else None
|
|
569
|
+
leases.append(Lease(kind=kind, token=token, region=cfg.region, target=args.target,
|
|
570
|
+
heartbeat_s=args.heartbeat_every, id=lease_id))
|
|
571
|
+
tasks.append(_fill_template(template, i))
|
|
572
|
+
policy = _lease_policy(args)
|
|
573
|
+
fm = FleetManager(cfg)
|
|
574
|
+
baseline = _baseline_mib(args, cfg)
|
|
575
|
+
plan = fm.plan(args.shards, baseline, policy)
|
|
576
|
+
console.print(plan.summary())
|
|
577
|
+
if plan.rejected:
|
|
578
|
+
sys.exit(2)
|
|
579
|
+
if plan.needs_approval and not args.approve:
|
|
580
|
+
print(f"mvm lease run: the plan needs approval (worst case ${plan.worst_case_usd:.2f} above "
|
|
581
|
+
f"approval_usd ${policy.approval_usd:g}); re-run with --approve", file=sys.stderr)
|
|
582
|
+
sys.exit(3)
|
|
583
|
+
if args.shards > plan.concurrency:
|
|
584
|
+
print(f"mvm lease run: {args.shards} leases exceed the concurrency limit {plan.concurrency}; "
|
|
585
|
+
f"launch in waves (--shards {plan.concurrency} at a time)", file=sys.stderr)
|
|
586
|
+
sys.exit(2)
|
|
587
|
+
try:
|
|
588
|
+
vms = fm.lease_many(args.image, leases, tasks, policy, baseline_mib=baseline, version=args.version,
|
|
589
|
+
execution_role=args.execution_role)
|
|
590
|
+
except LeasePlanRejected as e:
|
|
591
|
+
sys.exit(f"mvm lease run: {e}")
|
|
592
|
+
for i, vm in enumerate(vms):
|
|
593
|
+
console.print(f"[bold green]✓[/] shard {i}: {vm.microvm_id} {_state(vm.state)} "
|
|
594
|
+
f"[link=https://{vm.endpoint}]{vm.endpoint}[/link] lease {kind}")
|
|
595
|
+
if args.wait:
|
|
596
|
+
for vm in vms:
|
|
597
|
+
fm.wait_until(vm.microvm_id, "RUNNING")
|
|
598
|
+
console.print(f" all {len(vms)} RUNNING; follow them with: mvm watch --image {args.image}")
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def _plan_table(plan) -> Table:
|
|
602
|
+
lim = plan.limit
|
|
603
|
+
t = Table(title=f"lease plan: {plan.shards} shards", header_style="bold magenta", show_header=False)
|
|
604
|
+
t.add_column("field", style="cyan")
|
|
605
|
+
t.add_column("value")
|
|
606
|
+
t.add_row("shards", str(plan.shards))
|
|
607
|
+
t.add_row("baseline", f"{plan.baseline_mib} MiB")
|
|
608
|
+
t.add_row("memory quota", f"{lim.memory_quota_gb:g} GB" if lim.memory_quota_gb is not None
|
|
609
|
+
else "[dim]unknown[/]")
|
|
610
|
+
t.add_row("launch rate", f"{lim.launch_rate:g}/s")
|
|
611
|
+
t.add_row("concurrency", f"{plan.concurrency} [dim]({lim.reason})[/]")
|
|
612
|
+
t.add_row("waves", str(plan.waves))
|
|
613
|
+
t.add_row("launch to all running", f"~{plan.launch_to_all_running_s:.1f} s")
|
|
614
|
+
t.add_row("worst case VM-s", str(plan.worst_case_vm_seconds))
|
|
615
|
+
t.add_row("worst case USD", f"${plan.worst_case_usd:.4f}")
|
|
616
|
+
t.add_row("approval needed", "[yellow]yes[/]" if plan.needs_approval else "no")
|
|
617
|
+
t.add_row("rejected", f"[red]{plan.rejected}[/]" if plan.rejected else "no")
|
|
618
|
+
return t
|
|
619
|
+
|
|
620
|
+
|
|
621
|
+
def cmd_lease_plan(args):
|
|
622
|
+
"""Exit 0 when the plan can launch, 2 when it is rejected, 3 when it needs approval."""
|
|
623
|
+
cfg = _cfg(args)
|
|
624
|
+
fm = FleetManager(cfg)
|
|
625
|
+
plan = fm.plan(args.shards, _baseline_mib(args, cfg), _lease_policy(args))
|
|
626
|
+
if args.json:
|
|
627
|
+
print(json.dumps(plan.to_dict(), indent=2))
|
|
628
|
+
else:
|
|
629
|
+
console.print(_plan_table(plan))
|
|
630
|
+
console.print(plan.summary())
|
|
631
|
+
if plan.rejected:
|
|
632
|
+
sys.exit(2)
|
|
633
|
+
if plan.needs_approval:
|
|
634
|
+
sys.exit(3)
|
|
635
|
+
|
|
636
|
+
|
|
386
637
|
def cmd_playground(args):
|
|
387
638
|
from microvm.playground import serve
|
|
388
639
|
serve(host=args.host, port=args.port, dry_run=args.dry_run, open_browser=not args.no_open, cfg=_cfg(args))
|
|
@@ -486,21 +737,36 @@ def main(argv: list[str] | None = None):
|
|
|
486
737
|
st.add_argument("--port", type=int, default=None, help="non-default app port")
|
|
487
738
|
st.set_defaults(fn=cmd_status)
|
|
488
739
|
|
|
489
|
-
w = sub.add_parser("watch", help="live job table
|
|
490
|
-
w.add_argument("id")
|
|
740
|
+
w = sub.add_parser("watch", help="live job table for one VM (GET /events) or every VM of an image")
|
|
741
|
+
w.add_argument("id", nargs="?", help="microVM id (omit with --image)")
|
|
742
|
+
w.add_argument("--image", default=None, help="one table for every RUNNING member of this image")
|
|
491
743
|
w.add_argument("--port", type=int, default=None, help="non-default app port")
|
|
744
|
+
w.add_argument("--interval", type=float, default=2, help="seconds between polls with --image")
|
|
492
745
|
w.add_argument("--timeout", type=float, default=600, help="stop after N seconds")
|
|
493
746
|
w.set_defaults(fn=cmd_watch)
|
|
494
747
|
|
|
495
748
|
le = sub.add_parser("lease", help="hand a VM a task through a lease").add_subparsers(
|
|
496
749
|
dest="sub", required=True)
|
|
497
750
|
|
|
498
|
-
def _policy_flags(sp):
|
|
499
|
-
sp.add_argument("--budget", type=int, default=
|
|
500
|
-
help="seconds the orchestrator waits (task timeout)")
|
|
501
|
-
sp.add_argument("--heartbeat", type=int, default=
|
|
502
|
-
|
|
503
|
-
sp.add_argument("--
|
|
751
|
+
def _policy_flags(sp, heartbeat_every=True):
|
|
752
|
+
sp.add_argument("--budget", type=int, default=None,
|
|
753
|
+
help="seconds the orchestrator waits (task timeout; MVM_LEASE_BUDGET_S or 900)")
|
|
754
|
+
sp.add_argument("--heartbeat", type=int, default=None,
|
|
755
|
+
help="heartbeat timeout in seconds (MVM_LEASE_HEARTBEAT_TIMEOUT_S or 120)")
|
|
756
|
+
sp.add_argument("--slack", type=int, default=None,
|
|
757
|
+
help="VM outlives the budget by this many seconds (MVM_LEASE_SLACK_S or 120)")
|
|
758
|
+
sp.add_argument("--max-concurrency", type=int, default=None,
|
|
759
|
+
help="VMs in flight per fan-out (MVM_LEASE_MAX_CONCURRENCY; default: memory quota)")
|
|
760
|
+
sp.add_argument("--max-vm-seconds", type=int, default=None,
|
|
761
|
+
help="worst-case VM-seconds a fan-out may commit to (MVM_LEASE_MAX_VM_SECONDS)")
|
|
762
|
+
sp.add_argument("--approval-usd", type=float, default=None,
|
|
763
|
+
help="worst-case USD above which a plan needs approval (MVM_LEASE_APPROVAL_USD)")
|
|
764
|
+
if heartbeat_every:
|
|
765
|
+
sp.add_argument("--heartbeat-every", type=int, default=30, help="seconds between VM heartbeats")
|
|
766
|
+
|
|
767
|
+
def _baseline_flag(sp):
|
|
768
|
+
sp.add_argument("--baseline-mib", type=int, default=None,
|
|
769
|
+
help="image memory baseline in MiB (default: read from the image version)")
|
|
504
770
|
|
|
505
771
|
la = le.add_parser("asl", help="print the Step Functions state machine (JSONata) for one lease")
|
|
506
772
|
la.add_argument("--image", required=True, help="image name or ARN")
|
|
@@ -508,6 +774,15 @@ def main(argv: list[str] | None = None):
|
|
|
508
774
|
help="VM execution role ARN (default: MVM_EXECUTION_ROLE_ARN)")
|
|
509
775
|
la.add_argument("--name", default="Lease", help="name of the lease state")
|
|
510
776
|
la.add_argument("--task-expr", default="$states.input", help="JSONata expression for the task")
|
|
777
|
+
la.add_argument("--map", action="store_true",
|
|
778
|
+
help="emit a Map over --items-expr: one lease per item, MaxConcurrency from the plane")
|
|
779
|
+
la.add_argument("--items-expr", default="$states.input.shards",
|
|
780
|
+
help="JSONata array of task objects for --map")
|
|
781
|
+
la.add_argument("--approval-topic", default=None,
|
|
782
|
+
help="SNS topic ARN: plans above --approve-above-shards wait for a task token")
|
|
783
|
+
la.add_argument("--approve-above-shards", type=int, default=None,
|
|
784
|
+
help="shard count above which the Map waits for approval on --approval-topic")
|
|
785
|
+
_baseline_flag(la)
|
|
511
786
|
_policy_flags(la)
|
|
512
787
|
la.set_defaults(fn=cmd_lease_asl)
|
|
513
788
|
|
|
@@ -518,19 +793,38 @@ def main(argv: list[str] | None = None):
|
|
|
518
793
|
lp.add_argument("--execution-role", default=None, help="VM execution role for iam:PassRole")
|
|
519
794
|
lp.set_defaults(fn=cmd_lease_policy)
|
|
520
795
|
|
|
521
|
-
lr = le.add_parser("run", help="launch one lease by hand
|
|
796
|
+
lr = le.add_parser("run", help="launch one lease by hand, or --shards N of them through the plan")
|
|
522
797
|
lr.add_argument("image")
|
|
523
|
-
lr.add_argument("--kind",
|
|
798
|
+
lr.add_argument("--kind", default="none",
|
|
799
|
+
choices=["sfn", "durable", "http", "sqs", "eventbridge", "none"])
|
|
524
800
|
lr.add_argument("--token", default=None, help="task token, callback id, or bearer (not for kind none)")
|
|
525
801
|
lr.add_argument("--target", default=None, help="http URL, SQS queue URL, or event bus name")
|
|
526
802
|
lr.add_argument("--task", default=None, help="task JSON (pointers, not bodies; 4096 chars total)")
|
|
527
|
-
lr.add_argument("--id", default=None, help="human label carried as lease.id")
|
|
803
|
+
lr.add_argument("--id", default=None, help="human label carried as lease.id ({i} = shard index)")
|
|
528
804
|
lr.add_argument("--version", help="image version (default: latest ACTIVE)")
|
|
529
805
|
lr.add_argument("--execution-role", default=None)
|
|
530
806
|
lr.add_argument("--wait", action="store_true", help="poll until RUNNING")
|
|
807
|
+
lr.add_argument("--shards", type=int, default=None,
|
|
808
|
+
help="launch N leases at once after planning them (refuses above the limit)")
|
|
809
|
+
lr.add_argument("--task-template", default=None,
|
|
810
|
+
help="task JSON for --shards; {i} in string values becomes the shard index")
|
|
811
|
+
lr.add_argument("--token-template", default=None,
|
|
812
|
+
help="per-shard token for --shards with a kind other than none; {i} = shard index")
|
|
813
|
+
lr.add_argument("--approve", action="store_true",
|
|
814
|
+
help="launch even when the plan's worst case is above approval_usd")
|
|
815
|
+
_baseline_flag(lr)
|
|
531
816
|
_policy_flags(lr)
|
|
532
817
|
lr.set_defaults(fn=cmd_lease_run)
|
|
533
818
|
|
|
819
|
+
lpl = le.add_parser("plan", help="size a fan-out: concurrency, waves, launch time, worst case cost")
|
|
820
|
+
lpl.add_argument("--image", required=True, help="image name (baseline read from its ACTIVE version)")
|
|
821
|
+
lpl.add_argument("--shards", type=int, required=True, help="how many leases the fan-out launches")
|
|
822
|
+
lpl.add_argument("--version", default=None, help="image version to read the baseline from")
|
|
823
|
+
lpl.add_argument("--json", action="store_true", help="print the plan as JSON")
|
|
824
|
+
_baseline_flag(lpl)
|
|
825
|
+
_policy_flags(lpl, heartbeat_every=False)
|
|
826
|
+
lpl.set_defaults(fn=cmd_lease_plan)
|
|
827
|
+
|
|
534
828
|
tp = sub.add_parser("top", help="live fleet dashboard")
|
|
535
829
|
tp.add_argument("--image")
|
|
536
830
|
tp.add_argument("--watch", action="store_true")
|
|
@@ -129,12 +129,13 @@ class EndpointClient:
|
|
|
129
129
|
raise EndpointError(f"{self.microvm_id} not serving {path} after {timeout}s")
|
|
130
130
|
|
|
131
131
|
# -- job telemetry (HookApp built-in /status and /events) -------------------
|
|
132
|
-
def status(self, since: int | None = None) -> dict:
|
|
132
|
+
def status(self, since: int | None = None, *, timeout: float = 10, max_attempts: int = 6) -> dict:
|
|
133
133
|
"""The hook runtime's job snapshot (`GET /status`): phase, progress, counters,
|
|
134
134
|
the log tail, and the lease state. `since` returns only log lines with a
|
|
135
|
-
sequence number above it.
|
|
135
|
+
sequence number above it. `timeout` and `max_attempts` bound one poll, which
|
|
136
|
+
matters when many members are polled on a schedule."""
|
|
136
137
|
path = "/status" if since is None else f"/status?since={int(since)}"
|
|
137
|
-
resp = self.get(path, timeout=
|
|
138
|
+
resp = self.get(path, timeout=timeout, max_attempts=max_attempts)
|
|
138
139
|
if resp.status_code != 200:
|
|
139
140
|
raise EndpointError(f"{self.microvm_id} answered {resp.status_code} on {path}")
|
|
140
141
|
return resp.json()
|
|
@@ -18,7 +18,17 @@ from typing import Callable
|
|
|
18
18
|
|
|
19
19
|
from microvm.client import image_arn, microvm_client
|
|
20
20
|
from microvm.config import TPS, PlaneConfig
|
|
21
|
-
from microvm.lease import
|
|
21
|
+
from microvm.lease import (
|
|
22
|
+
FanoutLimit,
|
|
23
|
+
Lease,
|
|
24
|
+
LeasePlan,
|
|
25
|
+
LeasePlanRejected,
|
|
26
|
+
LeasePolicy,
|
|
27
|
+
client_token,
|
|
28
|
+
encode_payload,
|
|
29
|
+
fanout_limit,
|
|
30
|
+
plan_fanout,
|
|
31
|
+
)
|
|
22
32
|
from microvm.throttle import Throttled
|
|
23
33
|
|
|
24
34
|
ACTIVE_STATES = {"PENDING", "RUNNING", "SUSPENDING", "SUSPENDED"}
|
|
@@ -203,6 +213,51 @@ class FleetManager:
|
|
|
203
213
|
client_token=client_token(lease),
|
|
204
214
|
)
|
|
205
215
|
|
|
216
|
+
# -- fan-out -----------------------------------------------------------------
|
|
217
|
+
def fanout_limit(self, baseline_mib: int, policy: LeasePolicy | None = None) -> FanoutLimit:
|
|
218
|
+
"""How many leases of `baseline_mib` can be in flight at once, from the applied
|
|
219
|
+
memory quota (when Service Quotas answered) and the policy's max_concurrency."""
|
|
220
|
+
return fanout_limit(baseline_mib, policy or LeasePolicy(),
|
|
221
|
+
memory_quota_gb=self.memory_quota_gb, launch_rate=self.tps("RunMicrovm"))
|
|
222
|
+
|
|
223
|
+
def plan(self, shards: int, baseline_mib: int, policy: LeasePolicy | None = None) -> LeasePlan:
|
|
224
|
+
"""Size a fan-out of `shards` leases without launching anything."""
|
|
225
|
+
return plan_fanout(shards, baseline_mib, policy or LeasePolicy(),
|
|
226
|
+
memory_quota_gb=self.memory_quota_gb, launch_rate=self.tps("RunMicrovm"))
|
|
227
|
+
|
|
228
|
+
def lease_many(
|
|
229
|
+
self,
|
|
230
|
+
image: str,
|
|
231
|
+
leases: list[Lease],
|
|
232
|
+
tasks: list[dict],
|
|
233
|
+
policy: LeasePolicy | None = None,
|
|
234
|
+
*,
|
|
235
|
+
baseline_mib: int,
|
|
236
|
+
version: str | None = None,
|
|
237
|
+
execution_role: str | None = None,
|
|
238
|
+
ingress: list[str] | None = None,
|
|
239
|
+
egress: list[str] | None = None,
|
|
240
|
+
) -> list[Microvm]:
|
|
241
|
+
"""Pre-flight: plan(len(leases), baseline_mib, policy).check(); then REFUSE (LeasePlanRejected)
|
|
242
|
+
if len(leases) > plan.concurrency ("N leases exceed the concurrency limit M; launch in waves");
|
|
243
|
+
launch through the shared bucket on the Fleet thread pool (max 8 workers), preserving order;
|
|
244
|
+
return the Microvm list."""
|
|
245
|
+
if len(leases) != len(tasks):
|
|
246
|
+
raise ValueError(f"{len(leases)} leases but {len(tasks)} tasks: pass one task per lease")
|
|
247
|
+
plan = self.plan(len(leases), baseline_mib, policy)
|
|
248
|
+
plan.check()
|
|
249
|
+
if len(leases) > plan.concurrency:
|
|
250
|
+
raise LeasePlanRejected(
|
|
251
|
+
f"{len(leases)} leases exceed the concurrency limit {plan.concurrency}; launch in waves")
|
|
252
|
+
|
|
253
|
+
def one(pair):
|
|
254
|
+
lease, task = pair
|
|
255
|
+
return self.lease(image, lease, task, policy, version=version, execution_role=execution_role,
|
|
256
|
+
ingress=ingress, egress=egress)
|
|
257
|
+
|
|
258
|
+
with futures.ThreadPoolExecutor(max_workers=Fleet.MAX_WORKERS) as pool:
|
|
259
|
+
return list(pool.map(one, zip(leases, tasks)))
|
|
260
|
+
|
|
206
261
|
def get(self, microvm_id: str) -> Microvm:
|
|
207
262
|
return Microvm.from_api(self.api.get_microvm(microvmIdentifier=microvm_id))
|
|
208
263
|
|
|
@@ -240,6 +295,9 @@ class FleetManager:
|
|
|
240
295
|
class Fleet:
|
|
241
296
|
"""Declarative fleet of microVMs from one image: scale up, down, drain."""
|
|
242
297
|
|
|
298
|
+
#: threads used for parallel launches and lifecycle sweeps
|
|
299
|
+
MAX_WORKERS = 8
|
|
300
|
+
|
|
243
301
|
manager: FleetManager
|
|
244
302
|
image: str
|
|
245
303
|
version: str | None = None
|
|
@@ -251,7 +309,7 @@ class Fleet:
|
|
|
251
309
|
egress: list[str] | None = None
|
|
252
310
|
execution_role: str | None = None
|
|
253
311
|
_pool: futures.ThreadPoolExecutor = field(
|
|
254
|
-
default_factory=lambda: futures.ThreadPoolExecutor(max_workers=
|
|
312
|
+
default_factory=lambda: futures.ThreadPoolExecutor(max_workers=Fleet.MAX_WORKERS), repr=False
|
|
255
313
|
)
|
|
256
314
|
|
|
257
315
|
# -- observation -------------------------------------------------------------
|