microvm-ctl 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/PKG-INFO +11 -4
  2. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/README.md +10 -3
  3. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/__init__.py +6 -2
  4. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/cli.py +308 -14
  5. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/endpoint.py +4 -3
  6. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/fleet.py +60 -2
  7. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/hooks/server.py +8 -4
  8. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/images.py +33 -0
  9. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/integrations/__init__.py +5 -2
  10. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/integrations/durable.py +121 -2
  11. microvm_ctl-0.3.0/microvm/integrations/stepfunctions.py +353 -0
  12. microvm_ctl-0.3.0/microvm/lease.py +303 -0
  13. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/monitor.py +93 -0
  14. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/playground/server.py +234 -19
  15. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/playground/static/index.html +74 -7
  16. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/PKG-INFO +11 -4
  17. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/SOURCES.txt +5 -0
  18. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/pyproject.toml +1 -1
  19. microvm_ctl-0.3.0/tests/test_cli_scale.py +378 -0
  20. microvm_ctl-0.3.0/tests/test_integrations_durable_scale.py +232 -0
  21. microvm_ctl-0.3.0/tests/test_integrations_scale.py +218 -0
  22. microvm_ctl-0.3.0/tests/test_monitor_jobs.py +110 -0
  23. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_playground.py +126 -0
  24. microvm_ctl-0.3.0/tests/test_scale.py +331 -0
  25. microvm_ctl-0.2.0/microvm/integrations/stepfunctions.py +0 -210
  26. microvm_ctl-0.2.0/microvm/lease.py +0 -133
  27. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/LICENSE +0 -0
  28. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/bootstrap.py +0 -0
  29. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/client.py +0 -0
  30. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/config.py +0 -0
  31. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/data/lambda-microvms/2025-09-09/endpoint-rule-set-1.json.gz +0 -0
  32. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/data/lambda-microvms/2025-09-09/paginators-1.json +0 -0
  33. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/data/lambda-microvms/2025-09-09/service-2.json.gz +0 -0
  34. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/hooks/__init__.py +0 -0
  35. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/playground/__init__.py +0 -0
  36. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm/throttle.py +0 -0
  37. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/dependency_links.txt +0 -0
  38. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/entry_points.txt +0 -0
  39. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/requires.txt +0 -0
  40. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/microvm_ctl.egg-info/top_level.txt +0 -0
  41. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/setup.cfg +0 -0
  42. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_config.py +0 -0
  43. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_cost_model.py +0 -0
  44. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_endpoint.py +0 -0
  45. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_fleet.py +0 -0
  46. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_hooks.py +0 -0
  47. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_hooks_lease.py +0 -0
  48. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_hooks_validate.py +0 -0
  49. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_integrations.py +0 -0
  50. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_integrations_durable.py +0 -0
  51. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_lease.py +0 -0
  52. {microvm_ctl-0.2.0 → microvm_ctl-0.3.0}/tests/test_throttle.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: microvm-ctl
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Control and execution plane for AWS Lambda MicroVMs: build images, launch Firecracker microVMs, scale fleets inside your quotas, call into them securely, and watch it all live.
5
5
  Author: Vivek Raja P S
6
6
  License: Apache-2.0
@@ -69,6 +69,7 @@ Lambda MicroVMs exposes the primitive under Lambda itself: a Firecracker VM with
69
69
  | In-VM hook runtime | zero-dependency server for `/ready`, `/validate`, `/run`, `/resume`, `/suspend`, `/terminate` plus your own routes | `microvm/hooks/server.py` |
70
70
  | Observability and cost | live fleet table, CloudWatch tail, a cost model that prices a session shape before you commit to it | `microvm/monitor.py` |
71
71
  | Account bootstrap | artifact bucket plus separate build and execution roles | `microvm/bootstrap.py` |
72
+ | Leases at scale | `LeasePolicy` ceilings, `plan` with the honest concurrency and worst-case cost, `lease_many`, `mvm lease asl --map`, `lease_map`, `mvm watch --image` | `microvm/lease.py`, `microvm/integrations/` |
72
73
  | Leases | `Lease`, `LeasePolicy`, `FleetManager.lease`: one task per VM, completed by the VM through a task token, callback id, HTTP, SQS, or EventBridge | `microvm/lease.py` |
73
74
  | Orchestrator integrations | generated Step Functions state machine and IAM (`mvm lease asl`, `mvm lease policy`), `lease_microvm` for Lambda durable functions | `microvm/integrations/` |
74
75
 
@@ -170,9 +171,9 @@ mvm playground # http://127.0.0.1:8765
170
171
  - [The playground](docs/playground.md): the local web app over the whole SDK, and how to host it.
171
172
  - [Integrations](docs/integrations.md): the lease contract for handing a VM to Step Functions, Lambda durable functions, or your own orchestrator, and what the plane should own.
172
173
 
173
- ## See it working: eight examples and the article series
174
+ ## See it working: thirteen examples and the article series
174
175
 
175
- The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds eight production-shaped examples, each a Dockerfile plus a single-file app, deployed and recorded on the live service:
176
+ The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds thirteen production-shaped examples, each a Dockerfile plus a single-file app, deployed and recorded on the live service:
176
177
 
177
178
  | Example | What it shows |
178
179
  |---|---|
@@ -184,12 +185,18 @@ The fastest way to understand the plane is to read the apps built on it. The com
184
185
  | [ci-runner](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/ci-runner) | clone, test, report, terminate; `--max-duration` as the runaway cap |
185
186
  | [pdf-service](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/pdf-service) | untrusted HTML rendered in the VM; idle policy sleeps it between bursts |
186
187
  | [multi-tenant-agents](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/multi-tenant-agents) | one VM per tenant, identity via `runHookPayload`, `run_payload_factory` on a `Fleet` |
188
+ | [handoff-agent](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/handoff-agent) | the one image every orchestrator leases: `on_lease`, phases, progress, heartbeats, typed failures, in-VM parallel steps |
189
+ | [stepfunctions-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/stepfunctions-handoff) | `runMicrovm.waitForTaskToken` machines from `mvm lease asl`, single lease and a governed Map fan-out |
190
+ | [durable-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/durable-handoff) | a Lambda durable function leases a VM with `lease_with_relaunch` and fans out with `lease_map` |
191
+ | [generic-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/generic-handoff) | any controller: the same lease over SQS, EventBridge, or an HTTP collector |
192
+ | [circuit-breaker](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/circuit-breaker) | running-memory metric, alarm, and `Fleet.drain` for every image: the account-wide stop behind `LeasePolicy` |
187
193
 
188
- The three-part article series **Building on AWS Lambda MicroVMs** on the AWS Builder Center walks through them:
194
+ The four-part article series **Building on AWS Lambda MicroVMs** on the AWS Builder Center walks through them:
189
195
 
190
196
  1. [Control and scale AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JIDTpz0ZgatSBv24drra3gEod9/control-and-scale-aws-lambda-microvms-with-microvm-ctl): this plane and its measurements.
191
197
  2. [Seven workloads Lambda could never run, until MicroVMs](https://builder.aws.com/content/3JJ2oNWY9EsZzivMMx044cSlrFQ/seven-workloads-lambda-could-never-run-until-microvms): the first seven examples through build, run, cost, and gotchas.
192
198
  3. [A kernel for every customer: scaling AI agents to 1,000 tenants on AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JJ7tASPSSUUTrcpWnWtxOuu8g3/a-kernel-for-every-customer-scaling-ai-agents-to-1000-tenants-on-aws-lambda-microvms-with-microvm-ctl): the multi-tenant finale with the decision guide.
199
+ 4. Hand a task to a MicroVM from anywhere (part 4, [source in awesome-microvm](https://github.com/Vivek0712/awesome-microvm/blob/main/blog/03-handoff.md), Builder Center link to follow): the lease contract, Step Functions, durable functions, your own controller, and leases at scale.
193
200
 
194
201
  The article sources and a long-form deep dive per example live under [awesome-microvm/blog](https://github.com/Vivek0712/awesome-microvm/tree/main/blog).
195
202
 
@@ -33,6 +33,7 @@ Lambda MicroVMs exposes the primitive under Lambda itself: a Firecracker VM with
33
33
  | In-VM hook runtime | zero-dependency server for `/ready`, `/validate`, `/run`, `/resume`, `/suspend`, `/terminate` plus your own routes | `microvm/hooks/server.py` |
34
34
  | Observability and cost | live fleet table, CloudWatch tail, a cost model that prices a session shape before you commit to it | `microvm/monitor.py` |
35
35
  | Account bootstrap | artifact bucket plus separate build and execution roles | `microvm/bootstrap.py` |
36
+ | Leases at scale | `LeasePolicy` ceilings, `plan` with the honest concurrency and worst-case cost, `lease_many`, `mvm lease asl --map`, `lease_map`, `mvm watch --image` | `microvm/lease.py`, `microvm/integrations/` |
36
37
  | Leases | `Lease`, `LeasePolicy`, `FleetManager.lease`: one task per VM, completed by the VM through a task token, callback id, HTTP, SQS, or EventBridge | `microvm/lease.py` |
37
38
  | Orchestrator integrations | generated Step Functions state machine and IAM (`mvm lease asl`, `mvm lease policy`), `lease_microvm` for Lambda durable functions | `microvm/integrations/` |
38
39
 
@@ -134,9 +135,9 @@ mvm playground # http://127.0.0.1:8765
134
135
  - [The playground](docs/playground.md): the local web app over the whole SDK, and how to host it.
135
136
  - [Integrations](docs/integrations.md): the lease contract for handing a VM to Step Functions, Lambda durable functions, or your own orchestrator, and what the plane should own.
136
137
 
137
- ## See it working: eight examples and the article series
138
+ ## See it working: thirteen examples and the article series
138
139
 
139
- The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds eight production-shaped examples, each a Dockerfile plus a single-file app, deployed and recorded on the live service:
140
+ The fastest way to understand the plane is to read the apps built on it. The companion repo [awesome-microvm](https://github.com/Vivek0712/awesome-microvm) holds thirteen production-shaped examples, each a Dockerfile plus a single-file app, deployed and recorded on the live service:
140
141
 
141
142
  | Example | What it shows |
142
143
  |---|---|
@@ -148,12 +149,18 @@ The fastest way to understand the plane is to read the apps built on it. The com
148
149
  | [ci-runner](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/ci-runner) | clone, test, report, terminate; `--max-duration` as the runaway cap |
149
150
  | [pdf-service](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/pdf-service) | untrusted HTML rendered in the VM; idle policy sleeps it between bursts |
150
151
  | [multi-tenant-agents](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/multi-tenant-agents) | one VM per tenant, identity via `runHookPayload`, `run_payload_factory` on a `Fleet` |
152
+ | [handoff-agent](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/handoff-agent) | the one image every orchestrator leases: `on_lease`, phases, progress, heartbeats, typed failures, in-VM parallel steps |
153
+ | [stepfunctions-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/stepfunctions-handoff) | `runMicrovm.waitForTaskToken` machines from `mvm lease asl`, single lease and a governed Map fan-out |
154
+ | [durable-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/durable-handoff) | a Lambda durable function leases a VM with `lease_with_relaunch` and fans out with `lease_map` |
155
+ | [generic-handoff](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/generic-handoff) | any controller: the same lease over SQS, EventBridge, or an HTTP collector |
156
+ | [circuit-breaker](https://github.com/Vivek0712/awesome-microvm/tree/main/examples/circuit-breaker) | running-memory metric, alarm, and `Fleet.drain` for every image: the account-wide stop behind `LeasePolicy` |
151
157
 
152
- The three-part article series **Building on AWS Lambda MicroVMs** on the AWS Builder Center walks through them:
158
+ The four-part article series **Building on AWS Lambda MicroVMs** on the AWS Builder Center walks through them:
153
159
 
154
160
  1. [Control and scale AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JIDTpz0ZgatSBv24drra3gEod9/control-and-scale-aws-lambda-microvms-with-microvm-ctl): this plane and its measurements.
155
161
  2. [Seven workloads Lambda could never run, until MicroVMs](https://builder.aws.com/content/3JJ2oNWY9EsZzivMMx044cSlrFQ/seven-workloads-lambda-could-never-run-until-microvms): the first seven examples through build, run, cost, and gotchas.
156
162
  3. [A kernel for every customer: scaling AI agents to 1,000 tenants on AWS Lambda MicroVMs with microvm-ctl](https://builder.aws.com/content/3JJ7tASPSSUUTrcpWnWtxOuu8g3/a-kernel-for-every-customer-scaling-ai-agents-to-1000-tenants-on-aws-lambda-microvms-with-microvm-ctl): the multi-tenant finale with the decision guide.
163
+ 4. Hand a task to a MicroVM from anywhere (part 4, [source in awesome-microvm](https://github.com/Vivek0712/awesome-microvm/blob/main/blog/03-handoff.md), Builder Center link to follow): the lease contract, Step Functions, durable functions, your own controller, and leases at scale.
157
164
 
158
165
  The article sources and a long-form deep dive per example live under [awesome-microvm/blog](https://github.com/Vivek0712/awesome-microvm/tree/main/blog).
159
166
 
@@ -9,10 +9,10 @@ from microvm.config import PlaneConfig
9
9
  from microvm.endpoint import EndpointClient, EndpointError
10
10
  from microvm.fleet import Fleet, FleetManager
11
11
  from microvm.images import ImageBuilder, ImageBuildError
12
- from microvm.lease import Lease, LeasePolicy
12
+ from microvm.lease import FanoutLimit, Lease, LeasePlan, LeasePlanRejected, LeasePolicy, plan_fanout
13
13
  from microvm.monitor import CostModel, FleetMonitor
14
14
 
15
- __version__ = "0.2.0"
15
+ __version__ = "0.3.0"
16
16
 
17
17
  __all__ = [
18
18
  "microvm_client",
@@ -24,6 +24,10 @@ __all__ = [
24
24
  "FleetManager",
25
25
  "Lease",
26
26
  "LeasePolicy",
27
+ "LeasePlan",
28
+ "LeasePlanRejected",
29
+ "FanoutLimit",
30
+ "plan_fanout",
27
31
  "EndpointClient",
28
32
  "EndpointError",
29
33
  "FleetMonitor",
@@ -11,7 +11,8 @@
11
11
  mvm drain IMAGE terminate the whole fleet
12
12
  mvm call ID /path [-X POST -d '{}'] authenticated request into the VM
13
13
  mvm status ID | watch ID job telemetry from the hook runtime (/status, /events)
14
- mvm lease asl|policy|run Step Functions ASL, IAM statements, one manual lease
14
+ mvm watch --image NAME one live table for every leased VM of an image
15
+ mvm lease asl|policy|run|plan Step Functions ASL, IAM statements, leases, fan-out sizing
15
16
  mvm top [--image NAME] [--watch] live fleet dashboard
16
17
  mvm logs IMAGE CloudWatch tail
17
18
  mvm cost [--memory-gb 2 ...] session economics
@@ -95,6 +96,26 @@ def cmd_quotas(args):
95
96
  if not applied:
96
97
  console.print("[dim]Service Quotas did not answer (missing servicequotas:GetServiceQuota?); "
97
98
  "showing published defaults.[/]")
99
+ for line in fanout_tier_lines(mem):
100
+ console.print(line)
101
+
102
+
103
+ FANOUT_TIERS_GB = (0.5, 1, 2, 4, 8)
104
+
105
+
106
+ def fanout_tier_lines(memory_quota_gb) -> list[str]:
107
+ """One line per image size tier: how many can be in flight at once under the applied
108
+ memory quota, the same arithmetic `mvm lease plan` uses."""
109
+ from microvm.lease import DEFAULT_FANOUT_LIMIT, LeasePolicy, fanout_limit
110
+ if memory_quota_gb is None:
111
+ return [f"[dim]memory quota unknown: fan-outs default to {DEFAULT_FANOUT_LIMIT} at once "
112
+ f"(MVM_LEASE_MAX_CONCURRENCY overrides)[/]"]
113
+ policy = LeasePolicy()
114
+ lines = []
115
+ for tier in FANOUT_TIERS_GB:
116
+ lim = fanout_limit(int(tier * 1024), policy, memory_quota_gb=memory_quota_gb, launch_rate=1.0)
117
+ lines.append(f"{tier:g} GB images: [bold]{lim.limit}[/] at once")
118
+ return lines
98
119
 
99
120
 
100
121
  # ---------------------------------------------------------------- images
@@ -303,7 +324,88 @@ def cmd_status(args):
303
324
  console.print(_log_line(entry) if isinstance(entry, dict) else str(entry))
304
325
 
305
326
 
327
+ def _answered(row: dict) -> bool:
328
+ """job_status gives a member that did not answer /status only microvm_id and error."""
329
+ return "phase" in row
330
+
331
+
332
+ def _row_state(row: dict) -> str:
333
+ if not _answered(row):
334
+ return "[red]unreachable[/]"
335
+ if row.get("done"):
336
+ return "[red]failed[/]" if row.get("error") else "[bold green]done[/]"
337
+ if row.get("lost"):
338
+ return "[yellow]lost[/]"
339
+ return "[cyan]running[/]"
340
+
341
+
342
+ def _fleet_job_view(rows: list[dict], image: str):
343
+ """The `mvm watch --image` body: one row per RUNNING member plus the aggregate footer."""
344
+ from rich.console import Group
345
+ t = Table(title=f"fleet job: {image}", header_style="bold magenta")
346
+ for c in ("microvm id", "lease id", "phase", "progress", "elapsed", "state"):
347
+ t.add_column(c)
348
+ for r in rows:
349
+ prog = r.get("progress")
350
+ if isinstance(prog, dict):
351
+ done, total = prog.get("done"), prog.get("total")
352
+ prog = f"{done}/{total}" if total else (str(done) if done is not None else "-")
353
+ elapsed = r.get("elapsed_s")
354
+ t.add_row(
355
+ str(r.get("microvm_id", "-")), str(r.get("lease_id") or "-"),
356
+ str(r.get("phase") or "-") if _answered(r) else f"[dim]{r.get('error') or '-'}[/]",
357
+ str(prog if prog is not None else "-"),
358
+ f"{elapsed:.0f} s" if isinstance(elapsed, (int, float)) else "-", _row_state(r),
359
+ )
360
+ return Group(t, fleet_job_footer(rows))
361
+
362
+
363
+ def fleet_job_footer(rows: list[dict]) -> str:
364
+ """'done D/N, running R, lost L, slowest <id> <phase> <elapsed>' over job_status rows."""
365
+ answered = [r for r in rows if _answered(r)]
366
+ done = sum(1 for r in answered if r.get("done"))
367
+ lost = sum(1 for r in answered if r.get("lost") and not r.get("done"))
368
+ running = len(answered) - done - lost
369
+ text = f"done {done}/{len(rows)}, running {running}, lost {lost}"
370
+ if len(answered) < len(rows):
371
+ text += f", unreachable {len(rows) - len(answered)}"
372
+ timed = [r for r in answered if isinstance(r.get("elapsed_s"), (int, float))]
373
+ active = [r for r in timed if not r.get("done")] or timed
374
+ if active:
375
+ slow = max(active, key=lambda r: r["elapsed_s"])
376
+ text += f", slowest {slow.get('microvm_id')} {slow.get('phase') or '-'} {slow['elapsed_s']:.0f} s"
377
+ return text
378
+
379
+
380
+ def fleet_job_finished(rows: list[dict], seen_members: bool) -> bool:
381
+ """Stop when every member is done, or when members were seen earlier and are now gone."""
382
+ if rows:
383
+ return all(r.get("done") for r in rows)
384
+ return seen_members
385
+
386
+
387
+ def cmd_watch_image(args):
388
+ from rich.live import Live
389
+ mon = FleetMonitor(_cfg(args))
390
+ kw = {"port": args.port} if args.port else {}
391
+ rows = mon.job_status(args.image, **kw)
392
+ seen = bool(rows)
393
+ deadline = time.time() + args.timeout
394
+ with Live(_fleet_job_view(rows, args.image), refresh_per_second=2, console=console) as live:
395
+ while not fleet_job_finished(rows, seen) and time.time() < deadline:
396
+ time.sleep(args.interval)
397
+ rows = mon.job_status(args.image, **kw)
398
+ seen = seen or bool(rows)
399
+ live.update(_fleet_job_view(rows, args.image))
400
+ # the live view already ends on the footer; a trailing newline keeps the shell prompt off it
401
+ console.print()
402
+
403
+
306
404
  def cmd_watch(args):
405
+ if args.image:
406
+ return cmd_watch_image(args)
407
+ if not args.id:
408
+ sys.exit("mvm watch: pass a microVM ID or --image NAME")
307
409
  from rich.live import Live
308
410
  client = EndpointClient(_cfg(args), args.id, ports=[args.port] if args.port else None)
309
411
  snap = client.status()
@@ -318,9 +420,41 @@ def cmd_watch(args):
318
420
 
319
421
 
320
422
  # ---------------------------------------------------------------- leases
423
+ POLICY_FLAGS = (
424
+ ("budget", "budget_s"), ("heartbeat", "heartbeat_timeout_s"), ("slack", "slack_s"),
425
+ ("max_concurrency", "max_concurrency"), ("max_vm_seconds", "max_vm_seconds"),
426
+ ("approval_usd", "approval_usd"),
427
+ )
428
+
429
+
321
430
  def _lease_policy(args):
431
+ """LeasePolicy from the MVM_LEASE_* environment, then any flag that was given."""
322
432
  from microvm.lease import LeasePolicy
323
- return LeasePolicy(budget_s=args.budget, heartbeat_timeout_s=args.heartbeat, slack_s=args.slack)
433
+ policy = LeasePolicy.from_env()
434
+ for flag, field in POLICY_FLAGS:
435
+ value = getattr(args, flag, None)
436
+ if value is not None:
437
+ setattr(policy, field, value)
438
+ return policy
439
+
440
+
441
+ def _baseline_mib(args, cfg) -> int:
442
+ """--baseline-mib when given, else the image version's minimumMemoryInMiB."""
443
+ given = getattr(args, "baseline_mib", None)
444
+ if given:
445
+ return given
446
+ return ImageBuilder(cfg).baseline_mib(args.image, getattr(args, "version", None))
447
+
448
+
449
+ def _fill_template(obj, i: int):
450
+ """Replace `{i}` in every string value of a JSON-like object with the shard index."""
451
+ if isinstance(obj, str):
452
+ return obj.replace("{i}", str(i))
453
+ if isinstance(obj, list):
454
+ return [_fill_template(v, i) for v in obj]
455
+ if isinstance(obj, dict):
456
+ return {k: _fill_template(v, i) for k, v in obj.items()}
457
+ return obj
324
458
 
325
459
 
326
460
  def _image_arn_offline(image: str, region: str, role_arn: str | None, profile) -> str:
@@ -341,14 +475,40 @@ def cmd_lease_asl(args):
341
475
  role = args.execution_role or cfg.execution_role_arn
342
476
  if not role:
343
477
  sys.exit("mvm lease asl: --execution-role ARN (or MVM_EXECUTION_ROLE_ARN) is required")
478
+ kw = {}
479
+ if args.map:
480
+ from microvm.integrations.stepfunctions import FanoutSpec
481
+ max_concurrency = args.max_concurrency
482
+ if max_concurrency is None:
483
+ max_concurrency = _computed_fanout_limit(args, cfg)
484
+ kw["fanout"] = FanoutSpec(
485
+ items_expr=args.items_expr, max_concurrency=max_concurrency,
486
+ approval_topic_arn=args.approval_topic, approve_above_shards=args.approve_above_shards,
487
+ )
344
488
  asl = lease_state_machine(
345
489
  image_arn=_image_arn_offline(args.image, cfg.region, role, cfg.profile),
346
490
  execution_role_arn=role, policy=_lease_policy(args), region=cfg.region,
347
- name=args.name, task_expr=args.task_expr, heartbeat_s=args.heartbeat_every,
491
+ name=args.name, task_expr=args.task_expr, heartbeat_s=args.heartbeat_every, **kw,
348
492
  )
349
493
  print(json.dumps(asl, indent=2))
350
494
 
351
495
 
496
+ def _computed_fanout_limit(args, cfg) -> int:
497
+ """Map MaxConcurrency when --max-concurrency is omitted: the plane's fan-out limit for the
498
+ image's baseline, with the reason on stderr so the generated ASL is explainable."""
499
+ from microvm.lease import DEFAULT_FANOUT_LIMIT
500
+ try:
501
+ baseline = _baseline_mib(args, cfg)
502
+ limit = FleetManager(cfg).fanout_limit(baseline, _lease_policy(args))
503
+ except Exception as e: # no credentials, unknown image: still emit a usable machine
504
+ print(f"mvm lease asl: could not compute the fan-out limit ({e}); "
505
+ f"using MaxConcurrency {DEFAULT_FANOUT_LIMIT} (default)", file=sys.stderr)
506
+ return DEFAULT_FANOUT_LIMIT
507
+ print(f"mvm lease asl: MaxConcurrency {limit.limit} ({limit.reason}; "
508
+ f"{baseline} MiB baseline)", file=sys.stderr)
509
+ return limit.limit
510
+
511
+
352
512
  def cmd_lease_policy(args):
353
513
  from microvm.integrations.stepfunctions import iam_statements, orchestrator_statements, policy_document
354
514
  cfg = _cfg(args)
@@ -363,6 +523,8 @@ def cmd_lease_policy(args):
363
523
  def cmd_lease_run(args):
364
524
  from microvm.lease import Lease
365
525
  cfg = _cfg(args)
526
+ if args.shards is not None:
527
+ return cmd_lease_run_many(args, cfg)
366
528
  if args.kind != "none" and not args.token:
367
529
  sys.exit(f"mvm lease run: --token is required for kind {args.kind}")
368
530
  try:
@@ -383,6 +545,95 @@ def cmd_lease_run(args):
383
545
  console.print(f" now {_state(vm.state)}; follow it with: mvm watch {vm.microvm_id}")
384
546
 
385
547
 
548
+ def cmd_lease_run_many(args, cfg):
549
+ """`mvm lease run IMAGE --shards N`: plan, refuse or launch every shard through lease_many."""
550
+ from microvm.lease import Lease, LeasePlanRejected
551
+ if args.shards < 1:
552
+ sys.exit(f"mvm lease run: --shards must be at least 1, got {args.shards}")
553
+ kind = args.kind if (args.kind != "none" or not args.token_template) else "none"
554
+ if args.token_template and kind == "none":
555
+ sys.exit("mvm lease run: --token-template needs --kind (sfn, durable, http, sqs, eventbridge)")
556
+ if kind != "none" and not args.token_template:
557
+ sys.exit(f"mvm lease run: --token-template with {{i}} is required for kind {kind} and --shards")
558
+ try:
559
+ template = json.loads(args.task_template) if args.task_template else (
560
+ json.loads(args.task) if args.task else {})
561
+ except ValueError as e:
562
+ sys.exit(f"mvm lease run: --task-template is not valid JSON: {e}")
563
+ if not isinstance(template, dict):
564
+ sys.exit("mvm lease run: --task-template must be a JSON object")
565
+ leases, tasks = [], []
566
+ for i in range(args.shards):
567
+ token = args.token_template.replace("{i}", str(i)) if args.token_template else ""
568
+ lease_id = args.id.replace("{i}", str(i)) if args.id else None
569
+ leases.append(Lease(kind=kind, token=token, region=cfg.region, target=args.target,
570
+ heartbeat_s=args.heartbeat_every, id=lease_id))
571
+ tasks.append(_fill_template(template, i))
572
+ policy = _lease_policy(args)
573
+ fm = FleetManager(cfg)
574
+ baseline = _baseline_mib(args, cfg)
575
+ plan = fm.plan(args.shards, baseline, policy)
576
+ console.print(plan.summary())
577
+ if plan.rejected:
578
+ sys.exit(2)
579
+ if plan.needs_approval and not args.approve:
580
+ print(f"mvm lease run: the plan needs approval (worst case ${plan.worst_case_usd:.2f} above "
581
+ f"approval_usd ${policy.approval_usd:g}); re-run with --approve", file=sys.stderr)
582
+ sys.exit(3)
583
+ if args.shards > plan.concurrency:
584
+ print(f"mvm lease run: {args.shards} leases exceed the concurrency limit {plan.concurrency}; "
585
+ f"launch in waves (--shards {plan.concurrency} at a time)", file=sys.stderr)
586
+ sys.exit(2)
587
+ try:
588
+ vms = fm.lease_many(args.image, leases, tasks, policy, baseline_mib=baseline, version=args.version,
589
+ execution_role=args.execution_role)
590
+ except LeasePlanRejected as e:
591
+ sys.exit(f"mvm lease run: {e}")
592
+ for i, vm in enumerate(vms):
593
+ console.print(f"[bold green]✓[/] shard {i}: {vm.microvm_id} {_state(vm.state)} "
594
+ f"[link=https://{vm.endpoint}]{vm.endpoint}[/link] lease {kind}")
595
+ if args.wait:
596
+ for vm in vms:
597
+ fm.wait_until(vm.microvm_id, "RUNNING")
598
+ console.print(f" all {len(vms)} RUNNING; follow them with: mvm watch --image {args.image}")
599
+
600
+
601
+ def _plan_table(plan) -> Table:
602
+ lim = plan.limit
603
+ t = Table(title=f"lease plan: {plan.shards} shards", header_style="bold magenta", show_header=False)
604
+ t.add_column("field", style="cyan")
605
+ t.add_column("value")
606
+ t.add_row("shards", str(plan.shards))
607
+ t.add_row("baseline", f"{plan.baseline_mib} MiB")
608
+ t.add_row("memory quota", f"{lim.memory_quota_gb:g} GB" if lim.memory_quota_gb is not None
609
+ else "[dim]unknown[/]")
610
+ t.add_row("launch rate", f"{lim.launch_rate:g}/s")
611
+ t.add_row("concurrency", f"{plan.concurrency} [dim]({lim.reason})[/]")
612
+ t.add_row("waves", str(plan.waves))
613
+ t.add_row("launch to all running", f"~{plan.launch_to_all_running_s:.1f} s")
614
+ t.add_row("worst case VM-s", str(plan.worst_case_vm_seconds))
615
+ t.add_row("worst case USD", f"${plan.worst_case_usd:.4f}")
616
+ t.add_row("approval needed", "[yellow]yes[/]" if plan.needs_approval else "no")
617
+ t.add_row("rejected", f"[red]{plan.rejected}[/]" if plan.rejected else "no")
618
+ return t
619
+
620
+
621
+ def cmd_lease_plan(args):
622
+ """Exit 0 when the plan can launch, 2 when it is rejected, 3 when it needs approval."""
623
+ cfg = _cfg(args)
624
+ fm = FleetManager(cfg)
625
+ plan = fm.plan(args.shards, _baseline_mib(args, cfg), _lease_policy(args))
626
+ if args.json:
627
+ print(json.dumps(plan.to_dict(), indent=2))
628
+ else:
629
+ console.print(_plan_table(plan))
630
+ console.print(plan.summary())
631
+ if plan.rejected:
632
+ sys.exit(2)
633
+ if plan.needs_approval:
634
+ sys.exit(3)
635
+
636
+
386
637
  def cmd_playground(args):
387
638
  from microvm.playground import serve
388
639
  serve(host=args.host, port=args.port, dry_run=args.dry_run, open_browser=not args.no_open, cfg=_cfg(args))
@@ -486,21 +737,36 @@ def main(argv: list[str] | None = None):
486
737
  st.add_argument("--port", type=int, default=None, help="non-default app port")
487
738
  st.set_defaults(fn=cmd_status)
488
739
 
489
- w = sub.add_parser("watch", help="live job table plus streamed log lines (GET /events)")
490
- w.add_argument("id")
740
+ w = sub.add_parser("watch", help="live job table for one VM (GET /events) or every VM of an image")
741
+ w.add_argument("id", nargs="?", help="microVM id (omit with --image)")
742
+ w.add_argument("--image", default=None, help="one table for every RUNNING member of this image")
491
743
  w.add_argument("--port", type=int, default=None, help="non-default app port")
744
+ w.add_argument("--interval", type=float, default=2, help="seconds between polls with --image")
492
745
  w.add_argument("--timeout", type=float, default=600, help="stop after N seconds")
493
746
  w.set_defaults(fn=cmd_watch)
494
747
 
495
748
  le = sub.add_parser("lease", help="hand a VM a task through a lease").add_subparsers(
496
749
  dest="sub", required=True)
497
750
 
498
- def _policy_flags(sp):
499
- sp.add_argument("--budget", type=int, default=900,
500
- help="seconds the orchestrator waits (task timeout)")
501
- sp.add_argument("--heartbeat", type=int, default=120, help="heartbeat timeout in seconds")
502
- sp.add_argument("--slack", type=int, default=120, help="VM outlives the budget by this many seconds")
503
- sp.add_argument("--heartbeat-every", type=int, default=30, help="seconds between VM heartbeats")
751
+ def _policy_flags(sp, heartbeat_every=True):
752
+ sp.add_argument("--budget", type=int, default=None,
753
+ help="seconds the orchestrator waits (task timeout; MVM_LEASE_BUDGET_S or 900)")
754
+ sp.add_argument("--heartbeat", type=int, default=None,
755
+ help="heartbeat timeout in seconds (MVM_LEASE_HEARTBEAT_TIMEOUT_S or 120)")
756
+ sp.add_argument("--slack", type=int, default=None,
757
+ help="VM outlives the budget by this many seconds (MVM_LEASE_SLACK_S or 120)")
758
+ sp.add_argument("--max-concurrency", type=int, default=None,
759
+ help="VMs in flight per fan-out (MVM_LEASE_MAX_CONCURRENCY; default: memory quota)")
760
+ sp.add_argument("--max-vm-seconds", type=int, default=None,
761
+ help="worst-case VM-seconds a fan-out may commit to (MVM_LEASE_MAX_VM_SECONDS)")
762
+ sp.add_argument("--approval-usd", type=float, default=None,
763
+ help="worst-case USD above which a plan needs approval (MVM_LEASE_APPROVAL_USD)")
764
+ if heartbeat_every:
765
+ sp.add_argument("--heartbeat-every", type=int, default=30, help="seconds between VM heartbeats")
766
+
767
+ def _baseline_flag(sp):
768
+ sp.add_argument("--baseline-mib", type=int, default=None,
769
+ help="image memory baseline in MiB (default: read from the image version)")
504
770
 
505
771
  la = le.add_parser("asl", help="print the Step Functions state machine (JSONata) for one lease")
506
772
  la.add_argument("--image", required=True, help="image name or ARN")
@@ -508,6 +774,15 @@ def main(argv: list[str] | None = None):
508
774
  help="VM execution role ARN (default: MVM_EXECUTION_ROLE_ARN)")
509
775
  la.add_argument("--name", default="Lease", help="name of the lease state")
510
776
  la.add_argument("--task-expr", default="$states.input", help="JSONata expression for the task")
777
+ la.add_argument("--map", action="store_true",
778
+ help="emit a Map over --items-expr: one lease per item, MaxConcurrency from the plane")
779
+ la.add_argument("--items-expr", default="$states.input.shards",
780
+ help="JSONata array of task objects for --map")
781
+ la.add_argument("--approval-topic", default=None,
782
+ help="SNS topic ARN: plans above --approve-above-shards wait for a task token")
783
+ la.add_argument("--approve-above-shards", type=int, default=None,
784
+ help="shard count above which the Map waits for approval on --approval-topic")
785
+ _baseline_flag(la)
511
786
  _policy_flags(la)
512
787
  la.set_defaults(fn=cmd_lease_asl)
513
788
 
@@ -518,19 +793,38 @@ def main(argv: list[str] | None = None):
518
793
  lp.add_argument("--execution-role", default=None, help="VM execution role for iam:PassRole")
519
794
  lp.set_defaults(fn=cmd_lease_policy)
520
795
 
521
- lr = le.add_parser("run", help="launch one lease by hand (manual tests)")
796
+ lr = le.add_parser("run", help="launch one lease by hand, or --shards N of them through the plan")
522
797
  lr.add_argument("image")
523
- lr.add_argument("--kind", required=True, choices=["sfn", "durable", "http", "sqs", "eventbridge", "none"])
798
+ lr.add_argument("--kind", default="none",
799
+ choices=["sfn", "durable", "http", "sqs", "eventbridge", "none"])
524
800
  lr.add_argument("--token", default=None, help="task token, callback id, or bearer (not for kind none)")
525
801
  lr.add_argument("--target", default=None, help="http URL, SQS queue URL, or event bus name")
526
802
  lr.add_argument("--task", default=None, help="task JSON (pointers, not bodies; 4096 chars total)")
527
- lr.add_argument("--id", default=None, help="human label carried as lease.id")
803
+ lr.add_argument("--id", default=None, help="human label carried as lease.id ({i} = shard index)")
528
804
  lr.add_argument("--version", help="image version (default: latest ACTIVE)")
529
805
  lr.add_argument("--execution-role", default=None)
530
806
  lr.add_argument("--wait", action="store_true", help="poll until RUNNING")
807
+ lr.add_argument("--shards", type=int, default=None,
808
+ help="launch N leases at once after planning them (refuses above the limit)")
809
+ lr.add_argument("--task-template", default=None,
810
+ help="task JSON for --shards; {i} in string values becomes the shard index")
811
+ lr.add_argument("--token-template", default=None,
812
+ help="per-shard token for --shards with a kind other than none; {i} = shard index")
813
+ lr.add_argument("--approve", action="store_true",
814
+ help="launch even when the plan's worst case is above approval_usd")
815
+ _baseline_flag(lr)
531
816
  _policy_flags(lr)
532
817
  lr.set_defaults(fn=cmd_lease_run)
533
818
 
819
+ lpl = le.add_parser("plan", help="size a fan-out: concurrency, waves, launch time, worst case cost")
820
+ lpl.add_argument("--image", required=True, help="image name (baseline read from its ACTIVE version)")
821
+ lpl.add_argument("--shards", type=int, required=True, help="how many leases the fan-out launches")
822
+ lpl.add_argument("--version", default=None, help="image version to read the baseline from")
823
+ lpl.add_argument("--json", action="store_true", help="print the plan as JSON")
824
+ _baseline_flag(lpl)
825
+ _policy_flags(lpl, heartbeat_every=False)
826
+ lpl.set_defaults(fn=cmd_lease_plan)
827
+
534
828
  tp = sub.add_parser("top", help="live fleet dashboard")
535
829
  tp.add_argument("--image")
536
830
  tp.add_argument("--watch", action="store_true")
@@ -129,12 +129,13 @@ class EndpointClient:
129
129
  raise EndpointError(f"{self.microvm_id} not serving {path} after {timeout}s")
130
130
 
131
131
  # -- job telemetry (HookApp built-in /status and /events) -------------------
132
- def status(self, since: int | None = None) -> dict:
132
+ def status(self, since: int | None = None, *, timeout: float = 10, max_attempts: int = 6) -> dict:
133
133
  """The hook runtime's job snapshot (`GET /status`): phase, progress, counters,
134
134
  the log tail, and the lease state. `since` returns only log lines with a
135
- sequence number above it."""
135
+ sequence number above it. `timeout` and `max_attempts` bound one poll, which
136
+ matters when many members are polled on a schedule."""
136
137
  path = "/status" if since is None else f"/status?since={int(since)}"
137
- resp = self.get(path, timeout=10)
138
+ resp = self.get(path, timeout=timeout, max_attempts=max_attempts)
138
139
  if resp.status_code != 200:
139
140
  raise EndpointError(f"{self.microvm_id} answered {resp.status_code} on {path}")
140
141
  return resp.json()
@@ -18,7 +18,17 @@ from typing import Callable
18
18
 
19
19
  from microvm.client import image_arn, microvm_client
20
20
  from microvm.config import TPS, PlaneConfig
21
- from microvm.lease import Lease, LeasePolicy, client_token, encode_payload
21
+ from microvm.lease import (
22
+ FanoutLimit,
23
+ Lease,
24
+ LeasePlan,
25
+ LeasePlanRejected,
26
+ LeasePolicy,
27
+ client_token,
28
+ encode_payload,
29
+ fanout_limit,
30
+ plan_fanout,
31
+ )
22
32
  from microvm.throttle import Throttled
23
33
 
24
34
  ACTIVE_STATES = {"PENDING", "RUNNING", "SUSPENDING", "SUSPENDED"}
@@ -203,6 +213,51 @@ class FleetManager:
203
213
  client_token=client_token(lease),
204
214
  )
205
215
 
216
+ # -- fan-out -----------------------------------------------------------------
217
+ def fanout_limit(self, baseline_mib: int, policy: LeasePolicy | None = None) -> FanoutLimit:
218
+ """How many leases of `baseline_mib` can be in flight at once, from the applied
219
+ memory quota (when Service Quotas answered) and the policy's max_concurrency."""
220
+ return fanout_limit(baseline_mib, policy or LeasePolicy(),
221
+ memory_quota_gb=self.memory_quota_gb, launch_rate=self.tps("RunMicrovm"))
222
+
223
+ def plan(self, shards: int, baseline_mib: int, policy: LeasePolicy | None = None) -> LeasePlan:
224
+ """Size a fan-out of `shards` leases without launching anything."""
225
+ return plan_fanout(shards, baseline_mib, policy or LeasePolicy(),
226
+ memory_quota_gb=self.memory_quota_gb, launch_rate=self.tps("RunMicrovm"))
227
+
228
+ def lease_many(
229
+ self,
230
+ image: str,
231
+ leases: list[Lease],
232
+ tasks: list[dict],
233
+ policy: LeasePolicy | None = None,
234
+ *,
235
+ baseline_mib: int,
236
+ version: str | None = None,
237
+ execution_role: str | None = None,
238
+ ingress: list[str] | None = None,
239
+ egress: list[str] | None = None,
240
+ ) -> list[Microvm]:
241
+ """Pre-flight: plan(len(leases), baseline_mib, policy).check(); then REFUSE (LeasePlanRejected)
242
+ if len(leases) > plan.concurrency ("N leases exceed the concurrency limit M; launch in waves");
243
+ launch through the shared bucket on the Fleet thread pool (max 8 workers), preserving order;
244
+ return the Microvm list."""
245
+ if len(leases) != len(tasks):
246
+ raise ValueError(f"{len(leases)} leases but {len(tasks)} tasks: pass one task per lease")
247
+ plan = self.plan(len(leases), baseline_mib, policy)
248
+ plan.check()
249
+ if len(leases) > plan.concurrency:
250
+ raise LeasePlanRejected(
251
+ f"{len(leases)} leases exceed the concurrency limit {plan.concurrency}; launch in waves")
252
+
253
+ def one(pair):
254
+ lease, task = pair
255
+ return self.lease(image, lease, task, policy, version=version, execution_role=execution_role,
256
+ ingress=ingress, egress=egress)
257
+
258
+ with futures.ThreadPoolExecutor(max_workers=Fleet.MAX_WORKERS) as pool:
259
+ return list(pool.map(one, zip(leases, tasks)))
260
+
206
261
  def get(self, microvm_id: str) -> Microvm:
207
262
  return Microvm.from_api(self.api.get_microvm(microvmIdentifier=microvm_id))
208
263
 
@@ -240,6 +295,9 @@ class FleetManager:
240
295
  class Fleet:
241
296
  """Declarative fleet of microVMs from one image: scale up, down, drain."""
242
297
 
298
+ #: threads used for parallel launches and lifecycle sweeps
299
+ MAX_WORKERS = 8
300
+
243
301
  manager: FleetManager
244
302
  image: str
245
303
  version: str | None = None
@@ -251,7 +309,7 @@ class Fleet:
251
309
  egress: list[str] | None = None
252
310
  execution_role: str | None = None
253
311
  _pool: futures.ThreadPoolExecutor = field(
254
- default_factory=lambda: futures.ThreadPoolExecutor(max_workers=8), repr=False
312
+ default_factory=lambda: futures.ThreadPoolExecutor(max_workers=Fleet.MAX_WORKERS), repr=False
255
313
  )
256
314
 
257
315
  # -- observation -------------------------------------------------------------