sequence-base 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sequence_base/__init__.py +134 -0
- sequence_base/auth.py +35 -0
- sequence_base/codec/__init__.py +25 -0
- sequence_base/codec/image.py +136 -0
- sequence_base/codec/resize.py +147 -0
- sequence_base/constants/__init__.py +0 -0
- sequence_base/constants/env.py +14 -0
- sequence_base/constants/headers.py +5 -0
- sequence_base/constants/version.py +9 -0
- sequence_base/contract/__init__.py +111 -0
- sequence_base/contract/accounts.py +93 -0
- sequence_base/contract/act.py +89 -0
- sequence_base/contract/benchmarks.py +130 -0
- sequence_base/contract/connect.py +53 -0
- sequence_base/contract/deployments.py +197 -0
- sequence_base/contract/errors.py +45 -0
- sequence_base/contract/eval.py +201 -0
- sequence_base/contract/flags.py +92 -0
- sequence_base/contract/policy.py +172 -0
- sequence_base/errors/__init__.py +0 -0
- sequence_base/errors/transport.py +36 -0
- sequence_base/obs/__init__.py +0 -0
- sequence_base/obs/log.py +57 -0
- sequence_base/obs/rid.py +31 -0
- sequence_base/schema_export.py +238 -0
- sequence_base/schemas/ActionChunk.json +102 -0
- sequence_base/schemas/ActionContract.json +43 -0
- sequence_base/schemas/ActionStep.json +23 -0
- sequence_base/schemas/AdapterTransform.json +42 -0
- sequence_base/schemas/Autoscaler.json +43 -0
- sequence_base/schemas/BenchmarkRegisterRequest.json +302 -0
- sequence_base/schemas/BenchmarkSpec.json +401 -0
- sequence_base/schemas/BootstrapResponse.json +76 -0
- sequence_base/schemas/ClampInfo.json +27 -0
- sequence_base/schemas/CodeBundle.json +27 -0
- sequence_base/schemas/Concurrency.json +21 -0
- sequence_base/schemas/Conditions.json +59 -0
- sequence_base/schemas/ConnectRequest.json +16 -0
- sequence_base/schemas/ConnectResponse.json +81 -0
- sequence_base/schemas/DeploymentAccepted.json +22 -0
- sequence_base/schemas/DeploymentPolicy.json +101 -0
- sequence_base/schemas/DeploymentRequest.json +467 -0
- sequence_base/schemas/DeploymentStatusResponse.json +45 -0
- sequence_base/schemas/ErrorBody.json +32 -0
- sequence_base/schemas/Estimate.json +25 -0
- sequence_base/schemas/EvalDryRunResponse.json +41 -0
- sequence_base/schemas/EvalReport.json +179 -0
- sequence_base/schemas/EvalRequest.json +111 -0
- sequence_base/schemas/EvalResponse.json +88 -0
- sequence_base/schemas/ImageFrame.json +36 -0
- sequence_base/schemas/ImageSpec.json +123 -0
- sequence_base/schemas/ImageStepApt.json +18 -0
- sequence_base/schemas/ImageStepEnv.json +18 -0
- sequence_base/schemas/ImageStepRun.json +18 -0
- sequence_base/schemas/ImageStepUvPip.json +18 -0
- sequence_base/schemas/Lease.json +28 -0
- sequence_base/schemas/LossyTransform.json +21 -0
- sequence_base/schemas/MetricAgg.json +87 -0
- sequence_base/schemas/Observation.json +133 -0
- sequence_base/schemas/PolicyRegistration.json +438 -0
- sequence_base/schemas/PolicySpec.json +346 -0
- sequence_base/schemas/Proprioception.json +67 -0
- sequence_base/schemas/RenderSpec.json +22 -0
- sequence_base/schemas/Resources.json +29 -0
- sequence_base/schemas/RolloutResult.json +59 -0
- sequence_base/schemas/SecretCreateRequest.json +25 -0
- sequence_base/schemas/SecretInfo.json +44 -0
- sequence_base/schemas/SecretRef.json +16 -0
- sequence_base/schemas/SeqFlags.json +33 -0
- sequence_base/schemas/Step.json +68 -0
- sequence_base/schemas/SuiteAgg.json +26 -0
- sequence_base/schemas/TaskAgg.json +26 -0
- sequence_base/schemas/Timeouts.json +20 -0
- sequence_base/schemas/UsageReport.json +38 -0
- sequence_base/schemas/Variant.json +24 -0
- sequence_base/schemas/VolumeAttach.json +26 -0
- sequence_base/schemas/VolumeMount.json +21 -0
- sequence_base/schemas/VolumeRef.json +28 -0
- sequence_base/schemas/WarmingError.json +50 -0
- sequence_base/telemetry/__init__.py +13 -0
- sequence_base/telemetry/events.py +101 -0
- sequence_base-0.3.0.dist-info/METADATA +12 -0
- sequence_base-0.3.0.dist-info/RECORD +84 -0
- sequence_base-0.3.0.dist-info/WHEEL +4 -0
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
"""
|
|
2
|
+
契约 5 (run half) — `POST /v1/eval`, its response, and the `GET /v1/eval/{id}` report, plus the adapter
|
|
3
|
+
lossless/lossy classification wire types.
|
|
4
|
+
|
|
5
|
+
Two response shapes, deliberately distinguishable (goal 12): a real run returns eval_id / gpus_granted /
|
|
6
|
+
estimate; `dry_run=true` returns ONLY an estimate — no id to GET, no built-machine artifacts. The report
|
|
7
|
+
status is a restricted enum, matching the deployment-status discipline; every detail row is structured,
|
|
8
|
+
not a bare dict or free text.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from enum import Enum
|
|
13
|
+
|
|
14
|
+
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"EvalRequest",
|
|
18
|
+
"ClampInfo",
|
|
19
|
+
"Estimate",
|
|
20
|
+
"EvalResponse",
|
|
21
|
+
"EvalDryRunResponse",
|
|
22
|
+
"RolloutResult",
|
|
23
|
+
"EvalStatus",
|
|
24
|
+
"SuiteAgg",
|
|
25
|
+
"TaskAgg",
|
|
26
|
+
"MetricAgg",
|
|
27
|
+
"LossyTransform",
|
|
28
|
+
"EvalReport",
|
|
29
|
+
"AdapterKind",
|
|
30
|
+
"AdapterTransform",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class EvalRequest(BaseModel):
|
|
35
|
+
"""`model` and `benchmark` required; the rest are optional knobs. `dry_run` selects the estimate-only
|
|
36
|
+
response shape below."""
|
|
37
|
+
|
|
38
|
+
model_config = ConfigDict(extra="forbid")
|
|
39
|
+
|
|
40
|
+
model: str
|
|
41
|
+
benchmark: str
|
|
42
|
+
suite: str | None = None
|
|
43
|
+
trials: int | None = None
|
|
44
|
+
gpus: int | None = None
|
|
45
|
+
envs_per_gpu: int | None = None
|
|
46
|
+
vlm_max_concurrency: int | None = None
|
|
47
|
+
adapter: str | None = None
|
|
48
|
+
compare: str | None = Field(default=None, description="Another model id to compare against")
|
|
49
|
+
dry_run: bool = False
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class ClampInfo(BaseModel):
|
|
53
|
+
"""Why the granted GPU count was clamped below the request — structured, not a bare value."""
|
|
54
|
+
|
|
55
|
+
model_config = ConfigDict(extra="forbid")
|
|
56
|
+
|
|
57
|
+
limit: str = Field(..., description="Which limit bound it, e.g. 'account_quota'")
|
|
58
|
+
requested: int
|
|
59
|
+
granted: int
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class Estimate(BaseModel):
|
|
63
|
+
model_config = ConfigDict(extra="forbid")
|
|
64
|
+
|
|
65
|
+
wall_clock_s: float
|
|
66
|
+
cost_usd: float
|
|
67
|
+
vlm_calls: int
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class EvalResponse(BaseModel):
|
|
71
|
+
"""A real run — carries the built-machine artifacts (eval_id to GET, gpus_granted actually reserved)."""
|
|
72
|
+
|
|
73
|
+
model_config = ConfigDict(extra="forbid")
|
|
74
|
+
|
|
75
|
+
eval_id: str
|
|
76
|
+
gpus_granted: int
|
|
77
|
+
estimate: Estimate
|
|
78
|
+
clamped_by: ClampInfo | None = None
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class EvalDryRunResponse(BaseModel):
|
|
82
|
+
"""`dry_run=true` — estimate only. No eval_id, no gpus_granted: nothing was built, nothing to GET."""
|
|
83
|
+
|
|
84
|
+
model_config = ConfigDict(extra="forbid")
|
|
85
|
+
|
|
86
|
+
estimate: Estimate
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class RolloutResult(BaseModel):
|
|
90
|
+
"""One episode's result, ASSEMBLED BY THE PLATFORM from the per-step `@seq.check` returns — the user
|
|
91
|
+
never returns a success flag. `metrics` is the episode's closed-out user metrics (passthrough; the
|
|
92
|
+
platform does not know the field names). `video_uri` points at the mp4 the platform tool produced.
|
|
93
|
+
|
|
94
|
+
Scope: W0.1 delivers this wire shape only; the runtime that settles per-step `check.done` per-env and
|
|
95
|
+
assembles it lives in the container / gateway, not in this pure-types unit."""
|
|
96
|
+
|
|
97
|
+
model_config = ConfigDict(extra="forbid")
|
|
98
|
+
|
|
99
|
+
metrics: dict[str, float] = Field(
|
|
100
|
+
..., description="Episode's user-named metrics (passthrough; platform does not name them)"
|
|
101
|
+
)
|
|
102
|
+
steps: int
|
|
103
|
+
terminated: bool
|
|
104
|
+
truncated: bool
|
|
105
|
+
trajectory_uri: str | None = None
|
|
106
|
+
video_uri: str | None = None
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
class EvalStatus(str, Enum):
|
|
110
|
+
"""Restricted enum — same discipline as DeploymentState (not a free string)."""
|
|
111
|
+
|
|
112
|
+
PENDING = "pending"
|
|
113
|
+
RUNNING = "running"
|
|
114
|
+
COMPLETED = "completed"
|
|
115
|
+
FAILED = "failed"
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
class SuiteAgg(BaseModel):
|
|
119
|
+
"""Per-suite aggregate of ONE user-named metric — structured detail, not a bare dict. `value` is the
|
|
120
|
+
metric aggregated over that suite; the platform does not name or interpret the metric."""
|
|
121
|
+
|
|
122
|
+
model_config = ConfigDict(extra="forbid")
|
|
123
|
+
|
|
124
|
+
suite: str
|
|
125
|
+
value: float
|
|
126
|
+
n: int
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
class TaskAgg(BaseModel):
|
|
130
|
+
"""Per-task aggregate of ONE user-named metric — structured detail, not a bare dict."""
|
|
131
|
+
|
|
132
|
+
model_config = ConfigDict(extra="forbid")
|
|
133
|
+
|
|
134
|
+
task: str
|
|
135
|
+
value: float
|
|
136
|
+
n: int
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class MetricAgg(BaseModel):
|
|
140
|
+
"""The aggregate of one user-named metric key across the eval — overall plus per-suite / per-task
|
|
141
|
+
breakdowns and the sample count. The platform aggregates whatever the user returned in `Step.metrics`;
|
|
142
|
+
it neither defines nor injects `success_rate` (success is just one key the user may or may not name)."""
|
|
143
|
+
|
|
144
|
+
model_config = ConfigDict(extra="forbid")
|
|
145
|
+
|
|
146
|
+
overall: float
|
|
147
|
+
per_suite: list[SuiteAgg] = Field(default_factory=list)
|
|
148
|
+
per_task: list[TaskAgg] = Field(default_factory=list)
|
|
149
|
+
n: int
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
class LossyTransform(BaseModel):
|
|
153
|
+
"""A lossy adapter transform recorded in the report. Shape aligns with the lossy element of
|
|
154
|
+
AdapterTransform (both {transform, reason})."""
|
|
155
|
+
|
|
156
|
+
model_config = ConfigDict(extra="forbid")
|
|
157
|
+
|
|
158
|
+
transform: str
|
|
159
|
+
reason: str
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
class EvalReport(BaseModel):
|
|
163
|
+
"""GET /v1/eval/{id} (eval-raw revision). NO hard-coded `overall_sr` / `success_rate` top-level field —
|
|
164
|
+
`metrics` is a passthrough dict keyed by the user's metric names (from `@seq.check`'s `Step.metrics`),
|
|
165
|
+
each value a `MetricAgg`. The platform aggregates per key and invents nothing. `videos` carries the mp4
|
|
166
|
+
pointers the platform tool produced; `incomplete` flags that some condition did not finish (never pass a
|
|
167
|
+
subset off as the whole)."""
|
|
168
|
+
|
|
169
|
+
model_config = ConfigDict(extra="forbid")
|
|
170
|
+
|
|
171
|
+
status: EvalStatus
|
|
172
|
+
metrics: dict[str, MetricAgg] = Field(
|
|
173
|
+
..., description="User-named metric key → its aggregate; platform does not name the keys"
|
|
174
|
+
)
|
|
175
|
+
cost_usd: float
|
|
176
|
+
dashboard_url: str
|
|
177
|
+
adapter_lossy: list[LossyTransform]
|
|
178
|
+
videos: list[str] = Field(default_factory=list, description="mp4 pointers from the platform tool")
|
|
179
|
+
incomplete: bool | None = None
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
class AdapterKind(str, Enum):
|
|
183
|
+
LOSSLESS = "lossless"
|
|
184
|
+
LOSSY = "lossy"
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
class AdapterTransform(BaseModel):
|
|
188
|
+
"""One adapter transform, classified. A lossy transform must carry a non-empty reason; a lossless one
|
|
189
|
+
needs none. The lossy element's shape ({transform, reason}) aligns with EvalReport.adapter_lossy."""
|
|
190
|
+
|
|
191
|
+
model_config = ConfigDict(extra="forbid")
|
|
192
|
+
|
|
193
|
+
kind: AdapterKind
|
|
194
|
+
transform: str
|
|
195
|
+
reason: str | None = None
|
|
196
|
+
|
|
197
|
+
@model_validator(mode="after")
|
|
198
|
+
def _lossy_needs_reason(self) -> "AdapterTransform":
|
|
199
|
+
if self.kind is AdapterKind.LOSSY and not (self.reason and self.reason.strip()):
|
|
200
|
+
raise ValueError("a lossy transform must carry a non-empty reason")
|
|
201
|
+
return self
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""
|
|
2
|
+
契约 1 — `SeqFlags`: the lifecycle bitmask the SDK's `@seq.*` decorators write and the harness loader
|
|
3
|
+
reads back. One IntFlag, so a registration can carry several lifecycle entrypoints in one integer and
|
|
4
|
+
the harness recovers each by testing a bit.
|
|
5
|
+
|
|
6
|
+
Modeled on Modal's `_PartialFunctionFlags` (one IntFlag, one bit per lifecycle hook)
|
|
7
|
+
{modal _partial_function.py:29}. Four members map bit-for-bit onto Modal's flags so the harness can be
|
|
8
|
+
a thin adaptation of Modal's loader; the rest are platform-only lifecycle hooks Modal has no analogue
|
|
9
|
+
for (a warm-up hook, a plan step, and the benchmark SETUP / RESET / CHECK / SCORE quartet).
|
|
10
|
+
|
|
11
|
+
eval-raw revision: the platform provides the reset→step→check loop as a tool; the user defines the goal
|
|
12
|
+
through four hooks — SETUP (load sim + assets), RESET (build one episode's env), CHECK (per-step: read raw
|
|
13
|
+
env state → done + user-named metrics), SCORE (cross-episode aggregate). RESET=32768 and CHECK=65536 are
|
|
14
|
+
the new hook bits. The old single-hook `ROLLOUT` is DEPRECATED: bit 8192 stays reserved (never reused for
|
|
15
|
+
another member) but appears in NO bindings — the setup/reset/check/score model replaces it.
|
|
16
|
+
|
|
17
|
+
The exact bit values are the contract — see `MODAL_MAPPING` / `PLATFORM_ONLY` for the authoritative
|
|
18
|
+
"member → bit value → Modal counterpart" list. A member taking the wrong value, WARMUP collapsing back
|
|
19
|
+
into LOAD, RESET/CHECK overlapping an existing bit, or ROLLOUT being re-bound is caught bit-by-bit (see
|
|
20
|
+
tests), not waved away with "an IntFlag is naturally bit-composable".
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from enum import IntFlag
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SeqFlags(IntFlag):
|
|
28
|
+
# ── mapped bit-for-bit onto Modal `_PartialFunctionFlags` ────────────────────────────────────
|
|
29
|
+
LOAD = 2 # Modal ENTER — load weights into VRAM (one-shot; no snapshot bisection)
|
|
30
|
+
SHUTDOWN = 8 # Modal EXIT — teardown / release the container
|
|
31
|
+
INFER = 16 # Modal CALLABLE_INTERFACE — the obs→ActionChunk call
|
|
32
|
+
ON_EPISODE = 1024 # Modal ~SESSIONED — sticky / session boundary for stateful policies
|
|
33
|
+
|
|
34
|
+
# ── platform-only: no Modal counterpart; each an independent, disjoint bit ────────────────────
|
|
35
|
+
# WARMUP is its OWN bit, NOT an alias of LOAD (S1.yaml goal 5 — "WARMUP=1(独立 bit,明确不并入
|
|
36
|
+
# LOAD)"). A policy may bind @seq.load and @seq.warmup to two different methods → two independent
|
|
37
|
+
# bindings, so the bit must be distinct from LOAD. Value 1 is bit 0, unused by any mapped flag.
|
|
38
|
+
WARMUP = 1 # JIT warm-up hook (ours; Modal folded warm-up into ENTER, we keep it separate)
|
|
39
|
+
PLAN = 4 # a policy planning step (ours; Modal has none) — bit 2, unused by Modal
|
|
40
|
+
# benchmark lifecycle (eval-raw: platform provides the loop tool, user hooks the goal in) — each an
|
|
41
|
+
# independent, disjoint bit, disjoint too from every mapped bit:
|
|
42
|
+
SETUP = 4096 # load sim + assets, once (≈ @seq.load)
|
|
43
|
+
SCORE = 16384 # cross-episode aggregate (optional; default = mean of each numeric metric)
|
|
44
|
+
RESET = 32768 # build one episode's env (gym-style handle) — eval-raw new hook, bit 15
|
|
45
|
+
CHECK = 65536 # per step: read raw env state → done + user-named metrics — eval-raw new hook, bit 16
|
|
46
|
+
# DEPRECATED — the old single-hook benchmark model. Bit 8192 stays reserved (never reused for another
|
|
47
|
+
# member) but is NOT bound in any registration; setup/reset/check/score replaces it. Kept only so the
|
|
48
|
+
# value cannot be recycled and an old serialized 8192 still decodes to a known name.
|
|
49
|
+
ROLLOUT = 8192
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# Authoritative "member → Modal counterpart" list. Anything here must equal the bit above; anything in
|
|
53
|
+
# PLATFORM_ONLY must be disjoint from every mapped bit and from each other.
|
|
54
|
+
MODAL_MAPPING: dict[str, str] = {
|
|
55
|
+
"LOAD": "ENTER",
|
|
56
|
+
"SHUTDOWN": "EXIT",
|
|
57
|
+
"INFER": "CALLABLE_INTERFACE",
|
|
58
|
+
"ON_EPISODE": "~SESSIONED",
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
# Platform-only members Modal has no flag for. ROLLOUT is included: it is still a real (if deprecated)
|
|
62
|
+
# member and its bit must stay disjoint from every other, so the disjointness check must cover it.
|
|
63
|
+
PLATFORM_ONLY: frozenset[str] = frozenset(
|
|
64
|
+
{"WARMUP", "PLAN", "SETUP", "SCORE", "RESET", "CHECK", "ROLLOUT"}
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# The benchmark lifecycle bits a live registration may bind (eval-raw): SETUP / RESET / CHECK / SCORE.
|
|
68
|
+
# ROLLOUT is deliberately absent — deprecated, never bound.
|
|
69
|
+
BENCHMARK_BINDABLE: frozenset[SeqFlags] = frozenset(
|
|
70
|
+
{SeqFlags.SETUP, SeqFlags.RESET, SeqFlags.CHECK, SeqFlags.SCORE}
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
# Deprecated benchmark bits: reserved value, must not appear in any bindings.
|
|
74
|
+
DEPRECATED_FLAGS: frozenset[SeqFlags] = frozenset({SeqFlags.ROLLOUT})
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def seqflags_json_schema(contract_version: str) -> dict:
|
|
78
|
+
"""`SeqFlags` is an IntFlag, so `model_json_schema()` does not apply. Export it as an integer whose
|
|
79
|
+
named members are enumerated, with `x-flag-names` giving name→value for every member (WARMUP, PLAN and
|
|
80
|
+
the benchmark trio included). Every member is its own bit, so each is emitted once under its own name."""
|
|
81
|
+
# Every member has a distinct value (no aliases), so this maps value → name one-to-one.
|
|
82
|
+
canonical = {m.value: m.name for m in SeqFlags} # value → member name (one entry per distinct bit)
|
|
83
|
+
names = {name: value for value, name in sorted(canonical.items())}
|
|
84
|
+
return {
|
|
85
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
86
|
+
"title": "SeqFlags",
|
|
87
|
+
"description": "Lifecycle bitmask (IntFlag). One bit per lifecycle entrypoint.",
|
|
88
|
+
"type": "integer",
|
|
89
|
+
"enum": sorted(canonical),
|
|
90
|
+
"x-flag-names": names,
|
|
91
|
+
"x-contract-version": contract_version,
|
|
92
|
+
}
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
"""
|
|
2
|
+
契约 1 — the `@seq.policy` infra spec (SDK produces it, harness loader introspects it) and契约 2's
|
|
3
|
+
declarative building blocks (ImageSpec / VolumeMount / Concurrency), defined once and reused by the
|
|
4
|
+
deployments request so the two sides cannot drift.
|
|
5
|
+
|
|
6
|
+
Every shape here forbids unknown fields and pins exact field types — no `Optional[Any] = None`, no bare
|
|
7
|
+
`dict` standing in for a structured sub-shape (goal 2 / 4 / 5 / 18).
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
|
|
12
|
+
|
|
13
|
+
from sequence_base.contract.accounts import SecretRef
|
|
14
|
+
from sequence_base.contract.act import ActionSpace
|
|
15
|
+
from sequence_base.contract.flags import SeqFlags
|
|
16
|
+
|
|
17
|
+
__all__ = [
|
|
18
|
+
"ImageStepUvPip",
|
|
19
|
+
"ImageStepApt",
|
|
20
|
+
"ImageStepEnv",
|
|
21
|
+
"ImageStepRun",
|
|
22
|
+
"ImageStep",
|
|
23
|
+
"ImageSpec",
|
|
24
|
+
"VolumeMount",
|
|
25
|
+
"VolumeRef",
|
|
26
|
+
"Concurrency",
|
|
27
|
+
"ActionContract",
|
|
28
|
+
"PolicySpec",
|
|
29
|
+
"PolicyRegistration",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# ── ImageSpec — a discriminated union of exactly four step variants ───────────────────────────────
|
|
34
|
+
# Each variant carries exactly one key; extra="forbid" makes a dict that names two keys (or an unknown
|
|
35
|
+
# key) match NONE of them, so every step matches exactly one variant (goal 4). `run_commands` is its own
|
|
36
|
+
# variant (SDK contract S2 goal 5) so an arbitrary build command is an ordered step in the chain, never
|
|
37
|
+
# coalesced into pip/apt and never silently dropped for lack of a home.
|
|
38
|
+
class ImageStepUvPip(BaseModel):
|
|
39
|
+
model_config = ConfigDict(extra="forbid")
|
|
40
|
+
uv_pip_install: list[str]
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class ImageStepApt(BaseModel):
|
|
44
|
+
model_config = ConfigDict(extra="forbid")
|
|
45
|
+
apt_install: list[str]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
class ImageStepEnv(BaseModel):
|
|
49
|
+
model_config = ConfigDict(extra="forbid")
|
|
50
|
+
env: dict[str, str]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class ImageStepRun(BaseModel):
|
|
54
|
+
model_config = ConfigDict(extra="forbid")
|
|
55
|
+
run_commands: list[str]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
ImageStep = ImageStepUvPip | ImageStepApt | ImageStepEnv | ImageStepRun
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class ImageSpec(BaseModel):
|
|
62
|
+
"""Declarative image (gateway builds it via Cloud Build) — = Modal `Image` message. Exactly one of
|
|
63
|
+
`base` (a pullable registry ref) or `dockerfile` (inline Dockerfile source) is set — the Dockerfile
|
|
64
|
+
is its OWN component, never smuggled into the `base` string (S2 goal 4), so the gateway can tell a
|
|
65
|
+
registry build from a Dockerfile build. `steps` is an ordered list of the four-variant union above
|
|
66
|
+
(a bare list[dict] is refused), extending whichever base was chosen."""
|
|
67
|
+
|
|
68
|
+
model_config = ConfigDict(extra="forbid")
|
|
69
|
+
|
|
70
|
+
base: str | None = None
|
|
71
|
+
dockerfile: str | None = None
|
|
72
|
+
steps: list[ImageStep] = Field(default_factory=list)
|
|
73
|
+
|
|
74
|
+
@model_validator(mode="after")
|
|
75
|
+
def _base_xor_dockerfile(self) -> "ImageSpec":
|
|
76
|
+
# Exactly one origin. Neither → nothing to build; both → an ambiguous spec whose build order the
|
|
77
|
+
# gateway would have to guess. Either is a malformed image, caught at construction, not at build.
|
|
78
|
+
if (self.base is None) == (self.dockerfile is None):
|
|
79
|
+
raise ValueError(
|
|
80
|
+
"ImageSpec needs exactly one of base= (a registry ref) or dockerfile= "
|
|
81
|
+
"(inline Dockerfile source); got "
|
|
82
|
+
+ ("both" if self.base is not None else "neither")
|
|
83
|
+
)
|
|
84
|
+
return self
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# ── shared sub-shapes (referenced by PolicySpec and the deployments body) ─────────────────────────
|
|
88
|
+
class VolumeMount(BaseModel):
|
|
89
|
+
"""A concrete mount: where the data lives and where it lands in the container. Both required."""
|
|
90
|
+
|
|
91
|
+
model_config = ConfigDict(extra="forbid")
|
|
92
|
+
|
|
93
|
+
uri: str
|
|
94
|
+
mount: str
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
class VolumeRef(BaseModel):
|
|
98
|
+
"""PolicySpec.volumes value — at least a `uri`; a `mount` may pin the container path."""
|
|
99
|
+
|
|
100
|
+
model_config = ConfigDict(extra="forbid")
|
|
101
|
+
|
|
102
|
+
uri: str
|
|
103
|
+
mount: str | None = None
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class Concurrency(BaseModel):
|
|
107
|
+
"""= Modal max/target_concurrent_inputs. Both required."""
|
|
108
|
+
|
|
109
|
+
model_config = ConfigDict(extra="forbid")
|
|
110
|
+
|
|
111
|
+
max_inputs: int
|
|
112
|
+
target_inputs: int
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
class ActionContract(BaseModel):
|
|
116
|
+
"""The fixed obs→ActionChunk contract a policy promises. Every field required; `action_space` reuses
|
|
117
|
+
contract 4's enum (not a free string). Each ActionChunk echoes exactly these invariants back."""
|
|
118
|
+
|
|
119
|
+
model_config = ConfigDict(extra="forbid")
|
|
120
|
+
|
|
121
|
+
robot: str
|
|
122
|
+
action_space: ActionSpace
|
|
123
|
+
action_dim: int
|
|
124
|
+
control_frequency_hz: float
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
class PolicySpec(BaseModel):
|
|
128
|
+
"""What `@seq.policy(...)` accrues — a policy's complete infra declaration, losslessly. `action` is
|
|
129
|
+
required (no default): a policy without an obs→ActionChunk contract is not deployable."""
|
|
130
|
+
|
|
131
|
+
model_config = ConfigDict(extra="forbid")
|
|
132
|
+
|
|
133
|
+
name: str
|
|
134
|
+
gpu: str | list[str] = Field(..., description="A GPU type, or a list = cross-provider fallback")
|
|
135
|
+
image: ImageSpec
|
|
136
|
+
weights: str = Field(..., description="Weights URI; scheme restricted to gs:// or hf://")
|
|
137
|
+
volumes: dict[str, VolumeRef] = Field(default_factory=dict)
|
|
138
|
+
secrets: list[SecretRef] = Field(default_factory=list)
|
|
139
|
+
min_containers: int = 0
|
|
140
|
+
max_containers: int | None = None
|
|
141
|
+
buffer_containers: int | None = None
|
|
142
|
+
scaledown_window: int = 600
|
|
143
|
+
concurrency: Concurrency | None = None
|
|
144
|
+
streaming: bool = False
|
|
145
|
+
frames_per_chunk: int | None = None
|
|
146
|
+
action: ActionContract
|
|
147
|
+
|
|
148
|
+
@field_validator("weights")
|
|
149
|
+
@classmethod
|
|
150
|
+
def _weights_scheme(cls, v: str) -> str:
|
|
151
|
+
if not (v.startswith("gs://") or v.startswith("hf://")):
|
|
152
|
+
raise ValueError("weights must be a gs:// or hf:// URI")
|
|
153
|
+
return v
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
class PolicyRegistration(BaseModel):
|
|
157
|
+
"""What the harness loader introspects: the spec, the SeqFlags→method-name bindings, and the
|
|
158
|
+
`handler_ref` it importlib-loads the policy class from. From one of these the loader recovers which
|
|
159
|
+
method is which lifecycle entrypoint (load / infer / on_episode / plan …)."""
|
|
160
|
+
|
|
161
|
+
model_config = ConfigDict(extra="forbid")
|
|
162
|
+
|
|
163
|
+
spec: PolicySpec
|
|
164
|
+
bindings: dict[SeqFlags, str] = Field(..., description="lifecycle flag → method name on the handler")
|
|
165
|
+
handler_ref: str = Field(..., description='"module.path:ClassName" for importlib load')
|
|
166
|
+
# `bindings` is flag→method-name and cannot carry a per-binding parameter, so the one lifecycle hook
|
|
167
|
+
# that takes a schedule — @seq.plan(every_s=…), the low-frequency composite planner (契约 1) — needs a
|
|
168
|
+
# separate landing spot here. None when there is no PLAN binding; a positive seconds cadence otherwise.
|
|
169
|
+
# The harness reads bindings[PLAN] for the method and this for how often to call it.
|
|
170
|
+
plan_every_s: float | None = Field(
|
|
171
|
+
default=None, gt=0, description="Seconds between @seq.plan calls; set iff a PLAN binding exists"
|
|
172
|
+
)
|
|
File without changes
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Shared transport status taxonomy — the infrastructure outcomes all three sides must speak the same
|
|
3
|
+
way, so a caller can tell "retry in 2 min" from "stop and page someone" from "your payload is wrong".
|
|
4
|
+
{from general-sequences api/handoff/runpod.py — the four classes it defines}
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class TransportUnavailable(RuntimeError):
|
|
10
|
+
"""Endpoint unreachable / not configured / worker failed → 503. Retrying is meaningful (says
|
|
11
|
+
"this upstream is down now", not "we are broken")."""
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
class TransportWarming(RuntimeError):
|
|
15
|
+
"""Model loading, not ready yet → 503 + Retry-After. Distinct from Unavailable: it WILL succeed
|
|
16
|
+
shortly and the only useful action is to wait `eta_s`. `job_id` (if set) is the warming job, left
|
|
17
|
+
running so the retry lands on a hot worker."""
|
|
18
|
+
|
|
19
|
+
def __init__(self, message: str, eta_s: float, job_id: str | None = None) -> None:
|
|
20
|
+
super().__init__(message)
|
|
21
|
+
self.eta_s = eta_s
|
|
22
|
+
self.job_id = job_id
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class TransportRejected(RuntimeError):
|
|
26
|
+
"""The worker ran and refused the payload → 400, not 503. Never mislabel as a crash — a caller
|
|
27
|
+
must not retry a request that can never succeed."""
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class TransportOverloaded(RuntimeError):
|
|
31
|
+
"""The admission queue is full → 429 + Retry-After. Refuse here rather than pile onto the worker
|
|
32
|
+
(an unbounded pile starves /ping and turns one burst into a 503-storm)."""
|
|
33
|
+
|
|
34
|
+
def __init__(self, message: str, retry_after_s: float = 1.0) -> None:
|
|
35
|
+
super().__init__(message)
|
|
36
|
+
self.retry_after_s = retry_after_s
|
|
File without changes
|
sequence_base/obs/log.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Structured logging — one line per event: `seq ev=<what> rid=<which> k=v ...`, greppable, flushed.
|
|
3
|
+
Deliberately `print` (not `logging`): in Cloud Run and in a bare worker, stdout IS the log.
|
|
4
|
+
{from general-sequences obs.py}
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
import time
|
|
11
|
+
|
|
12
|
+
from sequence_base.obs.rid import clean
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def enabled() -> bool:
|
|
16
|
+
"""On by default — a fleet that can't be debugged is worse than a noisy one. SEQ_LOG=0 quiets it."""
|
|
17
|
+
return os.environ.get("SEQ_LOG", "1") != "0"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def log(ev: str, rid: str | None = None, **fields: object) -> None:
|
|
21
|
+
"""Print one event. `seq` first so a grep finds our lines among a vendor's; `ev` a dotted
|
|
22
|
+
noun.verb that narrows by prefix; `rid` threads the call. flush=True so a line isn't lost when a
|
|
23
|
+
worker is killed before its 30 s log-tail ships."""
|
|
24
|
+
parts = [f"seq ev={clean(ev)}"]
|
|
25
|
+
if rid:
|
|
26
|
+
parts.append(f"rid={clean(rid)}")
|
|
27
|
+
parts.extend(f"{k}={clean(v)}" for k, v in fields.items() if v is not None)
|
|
28
|
+
print(" ".join(parts), file=sys.stdout, flush=True)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class Timer:
|
|
32
|
+
"""Measure a phase and log it as one line when it ends — including when it raises, so every
|
|
33
|
+
start has an end and a gap in the log is real rather than a missing print.
|
|
34
|
+
|
|
35
|
+
with Timer("act.upstream", rid, model=m) as t:
|
|
36
|
+
...
|
|
37
|
+
t.add(status=200)
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
def __init__(self, ev: str, rid: str | None = None, **fields: object) -> None:
|
|
41
|
+
self.ev, self.rid, self.fields = ev, rid, dict(fields)
|
|
42
|
+
self.t0 = 0.0
|
|
43
|
+
|
|
44
|
+
def add(self, **fields: object) -> None:
|
|
45
|
+
self.fields.update(fields)
|
|
46
|
+
|
|
47
|
+
def __enter__(self) -> "Timer":
|
|
48
|
+
self.t0 = time.perf_counter()
|
|
49
|
+
return self
|
|
50
|
+
|
|
51
|
+
def __exit__(self, exc_type, exc, tb) -> bool:
|
|
52
|
+
ms = (time.perf_counter() - self.t0) * 1000
|
|
53
|
+
if exc_type is not None:
|
|
54
|
+
self.fields["ok"] = 0
|
|
55
|
+
self.fields["err"] = exc_type.__name__
|
|
56
|
+
log(self.ev, self.rid, ms=round(ms, 1), **self.fields)
|
|
57
|
+
return False
|
sequence_base/obs/rid.py
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Correlation id — one value threads a call through sdk → gateway → worker, so "what happened to that
|
|
3
|
+
call" is one grep instead of three log stores and a stopwatch. {from general-sequences obs.py}
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
import uuid
|
|
9
|
+
|
|
10
|
+
# The id travels in a HEADER, not a body field, so it survives paths that don't parse the body (the
|
|
11
|
+
# worker reads it without touching the observation; a JSON-rewriting proxy can't lose it).
|
|
12
|
+
HEADER = "X-Seq-Request-Id"
|
|
13
|
+
|
|
14
|
+
_UNSAFE = re.compile(r"\s+")
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def new_rid() -> str:
|
|
18
|
+
"""Short (read off a terminal, quoted in bug reports) + random (gateway instances don't
|
|
19
|
+
coordinate, so it must not collide)."""
|
|
20
|
+
return "r-" + uuid.uuid4().hex[:8]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def clean(v: object) -> str:
|
|
24
|
+
"""Collapse whitespace and cap length. A value with a space would split into two log fields and
|
|
25
|
+
let anything user-supplied forge keys into the line; 200 chars keeps a real error readable."""
|
|
26
|
+
s = _UNSAFE.sub("_", str(v))
|
|
27
|
+
return s if len(s) <= 200 else s[:197] + "..."
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
# Back-compat alias: general-sequences code calls obs._clean; keep it working after the switch.
|
|
31
|
+
_clean = clean
|