lablink-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,212 @@
1
+ """Configuration schema for LabLink deployments.
2
+
3
+ Re-exports the canonical Config dataclass from the allocator package
4
+ and adds CLI-specific helpers (YAML serialization, validation, reference data).
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import re
10
+ from dataclasses import fields
11
+ from pathlib import Path
12
+ from typing import Any
13
+
14
+ import yaml
15
+
16
+ # Re-export the canonical config dataclasses from the allocator.
17
+ from lablink_allocator_service.validate_config import (
18
+ get_config_errors,
19
+ )
20
+ from lablink_allocator_service.conf.structured_config import ( # noqa: F401
21
+ AllocatorConfig,
22
+ AppConfig,
23
+ Config,
24
+ DatabaseConfig,
25
+ DNSConfig,
26
+ EIPConfig,
27
+ MachineConfig,
28
+ SSLConfig,
29
+ StartupConfig,
30
+ )
31
+
32
+
33
+ def config_to_dict(cfg: Any) -> Any:
34
+ """Recursively convert a dataclass to a nested dict."""
35
+ if hasattr(cfg, "__dataclass_fields__"):
36
+ return {
37
+ f.name: config_to_dict(getattr(cfg, f.name))
38
+ for f in fields(cfg)
39
+ }
40
+ return cfg
41
+
42
+
43
+ def load_config(path: Path) -> Config:
44
+ """Load a Config from a YAML file."""
45
+ with open(path) as f:
46
+ data = yaml.safe_load(f)
47
+
48
+ cfg = Config()
49
+ for key, value in data.items():
50
+ if hasattr(cfg, key) and isinstance(value, dict):
51
+ sub = getattr(cfg, key)
52
+ for k, v in value.items():
53
+ if isinstance(v, dict):
54
+ # Nested sub-config (3-level YAML structure)
55
+ nested = getattr(sub, k, None)
56
+ if nested and hasattr(nested, "__dataclass_fields__"):
57
+ for nk, nv in v.items():
58
+ setattr(nested, nk, nv)
59
+ else:
60
+ setattr(sub, k, v)
61
+ else:
62
+ setattr(sub, k, v)
63
+ elif hasattr(cfg, key):
64
+ setattr(cfg, key, value)
65
+ return cfg
66
+
67
+
68
+ def save_config(cfg: Config, path: Path) -> None:
69
+ """Write a Config to a YAML file."""
70
+ path.parent.mkdir(parents=True, exist_ok=True)
71
+ with open(path, "w") as f:
72
+ yaml.dump(
73
+ config_to_dict(cfg),
74
+ f,
75
+ default_flow_style=False,
76
+ sort_keys=False,
77
+ )
78
+
79
+
80
+ DEPLOYMENT_NAME_RE = re.compile(r"^[a-z][a-z0-9-]*[a-z0-9]$")
81
+ VALID_ENVIRONMENTS = ("dev", "test", "ci-test", "prod")
82
+
83
+
84
+ def validate_config(cfg: Config) -> list[str]:
85
+ """Return a list of validation errors (empty = valid).
86
+
87
+ Mirrors the allocator's validate_config_logic but returns a
88
+ simple list instead of a tuple, suitable for TUI display.
89
+ """
90
+ errors: list[str] = []
91
+ # deployment_name validation
92
+ if not cfg.deployment_name:
93
+ errors.append(
94
+ "deployment_name is required "
95
+ "(e.g., 'sleap-lablink')"
96
+ )
97
+ elif (
98
+ len(cfg.deployment_name) < 3
99
+ or len(cfg.deployment_name) > 32
100
+ or not DEPLOYMENT_NAME_RE.match(cfg.deployment_name)
101
+ ):
102
+ errors.append(
103
+ "deployment_name must be 3-32 characters, "
104
+ "lowercase kebab-case (e.g., 'sleap-lablink')"
105
+ )
106
+ # environment validation
107
+ if cfg.environment not in VALID_ENVIRONMENTS:
108
+ errors.append(
109
+ f"environment must be one of: "
110
+ f"{', '.join(VALID_ENVIRONMENTS)}"
111
+ )
112
+ # DNS/SSL validation — shared with allocator's validate_config
113
+ errors.extend(get_config_errors(cfg))
114
+ return errors
115
+
116
+
117
+ # AMI IDs by region (Ubuntu 24.04 with Docker + Nvidia GPU Driver)
118
+ AMI_MAP: dict[str, str] = {
119
+ "us-east-1": "ami-0601752c11b394251",
120
+ "us-east-2": "ami-0601752c11b394251",
121
+ "us-west-1": "ami-0601752c11b394251",
122
+ "us-west-2": "ami-0601752c11b394251",
123
+ }
124
+
125
+ # Common GPU instance types
126
+ GPU_INSTANCE_TYPES: list[dict[str, str]] = [
127
+ {
128
+ "type": "g4dn.xlarge",
129
+ "gpu": "T4 16GB",
130
+ "vcpu": "4",
131
+ "ram": "16 GB",
132
+ "cost": "~$0.53/hr",
133
+ },
134
+ {
135
+ "type": "g4dn.2xlarge",
136
+ "gpu": "T4 16GB",
137
+ "vcpu": "8",
138
+ "ram": "32 GB",
139
+ "cost": "~$0.75/hr",
140
+ },
141
+ {
142
+ "type": "g5.xlarge",
143
+ "gpu": "A10G 24GB",
144
+ "vcpu": "4",
145
+ "ram": "16 GB",
146
+ "cost": "~$1.01/hr",
147
+ },
148
+ {
149
+ "type": "g5.2xlarge",
150
+ "gpu": "A10G 24GB",
151
+ "vcpu": "8",
152
+ "ram": "32 GB",
153
+ "cost": "~$1.21/hr",
154
+ },
155
+ {
156
+ "type": "p3.2xlarge",
157
+ "gpu": "V100 16GB",
158
+ "vcpu": "8",
159
+ "ram": "61 GB",
160
+ "cost": "~$3.06/hr",
161
+ },
162
+ ]
163
+
164
+ # Common CPU instance types (no GPU)
165
+ CPU_INSTANCE_TYPES: list[dict[str, str]] = [
166
+ {
167
+ "type": "t3.large",
168
+ "gpu": "—",
169
+ "vcpu": "2",
170
+ "ram": "8 GB",
171
+ "cost": "~$0.08/hr",
172
+ },
173
+ {
174
+ "type": "t3.xlarge",
175
+ "gpu": "—",
176
+ "vcpu": "4",
177
+ "ram": "16 GB",
178
+ "cost": "~$0.17/hr",
179
+ },
180
+ {
181
+ "type": "t3.2xlarge",
182
+ "gpu": "—",
183
+ "vcpu": "8",
184
+ "ram": "32 GB",
185
+ "cost": "~$0.33/hr",
186
+ },
187
+ {
188
+ "type": "m5.xlarge",
189
+ "gpu": "—",
190
+ "vcpu": "4",
191
+ "ram": "16 GB",
192
+ "cost": "~$0.19/hr",
193
+ },
194
+ {
195
+ "type": "m5.2xlarge",
196
+ "gpu": "—",
197
+ "vcpu": "8",
198
+ "ram": "32 GB",
199
+ "cost": "~$0.38/hr",
200
+ },
201
+ ]
202
+
203
+ AWS_REGIONS: list[dict[str, str]] = [
204
+ {"id": "us-east-1", "name": "US East (N. Virginia)"},
205
+ {"id": "us-east-2", "name": "US East (Ohio)"},
206
+ {"id": "us-west-1", "name": "US West (N. California)"},
207
+ {"id": "us-west-2", "name": "US West (Oregon)"},
208
+ {"id": "eu-west-1", "name": "Europe (Ireland)"},
209
+ {"id": "eu-central-1", "name": "Europe (Frankfurt)"},
210
+ {"id": "ap-northeast-1", "name": "Asia Pacific (Tokyo)"},
211
+ {"id": "ap-southeast-1", "name": "Asia Pacific (Singapore)"},
212
+ ]
@@ -0,0 +1,94 @@
1
+ """CLI-local cache for allocator deployment metrics (issue #317).
2
+
3
+ Stored on the operator's machine under ``~/.lablink/deployments/`` so that
4
+ metrics for failed deploys (or for already-destroyed allocators) survive.
5
+
6
+ Records start life with ``status="in_progress"`` and are promoted to
7
+ ``success`` / ``failed`` by :func:`~lablink_cli.commands.deploy.run_deploy`.
8
+ Plan-confirmation cancels and Ctrl-C leave the file in ``in_progress``
9
+ indefinitely — use ``lablink cache-clear --deployments --stale`` to prune
10
+ just those, or ``--deployments`` to wipe the whole cache.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import time
17
+ from contextlib import contextmanager
18
+ from dataclasses import asdict, dataclass
19
+ from datetime import datetime
20
+ from pathlib import Path
21
+ from typing import Iterator, Optional
22
+
23
+ DEPLOYMENTS_DIR = Path.home() / ".lablink" / "deployments"
24
+
25
+
26
+ @dataclass
27
+ class DeploymentMetrics:
28
+ deployment_name: str
29
+ # Which deploy path produced this record. deployment_name alone does not
30
+ # identify it: an operator can reuse one name across providers (observed:
31
+ # a `sleap-lablink` config switched from aws to manual), and the AWS
32
+ # OpenTofu timings below are meaningless for a compose deploy. Records
33
+ # written before this field existed are all AWS ones, which is why
34
+ # readers default a missing value to "aws" rather than to None.
35
+ provider: Optional[str] = None
36
+ region: Optional[str] = None
37
+ template_version: Optional[str] = None
38
+ ssl_enabled: Optional[bool] = None
39
+ allocator_deploy_start_time: Optional[str] = None
40
+ allocator_deploy_end_time: Optional[str] = None
41
+ allocator_tofu_init_duration_seconds: Optional[float] = None
42
+ allocator_tofu_plan_duration_seconds: Optional[float] = None
43
+ allocator_tofu_apply_duration_seconds: Optional[float] = None
44
+ # The manual provider's equivalent of the three OpenTofu phases above:
45
+ # one `docker compose up -d`, image pull included. Kept as its own field
46
+ # rather than reusing the apply timing, which would make a compose deploy
47
+ # look like an OpenTofu one in the exported CSV.
48
+ allocator_compose_up_duration_seconds: Optional[float] = None
49
+ allocator_health_check_duration_seconds: Optional[float] = None
50
+ allocator_total_deployment_duration_seconds: Optional[float] = None
51
+ status: str = "in_progress"
52
+ error: Optional[str] = None
53
+
54
+
55
+ def _slugify_timestamp(dt: datetime) -> str:
56
+ return dt.isoformat().replace(":", "-")
57
+
58
+
59
+ def cache_path_for(deployment_name: str, start_time: datetime) -> Path:
60
+ return DEPLOYMENTS_DIR / f"{deployment_name}-{_slugify_timestamp(start_time)}.json"
61
+
62
+
63
+ def write_metrics(path: Path, metrics: DeploymentMetrics) -> None:
64
+ path.parent.mkdir(parents=True, exist_ok=True)
65
+ tmp = path.with_suffix(".json.tmp")
66
+ tmp.write_text(json.dumps(asdict(metrics), indent=2, sort_keys=True))
67
+ tmp.replace(path)
68
+
69
+
70
+ def load_all_metrics() -> list[dict]:
71
+ if not DEPLOYMENTS_DIR.exists():
72
+ return []
73
+ out = []
74
+ for p in sorted(DEPLOYMENTS_DIR.glob("*.json")):
75
+ try:
76
+ out.append(json.loads(p.read_text()))
77
+ except json.JSONDecodeError:
78
+ continue
79
+ return out
80
+
81
+
82
+ @contextmanager
83
+ def phase_timer(
84
+ metrics: DeploymentMetrics,
85
+ field_name: str,
86
+ path: Path,
87
+ ) -> Iterator[None]:
88
+ """Time a code block with monotonic clock; persist on exit (even on error)."""
89
+ start = time.monotonic()
90
+ try:
91
+ yield
92
+ finally:
93
+ setattr(metrics, field_name, round(time.monotonic() - start, 3))
94
+ write_metrics(path, metrics)