lablink-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lablink_cli/__init__.py +8 -0
- lablink_cli/api.py +428 -0
- lablink_cli/app.py +938 -0
- lablink_cli/byo_detect.py +112 -0
- lablink_cli/commands/__init__.py +0 -0
- lablink_cli/commands/cleanup.py +647 -0
- lablink_cli/commands/deploy.py +863 -0
- lablink_cli/commands/deploy_compose.py +1203 -0
- lablink_cli/commands/doctor.py +549 -0
- lablink_cli/commands/export_metrics.py +244 -0
- lablink_cli/commands/launch.py +236 -0
- lablink_cli/commands/logs.py +434 -0
- lablink_cli/commands/register.py +839 -0
- lablink_cli/commands/reset_overlay.py +109 -0
- lablink_cli/commands/setup.py +347 -0
- lablink_cli/commands/stats.py +133 -0
- lablink_cli/commands/status.py +934 -0
- lablink_cli/commands/unregister.py +188 -0
- lablink_cli/commands/utils.py +552 -0
- lablink_cli/config/__init__.py +0 -0
- lablink_cli/config/schema.py +212 -0
- lablink_cli/deployment_metrics.py +94 -0
- lablink_cli/docker.py +419 -0
- lablink_cli/log_shipper.py +441 -0
- lablink_cli/templates/docker-compose.tailscale-override.yml +55 -0
- lablink_cli/templates/docker-compose.yml +67 -0
- lablink_cli/tofu_source.py +169 -0
- lablink_cli/tui/__init__.py +0 -0
- lablink_cli/tui/logs_viewer.py +413 -0
- lablink_cli/tui/wizard.py +1814 -0
- lablink_cli-0.1.0.dist-info/METADATA +76 -0
- lablink_cli-0.1.0.dist-info/RECORD +35 -0
- lablink_cli-0.1.0.dist-info/WHEEL +5 -0
- lablink_cli-0.1.0.dist-info/entry_points.txt +2 -0
- lablink_cli-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
"""Configuration schema for LabLink deployments.
|
|
2
|
+
|
|
3
|
+
Re-exports the canonical Config dataclass from the allocator package
|
|
4
|
+
and adds CLI-specific helpers (YAML serialization, validation, reference data).
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import re
|
|
10
|
+
from dataclasses import fields
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
import yaml
|
|
15
|
+
|
|
16
|
+
# Re-export the canonical config dataclasses from the allocator.
|
|
17
|
+
from lablink_allocator_service.validate_config import (
|
|
18
|
+
get_config_errors,
|
|
19
|
+
)
|
|
20
|
+
from lablink_allocator_service.conf.structured_config import ( # noqa: F401
|
|
21
|
+
AllocatorConfig,
|
|
22
|
+
AppConfig,
|
|
23
|
+
Config,
|
|
24
|
+
DatabaseConfig,
|
|
25
|
+
DNSConfig,
|
|
26
|
+
EIPConfig,
|
|
27
|
+
MachineConfig,
|
|
28
|
+
SSLConfig,
|
|
29
|
+
StartupConfig,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def config_to_dict(cfg: Any) -> Any:
|
|
34
|
+
"""Recursively convert a dataclass to a nested dict."""
|
|
35
|
+
if hasattr(cfg, "__dataclass_fields__"):
|
|
36
|
+
return {
|
|
37
|
+
f.name: config_to_dict(getattr(cfg, f.name))
|
|
38
|
+
for f in fields(cfg)
|
|
39
|
+
}
|
|
40
|
+
return cfg
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def load_config(path: Path) -> Config:
|
|
44
|
+
"""Load a Config from a YAML file."""
|
|
45
|
+
with open(path) as f:
|
|
46
|
+
data = yaml.safe_load(f)
|
|
47
|
+
|
|
48
|
+
cfg = Config()
|
|
49
|
+
for key, value in data.items():
|
|
50
|
+
if hasattr(cfg, key) and isinstance(value, dict):
|
|
51
|
+
sub = getattr(cfg, key)
|
|
52
|
+
for k, v in value.items():
|
|
53
|
+
if isinstance(v, dict):
|
|
54
|
+
# Nested sub-config (3-level YAML structure)
|
|
55
|
+
nested = getattr(sub, k, None)
|
|
56
|
+
if nested and hasattr(nested, "__dataclass_fields__"):
|
|
57
|
+
for nk, nv in v.items():
|
|
58
|
+
setattr(nested, nk, nv)
|
|
59
|
+
else:
|
|
60
|
+
setattr(sub, k, v)
|
|
61
|
+
else:
|
|
62
|
+
setattr(sub, k, v)
|
|
63
|
+
elif hasattr(cfg, key):
|
|
64
|
+
setattr(cfg, key, value)
|
|
65
|
+
return cfg
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def save_config(cfg: Config, path: Path) -> None:
|
|
69
|
+
"""Write a Config to a YAML file."""
|
|
70
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
71
|
+
with open(path, "w") as f:
|
|
72
|
+
yaml.dump(
|
|
73
|
+
config_to_dict(cfg),
|
|
74
|
+
f,
|
|
75
|
+
default_flow_style=False,
|
|
76
|
+
sort_keys=False,
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
DEPLOYMENT_NAME_RE = re.compile(r"^[a-z][a-z0-9-]*[a-z0-9]$")
|
|
81
|
+
VALID_ENVIRONMENTS = ("dev", "test", "ci-test", "prod")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def validate_config(cfg: Config) -> list[str]:
|
|
85
|
+
"""Return a list of validation errors (empty = valid).
|
|
86
|
+
|
|
87
|
+
Mirrors the allocator's validate_config_logic but returns a
|
|
88
|
+
simple list instead of a tuple, suitable for TUI display.
|
|
89
|
+
"""
|
|
90
|
+
errors: list[str] = []
|
|
91
|
+
# deployment_name validation
|
|
92
|
+
if not cfg.deployment_name:
|
|
93
|
+
errors.append(
|
|
94
|
+
"deployment_name is required "
|
|
95
|
+
"(e.g., 'sleap-lablink')"
|
|
96
|
+
)
|
|
97
|
+
elif (
|
|
98
|
+
len(cfg.deployment_name) < 3
|
|
99
|
+
or len(cfg.deployment_name) > 32
|
|
100
|
+
or not DEPLOYMENT_NAME_RE.match(cfg.deployment_name)
|
|
101
|
+
):
|
|
102
|
+
errors.append(
|
|
103
|
+
"deployment_name must be 3-32 characters, "
|
|
104
|
+
"lowercase kebab-case (e.g., 'sleap-lablink')"
|
|
105
|
+
)
|
|
106
|
+
# environment validation
|
|
107
|
+
if cfg.environment not in VALID_ENVIRONMENTS:
|
|
108
|
+
errors.append(
|
|
109
|
+
f"environment must be one of: "
|
|
110
|
+
f"{', '.join(VALID_ENVIRONMENTS)}"
|
|
111
|
+
)
|
|
112
|
+
# DNS/SSL validation — shared with allocator's validate_config
|
|
113
|
+
errors.extend(get_config_errors(cfg))
|
|
114
|
+
return errors
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
# AMI IDs by region (Ubuntu 24.04 with Docker + Nvidia GPU Driver)
|
|
118
|
+
AMI_MAP: dict[str, str] = {
|
|
119
|
+
"us-east-1": "ami-0601752c11b394251",
|
|
120
|
+
"us-east-2": "ami-0601752c11b394251",
|
|
121
|
+
"us-west-1": "ami-0601752c11b394251",
|
|
122
|
+
"us-west-2": "ami-0601752c11b394251",
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
# Common GPU instance types
|
|
126
|
+
GPU_INSTANCE_TYPES: list[dict[str, str]] = [
|
|
127
|
+
{
|
|
128
|
+
"type": "g4dn.xlarge",
|
|
129
|
+
"gpu": "T4 16GB",
|
|
130
|
+
"vcpu": "4",
|
|
131
|
+
"ram": "16 GB",
|
|
132
|
+
"cost": "~$0.53/hr",
|
|
133
|
+
},
|
|
134
|
+
{
|
|
135
|
+
"type": "g4dn.2xlarge",
|
|
136
|
+
"gpu": "T4 16GB",
|
|
137
|
+
"vcpu": "8",
|
|
138
|
+
"ram": "32 GB",
|
|
139
|
+
"cost": "~$0.75/hr",
|
|
140
|
+
},
|
|
141
|
+
{
|
|
142
|
+
"type": "g5.xlarge",
|
|
143
|
+
"gpu": "A10G 24GB",
|
|
144
|
+
"vcpu": "4",
|
|
145
|
+
"ram": "16 GB",
|
|
146
|
+
"cost": "~$1.01/hr",
|
|
147
|
+
},
|
|
148
|
+
{
|
|
149
|
+
"type": "g5.2xlarge",
|
|
150
|
+
"gpu": "A10G 24GB",
|
|
151
|
+
"vcpu": "8",
|
|
152
|
+
"ram": "32 GB",
|
|
153
|
+
"cost": "~$1.21/hr",
|
|
154
|
+
},
|
|
155
|
+
{
|
|
156
|
+
"type": "p3.2xlarge",
|
|
157
|
+
"gpu": "V100 16GB",
|
|
158
|
+
"vcpu": "8",
|
|
159
|
+
"ram": "61 GB",
|
|
160
|
+
"cost": "~$3.06/hr",
|
|
161
|
+
},
|
|
162
|
+
]
|
|
163
|
+
|
|
164
|
+
# Common CPU instance types (no GPU)
|
|
165
|
+
CPU_INSTANCE_TYPES: list[dict[str, str]] = [
|
|
166
|
+
{
|
|
167
|
+
"type": "t3.large",
|
|
168
|
+
"gpu": "—",
|
|
169
|
+
"vcpu": "2",
|
|
170
|
+
"ram": "8 GB",
|
|
171
|
+
"cost": "~$0.08/hr",
|
|
172
|
+
},
|
|
173
|
+
{
|
|
174
|
+
"type": "t3.xlarge",
|
|
175
|
+
"gpu": "—",
|
|
176
|
+
"vcpu": "4",
|
|
177
|
+
"ram": "16 GB",
|
|
178
|
+
"cost": "~$0.17/hr",
|
|
179
|
+
},
|
|
180
|
+
{
|
|
181
|
+
"type": "t3.2xlarge",
|
|
182
|
+
"gpu": "—",
|
|
183
|
+
"vcpu": "8",
|
|
184
|
+
"ram": "32 GB",
|
|
185
|
+
"cost": "~$0.33/hr",
|
|
186
|
+
},
|
|
187
|
+
{
|
|
188
|
+
"type": "m5.xlarge",
|
|
189
|
+
"gpu": "—",
|
|
190
|
+
"vcpu": "4",
|
|
191
|
+
"ram": "16 GB",
|
|
192
|
+
"cost": "~$0.19/hr",
|
|
193
|
+
},
|
|
194
|
+
{
|
|
195
|
+
"type": "m5.2xlarge",
|
|
196
|
+
"gpu": "—",
|
|
197
|
+
"vcpu": "8",
|
|
198
|
+
"ram": "32 GB",
|
|
199
|
+
"cost": "~$0.38/hr",
|
|
200
|
+
},
|
|
201
|
+
]
|
|
202
|
+
|
|
203
|
+
AWS_REGIONS: list[dict[str, str]] = [
|
|
204
|
+
{"id": "us-east-1", "name": "US East (N. Virginia)"},
|
|
205
|
+
{"id": "us-east-2", "name": "US East (Ohio)"},
|
|
206
|
+
{"id": "us-west-1", "name": "US West (N. California)"},
|
|
207
|
+
{"id": "us-west-2", "name": "US West (Oregon)"},
|
|
208
|
+
{"id": "eu-west-1", "name": "Europe (Ireland)"},
|
|
209
|
+
{"id": "eu-central-1", "name": "Europe (Frankfurt)"},
|
|
210
|
+
{"id": "ap-northeast-1", "name": "Asia Pacific (Tokyo)"},
|
|
211
|
+
{"id": "ap-southeast-1", "name": "Asia Pacific (Singapore)"},
|
|
212
|
+
]
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""CLI-local cache for allocator deployment metrics (issue #317).
|
|
2
|
+
|
|
3
|
+
Stored on the operator's machine under ``~/.lablink/deployments/`` so that
|
|
4
|
+
metrics for failed deploys (or for already-destroyed allocators) survive.
|
|
5
|
+
|
|
6
|
+
Records start life with ``status="in_progress"`` and are promoted to
|
|
7
|
+
``success`` / ``failed`` by :func:`~lablink_cli.commands.deploy.run_deploy`.
|
|
8
|
+
Plan-confirmation cancels and Ctrl-C leave the file in ``in_progress``
|
|
9
|
+
indefinitely — use ``lablink cache-clear --deployments --stale`` to prune
|
|
10
|
+
just those, or ``--deployments`` to wipe the whole cache.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import time
|
|
17
|
+
from contextlib import contextmanager
|
|
18
|
+
from dataclasses import asdict, dataclass
|
|
19
|
+
from datetime import datetime
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Iterator, Optional
|
|
22
|
+
|
|
23
|
+
DEPLOYMENTS_DIR = Path.home() / ".lablink" / "deployments"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class DeploymentMetrics:
|
|
28
|
+
deployment_name: str
|
|
29
|
+
# Which deploy path produced this record. deployment_name alone does not
|
|
30
|
+
# identify it: an operator can reuse one name across providers (observed:
|
|
31
|
+
# a `sleap-lablink` config switched from aws to manual), and the AWS
|
|
32
|
+
# OpenTofu timings below are meaningless for a compose deploy. Records
|
|
33
|
+
# written before this field existed are all AWS ones, which is why
|
|
34
|
+
# readers default a missing value to "aws" rather than to None.
|
|
35
|
+
provider: Optional[str] = None
|
|
36
|
+
region: Optional[str] = None
|
|
37
|
+
template_version: Optional[str] = None
|
|
38
|
+
ssl_enabled: Optional[bool] = None
|
|
39
|
+
allocator_deploy_start_time: Optional[str] = None
|
|
40
|
+
allocator_deploy_end_time: Optional[str] = None
|
|
41
|
+
allocator_tofu_init_duration_seconds: Optional[float] = None
|
|
42
|
+
allocator_tofu_plan_duration_seconds: Optional[float] = None
|
|
43
|
+
allocator_tofu_apply_duration_seconds: Optional[float] = None
|
|
44
|
+
# The manual provider's equivalent of the three OpenTofu phases above:
|
|
45
|
+
# one `docker compose up -d`, image pull included. Kept as its own field
|
|
46
|
+
# rather than reusing the apply timing, which would make a compose deploy
|
|
47
|
+
# look like an OpenTofu one in the exported CSV.
|
|
48
|
+
allocator_compose_up_duration_seconds: Optional[float] = None
|
|
49
|
+
allocator_health_check_duration_seconds: Optional[float] = None
|
|
50
|
+
allocator_total_deployment_duration_seconds: Optional[float] = None
|
|
51
|
+
status: str = "in_progress"
|
|
52
|
+
error: Optional[str] = None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def _slugify_timestamp(dt: datetime) -> str:
|
|
56
|
+
return dt.isoformat().replace(":", "-")
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def cache_path_for(deployment_name: str, start_time: datetime) -> Path:
|
|
60
|
+
return DEPLOYMENTS_DIR / f"{deployment_name}-{_slugify_timestamp(start_time)}.json"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def write_metrics(path: Path, metrics: DeploymentMetrics) -> None:
|
|
64
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
65
|
+
tmp = path.with_suffix(".json.tmp")
|
|
66
|
+
tmp.write_text(json.dumps(asdict(metrics), indent=2, sort_keys=True))
|
|
67
|
+
tmp.replace(path)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def load_all_metrics() -> list[dict]:
|
|
71
|
+
if not DEPLOYMENTS_DIR.exists():
|
|
72
|
+
return []
|
|
73
|
+
out = []
|
|
74
|
+
for p in sorted(DEPLOYMENTS_DIR.glob("*.json")):
|
|
75
|
+
try:
|
|
76
|
+
out.append(json.loads(p.read_text()))
|
|
77
|
+
except json.JSONDecodeError:
|
|
78
|
+
continue
|
|
79
|
+
return out
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@contextmanager
|
|
83
|
+
def phase_timer(
|
|
84
|
+
metrics: DeploymentMetrics,
|
|
85
|
+
field_name: str,
|
|
86
|
+
path: Path,
|
|
87
|
+
) -> Iterator[None]:
|
|
88
|
+
"""Time a code block with monotonic clock; persist on exit (even on error)."""
|
|
89
|
+
start = time.monotonic()
|
|
90
|
+
try:
|
|
91
|
+
yield
|
|
92
|
+
finally:
|
|
93
|
+
setattr(metrics, field_name, round(time.monotonic() - start, 3))
|
|
94
|
+
write_metrics(path, metrics)
|