duckless 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- duckless/__init__.py +5 -0
- duckless/adapters/__init__.py +1 -0
- duckless/adapters/batch.py +148 -0
- duckless/adapters/cloud_logging.py +45 -0
- duckless/adapters/compute_quotas.py +15 -0
- duckless/adapters/gcp_bootstrap.py +102 -0
- duckless/adapters/gcs.py +42 -0
- duckless/adapters/infra_manager.py +113 -0
- duckless/cli.py +331 -0
- duckless/core/__init__.py +1 -0
- duckless/core/errors.py +21 -0
- duckless/core/infra.py +128 -0
- duckless/core/job.py +175 -0
- duckless/core/machine.py +86 -0
- duckless/core/preflight.py +41 -0
- duckless/core/quota.py +59 -0
- duckless/ports.py +98 -0
- duckless/py.typed +0 -0
- duckless/service.py +119 -0
- duckless/settings.py +61 -0
- duckless/terraform/main.tf +109 -0
- duckless/terraform/outputs.tf +25 -0
- duckless/terraform/variables.tf +64 -0
- duckless/terraform/versions.tf +10 -0
- duckless/wiring.py +137 -0
- duckless-0.1.0.dist-info/METADATA +86 -0
- duckless-0.1.0.dist-info/RECORD +29 -0
- duckless-0.1.0.dist-info/WHEEL +4 -0
- duckless-0.1.0.dist-info/entry_points.txt +3 -0
duckless/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Driven adapters: GCP implementations of duckless.ports."""
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""Executor on Cloud Batch: one task, one VM per job, deleted when the job ends."""
|
|
2
|
+
|
|
3
|
+
from google.api_core.exceptions import NotFound
|
|
4
|
+
from google.cloud import batch_v1
|
|
5
|
+
|
|
6
|
+
from duckless.core.errors import JobNotFoundError
|
|
7
|
+
from duckless.core.job import JobEvent, JobKind, JobSpec, JobState, JobStatus
|
|
8
|
+
from duckless.core.machine import LOCAL_SSD_GB
|
|
9
|
+
from duckless.settings import Settings
|
|
10
|
+
|
|
11
|
+
SCRATCH_PATH = "/mnt/disks/scratch"
|
|
12
|
+
SCRATCH_DEVICE = "scratch"
|
|
13
|
+
# Runner exit codes (job failed / bad usage): fail at once, keep retries for infra failures
|
|
14
|
+
# such as a Spot preemption (exit 50001). Batch allows a single lifecycle policy per task.
|
|
15
|
+
RUNNER_EXIT_CODES = (1, 2)
|
|
16
|
+
MAX_RETRIES = 2
|
|
17
|
+
|
|
18
|
+
_STATES = {
|
|
19
|
+
batch_v1.JobStatus.State.QUEUED: JobState.QUEUED,
|
|
20
|
+
batch_v1.JobStatus.State.SCHEDULED: JobState.SCHEDULED,
|
|
21
|
+
batch_v1.JobStatus.State.RUNNING: JobState.RUNNING,
|
|
22
|
+
batch_v1.JobStatus.State.SUCCEEDED: JobState.SUCCEEDED,
|
|
23
|
+
batch_v1.JobStatus.State.FAILED: JobState.FAILED,
|
|
24
|
+
batch_v1.JobStatus.State.CANCELLED: JobState.CANCELLED,
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# ---------- pure ----------
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def container_invocation(spec: JobSpec) -> tuple[str, list[str]]:
|
|
32
|
+
"""(entrypoint, args). Runner jobs keep the image entrypoint (`python -m duckless_runtime`) and
|
|
33
|
+
pass `sql|py <uri>`; a command job replaces the entrypoint, so `-- dbt build` runs dbt itself."""
|
|
34
|
+
if spec.kind is JobKind.COMMAND:
|
|
35
|
+
return spec.command[0], list(spec.command[1:])
|
|
36
|
+
return "", list(spec.runner_args)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def build_job(spec: JobSpec, settings: Settings) -> batch_v1.Job:
|
|
40
|
+
has_scratch = spec.local_ssd_count > 0
|
|
41
|
+
entrypoint, args = container_invocation(spec)
|
|
42
|
+
# Local SSD is mounted by Batch on the host as root; open it to the non-root runner user.
|
|
43
|
+
prepare_scratch = batch_v1.Runnable(
|
|
44
|
+
script=batch_v1.Runnable.Script(text=f"mkdir -p {SCRATCH_PATH} && chmod 1777 {SCRATCH_PATH}")
|
|
45
|
+
)
|
|
46
|
+
runner = batch_v1.Runnable(
|
|
47
|
+
# Task volumes are bind-mounted into the container at the same path by default.
|
|
48
|
+
container=batch_v1.Runnable.Container(image_uri=spec.image, entrypoint=entrypoint, commands=args),
|
|
49
|
+
environment=batch_v1.Environment(variables={**dict(spec.env), "GOOGLE_CLOUD_PROJECT": settings.project}),
|
|
50
|
+
)
|
|
51
|
+
task = batch_v1.TaskSpec(
|
|
52
|
+
runnables=[prepare_scratch, runner] if has_scratch else [runner],
|
|
53
|
+
volumes=[batch_v1.Volume(device_name=SCRATCH_DEVICE, mount_path=SCRATCH_PATH)] if has_scratch else [],
|
|
54
|
+
max_run_duration=f"{spec.max_run_seconds}s",
|
|
55
|
+
max_retry_count=MAX_RETRIES,
|
|
56
|
+
lifecycle_policies=[
|
|
57
|
+
batch_v1.LifecyclePolicy(
|
|
58
|
+
action=batch_v1.LifecyclePolicy.Action.FAIL_TASK,
|
|
59
|
+
action_condition=batch_v1.LifecyclePolicy.ActionCondition(exit_codes=list(RUNNER_EXIT_CODES)),
|
|
60
|
+
)
|
|
61
|
+
],
|
|
62
|
+
)
|
|
63
|
+
policy = batch_v1.AllocationPolicy.InstancePolicy(
|
|
64
|
+
machine_type=spec.machine.name,
|
|
65
|
+
provisioning_model=(
|
|
66
|
+
batch_v1.AllocationPolicy.ProvisioningModel.SPOT
|
|
67
|
+
if spec.spot
|
|
68
|
+
else batch_v1.AllocationPolicy.ProvisioningModel.STANDARD
|
|
69
|
+
),
|
|
70
|
+
disks=[
|
|
71
|
+
batch_v1.AllocationPolicy.AttachedDisk(
|
|
72
|
+
new_disk=batch_v1.AllocationPolicy.Disk(type_="local-ssd", size_gb=LOCAL_SSD_GB * spec.local_ssd_count),
|
|
73
|
+
device_name=SCRATCH_DEVICE,
|
|
74
|
+
)
|
|
75
|
+
]
|
|
76
|
+
if has_scratch
|
|
77
|
+
else [],
|
|
78
|
+
)
|
|
79
|
+
allocation = batch_v1.AllocationPolicy(
|
|
80
|
+
# Any zone of the region: Spot + highmem + local SSD stocks out zone by zone.
|
|
81
|
+
location=batch_v1.AllocationPolicy.LocationPolicy(allowed_locations=[f"regions/{settings.region}"]),
|
|
82
|
+
instances=[batch_v1.AllocationPolicy.InstancePolicyOrTemplate(policy=policy)],
|
|
83
|
+
service_account=batch_v1.ServiceAccount(email=settings.service_account),
|
|
84
|
+
network=batch_v1.AllocationPolicy.NetworkPolicy(
|
|
85
|
+
network_interfaces=[
|
|
86
|
+
batch_v1.AllocationPolicy.NetworkInterface(
|
|
87
|
+
network=settings.network,
|
|
88
|
+
subnetwork=settings.subnetwork,
|
|
89
|
+
no_external_ip_address=not settings.external_ip,
|
|
90
|
+
)
|
|
91
|
+
]
|
|
92
|
+
),
|
|
93
|
+
)
|
|
94
|
+
return batch_v1.Job(
|
|
95
|
+
task_groups=[batch_v1.TaskGroup(task_spec=task, task_count=1)],
|
|
96
|
+
allocation_policy=allocation,
|
|
97
|
+
logs_policy=batch_v1.LogsPolicy(destination=batch_v1.LogsPolicy.Destination.CLOUD_LOGGING),
|
|
98
|
+
labels={"app": "duckless", "duckless-kind": spec.kind.value},
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def status_from_job(job: batch_v1.Job) -> JobStatus:
|
|
103
|
+
policy = job.allocation_policy.instances[0].policy
|
|
104
|
+
return JobStatus(
|
|
105
|
+
job_id=job.name.rsplit("/", 1)[-1],
|
|
106
|
+
uid=job.uid,
|
|
107
|
+
state=_STATES.get(job.status.state, JobState.UNKNOWN),
|
|
108
|
+
machine=policy.machine_type,
|
|
109
|
+
spot=policy.provisioning_model == batch_v1.AllocationPolicy.ProvisioningModel.SPOT,
|
|
110
|
+
created_at=job.create_time,
|
|
111
|
+
run_seconds=job.status.run_duration.total_seconds() if job.status.run_duration else None,
|
|
112
|
+
events=tuple(
|
|
113
|
+
JobEvent(at=e.event_time, description=e.description.split(" for job ")[0]) for e in job.status.status_events
|
|
114
|
+
),
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
# ---------- adapter ----------
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class BatchExecutor:
|
|
122
|
+
def __init__(self, client: batch_v1.BatchServiceClient, settings: Settings) -> None:
|
|
123
|
+
self._client = client
|
|
124
|
+
self._settings = settings
|
|
125
|
+
|
|
126
|
+
def _parent(self) -> str:
|
|
127
|
+
return f"projects/{self._settings.project}/locations/{self._settings.region}"
|
|
128
|
+
|
|
129
|
+
def _name(self, job_id: str) -> str:
|
|
130
|
+
return f"{self._parent()}/jobs/{job_id}"
|
|
131
|
+
|
|
132
|
+
def submit(self, spec: JobSpec) -> JobStatus:
|
|
133
|
+
created = self._client.create_job(
|
|
134
|
+
parent=self._parent(), job_id=spec.job_id, job=build_job(spec, self._settings)
|
|
135
|
+
)
|
|
136
|
+
return status_from_job(created)
|
|
137
|
+
|
|
138
|
+
def get(self, job_id: str) -> JobStatus:
|
|
139
|
+
try:
|
|
140
|
+
return status_from_job(self._client.get_job(name=self._name(job_id)))
|
|
141
|
+
except NotFound as e:
|
|
142
|
+
raise JobNotFoundError(job_id) from e
|
|
143
|
+
|
|
144
|
+
def cancel(self, job_id: str) -> None:
|
|
145
|
+
try:
|
|
146
|
+
self._client.cancel_job(name=self._name(job_id))
|
|
147
|
+
except NotFound as e:
|
|
148
|
+
raise JobNotFoundError(job_id) from e
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Runner logs from Cloud Logging: Batch writes task stdout to `batch_task_logs`, labelled by job uid."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
|
|
5
|
+
from google.cloud import logging as cloud_logging
|
|
6
|
+
|
|
7
|
+
from duckless.core.job import LogLine
|
|
8
|
+
|
|
9
|
+
TASK_LOG = "batch_task_logs"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def log_filter(project: str, job_uid: str, since: datetime | None) -> str:
|
|
13
|
+
clauses = (
|
|
14
|
+
f'logName="projects/{project}/logs/{TASK_LOG}"',
|
|
15
|
+
f'labels.job_uid="{job_uid}"',
|
|
16
|
+
*((f'timestamp>"{since.isoformat()}"',) if since else ()),
|
|
17
|
+
)
|
|
18
|
+
return " AND ".join(clauses)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def to_line(entry: cloud_logging.LogEntry) -> LogLine:
|
|
22
|
+
"""The runner logs JSON lines (jsonPayload); anything else (image pull, crash) is plain text."""
|
|
23
|
+
payload = entry.payload if isinstance(entry.payload, dict) else {"message": str(entry.payload)}
|
|
24
|
+
fields = {k: v for k, v in payload.items() if k not in {"message", "severity", "time"}}
|
|
25
|
+
return LogLine(
|
|
26
|
+
at=entry.timestamp,
|
|
27
|
+
severity=entry.severity or "DEFAULT",
|
|
28
|
+
message=str(payload.get("message", "")),
|
|
29
|
+
fields=fields,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class CloudLoggingLogReader:
|
|
34
|
+
def __init__(self, client: cloud_logging.Client, project: str) -> None:
|
|
35
|
+
self._client = client
|
|
36
|
+
self._project = project
|
|
37
|
+
|
|
38
|
+
def read(self, job_uid: str, since: datetime | None = None, limit: int = 200) -> tuple[LogLine, ...]:
|
|
39
|
+
entries = self._client.list_entries(
|
|
40
|
+
resource_names=[f"projects/{self._project}"],
|
|
41
|
+
filter_=log_filter(self._project, job_uid, since),
|
|
42
|
+
order_by=cloud_logging.ASCENDING,
|
|
43
|
+
max_results=limit,
|
|
44
|
+
)
|
|
45
|
+
return tuple(map(to_line, entries))
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from collections.abc import Mapping
|
|
2
|
+
|
|
3
|
+
from google.cloud import compute_v1
|
|
4
|
+
|
|
5
|
+
from duckless.core.quota import Quota
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ComputeQuotaReader:
|
|
9
|
+
def __init__(self, client: compute_v1.RegionsClient, project: str) -> None:
|
|
10
|
+
self._client = client
|
|
11
|
+
self._project = project
|
|
12
|
+
|
|
13
|
+
def regional_quotas(self, region: str) -> Mapping[str, Quota]:
|
|
14
|
+
regional = self._client.get(project=self._project, region=region)
|
|
15
|
+
return {q.metric: Quota(usage=q.usage, limit=q.limit) for q in regional.quotas}
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""InfraBootstrap with the caller's credentials: plain REST for the few one-off calls, the
|
|
2
|
+
storage client for the staging bucket."""
|
|
3
|
+
|
|
4
|
+
import time
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from google.auth.transport.requests import AuthorizedSession
|
|
8
|
+
from google.cloud import storage
|
|
9
|
+
|
|
10
|
+
from duckless.core.infra import with_bindings, without_bindings
|
|
11
|
+
|
|
12
|
+
SERVICE_USAGE = "https://serviceusage.googleapis.com/v1"
|
|
13
|
+
IAM = "https://iam.googleapis.com/v1"
|
|
14
|
+
RESOURCE_MANAGER = "https://cloudresourcemanager.googleapis.com/v3"
|
|
15
|
+
SKIPPED_PARTS = frozenset({".terraform", ".terraform.lock.hcl", "__pycache__"})
|
|
16
|
+
OPERATION_POLL_SECONDS = 3
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def module_files(local_dir: Path) -> list[Path]:
|
|
20
|
+
"""Files of the Terraform module, without local Terraform state or caches."""
|
|
21
|
+
return sorted(
|
|
22
|
+
p
|
|
23
|
+
for p in local_dir.rglob("*")
|
|
24
|
+
if p.is_file() and not SKIPPED_PARTS.intersection(p.relative_to(local_dir).parts)
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class GcpInfraBootstrap:
|
|
29
|
+
def __init__(self, session: AuthorizedSession, storage_client: storage.Client) -> None:
|
|
30
|
+
self._session = session
|
|
31
|
+
self._storage = storage_client
|
|
32
|
+
|
|
33
|
+
def _call(self, method: str, url: str, **kwargs) -> dict:
|
|
34
|
+
response = self._session.request(method, url, **kwargs)
|
|
35
|
+
response.raise_for_status()
|
|
36
|
+
return response.json() if response.content else {}
|
|
37
|
+
|
|
38
|
+
def _wait(self, base: str, operation: dict) -> None:
|
|
39
|
+
while not operation.get("done"):
|
|
40
|
+
time.sleep(OPERATION_POLL_SECONDS)
|
|
41
|
+
operation = self._call("GET", f"{base}/{operation['name']}")
|
|
42
|
+
if "error" in operation:
|
|
43
|
+
raise RuntimeError(f"operation {operation['name']} failed: {operation['error']}")
|
|
44
|
+
|
|
45
|
+
def enable_apis(self, project: str, apis: tuple[str, ...]) -> None:
|
|
46
|
+
operation = self._call(
|
|
47
|
+
"POST", f"{SERVICE_USAGE}/projects/{project}/services:batchEnable", json={"serviceIds": list(apis)}
|
|
48
|
+
)
|
|
49
|
+
self._wait(SERVICE_USAGE, operation)
|
|
50
|
+
|
|
51
|
+
def ensure_service_account(self, project: str, account_id: str, display_name: str) -> str:
|
|
52
|
+
email = f"{account_id}@{project}.iam.gserviceaccount.com"
|
|
53
|
+
existing = self._session.get(f"{IAM}/projects/{project}/serviceAccounts/{email}")
|
|
54
|
+
if existing.status_code == 404:
|
|
55
|
+
self._call(
|
|
56
|
+
"POST",
|
|
57
|
+
f"{IAM}/projects/{project}/serviceAccounts",
|
|
58
|
+
json={"accountId": account_id, "serviceAccount": {"displayName": display_name}},
|
|
59
|
+
)
|
|
60
|
+
else:
|
|
61
|
+
existing.raise_for_status()
|
|
62
|
+
return email
|
|
63
|
+
|
|
64
|
+
def _update_policy(self, project: str, change) -> bool:
|
|
65
|
+
resource = f"{RESOURCE_MANAGER}/projects/{project}"
|
|
66
|
+
policy = self._call("POST", f"{resource}:getIamPolicy", json={"options": {"requestedPolicyVersion": 3}})
|
|
67
|
+
updated, changed = change(policy)
|
|
68
|
+
if changed:
|
|
69
|
+
# The etag in `updated` makes a concurrent change fail instead of being overwritten.
|
|
70
|
+
self._call("POST", f"{resource}:setIamPolicy", json={"policy": updated})
|
|
71
|
+
return changed
|
|
72
|
+
|
|
73
|
+
def grant_project_roles(self, project: str, member: str, roles: tuple[str, ...]) -> bool:
|
|
74
|
+
return self._update_policy(project, lambda policy: with_bindings(policy, member, roles))
|
|
75
|
+
|
|
76
|
+
def revoke_project_roles(self, project: str, member: str, roles: tuple[str, ...]) -> None:
|
|
77
|
+
self._update_policy(project, lambda policy: without_bindings(policy, member, roles))
|
|
78
|
+
|
|
79
|
+
def delete_service_account(self, project: str, email: str) -> None:
|
|
80
|
+
response = self._session.delete(f"{IAM}/projects/{project}/serviceAccounts/{email}")
|
|
81
|
+
if response.status_code != 404:
|
|
82
|
+
response.raise_for_status()
|
|
83
|
+
|
|
84
|
+
def delete_bucket(self, bucket: str) -> None:
|
|
85
|
+
existing = self._storage.lookup_bucket(bucket)
|
|
86
|
+
if existing is not None:
|
|
87
|
+
existing.delete(force=True)
|
|
88
|
+
|
|
89
|
+
def ensure_bucket(self, project: str, region: str, bucket: str) -> None:
|
|
90
|
+
if self._storage.lookup_bucket(bucket) is not None:
|
|
91
|
+
return
|
|
92
|
+
new = self._storage.bucket(bucket)
|
|
93
|
+
new.iam_configuration.uniform_bucket_level_access_enabled = True
|
|
94
|
+
new.iam_configuration.public_access_prevention = "enforced"
|
|
95
|
+
new.labels = {"app": "duckless"}
|
|
96
|
+
self._storage.create_bucket(new, project=project, location=region)
|
|
97
|
+
|
|
98
|
+
def upload_directory(self, local_dir: Path, bucket: str, prefix: str) -> str:
|
|
99
|
+
target = self._storage.bucket(bucket)
|
|
100
|
+
for path in module_files(local_dir):
|
|
101
|
+
target.blob(f"{prefix}/{path.relative_to(local_dir).as_posix()}").upload_from_filename(str(path))
|
|
102
|
+
return f"gs://{bucket}/{prefix}"
|
duckless/adapters/gcs.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Work bucket layout: gs://<bucket>/runs/<job_id>/{<source>, metrics.json}."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from google.cloud import storage
|
|
9
|
+
|
|
10
|
+
RUNS_PREFIX = "runs"
|
|
11
|
+
METRICS_FILE = "metrics.json"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def run_object(job_id: str, name: str) -> str:
|
|
15
|
+
return f"{RUNS_PREFIX}/{job_id}/{name}"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def parse_metrics(raw: str) -> Mapping[str, Any] | None:
|
|
19
|
+
"""The runner writes metrics with DuckDB's COPY … (FORMAT json): one object per line."""
|
|
20
|
+
first = next((line for line in raw.splitlines() if line.strip()), None)
|
|
21
|
+
return json.loads(first) if first else None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class GcsArtifactStore:
|
|
25
|
+
def __init__(self, client: storage.Client, bucket: str) -> None:
|
|
26
|
+
self._bucket = client.bucket(bucket)
|
|
27
|
+
|
|
28
|
+
def upload_source(self, job_id: str, source: Path) -> str:
|
|
29
|
+
blob = self._bucket.blob(run_object(job_id, source.name))
|
|
30
|
+
blob.upload_from_filename(str(source))
|
|
31
|
+
return f"gs://{self._bucket.name}/{blob.name}"
|
|
32
|
+
|
|
33
|
+
def runner_env(self, job_id: str) -> Mapping[str, str]:
|
|
34
|
+
return {
|
|
35
|
+
"DUCKLESS_JOB_ID": job_id,
|
|
36
|
+
"DUCKLESS_BUCKET": self._bucket.name,
|
|
37
|
+
"DUCKLESS_METRICS_URI": f"gs://{self._bucket.name}/{run_object(job_id, METRICS_FILE)}",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
def read_metrics(self, job_id: str) -> Mapping[str, Any] | None:
|
|
41
|
+
blob = self._bucket.blob(run_object(job_id, METRICS_FILE))
|
|
42
|
+
return parse_metrics(blob.download_as_text()) if blob.exists() else None
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""InfraDeployer on Infrastructure Manager (Terraform run by Google, state kept in the project)."""
|
|
2
|
+
|
|
3
|
+
import contextlib
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from google.api_core.exceptions import GoogleAPICallError, NotFound
|
|
8
|
+
from google.cloud import config_v1
|
|
9
|
+
|
|
10
|
+
from duckless.core.infra import InfraStatus
|
|
11
|
+
|
|
12
|
+
APPLY_TIMEOUT_SECONDS = 30 * 60
|
|
13
|
+
LABELS = {"app": "duckless"}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def deployment_name(project: str, region: str, deployment_id: str) -> str:
|
|
17
|
+
return f"projects/{project}/locations/{region}/deployments/{deployment_id}"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def build_deployment(
|
|
21
|
+
name: str, source_uri: str, inputs: Mapping[str, Any], service_account: str
|
|
22
|
+
) -> config_v1.Deployment:
|
|
23
|
+
return config_v1.Deployment(
|
|
24
|
+
name=name,
|
|
25
|
+
service_account=service_account
|
|
26
|
+
if service_account.startswith("projects/")
|
|
27
|
+
else f"projects/{name.split('/')[1]}/serviceAccounts/{service_account}",
|
|
28
|
+
terraform_blueprint=config_v1.TerraformBlueprint(
|
|
29
|
+
gcs_source=source_uri,
|
|
30
|
+
input_values={k: config_v1.TerraformVariable(input_value=v) for k, v in inputs.items()},
|
|
31
|
+
),
|
|
32
|
+
labels=LABELS,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def failure_detail(deployment: config_v1.Deployment, revision: config_v1.Revision | None) -> str:
|
|
37
|
+
"""The most useful error Infra Manager kept: Terraform errors first, then the state detail."""
|
|
38
|
+
tf_errors = [e.error_description or e.resource_address for e in (revision.tf_errors if revision else ())]
|
|
39
|
+
tf_errors += [e.error_description or e.resource_address for e in deployment.tf_errors]
|
|
40
|
+
logs = revision.logs if revision and revision.logs else deployment.error_logs
|
|
41
|
+
parts = [
|
|
42
|
+
*tf_errors[:5],
|
|
43
|
+
deployment.state_detail or (revision.state_detail if revision else ""),
|
|
44
|
+
logs and f"logs: {logs}",
|
|
45
|
+
]
|
|
46
|
+
return "\n".join(p for p in parts if p) or deployment.state.name
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class InfraManagerDeployer:
|
|
50
|
+
def __init__(self, client: config_v1.ConfigClient) -> None:
|
|
51
|
+
self._client = client
|
|
52
|
+
|
|
53
|
+
def _revision(self, deployment: config_v1.Deployment) -> config_v1.Revision | None:
|
|
54
|
+
return self._client.get_revision(name=deployment.latest_revision) if deployment.latest_revision else None
|
|
55
|
+
|
|
56
|
+
def _status(self, deployment: config_v1.Deployment) -> InfraStatus:
|
|
57
|
+
revision = self._revision(deployment)
|
|
58
|
+
outputs = {k: o.value for k, o in revision.apply_results.outputs.items()} if revision else {}
|
|
59
|
+
failed = deployment.state == config_v1.Deployment.State.FAILED
|
|
60
|
+
return InfraStatus(
|
|
61
|
+
deployment=deployment.name,
|
|
62
|
+
state=deployment.state.name,
|
|
63
|
+
outputs=outputs,
|
|
64
|
+
error=failure_detail(deployment, revision) if failed else None,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
def get(self, project: str, region: str, deployment_id: str) -> InfraStatus | None:
|
|
68
|
+
try:
|
|
69
|
+
return self._status(self._client.get_deployment(name=deployment_name(project, region, deployment_id)))
|
|
70
|
+
except NotFound:
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
def apply(
|
|
74
|
+
self,
|
|
75
|
+
project: str,
|
|
76
|
+
region: str,
|
|
77
|
+
deployment_id: str,
|
|
78
|
+
source_uri: str,
|
|
79
|
+
inputs: Mapping[str, Any],
|
|
80
|
+
service_account: str,
|
|
81
|
+
) -> InfraStatus:
|
|
82
|
+
name = deployment_name(project, region, deployment_id)
|
|
83
|
+
deployment = build_deployment(name, source_uri, inputs, service_account)
|
|
84
|
+
exists = self.get(project, region, deployment_id) is not None
|
|
85
|
+
operation = (
|
|
86
|
+
self._client.update_deployment(deployment=deployment)
|
|
87
|
+
if exists
|
|
88
|
+
else self._client.create_deployment(
|
|
89
|
+
parent=f"projects/{project}/locations/{region}", deployment_id=deployment_id, deployment=deployment
|
|
90
|
+
)
|
|
91
|
+
)
|
|
92
|
+
# On failure the deployment carries the details, read below.
|
|
93
|
+
with contextlib.suppress(GoogleAPICallError):
|
|
94
|
+
operation.result(timeout=APPLY_TIMEOUT_SECONDS)
|
|
95
|
+
return self._status(self._client.get_deployment(name=name))
|
|
96
|
+
|
|
97
|
+
def destroy(self, project: str, region: str, deployment_id: str) -> InfraStatus:
|
|
98
|
+
name = deployment_name(project, region, deployment_id)
|
|
99
|
+
if self.get(project, region, deployment_id) is None:
|
|
100
|
+
return InfraStatus(name, "DELETED")
|
|
101
|
+
try:
|
|
102
|
+
# force: also delete the deployment's revisions (nested resources), else the call is refused.
|
|
103
|
+
self._client.delete_deployment(
|
|
104
|
+
request=config_v1.DeleteDeploymentRequest(
|
|
105
|
+
name=name, force=True, delete_policy=config_v1.DeleteDeploymentRequest.DeletePolicy.DELETE
|
|
106
|
+
)
|
|
107
|
+
).result(timeout=APPLY_TIMEOUT_SECONDS)
|
|
108
|
+
except GoogleAPICallError as e:
|
|
109
|
+
current = self.get(project, region, deployment_id)
|
|
110
|
+
return InfraStatus(
|
|
111
|
+
name, current.state if current else "UNKNOWN", error=current.error if current else str(e)
|
|
112
|
+
)
|
|
113
|
+
return InfraStatus(name, "DELETED")
|