duckless 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- duckless-0.1.0/PKG-INFO +86 -0
- duckless-0.1.0/README.md +68 -0
- duckless-0.1.0/duckless/__init__.py +5 -0
- duckless-0.1.0/duckless/adapters/__init__.py +1 -0
- duckless-0.1.0/duckless/adapters/batch.py +148 -0
- duckless-0.1.0/duckless/adapters/cloud_logging.py +45 -0
- duckless-0.1.0/duckless/adapters/compute_quotas.py +15 -0
- duckless-0.1.0/duckless/adapters/gcp_bootstrap.py +102 -0
- duckless-0.1.0/duckless/adapters/gcs.py +42 -0
- duckless-0.1.0/duckless/adapters/infra_manager.py +113 -0
- duckless-0.1.0/duckless/cli.py +331 -0
- duckless-0.1.0/duckless/core/__init__.py +1 -0
- duckless-0.1.0/duckless/core/errors.py +21 -0
- duckless-0.1.0/duckless/core/infra.py +128 -0
- duckless-0.1.0/duckless/core/job.py +175 -0
- duckless-0.1.0/duckless/core/machine.py +86 -0
- duckless-0.1.0/duckless/core/preflight.py +41 -0
- duckless-0.1.0/duckless/core/quota.py +59 -0
- duckless-0.1.0/duckless/ports.py +98 -0
- duckless-0.1.0/duckless/py.typed +0 -0
- duckless-0.1.0/duckless/service.py +119 -0
- duckless-0.1.0/duckless/settings.py +61 -0
- duckless-0.1.0/duckless/terraform/main.tf +109 -0
- duckless-0.1.0/duckless/terraform/outputs.tf +25 -0
- duckless-0.1.0/duckless/terraform/variables.tf +64 -0
- duckless-0.1.0/duckless/terraform/versions.tf +10 -0
- duckless-0.1.0/duckless/wiring.py +137 -0
- duckless-0.1.0/pyproject.toml +59 -0
- duckless-0.1.0/pyproject.toml.orig +48 -0
duckless-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: duckless
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Serverless DuckDB on GCP: run SQL or your own code on a right-sized VM in your project, then tear it down
|
|
5
|
+
License-Expression: Apache-2.0
|
|
6
|
+
Requires-Dist: typer>=0.15
|
|
7
|
+
Requires-Dist: google-cloud-batch>=0.17
|
|
8
|
+
Requires-Dist: google-cloud-config>=0.7
|
|
9
|
+
Requires-Dist: google-cloud-compute>=1.20
|
|
10
|
+
Requires-Dist: google-cloud-logging>=3.11
|
|
11
|
+
Requires-Dist: google-cloud-storage>=2.18
|
|
12
|
+
Requires-Python: >=3.13
|
|
13
|
+
Project-URL: Homepage, https://github.com/tosun-si/duckless
|
|
14
|
+
Project-URL: Repository, https://github.com/tosun-si/duckless
|
|
15
|
+
Project-URL: Changelog, https://github.com/tosun-si/duckless/releases
|
|
16
|
+
Project-URL: Issues, https://github.com/tosun-si/duckless/issues
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
|
|
19
|
+
# DuckLess
|
|
20
|
+
|
|
21
|
+
Serverless DuckDB on GCP. Submit SQL or your own code; DuckLess runs it on a right-sized
|
|
22
|
+
Compute Engine VM (Cloud Batch) **in your project**, reads and writes GCS through ADC
|
|
23
|
+
(no HMAC keys), spills on local SSD, then tears the VM down. Nothing runs between jobs.
|
|
24
|
+
|
|
25
|
+
> Status: v0.1.0, first release. Changes are listed in the [GitHub Releases](https://github.com/tosun-si/duckless/releases);
|
|
26
|
+
> `spike/README.md` has the measurements behind the design.
|
|
27
|
+
|
|
28
|
+
## Quick start
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
uv tool install duckless # or: pip install duckless
|
|
32
|
+
|
|
33
|
+
duckless init --project my-project --region europe-west1
|
|
34
|
+
# prints the lines to put in .envrc (DUCKLESS_PROJECT, DUCKLESS_BUCKET, DUCKLESS_SA, DUCKLESS_IMAGE)
|
|
35
|
+
|
|
36
|
+
duckless preflight --machine n2-highmem-32 --spot
|
|
37
|
+
duckless run job.sql --machine n2-highmem-32 --spot
|
|
38
|
+
duckless exec --image <your-image> --machine n2-highmem-16 -- dbt build
|
|
39
|
+
duckless status <job-id>
|
|
40
|
+
duckless logs <job-id> --follow
|
|
41
|
+
duckless result <job-id>
|
|
42
|
+
duckless destroy --project my-project # removes what init created
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
`init` applies the Terraform module shipped with the CLI (`duckless/terraform`) through
|
|
46
|
+
Infrastructure Manager: no local Terraform, state kept in your project, re-run it to upgrade.
|
|
47
|
+
Teams managing infra as code can use the same module directly instead.
|
|
48
|
+
|
|
49
|
+
## Writing jobs
|
|
50
|
+
|
|
51
|
+
- **SQL** (`.sql`): `${VAR}` placeholders come from the job env (`DUCKLESS_BUCKET`, `--env K=V`).
|
|
52
|
+
- **Python** (`.py`): `from duckless_runtime import connect` gives a DuckDB connection with GCS auth,
|
|
53
|
+
spill on local SSD, memory and threads sized to the VM.
|
|
54
|
+
- **Write to GCS in parallel**: `COPY … TO 'gs://…' (FORMAT parquet, PER_THREAD_OUTPUT, FILE_SIZE_BYTES '256MB')`
|
|
55
|
+
is 7-8x faster than the single-writer default (~900 MB/s vs ~110 MB/s on 32 vCPU).
|
|
56
|
+
- **Vectorize Python logic**: a row-wise Python UDF runs at ~8k rows/s on one thread; use an
|
|
57
|
+
Arrow UDF over numpy (`type="arrow"`) or SQL.
|
|
58
|
+
|
|
59
|
+
## Layout
|
|
60
|
+
|
|
61
|
+
Hexagonal, kept light: a pure core, ports, adapters, and one wiring point.
|
|
62
|
+
|
|
63
|
+
| Path | What |
|
|
64
|
+
| --- | --- |
|
|
65
|
+
| `duckless/core/` | Pure rules, no I/O: machine types and local SSD counts, job spec and planning, quotas, preflight |
|
|
66
|
+
| `duckless/ports.py` | What the service needs from outside: `Executor`, `ArtifactStore`, `LogReader`, `QuotaReader` (Protocols) |
|
|
67
|
+
| `duckless/service.py` | Operations shared by the CLI, the SDK and later the SaaS control plane: functions taking ports as arguments |
|
|
68
|
+
| `duckless/adapters/` | GCP implementations: Cloud Batch, GCS, Cloud Logging, Compute quotas |
|
|
69
|
+
| `duckless/wiring.py` | Binds the service functions to the adapters (lazily) |
|
|
70
|
+
| `duckless/cli.py` | `duckless` command, a driving adapter |
|
|
71
|
+
| `runtime/` | Runner image (`duckless_runtime`), published as `ghcr.io/tosun-si/duckless-runner`: DuckDB + `gcs` community extension, tuned for the VM |
|
|
72
|
+
| `duckless/terraform/` | APIs, work bucket, least-privilege runner service account, Artifact Registry remote repository proxying the runner image |
|
|
73
|
+
| `spike/` | The spike that validated the approach, kept as a record |
|
|
74
|
+
|
|
75
|
+
Dependency rule: `core` imports nothing else from DuckLess, `service` only `core` and `ports`,
|
|
76
|
+
and only `wiring` imports `adapters`.
|
|
77
|
+
|
|
78
|
+
## Development
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
uv sync
|
|
82
|
+
uv run pytest
|
|
83
|
+
uv run ruff check . && uv run ruff format --check .
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
License: Apache-2.0
|
duckless-0.1.0/README.md
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# DuckLess
|
|
2
|
+
|
|
3
|
+
Serverless DuckDB on GCP. Submit SQL or your own code; DuckLess runs it on a right-sized
|
|
4
|
+
Compute Engine VM (Cloud Batch) **in your project**, reads and writes GCS through ADC
|
|
5
|
+
(no HMAC keys), spills on local SSD, then tears the VM down. Nothing runs between jobs.
|
|
6
|
+
|
|
7
|
+
> Status: v0.1.0, first release. Changes are listed in the [GitHub Releases](https://github.com/tosun-si/duckless/releases);
|
|
8
|
+
> `spike/README.md` has the measurements behind the design.
|
|
9
|
+
|
|
10
|
+
## Quick start
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
uv tool install duckless # or: pip install duckless
|
|
14
|
+
|
|
15
|
+
duckless init --project my-project --region europe-west1
|
|
16
|
+
# prints the lines to put in .envrc (DUCKLESS_PROJECT, DUCKLESS_BUCKET, DUCKLESS_SA, DUCKLESS_IMAGE)
|
|
17
|
+
|
|
18
|
+
duckless preflight --machine n2-highmem-32 --spot
|
|
19
|
+
duckless run job.sql --machine n2-highmem-32 --spot
|
|
20
|
+
duckless exec --image <your-image> --machine n2-highmem-16 -- dbt build
|
|
21
|
+
duckless status <job-id>
|
|
22
|
+
duckless logs <job-id> --follow
|
|
23
|
+
duckless result <job-id>
|
|
24
|
+
duckless destroy --project my-project # removes what init created
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
`init` applies the Terraform module shipped with the CLI (`duckless/terraform`) through
|
|
28
|
+
Infrastructure Manager: no local Terraform, state kept in your project, re-run it to upgrade.
|
|
29
|
+
Teams managing infra as code can use the same module directly instead.
|
|
30
|
+
|
|
31
|
+
## Writing jobs
|
|
32
|
+
|
|
33
|
+
- **SQL** (`.sql`): `${VAR}` placeholders come from the job env (`DUCKLESS_BUCKET`, `--env K=V`).
|
|
34
|
+
- **Python** (`.py`): `from duckless_runtime import connect` gives a DuckDB connection with GCS auth,
|
|
35
|
+
spill on local SSD, memory and threads sized to the VM.
|
|
36
|
+
- **Write to GCS in parallel**: `COPY … TO 'gs://…' (FORMAT parquet, PER_THREAD_OUTPUT, FILE_SIZE_BYTES '256MB')`
|
|
37
|
+
is 7-8x faster than the single-writer default (~900 MB/s vs ~110 MB/s on 32 vCPU).
|
|
38
|
+
- **Vectorize Python logic**: a row-wise Python UDF runs at ~8k rows/s on one thread; use an
|
|
39
|
+
Arrow UDF over numpy (`type="arrow"`) or SQL.
|
|
40
|
+
|
|
41
|
+
## Layout
|
|
42
|
+
|
|
43
|
+
Hexagonal, kept light: a pure core, ports, adapters, and one wiring point.
|
|
44
|
+
|
|
45
|
+
| Path | What |
|
|
46
|
+
| --- | --- |
|
|
47
|
+
| `duckless/core/` | Pure rules, no I/O: machine types and local SSD counts, job spec and planning, quotas, preflight |
|
|
48
|
+
| `duckless/ports.py` | What the service needs from outside: `Executor`, `ArtifactStore`, `LogReader`, `QuotaReader` (Protocols) |
|
|
49
|
+
| `duckless/service.py` | Operations shared by the CLI, the SDK and later the SaaS control plane: functions taking ports as arguments |
|
|
50
|
+
| `duckless/adapters/` | GCP implementations: Cloud Batch, GCS, Cloud Logging, Compute quotas |
|
|
51
|
+
| `duckless/wiring.py` | Binds the service functions to the adapters (lazily) |
|
|
52
|
+
| `duckless/cli.py` | `duckless` command, a driving adapter |
|
|
53
|
+
| `runtime/` | Runner image (`duckless_runtime`), published as `ghcr.io/tosun-si/duckless-runner`: DuckDB + `gcs` community extension, tuned for the VM |
|
|
54
|
+
| `duckless/terraform/` | APIs, work bucket, least-privilege runner service account, Artifact Registry remote repository proxying the runner image |
|
|
55
|
+
| `spike/` | The spike that validated the approach, kept as a record |
|
|
56
|
+
|
|
57
|
+
Dependency rule: `core` imports nothing else from DuckLess, `service` only `core` and `ports`,
|
|
58
|
+
and only `wiring` imports `adapters`.
|
|
59
|
+
|
|
60
|
+
## Development
|
|
61
|
+
|
|
62
|
+
```bash
|
|
63
|
+
uv sync
|
|
64
|
+
uv run pytest
|
|
65
|
+
uv run ruff check . && uv run ruff format --check .
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
License: Apache-2.0
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Driven adapters: GCP implementations of duckless.ports."""
|
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
"""Executor on Cloud Batch: one task, one VM per job, deleted when the job ends."""
|
|
2
|
+
|
|
3
|
+
from google.api_core.exceptions import NotFound
|
|
4
|
+
from google.cloud import batch_v1
|
|
5
|
+
|
|
6
|
+
from duckless.core.errors import JobNotFoundError
|
|
7
|
+
from duckless.core.job import JobEvent, JobKind, JobSpec, JobState, JobStatus
|
|
8
|
+
from duckless.core.machine import LOCAL_SSD_GB
|
|
9
|
+
from duckless.settings import Settings
|
|
10
|
+
|
|
11
|
+
SCRATCH_PATH = "/mnt/disks/scratch"
|
|
12
|
+
SCRATCH_DEVICE = "scratch"
|
|
13
|
+
# Runner exit codes (job failed / bad usage): fail at once, keep retries for infra failures
|
|
14
|
+
# such as a Spot preemption (exit 50001). Batch allows a single lifecycle policy per task.
|
|
15
|
+
RUNNER_EXIT_CODES = (1, 2)
|
|
16
|
+
MAX_RETRIES = 2
|
|
17
|
+
|
|
18
|
+
_STATES = {
|
|
19
|
+
batch_v1.JobStatus.State.QUEUED: JobState.QUEUED,
|
|
20
|
+
batch_v1.JobStatus.State.SCHEDULED: JobState.SCHEDULED,
|
|
21
|
+
batch_v1.JobStatus.State.RUNNING: JobState.RUNNING,
|
|
22
|
+
batch_v1.JobStatus.State.SUCCEEDED: JobState.SUCCEEDED,
|
|
23
|
+
batch_v1.JobStatus.State.FAILED: JobState.FAILED,
|
|
24
|
+
batch_v1.JobStatus.State.CANCELLED: JobState.CANCELLED,
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# ---------- pure ----------
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def container_invocation(spec: JobSpec) -> tuple[str, list[str]]:
|
|
32
|
+
"""(entrypoint, args). Runner jobs keep the image entrypoint (`python -m duckless_runtime`) and
|
|
33
|
+
pass `sql|py <uri>`; a command job replaces the entrypoint, so `-- dbt build` runs dbt itself."""
|
|
34
|
+
if spec.kind is JobKind.COMMAND:
|
|
35
|
+
return spec.command[0], list(spec.command[1:])
|
|
36
|
+
return "", list(spec.runner_args)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def build_job(spec: JobSpec, settings: Settings) -> batch_v1.Job:
|
|
40
|
+
has_scratch = spec.local_ssd_count > 0
|
|
41
|
+
entrypoint, args = container_invocation(spec)
|
|
42
|
+
# Local SSD is mounted by Batch on the host as root; open it to the non-root runner user.
|
|
43
|
+
prepare_scratch = batch_v1.Runnable(
|
|
44
|
+
script=batch_v1.Runnable.Script(text=f"mkdir -p {SCRATCH_PATH} && chmod 1777 {SCRATCH_PATH}")
|
|
45
|
+
)
|
|
46
|
+
runner = batch_v1.Runnable(
|
|
47
|
+
# Task volumes are bind-mounted into the container at the same path by default.
|
|
48
|
+
container=batch_v1.Runnable.Container(image_uri=spec.image, entrypoint=entrypoint, commands=args),
|
|
49
|
+
environment=batch_v1.Environment(variables={**dict(spec.env), "GOOGLE_CLOUD_PROJECT": settings.project}),
|
|
50
|
+
)
|
|
51
|
+
task = batch_v1.TaskSpec(
|
|
52
|
+
runnables=[prepare_scratch, runner] if has_scratch else [runner],
|
|
53
|
+
volumes=[batch_v1.Volume(device_name=SCRATCH_DEVICE, mount_path=SCRATCH_PATH)] if has_scratch else [],
|
|
54
|
+
max_run_duration=f"{spec.max_run_seconds}s",
|
|
55
|
+
max_retry_count=MAX_RETRIES,
|
|
56
|
+
lifecycle_policies=[
|
|
57
|
+
batch_v1.LifecyclePolicy(
|
|
58
|
+
action=batch_v1.LifecyclePolicy.Action.FAIL_TASK,
|
|
59
|
+
action_condition=batch_v1.LifecyclePolicy.ActionCondition(exit_codes=list(RUNNER_EXIT_CODES)),
|
|
60
|
+
)
|
|
61
|
+
],
|
|
62
|
+
)
|
|
63
|
+
policy = batch_v1.AllocationPolicy.InstancePolicy(
|
|
64
|
+
machine_type=spec.machine.name,
|
|
65
|
+
provisioning_model=(
|
|
66
|
+
batch_v1.AllocationPolicy.ProvisioningModel.SPOT
|
|
67
|
+
if spec.spot
|
|
68
|
+
else batch_v1.AllocationPolicy.ProvisioningModel.STANDARD
|
|
69
|
+
),
|
|
70
|
+
disks=[
|
|
71
|
+
batch_v1.AllocationPolicy.AttachedDisk(
|
|
72
|
+
new_disk=batch_v1.AllocationPolicy.Disk(type_="local-ssd", size_gb=LOCAL_SSD_GB * spec.local_ssd_count),
|
|
73
|
+
device_name=SCRATCH_DEVICE,
|
|
74
|
+
)
|
|
75
|
+
]
|
|
76
|
+
if has_scratch
|
|
77
|
+
else [],
|
|
78
|
+
)
|
|
79
|
+
allocation = batch_v1.AllocationPolicy(
|
|
80
|
+
# Any zone of the region: Spot + highmem + local SSD stocks out zone by zone.
|
|
81
|
+
location=batch_v1.AllocationPolicy.LocationPolicy(allowed_locations=[f"regions/{settings.region}"]),
|
|
82
|
+
instances=[batch_v1.AllocationPolicy.InstancePolicyOrTemplate(policy=policy)],
|
|
83
|
+
service_account=batch_v1.ServiceAccount(email=settings.service_account),
|
|
84
|
+
network=batch_v1.AllocationPolicy.NetworkPolicy(
|
|
85
|
+
network_interfaces=[
|
|
86
|
+
batch_v1.AllocationPolicy.NetworkInterface(
|
|
87
|
+
network=settings.network,
|
|
88
|
+
subnetwork=settings.subnetwork,
|
|
89
|
+
no_external_ip_address=not settings.external_ip,
|
|
90
|
+
)
|
|
91
|
+
]
|
|
92
|
+
),
|
|
93
|
+
)
|
|
94
|
+
return batch_v1.Job(
|
|
95
|
+
task_groups=[batch_v1.TaskGroup(task_spec=task, task_count=1)],
|
|
96
|
+
allocation_policy=allocation,
|
|
97
|
+
logs_policy=batch_v1.LogsPolicy(destination=batch_v1.LogsPolicy.Destination.CLOUD_LOGGING),
|
|
98
|
+
labels={"app": "duckless", "duckless-kind": spec.kind.value},
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def status_from_job(job: batch_v1.Job) -> JobStatus:
|
|
103
|
+
policy = job.allocation_policy.instances[0].policy
|
|
104
|
+
return JobStatus(
|
|
105
|
+
job_id=job.name.rsplit("/", 1)[-1],
|
|
106
|
+
uid=job.uid,
|
|
107
|
+
state=_STATES.get(job.status.state, JobState.UNKNOWN),
|
|
108
|
+
machine=policy.machine_type,
|
|
109
|
+
spot=policy.provisioning_model == batch_v1.AllocationPolicy.ProvisioningModel.SPOT,
|
|
110
|
+
created_at=job.create_time,
|
|
111
|
+
run_seconds=job.status.run_duration.total_seconds() if job.status.run_duration else None,
|
|
112
|
+
events=tuple(
|
|
113
|
+
JobEvent(at=e.event_time, description=e.description.split(" for job ")[0]) for e in job.status.status_events
|
|
114
|
+
),
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
# ---------- adapter ----------
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class BatchExecutor:
|
|
122
|
+
def __init__(self, client: batch_v1.BatchServiceClient, settings: Settings) -> None:
|
|
123
|
+
self._client = client
|
|
124
|
+
self._settings = settings
|
|
125
|
+
|
|
126
|
+
def _parent(self) -> str:
|
|
127
|
+
return f"projects/{self._settings.project}/locations/{self._settings.region}"
|
|
128
|
+
|
|
129
|
+
def _name(self, job_id: str) -> str:
|
|
130
|
+
return f"{self._parent()}/jobs/{job_id}"
|
|
131
|
+
|
|
132
|
+
def submit(self, spec: JobSpec) -> JobStatus:
|
|
133
|
+
created = self._client.create_job(
|
|
134
|
+
parent=self._parent(), job_id=spec.job_id, job=build_job(spec, self._settings)
|
|
135
|
+
)
|
|
136
|
+
return status_from_job(created)
|
|
137
|
+
|
|
138
|
+
def get(self, job_id: str) -> JobStatus:
|
|
139
|
+
try:
|
|
140
|
+
return status_from_job(self._client.get_job(name=self._name(job_id)))
|
|
141
|
+
except NotFound as e:
|
|
142
|
+
raise JobNotFoundError(job_id) from e
|
|
143
|
+
|
|
144
|
+
def cancel(self, job_id: str) -> None:
|
|
145
|
+
try:
|
|
146
|
+
self._client.cancel_job(name=self._name(job_id))
|
|
147
|
+
except NotFound as e:
|
|
148
|
+
raise JobNotFoundError(job_id) from e
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Runner logs from Cloud Logging: Batch writes task stdout to `batch_task_logs`, labelled by job uid."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
|
|
5
|
+
from google.cloud import logging as cloud_logging
|
|
6
|
+
|
|
7
|
+
from duckless.core.job import LogLine
|
|
8
|
+
|
|
9
|
+
TASK_LOG = "batch_task_logs"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def log_filter(project: str, job_uid: str, since: datetime | None) -> str:
|
|
13
|
+
clauses = (
|
|
14
|
+
f'logName="projects/{project}/logs/{TASK_LOG}"',
|
|
15
|
+
f'labels.job_uid="{job_uid}"',
|
|
16
|
+
*((f'timestamp>"{since.isoformat()}"',) if since else ()),
|
|
17
|
+
)
|
|
18
|
+
return " AND ".join(clauses)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def to_line(entry: cloud_logging.LogEntry) -> LogLine:
|
|
22
|
+
"""The runner logs JSON lines (jsonPayload); anything else (image pull, crash) is plain text."""
|
|
23
|
+
payload = entry.payload if isinstance(entry.payload, dict) else {"message": str(entry.payload)}
|
|
24
|
+
fields = {k: v for k, v in payload.items() if k not in {"message", "severity", "time"}}
|
|
25
|
+
return LogLine(
|
|
26
|
+
at=entry.timestamp,
|
|
27
|
+
severity=entry.severity or "DEFAULT",
|
|
28
|
+
message=str(payload.get("message", "")),
|
|
29
|
+
fields=fields,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class CloudLoggingLogReader:
|
|
34
|
+
def __init__(self, client: cloud_logging.Client, project: str) -> None:
|
|
35
|
+
self._client = client
|
|
36
|
+
self._project = project
|
|
37
|
+
|
|
38
|
+
def read(self, job_uid: str, since: datetime | None = None, limit: int = 200) -> tuple[LogLine, ...]:
|
|
39
|
+
entries = self._client.list_entries(
|
|
40
|
+
resource_names=[f"projects/{self._project}"],
|
|
41
|
+
filter_=log_filter(self._project, job_uid, since),
|
|
42
|
+
order_by=cloud_logging.ASCENDING,
|
|
43
|
+
max_results=limit,
|
|
44
|
+
)
|
|
45
|
+
return tuple(map(to_line, entries))
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from collections.abc import Mapping
|
|
2
|
+
|
|
3
|
+
from google.cloud import compute_v1
|
|
4
|
+
|
|
5
|
+
from duckless.core.quota import Quota
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class ComputeQuotaReader:
|
|
9
|
+
def __init__(self, client: compute_v1.RegionsClient, project: str) -> None:
|
|
10
|
+
self._client = client
|
|
11
|
+
self._project = project
|
|
12
|
+
|
|
13
|
+
def regional_quotas(self, region: str) -> Mapping[str, Quota]:
|
|
14
|
+
regional = self._client.get(project=self._project, region=region)
|
|
15
|
+
return {q.metric: Quota(usage=q.usage, limit=q.limit) for q in regional.quotas}
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""InfraBootstrap with the caller's credentials: plain REST for the few one-off calls, the
|
|
2
|
+
storage client for the staging bucket."""
|
|
3
|
+
|
|
4
|
+
import time
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from google.auth.transport.requests import AuthorizedSession
|
|
8
|
+
from google.cloud import storage
|
|
9
|
+
|
|
10
|
+
from duckless.core.infra import with_bindings, without_bindings
|
|
11
|
+
|
|
12
|
+
SERVICE_USAGE = "https://serviceusage.googleapis.com/v1"
|
|
13
|
+
IAM = "https://iam.googleapis.com/v1"
|
|
14
|
+
RESOURCE_MANAGER = "https://cloudresourcemanager.googleapis.com/v3"
|
|
15
|
+
SKIPPED_PARTS = frozenset({".terraform", ".terraform.lock.hcl", "__pycache__"})
|
|
16
|
+
OPERATION_POLL_SECONDS = 3
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def module_files(local_dir: Path) -> list[Path]:
|
|
20
|
+
"""Files of the Terraform module, without local Terraform state or caches."""
|
|
21
|
+
return sorted(
|
|
22
|
+
p
|
|
23
|
+
for p in local_dir.rglob("*")
|
|
24
|
+
if p.is_file() and not SKIPPED_PARTS.intersection(p.relative_to(local_dir).parts)
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
class GcpInfraBootstrap:
|
|
29
|
+
def __init__(self, session: AuthorizedSession, storage_client: storage.Client) -> None:
|
|
30
|
+
self._session = session
|
|
31
|
+
self._storage = storage_client
|
|
32
|
+
|
|
33
|
+
def _call(self, method: str, url: str, **kwargs) -> dict:
|
|
34
|
+
response = self._session.request(method, url, **kwargs)
|
|
35
|
+
response.raise_for_status()
|
|
36
|
+
return response.json() if response.content else {}
|
|
37
|
+
|
|
38
|
+
def _wait(self, base: str, operation: dict) -> None:
|
|
39
|
+
while not operation.get("done"):
|
|
40
|
+
time.sleep(OPERATION_POLL_SECONDS)
|
|
41
|
+
operation = self._call("GET", f"{base}/{operation['name']}")
|
|
42
|
+
if "error" in operation:
|
|
43
|
+
raise RuntimeError(f"operation {operation['name']} failed: {operation['error']}")
|
|
44
|
+
|
|
45
|
+
def enable_apis(self, project: str, apis: tuple[str, ...]) -> None:
|
|
46
|
+
operation = self._call(
|
|
47
|
+
"POST", f"{SERVICE_USAGE}/projects/{project}/services:batchEnable", json={"serviceIds": list(apis)}
|
|
48
|
+
)
|
|
49
|
+
self._wait(SERVICE_USAGE, operation)
|
|
50
|
+
|
|
51
|
+
def ensure_service_account(self, project: str, account_id: str, display_name: str) -> str:
|
|
52
|
+
email = f"{account_id}@{project}.iam.gserviceaccount.com"
|
|
53
|
+
existing = self._session.get(f"{IAM}/projects/{project}/serviceAccounts/{email}")
|
|
54
|
+
if existing.status_code == 404:
|
|
55
|
+
self._call(
|
|
56
|
+
"POST",
|
|
57
|
+
f"{IAM}/projects/{project}/serviceAccounts",
|
|
58
|
+
json={"accountId": account_id, "serviceAccount": {"displayName": display_name}},
|
|
59
|
+
)
|
|
60
|
+
else:
|
|
61
|
+
existing.raise_for_status()
|
|
62
|
+
return email
|
|
63
|
+
|
|
64
|
+
def _update_policy(self, project: str, change) -> bool:
|
|
65
|
+
resource = f"{RESOURCE_MANAGER}/projects/{project}"
|
|
66
|
+
policy = self._call("POST", f"{resource}:getIamPolicy", json={"options": {"requestedPolicyVersion": 3}})
|
|
67
|
+
updated, changed = change(policy)
|
|
68
|
+
if changed:
|
|
69
|
+
# The etag in `updated` makes a concurrent change fail instead of being overwritten.
|
|
70
|
+
self._call("POST", f"{resource}:setIamPolicy", json={"policy": updated})
|
|
71
|
+
return changed
|
|
72
|
+
|
|
73
|
+
def grant_project_roles(self, project: str, member: str, roles: tuple[str, ...]) -> bool:
|
|
74
|
+
return self._update_policy(project, lambda policy: with_bindings(policy, member, roles))
|
|
75
|
+
|
|
76
|
+
def revoke_project_roles(self, project: str, member: str, roles: tuple[str, ...]) -> None:
|
|
77
|
+
self._update_policy(project, lambda policy: without_bindings(policy, member, roles))
|
|
78
|
+
|
|
79
|
+
def delete_service_account(self, project: str, email: str) -> None:
|
|
80
|
+
response = self._session.delete(f"{IAM}/projects/{project}/serviceAccounts/{email}")
|
|
81
|
+
if response.status_code != 404:
|
|
82
|
+
response.raise_for_status()
|
|
83
|
+
|
|
84
|
+
def delete_bucket(self, bucket: str) -> None:
|
|
85
|
+
existing = self._storage.lookup_bucket(bucket)
|
|
86
|
+
if existing is not None:
|
|
87
|
+
existing.delete(force=True)
|
|
88
|
+
|
|
89
|
+
def ensure_bucket(self, project: str, region: str, bucket: str) -> None:
|
|
90
|
+
if self._storage.lookup_bucket(bucket) is not None:
|
|
91
|
+
return
|
|
92
|
+
new = self._storage.bucket(bucket)
|
|
93
|
+
new.iam_configuration.uniform_bucket_level_access_enabled = True
|
|
94
|
+
new.iam_configuration.public_access_prevention = "enforced"
|
|
95
|
+
new.labels = {"app": "duckless"}
|
|
96
|
+
self._storage.create_bucket(new, project=project, location=region)
|
|
97
|
+
|
|
98
|
+
def upload_directory(self, local_dir: Path, bucket: str, prefix: str) -> str:
|
|
99
|
+
target = self._storage.bucket(bucket)
|
|
100
|
+
for path in module_files(local_dir):
|
|
101
|
+
target.blob(f"{prefix}/{path.relative_to(local_dir).as_posix()}").upload_from_filename(str(path))
|
|
102
|
+
return f"gs://{bucket}/{prefix}"
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Work bucket layout: gs://<bucket>/runs/<job_id>/{<source>, metrics.json}."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from google.cloud import storage
|
|
9
|
+
|
|
10
|
+
RUNS_PREFIX = "runs"
|
|
11
|
+
METRICS_FILE = "metrics.json"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def run_object(job_id: str, name: str) -> str:
|
|
15
|
+
return f"{RUNS_PREFIX}/{job_id}/{name}"
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def parse_metrics(raw: str) -> Mapping[str, Any] | None:
|
|
19
|
+
"""The runner writes metrics with DuckDB's COPY … (FORMAT json): one object per line."""
|
|
20
|
+
first = next((line for line in raw.splitlines() if line.strip()), None)
|
|
21
|
+
return json.loads(first) if first else None
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class GcsArtifactStore:
|
|
25
|
+
def __init__(self, client: storage.Client, bucket: str) -> None:
|
|
26
|
+
self._bucket = client.bucket(bucket)
|
|
27
|
+
|
|
28
|
+
def upload_source(self, job_id: str, source: Path) -> str:
|
|
29
|
+
blob = self._bucket.blob(run_object(job_id, source.name))
|
|
30
|
+
blob.upload_from_filename(str(source))
|
|
31
|
+
return f"gs://{self._bucket.name}/{blob.name}"
|
|
32
|
+
|
|
33
|
+
def runner_env(self, job_id: str) -> Mapping[str, str]:
|
|
34
|
+
return {
|
|
35
|
+
"DUCKLESS_JOB_ID": job_id,
|
|
36
|
+
"DUCKLESS_BUCKET": self._bucket.name,
|
|
37
|
+
"DUCKLESS_METRICS_URI": f"gs://{self._bucket.name}/{run_object(job_id, METRICS_FILE)}",
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
def read_metrics(self, job_id: str) -> Mapping[str, Any] | None:
|
|
41
|
+
blob = self._bucket.blob(run_object(job_id, METRICS_FILE))
|
|
42
|
+
return parse_metrics(blob.download_as_text()) if blob.exists() else None
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""InfraDeployer on Infrastructure Manager (Terraform run by Google, state kept in the project)."""
|
|
2
|
+
|
|
3
|
+
import contextlib
|
|
4
|
+
from collections.abc import Mapping
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from google.api_core.exceptions import GoogleAPICallError, NotFound
|
|
8
|
+
from google.cloud import config_v1
|
|
9
|
+
|
|
10
|
+
from duckless.core.infra import InfraStatus
|
|
11
|
+
|
|
12
|
+
APPLY_TIMEOUT_SECONDS = 30 * 60
|
|
13
|
+
LABELS = {"app": "duckless"}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def deployment_name(project: str, region: str, deployment_id: str) -> str:
|
|
17
|
+
return f"projects/{project}/locations/{region}/deployments/{deployment_id}"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def build_deployment(
|
|
21
|
+
name: str, source_uri: str, inputs: Mapping[str, Any], service_account: str
|
|
22
|
+
) -> config_v1.Deployment:
|
|
23
|
+
return config_v1.Deployment(
|
|
24
|
+
name=name,
|
|
25
|
+
service_account=service_account
|
|
26
|
+
if service_account.startswith("projects/")
|
|
27
|
+
else f"projects/{name.split('/')[1]}/serviceAccounts/{service_account}",
|
|
28
|
+
terraform_blueprint=config_v1.TerraformBlueprint(
|
|
29
|
+
gcs_source=source_uri,
|
|
30
|
+
input_values={k: config_v1.TerraformVariable(input_value=v) for k, v in inputs.items()},
|
|
31
|
+
),
|
|
32
|
+
labels=LABELS,
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def failure_detail(deployment: config_v1.Deployment, revision: config_v1.Revision | None) -> str:
|
|
37
|
+
"""The most useful error Infra Manager kept: Terraform errors first, then the state detail."""
|
|
38
|
+
tf_errors = [e.error_description or e.resource_address for e in (revision.tf_errors if revision else ())]
|
|
39
|
+
tf_errors += [e.error_description or e.resource_address for e in deployment.tf_errors]
|
|
40
|
+
logs = revision.logs if revision and revision.logs else deployment.error_logs
|
|
41
|
+
parts = [
|
|
42
|
+
*tf_errors[:5],
|
|
43
|
+
deployment.state_detail or (revision.state_detail if revision else ""),
|
|
44
|
+
logs and f"logs: {logs}",
|
|
45
|
+
]
|
|
46
|
+
return "\n".join(p for p in parts if p) or deployment.state.name
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class InfraManagerDeployer:
|
|
50
|
+
def __init__(self, client: config_v1.ConfigClient) -> None:
|
|
51
|
+
self._client = client
|
|
52
|
+
|
|
53
|
+
def _revision(self, deployment: config_v1.Deployment) -> config_v1.Revision | None:
|
|
54
|
+
return self._client.get_revision(name=deployment.latest_revision) if deployment.latest_revision else None
|
|
55
|
+
|
|
56
|
+
def _status(self, deployment: config_v1.Deployment) -> InfraStatus:
|
|
57
|
+
revision = self._revision(deployment)
|
|
58
|
+
outputs = {k: o.value for k, o in revision.apply_results.outputs.items()} if revision else {}
|
|
59
|
+
failed = deployment.state == config_v1.Deployment.State.FAILED
|
|
60
|
+
return InfraStatus(
|
|
61
|
+
deployment=deployment.name,
|
|
62
|
+
state=deployment.state.name,
|
|
63
|
+
outputs=outputs,
|
|
64
|
+
error=failure_detail(deployment, revision) if failed else None,
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
def get(self, project: str, region: str, deployment_id: str) -> InfraStatus | None:
|
|
68
|
+
try:
|
|
69
|
+
return self._status(self._client.get_deployment(name=deployment_name(project, region, deployment_id)))
|
|
70
|
+
except NotFound:
|
|
71
|
+
return None
|
|
72
|
+
|
|
73
|
+
def apply(
|
|
74
|
+
self,
|
|
75
|
+
project: str,
|
|
76
|
+
region: str,
|
|
77
|
+
deployment_id: str,
|
|
78
|
+
source_uri: str,
|
|
79
|
+
inputs: Mapping[str, Any],
|
|
80
|
+
service_account: str,
|
|
81
|
+
) -> InfraStatus:
|
|
82
|
+
name = deployment_name(project, region, deployment_id)
|
|
83
|
+
deployment = build_deployment(name, source_uri, inputs, service_account)
|
|
84
|
+
exists = self.get(project, region, deployment_id) is not None
|
|
85
|
+
operation = (
|
|
86
|
+
self._client.update_deployment(deployment=deployment)
|
|
87
|
+
if exists
|
|
88
|
+
else self._client.create_deployment(
|
|
89
|
+
parent=f"projects/{project}/locations/{region}", deployment_id=deployment_id, deployment=deployment
|
|
90
|
+
)
|
|
91
|
+
)
|
|
92
|
+
# On failure the deployment carries the details, read below.
|
|
93
|
+
with contextlib.suppress(GoogleAPICallError):
|
|
94
|
+
operation.result(timeout=APPLY_TIMEOUT_SECONDS)
|
|
95
|
+
return self._status(self._client.get_deployment(name=name))
|
|
96
|
+
|
|
97
|
+
def destroy(self, project: str, region: str, deployment_id: str) -> InfraStatus:
|
|
98
|
+
name = deployment_name(project, region, deployment_id)
|
|
99
|
+
if self.get(project, region, deployment_id) is None:
|
|
100
|
+
return InfraStatus(name, "DELETED")
|
|
101
|
+
try:
|
|
102
|
+
# force: also delete the deployment's revisions (nested resources), else the call is refused.
|
|
103
|
+
self._client.delete_deployment(
|
|
104
|
+
request=config_v1.DeleteDeploymentRequest(
|
|
105
|
+
name=name, force=True, delete_policy=config_v1.DeleteDeploymentRequest.DeletePolicy.DELETE
|
|
106
|
+
)
|
|
107
|
+
).result(timeout=APPLY_TIMEOUT_SECONDS)
|
|
108
|
+
except GoogleAPICallError as e:
|
|
109
|
+
current = self.get(project, region, deployment_id)
|
|
110
|
+
return InfraStatus(
|
|
111
|
+
name, current.state if current else "UNKNOWN", error=current.error if current else str(e)
|
|
112
|
+
)
|
|
113
|
+
return InfraStatus(name, "DELETED")
|