lamia-cloud 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lamia_cloud/__init__.py +59 -0
- lamia_cloud/_version.py +24 -0
- lamia_cloud/gcp/__init__.py +10 -0
- lamia_cloud/gcp/deployer.py +366 -0
- lamia_cloud/gcp/scheduler.py +213 -0
- lamia_cloud/gcp/vertex.py +451 -0
- lamia_cloud/interfaces.py +74 -0
- lamia_cloud/loader.py +53 -0
- lamia_cloud/templates/Dockerfile +13 -0
- lamia_cloud/templates/main.py +55 -0
- lamia_cloud/types.py +50 -0
- lamia_cloud-0.1.0.dist-info/METADATA +155 -0
- lamia_cloud-0.1.0.dist-info/RECORD +17 -0
- lamia_cloud-0.1.0.dist-info/WHEEL +5 -0
- lamia_cloud-0.1.0.dist-info/entry_points.txt +2 -0
- lamia_cloud-0.1.0.dist-info/licenses/LICENSE +21 -0
- lamia_cloud-0.1.0.dist-info/top_level.txt +1 -0
lamia_cloud/__init__.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""lamia-cloud — lamia cloud services package.
|
|
2
|
+
|
|
3
|
+
Public API:
|
|
4
|
+
get_cloud_llm() -> CloudLLM
|
|
5
|
+
get_scheduler(project_root) -> CloudScheduler
|
|
6
|
+
is_on_cloud() -> bool
|
|
7
|
+
Types: CloudLLMRequest, CloudLLMResponse, CloudScheduleJob, CloudJobStatus
|
|
8
|
+
"""
|
|
9
|
+
from lamia_cloud.interfaces import CloudLLM, CloudScheduler
|
|
10
|
+
from lamia_cloud.types import CloudLLMRequest, CloudLLMResponse, CloudScheduleJob, CloudJobStatus
|
|
11
|
+
from lamia_cloud.gcp import VertexLLM, is_on_gcp
|
|
12
|
+
from lamia_cloud.loader import get_scheduler
|
|
13
|
+
|
|
14
|
+
_llm_instance: CloudLLM = None
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def get_cloud_llm(region: str = "") -> CloudLLM:
|
|
18
|
+
"""Return the cloud LLM instance.
|
|
19
|
+
|
|
20
|
+
Only GCP is supported, so we return VertexLLM directly without
|
|
21
|
+
conditional provider checking.
|
|
22
|
+
"""
|
|
23
|
+
global _llm_instance
|
|
24
|
+
if _llm_instance is None:
|
|
25
|
+
_llm_instance = VertexLLM(region=region or _detect_region())
|
|
26
|
+
return _llm_instance
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _detect_region() -> str:
|
|
30
|
+
"""Auto-detect GCP region from Cloud Run metadata server."""
|
|
31
|
+
import urllib.request
|
|
32
|
+
try:
|
|
33
|
+
req = urllib.request.Request(
|
|
34
|
+
"http://metadata.google.internal/computeMetadata/v1/instance/region",
|
|
35
|
+
headers={"Metadata-Flavor": "Google"},
|
|
36
|
+
)
|
|
37
|
+
resp = urllib.request.urlopen(req, timeout=2)
|
|
38
|
+
full = resp.read().decode().strip()
|
|
39
|
+
return full.rsplit("/", 1)[-1]
|
|
40
|
+
except Exception:
|
|
41
|
+
return "us-central1"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def is_on_cloud() -> bool:
|
|
45
|
+
"""Check if running in a cloud environment."""
|
|
46
|
+
return is_on_gcp()
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
__all__ = [
|
|
50
|
+
"CloudLLM",
|
|
51
|
+
"CloudScheduler",
|
|
52
|
+
"CloudLLMRequest",
|
|
53
|
+
"CloudLLMResponse",
|
|
54
|
+
"CloudScheduleJob",
|
|
55
|
+
"CloudJobStatus",
|
|
56
|
+
"get_cloud_llm",
|
|
57
|
+
"get_scheduler",
|
|
58
|
+
"is_on_cloud",
|
|
59
|
+
]
|
lamia_cloud/_version.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# file generated by vcs-versioning
|
|
2
|
+
# don't change, don't track in version control
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"__version__",
|
|
7
|
+
"__version_tuple__",
|
|
8
|
+
"version",
|
|
9
|
+
"version_tuple",
|
|
10
|
+
"__commit_id__",
|
|
11
|
+
"commit_id",
|
|
12
|
+
]
|
|
13
|
+
|
|
14
|
+
version: str
|
|
15
|
+
__version__: str
|
|
16
|
+
__version_tuple__: tuple[int | str, ...]
|
|
17
|
+
version_tuple: tuple[int | str, ...]
|
|
18
|
+
commit_id: str | None
|
|
19
|
+
__commit_id__: str | None
|
|
20
|
+
|
|
21
|
+
__version__ = version = '0.1.0'
|
|
22
|
+
__version_tuple__ = version_tuple = (0, 1, 0)
|
|
23
|
+
|
|
24
|
+
__commit_id__ = commit_id = None
|
|
@@ -0,0 +1,366 @@
|
|
|
1
|
+
"""Deploys a lamia project to Cloud Run via Cloud Build.
|
|
2
|
+
|
|
3
|
+
Flow:
|
|
4
|
+
1. Package the .lm script + project files into a staging directory
|
|
5
|
+
2. Add Dockerfile + main.py handler + requirements.txt
|
|
6
|
+
3. Upload to GCS as source tarball
|
|
7
|
+
4. Submit Cloud Build to build the container
|
|
8
|
+
5. Deploy the container to Cloud Run (with Vertex AI IAM for LLM access)
|
|
9
|
+
6. Return the Cloud Run service URL
|
|
10
|
+
|
|
11
|
+
LLM authentication uses Vertex AI — the Cloud Run service account gets
|
|
12
|
+
roles/aiplatform.user, so no API keys are needed at runtime.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import io
|
|
16
|
+
import logging
|
|
17
|
+
import shutil
|
|
18
|
+
import tarfile
|
|
19
|
+
import tempfile
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
logger = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
# templates/ lives at the lamia_cloud package root, one level above this gcp/ subpackage
|
|
25
|
+
TEMPLATES_DIR = Path(__file__).parent.parent / "templates"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _service_name(schedule_id: str) -> str:
|
|
29
|
+
return f"lamia-{schedule_id}"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _image_name(project_id: str, schedule_id: str) -> str:
|
|
33
|
+
import time
|
|
34
|
+
ts = int(time.time())
|
|
35
|
+
return f"gcr.io/{project_id}/lamia-{schedule_id}:{ts}"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _collect_project_files(project_root: Path) -> list[Path]:
|
|
39
|
+
"""Collect .lm files, config.yaml, and supporting Python files from the project.
|
|
40
|
+
|
|
41
|
+
SECURITY: .env files are explicitly excluded — secrets must never be baked
|
|
42
|
+
into Docker image layers. API keys are injected via Cloud Run env vars at
|
|
43
|
+
deploy time (see deploy_cloud_run).
|
|
44
|
+
"""
|
|
45
|
+
files = []
|
|
46
|
+
for pattern in ("*.lm", "*.py", "*.yaml", "*.yml", "*.json", "*.txt", "*.csv"):
|
|
47
|
+
files.extend(project_root.glob(pattern))
|
|
48
|
+
files = [f for f in files if f.name != ".env"]
|
|
49
|
+
for subdir in project_root.iterdir():
|
|
50
|
+
if subdir.is_dir() and not subdir.name.startswith("."):
|
|
51
|
+
for pattern in ("**/*.lm", "**/*.py", "**/*.yaml", "**/*.json"):
|
|
52
|
+
files.extend(subdir.glob(pattern))
|
|
53
|
+
return files
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def package_deployment(
|
|
57
|
+
project_root: Path,
|
|
58
|
+
script_name: str,
|
|
59
|
+
schedule_id: str,
|
|
60
|
+
) -> Path:
|
|
61
|
+
"""Create a staging directory with everything needed for the Cloud Build."""
|
|
62
|
+
staging = Path(tempfile.mkdtemp(prefix="lamia-deploy-"))
|
|
63
|
+
|
|
64
|
+
project_dest = staging / "project"
|
|
65
|
+
project_dest.mkdir()
|
|
66
|
+
for f in _collect_project_files(project_root):
|
|
67
|
+
rel = f.relative_to(project_root)
|
|
68
|
+
dest = project_dest / rel
|
|
69
|
+
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
70
|
+
shutil.copy2(f, dest)
|
|
71
|
+
|
|
72
|
+
shutil.copy2(TEMPLATES_DIR / "Dockerfile", staging / "Dockerfile")
|
|
73
|
+
shutil.copy2(TEMPLATES_DIR / "main.py", staging / "main.py")
|
|
74
|
+
|
|
75
|
+
requirements = staging / "requirements.txt"
|
|
76
|
+
project_requirements = project_root / "requirements.txt"
|
|
77
|
+
if project_requirements.exists():
|
|
78
|
+
reqs = project_requirements.read_text()
|
|
79
|
+
else:
|
|
80
|
+
reqs = ""
|
|
81
|
+
if "lamia-lang" not in reqs:
|
|
82
|
+
reqs = "lamia-lang\n" + reqs
|
|
83
|
+
if "google-auth" not in reqs:
|
|
84
|
+
reqs += "google-auth\n"
|
|
85
|
+
requirements.write_text(reqs)
|
|
86
|
+
|
|
87
|
+
return staging
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def create_source_tarball(staging_dir: Path) -> bytes:
|
|
91
|
+
"""Create a gzipped tarball from the staging directory."""
|
|
92
|
+
buf = io.BytesIO()
|
|
93
|
+
with tarfile.open(fileobj=buf, mode="w:gz") as tar:
|
|
94
|
+
for item in staging_dir.iterdir():
|
|
95
|
+
tar.add(item, arcname=item.name)
|
|
96
|
+
return buf.getvalue()
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def upload_source(project_id: str, tarball: bytes, schedule_id: str) -> str:
|
|
100
|
+
"""Upload source tarball to GCS and return the gs:// URI."""
|
|
101
|
+
from google.cloud import storage
|
|
102
|
+
|
|
103
|
+
bucket_name = f"{project_id}_cloudbuild"
|
|
104
|
+
blob_name = f"lamia-source/{schedule_id}.tar.gz"
|
|
105
|
+
|
|
106
|
+
client = storage.Client(project=project_id)
|
|
107
|
+
bucket = client.bucket(bucket_name)
|
|
108
|
+
if not bucket.exists():
|
|
109
|
+
bucket = client.create_bucket(bucket_name, location="us")
|
|
110
|
+
|
|
111
|
+
blob = bucket.blob(blob_name)
|
|
112
|
+
blob.upload_from_string(tarball, content_type="application/gzip")
|
|
113
|
+
|
|
114
|
+
return f"gs://{bucket_name}/{blob_name}"
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def submit_build(
|
|
118
|
+
project_id: str,
|
|
119
|
+
source_uri: str,
|
|
120
|
+
image_name: str,
|
|
121
|
+
) -> None:
|
|
122
|
+
"""Submit a Cloud Build to build the container image."""
|
|
123
|
+
from google.cloud.devtools import cloudbuild_v1
|
|
124
|
+
|
|
125
|
+
client = cloudbuild_v1.CloudBuildClient()
|
|
126
|
+
|
|
127
|
+
build = cloudbuild_v1.Build(
|
|
128
|
+
source=cloudbuild_v1.Source(
|
|
129
|
+
storage_source=cloudbuild_v1.StorageSource(
|
|
130
|
+
bucket=source_uri.split("/")[2],
|
|
131
|
+
object_="/".join(source_uri.split("/")[3:]),
|
|
132
|
+
)
|
|
133
|
+
),
|
|
134
|
+
steps=[
|
|
135
|
+
cloudbuild_v1.BuildStep(
|
|
136
|
+
name="gcr.io/cloud-builders/docker",
|
|
137
|
+
args=["build", "-t", image_name, "."],
|
|
138
|
+
)
|
|
139
|
+
],
|
|
140
|
+
images=[image_name],
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
operation = client.create_build(project_id=project_id, build=build)
|
|
144
|
+
logger.info(f"Cloud Build submitted, waiting for completion...")
|
|
145
|
+
result = operation.result(timeout=600)
|
|
146
|
+
|
|
147
|
+
if result.status != cloudbuild_v1.Build.Status.SUCCESS:
|
|
148
|
+
raise RuntimeError(
|
|
149
|
+
f"Cloud Build failed with status {result.status.name}: "
|
|
150
|
+
f"{result.status_detail}"
|
|
151
|
+
)
|
|
152
|
+
logger.info(f"Cloud Build succeeded: {image_name}")
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def deploy_cloud_run(
|
|
156
|
+
project_id: str,
|
|
157
|
+
location: str,
|
|
158
|
+
service_name: str,
|
|
159
|
+
image_name: str,
|
|
160
|
+
script_name: str,
|
|
161
|
+
) -> str:
|
|
162
|
+
"""Deploy (or update) a Cloud Run service. Returns the service URL.
|
|
163
|
+
|
|
164
|
+
LLM auth is handled by Vertex AI — the service account gets
|
|
165
|
+
roles/aiplatform.user so no API keys are injected.
|
|
166
|
+
"""
|
|
167
|
+
from google.cloud import run_v2
|
|
168
|
+
|
|
169
|
+
client = run_v2.ServicesClient()
|
|
170
|
+
parent = f"projects/{project_id}/locations/{location}"
|
|
171
|
+
full_name = f"{parent}/services/{service_name}"
|
|
172
|
+
|
|
173
|
+
service_account = _ensure_service_account(project_id)
|
|
174
|
+
|
|
175
|
+
env_vars = [
|
|
176
|
+
run_v2.EnvVar(name="LAMIA_SCRIPT", value=script_name),
|
|
177
|
+
run_v2.EnvVar(name="GOOGLE_CLOUD_PROJECT", value=project_id),
|
|
178
|
+
]
|
|
179
|
+
|
|
180
|
+
container = run_v2.Container(
|
|
181
|
+
image=image_name,
|
|
182
|
+
env=env_vars,
|
|
183
|
+
resources=run_v2.ResourceRequirements(
|
|
184
|
+
limits={"memory": "512Mi", "cpu": "1"},
|
|
185
|
+
),
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
service = run_v2.Service(
|
|
189
|
+
template=run_v2.RevisionTemplate(
|
|
190
|
+
containers=[container],
|
|
191
|
+
service_account=service_account,
|
|
192
|
+
max_instance_request_concurrency=1,
|
|
193
|
+
timeout={"seconds": 540},
|
|
194
|
+
),
|
|
195
|
+
ingress=run_v2.IngressTraffic.INGRESS_TRAFFIC_INTERNAL_ONLY,
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
from google.api_core.exceptions import NotFound
|
|
199
|
+
|
|
200
|
+
service.name = full_name
|
|
201
|
+
|
|
202
|
+
try:
|
|
203
|
+
operation = client.update_service(service=service)
|
|
204
|
+
result = operation.result(timeout=300)
|
|
205
|
+
except NotFound:
|
|
206
|
+
service.name = ""
|
|
207
|
+
operation = client.create_service(
|
|
208
|
+
parent=parent,
|
|
209
|
+
service=service,
|
|
210
|
+
service_id=service_name,
|
|
211
|
+
)
|
|
212
|
+
result = operation.result(timeout=300)
|
|
213
|
+
|
|
214
|
+
url = result.uri
|
|
215
|
+
logger.info(f"Cloud Run deployed: {url}")
|
|
216
|
+
|
|
217
|
+
_allow_scheduler_invocation(project_id, location, service_name)
|
|
218
|
+
|
|
219
|
+
return url
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _ensure_service_account(project_id: str) -> str:
|
|
223
|
+
"""Create lamia-runner service account with required permissions.
|
|
224
|
+
|
|
225
|
+
Grants:
|
|
226
|
+
- roles/aiplatform.user — Vertex AI model access
|
|
227
|
+
- Cloud Scheduler agent gets token creator on lamia-runner (for OIDC signing)
|
|
228
|
+
"""
|
|
229
|
+
from google.cloud import iam_admin_v1
|
|
230
|
+
from google.cloud import resourcemanager_v3
|
|
231
|
+
from google.iam.v1 import policy_pb2
|
|
232
|
+
|
|
233
|
+
sa_email = f"lamia-runner@{project_id}.iam.gserviceaccount.com"
|
|
234
|
+
iam_client = iam_admin_v1.IAMClient()
|
|
235
|
+
|
|
236
|
+
try:
|
|
237
|
+
iam_client.get_service_account(
|
|
238
|
+
request={"name": f"projects/{project_id}/serviceAccounts/{sa_email}"}
|
|
239
|
+
)
|
|
240
|
+
except Exception as e:
|
|
241
|
+
if "NOT_FOUND" in str(e):
|
|
242
|
+
iam_client.create_service_account(
|
|
243
|
+
request={
|
|
244
|
+
"name": f"projects/{project_id}",
|
|
245
|
+
"account_id": "lamia-runner",
|
|
246
|
+
"service_account": {"display_name": "Lamia Cloud Runner"},
|
|
247
|
+
}
|
|
248
|
+
)
|
|
249
|
+
logger.info(f"Created service account: {sa_email}")
|
|
250
|
+
else:
|
|
251
|
+
raise
|
|
252
|
+
|
|
253
|
+
rm_client = resourcemanager_v3.ProjectsClient()
|
|
254
|
+
resource = f"projects/{project_id}"
|
|
255
|
+
policy = rm_client.get_iam_policy(request={"resource": resource})
|
|
256
|
+
|
|
257
|
+
project_number = _get_project_number(project_id)
|
|
258
|
+
scheduler_sa = f"service-{project_number}@gcp-sa-cloudscheduler.iam.gserviceaccount.com"
|
|
259
|
+
member = f"serviceAccount:{sa_email}"
|
|
260
|
+
|
|
261
|
+
required_bindings = {
|
|
262
|
+
"roles/aiplatform.user": [member],
|
|
263
|
+
"roles/iam.serviceAccountTokenCreator": [f"serviceAccount:{scheduler_sa}"],
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
changed = False
|
|
267
|
+
for role, members in required_bindings.items():
|
|
268
|
+
for m in members:
|
|
269
|
+
already = any(
|
|
270
|
+
b.role == role and m in b.members for b in policy.bindings
|
|
271
|
+
)
|
|
272
|
+
if not already:
|
|
273
|
+
policy.bindings.append(
|
|
274
|
+
policy_pb2.Binding(role=role, members=[m])
|
|
275
|
+
)
|
|
276
|
+
logger.info(f"Granted {role} to {m}")
|
|
277
|
+
changed = True
|
|
278
|
+
|
|
279
|
+
if changed:
|
|
280
|
+
rm_client.set_iam_policy(request={"resource": resource, "policy": policy})
|
|
281
|
+
|
|
282
|
+
return sa_email
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _allow_scheduler_invocation(project_id: str, location: str, service_name: str) -> None:
|
|
286
|
+
"""Grant Cloud Scheduler permission to invoke the Cloud Run service."""
|
|
287
|
+
from google.cloud import run_v2
|
|
288
|
+
from google.iam.v1 import iam_policy_pb2, policy_pb2
|
|
289
|
+
|
|
290
|
+
client = run_v2.ServicesClient()
|
|
291
|
+
resource = f"projects/{project_id}/locations/{location}/services/{service_name}"
|
|
292
|
+
|
|
293
|
+
try:
|
|
294
|
+
policy = client.get_iam_policy(request={"resource": resource})
|
|
295
|
+
except Exception:
|
|
296
|
+
policy = policy_pb2.Policy()
|
|
297
|
+
|
|
298
|
+
invoker_role = "roles/run.invoker"
|
|
299
|
+
scheduler_sa = f"service-{_get_project_number(project_id)}@gcp-sa-cloudscheduler.iam.gserviceaccount.com"
|
|
300
|
+
member = f"serviceAccount:{scheduler_sa}"
|
|
301
|
+
|
|
302
|
+
for binding in policy.bindings:
|
|
303
|
+
if binding.role == invoker_role and member in binding.members:
|
|
304
|
+
return
|
|
305
|
+
|
|
306
|
+
policy.bindings.append(
|
|
307
|
+
policy_pb2.Binding(role=invoker_role, members=[member])
|
|
308
|
+
)
|
|
309
|
+
client.set_iam_policy(request={"resource": resource, "policy": policy})
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def _get_project_number(project_id: str) -> str:
|
|
313
|
+
"""Get the project number from project ID."""
|
|
314
|
+
from google.cloud import resourcemanager_v3
|
|
315
|
+
|
|
316
|
+
client = resourcemanager_v3.ProjectsClient()
|
|
317
|
+
project = client.get_project(name=f"projects/{project_id}")
|
|
318
|
+
return project.name.split("/")[1]
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def deploy(
|
|
322
|
+
project_id: str,
|
|
323
|
+
location: str,
|
|
324
|
+
project_root: Path,
|
|
325
|
+
script_name: str,
|
|
326
|
+
schedule_id: str,
|
|
327
|
+
) -> str:
|
|
328
|
+
"""Full deploy pipeline. Returns the Cloud Run service URL."""
|
|
329
|
+
service_name = _service_name(schedule_id)
|
|
330
|
+
image = _image_name(project_id, schedule_id)
|
|
331
|
+
|
|
332
|
+
logger.info(f"Packaging {script_name} for deployment...")
|
|
333
|
+
staging = package_deployment(project_root, script_name, schedule_id)
|
|
334
|
+
|
|
335
|
+
try:
|
|
336
|
+
logger.info("Creating source tarball...")
|
|
337
|
+
tarball = create_source_tarball(staging)
|
|
338
|
+
|
|
339
|
+
logger.info("Uploading source to GCS...")
|
|
340
|
+
source_uri = upload_source(project_id, tarball, schedule_id)
|
|
341
|
+
|
|
342
|
+
logger.info("Submitting Cloud Build...")
|
|
343
|
+
submit_build(project_id, source_uri, image)
|
|
344
|
+
|
|
345
|
+
logger.info("Deploying to Cloud Run with Vertex AI access...")
|
|
346
|
+
url = deploy_cloud_run(project_id, location, service_name, image, script_name)
|
|
347
|
+
|
|
348
|
+
return url
|
|
349
|
+
finally:
|
|
350
|
+
shutil.rmtree(staging, ignore_errors=True)
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def teardown(project_id: str, location: str, schedule_id: str) -> None:
|
|
354
|
+
"""Remove the Cloud Run service for a schedule."""
|
|
355
|
+
from google.cloud import run_v2
|
|
356
|
+
|
|
357
|
+
client = run_v2.ServicesClient()
|
|
358
|
+
service_name = _service_name(schedule_id)
|
|
359
|
+
full_name = f"projects/{project_id}/locations/{location}/services/{service_name}"
|
|
360
|
+
|
|
361
|
+
try:
|
|
362
|
+
client.delete_service(name=full_name)
|
|
363
|
+
logger.info(f"Deleted Cloud Run service: {service_name}")
|
|
364
|
+
except Exception as e:
|
|
365
|
+
if "NOT_FOUND" not in str(e):
|
|
366
|
+
raise
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""GCP Cloud Scheduler backend.
|
|
2
|
+
|
|
3
|
+
Orchestrates: Cloud Build -> Cloud Run -> Cloud Scheduler.
|
|
4
|
+
The user only provides project_id and location. Everything else is automated.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import logging
|
|
9
|
+
import time
|
|
10
|
+
from typing import Optional
|
|
11
|
+
|
|
12
|
+
from lamia_cloud.interfaces import CloudScheduler
|
|
13
|
+
from lamia_cloud.types import CloudScheduleJob, CloudJobStatus
|
|
14
|
+
from lamia_cloud.gcp.deployer import deploy, teardown
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger(__name__)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _enable_apis(project_id: str) -> None:
|
|
20
|
+
"""Enable all required GCP APIs automatically.
|
|
21
|
+
|
|
22
|
+
Service Usage API must be enabled first (bootstraps the rest).
|
|
23
|
+
Most GCP projects have it enabled by default.
|
|
24
|
+
"""
|
|
25
|
+
try:
|
|
26
|
+
from google.cloud import service_usage_v1
|
|
27
|
+
client = service_usage_v1.ServiceUsageClient()
|
|
28
|
+
apis = [
|
|
29
|
+
"serviceusage.googleapis.com",
|
|
30
|
+
"cloudscheduler.googleapis.com",
|
|
31
|
+
"cloudbuild.googleapis.com",
|
|
32
|
+
"run.googleapis.com",
|
|
33
|
+
"storage.googleapis.com",
|
|
34
|
+
"aiplatform.googleapis.com",
|
|
35
|
+
"iam.googleapis.com",
|
|
36
|
+
]
|
|
37
|
+
for api in apis:
|
|
38
|
+
service_name = f"projects/{project_id}/services/{api}"
|
|
39
|
+
try:
|
|
40
|
+
client.enable_service(request={"name": service_name})
|
|
41
|
+
except Exception as e:
|
|
42
|
+
if "SERVICE_DISABLED" in str(e) and "serviceusage" in api:
|
|
43
|
+
logger.warning(
|
|
44
|
+
f"Service Usage API not enabled. Run once:\n"
|
|
45
|
+
f" gcloud services enable serviceusage.googleapis.com "
|
|
46
|
+
f"--project={project_id}"
|
|
47
|
+
)
|
|
48
|
+
return
|
|
49
|
+
except ImportError:
|
|
50
|
+
pass
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class GCPCloudScheduler(CloudScheduler):
|
|
54
|
+
"""GCP Cloud Scheduler backend with automatic deployment."""
|
|
55
|
+
|
|
56
|
+
def __init__(self, *, project_id: str, location: str):
|
|
57
|
+
self.project_id = project_id
|
|
58
|
+
self.location = location
|
|
59
|
+
self._parent = f"projects/{project_id}/locations/{location}"
|
|
60
|
+
|
|
61
|
+
@classmethod
|
|
62
|
+
def name(cls) -> str:
|
|
63
|
+
return "gcp-cloud"
|
|
64
|
+
|
|
65
|
+
@classmethod
|
|
66
|
+
def from_config(cls, cloud_cfg: dict) -> "GCPCloudScheduler":
|
|
67
|
+
"""Create instance from config.yaml cloud section."""
|
|
68
|
+
project_id = cloud_cfg.get("project_id")
|
|
69
|
+
if not project_id:
|
|
70
|
+
raise ValueError("cloud.project_id is required in config.yaml.")
|
|
71
|
+
|
|
72
|
+
location = cloud_cfg.get("location", "us-central1")
|
|
73
|
+
|
|
74
|
+
_enable_apis(project_id)
|
|
75
|
+
return cls(project_id=project_id, location=location)
|
|
76
|
+
|
|
77
|
+
def _scheduler_client(self):
|
|
78
|
+
from google.cloud import scheduler_v1
|
|
79
|
+
return scheduler_v1.CloudSchedulerClient()
|
|
80
|
+
|
|
81
|
+
def _job_name(self, job: CloudScheduleJob) -> str:
|
|
82
|
+
return f"{self._parent}/jobs/lamia-{job.schedule_id}"
|
|
83
|
+
|
|
84
|
+
def _build_scheduler_job(self, job: CloudScheduleJob, target_url: str):
|
|
85
|
+
from google.cloud import scheduler_v1
|
|
86
|
+
|
|
87
|
+
schedule = job.cron
|
|
88
|
+
if schedule == "@reboot":
|
|
89
|
+
schedule = "0 * * * *"
|
|
90
|
+
|
|
91
|
+
body = json.dumps({
|
|
92
|
+
"schedule_id": job.schedule_id,
|
|
93
|
+
"script": job.script,
|
|
94
|
+
}).encode()
|
|
95
|
+
|
|
96
|
+
return scheduler_v1.Job(
|
|
97
|
+
name=self._job_name(job),
|
|
98
|
+
schedule=schedule,
|
|
99
|
+
time_zone="UTC",
|
|
100
|
+
http_target=scheduler_v1.HttpTarget(
|
|
101
|
+
uri=target_url,
|
|
102
|
+
http_method=scheduler_v1.HttpMethod.POST,
|
|
103
|
+
headers={"Content-Type": "application/json"},
|
|
104
|
+
body=body,
|
|
105
|
+
oidc_token=scheduler_v1.OidcToken(
|
|
106
|
+
service_account_email=(
|
|
107
|
+
f"lamia-runner@{self.project_id}.iam.gserviceaccount.com"
|
|
108
|
+
),
|
|
109
|
+
audience=target_url,
|
|
110
|
+
),
|
|
111
|
+
),
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
def install(self, job: CloudScheduleJob, lamia_bin: str) -> None:
|
|
115
|
+
"""Deploy the script to Cloud Run and create a Cloud Scheduler trigger."""
|
|
116
|
+
logger.info(f"Deploying {job.script} to cloud...")
|
|
117
|
+
|
|
118
|
+
service_url = deploy(
|
|
119
|
+
project_id=self.project_id,
|
|
120
|
+
location=self.location,
|
|
121
|
+
project_root=job.project_root,
|
|
122
|
+
script_name=job.script,
|
|
123
|
+
schedule_id=job.schedule_id,
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
client = self._scheduler_client()
|
|
127
|
+
scheduler_job = self._build_scheduler_job(job, service_url)
|
|
128
|
+
|
|
129
|
+
from google.api_core.exceptions import AlreadyExists, NotFound
|
|
130
|
+
|
|
131
|
+
for attempt in range(5):
|
|
132
|
+
try:
|
|
133
|
+
client.create_job(parent=self._parent, job=scheduler_job)
|
|
134
|
+
logger.info(f"Created cloud schedule: {job.script} ({job.cron})")
|
|
135
|
+
break
|
|
136
|
+
except AlreadyExists:
|
|
137
|
+
client.update_job(job=scheduler_job)
|
|
138
|
+
logger.info(f"Updated cloud schedule: {job.script} ({job.cron})")
|
|
139
|
+
break
|
|
140
|
+
except NotFound:
|
|
141
|
+
if attempt == 4:
|
|
142
|
+
raise RuntimeError(
|
|
143
|
+
f"Cloud Scheduler location '{self.location}' not ready after retries. "
|
|
144
|
+
f"The API may still be propagating — wait a minute and retry."
|
|
145
|
+
)
|
|
146
|
+
logger.info(f"Cloud Scheduler not ready yet, retrying in 15s...")
|
|
147
|
+
time.sleep(15)
|
|
148
|
+
|
|
149
|
+
def uninstall(self, job: CloudScheduleJob) -> None:
|
|
150
|
+
"""Remove both the Cloud Scheduler job and the Cloud Run service."""
|
|
151
|
+
client = self._scheduler_client()
|
|
152
|
+
from google.api_core.exceptions import Aborted, NotFound
|
|
153
|
+
|
|
154
|
+
for attempt in range(5):
|
|
155
|
+
try:
|
|
156
|
+
client.delete_job(name=self._job_name(job))
|
|
157
|
+
logger.info(f"Deleted cloud schedule: {job.script}")
|
|
158
|
+
break
|
|
159
|
+
except NotFound:
|
|
160
|
+
break
|
|
161
|
+
except Aborted:
|
|
162
|
+
if attempt == 4:
|
|
163
|
+
raise
|
|
164
|
+
time.sleep(10)
|
|
165
|
+
|
|
166
|
+
teardown(self.project_id, self.location, job.schedule_id)
|
|
167
|
+
|
|
168
|
+
def is_installed(self, job: CloudScheduleJob) -> bool:
|
|
169
|
+
client = self._scheduler_client()
|
|
170
|
+
try:
|
|
171
|
+
client.get_job(name=self._job_name(job))
|
|
172
|
+
return True
|
|
173
|
+
except Exception:
|
|
174
|
+
return False
|
|
175
|
+
|
|
176
|
+
def get_status(self, job: CloudScheduleJob) -> CloudJobStatus:
|
|
177
|
+
client = self._scheduler_client()
|
|
178
|
+
try:
|
|
179
|
+
cloud_job = client.get_job(name=self._job_name(job))
|
|
180
|
+
from google.cloud.scheduler_v1 import Job
|
|
181
|
+
if cloud_job.state == Job.State.ENABLED:
|
|
182
|
+
return CloudJobStatus.ACTIVE
|
|
183
|
+
return CloudJobStatus.INACTIVE
|
|
184
|
+
except Exception:
|
|
185
|
+
return CloudJobStatus.UNKNOWN
|
|
186
|
+
|
|
187
|
+
def pause(self, job: CloudScheduleJob) -> None:
|
|
188
|
+
"""Pause the Cloud Scheduler job (stops triggering)."""
|
|
189
|
+
client = self._scheduler_client()
|
|
190
|
+
client.pause_job(name=self._job_name(job))
|
|
191
|
+
logger.info(f"Paused cloud schedule: {job.script}")
|
|
192
|
+
|
|
193
|
+
def resume(self, job: CloudScheduleJob) -> None:
|
|
194
|
+
"""Resume a paused Cloud Scheduler job."""
|
|
195
|
+
client = self._scheduler_client()
|
|
196
|
+
client.resume_job(name=self._job_name(job))
|
|
197
|
+
logger.info(f"Resumed cloud schedule: {job.script}")
|
|
198
|
+
|
|
199
|
+
def get_installed_config(self, job: CloudScheduleJob) -> Optional[dict]:
|
|
200
|
+
client = self._scheduler_client()
|
|
201
|
+
try:
|
|
202
|
+
cloud_job = client.get_job(name=self._job_name(job))
|
|
203
|
+
return {
|
|
204
|
+
"schedule": cloud_job.schedule,
|
|
205
|
+
"state": cloud_job.state.name,
|
|
206
|
+
"last_attempt_time": (
|
|
207
|
+
cloud_job.last_attempt_time.isoformat()
|
|
208
|
+
if cloud_job.last_attempt_time
|
|
209
|
+
else None
|
|
210
|
+
),
|
|
211
|
+
}
|
|
212
|
+
except Exception:
|
|
213
|
+
return None
|