buildai-cli 0.3.151__tar.gz → 0.3.153__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/PKG-INFO +1 -1
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/egoexo.py +2 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/egoexo_config.py +151 -17
- buildai_cli-0.3.153/cli/commands/egoexo_status.py +215 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/pyproject.toml +1 -1
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/.gitignore +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/AGENTS.md +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/CLAUDE.md +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/Dockerfile.rescaling-migration-jobs +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/Dockerfile.rescaling-migration-jobs.dockerignore +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/buildai_bootstrap.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/_has_core.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/auth_local.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/api_proxy.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/auth.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/broker.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/common.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/migrate.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/query.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/rescaling.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/schema.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/status.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/tunnel.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/dev.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/doctor.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/ego_frame_search.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/gigcamera.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/ingest.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/ingest_docs.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/intake.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/label_replay_recovery.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/labeling.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/processing.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/rescaling.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/spec.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/spec_pr.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/config.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/console.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/context.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/db_broker.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/guard.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/internal_api.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/labeling_onboarding.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/main.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/nl_query/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/nl_query/dataset_tools.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/ops_init.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/output.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/pagination.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/canonical.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/export.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/feed_proofs.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/fence.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/importer.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/jobs.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/management.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/orchestrator.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/probes.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/reconcile.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/run_record.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/runtime.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_export_closeout.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_export_prepare.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_fence.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_unfreeze.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_export_closeout.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_export_prepare.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_fence.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_unfreeze.sql +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/storage.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/transform.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rls_guard.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_logical_backup/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_logical_backup/__main__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_migration_job/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_migration_job/__main__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_recovery_extractor/__init__.py +0 -0
- {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_recovery_extractor/__main__.py +0 -0
|
@@ -11,11 +11,13 @@ import typer
|
|
|
11
11
|
|
|
12
12
|
from cli.commands.api_proxy import _build_url
|
|
13
13
|
from cli.commands.egoexo_config import app as config_app
|
|
14
|
+
from cli.commands.egoexo_status import status
|
|
14
15
|
from cli.config import resolve_api_url, resolve_credential
|
|
15
16
|
from cli.console import error
|
|
16
17
|
|
|
17
18
|
app = typer.Typer(help="Inspect EgoExo verdict-plane outcomes.", no_args_is_help=True)
|
|
18
19
|
app.add_typer(config_app, name="config")
|
|
20
|
+
app.command("status")(status)
|
|
19
21
|
|
|
20
22
|
|
|
21
23
|
def _request(path: str, *, query_params: list[str] | None = None) -> None:
|
|
@@ -333,12 +333,14 @@ def set_recipe(
|
|
|
333
333
|
|
|
334
334
|
@app.command("set-task-classes")
|
|
335
335
|
def set_task_classes(
|
|
336
|
-
file: Path = typer.Argument(
|
|
336
|
+
file: Path = typer.Argument(
|
|
337
|
+
..., help="Per-class overlays keyed by name, for example limits; images stay in env."
|
|
338
|
+
),
|
|
337
339
|
uri: str | None = URI_OPTION,
|
|
338
340
|
write: bool = WRITE_OPTION,
|
|
339
341
|
reason: str | None = REASON_OPTION,
|
|
340
342
|
) -> None:
|
|
341
|
-
"""Store
|
|
343
|
+
"""Store per-class field overlays applied on top of the environment task classes."""
|
|
342
344
|
classes = _load_json_file(file)
|
|
343
345
|
_revise(
|
|
344
346
|
_store(uri),
|
|
@@ -425,6 +427,152 @@ def clear(
|
|
|
425
427
|
)
|
|
426
428
|
|
|
427
429
|
|
|
430
|
+
def _effective_documents(store, deployment: Path | None):
|
|
431
|
+
"""Return the recipe and trusted assets the API would use, and where the recipe came from.
|
|
432
|
+
|
|
433
|
+
Returns ``(recipe, assets, recipe_source, generation, problem)``. Store values
|
|
434
|
+
win; the checked-in deployment files are the fallback. Both are deep copies.
|
|
435
|
+
"""
|
|
436
|
+
document, generation, problem = _read_base(store)
|
|
437
|
+
env, deployment_dir = _deployment_env(_deployment_dir(deployment))
|
|
438
|
+
if document is not None and document.recipe is not None:
|
|
439
|
+
recipe = json.loads(json.dumps(document.recipe))
|
|
440
|
+
recipe_source = "store"
|
|
441
|
+
elif env.recipe_json:
|
|
442
|
+
recipe = json.loads(env.recipe_json)
|
|
443
|
+
recipe_source = f"env ({deployment_dir / 'egoexo-recipe.json'})"
|
|
444
|
+
else:
|
|
445
|
+
error("No recipe in the store and no --deployment recipe to start from.")
|
|
446
|
+
raise typer.Exit(1)
|
|
447
|
+
if document is not None and document.trusted_assets is not None:
|
|
448
|
+
assets = json.loads(json.dumps(document.trusted_assets))
|
|
449
|
+
else:
|
|
450
|
+
assets = json.loads(env.trusted_assets_json)
|
|
451
|
+
return recipe, assets, recipe_source, generation, problem
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
IDENTITY_FIELDS = ("generation", "size_bytes", "crc32c_base64")
|
|
455
|
+
RECIPE_ASSET_KEYS = ("config", "board", "crown_observations_config", "crown_model_config")
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def _recipe_refs(recipe: dict[str, Any]) -> list[dict[str, Any]]:
|
|
459
|
+
"""Collect every artifact reference the recipe hands to a stage or the planner."""
|
|
460
|
+
refs = []
|
|
461
|
+
for stage, spec in sorted(recipe.get("preset", {}).items()):
|
|
462
|
+
for key in RECIPE_ASSET_KEYS:
|
|
463
|
+
if key in spec:
|
|
464
|
+
refs.append({"source": f"preset.{stage}.{key}", **spec[key]})
|
|
465
|
+
for index, checkpoint in enumerate(spec.get("checkpoints", [])):
|
|
466
|
+
refs.append({"source": f"preset.{stage}.checkpoints[{index}]", **checkpoint})
|
|
467
|
+
for key in ("keypoint_schema", "episode_selection"):
|
|
468
|
+
if recipe.get(key) is not None:
|
|
469
|
+
refs.append({"source": key, **recipe[key]})
|
|
470
|
+
return refs
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def _live_identity(storage, asset: dict[str, Any]) -> dict[str, Any]:
|
|
474
|
+
"""Read the pinned generation's metadata, and its bytes when small enough to hash."""
|
|
475
|
+
from egoexo.adapters.gcs import publication
|
|
476
|
+
|
|
477
|
+
try:
|
|
478
|
+
actual = {"present": True, **storage.metadata(asset["uri"], asset["generation"])}
|
|
479
|
+
except Exception as exc: # noqa: BLE001 - the storage SDK reports a miss as 404 or 412
|
|
480
|
+
if getattr(exc, "code", None) in {404, 412} or type(exc).__name__ in {
|
|
481
|
+
"NotFound",
|
|
482
|
+
"PreconditionFailed",
|
|
483
|
+
}:
|
|
484
|
+
return {"present": False}
|
|
485
|
+
raise
|
|
486
|
+
size = actual["size_bytes"]
|
|
487
|
+
if size is not None and size <= LOCAL_JSON_MAX_BYTES:
|
|
488
|
+
with tempfile.TemporaryDirectory() as scratch:
|
|
489
|
+
path = Path(scratch) / "asset"
|
|
490
|
+
storage.download(asset["uri"], asset["generation"], path, size)
|
|
491
|
+
_, sha, crc = publication.identity(path)
|
|
492
|
+
actual |= {"sha256": sha, "downloaded_crc32c_base64": crc}
|
|
493
|
+
return actual
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def _asset_failures(asset: dict[str, Any], actual: dict[str, Any]) -> list[str]:
|
|
497
|
+
"""Return every way the live object differs from the pinned identity."""
|
|
498
|
+
if not actual.get("present"):
|
|
499
|
+
return ["object_missing_at_pinned_generation"]
|
|
500
|
+
failures = [
|
|
501
|
+
f"{field}: pinned {asset[field]!r} live {actual[field]!r}"
|
|
502
|
+
for field in IDENTITY_FIELDS
|
|
503
|
+
if str(asset[field]) != str(actual[field])
|
|
504
|
+
]
|
|
505
|
+
if "sha256" in actual and actual["sha256"] != asset["sha256"]:
|
|
506
|
+
failures.append(f"sha256: pinned {asset['sha256']} live {actual['sha256']}")
|
|
507
|
+
if "downloaded_crc32c_base64" in actual and (
|
|
508
|
+
actual["downloaded_crc32c_base64"] != actual["crc32c_base64"]
|
|
509
|
+
):
|
|
510
|
+
failures.append("crc32c does not match the downloaded bytes")
|
|
511
|
+
return failures
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
@app.command("verify-assets")
|
|
515
|
+
def verify_assets(
|
|
516
|
+
uri: str | None = URI_OPTION,
|
|
517
|
+
deployment: Path | None = DEPLOYMENT_OPTION,
|
|
518
|
+
json_path: Path | None = typer.Option(
|
|
519
|
+
None, "--json", help="Also write the full report to this path."
|
|
520
|
+
),
|
|
521
|
+
) -> None:
|
|
522
|
+
"""Check every recipe asset and trusted asset against the live object it pins.
|
|
523
|
+
|
|
524
|
+
Compares generation, size and CRC32C for each pinned object, and SHA256 for
|
|
525
|
+
objects up to 1 MiB, which are downloaded. Larger checkpoints are checked by
|
|
526
|
+
metadata only, the same check the API makes before planning. Reads only.
|
|
527
|
+
"""
|
|
528
|
+
_runtime_config()
|
|
529
|
+
store = _store(uri)
|
|
530
|
+
recipe, assets, recipe_source, generation, problem = _effective_documents(store, deployment)
|
|
531
|
+
trusted = {asset["uri"]: asset for asset in assets}
|
|
532
|
+
problems = []
|
|
533
|
+
for stage, spec in sorted(recipe.get("preset", {}).items()):
|
|
534
|
+
config = spec.get("config")
|
|
535
|
+
if config is not None and spec["method"]["config_sha256"] != config["sha256"]:
|
|
536
|
+
problems.append(f"preset.{stage}: method.config_sha256 != config.sha256")
|
|
537
|
+
refs = _recipe_refs(recipe)
|
|
538
|
+
for ref in refs:
|
|
539
|
+
asset = trusted.get(ref["uri"])
|
|
540
|
+
if asset is None:
|
|
541
|
+
problems.append(f"{ref['source']}: {ref['uri']} is not a trusted asset")
|
|
542
|
+
continue
|
|
543
|
+
problems.extend(
|
|
544
|
+
f"{ref['source']}: {field} differs from the trusted asset entry"
|
|
545
|
+
for field in (*IDENTITY_FIELDS, "sha256")
|
|
546
|
+
if asset[field] != ref[field]
|
|
547
|
+
)
|
|
548
|
+
used = {ref["uri"] for ref in refs}
|
|
549
|
+
objects = {}
|
|
550
|
+
for asset in sorted(trusted.values(), key=lambda a: a["uri"]):
|
|
551
|
+
actual = _live_identity(store.storage, asset)
|
|
552
|
+
failures = _asset_failures(asset, actual)
|
|
553
|
+
objects[asset["uri"]] = {
|
|
554
|
+
"used_by_recipe": asset["uri"] in used,
|
|
555
|
+
"actual": actual,
|
|
556
|
+
"failures": failures,
|
|
557
|
+
}
|
|
558
|
+
problems.extend(f"{asset['uri']}: {failure}" for failure in failures)
|
|
559
|
+
report = {
|
|
560
|
+
"uri": store.uri,
|
|
561
|
+
"generation": None if generation == "0" else generation,
|
|
562
|
+
"current_problem": problem,
|
|
563
|
+
"recipe_source": recipe_source,
|
|
564
|
+
"recipe_references": len(refs),
|
|
565
|
+
"trusted_assets": len(assets),
|
|
566
|
+
"problems": problems,
|
|
567
|
+
"objects": objects,
|
|
568
|
+
}
|
|
569
|
+
if json_path is not None:
|
|
570
|
+
json_path.write_text(json.dumps(report, indent=2, sort_keys=True, default=str) + "\n")
|
|
571
|
+
_emit(report)
|
|
572
|
+
if problems:
|
|
573
|
+
raise typer.Exit(1)
|
|
574
|
+
|
|
575
|
+
|
|
428
576
|
def _asset_group(uri: str) -> str | None:
|
|
429
577
|
"""Return the <group> segment of derived/runtime-assets/<group>/<sha256>/<name>."""
|
|
430
578
|
marker = uri.find(ASSET_PREFIX)
|
|
@@ -461,21 +609,7 @@ def pin_config_asset(
|
|
|
461
609
|
|
|
462
610
|
_load_json_file(file)
|
|
463
611
|
store = _store(uri)
|
|
464
|
-
|
|
465
|
-
env, deployment_dir = _deployment_env(_deployment_dir(deployment))
|
|
466
|
-
if document is not None and document.recipe is not None:
|
|
467
|
-
recipe = json.loads(json.dumps(document.recipe))
|
|
468
|
-
recipe_source = "store"
|
|
469
|
-
elif env.recipe_json:
|
|
470
|
-
recipe = json.loads(env.recipe_json)
|
|
471
|
-
recipe_source = f"env ({deployment_dir / 'egoexo-recipe.json'})"
|
|
472
|
-
else:
|
|
473
|
-
error("No recipe in the store and no --deployment recipe to start from.")
|
|
474
|
-
raise typer.Exit(1)
|
|
475
|
-
if document is not None and document.trusted_assets is not None:
|
|
476
|
-
assets = json.loads(json.dumps(document.trusted_assets))
|
|
477
|
-
else:
|
|
478
|
-
assets = json.loads(env.trusted_assets_json)
|
|
612
|
+
recipe, assets, recipe_source, generation, problem = _effective_documents(store, deployment)
|
|
479
613
|
spec = recipe.get("preset", {}).get(stage)
|
|
480
614
|
if not isinstance(spec, dict) or not isinstance(spec.get("method"), dict):
|
|
481
615
|
error(f"Stage {stage!r} is not in the recipe preset.")
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""Read-only EgoExo pipeline status: work items per capture and stage, plus Batch jobs.
|
|
2
|
+
|
|
3
|
+
The processing API key is read from Secret Manager in process and never printed.
|
|
4
|
+
Batch jobs are listed through the REST API with a server-side filter, because
|
|
5
|
+
``gcloud batch jobs list`` filters on the client and takes minutes in this project.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import subprocess
|
|
12
|
+
import time
|
|
13
|
+
from collections import Counter, defaultdict
|
|
14
|
+
from datetime import UTC, datetime, timedelta
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
import httpx
|
|
19
|
+
import typer
|
|
20
|
+
|
|
21
|
+
from cli.config import resolve_api_url
|
|
22
|
+
|
|
23
|
+
PROJECT = "data-470400"
|
|
24
|
+
REGION = "us-central1"
|
|
25
|
+
BUCKET = "buildai-ego-exo"
|
|
26
|
+
ACTIVE_STATES = frozenset({"SCHEDULED", "QUEUED", "RUNNING"})
|
|
27
|
+
# The worker reads its assignment under the prefix of the image it runs in.
|
|
28
|
+
ASSIGNED_PREFIXES = ("EGOEXO_GPU", "EGOEXO_CPU")
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _gcloud(*args: str) -> str:
|
|
32
|
+
"""Run one gcloud command and return its stdout. Tests replace this function."""
|
|
33
|
+
return subprocess.run(["gcloud", *args], capture_output=True, text=True, check=True).stdout
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _client() -> httpx.Client:
|
|
37
|
+
"""HTTP client for the processing API and the Batch API. Tests replace this function."""
|
|
38
|
+
return httpx.Client(timeout=60.0)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _api_key(project: str) -> str:
|
|
42
|
+
"""Read the processing API key from Secret Manager."""
|
|
43
|
+
return _gcloud(
|
|
44
|
+
"secrets",
|
|
45
|
+
"versions",
|
|
46
|
+
"access",
|
|
47
|
+
"latest",
|
|
48
|
+
"--secret=PROCESSING_API_KEY",
|
|
49
|
+
f"--project={project}",
|
|
50
|
+
).strip()
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _raw_capture_ids(bucket: str) -> list[str]:
|
|
54
|
+
"""List capture UUIDs under raw/cn/<date>/."""
|
|
55
|
+
ids = []
|
|
56
|
+
for date in _gcloud("storage", "ls", f"gs://{bucket}/raw/cn/").split():
|
|
57
|
+
for capture in _gcloud("storage", "ls", date).split():
|
|
58
|
+
ids.append(capture.rstrip("/").rsplit("/", 1)[-1])
|
|
59
|
+
return sorted(ids)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _get_json(client: httpx.Client, url: str, token: str, params: dict | None = None) -> Any:
|
|
63
|
+
"""GET JSON with a bearer token. Server errors are retried a few times."""
|
|
64
|
+
for attempt in range(5):
|
|
65
|
+
response = client.get(url, params=params, headers={"Authorization": f"Bearer {token}"})
|
|
66
|
+
if response.status_code < 500 or attempt == 4:
|
|
67
|
+
response.raise_for_status()
|
|
68
|
+
return response.json()
|
|
69
|
+
time.sleep(2 * (attempt + 1))
|
|
70
|
+
raise AssertionError("unreachable")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _capture_items(client: httpx.Client, api_url: str, key: str, capture_id: str) -> list[dict]:
|
|
74
|
+
"""Page through one capture's work items."""
|
|
75
|
+
url = f"{api_url}/v1/processing/egoexo-captures/{capture_id}"
|
|
76
|
+
items: list[dict] = []
|
|
77
|
+
cursor = None
|
|
78
|
+
while True:
|
|
79
|
+
params = {"limit": 200} | ({"cursor": cursor} if cursor else {})
|
|
80
|
+
data = _get_json(client, url, key, params).get("data")
|
|
81
|
+
if not isinstance(data, dict):
|
|
82
|
+
return []
|
|
83
|
+
items.extend(data.get("items", []))
|
|
84
|
+
cursor = data.get("next_cursor")
|
|
85
|
+
if not data.get("truncated") or not cursor:
|
|
86
|
+
return items
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _stage_of(item: dict) -> str:
|
|
90
|
+
"""Derive the stage name from the request when present, else the work key."""
|
|
91
|
+
request = (item.get("metadata") or {}).get("request") or {}
|
|
92
|
+
return request.get("stage") or str(item.get("work_key", "?")).split(":", 1)[0]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _assigned_id(env: dict[str, str], suffix: str) -> str | None:
|
|
96
|
+
"""Read the assigned work item or attempt id under the GPU or the CPU worker prefix."""
|
|
97
|
+
for prefix in ASSIGNED_PREFIXES:
|
|
98
|
+
value = env.get(f"{prefix}_ASSIGNED_{suffix}")
|
|
99
|
+
if value:
|
|
100
|
+
return value
|
|
101
|
+
return None
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _batch_jobs(client: httpx.Client, project: str, region: str, since_hours: int) -> list[dict]:
|
|
105
|
+
"""List recent assigned Batch jobs. Only job metadata and the assignment ids are kept."""
|
|
106
|
+
token = _gcloud("auth", "print-access-token").strip()
|
|
107
|
+
since = (datetime.now(UTC) - timedelta(hours=since_hours)).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
108
|
+
url = f"https://batch.googleapis.com/v1/projects/{project}/locations/{region}/jobs"
|
|
109
|
+
query = {
|
|
110
|
+
"filter": f'labels.lane="egoexo-assigned" AND createTime>"{since}"',
|
|
111
|
+
"pageSize": "500",
|
|
112
|
+
}
|
|
113
|
+
rows: list[dict] = []
|
|
114
|
+
page_token = None
|
|
115
|
+
for _ in range(200):
|
|
116
|
+
params = query | ({"pageToken": page_token} if page_token else {})
|
|
117
|
+
payload = _get_json(client, url, token, params)
|
|
118
|
+
for job in payload.get("jobs", []):
|
|
119
|
+
env = job["taskGroups"][0]["taskSpec"].get("environment", {}).get("variables", {})
|
|
120
|
+
status = job.get("status", {})
|
|
121
|
+
rows.append(
|
|
122
|
+
{
|
|
123
|
+
"name": job["name"].rsplit("/", 1)[-1],
|
|
124
|
+
"state": status.get("state"),
|
|
125
|
+
"task_class": job.get("labels", {}).get("task-class"),
|
|
126
|
+
"created": job.get("createTime"),
|
|
127
|
+
"run_duration": status.get("runDuration"),
|
|
128
|
+
"work_item_id": _assigned_id(env, "WORK_ITEM_ID"),
|
|
129
|
+
"attempt_id": _assigned_id(env, "ATTEMPT_ID"),
|
|
130
|
+
"machine_type": job["allocationPolicy"]["instances"][0]["policy"].get(
|
|
131
|
+
"machineType"
|
|
132
|
+
),
|
|
133
|
+
"events": [
|
|
134
|
+
{"time": e.get("eventTime"), "description": e.get("description")}
|
|
135
|
+
for e in status.get("statusEvents", [])
|
|
136
|
+
],
|
|
137
|
+
}
|
|
138
|
+
)
|
|
139
|
+
page_token = payload.get("nextPageToken")
|
|
140
|
+
if not page_token:
|
|
141
|
+
break
|
|
142
|
+
return sorted(rows, key=lambda row: row["created"] or "")
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _capture_report(items: list[dict], *, include_items: bool) -> tuple[dict, str]:
|
|
146
|
+
"""Summarize one capture's work items by stage and state. Returns (report, one line)."""
|
|
147
|
+
by_stage: dict[str, Counter] = defaultdict(Counter)
|
|
148
|
+
errors = []
|
|
149
|
+
for item in items:
|
|
150
|
+
by_stage[_stage_of(item)][item.get("state", "?")] += 1
|
|
151
|
+
if item.get("last_error"):
|
|
152
|
+
errors.append({"stage": _stage_of(item), "error": str(item["last_error"])[:200]})
|
|
153
|
+
report = {
|
|
154
|
+
"items": len(items),
|
|
155
|
+
"stages": {stage: dict(counts) for stage, counts in sorted(by_stage.items())},
|
|
156
|
+
"errors": errors[:20],
|
|
157
|
+
}
|
|
158
|
+
if include_items:
|
|
159
|
+
report["raw_items"] = items
|
|
160
|
+
summary = ", ".join(
|
|
161
|
+
f"{stage}:{'/'.join(f'{state}={n}' for state, n in sorted(counts.items()))}"
|
|
162
|
+
for stage, counts in sorted(by_stage.items())
|
|
163
|
+
)
|
|
164
|
+
return report, f"items={len(items)} {summary or '(none)'}"
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def status(
|
|
168
|
+
captures: list[str] | None = typer.Option(
|
|
169
|
+
None,
|
|
170
|
+
"--capture",
|
|
171
|
+
help="Capture id. Repeat for several. Default: every capture under raw/cn/.",
|
|
172
|
+
),
|
|
173
|
+
json_path: Path | None = typer.Option(
|
|
174
|
+
None, "--json", help="Write the full report to this path."
|
|
175
|
+
),
|
|
176
|
+
items: bool = typer.Option(False, "--items", help="Include raw work items in the report."),
|
|
177
|
+
since_hours: int = typer.Option(24, "--since-hours", min=1, help="Batch job window."),
|
|
178
|
+
project: str = typer.Option(PROJECT, "--project", help="Google Cloud project."),
|
|
179
|
+
) -> None:
|
|
180
|
+
"""Show work items per capture and stage, and recent assigned Batch jobs. Reads only."""
|
|
181
|
+
key = _api_key(project)
|
|
182
|
+
api_url = resolve_api_url()
|
|
183
|
+
report: dict[str, Any] = {
|
|
184
|
+
"read_at": datetime.now(UTC).isoformat(),
|
|
185
|
+
"api_url": api_url,
|
|
186
|
+
"captures": {},
|
|
187
|
+
"batch_jobs": [],
|
|
188
|
+
}
|
|
189
|
+
with _client() as client:
|
|
190
|
+
for capture_id in captures or _raw_capture_ids(BUCKET):
|
|
191
|
+
try:
|
|
192
|
+
rows = _capture_items(client, api_url, key, capture_id)
|
|
193
|
+
except (httpx.HTTPError, ValueError) as exc:
|
|
194
|
+
report["captures"][capture_id] = {"error": str(exc)[:200]}
|
|
195
|
+
typer.echo(f"{capture_id} ERROR {str(exc)[:120]}")
|
|
196
|
+
continue
|
|
197
|
+
capture, line = _capture_report(rows, include_items=items)
|
|
198
|
+
report["captures"][capture_id] = capture
|
|
199
|
+
typer.echo(f"{capture_id} {line}")
|
|
200
|
+
for entry in capture["errors"][:5]:
|
|
201
|
+
typer.echo(f" ! {entry['stage']}: {entry['error']}")
|
|
202
|
+
try:
|
|
203
|
+
report["batch_jobs"] = _batch_jobs(client, project, REGION, since_hours)
|
|
204
|
+
except (httpx.HTTPError, subprocess.CalledProcessError, KeyError, ValueError) as exc:
|
|
205
|
+
report["batch_jobs"] = [{"error": str(exc)[:200]}]
|
|
206
|
+
jobs = Counter((job.get("task_class"), job.get("state")) for job in report["batch_jobs"])
|
|
207
|
+
typer.echo(f"batch jobs ({since_hours}h): {dict(jobs) or 'none'}")
|
|
208
|
+
for job in report["batch_jobs"]:
|
|
209
|
+
if job.get("state") in ACTIVE_STATES and job.get("task_class") == "gpu_inference":
|
|
210
|
+
last = job["events"][-1]["description"] if job["events"] else ""
|
|
211
|
+
typer.echo(
|
|
212
|
+
f" gpu {job['name']} {job['state']} item={job['work_item_id']} {last[:100]}"
|
|
213
|
+
)
|
|
214
|
+
if json_path is not None:
|
|
215
|
+
json_path.write_text(json.dumps(report, indent=2, default=str) + "\n")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{buildai_cli-0.3.151 → buildai_cli-0.3.153}/Dockerfile.rescaling-migration-jobs.dockerignore
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_export_closeout.sql
RENAMED
|
File without changes
|
{buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_export_prepare.sql
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_export_closeout.sql
RENAMED
|
File without changes
|
{buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_export_prepare.sql
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|