buildai-cli 0.3.151__tar.gz → 0.3.153__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/PKG-INFO +1 -1
  2. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/egoexo.py +2 -0
  3. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/egoexo_config.py +151 -17
  4. buildai_cli-0.3.153/cli/commands/egoexo_status.py +215 -0
  5. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/pyproject.toml +1 -1
  6. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/.gitignore +0 -0
  7. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/AGENTS.md +0 -0
  8. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/CLAUDE.md +0 -0
  9. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/Dockerfile.rescaling-migration-jobs +0 -0
  10. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/Dockerfile.rescaling-migration-jobs.dockerignore +0 -0
  11. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/buildai_bootstrap.py +0 -0
  12. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/__init__.py +0 -0
  13. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/_has_core.py +0 -0
  14. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/auth_local.py +0 -0
  15. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/__init__.py +0 -0
  16. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/api_proxy.py +0 -0
  17. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/auth.py +0 -0
  18. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/__init__.py +0 -0
  19. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/broker.py +0 -0
  20. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/common.py +0 -0
  21. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/migrate.py +0 -0
  22. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/query.py +0 -0
  23. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/rescaling.py +0 -0
  24. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/schema.py +0 -0
  25. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/status.py +0 -0
  26. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/db/tunnel.py +0 -0
  27. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/dev.py +0 -0
  28. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/doctor.py +0 -0
  29. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/ego_frame_search.py +0 -0
  30. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/gigcamera.py +0 -0
  31. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/ingest.py +0 -0
  32. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/ingest_docs.py +0 -0
  33. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/intake.py +0 -0
  34. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/label_replay_recovery.py +0 -0
  35. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/labeling.py +0 -0
  36. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/processing.py +0 -0
  37. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/rescaling.py +0 -0
  38. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/spec.py +0 -0
  39. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/commands/spec_pr.py +0 -0
  40. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/config.py +0 -0
  41. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/console.py +0 -0
  42. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/context.py +0 -0
  43. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/db_broker.py +0 -0
  44. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/guard.py +0 -0
  45. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/internal_api.py +0 -0
  46. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/labeling_onboarding.py +0 -0
  47. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/main.py +0 -0
  48. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/nl_query/__init__.py +0 -0
  49. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/nl_query/dataset_tools.py +0 -0
  50. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/ops_init.py +0 -0
  51. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/output.py +0 -0
  52. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/pagination.py +0 -0
  53. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/__init__.py +0 -0
  54. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/canonical.py +0 -0
  55. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/export.py +0 -0
  56. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/feed_proofs.py +0 -0
  57. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/fence.py +0 -0
  58. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/importer.py +0 -0
  59. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/jobs.py +0 -0
  60. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/management.py +0 -0
  61. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/orchestrator.py +0 -0
  62. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/probes.py +0 -0
  63. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/reconcile.py +0 -0
  64. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/run_record.py +0 -0
  65. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/runtime.py +0 -0
  66. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_export_closeout.sql +0 -0
  67. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_export_prepare.sql +0 -0
  68. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_fence.sql +0 -0
  69. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/legacy_unfreeze.sql +0 -0
  70. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_export_closeout.sql +0 -0
  71. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_export_prepare.sql +0 -0
  72. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_fence.sql +0 -0
  73. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/sql/v2_unfreeze.sql +0 -0
  74. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/storage.py +0 -0
  75. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rescaling_migration/transform.py +0 -0
  76. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/cli/rls_guard.py +0 -0
  77. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_logical_backup/__init__.py +0 -0
  78. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_logical_backup/__main__.py +0 -0
  79. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_migration_job/__init__.py +0 -0
  80. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_migration_job/__main__.py +0 -0
  81. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_recovery_extractor/__init__.py +0 -0
  82. {buildai_cli-0.3.151 → buildai_cli-0.3.153}/rescaling_recovery_extractor/__main__.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: buildai-cli
3
- Version: 0.3.151
3
+ Version: 0.3.153
4
4
  Summary: Build AI CLI (Typer)
5
5
  Requires-Python: >=3.11
6
6
  Requires-Dist: google-crc32c>=1.7.1
@@ -11,11 +11,13 @@ import typer
11
11
 
12
12
  from cli.commands.api_proxy import _build_url
13
13
  from cli.commands.egoexo_config import app as config_app
14
+ from cli.commands.egoexo_status import status
14
15
  from cli.config import resolve_api_url, resolve_credential
15
16
  from cli.console import error
16
17
 
17
18
  app = typer.Typer(help="Inspect EgoExo verdict-plane outcomes.", no_args_is_help=True)
18
19
  app.add_typer(config_app, name="config")
20
+ app.command("status")(status)
19
21
 
20
22
 
21
23
  def _request(path: str, *, query_params: list[str] | None = None) -> None:
@@ -333,12 +333,14 @@ def set_recipe(
333
333
 
334
334
  @app.command("set-task-classes")
335
335
  def set_task_classes(
336
- file: Path = typer.Argument(..., help="Task class list JSON, one to three classes."),
336
+ file: Path = typer.Argument(
337
+ ..., help="Per-class overlays keyed by name, for example limits; images stay in env."
338
+ ),
337
339
  uri: str | None = URI_OPTION,
338
340
  write: bool = WRITE_OPTION,
339
341
  reason: str | None = REASON_OPTION,
340
342
  ) -> None:
341
- """Store task classes that override BUILDAI_EGOEXO_ASSIGNED_TASK_CLASSES_JSON."""
343
+ """Store per-class field overlays applied on top of the environment task classes."""
342
344
  classes = _load_json_file(file)
343
345
  _revise(
344
346
  _store(uri),
@@ -425,6 +427,152 @@ def clear(
425
427
  )
426
428
 
427
429
 
430
+ def _effective_documents(store, deployment: Path | None):
431
+ """Return the recipe and trusted assets the API would use, and where the recipe came from.
432
+
433
+ Returns ``(recipe, assets, recipe_source, generation, problem)``. Store values
434
+ win; the checked-in deployment files are the fallback. Both are deep copies.
435
+ """
436
+ document, generation, problem = _read_base(store)
437
+ env, deployment_dir = _deployment_env(_deployment_dir(deployment))
438
+ if document is not None and document.recipe is not None:
439
+ recipe = json.loads(json.dumps(document.recipe))
440
+ recipe_source = "store"
441
+ elif env.recipe_json:
442
+ recipe = json.loads(env.recipe_json)
443
+ recipe_source = f"env ({deployment_dir / 'egoexo-recipe.json'})"
444
+ else:
445
+ error("No recipe in the store and no --deployment recipe to start from.")
446
+ raise typer.Exit(1)
447
+ if document is not None and document.trusted_assets is not None:
448
+ assets = json.loads(json.dumps(document.trusted_assets))
449
+ else:
450
+ assets = json.loads(env.trusted_assets_json)
451
+ return recipe, assets, recipe_source, generation, problem
452
+
453
+
454
+ IDENTITY_FIELDS = ("generation", "size_bytes", "crc32c_base64")
455
+ RECIPE_ASSET_KEYS = ("config", "board", "crown_observations_config", "crown_model_config")
456
+
457
+
458
+ def _recipe_refs(recipe: dict[str, Any]) -> list[dict[str, Any]]:
459
+ """Collect every artifact reference the recipe hands to a stage or the planner."""
460
+ refs = []
461
+ for stage, spec in sorted(recipe.get("preset", {}).items()):
462
+ for key in RECIPE_ASSET_KEYS:
463
+ if key in spec:
464
+ refs.append({"source": f"preset.{stage}.{key}", **spec[key]})
465
+ for index, checkpoint in enumerate(spec.get("checkpoints", [])):
466
+ refs.append({"source": f"preset.{stage}.checkpoints[{index}]", **checkpoint})
467
+ for key in ("keypoint_schema", "episode_selection"):
468
+ if recipe.get(key) is not None:
469
+ refs.append({"source": key, **recipe[key]})
470
+ return refs
471
+
472
+
473
+ def _live_identity(storage, asset: dict[str, Any]) -> dict[str, Any]:
474
+ """Read the pinned generation's metadata, and its bytes when small enough to hash."""
475
+ from egoexo.adapters.gcs import publication
476
+
477
+ try:
478
+ actual = {"present": True, **storage.metadata(asset["uri"], asset["generation"])}
479
+ except Exception as exc: # noqa: BLE001 - the storage SDK reports a miss as 404 or 412
480
+ if getattr(exc, "code", None) in {404, 412} or type(exc).__name__ in {
481
+ "NotFound",
482
+ "PreconditionFailed",
483
+ }:
484
+ return {"present": False}
485
+ raise
486
+ size = actual["size_bytes"]
487
+ if size is not None and size <= LOCAL_JSON_MAX_BYTES:
488
+ with tempfile.TemporaryDirectory() as scratch:
489
+ path = Path(scratch) / "asset"
490
+ storage.download(asset["uri"], asset["generation"], path, size)
491
+ _, sha, crc = publication.identity(path)
492
+ actual |= {"sha256": sha, "downloaded_crc32c_base64": crc}
493
+ return actual
494
+
495
+
496
+ def _asset_failures(asset: dict[str, Any], actual: dict[str, Any]) -> list[str]:
497
+ """Return every way the live object differs from the pinned identity."""
498
+ if not actual.get("present"):
499
+ return ["object_missing_at_pinned_generation"]
500
+ failures = [
501
+ f"{field}: pinned {asset[field]!r} live {actual[field]!r}"
502
+ for field in IDENTITY_FIELDS
503
+ if str(asset[field]) != str(actual[field])
504
+ ]
505
+ if "sha256" in actual and actual["sha256"] != asset["sha256"]:
506
+ failures.append(f"sha256: pinned {asset['sha256']} live {actual['sha256']}")
507
+ if "downloaded_crc32c_base64" in actual and (
508
+ actual["downloaded_crc32c_base64"] != actual["crc32c_base64"]
509
+ ):
510
+ failures.append("crc32c does not match the downloaded bytes")
511
+ return failures
512
+
513
+
514
+ @app.command("verify-assets")
515
+ def verify_assets(
516
+ uri: str | None = URI_OPTION,
517
+ deployment: Path | None = DEPLOYMENT_OPTION,
518
+ json_path: Path | None = typer.Option(
519
+ None, "--json", help="Also write the full report to this path."
520
+ ),
521
+ ) -> None:
522
+ """Check every recipe asset and trusted asset against the live object it pins.
523
+
524
+ Compares generation, size and CRC32C for each pinned object, and SHA256 for
525
+ objects up to 1 MiB, which are downloaded. Larger checkpoints are checked by
526
+ metadata only, the same check the API makes before planning. Reads only.
527
+ """
528
+ _runtime_config()
529
+ store = _store(uri)
530
+ recipe, assets, recipe_source, generation, problem = _effective_documents(store, deployment)
531
+ trusted = {asset["uri"]: asset for asset in assets}
532
+ problems = []
533
+ for stage, spec in sorted(recipe.get("preset", {}).items()):
534
+ config = spec.get("config")
535
+ if config is not None and spec["method"]["config_sha256"] != config["sha256"]:
536
+ problems.append(f"preset.{stage}: method.config_sha256 != config.sha256")
537
+ refs = _recipe_refs(recipe)
538
+ for ref in refs:
539
+ asset = trusted.get(ref["uri"])
540
+ if asset is None:
541
+ problems.append(f"{ref['source']}: {ref['uri']} is not a trusted asset")
542
+ continue
543
+ problems.extend(
544
+ f"{ref['source']}: {field} differs from the trusted asset entry"
545
+ for field in (*IDENTITY_FIELDS, "sha256")
546
+ if asset[field] != ref[field]
547
+ )
548
+ used = {ref["uri"] for ref in refs}
549
+ objects = {}
550
+ for asset in sorted(trusted.values(), key=lambda a: a["uri"]):
551
+ actual = _live_identity(store.storage, asset)
552
+ failures = _asset_failures(asset, actual)
553
+ objects[asset["uri"]] = {
554
+ "used_by_recipe": asset["uri"] in used,
555
+ "actual": actual,
556
+ "failures": failures,
557
+ }
558
+ problems.extend(f"{asset['uri']}: {failure}" for failure in failures)
559
+ report = {
560
+ "uri": store.uri,
561
+ "generation": None if generation == "0" else generation,
562
+ "current_problem": problem,
563
+ "recipe_source": recipe_source,
564
+ "recipe_references": len(refs),
565
+ "trusted_assets": len(assets),
566
+ "problems": problems,
567
+ "objects": objects,
568
+ }
569
+ if json_path is not None:
570
+ json_path.write_text(json.dumps(report, indent=2, sort_keys=True, default=str) + "\n")
571
+ _emit(report)
572
+ if problems:
573
+ raise typer.Exit(1)
574
+
575
+
428
576
  def _asset_group(uri: str) -> str | None:
429
577
  """Return the <group> segment of derived/runtime-assets/<group>/<sha256>/<name>."""
430
578
  marker = uri.find(ASSET_PREFIX)
@@ -461,21 +609,7 @@ def pin_config_asset(
461
609
 
462
610
  _load_json_file(file)
463
611
  store = _store(uri)
464
- document, generation, problem = _read_base(store)
465
- env, deployment_dir = _deployment_env(_deployment_dir(deployment))
466
- if document is not None and document.recipe is not None:
467
- recipe = json.loads(json.dumps(document.recipe))
468
- recipe_source = "store"
469
- elif env.recipe_json:
470
- recipe = json.loads(env.recipe_json)
471
- recipe_source = f"env ({deployment_dir / 'egoexo-recipe.json'})"
472
- else:
473
- error("No recipe in the store and no --deployment recipe to start from.")
474
- raise typer.Exit(1)
475
- if document is not None and document.trusted_assets is not None:
476
- assets = json.loads(json.dumps(document.trusted_assets))
477
- else:
478
- assets = json.loads(env.trusted_assets_json)
612
+ recipe, assets, recipe_source, generation, problem = _effective_documents(store, deployment)
479
613
  spec = recipe.get("preset", {}).get(stage)
480
614
  if not isinstance(spec, dict) or not isinstance(spec.get("method"), dict):
481
615
  error(f"Stage {stage!r} is not in the recipe preset.")
@@ -0,0 +1,215 @@
1
+ """Read-only EgoExo pipeline status: work items per capture and stage, plus Batch jobs.
2
+
3
+ The processing API key is read from Secret Manager in process and never printed.
4
+ Batch jobs are listed through the REST API with a server-side filter, because
5
+ ``gcloud batch jobs list`` filters on the client and takes minutes in this project.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import subprocess
12
+ import time
13
+ from collections import Counter, defaultdict
14
+ from datetime import UTC, datetime, timedelta
15
+ from pathlib import Path
16
+ from typing import Any
17
+
18
+ import httpx
19
+ import typer
20
+
21
+ from cli.config import resolve_api_url
22
+
23
+ PROJECT = "data-470400"
24
+ REGION = "us-central1"
25
+ BUCKET = "buildai-ego-exo"
26
+ ACTIVE_STATES = frozenset({"SCHEDULED", "QUEUED", "RUNNING"})
27
+ # The worker reads its assignment under the prefix of the image it runs in.
28
+ ASSIGNED_PREFIXES = ("EGOEXO_GPU", "EGOEXO_CPU")
29
+
30
+
31
+ def _gcloud(*args: str) -> str:
32
+ """Run one gcloud command and return its stdout. Tests replace this function."""
33
+ return subprocess.run(["gcloud", *args], capture_output=True, text=True, check=True).stdout
34
+
35
+
36
+ def _client() -> httpx.Client:
37
+ """HTTP client for the processing API and the Batch API. Tests replace this function."""
38
+ return httpx.Client(timeout=60.0)
39
+
40
+
41
+ def _api_key(project: str) -> str:
42
+ """Read the processing API key from Secret Manager."""
43
+ return _gcloud(
44
+ "secrets",
45
+ "versions",
46
+ "access",
47
+ "latest",
48
+ "--secret=PROCESSING_API_KEY",
49
+ f"--project={project}",
50
+ ).strip()
51
+
52
+
53
+ def _raw_capture_ids(bucket: str) -> list[str]:
54
+ """List capture UUIDs under raw/cn/<date>/."""
55
+ ids = []
56
+ for date in _gcloud("storage", "ls", f"gs://{bucket}/raw/cn/").split():
57
+ for capture in _gcloud("storage", "ls", date).split():
58
+ ids.append(capture.rstrip("/").rsplit("/", 1)[-1])
59
+ return sorted(ids)
60
+
61
+
62
+ def _get_json(client: httpx.Client, url: str, token: str, params: dict | None = None) -> Any:
63
+ """GET JSON with a bearer token. Server errors are retried a few times."""
64
+ for attempt in range(5):
65
+ response = client.get(url, params=params, headers={"Authorization": f"Bearer {token}"})
66
+ if response.status_code < 500 or attempt == 4:
67
+ response.raise_for_status()
68
+ return response.json()
69
+ time.sleep(2 * (attempt + 1))
70
+ raise AssertionError("unreachable")
71
+
72
+
73
+ def _capture_items(client: httpx.Client, api_url: str, key: str, capture_id: str) -> list[dict]:
74
+ """Page through one capture's work items."""
75
+ url = f"{api_url}/v1/processing/egoexo-captures/{capture_id}"
76
+ items: list[dict] = []
77
+ cursor = None
78
+ while True:
79
+ params = {"limit": 200} | ({"cursor": cursor} if cursor else {})
80
+ data = _get_json(client, url, key, params).get("data")
81
+ if not isinstance(data, dict):
82
+ return []
83
+ items.extend(data.get("items", []))
84
+ cursor = data.get("next_cursor")
85
+ if not data.get("truncated") or not cursor:
86
+ return items
87
+
88
+
89
+ def _stage_of(item: dict) -> str:
90
+ """Derive the stage name from the request when present, else the work key."""
91
+ request = (item.get("metadata") or {}).get("request") or {}
92
+ return request.get("stage") or str(item.get("work_key", "?")).split(":", 1)[0]
93
+
94
+
95
+ def _assigned_id(env: dict[str, str], suffix: str) -> str | None:
96
+ """Read the assigned work item or attempt id under the GPU or the CPU worker prefix."""
97
+ for prefix in ASSIGNED_PREFIXES:
98
+ value = env.get(f"{prefix}_ASSIGNED_{suffix}")
99
+ if value:
100
+ return value
101
+ return None
102
+
103
+
104
+ def _batch_jobs(client: httpx.Client, project: str, region: str, since_hours: int) -> list[dict]:
105
+ """List recent assigned Batch jobs. Only job metadata and the assignment ids are kept."""
106
+ token = _gcloud("auth", "print-access-token").strip()
107
+ since = (datetime.now(UTC) - timedelta(hours=since_hours)).strftime("%Y-%m-%dT%H:%M:%SZ")
108
+ url = f"https://batch.googleapis.com/v1/projects/{project}/locations/{region}/jobs"
109
+ query = {
110
+ "filter": f'labels.lane="egoexo-assigned" AND createTime>"{since}"',
111
+ "pageSize": "500",
112
+ }
113
+ rows: list[dict] = []
114
+ page_token = None
115
+ for _ in range(200):
116
+ params = query | ({"pageToken": page_token} if page_token else {})
117
+ payload = _get_json(client, url, token, params)
118
+ for job in payload.get("jobs", []):
119
+ env = job["taskGroups"][0]["taskSpec"].get("environment", {}).get("variables", {})
120
+ status = job.get("status", {})
121
+ rows.append(
122
+ {
123
+ "name": job["name"].rsplit("/", 1)[-1],
124
+ "state": status.get("state"),
125
+ "task_class": job.get("labels", {}).get("task-class"),
126
+ "created": job.get("createTime"),
127
+ "run_duration": status.get("runDuration"),
128
+ "work_item_id": _assigned_id(env, "WORK_ITEM_ID"),
129
+ "attempt_id": _assigned_id(env, "ATTEMPT_ID"),
130
+ "machine_type": job["allocationPolicy"]["instances"][0]["policy"].get(
131
+ "machineType"
132
+ ),
133
+ "events": [
134
+ {"time": e.get("eventTime"), "description": e.get("description")}
135
+ for e in status.get("statusEvents", [])
136
+ ],
137
+ }
138
+ )
139
+ page_token = payload.get("nextPageToken")
140
+ if not page_token:
141
+ break
142
+ return sorted(rows, key=lambda row: row["created"] or "")
143
+
144
+
145
+ def _capture_report(items: list[dict], *, include_items: bool) -> tuple[dict, str]:
146
+ """Summarize one capture's work items by stage and state. Returns (report, one line)."""
147
+ by_stage: dict[str, Counter] = defaultdict(Counter)
148
+ errors = []
149
+ for item in items:
150
+ by_stage[_stage_of(item)][item.get("state", "?")] += 1
151
+ if item.get("last_error"):
152
+ errors.append({"stage": _stage_of(item), "error": str(item["last_error"])[:200]})
153
+ report = {
154
+ "items": len(items),
155
+ "stages": {stage: dict(counts) for stage, counts in sorted(by_stage.items())},
156
+ "errors": errors[:20],
157
+ }
158
+ if include_items:
159
+ report["raw_items"] = items
160
+ summary = ", ".join(
161
+ f"{stage}:{'/'.join(f'{state}={n}' for state, n in sorted(counts.items()))}"
162
+ for stage, counts in sorted(by_stage.items())
163
+ )
164
+ return report, f"items={len(items)} {summary or '(none)'}"
165
+
166
+
167
+ def status(
168
+ captures: list[str] | None = typer.Option(
169
+ None,
170
+ "--capture",
171
+ help="Capture id. Repeat for several. Default: every capture under raw/cn/.",
172
+ ),
173
+ json_path: Path | None = typer.Option(
174
+ None, "--json", help="Write the full report to this path."
175
+ ),
176
+ items: bool = typer.Option(False, "--items", help="Include raw work items in the report."),
177
+ since_hours: int = typer.Option(24, "--since-hours", min=1, help="Batch job window."),
178
+ project: str = typer.Option(PROJECT, "--project", help="Google Cloud project."),
179
+ ) -> None:
180
+ """Show work items per capture and stage, and recent assigned Batch jobs. Reads only."""
181
+ key = _api_key(project)
182
+ api_url = resolve_api_url()
183
+ report: dict[str, Any] = {
184
+ "read_at": datetime.now(UTC).isoformat(),
185
+ "api_url": api_url,
186
+ "captures": {},
187
+ "batch_jobs": [],
188
+ }
189
+ with _client() as client:
190
+ for capture_id in captures or _raw_capture_ids(BUCKET):
191
+ try:
192
+ rows = _capture_items(client, api_url, key, capture_id)
193
+ except (httpx.HTTPError, ValueError) as exc:
194
+ report["captures"][capture_id] = {"error": str(exc)[:200]}
195
+ typer.echo(f"{capture_id} ERROR {str(exc)[:120]}")
196
+ continue
197
+ capture, line = _capture_report(rows, include_items=items)
198
+ report["captures"][capture_id] = capture
199
+ typer.echo(f"{capture_id} {line}")
200
+ for entry in capture["errors"][:5]:
201
+ typer.echo(f" ! {entry['stage']}: {entry['error']}")
202
+ try:
203
+ report["batch_jobs"] = _batch_jobs(client, project, REGION, since_hours)
204
+ except (httpx.HTTPError, subprocess.CalledProcessError, KeyError, ValueError) as exc:
205
+ report["batch_jobs"] = [{"error": str(exc)[:200]}]
206
+ jobs = Counter((job.get("task_class"), job.get("state")) for job in report["batch_jobs"])
207
+ typer.echo(f"batch jobs ({since_hours}h): {dict(jobs) or 'none'}")
208
+ for job in report["batch_jobs"]:
209
+ if job.get("state") in ACTIVE_STATES and job.get("task_class") == "gpu_inference":
210
+ last = job["events"][-1]["description"] if job["events"] else ""
211
+ typer.echo(
212
+ f" gpu {job['name']} {job['state']} item={job['work_item_id']} {last[:100]}"
213
+ )
214
+ if json_path is not None:
215
+ json_path.write_text(json.dumps(report, indent=2, default=str) + "\n")
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "buildai-cli"
7
- version = "0.3.151"
7
+ version = "0.3.153"
8
8
  description = "Build AI CLI (Typer)"
9
9
  requires-python = ">=3.11"
10
10
  dependencies = [
File without changes
File without changes
File without changes
File without changes