steve-cli 0.5.6__tar.gz → 1.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. {steve_cli-0.5.6 → steve_cli-1.0.1}/PKG-INFO +3 -1
  2. {steve_cli-0.5.6 → steve_cli-1.0.1}/README.md +2 -0
  3. {steve_cli-0.5.6 → steve_cli-1.0.1}/pyproject.toml +1 -1
  4. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/auth.py +10 -0
  5. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/cli.py +253 -2
  6. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/__init__.py +2 -0
  7. steve_cli-1.0.1/steve_cli/storage/branch.py +40 -0
  8. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/s3.py +15 -4
  9. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/trino.py +148 -27
  10. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli.egg-info/PKG-INFO +3 -1
  11. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli.egg-info/SOURCES.txt +1 -0
  12. {steve_cli-0.5.6 → steve_cli-1.0.1}/setup.cfg +0 -0
  13. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/__init__.py +0 -0
  14. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/decorators/__init__.py +0 -0
  15. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/decorators/lineage_job.py +0 -0
  16. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/__init__.py +0 -0
  17. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/adapters/__init__.py +0 -0
  18. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/adapters/composite.py +0 -0
  19. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/adapters/dataproductregistry.py +0 -0
  20. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/adapters/logging.py +0 -0
  21. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/adapters/null.py +0 -0
  22. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/adapters/openlineage.py +0 -0
  23. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/collector.py +0 -0
  24. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/port.py +0 -0
  25. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/registry.py +0 -0
  26. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/lineage/storage.py +0 -0
  27. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/ontology.py +0 -0
  28. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/policies/__init__.py +0 -0
  29. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/policies/client.py +0 -0
  30. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/__init__.py +0 -0
  31. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/extractors/__init__.py +0 -0
  32. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/extractors/csv.py +0 -0
  33. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/extractors/excel.py +0 -0
  34. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/extractors/generic.py +0 -0
  35. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/extractors/json.py +0 -0
  36. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/extractors/parquet.py +0 -0
  37. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/port.py +0 -0
  38. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/metadata/registry.py +0 -0
  39. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/parquet.py +0 -0
  40. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage/protocol.py +0 -0
  41. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/storage.py +0 -0
  42. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/validation/__init__.py +0 -0
  43. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/validation/adapters/__init__.py +0 -0
  44. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/validation/adapters/great_expectations.py +0 -0
  45. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/validation/adapters/null.py +0 -0
  46. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/validation/adapters/validoopsie.py +0 -0
  47. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/validation/port.py +0 -0
  48. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/validation/registry.py +0 -0
  49. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli/vkg.py +0 -0
  50. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli.egg-info/dependency_links.txt +0 -0
  51. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli.egg-info/entry_points.txt +0 -0
  52. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli.egg-info/requires.txt +0 -0
  53. {steve_cli-0.5.6 → steve_cli-1.0.1}/steve_cli.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steve-cli
3
- Version: 0.5.6
3
+ Version: 1.0.1
4
4
  Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
5
5
  Author: Frank
6
6
  License: MIT
@@ -67,6 +67,8 @@ Requires-Dist: visidata>=3.0; extra == "all"
67
67
 
68
68
  A CLI tool for the data platform. Commands are organized by concern: general workspace automation, data lakehouse operations, semantic knowledge graphs, and governance.
69
69
 
70
+ > **Upgrading from 0.x?** See [CHANGELOG.md](./CHANGELOG.md) — v1.0.0 introduces branch-isolated storage with breaking path changes that require a one-time data migration.
71
+
70
72
  ## Installation
71
73
 
72
74
  ```bash
@@ -2,6 +2,8 @@
2
2
 
3
3
  A CLI tool for the data platform. Commands are organized by concern: general workspace automation, data lakehouse operations, semantic knowledge graphs, and governance.
4
4
 
5
+ > **Upgrading from 0.x?** See [CHANGELOG.md](./CHANGELOG.md) — v1.0.0 introduces branch-isolated storage with breaking path changes that require a one-time data migration.
6
+
5
7
  ## Installation
6
8
 
7
9
  ```bash
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
5
5
 
6
6
  [project]
7
7
  name = "steve-cli"
8
- version = "0.5.6"
8
+ version = "1.0.1"
9
9
  description = "A simple CLI tool to run jobs from jobs.yaml with proper environment setup"
10
10
  readme = "README.md"
11
11
  license = {text = "MIT"}
@@ -21,6 +21,8 @@ def save_credentials(
21
21
  token: str,
22
22
  base_url: Optional[str] = None,
23
23
  workspace_name: Optional[str] = None,
24
+ expires_at: Optional[str] = None,
25
+ s3_endpoint: Optional[str] = None,
24
26
  ) -> None:
25
27
  _CREDENTIALS_FILE.parent.mkdir(parents=True, exist_ok=True)
26
28
  creds = load_credentials()
@@ -30,6 +32,10 @@ def save_credentials(
30
32
  entry["base_url"] = base_url.rstrip("/")
31
33
  if workspace_name:
32
34
  entry["workspace_name"] = workspace_name
35
+ if expires_at:
36
+ entry["expires_at"] = expires_at
37
+ if s3_endpoint:
38
+ entry["s3_endpoint"] = s3_endpoint
33
39
  workspaces[workspace_id] = entry
34
40
  creds["workspaces"] = workspaces
35
41
  _CREDENTIALS_FILE.write_text(json.dumps(creds, indent=2))
@@ -63,6 +69,10 @@ def get_token(workspace_id: Optional[str] = None) -> Optional[str]:
63
69
  return _get_workspace_entry(workspace_id).get("token")
64
70
 
65
71
 
72
+ def get_s3_endpoint(workspace_id: Optional[str] = None) -> Optional[str]:
73
+ return _get_workspace_entry(workspace_id).get("s3_endpoint")
74
+
75
+
66
76
  def get_service_url(service: str, workspace_id: Optional[str] = None) -> Optional[str]:
67
77
  entry = _get_workspace_entry(workspace_id)
68
78
  base_url = entry.get("base_url")
@@ -963,9 +963,10 @@ def policies_apply(file: Path | None):
963
963
  @click.option("--workspace", "workspace_id", default=None, help="Workspace ID (defaults to WORKSPACE_ID from .env)")
964
964
  @click.option("--url", "base_url", default=None, help="Platform root host, e.g. jds-dev.internal.jambit.io")
965
965
  @click.option("--workspace-name", "workspace_name_opt", default=None, help="Workspace name")
966
+ @click.option("--expires", "expires_at", default=None, help="Token expiry date/time (e.g. 2026-12-31 or ISO 8601). Informational only.")
966
967
  @click.option('--env-file', '-e', type=click.Path(path_type=Path), multiple=True,
967
968
  help='Path to .env file(s). Defaults to .env and .workspaces.env')
968
- def login(token: str | None, workspace_id: str | None, base_url: str | None, workspace_name_opt: str | None, env_file: tuple):
969
+ def login(token: str | None, workspace_id: str | None, base_url: str | None, workspace_name_opt: str | None, expires_at: str | None, env_file: tuple):
969
970
  """Print CLI access URLs or save a service principal token."""
970
971
  cwd = Path.cwd()
971
972
  env_files = [Path(f) for f in env_file] if env_file else [cwd / ".env", cwd / ".workspaces.env"]
@@ -984,9 +985,27 @@ def login(token: str | None, workspace_id: str | None, base_url: str | None, wor
984
985
  click.secho("Could not resolve workspace ID. Set WORKSPACE_ID in .env or pass --workspace.", fg="red", err=True)
985
986
  raise SystemExit(1)
986
987
  click.echo(f"\nYou are about to log in to workspace {click.style(workspace_name, fg='cyan', bold=True)} ({resolved_workspace})")
987
- save_credentials(resolved_workspace, token, base_url=base_url, workspace_name=workspace_name)
988
+ save_credentials(resolved_workspace, token, base_url=base_url, workspace_name=workspace_name, expires_at=expires_at)
988
989
  if base_url:
989
990
  click.echo(f"Platform: {base_url}")
991
+
992
+ s3_endpoint: str | None = None
993
+ if base_url:
994
+ try:
995
+ import urllib.request as _urlreq
996
+ config_url = f"https://datameshx.{base_url}/api/v1/cli/config"
997
+ req = _urlreq.Request(config_url, headers={"Authorization": f"Bearer {token}"})
998
+ with _urlreq.urlopen(req, timeout=5) as _r:
999
+ import json as _json
1000
+ cfg = _json.loads(_r.read().decode())
1001
+ s3_endpoint = cfg.get("s3Endpoint")
1002
+ except Exception as _e:
1003
+ click.secho(f"Could not fetch platform config (s3Endpoint will use .env fallback): {_e}", fg="yellow", err=True)
1004
+
1005
+ if s3_endpoint:
1006
+ save_credentials(resolved_workspace, token, base_url=base_url, workspace_name=workspace_name, expires_at=expires_at, s3_endpoint=s3_endpoint)
1007
+ click.echo(f"S3 endpoint: {s3_endpoint}")
1008
+
990
1009
  click.secho(f"Logged in. Credentials saved to ~/.steve/credentials.json", fg="green")
991
1010
  return
992
1011
 
@@ -1022,6 +1041,238 @@ def logout(workspace_id: str | None, env_file: tuple):
1022
1041
  click.secho("All credentials removed.", fg="green")
1023
1042
 
1024
1043
 
1044
+ @main.command("status")
1045
+ @click.option('--env-file', '-e', type=click.Path(path_type=Path), multiple=True,
1046
+ help='Path to .env file(s). Defaults to .env and .workspaces.env')
1047
+ @click.option('--no-check', is_flag=True, default=False, help='Skip connectivity checks.')
1048
+ def status(env_file: tuple, no_check: bool):
1049
+ """Show current steve-cli environment, storage paths, and connectivity."""
1050
+ import importlib.metadata
1051
+ import urllib.request
1052
+
1053
+ cwd = Path.cwd()
1054
+ env_files = [Path(f) for f in env_file] if env_file else [cwd / ".env", cwd / ".workspaces.env"]
1055
+ for ef in env_files:
1056
+ load_dotenv(ef)
1057
+
1058
+ try:
1059
+ version = importlib.metadata.version("steve-cli")
1060
+ except Exception:
1061
+ version = "unknown"
1062
+
1063
+ def _section(title: str) -> None:
1064
+ click.echo(f"\n{click.style(f'━━ {title} ', fg='bright_black')}{'━' * max(0, 48 - len(title))}")
1065
+
1066
+ def _row(label: str, value: str, value_color: str = "white") -> None:
1067
+ click.echo(f" {click.style(label.ljust(14), fg='bright_black')} {click.style(value, fg=value_color)}")
1068
+
1069
+ def _check(label: str, url: str, timeout: int = 2, token: str = "", suffix: str = "") -> None:
1070
+ if no_check:
1071
+ _row(label, "(skipped)", "bright_black")
1072
+ return
1073
+ try:
1074
+ req = urllib.request.Request(url)
1075
+ if token:
1076
+ req.add_header("Authorization", f"Bearer {token}")
1077
+ with urllib.request.urlopen(req, timeout=timeout) as r:
1078
+ _row(label, f"✓ reachable ({url}){suffix}", "green")
1079
+ return r.read()
1080
+ except urllib.error.HTTPError:
1081
+ _row(label, f"✓ reachable ({url}){suffix}", "green")
1082
+ except Exception as e:
1083
+ _row(label, f"✗ unreachable ({url}){suffix}: {e}", "red")
1084
+
1085
+ click.echo(f"\n{click.style('●', fg='green')} Steve CLI {click.style('v' + version, fg='cyan', bold=True)}")
1086
+
1087
+ from steve_cli.storage.branch import get_branch_prefix
1088
+ import subprocess as _sp
1089
+
1090
+ branch = get_branch_prefix()
1091
+ steve_branch_env = os.getenv("STEVE_BRANCH", "").strip()
1092
+ gh_head = os.getenv("GITHUB_HEAD_REF", "").strip() or os.getenv("GITHUB_REF_NAME", "").strip()
1093
+ if steve_branch_env:
1094
+ branch_source = f"STEVE_BRANCH={steve_branch_env!r}"
1095
+ elif gh_head:
1096
+ branch_source = f"GITHUB env ({gh_head})"
1097
+ else:
1098
+ try:
1099
+ r = _sp.run(["git", "rev-parse", "--abbrev-ref", "HEAD"], capture_output=True, text=True, timeout=3)
1100
+ branch_source = "git" if r.returncode == 0 and r.stdout.strip() else "fallback"
1101
+ except Exception:
1102
+ branch_source = "fallback"
1103
+
1104
+ workspace_id = os.getenv("WORKSPACE_ID", "")
1105
+ workspace_name = os.getenv("WORKSPACE_NAME", "")
1106
+ session = os.getenv("SESSION_NAME", "") or socket.gethostname()
1107
+
1108
+ from steve_cli.auth import _get_workspace_entry, get_service_url
1109
+ import datetime as _dt
1110
+ entry = _get_workspace_entry(workspace_id or None)
1111
+ token = entry.get("token", "")
1112
+ expires_at = entry.get("expires_at", "")
1113
+ if token:
1114
+ expiry_warn = False
1115
+ expiry_str = ", expires never" if not expires_at else ""
1116
+ if expires_at:
1117
+ try:
1118
+ exp = _dt.datetime.fromisoformat(expires_at.replace("Z", "+00:00"))
1119
+ now = _dt.datetime.now(_dt.timezone.utc) if exp.tzinfo else _dt.datetime.now()
1120
+ delta = exp - now
1121
+ total_seconds = int(delta.total_seconds())
1122
+ if total_seconds < 0:
1123
+ s = abs(total_seconds)
1124
+ if s < 3600:
1125
+ ago = f"{s // 60}m ago"
1126
+ elif s < 86400:
1127
+ ago = f"{s // 3600}h ago"
1128
+ else:
1129
+ ago = f"{s // 86400}d ago"
1130
+ expiry_str = f", expired {ago}"
1131
+ expiry_warn = True
1132
+ elif total_seconds < 3600:
1133
+ expiry_str = f", expires in {total_seconds // 60}m ⚠"
1134
+ expiry_warn = True
1135
+ elif total_seconds < 86400:
1136
+ expiry_str = f", expires in {total_seconds // 3600}h ⚠"
1137
+ expiry_warn = True
1138
+ elif delta.days <= 7:
1139
+ expiry_str = f", expires in {delta.days}d ⚠"
1140
+ expiry_warn = True
1141
+ else:
1142
+ expiry_str = f", expires {exp.strftime('%Y-%m-%d %H:%M')}"
1143
+ except Exception:
1144
+ expiry_str = f", expires {expires_at}"
1145
+ token_display = f"✓ logged in (stp_***{token[-4:]}{expiry_str})"
1146
+ token_color = "yellow" if expiry_warn else "green"
1147
+ else:
1148
+ token_display = "✗ not logged in — run: steve login"
1149
+ token_color = "yellow"
1150
+
1151
+ _section("Identity")
1152
+ _row("Branch", f"{branch} ({branch_source})", "cyan")
1153
+ _row("Workspace ID", workspace_id or "(not set)", "white" if workspace_id else "yellow")
1154
+ _row("Workspace", workspace_name or "(not set)", "white" if workspace_name else "yellow")
1155
+ _row("Session", session, "white")
1156
+ if token and not no_check:
1157
+ _base_url = entry.get("base_url", "")
1158
+ if _base_url:
1159
+ try:
1160
+ _req = urllib.request.Request(
1161
+ f"https://datameshx.{_base_url}/api/v1/cli/config",
1162
+ headers={"Authorization": f"Bearer {token}"},
1163
+ )
1164
+ with urllib.request.urlopen(_req, timeout=4) as _r:
1165
+ _r.read()
1166
+ _row("Auth", f"✓ token valid (stp_***{token[-4:]}{expiry_str})", "green" if not expiry_warn else "yellow")
1167
+ except urllib.error.HTTPError as _e:
1168
+ if _e.code == 401:
1169
+ _row("Auth", f"✗ token invalid/revoked (stp_***{token[-4:]}{expiry_str})", "red")
1170
+ else:
1171
+ _row("Auth", f"✓ token valid (stp_***{token[-4:]}{expiry_str})", "green" if not expiry_warn else "yellow")
1172
+ except Exception as _e:
1173
+ _row("Auth", f"{token_display} (validation failed: {_e})", token_color)
1174
+ else:
1175
+ _row("Auth", token_display, token_color)
1176
+ else:
1177
+ _row("Auth", token_display, token_color)
1178
+
1179
+ tiers = ["bronze", "silver", "gold"]
1180
+ s3_rows = []
1181
+ for tier in tiers:
1182
+ tier_up = tier.upper()
1183
+ bucket = os.getenv(f"{tier_up}_BUCKET", "")
1184
+ if bucket:
1185
+ s3_rows.append((f"S3 {tier}", f"s3://{bucket}/_store/{branch}/"))
1186
+ for ws in _detect_workspaces():
1187
+ for tier in tiers:
1188
+ bucket = os.getenv(f"{ws}_BUCKET_{tier.upper()}", "")
1189
+ if bucket:
1190
+ s3_rows.append((f"S3 {ws.lower()}/{tier}", f"s3://{bucket}/_store/{branch}/"))
1191
+
1192
+ _section("Storage Paths")
1193
+ if s3_rows:
1194
+ for label, path in s3_rows:
1195
+ _row(label, path, "cyan")
1196
+ else:
1197
+ _row("S3", "(no bucket env vars found)", "yellow")
1198
+ _row("Iceberg base", f"_iceberg/{branch}/", "cyan")
1199
+
1200
+ trino_endpoint = get_service_url("trino", workspace_id or None)
1201
+ resolved_workspace = workspace_name or workspace_id or ""
1202
+ trino_catalog = os.getenv("TRINO_CATALOG", "minio")
1203
+ trino_schema_explicit = os.getenv("TRINO_SCHEMA", "")
1204
+
1205
+ _section("Trino")
1206
+ if trino_endpoint:
1207
+ _row("Endpoint", trino_endpoint, "cyan")
1208
+ else:
1209
+ _row("Endpoint", "✗ not configured (login required)", "yellow")
1210
+ _row("Catalog", trino_catalog, "white")
1211
+ for tier in tiers:
1212
+ if trino_schema_explicit:
1213
+ schema = trino_schema_explicit
1214
+ elif resolved_workspace:
1215
+ ws_clean = resolved_workspace.lower().replace("-", "_")
1216
+ schema = f"{ws_clean}_{tier}__{branch}"
1217
+ else:
1218
+ schema = f"(workspace not set — set WORKSPACE_NAME)"
1219
+ _row(f"Schema {tier}", schema, "cyan" if resolved_workspace or trino_schema_explicit else "yellow")
1220
+ if trino_schema_explicit:
1221
+ break
1222
+
1223
+ lakekeeper = os.getenv("LAKEKEEPER_ENDPOINT", "") or get_service_url("lakekeeper", workspace_id or None)
1224
+
1225
+ _section("Connectivity")
1226
+ _cred_s3 = entry.get("s3_endpoint", "")
1227
+ _env_s3 = os.getenv("S3_ENDPOINT", "")
1228
+ if _cred_s3:
1229
+ s3_endpoint = _cred_s3
1230
+ s3_source = "credentials"
1231
+ elif _env_s3:
1232
+ s3_endpoint = _env_s3
1233
+ s3_source = "S3_ENDPOINT env"
1234
+ else:
1235
+ s3_endpoint = "http://localhost:9000"
1236
+ s3_source = "default"
1237
+ _check("S3 endpoint", s3_endpoint, suffix=f" (from {s3_source})")
1238
+ if trino_endpoint:
1239
+ if no_check:
1240
+ _row("Trino", "(skipped)", "bright_black")
1241
+ else:
1242
+ try:
1243
+ import json as _json
1244
+ req = urllib.request.Request(f"{trino_endpoint}/v1/info")
1245
+ if token:
1246
+ req.add_header("Authorization", f"Bearer {token}")
1247
+ req.add_header("X-Trino-User", os.getenv("TRINO_USER", "admin"))
1248
+ with urllib.request.urlopen(req, timeout=3) as r:
1249
+ info = _json.loads(r.read())
1250
+ version_str = info.get("nodeVersion", {}).get("version", "?")
1251
+ state = info.get("state", "?")
1252
+ uptime = info.get("uptime", "")
1253
+ detail = f"v{version_str} {state.lower()}" + (f", up {uptime}" if uptime else "")
1254
+ _row("Trino", f"✓ reachable ({detail}) {trino_endpoint}", "green")
1255
+ except Exception as e:
1256
+ _row("Trino", f"✗ unreachable: {e}", "red")
1257
+ else:
1258
+ _row("Trino", "✗ not configured", "yellow")
1259
+ if lakekeeper:
1260
+ _check("Lakekeeper", f"{lakekeeper}/catalog/v1/config")
1261
+ else:
1262
+ _row("Lakekeeper", "(not configured — set LAKEKEEPER_ENDPOINT or login with --url)", "bright_black")
1263
+
1264
+ _section("Environment")
1265
+ for ef in env_files:
1266
+ state = "✓ found" if ef.exists() else "✗ missing"
1267
+ color = "green" if ef.exists() else "bright_black"
1268
+ _row(ef.name, state, color)
1269
+ _row("STEVE_BRANCH", repr(steve_branch_env) if steve_branch_env else "(not set — auto-detected)", "white" if steve_branch_env else "bright_black")
1270
+ detected_ws = _detect_workspaces()
1271
+ if detected_ws:
1272
+ _row("Upstream ws", ", ".join(detected_ws), "white")
1273
+ click.echo()
1274
+
1275
+
1025
1276
  @main.command("upgrade")
1026
1277
  def upgrade():
1027
1278
  """Upgrade steve-cli to the latest version."""
@@ -1,3 +1,4 @@
1
+ from .branch import get_branch_prefix
1
2
  from .protocol import Storage
2
3
  from .s3 import S3Storage
3
4
  from .trino import TrinoStorage
@@ -5,6 +6,7 @@ from .parquet import ParquetMetadata, extract_parquet_metadata
5
6
  from .metadata import FileMetadata, ColumnMetadata, MetadataExtractorPort, MetadataRegistry
6
7
 
7
8
  __all__ = [
9
+ "get_branch_prefix",
8
10
  "Storage",
9
11
  "S3Storage",
10
12
  "TrinoStorage",
@@ -0,0 +1,40 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import os
5
+ import re
6
+ import subprocess
7
+
8
+ logger = logging.getLogger(__name__)
9
+
10
+ _MAIN_BRANCHES = {"main", "master"}
11
+
12
+
13
+ def _sanitize(branch: str) -> str:
14
+ branch = branch.lower().strip()
15
+ branch = re.sub(r"[^a-z0-9]+", "_", branch)
16
+ return branch.strip("_") or "main"
17
+
18
+
19
+ def get_branch_prefix() -> str:
20
+ for env_var in ("STEVE_BRANCH", "GITHUB_HEAD_REF", "GITHUB_REF_NAME"):
21
+ value = os.getenv(env_var, "").strip()
22
+ if value:
23
+ return _sanitize(value)
24
+
25
+ try:
26
+ result = subprocess.run(
27
+ ["git", "rev-parse", "--abbrev-ref", "HEAD"],
28
+ capture_output=True,
29
+ text=True,
30
+ timeout=3,
31
+ )
32
+ if result.returncode == 0:
33
+ branch = result.stdout.strip()
34
+ if branch and branch != "HEAD":
35
+ return _sanitize(branch)
36
+ except Exception as exc:
37
+ logger.debug("git branch detection failed: %s", exc)
38
+
39
+ logger.warning("Could not detect git branch; defaulting to 'main'")
40
+ return "main"
@@ -8,16 +8,27 @@ import boto3
8
8
  from botocore.client import Config
9
9
  from botocore.exceptions import ClientError, EndpointConnectionError
10
10
 
11
+ from .branch import get_branch_prefix
12
+ from steve_cli.auth import get_s3_endpoint
13
+
11
14
  logger = logging.getLogger(__name__)
12
15
 
13
16
 
14
17
  class S3Storage:
15
18
  def __init__(self, tier: str = "bronze", workspace: str | None = None):
16
19
  self.tier = tier.lower()
20
+ self._branch = get_branch_prefix()
17
21
  self.endpoint, self.bucket, access_key, secret_key, required_vars = (
18
22
  self._load_config(self.tier, workspace)
19
23
  )
20
24
 
25
+ cred_endpoint = get_s3_endpoint()
26
+ if cred_endpoint:
27
+ self.endpoint = cred_endpoint
28
+ self._endpoint_source = "credentials"
29
+ else:
30
+ self._endpoint_source = "env" if os.getenv("S3_ENDPOINT") else "default"
31
+
21
32
  missing = [v for v in required_vars if not os.getenv(v)]
22
33
  if missing:
23
34
  raise OSError(
@@ -66,11 +77,11 @@ class S3Storage:
66
77
  ]
67
78
  return endpoint, bucket, access_key, secret_key, required
68
79
 
69
- @staticmethod
70
- def _key(path: str) -> str:
80
+ def _key(self, path: str) -> str:
71
81
  if not path:
72
- return ""
73
- return str(PurePosixPath(path).as_posix().lstrip("/"))
82
+ return f"_store/{self._branch}/"
83
+ clean = str(PurePosixPath(path).as_posix().lstrip("/"))
84
+ return f"_store/{self._branch}/{clean}"
74
85
 
75
86
  def _object_location(self, path: str) -> str:
76
87
  return f"{self.endpoint}/{self.bucket}/{self._key(path)}"
@@ -10,19 +10,21 @@ import pyarrow as pa
10
10
  import pyarrow.parquet as pq
11
11
  import requests
12
12
 
13
+ from .branch import get_branch_prefix
14
+
13
15
  logger = logging.getLogger(__name__)
14
16
 
15
17
 
16
- def _derive_schema(tier: str, workspace: str | None) -> str | None:
17
- if workspace:
18
- prefix = workspace.lower().replace("-", "_")
19
- return f"ws_{prefix}_{tier.lower()}"
20
- return None
18
+ def _derive_schema(tier: str, workspace: str | None, branch: str) -> str:
19
+ base = f"{workspace.lower().replace('-', '_')}_{tier.lower()}" if workspace else tier.lower()
20
+ return f"{base}__{branch}"
21
21
 
22
22
 
23
23
  class TrinoStorage:
24
24
  def __init__(self, tier: str = "bronze", workspace: str | None = None):
25
25
  from steve_cli.auth import get_service_url
26
+ self._branch = get_branch_prefix()
27
+ self._tier = tier.lower()
26
28
  self._workspace_id = os.getenv("WORKSPACE_ID")
27
29
  trino_endpoint = get_service_url("trino", self._workspace_id)
28
30
  if not trino_endpoint:
@@ -36,16 +38,25 @@ class TrinoStorage:
36
38
  ]
37
39
  detail = "\n".join(f" {mark} {desc}" for mark, desc in checks)
38
40
  raise OSError(f"Trino endpoint not found:\n{detail}")
39
- self.catalog = os.getenv("TRINO_CATALOG", "minio")
40
41
  resolved_workspace = workspace or os.getenv("WORKSPACE_NAME")
41
- self.schema = _derive_schema(tier, resolved_workspace) or os.environ.get("TRINO_SCHEMA", "")
42
- if not self.schema:
43
- raise OSError("TRINO_SCHEMA is not set and no WORKSPACE_NAME or workspace was provided")
44
- self.user = os.getenv("TRINO_USER", "admin")
42
+ default_catalog = f"{resolved_workspace}-{tier}" if resolved_workspace else "minio"
43
+ self.catalog = os.getenv("TRINO_CATALOG", default_catalog)
44
+ self.schema = os.environ.get("TRINO_SCHEMA") or _derive_schema(tier, resolved_workspace, self._branch)
45
+ self.user = os.getenv("TRINO_USER") or resolved_workspace or "admin"
45
46
  self._base = trino_endpoint.rstrip("/")
46
- self._lakekeeper_endpoint = os.getenv("LAKEKEEPER_ENDPOINT")
47
- default_warehouse = f"{resolved_workspace}-{tier}" if resolved_workspace else "minio"
48
- self._lakekeeper_warehouse = os.getenv("LAKEKEEPER_WAREHOUSE", default_warehouse)
47
+ self._lakekeeper_endpoint = os.getenv("LAKEKEEPER_ENDPOINT") or get_service_url("lakekeeper", self._workspace_id)
48
+ self._lakekeeper_warehouse = os.getenv("LAKEKEEPER_WAREHOUSE", default_catalog)
49
+ _iceberg_bucket = os.getenv(f"{tier.upper()}_BUCKET", "")
50
+ if _iceberg_bucket:
51
+ self._iceberg_location_prefix = f"s3://{_iceberg_bucket}/_iceberg/{self._branch}"
52
+ else:
53
+ raise OSError(
54
+ f"TrinoStorage: missing bucket configuration for tier '{tier}' — "
55
+ f"{tier.upper()}_BUCKET is not set.\n"
56
+ f" Required env vars: {tier.upper()}_BUCKET, {tier.upper()}_ACCESS_KEY, {tier.upper()}_SECRET_KEY\n"
57
+ f" This workspace may only have read access to the '{tier}' tier. "
58
+ f"Write access requires provisioning credentials (run 'uv run steve setup env' or check workspace provisioning)."
59
+ )
49
60
  self.__iceberg_catalog = None
50
61
 
51
62
  @property
@@ -55,24 +66,44 @@ class TrinoStorage:
55
66
  from pyiceberg.io import load_file_io
56
67
  from pyiceberg.table import Table
57
68
 
58
- # FIXME: Interim workaround — pyiceberg cannot use Lakekeeper's remote signing
59
- # endpoint when running outside the cluster (hostname 'lakekeeper' doesn't resolve
60
- # locally). We subclass RestCatalog to inject our local signer URI after pyiceberg
61
- # merges table config (which overwrites s3.signer.uri with the internal hostname).
62
- # Remove once Increment 5 (STS credential vending) is wired.
63
- tier = self.schema.rsplit("_", 1)[-1].upper()
64
- lakekeeper_base = self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog"
69
+ # WORKAROUND: Lakekeeper's remote S3 presigner is bypassed for external access.
70
+ #
71
+ # Lakekeeper returns `s3.signer` / `s3.signer.uri` in the table config, which
72
+ # pyiceberg uses to sign S3 requests via Lakekeeper's signing endpoint. This works
73
+ # inside the cluster but fails from a laptop because:
74
+ # 1. The signed URI Lakekeeper produces contains the cluster-internal hostname
75
+ # (e.g. http://minio:9000) which is unreachable externally.
76
+ # 2. Even after overriding the signer URI to the public lakekeeper subdomain,
77
+ # Lakekeeper's signer returns HTTP 500 for URIs with the public minio hostname
78
+ # because it only knows the internal minio endpoint.
79
+ #
80
+ # Current workaround: strip all signer config from the merged table properties and
81
+ # use direct S3 credentials (access-key + secret-key) with the public minio endpoint
82
+ # from ~/.steve/credentials.json. This bypasses Lakekeeper's access control for S3
83
+ # writes — credentials live on disk rather than being vended by the cluster.
84
+ #
85
+ # To revert: configure Lakekeeper with its public S3 endpoint so presigned URIs
86
+ # contain a resolvable hostname, then remove the merged.pop() calls below and
87
+ # restore `s3.signer.uri: lakekeeper_base` in s3_overrides.
88
+ tier = self._tier.upper()
89
+ _lk = self._lakekeeper_endpoint or "http://lakekeeper:8181"
90
+ lakekeeper_base = _lk.rstrip("/") + "/catalog"
91
+ from steve_cli.auth import get_s3_endpoint as _get_s3_ep
92
+ _s3_ep = _get_s3_ep(self._workspace_id) or os.getenv("S3_ENDPOINT", "http://minio:9000")
65
93
  s3_overrides = {
66
- "s3.endpoint": os.getenv("S3_ENDPOINT", "http://minio:9000"),
94
+ "s3.endpoint": _s3_ep,
67
95
  "s3.access-key-id": os.getenv(f"{tier}_ACCESS_KEY") or os.getenv("AWS_ACCESS_KEY_ID", ""),
68
96
  "s3.secret-access-key": os.getenv(f"{tier}_SECRET_KEY") or os.getenv("AWS_SECRET_ACCESS_KEY", ""),
69
97
  "s3.path-style-access": "true",
70
- "s3.signer.uri": lakekeeper_base,
71
98
  } if self._lakekeeper_endpoint else {}
72
99
 
73
100
  class _PatchedRestCatalog(RestCatalog):
74
101
  def _response_to_table(self_, identifier_tuple, table_response):
75
102
  merged = {**table_response.metadata.properties, **table_response.config, **s3_overrides, "uri": lakekeeper_base}
103
+ # Strip remote-signer config — see WORKAROUND comment above.
104
+ merged.pop("s3.signer", None)
105
+ merged.pop("s3.signer.uri", None)
106
+ merged.pop("s3.signer.endpoint", None)
76
107
  return Table(
77
108
  identifier=identifier_tuple,
78
109
  metadata_location=table_response.metadata_location,
@@ -82,12 +113,16 @@ class TrinoStorage:
82
113
  config=table_response.config,
83
114
  )
84
115
 
116
+ from steve_cli.auth import get_token
117
+ stp_token = get_token(self._workspace_id)
118
+ token_kwargs = {"token": stp_token} if stp_token else {}
85
119
  try:
86
120
  catalog = _PatchedRestCatalog(
87
121
  "lakekeeper",
88
122
  uri=lakekeeper_base,
89
123
  warehouse=self._lakekeeper_warehouse,
90
124
  **s3_overrides,
125
+ **token_kwargs,
91
126
  )
92
127
  except Exception as exc:
93
128
  msg = str(exc)
@@ -172,7 +207,7 @@ class TrinoStorage:
172
207
  return path.strip("/").replace("/", "_").replace(".", "_")
173
208
 
174
209
  def list(self, prefix: str = "") -> list[str]:
175
- rows = self._execute(f"SHOW TABLES FROM {self.catalog}.{self.schema}")
210
+ rows = self._execute(f'SHOW TABLES FROM "{self.catalog}".{self.schema}')
176
211
  tables = [r["Table"] for r in rows]
177
212
  if prefix:
178
213
  clean = prefix.strip("/")
@@ -183,7 +218,10 @@ class TrinoStorage:
183
218
  return self.list()
184
219
 
185
220
  def get_bytes(self, path: str) -> bytes:
186
- table_ref = f"{self.catalog}.{self.schema}.{self._table_name(path)}"
221
+ table_name = self._table_name(path)
222
+ table_ref = f'"{self.catalog}".{self.schema}.{table_name}'
223
+ if os.getenv("STEVE_TRINO_TRACE") == "1":
224
+ logger.warning("[trino] get_bytes: %s (warehouse=%s)", table_ref, self._lakekeeper_warehouse)
187
225
  try:
188
226
  rows = self._execute(f"SELECT * FROM {table_ref}")
189
227
  except OSError:
@@ -193,7 +231,39 @@ class TrinoStorage:
193
231
  f"TrinoStorage: read from '{table_ref}' failed — {type(exc).__name__}: {exc}"
194
232
  ) from exc
195
233
  if not rows:
196
- return b""
234
+ list_err: str | None = None
235
+ table_names: list[str] | None = None
236
+ try:
237
+ existing = self._execute(f'SHOW TABLES FROM "{self.catalog}".{self.schema}')
238
+ table_names = [r.get("Table", "") for r in existing]
239
+ except Exception as list_exc:
240
+ list_err = str(list_exc)
241
+
242
+ if table_names is not None and table_name in table_names:
243
+ raise OSError(
244
+ f"TrinoStorage: table '{table_ref}' exists in Trino but returned 0 rows — "
245
+ f"data may not have been written yet, or the last write produced an empty result."
246
+ )
247
+ elif list_err is not None:
248
+ raise OSError(
249
+ f"TrinoStorage: table '{table_name}' not found in catalog='{self.catalog}' schema='{self.schema}'.\n"
250
+ f" Looked in: {table_ref}\n"
251
+ f" Could not list tables in that schema: {list_err}\n"
252
+ f" Hint: if data was written with get_storage() (S3/Parquet), read it back with get_storage(), not get_tables()."
253
+ )
254
+ elif table_names == []:
255
+ raise OSError(
256
+ f"TrinoStorage: table '{table_name}' not found — schema '{self.catalog}'.'{self.schema}' exists but contains no tables.\n"
257
+ f" Looked in: {table_ref}\n"
258
+ f" Hint: if data was written with get_storage() (S3/Parquet), read it back with get_storage(), not get_tables()."
259
+ )
260
+ else:
261
+ raise OSError(
262
+ f"TrinoStorage: table '{table_name}' not found in catalog='{self.catalog}' schema='{self.schema}'.\n"
263
+ f" Looked in: {table_ref}\n"
264
+ f" Tables found in that schema: {', '.join(table_names or [])}\n"
265
+ f" Hint: if data was written with get_storage() (S3/Parquet), read it back with get_storage(), not get_tables()."
266
+ )
197
267
  arrow_table = pa.Table.from_pylist(rows)
198
268
  buf = io.BytesIO()
199
269
  pq.write_table(arrow_table, buf)
@@ -204,11 +274,17 @@ class TrinoStorage:
204
274
  Path(local_path).write_bytes(self.get_bytes(path))
205
275
 
206
276
  def _ensure_namespace(self) -> None:
277
+ import datetime
278
+
207
279
  from pyiceberg.exceptions import NamespaceAlreadyExistsError
208
- lakekeeper_base = self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog"
280
+ _lk = self._lakekeeper_endpoint or "http://lakekeeper:8181"
281
+ lakekeeper_base = _lk.rstrip("/") + "/catalog"
282
+ namespace_props = {"location": self._iceberg_location_prefix}
283
+ created = False
209
284
  try:
210
- self._iceberg_catalog.create_namespace(self.schema)
285
+ self._iceberg_catalog.create_namespace(self.schema, properties=namespace_props)
211
286
  logger.info("Created namespace %s", self.schema)
287
+ created = True
212
288
  except NamespaceAlreadyExistsError:
213
289
  pass
214
290
  except Exception as exc:
@@ -224,6 +300,39 @@ class TrinoStorage:
224
300
  f"[{lakekeeper_base}] — {type(exc).__name__}: {exc}"
225
301
  ) from exc
226
302
 
303
+ if created:
304
+ from pyiceberg.schema import Schema
305
+ from pyiceberg.types import NestedField, StringType
306
+ meta_schema = Schema(
307
+ NestedField(1, "branch", StringType(), required=False),
308
+ NestedField(2, "tier", StringType(), required=False),
309
+ NestedField(3, "schema_name", StringType(), required=False),
310
+ NestedField(4, "iceberg_location", StringType(), required=False),
311
+ NestedField(5, "created_at", StringType(), required=False),
312
+ )
313
+ meta_table_id = f"{self.schema}._meta"
314
+ try:
315
+ meta_table = self._iceberg_catalog.create_table(meta_table_id, schema=meta_schema)
316
+ meta_table.append(pa.table({
317
+ "branch": [self._branch],
318
+ "tier": [self._tier],
319
+ "schema_name": [self.schema],
320
+ "iceberg_location": [self._iceberg_location_prefix],
321
+ "created_at": [datetime.datetime.now(datetime.timezone.utc).isoformat()],
322
+ }))
323
+ logger.info("Created _meta table in namespace %s", self.schema)
324
+ except Exception as exc:
325
+ logger.warning("Could not create _meta table in %s: %s", self.schema, exc)
326
+
327
+ try:
328
+ self._execute(
329
+ f'CREATE SCHEMA IF NOT EXISTS "{self.catalog}".{self.schema} '
330
+ f"WITH (location = '{self._iceberg_location_prefix}')"
331
+ )
332
+ logger.debug("Ensured Trino schema %s.%s", self.catalog, self.schema)
333
+ except Exception as exc:
334
+ logger.warning("Could not ensure Trino schema %s.%s: %s", self.catalog, self.schema, exc)
335
+
227
336
  def put_bytes(self, data: bytes, path: str) -> None:
228
337
  from pyiceberg.exceptions import NoSuchTableError
229
338
 
@@ -305,6 +414,18 @@ class TrinoStorage:
305
414
  ) from exc
306
415
  _append(iceberg_table, arrow_table, table_id)
307
416
 
417
+ if os.getenv("STEVE_TRINO_TRACE") == "1":
418
+ snapshot = None
419
+ try:
420
+ snapshot = catalog.load_table(f"{self.schema}.{self._table_name(path)}").current_snapshot()
421
+ except Exception:
422
+ pass
423
+ logger.warning(
424
+ "[trino] put_bytes: wrote %d rows → %s.%s (warehouse=%s, location=%s%s)",
425
+ len(arrow_table), self.catalog, table_id,
426
+ self._lakekeeper_warehouse, self._iceberg_location_prefix,
427
+ f", snapshot={snapshot.snapshot_id}" if snapshot else "",
428
+ )
308
429
  logger.info("Written %d rows to %s", len(arrow_table), table_id)
309
430
 
310
431
  def put_file(self, local_path: str, path: str) -> None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steve-cli
3
- Version: 0.5.6
3
+ Version: 1.0.1
4
4
  Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
5
5
  Author: Frank
6
6
  License: MIT
@@ -67,6 +67,8 @@ Requires-Dist: visidata>=3.0; extra == "all"
67
67
 
68
68
  A CLI tool for the data platform. Commands are organized by concern: general workspace automation, data lakehouse operations, semantic knowledge graphs, and governance.
69
69
 
70
+ > **Upgrading from 0.x?** See [CHANGELOG.md](./CHANGELOG.md) — v1.0.0 introduces branch-isolated storage with breaking path changes that require a one-time data migration.
71
+
70
72
  ## Installation
71
73
 
72
74
  ```bash
@@ -28,6 +28,7 @@ steve_cli/lineage/adapters/openlineage.py
28
28
  steve_cli/policies/__init__.py
29
29
  steve_cli/policies/client.py
30
30
  steve_cli/storage/__init__.py
31
+ steve_cli/storage/branch.py
31
32
  steve_cli/storage/parquet.py
32
33
  steve_cli/storage/protocol.py
33
34
  steve_cli/storage/s3.py
File without changes
File without changes