steve-cli 0.5.6__tar.gz → 1.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. {steve_cli-0.5.6 → steve_cli-1.0.2}/PKG-INFO +3 -1
  2. {steve_cli-0.5.6 → steve_cli-1.0.2}/README.md +2 -0
  3. {steve_cli-0.5.6 → steve_cli-1.0.2}/pyproject.toml +1 -1
  4. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/auth.py +10 -0
  5. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/cli.py +253 -2
  6. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/__init__.py +2 -0
  7. steve_cli-1.0.2/steve_cli/storage/branch.py +40 -0
  8. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/s3.py +15 -4
  9. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/trino.py +146 -27
  10. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli.egg-info/PKG-INFO +3 -1
  11. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli.egg-info/SOURCES.txt +1 -0
  12. {steve_cli-0.5.6 → steve_cli-1.0.2}/setup.cfg +0 -0
  13. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/__init__.py +0 -0
  14. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/decorators/__init__.py +0 -0
  15. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/decorators/lineage_job.py +0 -0
  16. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/__init__.py +0 -0
  17. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/adapters/__init__.py +0 -0
  18. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/adapters/composite.py +0 -0
  19. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/adapters/dataproductregistry.py +0 -0
  20. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/adapters/logging.py +0 -0
  21. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/adapters/null.py +0 -0
  22. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/adapters/openlineage.py +0 -0
  23. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/collector.py +0 -0
  24. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/port.py +0 -0
  25. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/registry.py +0 -0
  26. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/lineage/storage.py +0 -0
  27. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/ontology.py +0 -0
  28. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/policies/__init__.py +0 -0
  29. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/policies/client.py +0 -0
  30. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/__init__.py +0 -0
  31. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/extractors/__init__.py +0 -0
  32. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/extractors/csv.py +0 -0
  33. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/extractors/excel.py +0 -0
  34. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/extractors/generic.py +0 -0
  35. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/extractors/json.py +0 -0
  36. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/extractors/parquet.py +0 -0
  37. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/port.py +0 -0
  38. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/metadata/registry.py +0 -0
  39. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/parquet.py +0 -0
  40. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage/protocol.py +0 -0
  41. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/storage.py +0 -0
  42. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/validation/__init__.py +0 -0
  43. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/validation/adapters/__init__.py +0 -0
  44. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/validation/adapters/great_expectations.py +0 -0
  45. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/validation/adapters/null.py +0 -0
  46. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/validation/adapters/validoopsie.py +0 -0
  47. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/validation/port.py +0 -0
  48. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/validation/registry.py +0 -0
  49. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli/vkg.py +0 -0
  50. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli.egg-info/dependency_links.txt +0 -0
  51. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli.egg-info/entry_points.txt +0 -0
  52. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli.egg-info/requires.txt +0 -0
  53. {steve_cli-0.5.6 → steve_cli-1.0.2}/steve_cli.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steve-cli
3
- Version: 0.5.6
3
+ Version: 1.0.2
4
4
  Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
5
5
  Author: Frank
6
6
  License: MIT
@@ -67,6 +67,8 @@ Requires-Dist: visidata>=3.0; extra == "all"
67
67
 
68
68
  A CLI tool for the data platform. Commands are organized by concern: general workspace automation, data lakehouse operations, semantic knowledge graphs, and governance.
69
69
 
70
+ > **Upgrading from 0.x?** See [CHANGELOG.md](./CHANGELOG.md) — v1.0.0 introduces branch-isolated storage with breaking path changes that require a one-time data migration.
71
+
70
72
  ## Installation
71
73
 
72
74
  ```bash
@@ -2,6 +2,8 @@
2
2
 
3
3
  A CLI tool for the data platform. Commands are organized by concern: general workspace automation, data lakehouse operations, semantic knowledge graphs, and governance.
4
4
 
5
+ > **Upgrading from 0.x?** See [CHANGELOG.md](./CHANGELOG.md) — v1.0.0 introduces branch-isolated storage with breaking path changes that require a one-time data migration.
6
+
5
7
  ## Installation
6
8
 
7
9
  ```bash
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
5
5
 
6
6
  [project]
7
7
  name = "steve-cli"
8
- version = "0.5.6"
8
+ version = "1.0.2"
9
9
  description = "A simple CLI tool to run jobs from jobs.yaml with proper environment setup"
10
10
  readme = "README.md"
11
11
  license = {text = "MIT"}
@@ -21,6 +21,8 @@ def save_credentials(
21
21
  token: str,
22
22
  base_url: Optional[str] = None,
23
23
  workspace_name: Optional[str] = None,
24
+ expires_at: Optional[str] = None,
25
+ s3_endpoint: Optional[str] = None,
24
26
  ) -> None:
25
27
  _CREDENTIALS_FILE.parent.mkdir(parents=True, exist_ok=True)
26
28
  creds = load_credentials()
@@ -30,6 +32,10 @@ def save_credentials(
30
32
  entry["base_url"] = base_url.rstrip("/")
31
33
  if workspace_name:
32
34
  entry["workspace_name"] = workspace_name
35
+ if expires_at:
36
+ entry["expires_at"] = expires_at
37
+ if s3_endpoint:
38
+ entry["s3_endpoint"] = s3_endpoint
33
39
  workspaces[workspace_id] = entry
34
40
  creds["workspaces"] = workspaces
35
41
  _CREDENTIALS_FILE.write_text(json.dumps(creds, indent=2))
@@ -63,6 +69,10 @@ def get_token(workspace_id: Optional[str] = None) -> Optional[str]:
63
69
  return _get_workspace_entry(workspace_id).get("token")
64
70
 
65
71
 
72
+ def get_s3_endpoint(workspace_id: Optional[str] = None) -> Optional[str]:
73
+ return _get_workspace_entry(workspace_id).get("s3_endpoint")
74
+
75
+
66
76
  def get_service_url(service: str, workspace_id: Optional[str] = None) -> Optional[str]:
67
77
  entry = _get_workspace_entry(workspace_id)
68
78
  base_url = entry.get("base_url")
@@ -963,9 +963,10 @@ def policies_apply(file: Path | None):
963
963
  @click.option("--workspace", "workspace_id", default=None, help="Workspace ID (defaults to WORKSPACE_ID from .env)")
964
964
  @click.option("--url", "base_url", default=None, help="Platform root host, e.g. jds-dev.internal.jambit.io")
965
965
  @click.option("--workspace-name", "workspace_name_opt", default=None, help="Workspace name")
966
+ @click.option("--expires", "expires_at", default=None, help="Token expiry date/time (e.g. 2026-12-31 or ISO 8601). Informational only.")
966
967
  @click.option('--env-file', '-e', type=click.Path(path_type=Path), multiple=True,
967
968
  help='Path to .env file(s). Defaults to .env and .workspaces.env')
968
- def login(token: str | None, workspace_id: str | None, base_url: str | None, workspace_name_opt: str | None, env_file: tuple):
969
+ def login(token: str | None, workspace_id: str | None, base_url: str | None, workspace_name_opt: str | None, expires_at: str | None, env_file: tuple):
969
970
  """Print CLI access URLs or save a service principal token."""
970
971
  cwd = Path.cwd()
971
972
  env_files = [Path(f) for f in env_file] if env_file else [cwd / ".env", cwd / ".workspaces.env"]
@@ -984,9 +985,27 @@ def login(token: str | None, workspace_id: str | None, base_url: str | None, wor
984
985
  click.secho("Could not resolve workspace ID. Set WORKSPACE_ID in .env or pass --workspace.", fg="red", err=True)
985
986
  raise SystemExit(1)
986
987
  click.echo(f"\nYou are about to log in to workspace {click.style(workspace_name, fg='cyan', bold=True)} ({resolved_workspace})")
987
- save_credentials(resolved_workspace, token, base_url=base_url, workspace_name=workspace_name)
988
+ save_credentials(resolved_workspace, token, base_url=base_url, workspace_name=workspace_name, expires_at=expires_at)
988
989
  if base_url:
989
990
  click.echo(f"Platform: {base_url}")
991
+
992
+ s3_endpoint: str | None = None
993
+ if base_url:
994
+ try:
995
+ import urllib.request as _urlreq
996
+ config_url = f"https://datameshx.{base_url}/api/v1/cli/config"
997
+ req = _urlreq.Request(config_url, headers={"Authorization": f"Bearer {token}"})
998
+ with _urlreq.urlopen(req, timeout=5) as _r:
999
+ import json as _json
1000
+ cfg = _json.loads(_r.read().decode())
1001
+ s3_endpoint = cfg.get("s3Endpoint")
1002
+ except Exception as _e:
1003
+ click.secho(f"Could not fetch platform config (s3Endpoint will use .env fallback): {_e}", fg="yellow", err=True)
1004
+
1005
+ if s3_endpoint:
1006
+ save_credentials(resolved_workspace, token, base_url=base_url, workspace_name=workspace_name, expires_at=expires_at, s3_endpoint=s3_endpoint)
1007
+ click.echo(f"S3 endpoint: {s3_endpoint}")
1008
+
990
1009
  click.secho(f"Logged in. Credentials saved to ~/.steve/credentials.json", fg="green")
991
1010
  return
992
1011
 
@@ -1022,6 +1041,238 @@ def logout(workspace_id: str | None, env_file: tuple):
1022
1041
  click.secho("All credentials removed.", fg="green")
1023
1042
 
1024
1043
 
1044
+ @main.command("status")
1045
+ @click.option('--env-file', '-e', type=click.Path(path_type=Path), multiple=True,
1046
+ help='Path to .env file(s). Defaults to .env and .workspaces.env')
1047
+ @click.option('--no-check', is_flag=True, default=False, help='Skip connectivity checks.')
1048
+ def status(env_file: tuple, no_check: bool):
1049
+ """Show current steve-cli environment, storage paths, and connectivity."""
1050
+ import importlib.metadata
1051
+ import urllib.request
1052
+
1053
+ cwd = Path.cwd()
1054
+ env_files = [Path(f) for f in env_file] if env_file else [cwd / ".env", cwd / ".workspaces.env"]
1055
+ for ef in env_files:
1056
+ load_dotenv(ef)
1057
+
1058
+ try:
1059
+ version = importlib.metadata.version("steve-cli")
1060
+ except Exception:
1061
+ version = "unknown"
1062
+
1063
+ def _section(title: str) -> None:
1064
+ click.echo(f"\n{click.style(f'━━ {title} ', fg='bright_black')}{'━' * max(0, 48 - len(title))}")
1065
+
1066
+ def _row(label: str, value: str, value_color: str = "white") -> None:
1067
+ click.echo(f" {click.style(label.ljust(14), fg='bright_black')} {click.style(value, fg=value_color)}")
1068
+
1069
+ def _check(label: str, url: str, timeout: int = 2, token: str = "", suffix: str = "") -> None:
1070
+ if no_check:
1071
+ _row(label, "(skipped)", "bright_black")
1072
+ return
1073
+ try:
1074
+ req = urllib.request.Request(url)
1075
+ if token:
1076
+ req.add_header("Authorization", f"Bearer {token}")
1077
+ with urllib.request.urlopen(req, timeout=timeout) as r:
1078
+ _row(label, f"✓ reachable ({url}){suffix}", "green")
1079
+ return r.read()
1080
+ except urllib.error.HTTPError:
1081
+ _row(label, f"✓ reachable ({url}){suffix}", "green")
1082
+ except Exception as e:
1083
+ _row(label, f"✗ unreachable ({url}){suffix}: {e}", "red")
1084
+
1085
+ click.echo(f"\n{click.style('●', fg='green')} Steve CLI {click.style('v' + version, fg='cyan', bold=True)}")
1086
+
1087
+ from steve_cli.storage.branch import get_branch_prefix
1088
+ import subprocess as _sp
1089
+
1090
+ branch = get_branch_prefix()
1091
+ steve_branch_env = os.getenv("STEVE_BRANCH", "").strip()
1092
+ gh_head = os.getenv("GITHUB_HEAD_REF", "").strip() or os.getenv("GITHUB_REF_NAME", "").strip()
1093
+ if steve_branch_env:
1094
+ branch_source = f"STEVE_BRANCH={steve_branch_env!r}"
1095
+ elif gh_head:
1096
+ branch_source = f"GITHUB env ({gh_head})"
1097
+ else:
1098
+ try:
1099
+ r = _sp.run(["git", "rev-parse", "--abbrev-ref", "HEAD"], capture_output=True, text=True, timeout=3)
1100
+ branch_source = "git" if r.returncode == 0 and r.stdout.strip() else "fallback"
1101
+ except Exception:
1102
+ branch_source = "fallback"
1103
+
1104
+ workspace_id = os.getenv("WORKSPACE_ID", "")
1105
+ workspace_name = os.getenv("WORKSPACE_NAME", "")
1106
+ session = os.getenv("SESSION_NAME", "") or socket.gethostname()
1107
+
1108
+ from steve_cli.auth import _get_workspace_entry, get_service_url
1109
+ import datetime as _dt
1110
+ entry = _get_workspace_entry(workspace_id or None)
1111
+ token = entry.get("token", "")
1112
+ expires_at = entry.get("expires_at", "")
1113
+ if token:
1114
+ expiry_warn = False
1115
+ expiry_str = ", expires never" if not expires_at else ""
1116
+ if expires_at:
1117
+ try:
1118
+ exp = _dt.datetime.fromisoformat(expires_at.replace("Z", "+00:00"))
1119
+ now = _dt.datetime.now(_dt.timezone.utc) if exp.tzinfo else _dt.datetime.now()
1120
+ delta = exp - now
1121
+ total_seconds = int(delta.total_seconds())
1122
+ if total_seconds < 0:
1123
+ s = abs(total_seconds)
1124
+ if s < 3600:
1125
+ ago = f"{s // 60}m ago"
1126
+ elif s < 86400:
1127
+ ago = f"{s // 3600}h ago"
1128
+ else:
1129
+ ago = f"{s // 86400}d ago"
1130
+ expiry_str = f", expired {ago}"
1131
+ expiry_warn = True
1132
+ elif total_seconds < 3600:
1133
+ expiry_str = f", expires in {total_seconds // 60}m ⚠"
1134
+ expiry_warn = True
1135
+ elif total_seconds < 86400:
1136
+ expiry_str = f", expires in {total_seconds // 3600}h ⚠"
1137
+ expiry_warn = True
1138
+ elif delta.days <= 7:
1139
+ expiry_str = f", expires in {delta.days}d ⚠"
1140
+ expiry_warn = True
1141
+ else:
1142
+ expiry_str = f", expires {exp.strftime('%Y-%m-%d %H:%M')}"
1143
+ except Exception:
1144
+ expiry_str = f", expires {expires_at}"
1145
+ token_display = f"✓ logged in (stp_***{token[-4:]}{expiry_str})"
1146
+ token_color = "yellow" if expiry_warn else "green"
1147
+ else:
1148
+ token_display = "✗ not logged in — run: steve login"
1149
+ token_color = "yellow"
1150
+
1151
+ _section("Identity")
1152
+ _row("Branch", f"{branch} ({branch_source})", "cyan")
1153
+ _row("Workspace ID", workspace_id or "(not set)", "white" if workspace_id else "yellow")
1154
+ _row("Workspace", workspace_name or "(not set)", "white" if workspace_name else "yellow")
1155
+ _row("Session", session, "white")
1156
+ if token and not no_check:
1157
+ _base_url = entry.get("base_url", "")
1158
+ if _base_url:
1159
+ try:
1160
+ _req = urllib.request.Request(
1161
+ f"https://datameshx.{_base_url}/api/v1/cli/config",
1162
+ headers={"Authorization": f"Bearer {token}"},
1163
+ )
1164
+ with urllib.request.urlopen(_req, timeout=4) as _r:
1165
+ _r.read()
1166
+ _row("Auth", f"✓ token valid (stp_***{token[-4:]}{expiry_str})", "green" if not expiry_warn else "yellow")
1167
+ except urllib.error.HTTPError as _e:
1168
+ if _e.code == 401:
1169
+ _row("Auth", f"✗ token invalid/revoked (stp_***{token[-4:]}{expiry_str})", "red")
1170
+ else:
1171
+ _row("Auth", f"✓ token valid (stp_***{token[-4:]}{expiry_str})", "green" if not expiry_warn else "yellow")
1172
+ except Exception as _e:
1173
+ _row("Auth", f"{token_display} (validation failed: {_e})", token_color)
1174
+ else:
1175
+ _row("Auth", token_display, token_color)
1176
+ else:
1177
+ _row("Auth", token_display, token_color)
1178
+
1179
+ tiers = ["bronze", "silver", "gold"]
1180
+ s3_rows = []
1181
+ for tier in tiers:
1182
+ tier_up = tier.upper()
1183
+ bucket = os.getenv(f"{tier_up}_BUCKET", "")
1184
+ if bucket:
1185
+ s3_rows.append((f"S3 {tier}", f"s3://{bucket}/_store/{branch}/"))
1186
+ for ws in _detect_workspaces():
1187
+ for tier in tiers:
1188
+ bucket = os.getenv(f"{ws}_BUCKET_{tier.upper()}", "")
1189
+ if bucket:
1190
+ s3_rows.append((f"S3 {ws.lower()}/{tier}", f"s3://{bucket}/_store/{branch}/"))
1191
+
1192
+ _section("Storage Paths")
1193
+ if s3_rows:
1194
+ for label, path in s3_rows:
1195
+ _row(label, path, "cyan")
1196
+ else:
1197
+ _row("S3", "(no bucket env vars found)", "yellow")
1198
+ _row("Iceberg base", f"_iceberg/{branch}/", "cyan")
1199
+
1200
+ trino_endpoint = get_service_url("trino", workspace_id or None)
1201
+ resolved_workspace = workspace_name or workspace_id or ""
1202
+ trino_catalog = os.getenv("TRINO_CATALOG", "minio")
1203
+ trino_schema_explicit = os.getenv("TRINO_SCHEMA", "")
1204
+
1205
+ _section("Trino")
1206
+ if trino_endpoint:
1207
+ _row("Endpoint", trino_endpoint, "cyan")
1208
+ else:
1209
+ _row("Endpoint", "✗ not configured (login required)", "yellow")
1210
+ _row("Catalog", trino_catalog, "white")
1211
+ for tier in tiers:
1212
+ if trino_schema_explicit:
1213
+ schema = trino_schema_explicit
1214
+ elif resolved_workspace:
1215
+ ws_clean = resolved_workspace.lower().replace("-", "_")
1216
+ schema = f"{ws_clean}_{tier}__{branch}"
1217
+ else:
1218
+ schema = f"(workspace not set — set WORKSPACE_NAME)"
1219
+ _row(f"Schema {tier}", schema, "cyan" if resolved_workspace or trino_schema_explicit else "yellow")
1220
+ if trino_schema_explicit:
1221
+ break
1222
+
1223
+ lakekeeper = os.getenv("LAKEKEEPER_ENDPOINT", "") or get_service_url("lakekeeper", workspace_id or None)
1224
+
1225
+ _section("Connectivity")
1226
+ _cred_s3 = entry.get("s3_endpoint", "")
1227
+ _env_s3 = os.getenv("S3_ENDPOINT", "")
1228
+ if _cred_s3:
1229
+ s3_endpoint = _cred_s3
1230
+ s3_source = "credentials"
1231
+ elif _env_s3:
1232
+ s3_endpoint = _env_s3
1233
+ s3_source = "S3_ENDPOINT env"
1234
+ else:
1235
+ s3_endpoint = "http://localhost:9000"
1236
+ s3_source = "default"
1237
+ _check("S3 endpoint", s3_endpoint, suffix=f" (from {s3_source})")
1238
+ if trino_endpoint:
1239
+ if no_check:
1240
+ _row("Trino", "(skipped)", "bright_black")
1241
+ else:
1242
+ try:
1243
+ import json as _json
1244
+ req = urllib.request.Request(f"{trino_endpoint}/v1/info")
1245
+ if token:
1246
+ req.add_header("Authorization", f"Bearer {token}")
1247
+ req.add_header("X-Trino-User", os.getenv("TRINO_USER", "admin"))
1248
+ with urllib.request.urlopen(req, timeout=3) as r:
1249
+ info = _json.loads(r.read())
1250
+ version_str = info.get("nodeVersion", {}).get("version", "?")
1251
+ state = info.get("state", "?")
1252
+ uptime = info.get("uptime", "")
1253
+ detail = f"v{version_str} {state.lower()}" + (f", up {uptime}" if uptime else "")
1254
+ _row("Trino", f"✓ reachable ({detail}) {trino_endpoint}", "green")
1255
+ except Exception as e:
1256
+ _row("Trino", f"✗ unreachable: {e}", "red")
1257
+ else:
1258
+ _row("Trino", "✗ not configured", "yellow")
1259
+ if lakekeeper:
1260
+ _check("Lakekeeper", f"{lakekeeper}/catalog/v1/config")
1261
+ else:
1262
+ _row("Lakekeeper", "(not configured — set LAKEKEEPER_ENDPOINT or login with --url)", "bright_black")
1263
+
1264
+ _section("Environment")
1265
+ for ef in env_files:
1266
+ state = "✓ found" if ef.exists() else "✗ missing"
1267
+ color = "green" if ef.exists() else "bright_black"
1268
+ _row(ef.name, state, color)
1269
+ _row("STEVE_BRANCH", repr(steve_branch_env) if steve_branch_env else "(not set — auto-detected)", "white" if steve_branch_env else "bright_black")
1270
+ detected_ws = _detect_workspaces()
1271
+ if detected_ws:
1272
+ _row("Upstream ws", ", ".join(detected_ws), "white")
1273
+ click.echo()
1274
+
1275
+
1025
1276
  @main.command("upgrade")
1026
1277
  def upgrade():
1027
1278
  """Upgrade steve-cli to the latest version."""
@@ -1,3 +1,4 @@
1
+ from .branch import get_branch_prefix
1
2
  from .protocol import Storage
2
3
  from .s3 import S3Storage
3
4
  from .trino import TrinoStorage
@@ -5,6 +6,7 @@ from .parquet import ParquetMetadata, extract_parquet_metadata
5
6
  from .metadata import FileMetadata, ColumnMetadata, MetadataExtractorPort, MetadataRegistry
6
7
 
7
8
  __all__ = [
9
+ "get_branch_prefix",
8
10
  "Storage",
9
11
  "S3Storage",
10
12
  "TrinoStorage",
@@ -0,0 +1,40 @@
1
+ from __future__ import annotations
2
+
3
+ import logging
4
+ import os
5
+ import re
6
+ import subprocess
7
+
8
+ logger = logging.getLogger(__name__)
9
+
10
+ _MAIN_BRANCHES = {"main", "master"}
11
+
12
+
13
+ def _sanitize(branch: str) -> str:
14
+ branch = branch.lower().strip()
15
+ branch = re.sub(r"[^a-z0-9]+", "_", branch)
16
+ return branch.strip("_") or "main"
17
+
18
+
19
+ def get_branch_prefix() -> str:
20
+ for env_var in ("STEVE_BRANCH", "GITHUB_HEAD_REF", "GITHUB_REF_NAME"):
21
+ value = os.getenv(env_var, "").strip()
22
+ if value:
23
+ return _sanitize(value)
24
+
25
+ try:
26
+ result = subprocess.run(
27
+ ["git", "rev-parse", "--abbrev-ref", "HEAD"],
28
+ capture_output=True,
29
+ text=True,
30
+ timeout=3,
31
+ )
32
+ if result.returncode == 0:
33
+ branch = result.stdout.strip()
34
+ if branch and branch != "HEAD":
35
+ return _sanitize(branch)
36
+ except Exception as exc:
37
+ logger.debug("git branch detection failed: %s", exc)
38
+
39
+ logger.warning("Could not detect git branch; defaulting to 'main'")
40
+ return "main"
@@ -8,16 +8,27 @@ import boto3
8
8
  from botocore.client import Config
9
9
  from botocore.exceptions import ClientError, EndpointConnectionError
10
10
 
11
+ from .branch import get_branch_prefix
12
+ from steve_cli.auth import get_s3_endpoint
13
+
11
14
  logger = logging.getLogger(__name__)
12
15
 
13
16
 
14
17
  class S3Storage:
15
18
  def __init__(self, tier: str = "bronze", workspace: str | None = None):
16
19
  self.tier = tier.lower()
20
+ self._branch = get_branch_prefix()
17
21
  self.endpoint, self.bucket, access_key, secret_key, required_vars = (
18
22
  self._load_config(self.tier, workspace)
19
23
  )
20
24
 
25
+ cred_endpoint = get_s3_endpoint()
26
+ if cred_endpoint:
27
+ self.endpoint = cred_endpoint
28
+ self._endpoint_source = "credentials"
29
+ else:
30
+ self._endpoint_source = "env" if os.getenv("S3_ENDPOINT") else "default"
31
+
21
32
  missing = [v for v in required_vars if not os.getenv(v)]
22
33
  if missing:
23
34
  raise OSError(
@@ -66,11 +77,11 @@ class S3Storage:
66
77
  ]
67
78
  return endpoint, bucket, access_key, secret_key, required
68
79
 
69
- @staticmethod
70
- def _key(path: str) -> str:
80
+ def _key(self, path: str) -> str:
71
81
  if not path:
72
- return ""
73
- return str(PurePosixPath(path).as_posix().lstrip("/"))
82
+ return f"_store/{self._branch}/"
83
+ clean = str(PurePosixPath(path).as_posix().lstrip("/"))
84
+ return f"_store/{self._branch}/{clean}"
74
85
 
75
86
  def _object_location(self, path: str) -> str:
76
87
  return f"{self.endpoint}/{self.bucket}/{self._key(path)}"
@@ -10,19 +10,21 @@ import pyarrow as pa
10
10
  import pyarrow.parquet as pq
11
11
  import requests
12
12
 
13
+ from .branch import get_branch_prefix
14
+
13
15
  logger = logging.getLogger(__name__)
14
16
 
15
17
 
16
- def _derive_schema(tier: str, workspace: str | None) -> str | None:
17
- if workspace:
18
- prefix = workspace.lower().replace("-", "_")
19
- return f"ws_{prefix}_{tier.lower()}"
20
- return None
18
+ def _derive_schema(tier: str, workspace: str | None, branch: str) -> str:
19
+ base = f"{workspace.lower().replace('-', '_')}_{tier.lower()}" if workspace else tier.lower()
20
+ return f"{base}__{branch}"
21
21
 
22
22
 
23
23
  class TrinoStorage:
24
24
  def __init__(self, tier: str = "bronze", workspace: str | None = None):
25
25
  from steve_cli.auth import get_service_url
26
+ self._branch = get_branch_prefix()
27
+ self._tier = tier.lower()
26
28
  self._workspace_id = os.getenv("WORKSPACE_ID")
27
29
  trino_endpoint = get_service_url("trino", self._workspace_id)
28
30
  if not trino_endpoint:
@@ -36,16 +38,25 @@ class TrinoStorage:
36
38
  ]
37
39
  detail = "\n".join(f" {mark} {desc}" for mark, desc in checks)
38
40
  raise OSError(f"Trino endpoint not found:\n{detail}")
39
- self.catalog = os.getenv("TRINO_CATALOG", "minio")
40
41
  resolved_workspace = workspace or os.getenv("WORKSPACE_NAME")
41
- self.schema = _derive_schema(tier, resolved_workspace) or os.environ.get("TRINO_SCHEMA", "")
42
- if not self.schema:
43
- raise OSError("TRINO_SCHEMA is not set and no WORKSPACE_NAME or workspace was provided")
44
- self.user = os.getenv("TRINO_USER", "admin")
42
+ default_catalog = f"{resolved_workspace}-{tier}" if resolved_workspace else "minio"
43
+ self.catalog = os.getenv("TRINO_CATALOG", default_catalog)
44
+ self.schema = os.environ.get("TRINO_SCHEMA") or _derive_schema(tier, resolved_workspace, self._branch)
45
+ self.user = os.getenv("TRINO_USER") or resolved_workspace or "admin"
45
46
  self._base = trino_endpoint.rstrip("/")
46
- self._lakekeeper_endpoint = os.getenv("LAKEKEEPER_ENDPOINT")
47
- default_warehouse = f"{resolved_workspace}-{tier}" if resolved_workspace else "minio"
48
- self._lakekeeper_warehouse = os.getenv("LAKEKEEPER_WAREHOUSE", default_warehouse)
47
+ self._lakekeeper_endpoint = os.getenv("LAKEKEEPER_ENDPOINT") or get_service_url("lakekeeper", self._workspace_id)
48
+ self._lakekeeper_warehouse = os.getenv("LAKEKEEPER_WAREHOUSE", default_catalog)
49
+ _iceberg_bucket = os.getenv(f"{tier.upper()}_BUCKET", "")
50
+ if _iceberg_bucket:
51
+ self._iceberg_location_prefix = f"s3://{_iceberg_bucket}/_iceberg/{self._branch}"
52
+ else:
53
+ raise OSError(
54
+ f"TrinoStorage: missing bucket configuration for tier '{tier}' — "
55
+ f"{tier.upper()}_BUCKET is not set.\n"
56
+ f" Required env vars: {tier.upper()}_BUCKET, {tier.upper()}_ACCESS_KEY, {tier.upper()}_SECRET_KEY\n"
57
+ f" This workspace may only have read access to the '{tier}' tier. "
58
+ f"Write access requires provisioning credentials (run 'uv run steve setup env' or check workspace provisioning)."
59
+ )
49
60
  self.__iceberg_catalog = None
50
61
 
51
62
  @property
@@ -55,24 +66,43 @@ class TrinoStorage:
55
66
  from pyiceberg.io import load_file_io
56
67
  from pyiceberg.table import Table
57
68
 
58
- # FIXME: Interim workaround — pyiceberg cannot use Lakekeeper's remote signing
59
- # endpoint when running outside the cluster (hostname 'lakekeeper' doesn't resolve
60
- # locally). We subclass RestCatalog to inject our local signer URI after pyiceberg
61
- # merges table config (which overwrites s3.signer.uri with the internal hostname).
62
- # Remove once Increment 5 (STS credential vending) is wired.
63
- tier = self.schema.rsplit("_", 1)[-1].upper()
64
- lakekeeper_base = self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog"
69
+ # WORKAROUND: Lakekeeper's remote S3 presigner is bypassed for external access.
70
+ #
71
+ # Lakekeeper returns `s3.signer` / `s3.signer.uri` in the table config, which
72
+ # pyiceberg uses to sign S3 requests via Lakekeeper's signing endpoint. This works
73
+ # inside the cluster but fails from a laptop because:
74
+ # 1. The signed URI Lakekeeper produces contains the cluster-internal hostname
75
+ # (e.g. http://minio:9000) which is unreachable externally.
76
+ # 2. Even after overriding the signer URI to the public lakekeeper subdomain,
77
+ # Lakekeeper's signer returns HTTP 500 for URIs with the public minio hostname
78
+ # because it only knows the internal minio endpoint.
79
+ #
80
+ # Current workaround: strip all signer config from the merged table properties and
81
+ # use direct S3 credentials (access-key + secret-key) with the public minio endpoint
82
+ # from ~/.steve/credentials.json. This bypasses Lakekeeper's access control for S3
83
+ # writes — credentials live on disk rather than being vended by the cluster.
84
+ #
85
+ # To revert: configure Lakekeeper with its public S3 endpoint so presigned URIs
86
+ # contain a resolvable hostname, then remove the merged.pop() calls below and
87
+ # restore `s3.signer.uri: lakekeeper_base` in s3_overrides.
88
+ tier = self._tier.upper()
89
+ lakekeeper_base = (self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog").rstrip("/")
90
+ from steve_cli.auth import get_s3_endpoint as _get_s3_ep
91
+ _s3_ep = _get_s3_ep(self._workspace_id) or os.getenv("S3_ENDPOINT", "http://minio:9000")
65
92
  s3_overrides = {
66
- "s3.endpoint": os.getenv("S3_ENDPOINT", "http://minio:9000"),
93
+ "s3.endpoint": _s3_ep,
67
94
  "s3.access-key-id": os.getenv(f"{tier}_ACCESS_KEY") or os.getenv("AWS_ACCESS_KEY_ID", ""),
68
95
  "s3.secret-access-key": os.getenv(f"{tier}_SECRET_KEY") or os.getenv("AWS_SECRET_ACCESS_KEY", ""),
69
96
  "s3.path-style-access": "true",
70
- "s3.signer.uri": lakekeeper_base,
71
97
  } if self._lakekeeper_endpoint else {}
72
98
 
73
99
  class _PatchedRestCatalog(RestCatalog):
74
100
  def _response_to_table(self_, identifier_tuple, table_response):
75
101
  merged = {**table_response.metadata.properties, **table_response.config, **s3_overrides, "uri": lakekeeper_base}
102
+ # Strip remote-signer config — see WORKAROUND comment above.
103
+ merged.pop("s3.signer", None)
104
+ merged.pop("s3.signer.uri", None)
105
+ merged.pop("s3.signer.endpoint", None)
76
106
  return Table(
77
107
  identifier=identifier_tuple,
78
108
  metadata_location=table_response.metadata_location,
@@ -82,12 +112,16 @@ class TrinoStorage:
82
112
  config=table_response.config,
83
113
  )
84
114
 
115
+ from steve_cli.auth import get_token
116
+ stp_token = get_token(self._workspace_id)
117
+ token_kwargs = {"token": stp_token} if stp_token else {}
85
118
  try:
86
119
  catalog = _PatchedRestCatalog(
87
120
  "lakekeeper",
88
121
  uri=lakekeeper_base,
89
122
  warehouse=self._lakekeeper_warehouse,
90
123
  **s3_overrides,
124
+ **token_kwargs,
91
125
  )
92
126
  except Exception as exc:
93
127
  msg = str(exc)
@@ -172,7 +206,7 @@ class TrinoStorage:
172
206
  return path.strip("/").replace("/", "_").replace(".", "_")
173
207
 
174
208
  def list(self, prefix: str = "") -> list[str]:
175
- rows = self._execute(f"SHOW TABLES FROM {self.catalog}.{self.schema}")
209
+ rows = self._execute(f'SHOW TABLES FROM "{self.catalog}".{self.schema}')
176
210
  tables = [r["Table"] for r in rows]
177
211
  if prefix:
178
212
  clean = prefix.strip("/")
@@ -183,7 +217,10 @@ class TrinoStorage:
183
217
  return self.list()
184
218
 
185
219
  def get_bytes(self, path: str) -> bytes:
186
- table_ref = f"{self.catalog}.{self.schema}.{self._table_name(path)}"
220
+ table_name = self._table_name(path)
221
+ table_ref = f'"{self.catalog}".{self.schema}.{table_name}'
222
+ if os.getenv("STEVE_TRINO_TRACE") == "1":
223
+ logger.warning("[trino] get_bytes: %s (warehouse=%s)", table_ref, self._lakekeeper_warehouse)
187
224
  try:
188
225
  rows = self._execute(f"SELECT * FROM {table_ref}")
189
226
  except OSError:
@@ -193,7 +230,39 @@ class TrinoStorage:
193
230
  f"TrinoStorage: read from '{table_ref}' failed — {type(exc).__name__}: {exc}"
194
231
  ) from exc
195
232
  if not rows:
196
- return b""
233
+ list_err: str | None = None
234
+ table_names: list[str] | None = None
235
+ try:
236
+ existing = self._execute(f'SHOW TABLES FROM "{self.catalog}".{self.schema}')
237
+ table_names = [r.get("Table", "") for r in existing]
238
+ except Exception as list_exc:
239
+ list_err = str(list_exc)
240
+
241
+ if table_names is not None and table_name in table_names:
242
+ raise OSError(
243
+ f"TrinoStorage: table '{table_ref}' exists in Trino but returned 0 rows — "
244
+ f"data may not have been written yet, or the last write produced an empty result."
245
+ )
246
+ elif list_err is not None:
247
+ raise OSError(
248
+ f"TrinoStorage: table '{table_name}' not found in catalog='{self.catalog}' schema='{self.schema}'.\n"
249
+ f" Looked in: {table_ref}\n"
250
+ f" Could not list tables in that schema: {list_err}\n"
251
+ f" Hint: if data was written with get_storage() (S3/Parquet), read it back with get_storage(), not get_tables()."
252
+ )
253
+ elif table_names == []:
254
+ raise OSError(
255
+ f"TrinoStorage: table '{table_name}' not found — schema '{self.catalog}'.'{self.schema}' exists but contains no tables.\n"
256
+ f" Looked in: {table_ref}\n"
257
+ f" Hint: if data was written with get_storage() (S3/Parquet), read it back with get_storage(), not get_tables()."
258
+ )
259
+ else:
260
+ raise OSError(
261
+ f"TrinoStorage: table '{table_name}' not found in catalog='{self.catalog}' schema='{self.schema}'.\n"
262
+ f" Looked in: {table_ref}\n"
263
+ f" Tables found in that schema: {', '.join(table_names or [])}\n"
264
+ f" Hint: if data was written with get_storage() (S3/Parquet), read it back with get_storage(), not get_tables()."
265
+ )
197
266
  arrow_table = pa.Table.from_pylist(rows)
198
267
  buf = io.BytesIO()
199
268
  pq.write_table(arrow_table, buf)
@@ -204,11 +273,16 @@ class TrinoStorage:
204
273
  Path(local_path).write_bytes(self.get_bytes(path))
205
274
 
206
275
  def _ensure_namespace(self) -> None:
276
+ import datetime
277
+
207
278
  from pyiceberg.exceptions import NamespaceAlreadyExistsError
208
- lakekeeper_base = self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog"
279
+ lakekeeper_base = (self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog").rstrip("/")
280
+ namespace_props = {"location": self._iceberg_location_prefix}
281
+ created = False
209
282
  try:
210
- self._iceberg_catalog.create_namespace(self.schema)
283
+ self._iceberg_catalog.create_namespace(self.schema, properties=namespace_props)
211
284
  logger.info("Created namespace %s", self.schema)
285
+ created = True
212
286
  except NamespaceAlreadyExistsError:
213
287
  pass
214
288
  except Exception as exc:
@@ -224,6 +298,39 @@ class TrinoStorage:
224
298
  f"[{lakekeeper_base}] — {type(exc).__name__}: {exc}"
225
299
  ) from exc
226
300
 
301
+ if created:
302
+ from pyiceberg.schema import Schema
303
+ from pyiceberg.types import NestedField, StringType
304
+ meta_schema = Schema(
305
+ NestedField(1, "branch", StringType(), required=False),
306
+ NestedField(2, "tier", StringType(), required=False),
307
+ NestedField(3, "schema_name", StringType(), required=False),
308
+ NestedField(4, "iceberg_location", StringType(), required=False),
309
+ NestedField(5, "created_at", StringType(), required=False),
310
+ )
311
+ meta_table_id = f"{self.schema}._meta"
312
+ try:
313
+ meta_table = self._iceberg_catalog.create_table(meta_table_id, schema=meta_schema)
314
+ meta_table.append(pa.table({
315
+ "branch": [self._branch],
316
+ "tier": [self._tier],
317
+ "schema_name": [self.schema],
318
+ "iceberg_location": [self._iceberg_location_prefix],
319
+ "created_at": [datetime.datetime.now(datetime.timezone.utc).isoformat()],
320
+ }))
321
+ logger.info("Created _meta table in namespace %s", self.schema)
322
+ except Exception as exc:
323
+ logger.warning("Could not create _meta table in %s: %s", self.schema, exc)
324
+
325
+ try:
326
+ self._execute(
327
+ f'CREATE SCHEMA IF NOT EXISTS "{self.catalog}".{self.schema} '
328
+ f"WITH (location = '{self._iceberg_location_prefix}')"
329
+ )
330
+ logger.debug("Ensured Trino schema %s.%s", self.catalog, self.schema)
331
+ except Exception as exc:
332
+ logger.warning("Could not ensure Trino schema %s.%s: %s", self.catalog, self.schema, exc)
333
+
227
334
  def put_bytes(self, data: bytes, path: str) -> None:
228
335
  from pyiceberg.exceptions import NoSuchTableError
229
336
 
@@ -305,6 +412,18 @@ class TrinoStorage:
305
412
  ) from exc
306
413
  _append(iceberg_table, arrow_table, table_id)
307
414
 
415
+ if os.getenv("STEVE_TRINO_TRACE") == "1":
416
+ snapshot = None
417
+ try:
418
+ snapshot = catalog.load_table(f"{self.schema}.{self._table_name(path)}").current_snapshot()
419
+ except Exception:
420
+ pass
421
+ logger.warning(
422
+ "[trino] put_bytes: wrote %d rows → %s.%s (warehouse=%s, location=%s%s)",
423
+ len(arrow_table), self.catalog, table_id,
424
+ self._lakekeeper_warehouse, self._iceberg_location_prefix,
425
+ f", snapshot={snapshot.snapshot_id}" if snapshot else "",
426
+ )
308
427
  logger.info("Written %d rows to %s", len(arrow_table), table_id)
309
428
 
310
429
  def put_file(self, local_path: str, path: str) -> None:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steve-cli
3
- Version: 0.5.6
3
+ Version: 1.0.2
4
4
  Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
5
5
  Author: Frank
6
6
  License: MIT
@@ -67,6 +67,8 @@ Requires-Dist: visidata>=3.0; extra == "all"
67
67
 
68
68
  A CLI tool for the data platform. Commands are organized by concern: general workspace automation, data lakehouse operations, semantic knowledge graphs, and governance.
69
69
 
70
+ > **Upgrading from 0.x?** See [CHANGELOG.md](./CHANGELOG.md) — v1.0.0 introduces branch-isolated storage with breaking path changes that require a one-time data migration.
71
+
70
72
  ## Installation
71
73
 
72
74
  ```bash
@@ -28,6 +28,7 @@ steve_cli/lineage/adapters/openlineage.py
28
28
  steve_cli/policies/__init__.py
29
29
  steve_cli/policies/client.py
30
30
  steve_cli/storage/__init__.py
31
+ steve_cli/storage/branch.py
31
32
  steve_cli/storage/parquet.py
32
33
  steve_cli/storage/protocol.py
33
34
  steve_cli/storage/s3.py
File without changes
File without changes