steve-cli 0.3.20__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. {steve_cli-0.3.20 → steve_cli-0.4.1}/PKG-INFO +24 -12
  2. {steve_cli-0.3.20 → steve_cli-0.4.1}/README.md +17 -11
  3. {steve_cli-0.3.20 → steve_cli-0.4.1}/pyproject.toml +8 -1
  4. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/cli.py +120 -0
  5. steve_cli-0.4.1/steve_cli/decorators/__init__.py +10 -0
  6. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/decorators/lineage_job.py +13 -2
  7. steve_cli-0.4.1/steve_cli/policies/__init__.py +3 -0
  8. steve_cli-0.4.1/steve_cli/policies/client.py +89 -0
  9. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/__init__.py +2 -0
  10. steve_cli-0.4.1/steve_cli/storage/trino.py +199 -0
  11. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/PKG-INFO +24 -12
  12. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/SOURCES.txt +3 -0
  13. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/requires.txt +7 -0
  14. steve_cli-0.3.20/steve_cli/decorators/__init__.py +0 -3
  15. {steve_cli-0.3.20 → steve_cli-0.4.1}/setup.cfg +0 -0
  16. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/__init__.py +0 -0
  17. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/__init__.py +0 -0
  18. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/__init__.py +0 -0
  19. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/logging.py +0 -0
  20. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/null.py +0 -0
  21. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/openlineage.py +0 -0
  22. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/collector.py +0 -0
  23. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/port.py +0 -0
  24. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/registry.py +0 -0
  25. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/storage.py +0 -0
  26. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/__init__.py +0 -0
  27. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/__init__.py +0 -0
  28. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/csv.py +0 -0
  29. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/excel.py +0 -0
  30. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/generic.py +0 -0
  31. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/json.py +0 -0
  32. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/parquet.py +0 -0
  33. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/port.py +0 -0
  34. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/registry.py +0 -0
  35. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/parquet.py +0 -0
  36. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/protocol.py +0 -0
  37. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/s3.py +0 -0
  38. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage.py +0 -0
  39. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/__init__.py +0 -0
  40. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/__init__.py +0 -0
  41. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/great_expectations.py +0 -0
  42. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/null.py +0 -0
  43. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/validoopsie.py +0 -0
  44. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/port.py +0 -0
  45. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/registry.py +0 -0
  46. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/dependency_links.txt +0 -0
  47. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/entry_points.txt +0 -0
  48. {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steve-cli
3
- Version: 0.3.20
3
+ Version: 0.4.1
4
4
  Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
5
5
  Author: Frank
6
6
  License: MIT
@@ -44,6 +44,12 @@ Requires-Dist: great-expectations>=0.18.0; extra == "great-expectations"
44
44
  Requires-Dist: pandas>=2.0.3; extra == "great-expectations"
45
45
  Provides-Extra: excel
46
46
  Requires-Dist: openpyxl>=3.1.0; extra == "excel"
47
+ Provides-Extra: trino
48
+ Requires-Dist: trino>=0.330; extra == "trino"
49
+ Requires-Dist: pyarrow>=17.0.0; extra == "trino"
50
+ Requires-Dist: requests>=2.31.0; extra == "trino"
51
+ Requires-Dist: pyiceberg>=0.7.0; extra == "trino"
52
+ Requires-Dist: s3fs>=2024.1.0; extra == "trino"
47
53
  Provides-Extra: visidata
48
54
  Requires-Dist: visidata>=3.0; extra == "visidata"
49
55
  Provides-Extra: all
@@ -72,7 +78,7 @@ uv pip install -e .
72
78
 
73
79
  ### Install from GitHub
74
80
 
75
- `uv add steve-cli`
81
+ `uv add steve-cli`
76
82
 
77
83
  OR
78
84
 
@@ -80,7 +86,14 @@ OR
80
86
  uv add git+https://github.com/your-org/tilt-ts-4.git#subdirectory=apps/example-hehnke/packages/steve-cli
81
87
  ```
82
88
 
89
+ ### dev
83
90
 
91
+ For local development add the following to your pyproj.toml which allows using the local copy instead of the one from the registry
92
+
93
+ ```toml
94
+ [tool.uv.sources]
95
+ steve-cli = { path = "../../packages/steve-cli", editable = true }
96
+ ```
84
97
 
85
98
  ## Usage
86
99
 
@@ -162,13 +175,13 @@ Running `steve extract-data` will:
162
175
 
163
176
  Steve automatically extracts metadata from files read or written via `MetadataRegistry`. The extractor is chosen by file extension — no configuration needed.
164
177
 
165
- | Extension | Extractor | Requires |
166
- |---|---|---|
167
- | `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
168
- | `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
169
- | `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
170
- | `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
171
- | anything else | `GenericExtractor` | stdlib only |
178
+ | Extension | Extractor | Requires |
179
+ | ---------------------------- | ------------------ | ------------------------------- |
180
+ | `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
181
+ | `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
182
+ | `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
183
+ | `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
184
+ | anything else | `GenericExtractor` | stdlib only |
172
185
 
173
186
  ### Adding a custom extractor
174
187
 
@@ -230,10 +243,9 @@ black steve_cli/
230
243
  isort steve_cli/
231
244
  ```
232
245
 
233
-
234
-
235
246
  # use cli
236
- source .venv/bin/activate
247
+
248
+ source .venv/bin/activate
237
249
  steve
238
250
 
239
251
  #
@@ -14,7 +14,7 @@ uv pip install -e .
14
14
 
15
15
  ### Install from GitHub
16
16
 
17
- `uv add steve-cli`
17
+ `uv add steve-cli`
18
18
 
19
19
  OR
20
20
 
@@ -22,7 +22,14 @@ OR
22
22
  uv add git+https://github.com/your-org/tilt-ts-4.git#subdirectory=apps/example-hehnke/packages/steve-cli
23
23
  ```
24
24
 
25
+ ### dev
25
26
 
27
+ For local development add the following to your pyproj.toml which allows using the local copy instead of the one from the registry
28
+
29
+ ```toml
30
+ [tool.uv.sources]
31
+ steve-cli = { path = "../../packages/steve-cli", editable = true }
32
+ ```
26
33
 
27
34
  ## Usage
28
35
 
@@ -104,13 +111,13 @@ Running `steve extract-data` will:
104
111
 
105
112
  Steve automatically extracts metadata from files read or written via `MetadataRegistry`. The extractor is chosen by file extension — no configuration needed.
106
113
 
107
- | Extension | Extractor | Requires |
108
- |---|---|---|
109
- | `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
110
- | `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
111
- | `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
112
- | `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
113
- | anything else | `GenericExtractor` | stdlib only |
114
+ | Extension | Extractor | Requires |
115
+ | ---------------------------- | ------------------ | ------------------------------- |
116
+ | `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
117
+ | `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
118
+ | `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
119
+ | `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
120
+ | anything else | `GenericExtractor` | stdlib only |
114
121
 
115
122
  ### Adding a custom extractor
116
123
 
@@ -172,10 +179,9 @@ black steve_cli/
172
179
  isort steve_cli/
173
180
  ```
174
181
 
175
-
176
-
177
182
  # use cli
178
- source .venv/bin/activate
183
+
184
+ source .venv/bin/activate
179
185
  steve
180
186
 
181
187
  #
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
5
5
 
6
6
  [project]
7
7
  name = "steve-cli"
8
- version = "0.3.20"
8
+ version = "0.4.1"
9
9
  description = "A simple CLI tool to run jobs from jobs.yaml with proper environment setup"
10
10
  readme = "README.md"
11
11
  license = {text = "MIT"}
@@ -60,6 +60,13 @@ great-expectations = [
60
60
  excel = [
61
61
  "openpyxl>=3.1.0",
62
62
  ]
63
+ trino = [
64
+ "trino>=0.330",
65
+ "pyarrow>=17.0.0",
66
+ "requests>=2.31.0",
67
+ "pyiceberg>=0.7.0",
68
+ "s3fs>=2024.1.0",
69
+ ]
63
70
  visidata = [
64
71
  "visidata>=3.0",
65
72
  ]
@@ -781,6 +781,126 @@ def buckets(env_file: tuple):
781
781
  click.secho(f"❌ Could not open file: {e}", fg="red", err=True)
782
782
 
783
783
 
784
+ def _list_trino_tables(storage_kwargs: dict, label: str) -> tuple:
785
+ click.echo(f" {click.style(label, fg='cyan')}")
786
+ try:
787
+ from steve_cli.storage.trino import TrinoStorage
788
+ storage = TrinoStorage(**storage_kwargs)
789
+ tables = storage.list()
790
+ if not tables:
791
+ click.echo(" (no tables)")
792
+ else:
793
+ for t in tables:
794
+ click.echo(f" ├── {t}")
795
+ return storage, tables
796
+ except EnvironmentError as e:
797
+ click.echo(f" ⚠️ {e}", err=True)
798
+ except Exception as e:
799
+ click.echo(f" ❌ {e}", err=True)
800
+ return None, []
801
+
802
+
803
+ @main.command("tables")
804
+ @click.option('--env-file', '-e', type=click.Path(path_type=Path), multiple=True,
805
+ help='Path to .env file(s). Can be specified multiple times. Defaults to .env and .workspaces.env')
806
+ def tables(env_file: tuple):
807
+ """List Iceberg tables via Trino and view their contents."""
808
+ if not os.getenv("TRINO_ENDPOINT"):
809
+ click.secho("TRINO_ENDPOINT is not set — Trino is not available.", fg="yellow")
810
+ return
811
+
812
+ cwd = Path.cwd()
813
+ env_files = [Path(f) for f in env_file] if env_file else [cwd / ".env", cwd / ".workspaces.env"]
814
+ for ef in env_files:
815
+ load_dotenv(ef)
816
+
817
+ tiers = ["bronze", "silver", "gold"]
818
+ options: List[Dict[str, Any]] = []
819
+
820
+ bare_tiers = [t for t in tiers if os.getenv(f"{t.upper()}_ACCESS_KEY")]
821
+ for tier in bare_tiers:
822
+ options.append({
823
+ "label": f"default / {tier}",
824
+ "kwargs": {"tier": tier},
825
+ })
826
+
827
+ for ws in _detect_workspaces():
828
+ for tier in tiers:
829
+ if not os.getenv(f"{ws}_ACCESS_KEY_{tier.upper()}") and not os.getenv(f"{ws}_ACCESS_KEY"):
830
+ continue
831
+ options.append({
832
+ "label": f"{ws} / {tier}",
833
+ "kwargs": {"tier": tier, "workspace": ws.lower().replace("_", "-")},
834
+ })
835
+
836
+ if not options:
837
+ click.echo("No Trino storage env variables found (expected: TRINO_ENDPOINT + BRONZE_ACCESS_KEY or {WORKSPACE}_ACCESS_KEY).")
838
+ return
839
+
840
+ choice = questionary.select(
841
+ "Select a schema to list:",
842
+ choices=[o["label"] for o in options],
843
+ ).ask()
844
+
845
+ if choice is None:
846
+ sys.exit(0)
847
+
848
+ selected = next(o for o in options if o["label"] == choice)
849
+ storage, table_names = _list_trino_tables(selected["kwargs"], selected["label"])
850
+
851
+ if not storage or not table_names:
852
+ return
853
+
854
+ table_choice = questionary.select(
855
+ "View a table (or press Esc to exit):",
856
+ choices=["(done)"] + table_names,
857
+ ).ask()
858
+
859
+ if not table_choice or table_choice == "(done)":
860
+ return
861
+
862
+ click.echo(f"\n📊 {click.style(table_choice, fg='cyan')}\n")
863
+ try:
864
+ import tempfile
865
+ data = storage.get_bytes(table_choice)
866
+ with tempfile.NamedTemporaryFile(suffix=".parquet", delete=False) as tmp:
867
+ tmp.write(data)
868
+ tmp_path = tmp.name
869
+ if not shutil.which("vd"):
870
+ click.secho("visidata not found. Install it with: uv pip install 'steve-cli[visidata]'", fg="yellow")
871
+ return
872
+ subprocess.call(["vd", tmp_path])
873
+ except Exception as e:
874
+ click.secho(f"❌ Could not open table: {e}", fg="red", err=True)
875
+
876
+
877
+ @main.group()
878
+ def policies():
879
+ """Manage and apply access policies via the Policy Control Plane."""
880
+ pass
881
+
882
+
883
+ @policies.command("apply")
884
+ @click.option('--file', '-f', type=click.Path(path_type=Path),
885
+ help='Path to policies YAML file (default: ./policies/access.yaml)')
886
+ def policies_apply(file: Path | None):
887
+ """Apply access policies from a YAML file to the Policy Control Plane."""
888
+ from dotenv import load_dotenv
889
+ load_dotenv(".env", override=False)
890
+
891
+ from steve_cli.policies import PolicyClient
892
+ policies_file = Path(file) if file else None
893
+ try:
894
+ PolicyClient().apply_from_file(policies_file)
895
+ click.secho("Policies applied successfully.", fg="green")
896
+ except FileNotFoundError as e:
897
+ click.secho(str(e), fg="red", err=True)
898
+ sys.exit(1)
899
+ except Exception as e:
900
+ click.secho(f"ERROR: {e}", fg="red", bold=True, err=True)
901
+ sys.exit(1)
902
+
903
+
784
904
  @main.command("upgrade")
785
905
  def upgrade():
786
906
  """Upgrade steve-cli to the latest version."""
@@ -0,0 +1,10 @@
1
+ from collections.abc import Callable
2
+
3
+ from steve_cli.lineage.storage import LineageStorage
4
+
5
+ from .lineage_job import lineage_job
6
+
7
+ GetTables = Callable[[str, str | None], LineageStorage]
8
+ GetStorage = Callable[[str, str | None], LineageStorage]
9
+
10
+ __all__ = ["GetStorage", "GetTables", "lineage_job"]
@@ -7,6 +7,7 @@ from typing import Any, Callable
7
7
 
8
8
  from steve_cli.lineage.collector import make_session
9
9
  from steve_cli.storage.s3 import S3Storage
10
+ from steve_cli.storage.trino import TrinoStorage
10
11
  from steve_cli.lineage.storage import LineageStorage
11
12
 
12
13
 
@@ -48,14 +49,24 @@ def lineage_job(
48
49
  session.namespace = ls._dataset_namespace
49
50
  return ls
50
51
 
52
+ def get_tables(tier: str = "bronze", workspace: str | None = None) -> LineageStorage:
53
+ return LineageStorage(
54
+ storage=lambda: TrinoStorage(tier=tier, workspace=workspace),
55
+ session=session,
56
+ )
57
+
58
+ sig = inspect.signature(fn)
59
+ available = {"get_storage": get_storage, "get_tables": get_tables}
60
+ injectable = {k: v for k, v in available.items() if k in sig.parameters}
61
+
51
62
  try:
52
- result = fn(*args, get_storage=get_storage, **kwargs)
63
+ result = fn(*args, **injectable, **kwargs)
53
64
  except Exception as exc:
54
65
  session.fail(exc)
55
66
  import sys
67
+ import click
56
68
  from steve_cli.validation.port import DataQualityError
57
69
  if isinstance(exc, (DataQualityError, EnvironmentError)):
58
- import click
59
70
  click.secho(f"ERROR {exc}", fg="red", bold=True, err=True)
60
71
  sys.exit(1)
61
72
  raise
@@ -0,0 +1,3 @@
1
+ from .client import PolicyClient
2
+
3
+ __all__ = ["PolicyClient"]
@@ -0,0 +1,89 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import os
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+ import httpx
9
+
10
+ _POLICY_VERSION = "v0"
11
+
12
+
13
+ class PolicyClient:
14
+ def __init__(
15
+ self,
16
+ endpoint: str | None = None,
17
+ token: str | None = None,
18
+ ):
19
+ self.endpoint = (endpoint or os.environ.get("PCP_ENDPOINT", "http://policy-control-plane:3010")).rstrip("/")
20
+ self.token = token or os.environ.get("PCP_TOKEN", "")
21
+
22
+ def __trpc(self, method: str, path: str, payload: dict | None = None) -> dict:
23
+ url = f"{self.endpoint}/api/trpc/{path}"
24
+ headers = {"Authorization": f"Bearer {self.token}"} if self.token else {}
25
+ if method == "query":
26
+ resp = httpx.get(url, params={"input": json.dumps(payload or {})}, headers=headers, timeout=10)
27
+ else:
28
+ resp = httpx.post(url, json=payload or {}, headers=headers, timeout=10)
29
+ resp.raise_for_status()
30
+ data = resp.json()
31
+ if "error" in data:
32
+ raise RuntimeError(f"tRPC error on {path}: {data['error']}")
33
+ return data.get("result", {}).get("data", data.get("result", {}))
34
+
35
+ def _trpc(self, method: str, path: str, payload: dict | None = None) -> dict:
36
+ try:
37
+ return self.__trpc(method, path, payload)
38
+ except httpx.ConnectError as e:
39
+ raise EnvironmentError(
40
+ f"Cannot reach Policy Control Plane at {self.endpoint} — "
41
+ f"set PCP_ENDPOINT in your .env"
42
+ ) from e
43
+
44
+ def register_pdp(self, name: str, endpoint: str, pdp_type: str = "opa", is_default: bool = True) -> dict:
45
+ return self._trpc("mutation", "pdp.register", {
46
+ "name": name,
47
+ "type": pdp_type,
48
+ "endpoint": endpoint,
49
+ "isDefault": is_default,
50
+ })
51
+
52
+ def attach_pdp(self, workspace: str, pdp_id: str) -> None:
53
+ self._trpc("mutation", "workspace.attachPdp", {"workspace": workspace, "pdpId": pdp_id})
54
+
55
+ def apply_policies(self, workspace: str, pdp_name: str, rules: list[dict[str, Any]]) -> dict:
56
+ return self._trpc("mutation", "workspace.applyPolicies", {
57
+ "workspace": workspace,
58
+ "fragments": [
59
+ {
60
+ "syntax": "dsl",
61
+ "target": pdp_name,
62
+ "content": {"rules": rules},
63
+ }
64
+ ],
65
+ })
66
+
67
+ def apply_from_file(self, policies_file: Path | None = None) -> None:
68
+ path = policies_file or Path.cwd() / "policies" / "access.yaml"
69
+ if not path.exists():
70
+ raise FileNotFoundError(f"Policies file not found: {path}")
71
+
72
+ import yaml
73
+ config: dict[str, Any] = yaml.safe_load(path.read_text())
74
+ workspace = config["workspace"]
75
+ pdp_cfg = config["pdp"]
76
+ rules = config.get("rules", [])
77
+
78
+ pdp = self.register_pdp(
79
+ name=pdp_cfg["name"],
80
+ endpoint=pdp_cfg["endpoint"],
81
+ pdp_type=pdp_cfg.get("type", "opa"),
82
+ )
83
+ self.attach_pdp(workspace, pdp["id"])
84
+ result = self.apply_policies(workspace, pdp_cfg["name"], rules)
85
+
86
+ errors = result.get("errors", [])
87
+ if errors:
88
+ reasons = "; ".join(e["reason"] for e in errors)
89
+ raise RuntimeError(f"Failed to apply {len(errors)} policy fragment(s): {reasons}")
@@ -1,11 +1,13 @@
1
1
  from .protocol import Storage
2
2
  from .s3 import S3Storage
3
+ from .trino import TrinoStorage
3
4
  from .parquet import ParquetMetadata, extract_parquet_metadata
4
5
  from .metadata import FileMetadata, ColumnMetadata, MetadataExtractorPort, MetadataRegistry
5
6
 
6
7
  __all__ = [
7
8
  "Storage",
8
9
  "S3Storage",
10
+ "TrinoStorage",
9
11
  "ParquetMetadata",
10
12
  "extract_parquet_metadata",
11
13
  "FileMetadata",
@@ -0,0 +1,199 @@
1
+ from __future__ import annotations
2
+
3
+ import io
4
+ import logging
5
+ import os
6
+ import time
7
+ from pathlib import Path
8
+
9
+ import pyarrow as pa
10
+ import pyarrow.parquet as pq
11
+ import requests
12
+
13
+ logger = logging.getLogger(__name__)
14
+
15
+
16
+ def _derive_schema(tier: str, workspace: str | None) -> str | None:
17
+ if workspace:
18
+ prefix = workspace.lower().replace("-", "_")
19
+ return f"ws_{prefix}_{tier.lower()}"
20
+ return None
21
+
22
+
23
+ class TrinoStorage:
24
+ def __init__(self, tier: str = "bronze", workspace: str | None = None):
25
+ trino_endpoint = os.getenv("TRINO_ENDPOINT")
26
+ if not trino_endpoint:
27
+ raise EnvironmentError("TRINO_ENDPOINT is not set — Trino is not available in this environment")
28
+ self.catalog = os.getenv("TRINO_CATALOG", "minio")
29
+ resolved_workspace = workspace or os.getenv("WORKSPACE_NAME")
30
+ self.schema = _derive_schema(tier, resolved_workspace) or os.environ.get("TRINO_SCHEMA", "")
31
+ if not self.schema:
32
+ raise EnvironmentError("TRINO_SCHEMA is not set and no WORKSPACE_NAME or workspace was provided")
33
+ self.user = os.getenv("TRINO_USER", "admin")
34
+ self._base = trino_endpoint.rstrip("/")
35
+ self._lakekeeper_endpoint = os.getenv("LAKEKEEPER_ENDPOINT")
36
+ default_warehouse = f"{resolved_workspace}-{tier}" if resolved_workspace else "minio"
37
+ self._lakekeeper_warehouse = os.getenv("LAKEKEEPER_WAREHOUSE", default_warehouse)
38
+ self.__iceberg_catalog = None
39
+
40
+ @property
41
+ def _iceberg_catalog(self):
42
+ if self.__iceberg_catalog is None:
43
+ from pyiceberg.catalog.rest import RestCatalog
44
+ from pyiceberg.io import load_file_io
45
+ from pyiceberg.table import Table
46
+
47
+ # FIXME: Interim workaround — pyiceberg cannot use Lakekeeper's remote signing
48
+ # endpoint when running outside the cluster (hostname 'lakekeeper' doesn't resolve
49
+ # locally). We subclass RestCatalog to inject our local signer URI after pyiceberg
50
+ # merges table config (which overwrites s3.signer.uri with the internal hostname).
51
+ # Remove once Increment 5 (STS credential vending) is wired.
52
+ tier = self.schema.rsplit("_", 1)[-1].upper()
53
+ lakekeeper_base = self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog"
54
+ s3_overrides = {
55
+ "s3.endpoint": os.getenv("S3_ENDPOINT", "http://minio:9000"),
56
+ "s3.access-key-id": os.getenv(f"{tier}_ACCESS_KEY") or os.getenv("AWS_ACCESS_KEY_ID", ""),
57
+ "s3.secret-access-key": os.getenv(f"{tier}_SECRET_KEY") or os.getenv("AWS_SECRET_ACCESS_KEY", ""),
58
+ "s3.path-style-access": "true",
59
+ "s3.signer.uri": lakekeeper_base,
60
+ } if self._lakekeeper_endpoint else {}
61
+
62
+ class _PatchedRestCatalog(RestCatalog):
63
+ def _response_to_table(self_, identifier_tuple, table_response):
64
+ merged = {**table_response.metadata.properties, **table_response.config, **s3_overrides, "uri": lakekeeper_base}
65
+ return Table(
66
+ identifier=identifier_tuple,
67
+ metadata_location=table_response.metadata_location,
68
+ metadata=table_response.metadata,
69
+ io=load_file_io(merged, table_response.metadata_location),
70
+ catalog=self_,
71
+ config=table_response.config,
72
+ )
73
+
74
+ catalog = _PatchedRestCatalog(
75
+ "lakekeeper",
76
+ uri=lakekeeper_base,
77
+ warehouse=self._lakekeeper_warehouse,
78
+ **s3_overrides,
79
+ )
80
+ if self._lakekeeper_endpoint:
81
+ catalog.uri = lakekeeper_base
82
+ self.__iceberg_catalog = catalog
83
+ return self.__iceberg_catalog
84
+
85
+ def _execute(self, sql: str) -> list[dict]:
86
+ headers = {
87
+ "X-Trino-User": self.user,
88
+ "X-Trino-Catalog": self.catalog,
89
+ "X-Trino-Schema": self.schema,
90
+ }
91
+ resp = requests.post(f"{self._base}/v1/statement", data=sql, headers=headers)
92
+ resp.raise_for_status()
93
+ data = resp.json()
94
+ rows: list[dict] = []
95
+ while True:
96
+ if "data" in data and "columns" in data:
97
+ col_names = [c["name"] for c in data["columns"]]
98
+ for row in data["data"]:
99
+ rows.append(dict(zip(col_names, row)))
100
+ next_uri = data.get("nextUri")
101
+ if not next_uri:
102
+ break
103
+ time.sleep(0.1)
104
+ resp = requests.get(next_uri, headers=headers)
105
+ resp.raise_for_status()
106
+ data = resp.json()
107
+ return rows
108
+
109
+ @staticmethod
110
+ def _table_name(path: str) -> str:
111
+ return path.strip("/").replace("/", "_").replace(".", "_")
112
+
113
+ def list(self, prefix: str = "") -> list[str]:
114
+ rows = self._execute(f"SHOW TABLES FROM {self.catalog}.{self.schema}")
115
+ tables = [r["Table"] for r in rows]
116
+ if prefix:
117
+ clean = prefix.strip("/")
118
+ tables = [t for t in tables if t.startswith(clean)]
119
+ return tables
120
+
121
+ def list_all(self) -> list[str]:
122
+ return self.list()
123
+
124
+ def get_bytes(self, path: str) -> bytes:
125
+ rows = self._execute(f"SELECT * FROM {self.catalog}.{self.schema}.{self._table_name(path)}")
126
+ if not rows:
127
+ return b""
128
+ arrow_table = pa.Table.from_pylist(rows)
129
+ buf = io.BytesIO()
130
+ pq.write_table(arrow_table, buf)
131
+ return buf.getvalue()
132
+
133
+ def get_file(self, path: str, local_path: str) -> None:
134
+ Path(local_path).parent.mkdir(parents=True, exist_ok=True)
135
+ Path(local_path).write_bytes(self.get_bytes(path))
136
+
137
+ def _ensure_namespace(self) -> None:
138
+ from pyiceberg.exceptions import NamespaceAlreadyExistsError
139
+ try:
140
+ self._iceberg_catalog.create_namespace(self.schema)
141
+ logger.info("Created namespace %s", self.schema)
142
+ except NamespaceAlreadyExistsError:
143
+ pass
144
+
145
+ def put_bytes(self, data: bytes, path: str) -> None:
146
+ from pyiceberg.exceptions import NoSuchTableError
147
+
148
+ arrow_table = pq.read_table(io.BytesIO(data))
149
+ table_id = f"{self.schema}.{self._table_name(path)}"
150
+ catalog = self._iceberg_catalog
151
+
152
+ self._ensure_namespace()
153
+
154
+ try:
155
+ iceberg_table = catalog.load_table(table_id)
156
+ iceberg_table.append(arrow_table)
157
+ except NoSuchTableError:
158
+ from pyiceberg.schema import Schema
159
+ from pyiceberg.types import (
160
+ BinaryType,
161
+ BooleanType,
162
+ DateType,
163
+ DoubleType,
164
+ FloatType,
165
+ IntegerType,
166
+ LongType,
167
+ NestedField,
168
+ StringType,
169
+ TimestampType,
170
+ )
171
+
172
+ _PYARROW_TO_ICEBERG = {
173
+ pa.bool_(): BooleanType(),
174
+ pa.int32(): IntegerType(),
175
+ pa.int64(): LongType(),
176
+ pa.float32(): FloatType(),
177
+ pa.float64(): DoubleType(),
178
+ pa.large_utf8(): StringType(),
179
+ pa.utf8(): StringType(),
180
+ pa.large_binary(): BinaryType(),
181
+ pa.binary(): BinaryType(),
182
+ pa.date32(): DateType(),
183
+ pa.timestamp("us"): TimestampType(),
184
+ pa.timestamp("us", tz="UTC"): TimestampType(),
185
+ }
186
+
187
+ fields = []
188
+ for i, field in enumerate(arrow_table.schema):
189
+ iceberg_type = _PYARROW_TO_ICEBERG.get(field.type, StringType())
190
+ fields.append(NestedField(field_id=i + 1, name=field.name, field_type=iceberg_type, required=not field.nullable))
191
+
192
+ iceberg_schema = Schema(*fields)
193
+ iceberg_table = catalog.create_table(table_id, schema=iceberg_schema)
194
+ iceberg_table.append(arrow_table)
195
+
196
+ logger.info("Written %d rows to %s", len(arrow_table), table_id)
197
+
198
+ def put_file(self, local_path: str, path: str) -> None:
199
+ self.put_bytes(Path(local_path).read_bytes(), path)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: steve-cli
3
- Version: 0.3.20
3
+ Version: 0.4.1
4
4
  Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
5
5
  Author: Frank
6
6
  License: MIT
@@ -44,6 +44,12 @@ Requires-Dist: great-expectations>=0.18.0; extra == "great-expectations"
44
44
  Requires-Dist: pandas>=2.0.3; extra == "great-expectations"
45
45
  Provides-Extra: excel
46
46
  Requires-Dist: openpyxl>=3.1.0; extra == "excel"
47
+ Provides-Extra: trino
48
+ Requires-Dist: trino>=0.330; extra == "trino"
49
+ Requires-Dist: pyarrow>=17.0.0; extra == "trino"
50
+ Requires-Dist: requests>=2.31.0; extra == "trino"
51
+ Requires-Dist: pyiceberg>=0.7.0; extra == "trino"
52
+ Requires-Dist: s3fs>=2024.1.0; extra == "trino"
47
53
  Provides-Extra: visidata
48
54
  Requires-Dist: visidata>=3.0; extra == "visidata"
49
55
  Provides-Extra: all
@@ -72,7 +78,7 @@ uv pip install -e .
72
78
 
73
79
  ### Install from GitHub
74
80
 
75
- `uv add steve-cli`
81
+ `uv add steve-cli`
76
82
 
77
83
  OR
78
84
 
@@ -80,7 +86,14 @@ OR
80
86
  uv add git+https://github.com/your-org/tilt-ts-4.git#subdirectory=apps/example-hehnke/packages/steve-cli
81
87
  ```
82
88
 
89
+ ### dev
83
90
 
91
+ For local development add the following to your pyproj.toml which allows using the local copy instead of the one from the registry
92
+
93
+ ```toml
94
+ [tool.uv.sources]
95
+ steve-cli = { path = "../../packages/steve-cli", editable = true }
96
+ ```
84
97
 
85
98
  ## Usage
86
99
 
@@ -162,13 +175,13 @@ Running `steve extract-data` will:
162
175
 
163
176
  Steve automatically extracts metadata from files read or written via `MetadataRegistry`. The extractor is chosen by file extension — no configuration needed.
164
177
 
165
- | Extension | Extractor | Requires |
166
- |---|---|---|
167
- | `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
168
- | `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
169
- | `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
170
- | `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
171
- | anything else | `GenericExtractor` | stdlib only |
178
+ | Extension | Extractor | Requires |
179
+ | ---------------------------- | ------------------ | ------------------------------- |
180
+ | `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
181
+ | `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
182
+ | `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
183
+ | `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
184
+ | anything else | `GenericExtractor` | stdlib only |
172
185
 
173
186
  ### Adding a custom extractor
174
187
 
@@ -230,10 +243,9 @@ black steve_cli/
230
243
  isort steve_cli/
231
244
  ```
232
245
 
233
-
234
-
235
246
  # use cli
236
- source .venv/bin/activate
247
+
248
+ source .venv/bin/activate
237
249
  steve
238
250
 
239
251
  #
@@ -20,10 +20,13 @@ steve_cli/lineage/adapters/__init__.py
20
20
  steve_cli/lineage/adapters/logging.py
21
21
  steve_cli/lineage/adapters/null.py
22
22
  steve_cli/lineage/adapters/openlineage.py
23
+ steve_cli/policies/__init__.py
24
+ steve_cli/policies/client.py
23
25
  steve_cli/storage/__init__.py
24
26
  steve_cli/storage/parquet.py
25
27
  steve_cli/storage/protocol.py
26
28
  steve_cli/storage/s3.py
29
+ steve_cli/storage/trino.py
27
30
  steve_cli/storage/metadata/__init__.py
28
31
  steve_cli/storage/metadata/port.py
29
32
  steve_cli/storage/metadata/registry.py
@@ -41,6 +41,13 @@ pyarrow>=17.0.0
41
41
  polars>=1.8.2
42
42
  pyarrow>=17.0.0
43
43
 
44
+ [trino]
45
+ trino>=0.330
46
+ pyarrow>=17.0.0
47
+ requests>=2.31.0
48
+ pyiceberg>=0.7.0
49
+ s3fs>=2024.1.0
50
+
44
51
  [validoopsie]
45
52
  validoopsie>=0.1.0
46
53
  polars>=1.8.2
@@ -1,3 +0,0 @@
1
- from .lineage_job import lineage_job
2
-
3
- __all__ = ["lineage_job"]
File without changes