steve-cli 0.3.20__tar.gz → 0.4.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {steve_cli-0.3.20 → steve_cli-0.4.1}/PKG-INFO +24 -12
- {steve_cli-0.3.20 → steve_cli-0.4.1}/README.md +17 -11
- {steve_cli-0.3.20 → steve_cli-0.4.1}/pyproject.toml +8 -1
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/cli.py +120 -0
- steve_cli-0.4.1/steve_cli/decorators/__init__.py +10 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/decorators/lineage_job.py +13 -2
- steve_cli-0.4.1/steve_cli/policies/__init__.py +3 -0
- steve_cli-0.4.1/steve_cli/policies/client.py +89 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/__init__.py +2 -0
- steve_cli-0.4.1/steve_cli/storage/trino.py +199 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/PKG-INFO +24 -12
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/SOURCES.txt +3 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/requires.txt +7 -0
- steve_cli-0.3.20/steve_cli/decorators/__init__.py +0 -3
- {steve_cli-0.3.20 → steve_cli-0.4.1}/setup.cfg +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/__init__.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/__init__.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/__init__.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/logging.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/null.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/adapters/openlineage.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/collector.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/port.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/registry.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/lineage/storage.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/__init__.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/__init__.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/csv.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/excel.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/generic.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/json.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/extractors/parquet.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/port.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/metadata/registry.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/parquet.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/protocol.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage/s3.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/storage.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/__init__.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/__init__.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/great_expectations.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/null.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/adapters/validoopsie.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/port.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli/validation/registry.py +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/dependency_links.txt +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/entry_points.txt +0 -0
- {steve_cli-0.3.20 → steve_cli-0.4.1}/steve_cli.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: steve-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
|
|
5
5
|
Author: Frank
|
|
6
6
|
License: MIT
|
|
@@ -44,6 +44,12 @@ Requires-Dist: great-expectations>=0.18.0; extra == "great-expectations"
|
|
|
44
44
|
Requires-Dist: pandas>=2.0.3; extra == "great-expectations"
|
|
45
45
|
Provides-Extra: excel
|
|
46
46
|
Requires-Dist: openpyxl>=3.1.0; extra == "excel"
|
|
47
|
+
Provides-Extra: trino
|
|
48
|
+
Requires-Dist: trino>=0.330; extra == "trino"
|
|
49
|
+
Requires-Dist: pyarrow>=17.0.0; extra == "trino"
|
|
50
|
+
Requires-Dist: requests>=2.31.0; extra == "trino"
|
|
51
|
+
Requires-Dist: pyiceberg>=0.7.0; extra == "trino"
|
|
52
|
+
Requires-Dist: s3fs>=2024.1.0; extra == "trino"
|
|
47
53
|
Provides-Extra: visidata
|
|
48
54
|
Requires-Dist: visidata>=3.0; extra == "visidata"
|
|
49
55
|
Provides-Extra: all
|
|
@@ -72,7 +78,7 @@ uv pip install -e .
|
|
|
72
78
|
|
|
73
79
|
### Install from GitHub
|
|
74
80
|
|
|
75
|
-
`uv add steve-cli`
|
|
81
|
+
`uv add steve-cli`
|
|
76
82
|
|
|
77
83
|
OR
|
|
78
84
|
|
|
@@ -80,7 +86,14 @@ OR
|
|
|
80
86
|
uv add git+https://github.com/your-org/tilt-ts-4.git#subdirectory=apps/example-hehnke/packages/steve-cli
|
|
81
87
|
```
|
|
82
88
|
|
|
89
|
+
### dev
|
|
83
90
|
|
|
91
|
+
For local development add the following to your pyproj.toml which allows using the local copy instead of the one from the registry
|
|
92
|
+
|
|
93
|
+
```toml
|
|
94
|
+
[tool.uv.sources]
|
|
95
|
+
steve-cli = { path = "../../packages/steve-cli", editable = true }
|
|
96
|
+
```
|
|
84
97
|
|
|
85
98
|
## Usage
|
|
86
99
|
|
|
@@ -162,13 +175,13 @@ Running `steve extract-data` will:
|
|
|
162
175
|
|
|
163
176
|
Steve automatically extracts metadata from files read or written via `MetadataRegistry`. The extractor is chosen by file extension — no configuration needed.
|
|
164
177
|
|
|
165
|
-
| Extension
|
|
166
|
-
|
|
167
|
-
| `.parquet`, `.pq`
|
|
168
|
-
| `.csv`, `.tsv`, `.txt`
|
|
169
|
-
| `.json`, `.jsonl`, `.ndjson` | `JsonExtractor`
|
|
170
|
-
| `.xlsx`, `.xls`, `.xlsm`
|
|
171
|
-
| anything else
|
|
178
|
+
| Extension | Extractor | Requires |
|
|
179
|
+
| ---------------------------- | ------------------ | ------------------------------- |
|
|
180
|
+
| `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
|
|
181
|
+
| `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
|
|
182
|
+
| `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
|
|
183
|
+
| `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
|
|
184
|
+
| anything else | `GenericExtractor` | stdlib only |
|
|
172
185
|
|
|
173
186
|
### Adding a custom extractor
|
|
174
187
|
|
|
@@ -230,10 +243,9 @@ black steve_cli/
|
|
|
230
243
|
isort steve_cli/
|
|
231
244
|
```
|
|
232
245
|
|
|
233
|
-
|
|
234
|
-
|
|
235
246
|
# use cli
|
|
236
|
-
|
|
247
|
+
|
|
248
|
+
source .venv/bin/activate
|
|
237
249
|
steve
|
|
238
250
|
|
|
239
251
|
#
|
|
@@ -14,7 +14,7 @@ uv pip install -e .
|
|
|
14
14
|
|
|
15
15
|
### Install from GitHub
|
|
16
16
|
|
|
17
|
-
`uv add steve-cli`
|
|
17
|
+
`uv add steve-cli`
|
|
18
18
|
|
|
19
19
|
OR
|
|
20
20
|
|
|
@@ -22,7 +22,14 @@ OR
|
|
|
22
22
|
uv add git+https://github.com/your-org/tilt-ts-4.git#subdirectory=apps/example-hehnke/packages/steve-cli
|
|
23
23
|
```
|
|
24
24
|
|
|
25
|
+
### dev
|
|
25
26
|
|
|
27
|
+
For local development add the following to your pyproj.toml which allows using the local copy instead of the one from the registry
|
|
28
|
+
|
|
29
|
+
```toml
|
|
30
|
+
[tool.uv.sources]
|
|
31
|
+
steve-cli = { path = "../../packages/steve-cli", editable = true }
|
|
32
|
+
```
|
|
26
33
|
|
|
27
34
|
## Usage
|
|
28
35
|
|
|
@@ -104,13 +111,13 @@ Running `steve extract-data` will:
|
|
|
104
111
|
|
|
105
112
|
Steve automatically extracts metadata from files read or written via `MetadataRegistry`. The extractor is chosen by file extension — no configuration needed.
|
|
106
113
|
|
|
107
|
-
| Extension
|
|
108
|
-
|
|
109
|
-
| `.parquet`, `.pq`
|
|
110
|
-
| `.csv`, `.tsv`, `.txt`
|
|
111
|
-
| `.json`, `.jsonl`, `.ndjson` | `JsonExtractor`
|
|
112
|
-
| `.xlsx`, `.xls`, `.xlsm`
|
|
113
|
-
| anything else
|
|
114
|
+
| Extension | Extractor | Requires |
|
|
115
|
+
| ---------------------------- | ------------------ | ------------------------------- |
|
|
116
|
+
| `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
|
|
117
|
+
| `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
|
|
118
|
+
| `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
|
|
119
|
+
| `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
|
|
120
|
+
| anything else | `GenericExtractor` | stdlib only |
|
|
114
121
|
|
|
115
122
|
### Adding a custom extractor
|
|
116
123
|
|
|
@@ -172,10 +179,9 @@ black steve_cli/
|
|
|
172
179
|
isort steve_cli/
|
|
173
180
|
```
|
|
174
181
|
|
|
175
|
-
|
|
176
|
-
|
|
177
182
|
# use cli
|
|
178
|
-
|
|
183
|
+
|
|
184
|
+
source .venv/bin/activate
|
|
179
185
|
steve
|
|
180
186
|
|
|
181
187
|
#
|
|
@@ -5,7 +5,7 @@ build-backend = "setuptools.build_meta"
|
|
|
5
5
|
|
|
6
6
|
[project]
|
|
7
7
|
name = "steve-cli"
|
|
8
|
-
version = "0.
|
|
8
|
+
version = "0.4.1"
|
|
9
9
|
description = "A simple CLI tool to run jobs from jobs.yaml with proper environment setup"
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
license = {text = "MIT"}
|
|
@@ -60,6 +60,13 @@ great-expectations = [
|
|
|
60
60
|
excel = [
|
|
61
61
|
"openpyxl>=3.1.0",
|
|
62
62
|
]
|
|
63
|
+
trino = [
|
|
64
|
+
"trino>=0.330",
|
|
65
|
+
"pyarrow>=17.0.0",
|
|
66
|
+
"requests>=2.31.0",
|
|
67
|
+
"pyiceberg>=0.7.0",
|
|
68
|
+
"s3fs>=2024.1.0",
|
|
69
|
+
]
|
|
63
70
|
visidata = [
|
|
64
71
|
"visidata>=3.0",
|
|
65
72
|
]
|
|
@@ -781,6 +781,126 @@ def buckets(env_file: tuple):
|
|
|
781
781
|
click.secho(f"❌ Could not open file: {e}", fg="red", err=True)
|
|
782
782
|
|
|
783
783
|
|
|
784
|
+
def _list_trino_tables(storage_kwargs: dict, label: str) -> tuple:
|
|
785
|
+
click.echo(f" {click.style(label, fg='cyan')}")
|
|
786
|
+
try:
|
|
787
|
+
from steve_cli.storage.trino import TrinoStorage
|
|
788
|
+
storage = TrinoStorage(**storage_kwargs)
|
|
789
|
+
tables = storage.list()
|
|
790
|
+
if not tables:
|
|
791
|
+
click.echo(" (no tables)")
|
|
792
|
+
else:
|
|
793
|
+
for t in tables:
|
|
794
|
+
click.echo(f" ├── {t}")
|
|
795
|
+
return storage, tables
|
|
796
|
+
except EnvironmentError as e:
|
|
797
|
+
click.echo(f" ⚠️ {e}", err=True)
|
|
798
|
+
except Exception as e:
|
|
799
|
+
click.echo(f" ❌ {e}", err=True)
|
|
800
|
+
return None, []
|
|
801
|
+
|
|
802
|
+
|
|
803
|
+
@main.command("tables")
|
|
804
|
+
@click.option('--env-file', '-e', type=click.Path(path_type=Path), multiple=True,
|
|
805
|
+
help='Path to .env file(s). Can be specified multiple times. Defaults to .env and .workspaces.env')
|
|
806
|
+
def tables(env_file: tuple):
|
|
807
|
+
"""List Iceberg tables via Trino and view their contents."""
|
|
808
|
+
if not os.getenv("TRINO_ENDPOINT"):
|
|
809
|
+
click.secho("TRINO_ENDPOINT is not set — Trino is not available.", fg="yellow")
|
|
810
|
+
return
|
|
811
|
+
|
|
812
|
+
cwd = Path.cwd()
|
|
813
|
+
env_files = [Path(f) for f in env_file] if env_file else [cwd / ".env", cwd / ".workspaces.env"]
|
|
814
|
+
for ef in env_files:
|
|
815
|
+
load_dotenv(ef)
|
|
816
|
+
|
|
817
|
+
tiers = ["bronze", "silver", "gold"]
|
|
818
|
+
options: List[Dict[str, Any]] = []
|
|
819
|
+
|
|
820
|
+
bare_tiers = [t for t in tiers if os.getenv(f"{t.upper()}_ACCESS_KEY")]
|
|
821
|
+
for tier in bare_tiers:
|
|
822
|
+
options.append({
|
|
823
|
+
"label": f"default / {tier}",
|
|
824
|
+
"kwargs": {"tier": tier},
|
|
825
|
+
})
|
|
826
|
+
|
|
827
|
+
for ws in _detect_workspaces():
|
|
828
|
+
for tier in tiers:
|
|
829
|
+
if not os.getenv(f"{ws}_ACCESS_KEY_{tier.upper()}") and not os.getenv(f"{ws}_ACCESS_KEY"):
|
|
830
|
+
continue
|
|
831
|
+
options.append({
|
|
832
|
+
"label": f"{ws} / {tier}",
|
|
833
|
+
"kwargs": {"tier": tier, "workspace": ws.lower().replace("_", "-")},
|
|
834
|
+
})
|
|
835
|
+
|
|
836
|
+
if not options:
|
|
837
|
+
click.echo("No Trino storage env variables found (expected: TRINO_ENDPOINT + BRONZE_ACCESS_KEY or {WORKSPACE}_ACCESS_KEY).")
|
|
838
|
+
return
|
|
839
|
+
|
|
840
|
+
choice = questionary.select(
|
|
841
|
+
"Select a schema to list:",
|
|
842
|
+
choices=[o["label"] for o in options],
|
|
843
|
+
).ask()
|
|
844
|
+
|
|
845
|
+
if choice is None:
|
|
846
|
+
sys.exit(0)
|
|
847
|
+
|
|
848
|
+
selected = next(o for o in options if o["label"] == choice)
|
|
849
|
+
storage, table_names = _list_trino_tables(selected["kwargs"], selected["label"])
|
|
850
|
+
|
|
851
|
+
if not storage or not table_names:
|
|
852
|
+
return
|
|
853
|
+
|
|
854
|
+
table_choice = questionary.select(
|
|
855
|
+
"View a table (or press Esc to exit):",
|
|
856
|
+
choices=["(done)"] + table_names,
|
|
857
|
+
).ask()
|
|
858
|
+
|
|
859
|
+
if not table_choice or table_choice == "(done)":
|
|
860
|
+
return
|
|
861
|
+
|
|
862
|
+
click.echo(f"\n📊 {click.style(table_choice, fg='cyan')}\n")
|
|
863
|
+
try:
|
|
864
|
+
import tempfile
|
|
865
|
+
data = storage.get_bytes(table_choice)
|
|
866
|
+
with tempfile.NamedTemporaryFile(suffix=".parquet", delete=False) as tmp:
|
|
867
|
+
tmp.write(data)
|
|
868
|
+
tmp_path = tmp.name
|
|
869
|
+
if not shutil.which("vd"):
|
|
870
|
+
click.secho("visidata not found. Install it with: uv pip install 'steve-cli[visidata]'", fg="yellow")
|
|
871
|
+
return
|
|
872
|
+
subprocess.call(["vd", tmp_path])
|
|
873
|
+
except Exception as e:
|
|
874
|
+
click.secho(f"❌ Could not open table: {e}", fg="red", err=True)
|
|
875
|
+
|
|
876
|
+
|
|
877
|
+
@main.group()
|
|
878
|
+
def policies():
|
|
879
|
+
"""Manage and apply access policies via the Policy Control Plane."""
|
|
880
|
+
pass
|
|
881
|
+
|
|
882
|
+
|
|
883
|
+
@policies.command("apply")
|
|
884
|
+
@click.option('--file', '-f', type=click.Path(path_type=Path),
|
|
885
|
+
help='Path to policies YAML file (default: ./policies/access.yaml)')
|
|
886
|
+
def policies_apply(file: Path | None):
|
|
887
|
+
"""Apply access policies from a YAML file to the Policy Control Plane."""
|
|
888
|
+
from dotenv import load_dotenv
|
|
889
|
+
load_dotenv(".env", override=False)
|
|
890
|
+
|
|
891
|
+
from steve_cli.policies import PolicyClient
|
|
892
|
+
policies_file = Path(file) if file else None
|
|
893
|
+
try:
|
|
894
|
+
PolicyClient().apply_from_file(policies_file)
|
|
895
|
+
click.secho("Policies applied successfully.", fg="green")
|
|
896
|
+
except FileNotFoundError as e:
|
|
897
|
+
click.secho(str(e), fg="red", err=True)
|
|
898
|
+
sys.exit(1)
|
|
899
|
+
except Exception as e:
|
|
900
|
+
click.secho(f"ERROR: {e}", fg="red", bold=True, err=True)
|
|
901
|
+
sys.exit(1)
|
|
902
|
+
|
|
903
|
+
|
|
784
904
|
@main.command("upgrade")
|
|
785
905
|
def upgrade():
|
|
786
906
|
"""Upgrade steve-cli to the latest version."""
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
from collections.abc import Callable
|
|
2
|
+
|
|
3
|
+
from steve_cli.lineage.storage import LineageStorage
|
|
4
|
+
|
|
5
|
+
from .lineage_job import lineage_job
|
|
6
|
+
|
|
7
|
+
GetTables = Callable[[str, str | None], LineageStorage]
|
|
8
|
+
GetStorage = Callable[[str, str | None], LineageStorage]
|
|
9
|
+
|
|
10
|
+
__all__ = ["GetStorage", "GetTables", "lineage_job"]
|
|
@@ -7,6 +7,7 @@ from typing import Any, Callable
|
|
|
7
7
|
|
|
8
8
|
from steve_cli.lineage.collector import make_session
|
|
9
9
|
from steve_cli.storage.s3 import S3Storage
|
|
10
|
+
from steve_cli.storage.trino import TrinoStorage
|
|
10
11
|
from steve_cli.lineage.storage import LineageStorage
|
|
11
12
|
|
|
12
13
|
|
|
@@ -48,14 +49,24 @@ def lineage_job(
|
|
|
48
49
|
session.namespace = ls._dataset_namespace
|
|
49
50
|
return ls
|
|
50
51
|
|
|
52
|
+
def get_tables(tier: str = "bronze", workspace: str | None = None) -> LineageStorage:
|
|
53
|
+
return LineageStorage(
|
|
54
|
+
storage=lambda: TrinoStorage(tier=tier, workspace=workspace),
|
|
55
|
+
session=session,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
sig = inspect.signature(fn)
|
|
59
|
+
available = {"get_storage": get_storage, "get_tables": get_tables}
|
|
60
|
+
injectable = {k: v for k, v in available.items() if k in sig.parameters}
|
|
61
|
+
|
|
51
62
|
try:
|
|
52
|
-
result = fn(*args,
|
|
63
|
+
result = fn(*args, **injectable, **kwargs)
|
|
53
64
|
except Exception as exc:
|
|
54
65
|
session.fail(exc)
|
|
55
66
|
import sys
|
|
67
|
+
import click
|
|
56
68
|
from steve_cli.validation.port import DataQualityError
|
|
57
69
|
if isinstance(exc, (DataQualityError, EnvironmentError)):
|
|
58
|
-
import click
|
|
59
70
|
click.secho(f"ERROR {exc}", fg="red", bold=True, err=True)
|
|
60
71
|
sys.exit(1)
|
|
61
72
|
raise
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import httpx
|
|
9
|
+
|
|
10
|
+
_POLICY_VERSION = "v0"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class PolicyClient:
|
|
14
|
+
def __init__(
|
|
15
|
+
self,
|
|
16
|
+
endpoint: str | None = None,
|
|
17
|
+
token: str | None = None,
|
|
18
|
+
):
|
|
19
|
+
self.endpoint = (endpoint or os.environ.get("PCP_ENDPOINT", "http://policy-control-plane:3010")).rstrip("/")
|
|
20
|
+
self.token = token or os.environ.get("PCP_TOKEN", "")
|
|
21
|
+
|
|
22
|
+
def __trpc(self, method: str, path: str, payload: dict | None = None) -> dict:
|
|
23
|
+
url = f"{self.endpoint}/api/trpc/{path}"
|
|
24
|
+
headers = {"Authorization": f"Bearer {self.token}"} if self.token else {}
|
|
25
|
+
if method == "query":
|
|
26
|
+
resp = httpx.get(url, params={"input": json.dumps(payload or {})}, headers=headers, timeout=10)
|
|
27
|
+
else:
|
|
28
|
+
resp = httpx.post(url, json=payload or {}, headers=headers, timeout=10)
|
|
29
|
+
resp.raise_for_status()
|
|
30
|
+
data = resp.json()
|
|
31
|
+
if "error" in data:
|
|
32
|
+
raise RuntimeError(f"tRPC error on {path}: {data['error']}")
|
|
33
|
+
return data.get("result", {}).get("data", data.get("result", {}))
|
|
34
|
+
|
|
35
|
+
def _trpc(self, method: str, path: str, payload: dict | None = None) -> dict:
|
|
36
|
+
try:
|
|
37
|
+
return self.__trpc(method, path, payload)
|
|
38
|
+
except httpx.ConnectError as e:
|
|
39
|
+
raise EnvironmentError(
|
|
40
|
+
f"Cannot reach Policy Control Plane at {self.endpoint} — "
|
|
41
|
+
f"set PCP_ENDPOINT in your .env"
|
|
42
|
+
) from e
|
|
43
|
+
|
|
44
|
+
def register_pdp(self, name: str, endpoint: str, pdp_type: str = "opa", is_default: bool = True) -> dict:
|
|
45
|
+
return self._trpc("mutation", "pdp.register", {
|
|
46
|
+
"name": name,
|
|
47
|
+
"type": pdp_type,
|
|
48
|
+
"endpoint": endpoint,
|
|
49
|
+
"isDefault": is_default,
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
def attach_pdp(self, workspace: str, pdp_id: str) -> None:
|
|
53
|
+
self._trpc("mutation", "workspace.attachPdp", {"workspace": workspace, "pdpId": pdp_id})
|
|
54
|
+
|
|
55
|
+
def apply_policies(self, workspace: str, pdp_name: str, rules: list[dict[str, Any]]) -> dict:
|
|
56
|
+
return self._trpc("mutation", "workspace.applyPolicies", {
|
|
57
|
+
"workspace": workspace,
|
|
58
|
+
"fragments": [
|
|
59
|
+
{
|
|
60
|
+
"syntax": "dsl",
|
|
61
|
+
"target": pdp_name,
|
|
62
|
+
"content": {"rules": rules},
|
|
63
|
+
}
|
|
64
|
+
],
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
def apply_from_file(self, policies_file: Path | None = None) -> None:
|
|
68
|
+
path = policies_file or Path.cwd() / "policies" / "access.yaml"
|
|
69
|
+
if not path.exists():
|
|
70
|
+
raise FileNotFoundError(f"Policies file not found: {path}")
|
|
71
|
+
|
|
72
|
+
import yaml
|
|
73
|
+
config: dict[str, Any] = yaml.safe_load(path.read_text())
|
|
74
|
+
workspace = config["workspace"]
|
|
75
|
+
pdp_cfg = config["pdp"]
|
|
76
|
+
rules = config.get("rules", [])
|
|
77
|
+
|
|
78
|
+
pdp = self.register_pdp(
|
|
79
|
+
name=pdp_cfg["name"],
|
|
80
|
+
endpoint=pdp_cfg["endpoint"],
|
|
81
|
+
pdp_type=pdp_cfg.get("type", "opa"),
|
|
82
|
+
)
|
|
83
|
+
self.attach_pdp(workspace, pdp["id"])
|
|
84
|
+
result = self.apply_policies(workspace, pdp_cfg["name"], rules)
|
|
85
|
+
|
|
86
|
+
errors = result.get("errors", [])
|
|
87
|
+
if errors:
|
|
88
|
+
reasons = "; ".join(e["reason"] for e in errors)
|
|
89
|
+
raise RuntimeError(f"Failed to apply {len(errors)} policy fragment(s): {reasons}")
|
|
@@ -1,11 +1,13 @@
|
|
|
1
1
|
from .protocol import Storage
|
|
2
2
|
from .s3 import S3Storage
|
|
3
|
+
from .trino import TrinoStorage
|
|
3
4
|
from .parquet import ParquetMetadata, extract_parquet_metadata
|
|
4
5
|
from .metadata import FileMetadata, ColumnMetadata, MetadataExtractorPort, MetadataRegistry
|
|
5
6
|
|
|
6
7
|
__all__ = [
|
|
7
8
|
"Storage",
|
|
8
9
|
"S3Storage",
|
|
10
|
+
"TrinoStorage",
|
|
9
11
|
"ParquetMetadata",
|
|
10
12
|
"extract_parquet_metadata",
|
|
11
13
|
"FileMetadata",
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import io
|
|
4
|
+
import logging
|
|
5
|
+
import os
|
|
6
|
+
import time
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
import pyarrow as pa
|
|
10
|
+
import pyarrow.parquet as pq
|
|
11
|
+
import requests
|
|
12
|
+
|
|
13
|
+
logger = logging.getLogger(__name__)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _derive_schema(tier: str, workspace: str | None) -> str | None:
|
|
17
|
+
if workspace:
|
|
18
|
+
prefix = workspace.lower().replace("-", "_")
|
|
19
|
+
return f"ws_{prefix}_{tier.lower()}"
|
|
20
|
+
return None
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class TrinoStorage:
|
|
24
|
+
def __init__(self, tier: str = "bronze", workspace: str | None = None):
|
|
25
|
+
trino_endpoint = os.getenv("TRINO_ENDPOINT")
|
|
26
|
+
if not trino_endpoint:
|
|
27
|
+
raise EnvironmentError("TRINO_ENDPOINT is not set — Trino is not available in this environment")
|
|
28
|
+
self.catalog = os.getenv("TRINO_CATALOG", "minio")
|
|
29
|
+
resolved_workspace = workspace or os.getenv("WORKSPACE_NAME")
|
|
30
|
+
self.schema = _derive_schema(tier, resolved_workspace) or os.environ.get("TRINO_SCHEMA", "")
|
|
31
|
+
if not self.schema:
|
|
32
|
+
raise EnvironmentError("TRINO_SCHEMA is not set and no WORKSPACE_NAME or workspace was provided")
|
|
33
|
+
self.user = os.getenv("TRINO_USER", "admin")
|
|
34
|
+
self._base = trino_endpoint.rstrip("/")
|
|
35
|
+
self._lakekeeper_endpoint = os.getenv("LAKEKEEPER_ENDPOINT")
|
|
36
|
+
default_warehouse = f"{resolved_workspace}-{tier}" if resolved_workspace else "minio"
|
|
37
|
+
self._lakekeeper_warehouse = os.getenv("LAKEKEEPER_WAREHOUSE", default_warehouse)
|
|
38
|
+
self.__iceberg_catalog = None
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def _iceberg_catalog(self):
|
|
42
|
+
if self.__iceberg_catalog is None:
|
|
43
|
+
from pyiceberg.catalog.rest import RestCatalog
|
|
44
|
+
from pyiceberg.io import load_file_io
|
|
45
|
+
from pyiceberg.table import Table
|
|
46
|
+
|
|
47
|
+
# FIXME: Interim workaround — pyiceberg cannot use Lakekeeper's remote signing
|
|
48
|
+
# endpoint when running outside the cluster (hostname 'lakekeeper' doesn't resolve
|
|
49
|
+
# locally). We subclass RestCatalog to inject our local signer URI after pyiceberg
|
|
50
|
+
# merges table config (which overwrites s3.signer.uri with the internal hostname).
|
|
51
|
+
# Remove once Increment 5 (STS credential vending) is wired.
|
|
52
|
+
tier = self.schema.rsplit("_", 1)[-1].upper()
|
|
53
|
+
lakekeeper_base = self._lakekeeper_endpoint or "http://lakekeeper:8181/catalog"
|
|
54
|
+
s3_overrides = {
|
|
55
|
+
"s3.endpoint": os.getenv("S3_ENDPOINT", "http://minio:9000"),
|
|
56
|
+
"s3.access-key-id": os.getenv(f"{tier}_ACCESS_KEY") or os.getenv("AWS_ACCESS_KEY_ID", ""),
|
|
57
|
+
"s3.secret-access-key": os.getenv(f"{tier}_SECRET_KEY") or os.getenv("AWS_SECRET_ACCESS_KEY", ""),
|
|
58
|
+
"s3.path-style-access": "true",
|
|
59
|
+
"s3.signer.uri": lakekeeper_base,
|
|
60
|
+
} if self._lakekeeper_endpoint else {}
|
|
61
|
+
|
|
62
|
+
class _PatchedRestCatalog(RestCatalog):
|
|
63
|
+
def _response_to_table(self_, identifier_tuple, table_response):
|
|
64
|
+
merged = {**table_response.metadata.properties, **table_response.config, **s3_overrides, "uri": lakekeeper_base}
|
|
65
|
+
return Table(
|
|
66
|
+
identifier=identifier_tuple,
|
|
67
|
+
metadata_location=table_response.metadata_location,
|
|
68
|
+
metadata=table_response.metadata,
|
|
69
|
+
io=load_file_io(merged, table_response.metadata_location),
|
|
70
|
+
catalog=self_,
|
|
71
|
+
config=table_response.config,
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
catalog = _PatchedRestCatalog(
|
|
75
|
+
"lakekeeper",
|
|
76
|
+
uri=lakekeeper_base,
|
|
77
|
+
warehouse=self._lakekeeper_warehouse,
|
|
78
|
+
**s3_overrides,
|
|
79
|
+
)
|
|
80
|
+
if self._lakekeeper_endpoint:
|
|
81
|
+
catalog.uri = lakekeeper_base
|
|
82
|
+
self.__iceberg_catalog = catalog
|
|
83
|
+
return self.__iceberg_catalog
|
|
84
|
+
|
|
85
|
+
def _execute(self, sql: str) -> list[dict]:
|
|
86
|
+
headers = {
|
|
87
|
+
"X-Trino-User": self.user,
|
|
88
|
+
"X-Trino-Catalog": self.catalog,
|
|
89
|
+
"X-Trino-Schema": self.schema,
|
|
90
|
+
}
|
|
91
|
+
resp = requests.post(f"{self._base}/v1/statement", data=sql, headers=headers)
|
|
92
|
+
resp.raise_for_status()
|
|
93
|
+
data = resp.json()
|
|
94
|
+
rows: list[dict] = []
|
|
95
|
+
while True:
|
|
96
|
+
if "data" in data and "columns" in data:
|
|
97
|
+
col_names = [c["name"] for c in data["columns"]]
|
|
98
|
+
for row in data["data"]:
|
|
99
|
+
rows.append(dict(zip(col_names, row)))
|
|
100
|
+
next_uri = data.get("nextUri")
|
|
101
|
+
if not next_uri:
|
|
102
|
+
break
|
|
103
|
+
time.sleep(0.1)
|
|
104
|
+
resp = requests.get(next_uri, headers=headers)
|
|
105
|
+
resp.raise_for_status()
|
|
106
|
+
data = resp.json()
|
|
107
|
+
return rows
|
|
108
|
+
|
|
109
|
+
@staticmethod
|
|
110
|
+
def _table_name(path: str) -> str:
|
|
111
|
+
return path.strip("/").replace("/", "_").replace(".", "_")
|
|
112
|
+
|
|
113
|
+
def list(self, prefix: str = "") -> list[str]:
|
|
114
|
+
rows = self._execute(f"SHOW TABLES FROM {self.catalog}.{self.schema}")
|
|
115
|
+
tables = [r["Table"] for r in rows]
|
|
116
|
+
if prefix:
|
|
117
|
+
clean = prefix.strip("/")
|
|
118
|
+
tables = [t for t in tables if t.startswith(clean)]
|
|
119
|
+
return tables
|
|
120
|
+
|
|
121
|
+
def list_all(self) -> list[str]:
|
|
122
|
+
return self.list()
|
|
123
|
+
|
|
124
|
+
def get_bytes(self, path: str) -> bytes:
|
|
125
|
+
rows = self._execute(f"SELECT * FROM {self.catalog}.{self.schema}.{self._table_name(path)}")
|
|
126
|
+
if not rows:
|
|
127
|
+
return b""
|
|
128
|
+
arrow_table = pa.Table.from_pylist(rows)
|
|
129
|
+
buf = io.BytesIO()
|
|
130
|
+
pq.write_table(arrow_table, buf)
|
|
131
|
+
return buf.getvalue()
|
|
132
|
+
|
|
133
|
+
def get_file(self, path: str, local_path: str) -> None:
|
|
134
|
+
Path(local_path).parent.mkdir(parents=True, exist_ok=True)
|
|
135
|
+
Path(local_path).write_bytes(self.get_bytes(path))
|
|
136
|
+
|
|
137
|
+
def _ensure_namespace(self) -> None:
|
|
138
|
+
from pyiceberg.exceptions import NamespaceAlreadyExistsError
|
|
139
|
+
try:
|
|
140
|
+
self._iceberg_catalog.create_namespace(self.schema)
|
|
141
|
+
logger.info("Created namespace %s", self.schema)
|
|
142
|
+
except NamespaceAlreadyExistsError:
|
|
143
|
+
pass
|
|
144
|
+
|
|
145
|
+
def put_bytes(self, data: bytes, path: str) -> None:
|
|
146
|
+
from pyiceberg.exceptions import NoSuchTableError
|
|
147
|
+
|
|
148
|
+
arrow_table = pq.read_table(io.BytesIO(data))
|
|
149
|
+
table_id = f"{self.schema}.{self._table_name(path)}"
|
|
150
|
+
catalog = self._iceberg_catalog
|
|
151
|
+
|
|
152
|
+
self._ensure_namespace()
|
|
153
|
+
|
|
154
|
+
try:
|
|
155
|
+
iceberg_table = catalog.load_table(table_id)
|
|
156
|
+
iceberg_table.append(arrow_table)
|
|
157
|
+
except NoSuchTableError:
|
|
158
|
+
from pyiceberg.schema import Schema
|
|
159
|
+
from pyiceberg.types import (
|
|
160
|
+
BinaryType,
|
|
161
|
+
BooleanType,
|
|
162
|
+
DateType,
|
|
163
|
+
DoubleType,
|
|
164
|
+
FloatType,
|
|
165
|
+
IntegerType,
|
|
166
|
+
LongType,
|
|
167
|
+
NestedField,
|
|
168
|
+
StringType,
|
|
169
|
+
TimestampType,
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
_PYARROW_TO_ICEBERG = {
|
|
173
|
+
pa.bool_(): BooleanType(),
|
|
174
|
+
pa.int32(): IntegerType(),
|
|
175
|
+
pa.int64(): LongType(),
|
|
176
|
+
pa.float32(): FloatType(),
|
|
177
|
+
pa.float64(): DoubleType(),
|
|
178
|
+
pa.large_utf8(): StringType(),
|
|
179
|
+
pa.utf8(): StringType(),
|
|
180
|
+
pa.large_binary(): BinaryType(),
|
|
181
|
+
pa.binary(): BinaryType(),
|
|
182
|
+
pa.date32(): DateType(),
|
|
183
|
+
pa.timestamp("us"): TimestampType(),
|
|
184
|
+
pa.timestamp("us", tz="UTC"): TimestampType(),
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
fields = []
|
|
188
|
+
for i, field in enumerate(arrow_table.schema):
|
|
189
|
+
iceberg_type = _PYARROW_TO_ICEBERG.get(field.type, StringType())
|
|
190
|
+
fields.append(NestedField(field_id=i + 1, name=field.name, field_type=iceberg_type, required=not field.nullable))
|
|
191
|
+
|
|
192
|
+
iceberg_schema = Schema(*fields)
|
|
193
|
+
iceberg_table = catalog.create_table(table_id, schema=iceberg_schema)
|
|
194
|
+
iceberg_table.append(arrow_table)
|
|
195
|
+
|
|
196
|
+
logger.info("Written %d rows to %s", len(arrow_table), table_id)
|
|
197
|
+
|
|
198
|
+
def put_file(self, local_path: str, path: str) -> None:
|
|
199
|
+
self.put_bytes(Path(local_path).read_bytes(), path)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: steve-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.4.1
|
|
4
4
|
Summary: A simple CLI tool to run jobs from jobs.yaml with proper environment setup
|
|
5
5
|
Author: Frank
|
|
6
6
|
License: MIT
|
|
@@ -44,6 +44,12 @@ Requires-Dist: great-expectations>=0.18.0; extra == "great-expectations"
|
|
|
44
44
|
Requires-Dist: pandas>=2.0.3; extra == "great-expectations"
|
|
45
45
|
Provides-Extra: excel
|
|
46
46
|
Requires-Dist: openpyxl>=3.1.0; extra == "excel"
|
|
47
|
+
Provides-Extra: trino
|
|
48
|
+
Requires-Dist: trino>=0.330; extra == "trino"
|
|
49
|
+
Requires-Dist: pyarrow>=17.0.0; extra == "trino"
|
|
50
|
+
Requires-Dist: requests>=2.31.0; extra == "trino"
|
|
51
|
+
Requires-Dist: pyiceberg>=0.7.0; extra == "trino"
|
|
52
|
+
Requires-Dist: s3fs>=2024.1.0; extra == "trino"
|
|
47
53
|
Provides-Extra: visidata
|
|
48
54
|
Requires-Dist: visidata>=3.0; extra == "visidata"
|
|
49
55
|
Provides-Extra: all
|
|
@@ -72,7 +78,7 @@ uv pip install -e .
|
|
|
72
78
|
|
|
73
79
|
### Install from GitHub
|
|
74
80
|
|
|
75
|
-
`uv add steve-cli`
|
|
81
|
+
`uv add steve-cli`
|
|
76
82
|
|
|
77
83
|
OR
|
|
78
84
|
|
|
@@ -80,7 +86,14 @@ OR
|
|
|
80
86
|
uv add git+https://github.com/your-org/tilt-ts-4.git#subdirectory=apps/example-hehnke/packages/steve-cli
|
|
81
87
|
```
|
|
82
88
|
|
|
89
|
+
### dev
|
|
83
90
|
|
|
91
|
+
For local development add the following to your pyproj.toml which allows using the local copy instead of the one from the registry
|
|
92
|
+
|
|
93
|
+
```toml
|
|
94
|
+
[tool.uv.sources]
|
|
95
|
+
steve-cli = { path = "../../packages/steve-cli", editable = true }
|
|
96
|
+
```
|
|
84
97
|
|
|
85
98
|
## Usage
|
|
86
99
|
|
|
@@ -162,13 +175,13 @@ Running `steve extract-data` will:
|
|
|
162
175
|
|
|
163
176
|
Steve automatically extracts metadata from files read or written via `MetadataRegistry`. The extractor is chosen by file extension — no configuration needed.
|
|
164
177
|
|
|
165
|
-
| Extension
|
|
166
|
-
|
|
167
|
-
| `.parquet`, `.pq`
|
|
168
|
-
| `.csv`, `.tsv`, `.txt`
|
|
169
|
-
| `.json`, `.jsonl`, `.ndjson` | `JsonExtractor`
|
|
170
|
-
| `.xlsx`, `.xls`, `.xlsm`
|
|
171
|
-
| anything else
|
|
178
|
+
| Extension | Extractor | Requires |
|
|
179
|
+
| ---------------------------- | ------------------ | ------------------------------- |
|
|
180
|
+
| `.parquet`, `.pq` | `ParquetExtractor` | `pip install steve-cli[polars]` |
|
|
181
|
+
| `.csv`, `.tsv`, `.txt` | `CsvExtractor` | stdlib only |
|
|
182
|
+
| `.json`, `.jsonl`, `.ndjson` | `JsonExtractor` | stdlib only |
|
|
183
|
+
| `.xlsx`, `.xls`, `.xlsm` | `ExcelExtractor` | `pip install steve-cli[excel]` |
|
|
184
|
+
| anything else | `GenericExtractor` | stdlib only |
|
|
172
185
|
|
|
173
186
|
### Adding a custom extractor
|
|
174
187
|
|
|
@@ -230,10 +243,9 @@ black steve_cli/
|
|
|
230
243
|
isort steve_cli/
|
|
231
244
|
```
|
|
232
245
|
|
|
233
|
-
|
|
234
|
-
|
|
235
246
|
# use cli
|
|
236
|
-
|
|
247
|
+
|
|
248
|
+
source .venv/bin/activate
|
|
237
249
|
steve
|
|
238
250
|
|
|
239
251
|
#
|
|
@@ -20,10 +20,13 @@ steve_cli/lineage/adapters/__init__.py
|
|
|
20
20
|
steve_cli/lineage/adapters/logging.py
|
|
21
21
|
steve_cli/lineage/adapters/null.py
|
|
22
22
|
steve_cli/lineage/adapters/openlineage.py
|
|
23
|
+
steve_cli/policies/__init__.py
|
|
24
|
+
steve_cli/policies/client.py
|
|
23
25
|
steve_cli/storage/__init__.py
|
|
24
26
|
steve_cli/storage/parquet.py
|
|
25
27
|
steve_cli/storage/protocol.py
|
|
26
28
|
steve_cli/storage/s3.py
|
|
29
|
+
steve_cli/storage/trino.py
|
|
27
30
|
steve_cli/storage/metadata/__init__.py
|
|
28
31
|
steve_cli/storage/metadata/port.py
|
|
29
32
|
steve_cli/storage/metadata/registry.py
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|