ripple-sql 0.1.7__tar.gz → 0.1.8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/CHANGELOG.md +8 -1
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/PKG-INFO +48 -1
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/README.md +47 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/pyproject.toml +1 -1
- ripple_sql-0.1.8/src/ripple/check_file.py +336 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/cli.py +47 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/dbt_config.py +44 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/identity.py +11 -2
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/.gitignore +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/LICENSE +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/answer.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/answer_page.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/cache.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/ci.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/ci_signature.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/doctor.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/budget.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/column_lineage.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/column_ref.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/cte_tracing.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/dependencies.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/dialect.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/dispatch.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/extraction.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/generators.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/jinja.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/json_sources.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/macro_source.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/pipeline.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/preprocess.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/safe_gen.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/schema_qualification.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/scope.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/select_sources.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/sql_script.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/statement.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tech_debt.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_catalog.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_scalar_vars.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_tvf.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_xml.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/types.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/unused_deps.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/validation.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/graph.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/home.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/dbt.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/sidecar.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/sqldir.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/types.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/lookml.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/mcp_server.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/names.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/project.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/py.typed +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/render.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/render_shims.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/schemas.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/semantic.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/server.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/sourcefiles.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/star_resolution.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/answer.css +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/answer.html +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/answer_twin.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/explore.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/find.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/focus.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/cli.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/collect.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/discover.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/ingest.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/report.py +0 -0
|
@@ -6,6 +6,12 @@ Notable changes to Ripple. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.8] - 2026-09-15
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- `ripple check-file FILE [--source TABLE]`: a file arrives from an external team (csv, tsv, xlsx, parquet), and before anyone loads it Ripple diffs its header against the landing table's declared columns and prints who reads every column that is gone: models, dashboards, hit hardest. The verdict is the first line and the last; a similar name is a hint ("looks like customer_name; a similar name does not prove the same contents"), never an assumption; exit 0 when nothing downstream changes (new columns nobody reads included), 1 when something breaks, 2 when the check could not run. The file binds to its table by `--source`, or by a `files:` pattern the project declares (`meta: {ripple: {files: ["vendor_feed_*.csv"]}}` on a dbt source, or `files:` in `ripple.yml` for a plain SQL folder); with neither, the command prints the closest table and the exact line to add and stops. Header only: no data row is read and nothing leaves the machine. Shaped by a two-model debate; the spec's sample output is the command's own.
|
|
13
|
+
- A dbt source's `columns:` in sources.yml are its declared columns. The graph read only ingested schemas and seeds before, so every dbt source showed as "not declared" and a select star through it stayed unresolved even when the yml listed every column.
|
|
14
|
+
|
|
9
15
|
## [0.1.7] - 2026-09-10
|
|
10
16
|
|
|
11
17
|
### Fixed
|
|
@@ -264,7 +270,8 @@ First public release.
|
|
|
264
270
|
### Removed
|
|
265
271
|
- The dark whole-graph canvas that `ripple serve` used to open (`static/index.html`). Every surface now draws one answer.
|
|
266
272
|
|
|
267
|
-
[Unreleased]: https://github.com/bteh/ripple/compare/v0.1.
|
|
273
|
+
[Unreleased]: https://github.com/bteh/ripple/compare/v0.1.8...HEAD
|
|
274
|
+
[0.1.8]: https://github.com/bteh/ripple/releases/tag/v0.1.8
|
|
268
275
|
[0.1.7]: https://github.com/bteh/ripple/releases/tag/v0.1.7
|
|
269
276
|
[0.1.6]: https://github.com/bteh/ripple/releases/tag/v0.1.6
|
|
270
277
|
[0.1.5]: https://github.com/bteh/ripple/releases/tag/v0.1.5
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: ripple-sql
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.8
|
|
4
4
|
Summary: Offline column-level SQL lineage. See what breaks before you merge.
|
|
5
5
|
Project-URL: Homepage, https://github.com/bteh/ripple
|
|
6
6
|
Project-URL: Repository, https://github.com/bteh/ripple
|
|
@@ -210,6 +210,53 @@ ripple unresolved # which tables block the most coverage, worst
|
|
|
210
210
|
ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
|
|
211
211
|
```
|
|
212
212
|
|
|
213
|
+
## A file arrives from another team
|
|
214
|
+
|
|
215
|
+
Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
ripple check-file vendor_feed_2026-09.csv
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
```
|
|
222
|
+
Stop: 2 columns that 4 models read are missing from this file.
|
|
223
|
+
|
|
224
|
+
File vendor_feed_2026-09.csv
|
|
225
|
+
Table raw.vendor_feed (files: vendor_feed_*.csv)
|
|
226
|
+
Checked: column names only. Not checked: values, types, whether the file loads.
|
|
227
|
+
|
|
228
|
+
6 columns: 2 unchanged, 2 missing, 4 new
|
|
229
|
+
|
|
230
|
+
MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
|
|
231
|
+
MISSING amount_usd read by 2 models
|
|
232
|
+
NEW cust_nm looks like customer_name; a similar name does not prove the same contents
|
|
233
|
+
3 more new columns nobody reads yet: region, notes, batch_id
|
|
234
|
+
|
|
235
|
+
To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
|
|
236
|
+
Details: ripple breaks raw.vendor_feed.customer_name
|
|
237
|
+
|
|
238
|
+
Stop: 2 columns that 4 models read are missing from this file. (exit 1)
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
|
|
242
|
+
|
|
243
|
+
The file finds its table through `--source raw.vendor_feed`, or through a pattern the project declares once:
|
|
244
|
+
|
|
245
|
+
```yaml
|
|
246
|
+
sources:
|
|
247
|
+
- name: raw
|
|
248
|
+
tables:
|
|
249
|
+
- name: vendor_feed
|
|
250
|
+
meta:
|
|
251
|
+
ripple:
|
|
252
|
+
files: ["vendor_feed_*.csv"]
|
|
253
|
+
columns:
|
|
254
|
+
- name: customer_id
|
|
255
|
+
- name: customer_name
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow installed). `--json` carries the same facts for an ingest job.
|
|
259
|
+
|
|
213
260
|
## Bring your query history (optional)
|
|
214
261
|
|
|
215
262
|
Your SQL files say what is supposed to happen. The warehouse's query log says what
|
|
@@ -170,6 +170,53 @@ ripple unresolved # which tables block the most coverage, worst
|
|
|
170
170
|
ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
|
|
171
171
|
```
|
|
172
172
|
|
|
173
|
+
## A file arrives from another team
|
|
174
|
+
|
|
175
|
+
Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
ripple check-file vendor_feed_2026-09.csv
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
```
|
|
182
|
+
Stop: 2 columns that 4 models read are missing from this file.
|
|
183
|
+
|
|
184
|
+
File vendor_feed_2026-09.csv
|
|
185
|
+
Table raw.vendor_feed (files: vendor_feed_*.csv)
|
|
186
|
+
Checked: column names only. Not checked: values, types, whether the file loads.
|
|
187
|
+
|
|
188
|
+
6 columns: 2 unchanged, 2 missing, 4 new
|
|
189
|
+
|
|
190
|
+
MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
|
|
191
|
+
MISSING amount_usd read by 2 models
|
|
192
|
+
NEW cust_nm looks like customer_name; a similar name does not prove the same contents
|
|
193
|
+
3 more new columns nobody reads yet: region, notes, batch_id
|
|
194
|
+
|
|
195
|
+
To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
|
|
196
|
+
Details: ripple breaks raw.vendor_feed.customer_name
|
|
197
|
+
|
|
198
|
+
Stop: 2 columns that 4 models read are missing from this file. (exit 1)
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
|
|
202
|
+
|
|
203
|
+
The file finds its table through `--source raw.vendor_feed`, or through a pattern the project declares once:
|
|
204
|
+
|
|
205
|
+
```yaml
|
|
206
|
+
sources:
|
|
207
|
+
- name: raw
|
|
208
|
+
tables:
|
|
209
|
+
- name: vendor_feed
|
|
210
|
+
meta:
|
|
211
|
+
ripple:
|
|
212
|
+
files: ["vendor_feed_*.csv"]
|
|
213
|
+
columns:
|
|
214
|
+
- name: customer_id
|
|
215
|
+
- name: customer_name
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow installed). `--json` carries the same facts for an ingest job.
|
|
219
|
+
|
|
173
220
|
## Bring your query history (optional)
|
|
174
221
|
|
|
175
222
|
Your SQL files say what is supposed to happen. The warehouse's query log says what
|
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
"""A file arrives from an external team: what breaks if it is loaded.
|
|
2
|
+
|
|
3
|
+
The file feeds one landing table the graph already knows. Its header is
|
|
4
|
+
diffed against the table's declared columns, and every column that is gone
|
|
5
|
+
gets the blast radius `ripple breaks` would give it. Header only: no data
|
|
6
|
+
row is read, nothing leaves the machine, and a similar name is a hint,
|
|
7
|
+
never an assumption.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import csv
|
|
13
|
+
import difflib
|
|
14
|
+
import fnmatch
|
|
15
|
+
import zipfile
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from xml.etree import ElementTree
|
|
18
|
+
|
|
19
|
+
from ripple.answer import breaks_answer, hit_hardest
|
|
20
|
+
from ripple.graph import UnknownTarget
|
|
21
|
+
|
|
22
|
+
HEADER_ONLY = "Checked: column names only. Not checked: values, types, whether the file loads."
|
|
23
|
+
KINDS = ("csv", "tsv", "xlsx", "parquet")
|
|
24
|
+
LOOKS_LIKE = 0.6
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class CannotCheck(Exception):
|
|
28
|
+
"""The check could not run; the message says why and what to do."""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def read_header(path: Path) -> list[str]:
|
|
32
|
+
"""The column names of a file, from its header only."""
|
|
33
|
+
kind = path.suffix.lower().lstrip(".")
|
|
34
|
+
if kind in ("csv", "tsv", "txt"):
|
|
35
|
+
return _delimited(path, "\t" if kind == "tsv" else ",")
|
|
36
|
+
if kind == "xlsx":
|
|
37
|
+
return _xlsx(path)
|
|
38
|
+
if kind == "parquet":
|
|
39
|
+
return _parquet(path)
|
|
40
|
+
raise CannotCheck(f"{path.name}: Ripple reads a csv, tsv, xlsx or parquet header, not .{kind}")
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _delimited(path: Path, delimiter: str) -> list[str]:
|
|
44
|
+
# newline="" lets the csv module see a quoted header that spans lines
|
|
45
|
+
with open(path, encoding="utf-8-sig", errors="replace", newline="") as f:
|
|
46
|
+
try:
|
|
47
|
+
header = next(csv.reader(f, delimiter=delimiter))
|
|
48
|
+
except StopIteration:
|
|
49
|
+
raise CannotCheck(f"{path.name} is empty; no header row to check") from None
|
|
50
|
+
# a quoted header that spans lines keeps the file's own line ending
|
|
51
|
+
return [h.replace("\r\n", "\n").strip() for h in header]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _local(tag: str) -> str:
|
|
55
|
+
return tag.rsplit("}", 1)[-1]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _xlsx(path: Path) -> list[str]:
|
|
59
|
+
"""Row 1 of the first sheet, without a spreadsheet library."""
|
|
60
|
+
try:
|
|
61
|
+
with zipfile.ZipFile(path) as z:
|
|
62
|
+
names = z.namelist()
|
|
63
|
+
shared: list[str] = []
|
|
64
|
+
if "xl/sharedStrings.xml" in names:
|
|
65
|
+
for si in ElementTree.fromstring(z.read("xl/sharedStrings.xml")):
|
|
66
|
+
shared.append("".join(t.text or "" for t in si.iter() if _local(t.tag) == "t"))
|
|
67
|
+
sheet = next((n for n in sorted(names) if n.startswith("xl/worksheets/sheet")), None)
|
|
68
|
+
if not sheet:
|
|
69
|
+
raise CannotCheck(f"{path.name} has no worksheet")
|
|
70
|
+
root = ElementTree.fromstring(z.read(sheet))
|
|
71
|
+
except zipfile.BadZipFile:
|
|
72
|
+
raise CannotCheck(f"{path.name} is not an xlsx workbook") from None
|
|
73
|
+
row = next((r for r in root.iter() if _local(r.tag) == "row"), None)
|
|
74
|
+
if row is None:
|
|
75
|
+
raise CannotCheck(f"{path.name} is empty; no header row to check")
|
|
76
|
+
header = []
|
|
77
|
+
for cell in row:
|
|
78
|
+
if _local(cell.tag) != "c":
|
|
79
|
+
continue
|
|
80
|
+
kind = cell.get("t")
|
|
81
|
+
value = next((v.text or "" for v in cell.iter() if _local(v.tag) in ("v", "t")), "")
|
|
82
|
+
if kind == "s" and value.isdigit() and int(value) < len(shared):
|
|
83
|
+
value = shared[int(value)]
|
|
84
|
+
header.append(value.strip())
|
|
85
|
+
return header
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _parquet(path: Path) -> list[str]:
|
|
89
|
+
try:
|
|
90
|
+
import pyarrow.parquet as pq
|
|
91
|
+
except ImportError:
|
|
92
|
+
raise CannotCheck(
|
|
93
|
+
f"{path.name}: reading a Parquet schema needs pyarrow (pip install pyarrow), "
|
|
94
|
+
"or export the header as csv"
|
|
95
|
+
) from None
|
|
96
|
+
return list(pq.read_schema(path).names)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def file_map(root: Path) -> dict[str, str]:
|
|
100
|
+
"""File-name patterns to the tables they feed, from the project.
|
|
101
|
+
|
|
102
|
+
A dbt source table carries them as `meta: {ripple: {files: [...]}}`; a
|
|
103
|
+
plain SQL folder keeps a `files:` map in ripple.yml at its root."""
|
|
104
|
+
import yaml
|
|
105
|
+
|
|
106
|
+
from ripple.loaders.dbt_config import source_tables
|
|
107
|
+
|
|
108
|
+
root = Path(root)
|
|
109
|
+
patterns: dict[str, str] = {}
|
|
110
|
+
own = root / "ripple.yml"
|
|
111
|
+
if own.is_file():
|
|
112
|
+
parsed = yaml.safe_load(own.read_text(encoding="utf-8", errors="replace")) or {}
|
|
113
|
+
for pattern, table in (parsed.get("files") or {}).items():
|
|
114
|
+
patterns[str(pattern)] = str(table)
|
|
115
|
+
for source, table in source_tables(root):
|
|
116
|
+
meta = (table.get("meta") or {}).get("ripple") or {}
|
|
117
|
+
for pattern in meta.get("files") or []:
|
|
118
|
+
patterns[str(pattern)] = f"{source}.{table['name']}"
|
|
119
|
+
return patterns
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def bind(
|
|
123
|
+
file_name: str, explicit: str | None, patterns: dict[str, str]
|
|
124
|
+
) -> tuple[str | None, str | None]:
|
|
125
|
+
"""(table, how) for a file: --source, else the first matching pattern."""
|
|
126
|
+
if explicit:
|
|
127
|
+
return explicit, "--source"
|
|
128
|
+
for pattern, table in patterns.items():
|
|
129
|
+
if fnmatch.fnmatch(file_name, pattern):
|
|
130
|
+
return table, f"files: {pattern}"
|
|
131
|
+
return None, None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def find_source(graph, table: str):
|
|
135
|
+
"""The source table by its name or an alias, case-insensitive."""
|
|
136
|
+
wanted = table.lower()
|
|
137
|
+
for source in graph.project.sources:
|
|
138
|
+
if source.name.lower() == wanted or wanted in {a.lower() for a in source.aliases}:
|
|
139
|
+
return source
|
|
140
|
+
return None
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def closest_source(graph, file_name: str) -> str | None:
|
|
144
|
+
stem = Path(file_name).stem.lower()
|
|
145
|
+
names = [s.name for s in graph.project.sources]
|
|
146
|
+
by_tail = {n.rsplit(".", 1)[-1].lower(): n for n in names}
|
|
147
|
+
match = difflib.get_close_matches(stem, list(by_tail), n=1, cutoff=0.4)
|
|
148
|
+
return by_tail[match[0]] if match else (names[0] if names else None)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _readers(graph, table: str, column: str) -> dict:
|
|
152
|
+
try:
|
|
153
|
+
answer = breaks_answer(graph.breaks(table, column), "model")
|
|
154
|
+
except UnknownTarget:
|
|
155
|
+
return {"models": set(), "dashboards": set(), "hit_hardest": []}
|
|
156
|
+
nodes = [n for n in answer["nodes"] if n["depth"] > 0]
|
|
157
|
+
return {
|
|
158
|
+
"models": {n["model"] for n in nodes if n["kind"] == "model"},
|
|
159
|
+
"dashboards": {n["model"] for n in nodes if n["kind"] == "dashboard"},
|
|
160
|
+
"hit_hardest": hit_hardest(answer, 1),
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def check(
|
|
165
|
+
graph, table: str, header: list[str], file_name: str, how: str, noun: str = "model"
|
|
166
|
+
) -> dict:
|
|
167
|
+
"""The report: every column's status, who reads the missing ones, the verdict."""
|
|
168
|
+
source = find_source(graph, table)
|
|
169
|
+
if source is None:
|
|
170
|
+
names = [s.name for s in graph.project.sources]
|
|
171
|
+
near = difflib.get_close_matches(table, names, n=3, cutoff=0.5)
|
|
172
|
+
hint = f" Did you mean {', '.join(near)}?" if near else ""
|
|
173
|
+
raise CannotCheck(f"no source table called {table} in this project.{hint}")
|
|
174
|
+
declared = list(source.declared_columns)
|
|
175
|
+
if not declared:
|
|
176
|
+
raise CannotCheck(
|
|
177
|
+
f"{source.name} has no declared columns, so there is nothing to check the file against. "
|
|
178
|
+
"Add them to its yml, or run ripple ingest-schema with your warehouse's column list."
|
|
179
|
+
)
|
|
180
|
+
present = {h.lower() for h in header}
|
|
181
|
+
known = {c.lower() for c in declared}
|
|
182
|
+
missing = [c for c in declared if c.lower() not in present]
|
|
183
|
+
new = [h for h in header if h.lower() not in known]
|
|
184
|
+
columns: list[dict] = []
|
|
185
|
+
for c in declared:
|
|
186
|
+
if c.lower() in present:
|
|
187
|
+
columns.append({"name": c, "status": "unchanged"})
|
|
188
|
+
continue
|
|
189
|
+
r = _readers(graph, source.name, c)
|
|
190
|
+
entry = {
|
|
191
|
+
"name": c,
|
|
192
|
+
"status": "missing",
|
|
193
|
+
"readers": {"models": len(r["models"]), "dashboards": len(r["dashboards"])},
|
|
194
|
+
"hit_hardest": [{"model": m, "columns": n} for m, n in r["hit_hardest"]],
|
|
195
|
+
}
|
|
196
|
+
entry["_who"] = r
|
|
197
|
+
columns.append(entry)
|
|
198
|
+
read_missing = [c["name"] for c in columns if c["status"] == "missing" and _count(c)]
|
|
199
|
+
for h in new:
|
|
200
|
+
like = difflib.get_close_matches(
|
|
201
|
+
h.lower(), [m.lower() for m in read_missing], n=1, cutoff=LOOKS_LIKE
|
|
202
|
+
)
|
|
203
|
+
entry = {"name": h, "status": "new"}
|
|
204
|
+
if like:
|
|
205
|
+
entry["looks_like"] = next(m for m in read_missing if m.lower() == like[0])
|
|
206
|
+
columns.append(entry)
|
|
207
|
+
models: set[str] = set()
|
|
208
|
+
dashboards: set[str] = set()
|
|
209
|
+
for c in columns:
|
|
210
|
+
if c["status"] == "missing" and _count(c):
|
|
211
|
+
models |= c["_who"]["models"]
|
|
212
|
+
dashboards |= c["_who"]["dashboards"]
|
|
213
|
+
for c in columns:
|
|
214
|
+
c.pop("_who", None)
|
|
215
|
+
breaking = len(read_missing)
|
|
216
|
+
if breaking:
|
|
217
|
+
verdict = (
|
|
218
|
+
f"Stop: {_plural(breaking, 'column')} that {_where(len(models), len(dashboards), noun)} "
|
|
219
|
+
f"read {'is' if breaking == 1 else 'are'} missing from this file."
|
|
220
|
+
)
|
|
221
|
+
else:
|
|
222
|
+
verdict = "Nothing downstream changes."
|
|
223
|
+
report = {
|
|
224
|
+
"file": file_name,
|
|
225
|
+
"table": source.name,
|
|
226
|
+
"method": how,
|
|
227
|
+
"checked": "names",
|
|
228
|
+
"columns": columns,
|
|
229
|
+
"counts": {
|
|
230
|
+
"file": len(header),
|
|
231
|
+
"unchanged": len(declared) - len(missing),
|
|
232
|
+
"missing": len(missing),
|
|
233
|
+
"new": len(new),
|
|
234
|
+
},
|
|
235
|
+
"verdict": verdict,
|
|
236
|
+
"exit": 1 if breaking else 0,
|
|
237
|
+
"noun": noun,
|
|
238
|
+
}
|
|
239
|
+
report["lines"] = lines(report)
|
|
240
|
+
return report
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def _count(c: dict) -> int:
|
|
244
|
+
return c["readers"]["models"] + c["readers"]["dashboards"]
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _plural(n: int, word: str) -> str:
|
|
248
|
+
return f"{n} {word}" + ("" if n == 1 else "s")
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _where(models: int, dashboards: int, noun: str) -> str:
|
|
252
|
+
parts = []
|
|
253
|
+
if models:
|
|
254
|
+
parts.append(_plural(models, noun))
|
|
255
|
+
if dashboards:
|
|
256
|
+
parts.append(_plural(dashboards, "dashboard"))
|
|
257
|
+
return " and ".join(parts) if parts else "nobody"
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def lines(report: dict) -> list[str]:
|
|
261
|
+
"""The terminal text: verdict first and last, damage in between."""
|
|
262
|
+
noun = report.get("noun", "model")
|
|
263
|
+
counts = report["counts"]
|
|
264
|
+
listed = [c for c in report["columns"] if c["status"] == "missing" and _count(c)]
|
|
265
|
+
hinted = [c for c in report["columns"] if c["status"] == "new" and c.get("looks_like")]
|
|
266
|
+
quiet_new = [
|
|
267
|
+
c["name"] for c in report["columns"] if c["status"] == "new" and not c.get("looks_like")
|
|
268
|
+
]
|
|
269
|
+
quiet_missing = [
|
|
270
|
+
c["name"] for c in report["columns"] if c["status"] == "missing" and not _count(c)
|
|
271
|
+
]
|
|
272
|
+
width = max((len(c["name"]) for c in listed + hinted), default=0) + 3
|
|
273
|
+
out = [report["verdict"], ""]
|
|
274
|
+
out.append(f"File {report['file']}")
|
|
275
|
+
out.append(f"Table {report['table']} ({report['method']})")
|
|
276
|
+
out.append(HEADER_ONLY)
|
|
277
|
+
out.append("")
|
|
278
|
+
parts = [f"{counts['unchanged']} unchanged"]
|
|
279
|
+
if counts["missing"]:
|
|
280
|
+
parts.append(f"{counts['missing']} missing")
|
|
281
|
+
if counts["new"]:
|
|
282
|
+
parts.append(f"{counts['new']} new")
|
|
283
|
+
out.append(f"{_plural(counts['file'], 'column')}: {', '.join(parts)}")
|
|
284
|
+
out.append("")
|
|
285
|
+
for c in sorted(listed, key=lambda c: (-_count(c), c["name"])):
|
|
286
|
+
who = _where(c["readers"]["models"], c["readers"]["dashboards"], noun)
|
|
287
|
+
line = f"MISSING {c['name']:<{width}}read by {who}"
|
|
288
|
+
# worth a name only when one reader is hit in more than one column
|
|
289
|
+
if c["hit_hardest"] and c["hit_hardest"][0]["columns"] > 1:
|
|
290
|
+
top = c["hit_hardest"][0]
|
|
291
|
+
line += f"; hit hardest: {top['model']} ({_plural(top['columns'], 'column')})"
|
|
292
|
+
out.append(line)
|
|
293
|
+
for c in hinted:
|
|
294
|
+
out.append(
|
|
295
|
+
f"NEW {c['name']:<{width}}looks like {c['looks_like']}; "
|
|
296
|
+
"a similar name does not prove the same contents"
|
|
297
|
+
)
|
|
298
|
+
if quiet_new:
|
|
299
|
+
more = "more " if hinted else ""
|
|
300
|
+
out.append(
|
|
301
|
+
f"{len(quiet_new)} {more}new column{'s' if len(quiet_new) != 1 else ''} nobody reads yet: {', '.join(quiet_new)}"
|
|
302
|
+
)
|
|
303
|
+
if quiet_missing:
|
|
304
|
+
out.append(
|
|
305
|
+
f"{len(quiet_missing)} missing column{'s' if len(quiet_missing) != 1 else ''} nobody reads yet: {', '.join(quiet_missing)}"
|
|
306
|
+
)
|
|
307
|
+
if listed:
|
|
308
|
+
out.append("")
|
|
309
|
+
fixes = []
|
|
310
|
+
renamed = {c["looks_like"]: c["name"] for c in hinted}
|
|
311
|
+
for c in listed:
|
|
312
|
+
if c["name"] in renamed:
|
|
313
|
+
fixes.append(
|
|
314
|
+
f"if {renamed[c['name']]} holds the same data as {c['name']}, rename that header to {c['name']}; "
|
|
315
|
+
f"otherwise restore {c['name']}."
|
|
316
|
+
)
|
|
317
|
+
else:
|
|
318
|
+
fixes.append(f"Restore {c['name']}.")
|
|
319
|
+
out.append("To fix: " + " ".join(fixes) + " Then run this check again.")
|
|
320
|
+
out.append(f"Details: ripple breaks {report['table']}.{listed[0]['name']}")
|
|
321
|
+
out.append("")
|
|
322
|
+
out.append(f"{report['verdict']} (exit {report['exit']})")
|
|
323
|
+
return out
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def no_binding_lines(file_name: str, graph) -> list[str]:
|
|
327
|
+
"""What to do when no source claims the file: a hint, never a guess."""
|
|
328
|
+
guess = closest_source(graph, file_name)
|
|
329
|
+
out = [f"Cannot check: no source is mapped to {file_name}."]
|
|
330
|
+
if guess:
|
|
331
|
+
out.append(
|
|
332
|
+
f"It looks like {guess}. Run again with --source {guess}, or add to that source's yml:"
|
|
333
|
+
)
|
|
334
|
+
out.append(f' meta: {{ripple: {{files: ["{file_name}"]}}}}')
|
|
335
|
+
out.append("A pattern like vendor_feed_*.csv covers every drop of the same feed.")
|
|
336
|
+
return out
|
|
@@ -258,6 +258,43 @@ def cmd_breaks(args) -> None:
|
|
|
258
258
|
_page_door(args, project, graph, answer, plain_text(breaks_lines(answer, full=True)))
|
|
259
259
|
|
|
260
260
|
|
|
261
|
+
def cmd_check_file(args) -> None:
|
|
262
|
+
from ripple import check_file
|
|
263
|
+
from ripple.project import find_project_root
|
|
264
|
+
|
|
265
|
+
project, graph = _load_graph(args)
|
|
266
|
+
path = Path(args.file)
|
|
267
|
+
root = find_project_root(Path(args.path))
|
|
268
|
+
try:
|
|
269
|
+
header = check_file.read_header(path)
|
|
270
|
+
table, how = check_file.bind(path.name, args.source, check_file.file_map(root))
|
|
271
|
+
if table is None:
|
|
272
|
+
lines = check_file.no_binding_lines(path.name, graph)
|
|
273
|
+
print(
|
|
274
|
+
json.dumps({"file": path.name, "exit": 2, "lines": lines}, indent=2)
|
|
275
|
+
if args.json
|
|
276
|
+
else "\n".join(lines)
|
|
277
|
+
)
|
|
278
|
+
sys.exit(2)
|
|
279
|
+
report = check_file.check(graph, table, header, path.name, how, noun_for(project.mode))
|
|
280
|
+
except check_file.CannotCheck as e:
|
|
281
|
+
print(
|
|
282
|
+
json.dumps({"file": path.name, "exit": 2, "lines": [f"Cannot check: {e}"]}, indent=2)
|
|
283
|
+
if args.json
|
|
284
|
+
else f"Cannot check: {e}"
|
|
285
|
+
)
|
|
286
|
+
sys.exit(2)
|
|
287
|
+
except OSError as e:
|
|
288
|
+
print(f"Cannot check: {e}")
|
|
289
|
+
sys.exit(2)
|
|
290
|
+
if args.json:
|
|
291
|
+
print(json.dumps(report, indent=2))
|
|
292
|
+
else:
|
|
293
|
+
print("\n".join(report["lines"]))
|
|
294
|
+
if report["exit"]:
|
|
295
|
+
sys.exit(report["exit"])
|
|
296
|
+
|
|
297
|
+
|
|
261
298
|
def cmd_trace(args) -> None:
|
|
262
299
|
project, graph = _load_graph(args)
|
|
263
300
|
model, column = _parse_target(args.target)
|
|
@@ -604,6 +641,15 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
604
641
|
)
|
|
605
642
|
p_columns.add_argument("model")
|
|
606
643
|
|
|
644
|
+
p_check = sub.add_parser(
|
|
645
|
+
"check-file",
|
|
646
|
+
help="a file from an external team: what breaks if it is loaded",
|
|
647
|
+
parents=[common],
|
|
648
|
+
)
|
|
649
|
+
p_check.add_argument("file", help="the csv, tsv, xlsx or parquet file that arrived")
|
|
650
|
+
p_check.add_argument(
|
|
651
|
+
"--source", help="the table it feeds (raw.vendor_feed); else the project's files map"
|
|
652
|
+
)
|
|
607
653
|
p_breaks = sub.add_parser("breaks", help="what breaks if this column changes", parents=[common])
|
|
608
654
|
p_breaks.add_argument("target", help="model.column")
|
|
609
655
|
p_breaks.add_argument(
|
|
@@ -716,6 +762,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
716
762
|
"models": cmd_models,
|
|
717
763
|
"columns": cmd_columns,
|
|
718
764
|
"breaks": cmd_breaks,
|
|
765
|
+
"check-file": cmd_check_file,
|
|
719
766
|
"trace": cmd_trace,
|
|
720
767
|
"graph": cmd_graph,
|
|
721
768
|
"ci": cmd_ci,
|
|
@@ -339,3 +339,47 @@ def _dbt_owned_dirs(project_dir: Path) -> list[Path]:
|
|
|
339
339
|
owned += _dbt_model_dirs(project_dir)
|
|
340
340
|
owned += [project_dir / name for name in _macro_dir_names(project_dir / "dbt_project.yml")]
|
|
341
341
|
return [d for d in owned if d.is_dir()]
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def source_tables(root: Path):
|
|
345
|
+
"""Every (source name, table dict) a dbt project declares in its yml.
|
|
346
|
+
|
|
347
|
+
Walks the model paths only (dbt_project.yml's model-paths), so a
|
|
348
|
+
package's or a build folder's yml never counts as this project's."""
|
|
349
|
+
import yaml
|
|
350
|
+
|
|
351
|
+
root = Path(root)
|
|
352
|
+
if not (root / "dbt_project.yml").is_file():
|
|
353
|
+
return
|
|
354
|
+
for model_dir in _dbt_model_dirs(root):
|
|
355
|
+
for yml in [*files_under(model_dir, "*.yml"), *files_under(model_dir, "*.yaml")]:
|
|
356
|
+
try:
|
|
357
|
+
parsed = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
|
|
358
|
+
except yaml.YAMLError:
|
|
359
|
+
continue
|
|
360
|
+
if not isinstance(parsed, dict):
|
|
361
|
+
continue
|
|
362
|
+
for source in parsed.get("sources") or []:
|
|
363
|
+
if not isinstance(source, dict) or not source.get("name"):
|
|
364
|
+
continue
|
|
365
|
+
for table in source.get("tables") or []:
|
|
366
|
+
if isinstance(table, dict) and table.get("name"):
|
|
367
|
+
yield str(source["name"]), table
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def source_columns(root: Path) -> dict[str, list[str]]:
|
|
371
|
+
"""The columns sources.yml declares, as {"schema.table": [names]}.
|
|
372
|
+
|
|
373
|
+
dbt users list a source's columns in yml for docs and tests; those are
|
|
374
|
+
the landing table's contract, and the graph read only ingested schemas
|
|
375
|
+
and seeds before (a dbt source always showed "not declared")."""
|
|
376
|
+
out: dict[str, list[str]] = {}
|
|
377
|
+
for source, table in source_tables(root):
|
|
378
|
+
names = []
|
|
379
|
+
for column in table.get("columns") or []:
|
|
380
|
+
name = column.get("name") if isinstance(column, dict) else column
|
|
381
|
+
if name:
|
|
382
|
+
names.append(str(name))
|
|
383
|
+
if names:
|
|
384
|
+
out[f"{source}.{table['name']}"] = names
|
|
385
|
+
return out
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import re
|
|
6
|
+
from pathlib import Path
|
|
6
7
|
|
|
7
8
|
from ripple.loaders.types import Model, Project
|
|
8
9
|
|
|
@@ -93,14 +94,22 @@ def _apply_ingested_schemas(project: Project, extra_roots: tuple = ()) -> None:
|
|
|
93
94
|
a new source with declared columns. A model the project actually derives
|
|
94
95
|
keeps its own analysis: ingested schemas fill gaps, they never override.
|
|
95
96
|
"""
|
|
97
|
+
from ripple.loaders.dbt_config import source_columns
|
|
96
98
|
from ripple.schemas import load_schemas
|
|
97
99
|
|
|
98
100
|
# the CLI and MCP write .ripple/schemas.json at the repo root they were
|
|
99
101
|
# run from; a dbt project in a subfolder (balboa's transform/) has its
|
|
100
|
-
# own root, and reading only there made every ingest a silent no-op
|
|
102
|
+
# own root, and reading only there made every ingest a silent no-op.
|
|
103
|
+
# sources.yml columns come first; an ingest adds to them
|
|
101
104
|
tables: dict = {}
|
|
102
105
|
for root in dict.fromkeys([project.root, *extra_roots]):
|
|
103
|
-
tables.update(
|
|
106
|
+
tables.update(source_columns(Path(root)))
|
|
107
|
+
for root in dict.fromkeys([project.root, *extra_roots]):
|
|
108
|
+
for name, columns in (load_schemas(root) or {}).items():
|
|
109
|
+
tables[name] = [
|
|
110
|
+
*tables.get(name, []),
|
|
111
|
+
*[c for c in columns if c not in tables.get(name, [])],
|
|
112
|
+
]
|
|
104
113
|
if not tables:
|
|
105
114
|
return
|
|
106
115
|
# every relation a spelling names, not the first: a monorepo declares the
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|