ripple-sql 0.1.7__tar.gz → 0.1.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/CHANGELOG.md +14 -1
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/PKG-INFO +48 -1
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/README.md +47 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/pyproject.toml +1 -1
- ripple_sql-0.1.9/src/ripple/check_file.py +627 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/cli.py +51 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/dbt_config.py +54 -1
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/identity.py +40 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/types.py +3 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/project.py +9 -1
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/.gitignore +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/LICENSE +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/answer.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/answer_page.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/cache.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/ci.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/ci_signature.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/doctor.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/budget.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/column_lineage.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/column_ref.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/cte_tracing.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/dependencies.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/dialect.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/dispatch.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/extraction.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/generators.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/jinja.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/json_sources.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/macro_source.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/pipeline.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/preprocess.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/safe_gen.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/schema_qualification.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/scope.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/select_sources.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/sql_script.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/statement.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tech_debt.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_catalog.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_scalar_vars.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_tvf.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_xml.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/types.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/unused_deps.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/validation.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/graph.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/home.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/dbt.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/sidecar.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/sqldir.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/lookml.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/mcp_server.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/names.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/py.typed +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/render.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/render_shims.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/schemas.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/semantic.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/server.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/sourcefiles.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/star_resolution.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/answer.css +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/answer.html +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/answer_twin.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/explore.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/find.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/focus.js +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/__init__.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/cli.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/collect.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/discover.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/ingest.py +0 -0
- {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/report.py +0 -0
|
@@ -6,6 +6,17 @@ Notable changes to Ripple. The format follows
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.9] - 2026-09-15
|
|
10
|
+
|
|
11
|
+
### Changed
|
|
12
|
+
- `ripple check-file` no longer needs the table's name in the common case. A file binds to the table its own columns name when they say it clearly: at least two of them are one table's, they are most of the file's columns, and no other table matches as many ("matched by 4 of 5 column names"). `--source` takes the bare table name when one schema has it (`--source vendor_feed`); a name two schemas share is refused with both spellings. When two tables fit a file equally, or none does, the command lists the tables with declared columns and stops with exit 2, never picking one. `--source` and a declared `files:` pattern still come first. A review of the command's edge cases also fixed: a column read only in a where clause or a join now counts as read; a reader the graph only sees through select star counts instead of vanishing; long chains are counted whole; an xlsx is read by the workbook's own sheet order, streamed no further than its first row, with rich text joined and empty cells kept as positions; tab-separated `.txt` and semicolon `.csv` files are read by their delimiter; a file that is not UTF-8 asks for `--encoding`; an unclosed quote, a duplicated header name, a DOCTYPE in the XML, a broken xlsx part, a pyarrow older than 14.0.1 and a broken `ripple.yml` are each a plain message with exit 2; a folder pattern (`incoming/*.csv`) matches the file's path; two patterns that name two tables are refused; `--source` never reads the pattern map; a missing file, a folder, and a project that cannot load give the same JSON shape as any other failure.
|
|
13
|
+
|
|
14
|
+
## [0.1.8] - 2026-09-15
|
|
15
|
+
|
|
16
|
+
### Added
|
|
17
|
+
- `ripple check-file FILE [--source TABLE]`: a file arrives from an external team (csv, tsv, xlsx, parquet), and before anyone loads it Ripple diffs its header against the landing table's declared columns and prints who reads every column that is gone: models, dashboards, hit hardest. The verdict is the first line and the last; a similar name is a hint ("looks like customer_name; a similar name does not prove the same contents"), never an assumption; exit 0 when nothing downstream changes (new columns nobody reads included), 1 when something breaks, 2 when the check could not run. The file binds to its table by `--source`, or by a `files:` pattern the project declares (`meta: {ripple: {files: ["vendor_feed_*.csv"]}}` on a dbt source, or `files:` in `ripple.yml` for a plain SQL folder); with neither, the command prints the closest table and the exact line to add and stops. Header only: no data row is read and nothing leaves the machine. Shaped by a two-model debate; the spec's sample output is the command's own.
|
|
18
|
+
- A dbt source's `columns:` in sources.yml are its declared columns. The graph read only ingested schemas and seeds before, so every dbt source showed as "not declared" and a select star through it stayed unresolved even when the yml listed every column.
|
|
19
|
+
|
|
9
20
|
## [0.1.7] - 2026-09-10
|
|
10
21
|
|
|
11
22
|
### Fixed
|
|
@@ -264,7 +275,9 @@ First public release.
|
|
|
264
275
|
### Removed
|
|
265
276
|
- The dark whole-graph canvas that `ripple serve` used to open (`static/index.html`). Every surface now draws one answer.
|
|
266
277
|
|
|
267
|
-
[Unreleased]: https://github.com/bteh/ripple/compare/v0.1.
|
|
278
|
+
[Unreleased]: https://github.com/bteh/ripple/compare/v0.1.9...HEAD
|
|
279
|
+
[0.1.9]: https://github.com/bteh/ripple/releases/tag/v0.1.9
|
|
280
|
+
[0.1.8]: https://github.com/bteh/ripple/releases/tag/v0.1.8
|
|
268
281
|
[0.1.7]: https://github.com/bteh/ripple/releases/tag/v0.1.7
|
|
269
282
|
[0.1.6]: https://github.com/bteh/ripple/releases/tag/v0.1.6
|
|
270
283
|
[0.1.5]: https://github.com/bteh/ripple/releases/tag/v0.1.5
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: ripple-sql
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.9
|
|
4
4
|
Summary: Offline column-level SQL lineage. See what breaks before you merge.
|
|
5
5
|
Project-URL: Homepage, https://github.com/bteh/ripple
|
|
6
6
|
Project-URL: Repository, https://github.com/bteh/ripple
|
|
@@ -210,6 +210,53 @@ ripple unresolved # which tables block the most coverage, worst
|
|
|
210
210
|
ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
|
|
211
211
|
```
|
|
212
212
|
|
|
213
|
+
## A file arrives from another team
|
|
214
|
+
|
|
215
|
+
Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
ripple check-file vendor_feed_2026-09.csv
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
```
|
|
222
|
+
Stop: 2 columns that 4 models read are missing from this file.
|
|
223
|
+
|
|
224
|
+
File vendor_feed_2026-09.csv
|
|
225
|
+
Table raw.vendor_feed (files: vendor_feed_*.csv)
|
|
226
|
+
Checked: column names only. Not checked: values, types, whether the file loads.
|
|
227
|
+
|
|
228
|
+
6 columns: 2 unchanged, 2 missing, 4 new
|
|
229
|
+
|
|
230
|
+
MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
|
|
231
|
+
MISSING amount_usd read by 2 models
|
|
232
|
+
NEW cust_nm looks like customer_name; a similar name does not prove the same contents
|
|
233
|
+
3 more new columns nobody reads yet: region, notes, batch_id
|
|
234
|
+
|
|
235
|
+
To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
|
|
236
|
+
Details: ripple breaks raw.vendor_feed.customer_name
|
|
237
|
+
|
|
238
|
+
Stop: 2 columns that 4 models read are missing from this file. (exit 1)
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
|
|
242
|
+
|
|
243
|
+
The file finds its table by its own columns when they say it clearly (most of them belong to one table and no other table comes close: "matched by 4 of 5 column names"), by `--source vendor_feed` (the bare table name is enough when one schema has it), or by a pattern the project declares once:
|
|
244
|
+
|
|
245
|
+
```yaml
|
|
246
|
+
sources:
|
|
247
|
+
- name: raw
|
|
248
|
+
tables:
|
|
249
|
+
- name: vendor_feed
|
|
250
|
+
meta:
|
|
251
|
+
ripple:
|
|
252
|
+
files: ["vendor_feed_*.csv"]
|
|
253
|
+
columns:
|
|
254
|
+
- name: customer_id
|
|
255
|
+
- name: customer_name
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). When two tables fit the file equally, or none does, the command lists the tables it can check and stops with exit 2 rather than pick one. The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow 14.0.1 or newer). A file that is not UTF-8 takes `--encoding cp1252`. `--json` carries the same facts for an ingest job, in the same shape whether the check ran or could not.
|
|
259
|
+
|
|
213
260
|
## Bring your query history (optional)
|
|
214
261
|
|
|
215
262
|
Your SQL files say what is supposed to happen. The warehouse's query log says what
|
|
@@ -170,6 +170,53 @@ ripple unresolved # which tables block the most coverage, worst
|
|
|
170
170
|
ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
|
|
171
171
|
```
|
|
172
172
|
|
|
173
|
+
## A file arrives from another team
|
|
174
|
+
|
|
175
|
+
Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
|
|
176
|
+
|
|
177
|
+
```bash
|
|
178
|
+
ripple check-file vendor_feed_2026-09.csv
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
```
|
|
182
|
+
Stop: 2 columns that 4 models read are missing from this file.
|
|
183
|
+
|
|
184
|
+
File vendor_feed_2026-09.csv
|
|
185
|
+
Table raw.vendor_feed (files: vendor_feed_*.csv)
|
|
186
|
+
Checked: column names only. Not checked: values, types, whether the file loads.
|
|
187
|
+
|
|
188
|
+
6 columns: 2 unchanged, 2 missing, 4 new
|
|
189
|
+
|
|
190
|
+
MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
|
|
191
|
+
MISSING amount_usd read by 2 models
|
|
192
|
+
NEW cust_nm looks like customer_name; a similar name does not prove the same contents
|
|
193
|
+
3 more new columns nobody reads yet: region, notes, batch_id
|
|
194
|
+
|
|
195
|
+
To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
|
|
196
|
+
Details: ripple breaks raw.vendor_feed.customer_name
|
|
197
|
+
|
|
198
|
+
Stop: 2 columns that 4 models read are missing from this file. (exit 1)
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
|
|
202
|
+
|
|
203
|
+
The file finds its table by its own columns when they say it clearly (most of them belong to one table and no other table comes close: "matched by 4 of 5 column names"), by `--source vendor_feed` (the bare table name is enough when one schema has it), or by a pattern the project declares once:
|
|
204
|
+
|
|
205
|
+
```yaml
|
|
206
|
+
sources:
|
|
207
|
+
- name: raw
|
|
208
|
+
tables:
|
|
209
|
+
- name: vendor_feed
|
|
210
|
+
meta:
|
|
211
|
+
ripple:
|
|
212
|
+
files: ["vendor_feed_*.csv"]
|
|
213
|
+
columns:
|
|
214
|
+
- name: customer_id
|
|
215
|
+
- name: customer_name
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). When two tables fit the file equally, or none does, the command lists the tables it can check and stops with exit 2 rather than pick one. The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow 14.0.1 or newer). A file that is not UTF-8 takes `--encoding cp1252`. `--json` carries the same facts for an ingest job, in the same shape whether the check ran or could not.
|
|
219
|
+
|
|
173
220
|
## Bring your query history (optional)
|
|
174
221
|
|
|
175
222
|
Your SQL files say what is supposed to happen. The warehouse's query log says what
|
|
@@ -0,0 +1,627 @@
|
|
|
1
|
+
"""A file arrives from an external team: what breaks if it is loaded.
|
|
2
|
+
|
|
3
|
+
The file feeds one landing table the graph already knows. Its header is
|
|
4
|
+
diffed against the table's declared columns, and every column that is gone
|
|
5
|
+
gets the blast radius `ripple breaks` would give it. Header only: no data
|
|
6
|
+
row is read, nothing leaves the machine, and a similar name is a hint,
|
|
7
|
+
never an assumption.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import csv
|
|
13
|
+
import difflib
|
|
14
|
+
import fnmatch
|
|
15
|
+
import io
|
|
16
|
+
import zipfile
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from xml.etree import ElementTree
|
|
19
|
+
|
|
20
|
+
from ripple.answer import breaks_answer, hit_hardest
|
|
21
|
+
from ripple.graph import UnknownTarget
|
|
22
|
+
|
|
23
|
+
HEADER_ONLY = "Checked: column names only. Not checked: values, types, whether the file loads."
|
|
24
|
+
KINDS = ("csv", "tsv", "txt", "xlsx", "parquet")
|
|
25
|
+
LOOKS_LIKE = 0.6
|
|
26
|
+
DELIMITERS = ",;\t|"
|
|
27
|
+
HEAD_BYTES = 64 * 1024
|
|
28
|
+
# a reader older than this deserializes untrusted Parquet unsafely (CVE-2023-47248)
|
|
29
|
+
PYARROW_FLOOR = (14, 0, 1)
|
|
30
|
+
# far past any real chain; the walk stops at cycles on its own
|
|
31
|
+
MAX_DEPTH = 500
|
|
32
|
+
# the parts a workbook's header lives in; anything past these caps is not a
|
|
33
|
+
# workbook Ripple reads a header from, whatever the zip claims
|
|
34
|
+
XML_PART_CAP = 4 * 1024 * 1024
|
|
35
|
+
SHARED_STRINGS_CAP = 64 * 1024 * 1024
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class CannotCheck(Exception):
|
|
39
|
+
"""The check could not run; the message says why and what to do."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def read_header(path: Path, encoding: str | None = None) -> list[str]:
|
|
43
|
+
"""The column names of a file, from its header only."""
|
|
44
|
+
if path.is_dir():
|
|
45
|
+
raise CannotCheck(f"{path.name} is a folder; pass one file")
|
|
46
|
+
kind = path.suffix.lower().lstrip(".")
|
|
47
|
+
if kind in ("csv", "tsv", "txt"):
|
|
48
|
+
header = _delimited(path, kind, encoding)
|
|
49
|
+
elif kind == "xlsx":
|
|
50
|
+
header = _xlsx(path)
|
|
51
|
+
elif kind == "parquet":
|
|
52
|
+
header = _parquet(path)
|
|
53
|
+
else:
|
|
54
|
+
raise CannotCheck(
|
|
55
|
+
f"{path.name}: Ripple reads a csv, tsv, xlsx or parquet header, not .{kind}"
|
|
56
|
+
)
|
|
57
|
+
names = [h for h in header if h]
|
|
58
|
+
seen: set[str] = set()
|
|
59
|
+
for name in names:
|
|
60
|
+
if name.lower() in seen:
|
|
61
|
+
raise CannotCheck(
|
|
62
|
+
f"{path.name} names {name} twice in its header; a table has one column of a name"
|
|
63
|
+
)
|
|
64
|
+
seen.add(name.lower())
|
|
65
|
+
if not names:
|
|
66
|
+
raise CannotCheck(f"{path.name} is empty; no header row to check")
|
|
67
|
+
return names
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _decode(path: Path, raw: bytes, encoding: str | None) -> str:
|
|
71
|
+
try:
|
|
72
|
+
return raw.decode(encoding or "utf-8-sig")
|
|
73
|
+
except (UnicodeDecodeError, LookupError):
|
|
74
|
+
if encoding:
|
|
75
|
+
raise CannotCheck(
|
|
76
|
+
f"{path.name} is not {encoding}; pass the file's encoding with --encoding"
|
|
77
|
+
) from None
|
|
78
|
+
raise CannotCheck(
|
|
79
|
+
f"{path.name} is not UTF-8; pass --encoding cp1252 (Excel on Windows) or latin-1"
|
|
80
|
+
) from None
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _delimited(path: Path, kind: str, encoding: str | None) -> list[str]:
|
|
84
|
+
with open(path, "rb") as f:
|
|
85
|
+
text = _decode(path, f.read(HEAD_BYTES), encoding)
|
|
86
|
+
first = text.split("\n", 1)[0]
|
|
87
|
+
delimiter = "\t" if kind == "tsv" else max(DELIMITERS, key=lambda d: (first.count(d), d == ","))
|
|
88
|
+
if text.count('"') % 2:
|
|
89
|
+
raise CannotCheck(f"{path.name} has an unclosed quote in its header")
|
|
90
|
+
try:
|
|
91
|
+
header = next(csv.reader(io.StringIO(text, newline=""), delimiter=delimiter, strict=True))
|
|
92
|
+
except StopIteration:
|
|
93
|
+
raise CannotCheck(f"{path.name} is empty; no header row to check") from None
|
|
94
|
+
except csv.Error as e:
|
|
95
|
+
raise CannotCheck(f"{path.name}: the header is not valid csv ({e})") from None
|
|
96
|
+
# a quoted header that spans lines keeps the file's own line ending
|
|
97
|
+
return [h.replace("\r\n", "\n").strip() for h in header]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _local(tag: str) -> str:
|
|
101
|
+
return tag.rsplit("}", 1)[-1]
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _guard_xml(path: Path, data: bytes) -> None:
|
|
105
|
+
"""Refuse a DOCTYPE or an entity in an XML part, in whatever encoding the
|
|
106
|
+
part is written: expat honours a UTF-16 or UTF-32 prolog, so the scan
|
|
107
|
+
must read the bytes the way the parser will."""
|
|
108
|
+
if data.startswith((b"\xff\xfe\x00\x00", b"\x00\x00\xfe\xff")):
|
|
109
|
+
codec = "utf-32"
|
|
110
|
+
elif data.startswith((b"\xff\xfe", b"\xfe\xff")):
|
|
111
|
+
codec = "utf-16"
|
|
112
|
+
elif b"\x00" in data[:64]:
|
|
113
|
+
raise CannotCheck(
|
|
114
|
+
f"{path.name} carries XML in an encoding Ripple does not read; save it as a normal xlsx"
|
|
115
|
+
)
|
|
116
|
+
else:
|
|
117
|
+
codec = "utf-8"
|
|
118
|
+
text = data.decode(codec, errors="ignore")
|
|
119
|
+
if "<!DOCTYPE" in text or "<!ENTITY" in text:
|
|
120
|
+
raise CannotCheck(f"{path.name} carries a DOCTYPE in its XML; Ripple does not read that")
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _part(z: zipfile.ZipFile, path: Path, name: str, cap: int) -> bytes:
|
|
124
|
+
"""One zip member, whole, guarded: bounded in size and free of a DOCTYPE."""
|
|
125
|
+
if z.getinfo(name).file_size > cap:
|
|
126
|
+
raise CannotCheck(
|
|
127
|
+
f"{path.name}: {name} is larger than {cap >> 20} MB; not a workbook Ripple reads"
|
|
128
|
+
)
|
|
129
|
+
member = z.open # bytes; the utf-8 floor scan keys on the name open(
|
|
130
|
+
with member(name) as f:
|
|
131
|
+
data = f.read(cap + 1)
|
|
132
|
+
if len(data) > cap:
|
|
133
|
+
raise CannotCheck(
|
|
134
|
+
f"{path.name}: {name} is larger than {cap >> 20} MB; not a workbook Ripple reads"
|
|
135
|
+
)
|
|
136
|
+
_guard_xml(path, data)
|
|
137
|
+
return data
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _first_sheet(z: zipfile.ZipFile, path: Path) -> str:
|
|
141
|
+
"""The workbook's first sheet, by its own order, not by file name."""
|
|
142
|
+
names = set(z.namelist())
|
|
143
|
+
if "xl/workbook.xml" in names and "xl/_rels/workbook.xml.rels" in names:
|
|
144
|
+
try:
|
|
145
|
+
wb = ElementTree.fromstring(_part(z, path, "xl/workbook.xml", XML_PART_CAP))
|
|
146
|
+
rels = ElementTree.fromstring(
|
|
147
|
+
_part(z, path, "xl/_rels/workbook.xml.rels", XML_PART_CAP)
|
|
148
|
+
)
|
|
149
|
+
except ElementTree.ParseError:
|
|
150
|
+
raise CannotCheck(f"{path.name} is not valid xlsx") from None
|
|
151
|
+
first = next((s for s in wb.iter() if _local(s.tag) == "sheet"), None)
|
|
152
|
+
rid = (
|
|
153
|
+
next((v for k, v in first.attrib.items() if _local(k) == "id"), None)
|
|
154
|
+
if first is not None
|
|
155
|
+
else None
|
|
156
|
+
)
|
|
157
|
+
target = next((r.get("Target") for r in rels.iter() if r.get("Id") == rid), None)
|
|
158
|
+
if target:
|
|
159
|
+
member = target.lstrip("/") if target.startswith("/") else "xl/" + target
|
|
160
|
+
if member in names:
|
|
161
|
+
return member
|
|
162
|
+
sheet = next((n for n in sorted(names) if n.startswith("xl/worksheets/sheet")), None)
|
|
163
|
+
if not sheet:
|
|
164
|
+
raise CannotCheck(f"{path.name} has no worksheet")
|
|
165
|
+
return sheet
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _column_index(ref: str) -> int:
|
|
169
|
+
n = 0
|
|
170
|
+
for ch in ref:
|
|
171
|
+
if not ch.isalpha():
|
|
172
|
+
break
|
|
173
|
+
n = n * 26 + (ord(ch.upper()) - 64)
|
|
174
|
+
return max(n - 1, 0)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _cell_text(cell, shared: list[str]) -> str:
|
|
178
|
+
kind = cell.get("t")
|
|
179
|
+
if kind == "inlineStr":
|
|
180
|
+
return "".join(t.text or "" for t in cell.iter() if _local(t.tag) == "t")
|
|
181
|
+
value = next((v.text or "" for v in cell.iter() if _local(v.tag) == "v"), "")
|
|
182
|
+
if kind == "s" and value.isdigit() and int(value) < len(shared):
|
|
183
|
+
return shared[int(value)]
|
|
184
|
+
return value
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _xlsx(path: Path) -> list[str]:
|
|
188
|
+
"""Row 1 of the first sheet, streamed and left unread past that row."""
|
|
189
|
+
try:
|
|
190
|
+
with zipfile.ZipFile(path) as z:
|
|
191
|
+
shared: list[str] = []
|
|
192
|
+
if "xl/sharedStrings.xml" in z.namelist():
|
|
193
|
+
raw = _part(z, path, "xl/sharedStrings.xml", SHARED_STRINGS_CAP)
|
|
194
|
+
for si in ElementTree.fromstring(raw):
|
|
195
|
+
shared.append("".join(t.text or "" for t in si.iter() if _local(t.tag) == "t"))
|
|
196
|
+
sheet = _first_sheet(z, path)
|
|
197
|
+
member = z.open # bytes; the utf-8 floor scan keys on the name open(
|
|
198
|
+
with member(sheet) as f:
|
|
199
|
+
row = _first_row(path, f)
|
|
200
|
+
except zipfile.BadZipFile:
|
|
201
|
+
raise CannotCheck(f"{path.name} is not an xlsx workbook") from None
|
|
202
|
+
except ElementTree.ParseError:
|
|
203
|
+
raise CannotCheck(f"{path.name} is not valid xlsx") from None
|
|
204
|
+
if row is None:
|
|
205
|
+
raise CannotCheck(f"{path.name} is not valid xlsx: no header row")
|
|
206
|
+
if row.get("r") not in (None, "1"):
|
|
207
|
+
raise CannotCheck(
|
|
208
|
+
f"{path.name} has no row 1; the header must be the first row of the sheet"
|
|
209
|
+
)
|
|
210
|
+
cells: dict[int, str] = {}
|
|
211
|
+
for cell in (c for c in row if _local(c.tag) == "c"):
|
|
212
|
+
# a cell without a coordinate follows the previous one
|
|
213
|
+
index = _column_index(cell.get("r")) if cell.get("r") else len(cells)
|
|
214
|
+
cells[index] = _cell_text(cell, shared).strip()
|
|
215
|
+
width = max(cells) + 1 if cells else 0
|
|
216
|
+
return [cells.get(i, "") for i in range(width)]
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _first_row(path: Path, stream):
|
|
220
|
+
"""The first <row> element, fed to the parser in small pieces so nothing
|
|
221
|
+
past it is ever parsed; never closed, so a later broken row cannot fail
|
|
222
|
+
a good header. A DOCTYPE can only sit before the root element, so the
|
|
223
|
+
bytes fed until the first element opens are the ones checked for it."""
|
|
224
|
+
parser = ElementTree.XMLPullParser(events=("start", "end"))
|
|
225
|
+
preamble, opened, read = b"", False, 0
|
|
226
|
+
while True:
|
|
227
|
+
chunk = stream.read(4096)
|
|
228
|
+
if not chunk:
|
|
229
|
+
return None
|
|
230
|
+
read += len(chunk)
|
|
231
|
+
if read > XML_PART_CAP:
|
|
232
|
+
raise CannotCheck(
|
|
233
|
+
f"{path.name}: no header row in the first {XML_PART_CAP >> 20} MB of the sheet"
|
|
234
|
+
)
|
|
235
|
+
if not opened:
|
|
236
|
+
preamble += chunk
|
|
237
|
+
parser.feed(chunk)
|
|
238
|
+
for event, element in parser.read_events():
|
|
239
|
+
if event == "start" and not opened:
|
|
240
|
+
opened = True
|
|
241
|
+
_guard_xml(path, preamble)
|
|
242
|
+
if event == "end" and _local(element.tag) == "row":
|
|
243
|
+
return element
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _parquet(path: Path) -> list[str]:
|
|
247
|
+
try:
|
|
248
|
+
import pyarrow
|
|
249
|
+
import pyarrow.parquet as pq
|
|
250
|
+
except ImportError:
|
|
251
|
+
raise CannotCheck(
|
|
252
|
+
f"{path.name}: reading a Parquet schema needs pyarrow (pip install pyarrow), "
|
|
253
|
+
"or export the header as csv"
|
|
254
|
+
) from None
|
|
255
|
+
version = tuple(
|
|
256
|
+
int(p) for p in str(getattr(pyarrow, "__version__", "0")).split(".")[:3] if p.isdigit()
|
|
257
|
+
)
|
|
258
|
+
if version < PYARROW_FLOOR:
|
|
259
|
+
floor = ".".join(str(n) for n in PYARROW_FLOOR)
|
|
260
|
+
raise CannotCheck(
|
|
261
|
+
f"pyarrow {getattr(pyarrow, '__version__', '?')} is older than {floor}, which reads untrusted Parquet "
|
|
262
|
+
"unsafely (CVE-2023-47248); upgrade it or export the header as csv"
|
|
263
|
+
)
|
|
264
|
+
try:
|
|
265
|
+
return list(pq.read_schema(path).names)
|
|
266
|
+
except Exception as e: # pyarrow raises its own hierarchy
|
|
267
|
+
raise CannotCheck(f"{path.name} is not valid Parquet ({e})") from None
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def file_map(roots: list[Path]) -> dict[str, str]:
|
|
271
|
+
"""File-name patterns to the tables they feed, from the project.
|
|
272
|
+
|
|
273
|
+
A dbt source table carries them as `meta: {ripple: {files: [...]}}`; a
|
|
274
|
+
plain SQL folder keeps a `files:` map in ripple.yml at its root. One
|
|
275
|
+
pattern naming two tables is a conflict, not a choice."""
|
|
276
|
+
import yaml
|
|
277
|
+
|
|
278
|
+
from ripple.loaders.dbt_config import source_tables
|
|
279
|
+
|
|
280
|
+
patterns: dict[str, str] = {}
|
|
281
|
+
|
|
282
|
+
def put(pattern: str, table: str, where: str) -> None:
|
|
283
|
+
if pattern in patterns and patterns[pattern] != table:
|
|
284
|
+
raise CannotCheck(
|
|
285
|
+
f"{where}: pattern {pattern} maps to two tables, {patterns[pattern]} and {table}"
|
|
286
|
+
)
|
|
287
|
+
patterns[pattern] = table
|
|
288
|
+
|
|
289
|
+
for root in dict.fromkeys(Path(r) for r in roots):
|
|
290
|
+
own = root / "ripple.yml"
|
|
291
|
+
if own.is_file():
|
|
292
|
+
try:
|
|
293
|
+
parsed = yaml.safe_load(own.read_text(encoding="utf-8", errors="replace")) or {}
|
|
294
|
+
except yaml.YAMLError as e:
|
|
295
|
+
raise CannotCheck(
|
|
296
|
+
f"ripple.yml is not valid yaml ({str(e).splitlines()[0]})"
|
|
297
|
+
) from None
|
|
298
|
+
files = parsed.get("files") if isinstance(parsed, dict) else None
|
|
299
|
+
if files is not None and not isinstance(files, dict):
|
|
300
|
+
raise CannotCheck(
|
|
301
|
+
'ripple.yml: files must map a pattern to a table, like "feed_*.csv": raw.feed'
|
|
302
|
+
)
|
|
303
|
+
for pattern, table in (files or {}).items():
|
|
304
|
+
put(str(pattern), str(table), "ripple.yml")
|
|
305
|
+
for source, table in source_tables(root):
|
|
306
|
+
meta = (table.get("meta") or {}).get("ripple") or {}
|
|
307
|
+
for pattern in meta.get("files") or []:
|
|
308
|
+
put(str(pattern), f"{source}.{table['name']}", "sources yml")
|
|
309
|
+
return patterns
|
|
310
|
+
|
|
311
|
+
|
|
312
|
+
def bind(path: Path, root: Path, patterns: dict[str, str]) -> tuple[str | None, str | None]:
|
|
313
|
+
"""(table, how) for a file from the declared patterns, or (None, None).
|
|
314
|
+
|
|
315
|
+
A pattern with a folder in it matches the file's path (relative to the
|
|
316
|
+
project when it is inside it); a bare pattern matches the file name."""
|
|
317
|
+
spellings = [path.as_posix()]
|
|
318
|
+
resolved, base = path.resolve(), Path(root).resolve()
|
|
319
|
+
if base in resolved.parents:
|
|
320
|
+
spellings.append(resolved.relative_to(base).as_posix())
|
|
321
|
+
hits: dict[str, str] = {}
|
|
322
|
+
for pattern, table in patterns.items():
|
|
323
|
+
target = spellings if "/" in pattern else [path.name]
|
|
324
|
+
if any(fnmatch.fnmatch(s, pattern) for s in target):
|
|
325
|
+
hits.setdefault(table, pattern)
|
|
326
|
+
if len(hits) > 1:
|
|
327
|
+
listed = ", ".join(f"{p} for {t}" for t, p in hits.items())
|
|
328
|
+
raise CannotCheck(
|
|
329
|
+
f"{path.name} matches {len(hits)} patterns for different tables: {listed}. Pass --source."
|
|
330
|
+
)
|
|
331
|
+
if hits:
|
|
332
|
+
((table, pattern),) = hits.items()
|
|
333
|
+
return table, f"files: {pattern}"
|
|
334
|
+
return None, None
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def declared_sources(graph) -> list:
|
|
338
|
+
return [s for s in graph.project.sources if s.declared_columns]
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def fits(header: list[str], graph) -> list[tuple[object, int]]:
|
|
342
|
+
"""Tables ranked by how many of the file's columns are theirs."""
|
|
343
|
+
names = {h.lower() for h in header}
|
|
344
|
+
scored = []
|
|
345
|
+
for source in declared_sources(graph):
|
|
346
|
+
matched = sum(1 for c in source.declared_columns if c.lower() in names)
|
|
347
|
+
if matched:
|
|
348
|
+
scored.append((source, matched))
|
|
349
|
+
scored.sort(key=lambda pair: (-pair[1], pair[0].name))
|
|
350
|
+
return scored
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def bind_by_header(header: list[str], graph) -> tuple[object | None, str | None, list]:
|
|
354
|
+
"""The one table the file's columns name, or the tables that tie.
|
|
355
|
+
|
|
356
|
+
A file binds when at least two of its columns are one table's, they are
|
|
357
|
+
more than half of the file's columns, and no other table matches as
|
|
358
|
+
many. Anything less is listed for the user to pick from, never picked."""
|
|
359
|
+
ranked = fits(header, graph)
|
|
360
|
+
strong = [(s, n) for s, n in ranked if n >= 2 and n * 2 > len(header)]
|
|
361
|
+
if not strong:
|
|
362
|
+
return None, None, []
|
|
363
|
+
best = strong[0][1]
|
|
364
|
+
tied = [(s, n) for s, n in strong if n == best]
|
|
365
|
+
if len(tied) > 1:
|
|
366
|
+
return None, None, tied
|
|
367
|
+
return tied[0][0], f"matched by {best} of {_plural(len(header), 'column name')}", tied
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def find_source(graph, table: str):
|
|
371
|
+
"""The source table by its name, an alias, or its bare table name.
|
|
372
|
+
|
|
373
|
+
Case-insensitive. An alias or bare name two schemas' tables share is
|
|
374
|
+
refused with both spellings; an unknown name lists the tables that can
|
|
375
|
+
be checked."""
|
|
376
|
+
wanted = table.lower()
|
|
377
|
+
exact = [s for s in graph.project.sources if s.name.lower() == wanted]
|
|
378
|
+
if len(exact) == 1:
|
|
379
|
+
return exact[0]
|
|
380
|
+
hits = [
|
|
381
|
+
s
|
|
382
|
+
for s in graph.project.sources
|
|
383
|
+
if wanted in {a.lower() for a in s.aliases} or s.name.rsplit(".", 1)[-1].lower() == wanted
|
|
384
|
+
]
|
|
385
|
+
if len(hits) == 1:
|
|
386
|
+
return hits[0]
|
|
387
|
+
if hits:
|
|
388
|
+
names = ", ".join(sorted(s.name for s in hits))
|
|
389
|
+
raise CannotCheck(
|
|
390
|
+
f"{table} names {len(hits)} tables: {names}. Pass --source with the schema."
|
|
391
|
+
)
|
|
392
|
+
raise CannotCheck(f"no table called {table} in this project. {table_list(graph)}")
|
|
393
|
+
|
|
394
|
+
|
|
395
|
+
def table_list(graph) -> str:
|
|
396
|
+
tables = declared_sources(graph)
|
|
397
|
+
if not tables:
|
|
398
|
+
return "No source table has declared columns yet; add them to its yml or run ripple ingest-schema."
|
|
399
|
+
listed = ", ".join(f"{s.name} ({_plural(len(s.declared_columns), 'column')})" for s in tables)
|
|
400
|
+
return f"Tables with declared columns: {listed}."
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def _readers(graph, table: str, column: str) -> dict:
|
|
404
|
+
"""Who reads one column of the table: models (a filter counts), dashboards,
|
|
405
|
+
the hardest hit; through select * when the graph cannot tell which
|
|
406
|
+
columns those readers use."""
|
|
407
|
+
try:
|
|
408
|
+
answer = breaks_answer(graph.breaks(table, column, max_depth=MAX_DEPTH), "model")
|
|
409
|
+
except UnknownTarget:
|
|
410
|
+
star = {e.dst_model for e in graph.edges if e.src_model == table and e.src_column == "*"}
|
|
411
|
+
return {
|
|
412
|
+
"models": star,
|
|
413
|
+
"dashboards": set(),
|
|
414
|
+
"filters": set(),
|
|
415
|
+
"hit_hardest": [],
|
|
416
|
+
"via_star": bool(star),
|
|
417
|
+
}
|
|
418
|
+
nodes = [n for n in answer["nodes"] if n["depth"] > 0]
|
|
419
|
+
filters = {r["model"] for r in answer["row_level"]}
|
|
420
|
+
return {
|
|
421
|
+
"models": {n["model"] for n in nodes if n["kind"] == "model"} | filters,
|
|
422
|
+
"dashboards": {n["model"] for n in nodes if n["kind"] == "dashboard"},
|
|
423
|
+
"filters": filters,
|
|
424
|
+
"hit_hardest": hit_hardest(answer, 1),
|
|
425
|
+
"via_star": False,
|
|
426
|
+
}
|
|
427
|
+
|
|
428
|
+
|
|
429
|
+
def check(
|
|
430
|
+
graph, table: str, header: list[str], file_name: str, how: str, noun: str = "model"
|
|
431
|
+
) -> dict:
|
|
432
|
+
"""The report: every column's status, who reads the missing ones, the verdict."""
|
|
433
|
+
source = find_source(graph, table)
|
|
434
|
+
declared = list(source.declared_columns)
|
|
435
|
+
if not declared:
|
|
436
|
+
raise CannotCheck(
|
|
437
|
+
f"{source.name} has no declared columns, so there is nothing to check the file against. "
|
|
438
|
+
"Add them to its yml, or run ripple ingest-schema with your warehouse's column list."
|
|
439
|
+
)
|
|
440
|
+
present = {h.lower() for h in header}
|
|
441
|
+
known = {c.lower() for c in declared}
|
|
442
|
+
missing = [c for c in declared if c.lower() not in present]
|
|
443
|
+
new = [h for h in header if h.lower() not in known]
|
|
444
|
+
columns: list[dict] = []
|
|
445
|
+
who: dict[str, dict] = {}
|
|
446
|
+
for c in declared:
|
|
447
|
+
if c.lower() in present:
|
|
448
|
+
columns.append({"name": c, "status": "unchanged"})
|
|
449
|
+
continue
|
|
450
|
+
r = who[c] = _readers(graph, source.name, c)
|
|
451
|
+
columns.append(
|
|
452
|
+
{
|
|
453
|
+
"name": c,
|
|
454
|
+
"status": "missing",
|
|
455
|
+
"readers": {"models": len(r["models"]), "dashboards": len(r["dashboards"])},
|
|
456
|
+
"filters": len(r["filters"]),
|
|
457
|
+
"via_star": r["via_star"],
|
|
458
|
+
"hit_hardest": [{"model": m, "columns": n} for m, n in r["hit_hardest"]],
|
|
459
|
+
}
|
|
460
|
+
)
|
|
461
|
+
read_missing = [c["name"] for c in columns if c["status"] == "missing" and _count(c)]
|
|
462
|
+
for h in new:
|
|
463
|
+
like = difflib.get_close_matches(
|
|
464
|
+
h.lower(), [m.lower() for m in read_missing], n=1, cutoff=LOOKS_LIKE
|
|
465
|
+
)
|
|
466
|
+
entry = {"name": h, "status": "new"}
|
|
467
|
+
if like:
|
|
468
|
+
entry["looks_like"] = next(m for m in read_missing if m.lower() == like[0])
|
|
469
|
+
columns.append(entry)
|
|
470
|
+
models: set[str] = set()
|
|
471
|
+
dashboards: set[str] = set()
|
|
472
|
+
for name in read_missing:
|
|
473
|
+
models |= who[name]["models"]
|
|
474
|
+
dashboards |= who[name]["dashboards"]
|
|
475
|
+
breaking = len(read_missing)
|
|
476
|
+
if breaking:
|
|
477
|
+
readers = len(models) + len(dashboards)
|
|
478
|
+
verdict = (
|
|
479
|
+
f"Stop: {_plural(breaking, 'column')} that {_where(len(models), len(dashboards), noun)} "
|
|
480
|
+
f"read{'s' if readers == 1 else ''} {'is' if breaking == 1 else 'are'} missing from this file."
|
|
481
|
+
)
|
|
482
|
+
else:
|
|
483
|
+
verdict = "Nothing downstream changes."
|
|
484
|
+
report = {
|
|
485
|
+
"file": file_name,
|
|
486
|
+
"table": source.name,
|
|
487
|
+
"method": how,
|
|
488
|
+
"checked": "names",
|
|
489
|
+
"columns": columns,
|
|
490
|
+
"counts": {
|
|
491
|
+
"file": len(header),
|
|
492
|
+
"unchanged": len(declared) - len(missing),
|
|
493
|
+
"missing": len(missing),
|
|
494
|
+
"new": len(new),
|
|
495
|
+
},
|
|
496
|
+
"verdict": verdict,
|
|
497
|
+
"exit": 1 if breaking else 0,
|
|
498
|
+
"noun": noun,
|
|
499
|
+
}
|
|
500
|
+
report["lines"] = lines(report)
|
|
501
|
+
return report
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def failure_report(file_name: str, lines: list[str]) -> dict:
|
|
505
|
+
"""The same shape as a finished check, for a check that could not run."""
|
|
506
|
+
return {
|
|
507
|
+
"file": file_name,
|
|
508
|
+
"table": None,
|
|
509
|
+
"method": None,
|
|
510
|
+
"checked": "names",
|
|
511
|
+
"columns": [],
|
|
512
|
+
"counts": None,
|
|
513
|
+
"verdict": lines[0],
|
|
514
|
+
"exit": 2,
|
|
515
|
+
"lines": lines,
|
|
516
|
+
}
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _count(c: dict) -> int:
|
|
520
|
+
return c["readers"]["models"] + c["readers"]["dashboards"]
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _plural(n: int, word: str) -> str:
|
|
524
|
+
return f"{n} {word}" + ("" if n == 1 else "s")
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def _where(models: int, dashboards: int, noun: str) -> str:
|
|
528
|
+
parts = []
|
|
529
|
+
if models:
|
|
530
|
+
parts.append(_plural(models, noun))
|
|
531
|
+
if dashboards:
|
|
532
|
+
parts.append(_plural(dashboards, "dashboard"))
|
|
533
|
+
return " and ".join(parts) if parts else "nobody"
|
|
534
|
+
|
|
535
|
+
|
|
536
|
+
def _missing_line(c: dict, noun: str, width: int) -> str:
|
|
537
|
+
who = _where(c["readers"]["models"], c["readers"]["dashboards"], noun)
|
|
538
|
+
line = f"MISSING {c['name']:<{width}}read by {who}"
|
|
539
|
+
if c.get("via_star"):
|
|
540
|
+
return line + " through select * (which columns, Ripple cannot tell)"
|
|
541
|
+
if c.get("filters"):
|
|
542
|
+
line += f"; filters rows in {_plural(c['filters'], noun)}"
|
|
543
|
+
# worth a name only when one reader is hit in more than one column
|
|
544
|
+
if c["hit_hardest"] and c["hit_hardest"][0]["columns"] > 1:
|
|
545
|
+
top = c["hit_hardest"][0]
|
|
546
|
+
line += f"; hit hardest: {top['model']} ({_plural(top['columns'], 'column')})"
|
|
547
|
+
return line
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def lines(report: dict) -> list[str]:
|
|
551
|
+
"""The terminal text: verdict first and last, damage in between."""
|
|
552
|
+
noun = report.get("noun", "model")
|
|
553
|
+
counts = report["counts"]
|
|
554
|
+
listed = [c for c in report["columns"] if c["status"] == "missing" and _count(c)]
|
|
555
|
+
hinted = [c for c in report["columns"] if c["status"] == "new" and c.get("looks_like")]
|
|
556
|
+
quiet_new = [
|
|
557
|
+
c["name"] for c in report["columns"] if c["status"] == "new" and not c.get("looks_like")
|
|
558
|
+
]
|
|
559
|
+
quiet_missing = [
|
|
560
|
+
c["name"] for c in report["columns"] if c["status"] == "missing" and not _count(c)
|
|
561
|
+
]
|
|
562
|
+
width = max((len(c["name"]) for c in listed + hinted), default=0) + 3
|
|
563
|
+
out = [report["verdict"], ""]
|
|
564
|
+
out.append(f"File {report['file']}")
|
|
565
|
+
out.append(f"Table {report['table']} ({report['method']})")
|
|
566
|
+
out.append(HEADER_ONLY)
|
|
567
|
+
out.append("")
|
|
568
|
+
parts = [f"{counts['unchanged']} unchanged"]
|
|
569
|
+
if counts["missing"]:
|
|
570
|
+
parts.append(f"{counts['missing']} missing")
|
|
571
|
+
if counts["new"]:
|
|
572
|
+
parts.append(f"{counts['new']} new")
|
|
573
|
+
out.append(f"{_plural(counts['file'], 'column')}: {', '.join(parts)}")
|
|
574
|
+
out.append("")
|
|
575
|
+
for c in sorted(listed, key=lambda c: (-_count(c), c["name"])):
|
|
576
|
+
out.append(_missing_line(c, noun, width))
|
|
577
|
+
for c in hinted:
|
|
578
|
+
out.append(
|
|
579
|
+
f"NEW {c['name']:<{width}}looks like {c['looks_like']}; "
|
|
580
|
+
"a similar name does not prove the same contents"
|
|
581
|
+
)
|
|
582
|
+
if quiet_new:
|
|
583
|
+
more = "more " if hinted else ""
|
|
584
|
+
out.append(
|
|
585
|
+
f"{len(quiet_new)} {more}new column{'s' if len(quiet_new) != 1 else ''} nobody reads yet: {', '.join(quiet_new)}"
|
|
586
|
+
)
|
|
587
|
+
if quiet_missing:
|
|
588
|
+
out.append(
|
|
589
|
+
f"{len(quiet_missing)} missing column{'s' if len(quiet_missing) != 1 else ''} nobody reads yet: {', '.join(quiet_missing)}"
|
|
590
|
+
)
|
|
591
|
+
if listed:
|
|
592
|
+
out.append("")
|
|
593
|
+
fixes = []
|
|
594
|
+
renamed = {c["looks_like"]: c["name"] for c in hinted}
|
|
595
|
+
for c in listed:
|
|
596
|
+
if c["name"] in renamed:
|
|
597
|
+
fixes.append(
|
|
598
|
+
f"if {renamed[c['name']]} holds the same data as {c['name']}, rename that header to {c['name']}; "
|
|
599
|
+
f"otherwise restore {c['name']}."
|
|
600
|
+
)
|
|
601
|
+
else:
|
|
602
|
+
fixes.append(f"Restore {c['name']}.")
|
|
603
|
+
out.append("To fix: " + " ".join(fixes) + " Then run this check again.")
|
|
604
|
+
out.append(f"Details: ripple breaks {report['table']}.{listed[0]['name']}")
|
|
605
|
+
out.append("")
|
|
606
|
+
out.append(f"{report['verdict']} (exit {report['exit']})")
|
|
607
|
+
return out
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
def no_binding_lines(file_name: str, header: list[str], graph, tied: list) -> list[str]:
|
|
611
|
+
"""What to do when no table claims the file: the choices, never a guess."""
|
|
612
|
+
if tied:
|
|
613
|
+
fits_line = ", ".join(
|
|
614
|
+
f"{s.name} ({n} of {_plural(len(header), 'column name')})" for s, n in tied
|
|
615
|
+
)
|
|
616
|
+
out = [f"Cannot check: {file_name} fits more than one table: {fits_line}."]
|
|
617
|
+
else:
|
|
618
|
+
out = [
|
|
619
|
+
f"Cannot check: no table is mapped to {file_name} and its columns fit none.",
|
|
620
|
+
table_list(graph),
|
|
621
|
+
]
|
|
622
|
+
out.append(
|
|
623
|
+
"Run again with --source TABLE, or add to that source's yml so the next drop binds itself:"
|
|
624
|
+
)
|
|
625
|
+
out.append(f' meta: {{ripple: {{files: ["{file_name}"]}}}}')
|
|
626
|
+
out.append("A pattern like vendor_feed_*.csv covers every drop of the same feed.")
|
|
627
|
+
return out
|
|
@@ -258,6 +258,43 @@ def cmd_breaks(args) -> None:
|
|
|
258
258
|
_page_door(args, project, graph, answer, plain_text(breaks_lines(answer, full=True)))
|
|
259
259
|
|
|
260
260
|
|
|
261
|
+
def cmd_check_file(args) -> None:
|
|
262
|
+
from ripple import check_file
|
|
263
|
+
from ripple.loaders.types import NotAProject
|
|
264
|
+
from ripple.project import find_project_root
|
|
265
|
+
|
|
266
|
+
path = Path(args.file)
|
|
267
|
+
|
|
268
|
+
def stop(lines: list[str]) -> None:
|
|
269
|
+
report = check_file.failure_report(path.name, lines)
|
|
270
|
+
print(json.dumps(report, indent=2) if args.json else "\n".join(lines))
|
|
271
|
+
sys.exit(2)
|
|
272
|
+
|
|
273
|
+
try:
|
|
274
|
+
project, graph = _load_graph(args)
|
|
275
|
+
root = find_project_root(Path(args.path))
|
|
276
|
+
header = check_file.read_header(path, args.encoding)
|
|
277
|
+
if args.source:
|
|
278
|
+
table, how = args.source, "--source"
|
|
279
|
+
else:
|
|
280
|
+
roots = [root, *getattr(project, "dbt_roots", [])]
|
|
281
|
+
table, how = check_file.bind(path, root, check_file.file_map(roots))
|
|
282
|
+
if table is None:
|
|
283
|
+
source, how, tied = check_file.bind_by_header(header, graph)
|
|
284
|
+
if source is None:
|
|
285
|
+
stop(check_file.no_binding_lines(path.name, header, graph, tied))
|
|
286
|
+
table = source.name
|
|
287
|
+
report = check_file.check(graph, table, header, path.name, how, noun_for(project.mode))
|
|
288
|
+
except (check_file.CannotCheck, NotAProject, OSError) as e:
|
|
289
|
+
stop([f"Cannot check: {e}"])
|
|
290
|
+
if args.json:
|
|
291
|
+
print(json.dumps(report, indent=2))
|
|
292
|
+
else:
|
|
293
|
+
print("\n".join(report["lines"]))
|
|
294
|
+
if report["exit"]:
|
|
295
|
+
sys.exit(report["exit"])
|
|
296
|
+
|
|
297
|
+
|
|
261
298
|
def cmd_trace(args) -> None:
|
|
262
299
|
project, graph = _load_graph(args)
|
|
263
300
|
model, column = _parse_target(args.target)
|
|
@@ -604,6 +641,19 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
604
641
|
)
|
|
605
642
|
p_columns.add_argument("model")
|
|
606
643
|
|
|
644
|
+
p_check = sub.add_parser(
|
|
645
|
+
"check-file",
|
|
646
|
+
help="a file from an external team: what breaks if it is loaded",
|
|
647
|
+
parents=[common],
|
|
648
|
+
)
|
|
649
|
+
p_check.add_argument("file", help="the csv, tsv, xlsx or parquet file that arrived")
|
|
650
|
+
p_check.add_argument(
|
|
651
|
+
"--source",
|
|
652
|
+
help="the table it feeds (raw.vendor_feed, or just vendor_feed); else the project's files map",
|
|
653
|
+
)
|
|
654
|
+
p_check.add_argument(
|
|
655
|
+
"--encoding", help="the file's text encoding when it is not UTF-8 (cp1252, latin-1)"
|
|
656
|
+
)
|
|
607
657
|
p_breaks = sub.add_parser("breaks", help="what breaks if this column changes", parents=[common])
|
|
608
658
|
p_breaks.add_argument("target", help="model.column")
|
|
609
659
|
p_breaks.add_argument(
|
|
@@ -716,6 +766,7 @@ def main(argv: list[str] | None = None) -> None:
|
|
|
716
766
|
"models": cmd_models,
|
|
717
767
|
"columns": cmd_columns,
|
|
718
768
|
"breaks": cmd_breaks,
|
|
769
|
+
"check-file": cmd_check_file,
|
|
719
770
|
"trace": cmd_trace,
|
|
720
771
|
"graph": cmd_graph,
|
|
721
772
|
"ci": cmd_ci,
|
|
@@ -5,7 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
import re
|
|
6
6
|
from pathlib import Path
|
|
7
7
|
|
|
8
|
-
from ripple.loaders.types import files_under
|
|
8
|
+
from ripple.loaders.types import IGNORE_DIRS, files_under
|
|
9
9
|
|
|
10
10
|
|
|
11
11
|
def _project_vars(root: Path) -> dict:
|
|
@@ -339,3 +339,56 @@ def _dbt_owned_dirs(project_dir: Path) -> list[Path]:
|
|
|
339
339
|
owned += _dbt_model_dirs(project_dir)
|
|
340
340
|
owned += [project_dir / name for name in _macro_dir_names(project_dir / "dbt_project.yml")]
|
|
341
341
|
return [d for d in owned if d.is_dir()]
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def source_tables(root: Path):
|
|
345
|
+
"""Every (source name, table dict) a dbt project declares in its yml.
|
|
346
|
+
|
|
347
|
+
Walks the model paths only (dbt_project.yml's model-paths), so a
|
|
348
|
+
package's or a build folder's yml never counts as this project's."""
|
|
349
|
+
import yaml
|
|
350
|
+
|
|
351
|
+
root = Path(root)
|
|
352
|
+
if not (root / "dbt_project.yml").is_file():
|
|
353
|
+
return
|
|
354
|
+
for model_dir in _dbt_model_dirs(root):
|
|
355
|
+
for yml in [*files_under(model_dir, "*.yml"), *files_under(model_dir, "*.yaml")]:
|
|
356
|
+
# a package's or a build folder's yml is not this project's, even
|
|
357
|
+
# when model-paths is the project root
|
|
358
|
+
if IGNORE_DIRS & set(yml.relative_to(model_dir).parts[:-1]):
|
|
359
|
+
continue
|
|
360
|
+
try:
|
|
361
|
+
parsed = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
|
|
362
|
+
except yaml.YAMLError:
|
|
363
|
+
continue
|
|
364
|
+
if not isinstance(parsed, dict):
|
|
365
|
+
continue
|
|
366
|
+
for source in parsed.get("sources") or []:
|
|
367
|
+
if not isinstance(source, dict) or not source.get("name"):
|
|
368
|
+
continue
|
|
369
|
+
for table in source.get("tables") or []:
|
|
370
|
+
if isinstance(table, dict) and table.get("name"):
|
|
371
|
+
yield str(source["name"]), table
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
def source_columns(root: Path) -> dict[str, list[str]]:
|
|
375
|
+
"""The columns sources.yml declares, as {"schema.table": [names]}.
|
|
376
|
+
|
|
377
|
+
dbt users list a source's columns in yml for docs and tests; those are
|
|
378
|
+
the landing table's contract, and the graph read only ingested schemas
|
|
379
|
+
and seeds before (a dbt source always showed "not declared")."""
|
|
380
|
+
out: dict[str, list[str]] = {}
|
|
381
|
+
spelling: dict[str, str] = {}
|
|
382
|
+
for source, table in source_tables(root):
|
|
383
|
+
key = f"{source}.{table['name']}".lower()
|
|
384
|
+
name = spelling.setdefault(key, f"{source}.{table['name']}")
|
|
385
|
+
# two files may describe one table (docs in one, tests in another);
|
|
386
|
+
# the union is the contract, the first spelling stays
|
|
387
|
+
columns = out.setdefault(name, [])
|
|
388
|
+
known = {c.lower() for c in columns}
|
|
389
|
+
for column in table.get("columns") or []:
|
|
390
|
+
column_name = column.get("name") if isinstance(column, dict) else column
|
|
391
|
+
if column_name and str(column_name).lower() not in known:
|
|
392
|
+
columns.append(str(column_name))
|
|
393
|
+
known.add(str(column_name).lower())
|
|
394
|
+
return {name: columns for name, columns in out.items() if columns}
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import re
|
|
6
|
+
from pathlib import Path
|
|
6
7
|
|
|
7
8
|
from ripple.loaders.types import Model, Project
|
|
8
9
|
|
|
@@ -85,6 +86,45 @@ def _qualifiers_by_bare_name(project: Project) -> dict[str, set[str]]:
|
|
|
85
86
|
return seen
|
|
86
87
|
|
|
87
88
|
|
|
89
|
+
def _apply_declared_sources(project: Project) -> None:
|
|
90
|
+
"""Expose sources.yml column lists as the sources' declared columns.
|
|
91
|
+
|
|
92
|
+
By exact name only (schema.table, case-insensitive): a declaration names
|
|
93
|
+
its own table, so it never rides the suffix attachment an ingested schema
|
|
94
|
+
needs, which merged raw.feed and staging.feed into one source. A table
|
|
95
|
+
no SQL references still becomes a source, so a file can be checked
|
|
96
|
+
against it. Every dbt project in a monorepo declares its own."""
|
|
97
|
+
from ripple.loaders.dbt_config import source_columns
|
|
98
|
+
|
|
99
|
+
declared: dict[str, list[str]] = {}
|
|
100
|
+
spelling: dict[str, str] = {}
|
|
101
|
+
for root in project.dbt_roots:
|
|
102
|
+
for name, columns in source_columns(Path(root)).items():
|
|
103
|
+
key = name.lower()
|
|
104
|
+
known = declared.setdefault(spelling.setdefault(key, name), [])
|
|
105
|
+
seen = {c.lower() for c in known}
|
|
106
|
+
known.extend(c for c in columns if c.lower() not in seen)
|
|
107
|
+
if not declared:
|
|
108
|
+
return
|
|
109
|
+
by_name = {s.name.lower(): s for s in project.sources}
|
|
110
|
+
owned = {m.name.lower() for m in project.models if m.sql.strip()}
|
|
111
|
+
for name, columns in declared.items():
|
|
112
|
+
target = by_name.get(name.lower())
|
|
113
|
+
if target is None:
|
|
114
|
+
target = Model(
|
|
115
|
+
name=name,
|
|
116
|
+
sql="",
|
|
117
|
+
path="",
|
|
118
|
+
aliases={a for a in _name_suffixes(name) if a.lower() not in owned},
|
|
119
|
+
is_source=True,
|
|
120
|
+
evidence="declared",
|
|
121
|
+
)
|
|
122
|
+
project.sources.append(target)
|
|
123
|
+
by_name[name.lower()] = target
|
|
124
|
+
seen = {c.lower() for c in target.declared_columns}
|
|
125
|
+
target.declared_columns.extend(c for c in columns if c.lower() not in seen)
|
|
126
|
+
|
|
127
|
+
|
|
88
128
|
def _apply_ingested_schemas(project: Project, extra_roots: tuple = ()) -> None:
|
|
89
129
|
"""Expose .ripple/schemas.json tables as known external relations.
|
|
90
130
|
|
|
@@ -122,6 +122,9 @@ class Project:
|
|
|
122
122
|
downstream: list = field(default_factory=list)
|
|
123
123
|
# vars declared in dbt_project.yml, resolved by var() during tier-b renders
|
|
124
124
|
jinja_vars: dict = field(default_factory=dict)
|
|
125
|
+
# every dbt project directory that was loaded: one for a dbt project, all
|
|
126
|
+
# of them for a monorepo, none for a plain SQL folder
|
|
127
|
+
dbt_roots: list = field(default_factory=list)
|
|
125
128
|
|
|
126
129
|
@property
|
|
127
130
|
def name_candidates(self) -> dict[str, list[str]]:
|
|
@@ -22,7 +22,11 @@ from pathlib import Path
|
|
|
22
22
|
|
|
23
23
|
from ripple.loaders.dbt import _load_one_dbt, resolve_dbt_dialect
|
|
24
24
|
from ripple.loaders.dbt_config import _dbt_owned_dirs
|
|
25
|
-
from ripple.loaders.identity import
|
|
25
|
+
from ripple.loaders.identity import (
|
|
26
|
+
_apply_declared_sources,
|
|
27
|
+
_apply_ingested_schemas,
|
|
28
|
+
_qualify_collisions,
|
|
29
|
+
)
|
|
26
30
|
from ripple.loaders.sqldir import _load_sql_dir
|
|
27
31
|
from ripple.loaders.types import (
|
|
28
32
|
ADAPTER_TO_DIALECT,
|
|
@@ -83,7 +87,11 @@ def load_project(path: str | Path = ".", dialect: str | None = None) -> Project:
|
|
|
83
87
|
f"{len(stray.models)} plain-SQL models outside the dbt "
|
|
84
88
|
f"project{'s' if len(dbt_dirs) != 1 else ''} were loaded as well."
|
|
85
89
|
)
|
|
90
|
+
project.dbt_roots = (
|
|
91
|
+
list(dbt_dirs) if dbt_dirs else ([root] if (root / "dbt_project.yml").exists() else [])
|
|
92
|
+
)
|
|
86
93
|
_qualify_collisions(project)
|
|
94
|
+
_apply_declared_sources(project)
|
|
87
95
|
_apply_ingested_schemas(project, extra_roots=(scan_root, start))
|
|
88
96
|
from ripple.lookml import load_lookml_downstream
|
|
89
97
|
from ripple.semantic import load_downstream
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|