ripple-sql 0.1.7__tar.gz → 0.1.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/CHANGELOG.md +14 -1
  2. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/PKG-INFO +48 -1
  3. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/README.md +47 -0
  4. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/pyproject.toml +1 -1
  5. ripple_sql-0.1.9/src/ripple/check_file.py +627 -0
  6. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/cli.py +51 -0
  7. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/dbt_config.py +54 -1
  8. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/identity.py +40 -0
  9. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/types.py +3 -0
  10. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/project.py +9 -1
  11. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/.gitignore +0 -0
  12. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/LICENSE +0 -0
  13. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/__init__.py +0 -0
  14. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/answer.py +0 -0
  15. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/answer_page.py +0 -0
  16. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/cache.py +0 -0
  17. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/ci.py +0 -0
  18. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/ci_signature.py +0 -0
  19. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/doctor.py +0 -0
  20. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/__init__.py +0 -0
  21. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/budget.py +0 -0
  22. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/column_lineage.py +0 -0
  23. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/column_ref.py +0 -0
  24. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/cte_tracing.py +0 -0
  25. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/dependencies.py +0 -0
  26. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/dialect.py +0 -0
  27. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/dispatch.py +0 -0
  28. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/extraction.py +0 -0
  29. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/generators.py +0 -0
  30. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/jinja.py +0 -0
  31. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/json_sources.py +0 -0
  32. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/macro_source.py +0 -0
  33. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/pipeline.py +0 -0
  34. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/preprocess.py +0 -0
  35. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/safe_gen.py +0 -0
  36. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/schema_qualification.py +0 -0
  37. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/scope.py +0 -0
  38. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/select_sources.py +0 -0
  39. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/sql_script.py +0 -0
  40. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/statement.py +0 -0
  41. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tech_debt.py +0 -0
  42. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_catalog.py +0 -0
  43. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_scalar_vars.py +0 -0
  44. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_tvf.py +0 -0
  45. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/tsql_xml.py +0 -0
  46. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/types.py +0 -0
  47. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/unused_deps.py +0 -0
  48. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/engine/validation.py +0 -0
  49. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/graph.py +0 -0
  50. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/home.py +0 -0
  51. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/__init__.py +0 -0
  52. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/dbt.py +0 -0
  53. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/sidecar.py +0 -0
  54. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/loaders/sqldir.py +0 -0
  55. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/lookml.py +0 -0
  56. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/mcp_server.py +0 -0
  57. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/names.py +0 -0
  58. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/py.typed +0 -0
  59. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/render.py +0 -0
  60. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/render_shims.py +0 -0
  61. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/schemas.py +0 -0
  62. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/semantic.py +0 -0
  63. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/server.py +0 -0
  64. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/sourcefiles.py +0 -0
  65. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/star_resolution.py +0 -0
  66. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/answer.css +0 -0
  67. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/answer.html +0 -0
  68. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/answer_twin.js +0 -0
  69. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/explore.js +0 -0
  70. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/find.js +0 -0
  71. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/static/focus.js +0 -0
  72. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/__init__.py +0 -0
  73. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/cli.py +0 -0
  74. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/collect.py +0 -0
  75. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/discover.py +0 -0
  76. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/ingest.py +0 -0
  77. {ripple_sql-0.1.7 → ripple_sql-0.1.9}/src/ripple/usage/report.py +0 -0
@@ -6,6 +6,17 @@ Notable changes to Ripple. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.9] - 2026-09-15
10
+
11
+ ### Changed
12
+ - `ripple check-file` no longer needs the table's name in the common case. A file binds to the table its own columns name when they say it clearly: at least two of them are one table's, they are most of the file's columns, and no other table matches as many ("matched by 4 of 5 column names"). `--source` takes the bare table name when one schema has it (`--source vendor_feed`); a name two schemas share is refused with both spellings. When two tables fit a file equally, or none does, the command lists the tables with declared columns and stops with exit 2, never picking one. `--source` and a declared `files:` pattern still come first. A review of the command's edge cases also fixed: a column read only in a where clause or a join now counts as read; a reader the graph only sees through select star counts instead of vanishing; long chains are counted whole; an xlsx is read by the workbook's own sheet order, streamed no further than its first row, with rich text joined and empty cells kept as positions; tab-separated `.txt` and semicolon `.csv` files are read by their delimiter; a file that is not UTF-8 asks for `--encoding`; an unclosed quote, a duplicated header name, a DOCTYPE in the XML, a broken xlsx part, a pyarrow older than 14.0.1 and a broken `ripple.yml` are each a plain message with exit 2; a folder pattern (`incoming/*.csv`) matches the file's path; two patterns that name two tables are refused; `--source` never reads the pattern map; a missing file, a folder, and a project that cannot load give the same JSON shape as any other failure.
13
+
14
+ ## [0.1.8] - 2026-09-15
15
+
16
+ ### Added
17
+ - `ripple check-file FILE [--source TABLE]`: a file arrives from an external team (csv, tsv, xlsx, parquet), and before anyone loads it Ripple diffs its header against the landing table's declared columns and prints who reads every column that is gone: models, dashboards, hit hardest. The verdict is the first line and the last; a similar name is a hint ("looks like customer_name; a similar name does not prove the same contents"), never an assumption; exit 0 when nothing downstream changes (new columns nobody reads included), 1 when something breaks, 2 when the check could not run. The file binds to its table by `--source`, or by a `files:` pattern the project declares (`meta: {ripple: {files: ["vendor_feed_*.csv"]}}` on a dbt source, or `files:` in `ripple.yml` for a plain SQL folder); with neither, the command prints the closest table and the exact line to add and stops. Header only: no data row is read and nothing leaves the machine. Shaped by a two-model debate; the spec's sample output is the command's own.
18
+ - A dbt source's `columns:` in sources.yml are its declared columns. The graph read only ingested schemas and seeds before, so every dbt source showed as "not declared" and a select star through it stayed unresolved even when the yml listed every column.
19
+
9
20
  ## [0.1.7] - 2026-09-10
10
21
 
11
22
  ### Fixed
@@ -264,7 +275,9 @@ First public release.
264
275
  ### Removed
265
276
  - The dark whole-graph canvas that `ripple serve` used to open (`static/index.html`). Every surface now draws one answer.
266
277
 
267
- [Unreleased]: https://github.com/bteh/ripple/compare/v0.1.7...HEAD
278
+ [Unreleased]: https://github.com/bteh/ripple/compare/v0.1.9...HEAD
279
+ [0.1.9]: https://github.com/bteh/ripple/releases/tag/v0.1.9
280
+ [0.1.8]: https://github.com/bteh/ripple/releases/tag/v0.1.8
268
281
  [0.1.7]: https://github.com/bteh/ripple/releases/tag/v0.1.7
269
282
  [0.1.6]: https://github.com/bteh/ripple/releases/tag/v0.1.6
270
283
  [0.1.5]: https://github.com/bteh/ripple/releases/tag/v0.1.5
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: ripple-sql
3
- Version: 0.1.7
3
+ Version: 0.1.9
4
4
  Summary: Offline column-level SQL lineage. See what breaks before you merge.
5
5
  Project-URL: Homepage, https://github.com/bteh/ripple
6
6
  Project-URL: Repository, https://github.com/bteh/ripple
@@ -210,6 +210,53 @@ ripple unresolved # which tables block the most coverage, worst
210
210
  ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
211
211
  ```
212
212
 
213
+ ## A file arrives from another team
214
+
215
+ Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
216
+
217
+ ```bash
218
+ ripple check-file vendor_feed_2026-09.csv
219
+ ```
220
+
221
+ ```
222
+ Stop: 2 columns that 4 models read are missing from this file.
223
+
224
+ File vendor_feed_2026-09.csv
225
+ Table raw.vendor_feed (files: vendor_feed_*.csv)
226
+ Checked: column names only. Not checked: values, types, whether the file loads.
227
+
228
+ 6 columns: 2 unchanged, 2 missing, 4 new
229
+
230
+ MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
231
+ MISSING amount_usd read by 2 models
232
+ NEW cust_nm looks like customer_name; a similar name does not prove the same contents
233
+ 3 more new columns nobody reads yet: region, notes, batch_id
234
+
235
+ To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
236
+ Details: ripple breaks raw.vendor_feed.customer_name
237
+
238
+ Stop: 2 columns that 4 models read are missing from this file. (exit 1)
239
+ ```
240
+
241
+ The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
242
+
243
+ The file finds its table by its own columns when they say it clearly (most of them belong to one table and no other table comes close: "matched by 4 of 5 column names"), by `--source vendor_feed` (the bare table name is enough when one schema has it), or by a pattern the project declares once:
244
+
245
+ ```yaml
246
+ sources:
247
+ - name: raw
248
+ tables:
249
+ - name: vendor_feed
250
+ meta:
251
+ ripple:
252
+ files: ["vendor_feed_*.csv"]
253
+ columns:
254
+ - name: customer_id
255
+ - name: customer_name
256
+ ```
257
+
258
+ A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). When two tables fit the file equally, or none does, the command lists the tables it can check and stops with exit 2 rather than pick one. The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow 14.0.1 or newer). A file that is not UTF-8 takes `--encoding cp1252`. `--json` carries the same facts for an ingest job, in the same shape whether the check ran or could not.
259
+
213
260
  ## Bring your query history (optional)
214
261
 
215
262
  Your SQL files say what is supposed to happen. The warehouse's query log says what
@@ -170,6 +170,53 @@ ripple unresolved # which tables block the most coverage, worst
170
170
  ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
171
171
  ```
172
172
 
173
+ ## A file arrives from another team
174
+
175
+ Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
176
+
177
+ ```bash
178
+ ripple check-file vendor_feed_2026-09.csv
179
+ ```
180
+
181
+ ```
182
+ Stop: 2 columns that 4 models read are missing from this file.
183
+
184
+ File vendor_feed_2026-09.csv
185
+ Table raw.vendor_feed (files: vendor_feed_*.csv)
186
+ Checked: column names only. Not checked: values, types, whether the file loads.
187
+
188
+ 6 columns: 2 unchanged, 2 missing, 4 new
189
+
190
+ MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
191
+ MISSING amount_usd read by 2 models
192
+ NEW cust_nm looks like customer_name; a similar name does not prove the same contents
193
+ 3 more new columns nobody reads yet: region, notes, batch_id
194
+
195
+ To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
196
+ Details: ripple breaks raw.vendor_feed.customer_name
197
+
198
+ Stop: 2 columns that 4 models read are missing from this file. (exit 1)
199
+ ```
200
+
201
+ The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
202
+
203
+ The file finds its table by its own columns when they say it clearly (most of them belong to one table and no other table comes close: "matched by 4 of 5 column names"), by `--source vendor_feed` (the bare table name is enough when one schema has it), or by a pattern the project declares once:
204
+
205
+ ```yaml
206
+ sources:
207
+ - name: raw
208
+ tables:
209
+ - name: vendor_feed
210
+ meta:
211
+ ripple:
212
+ files: ["vendor_feed_*.csv"]
213
+ columns:
214
+ - name: customer_id
215
+ - name: customer_name
216
+ ```
217
+
218
+ A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). When two tables fit the file equally, or none does, the command lists the tables it can check and stops with exit 2 rather than pick one. The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow 14.0.1 or newer). A file that is not UTF-8 takes `--encoding cp1252`. `--json` carries the same facts for an ingest job, in the same shape whether the check ran or could not.
219
+
173
220
  ## Bring your query history (optional)
174
221
 
175
222
  Your SQL files say what is supposed to happen. The warehouse's query log says what
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "ripple-sql"
3
- version = "0.1.7"
3
+ version = "0.1.9"
4
4
  description = "Offline column-level SQL lineage. See what breaks before you merge."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -0,0 +1,627 @@
1
+ """A file arrives from an external team: what breaks if it is loaded.
2
+
3
+ The file feeds one landing table the graph already knows. Its header is
4
+ diffed against the table's declared columns, and every column that is gone
5
+ gets the blast radius `ripple breaks` would give it. Header only: no data
6
+ row is read, nothing leaves the machine, and a similar name is a hint,
7
+ never an assumption.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import csv
13
+ import difflib
14
+ import fnmatch
15
+ import io
16
+ import zipfile
17
+ from pathlib import Path
18
+ from xml.etree import ElementTree
19
+
20
+ from ripple.answer import breaks_answer, hit_hardest
21
+ from ripple.graph import UnknownTarget
22
+
23
+ HEADER_ONLY = "Checked: column names only. Not checked: values, types, whether the file loads."
24
+ KINDS = ("csv", "tsv", "txt", "xlsx", "parquet")
25
+ LOOKS_LIKE = 0.6
26
+ DELIMITERS = ",;\t|"
27
+ HEAD_BYTES = 64 * 1024
28
+ # a reader older than this deserializes untrusted Parquet unsafely (CVE-2023-47248)
29
+ PYARROW_FLOOR = (14, 0, 1)
30
+ # far past any real chain; the walk stops at cycles on its own
31
+ MAX_DEPTH = 500
32
+ # the parts a workbook's header lives in; anything past these caps is not a
33
+ # workbook Ripple reads a header from, whatever the zip claims
34
+ XML_PART_CAP = 4 * 1024 * 1024
35
+ SHARED_STRINGS_CAP = 64 * 1024 * 1024
36
+
37
+
38
+ class CannotCheck(Exception):
39
+ """The check could not run; the message says why and what to do."""
40
+
41
+
42
+ def read_header(path: Path, encoding: str | None = None) -> list[str]:
43
+ """The column names of a file, from its header only."""
44
+ if path.is_dir():
45
+ raise CannotCheck(f"{path.name} is a folder; pass one file")
46
+ kind = path.suffix.lower().lstrip(".")
47
+ if kind in ("csv", "tsv", "txt"):
48
+ header = _delimited(path, kind, encoding)
49
+ elif kind == "xlsx":
50
+ header = _xlsx(path)
51
+ elif kind == "parquet":
52
+ header = _parquet(path)
53
+ else:
54
+ raise CannotCheck(
55
+ f"{path.name}: Ripple reads a csv, tsv, xlsx or parquet header, not .{kind}"
56
+ )
57
+ names = [h for h in header if h]
58
+ seen: set[str] = set()
59
+ for name in names:
60
+ if name.lower() in seen:
61
+ raise CannotCheck(
62
+ f"{path.name} names {name} twice in its header; a table has one column of a name"
63
+ )
64
+ seen.add(name.lower())
65
+ if not names:
66
+ raise CannotCheck(f"{path.name} is empty; no header row to check")
67
+ return names
68
+
69
+
70
+ def _decode(path: Path, raw: bytes, encoding: str | None) -> str:
71
+ try:
72
+ return raw.decode(encoding or "utf-8-sig")
73
+ except (UnicodeDecodeError, LookupError):
74
+ if encoding:
75
+ raise CannotCheck(
76
+ f"{path.name} is not {encoding}; pass the file's encoding with --encoding"
77
+ ) from None
78
+ raise CannotCheck(
79
+ f"{path.name} is not UTF-8; pass --encoding cp1252 (Excel on Windows) or latin-1"
80
+ ) from None
81
+
82
+
83
+ def _delimited(path: Path, kind: str, encoding: str | None) -> list[str]:
84
+ with open(path, "rb") as f:
85
+ text = _decode(path, f.read(HEAD_BYTES), encoding)
86
+ first = text.split("\n", 1)[0]
87
+ delimiter = "\t" if kind == "tsv" else max(DELIMITERS, key=lambda d: (first.count(d), d == ","))
88
+ if text.count('"') % 2:
89
+ raise CannotCheck(f"{path.name} has an unclosed quote in its header")
90
+ try:
91
+ header = next(csv.reader(io.StringIO(text, newline=""), delimiter=delimiter, strict=True))
92
+ except StopIteration:
93
+ raise CannotCheck(f"{path.name} is empty; no header row to check") from None
94
+ except csv.Error as e:
95
+ raise CannotCheck(f"{path.name}: the header is not valid csv ({e})") from None
96
+ # a quoted header that spans lines keeps the file's own line ending
97
+ return [h.replace("\r\n", "\n").strip() for h in header]
98
+
99
+
100
+ def _local(tag: str) -> str:
101
+ return tag.rsplit("}", 1)[-1]
102
+
103
+
104
+ def _guard_xml(path: Path, data: bytes) -> None:
105
+ """Refuse a DOCTYPE or an entity in an XML part, in whatever encoding the
106
+ part is written: expat honours a UTF-16 or UTF-32 prolog, so the scan
107
+ must read the bytes the way the parser will."""
108
+ if data.startswith((b"\xff\xfe\x00\x00", b"\x00\x00\xfe\xff")):
109
+ codec = "utf-32"
110
+ elif data.startswith((b"\xff\xfe", b"\xfe\xff")):
111
+ codec = "utf-16"
112
+ elif b"\x00" in data[:64]:
113
+ raise CannotCheck(
114
+ f"{path.name} carries XML in an encoding Ripple does not read; save it as a normal xlsx"
115
+ )
116
+ else:
117
+ codec = "utf-8"
118
+ text = data.decode(codec, errors="ignore")
119
+ if "<!DOCTYPE" in text or "<!ENTITY" in text:
120
+ raise CannotCheck(f"{path.name} carries a DOCTYPE in its XML; Ripple does not read that")
121
+
122
+
123
+ def _part(z: zipfile.ZipFile, path: Path, name: str, cap: int) -> bytes:
124
+ """One zip member, whole, guarded: bounded in size and free of a DOCTYPE."""
125
+ if z.getinfo(name).file_size > cap:
126
+ raise CannotCheck(
127
+ f"{path.name}: {name} is larger than {cap >> 20} MB; not a workbook Ripple reads"
128
+ )
129
+ member = z.open # bytes; the utf-8 floor scan keys on the name open(
130
+ with member(name) as f:
131
+ data = f.read(cap + 1)
132
+ if len(data) > cap:
133
+ raise CannotCheck(
134
+ f"{path.name}: {name} is larger than {cap >> 20} MB; not a workbook Ripple reads"
135
+ )
136
+ _guard_xml(path, data)
137
+ return data
138
+
139
+
140
+ def _first_sheet(z: zipfile.ZipFile, path: Path) -> str:
141
+ """The workbook's first sheet, by its own order, not by file name."""
142
+ names = set(z.namelist())
143
+ if "xl/workbook.xml" in names and "xl/_rels/workbook.xml.rels" in names:
144
+ try:
145
+ wb = ElementTree.fromstring(_part(z, path, "xl/workbook.xml", XML_PART_CAP))
146
+ rels = ElementTree.fromstring(
147
+ _part(z, path, "xl/_rels/workbook.xml.rels", XML_PART_CAP)
148
+ )
149
+ except ElementTree.ParseError:
150
+ raise CannotCheck(f"{path.name} is not valid xlsx") from None
151
+ first = next((s for s in wb.iter() if _local(s.tag) == "sheet"), None)
152
+ rid = (
153
+ next((v for k, v in first.attrib.items() if _local(k) == "id"), None)
154
+ if first is not None
155
+ else None
156
+ )
157
+ target = next((r.get("Target") for r in rels.iter() if r.get("Id") == rid), None)
158
+ if target:
159
+ member = target.lstrip("/") if target.startswith("/") else "xl/" + target
160
+ if member in names:
161
+ return member
162
+ sheet = next((n for n in sorted(names) if n.startswith("xl/worksheets/sheet")), None)
163
+ if not sheet:
164
+ raise CannotCheck(f"{path.name} has no worksheet")
165
+ return sheet
166
+
167
+
168
+ def _column_index(ref: str) -> int:
169
+ n = 0
170
+ for ch in ref:
171
+ if not ch.isalpha():
172
+ break
173
+ n = n * 26 + (ord(ch.upper()) - 64)
174
+ return max(n - 1, 0)
175
+
176
+
177
+ def _cell_text(cell, shared: list[str]) -> str:
178
+ kind = cell.get("t")
179
+ if kind == "inlineStr":
180
+ return "".join(t.text or "" for t in cell.iter() if _local(t.tag) == "t")
181
+ value = next((v.text or "" for v in cell.iter() if _local(v.tag) == "v"), "")
182
+ if kind == "s" and value.isdigit() and int(value) < len(shared):
183
+ return shared[int(value)]
184
+ return value
185
+
186
+
187
+ def _xlsx(path: Path) -> list[str]:
188
+ """Row 1 of the first sheet, streamed and left unread past that row."""
189
+ try:
190
+ with zipfile.ZipFile(path) as z:
191
+ shared: list[str] = []
192
+ if "xl/sharedStrings.xml" in z.namelist():
193
+ raw = _part(z, path, "xl/sharedStrings.xml", SHARED_STRINGS_CAP)
194
+ for si in ElementTree.fromstring(raw):
195
+ shared.append("".join(t.text or "" for t in si.iter() if _local(t.tag) == "t"))
196
+ sheet = _first_sheet(z, path)
197
+ member = z.open # bytes; the utf-8 floor scan keys on the name open(
198
+ with member(sheet) as f:
199
+ row = _first_row(path, f)
200
+ except zipfile.BadZipFile:
201
+ raise CannotCheck(f"{path.name} is not an xlsx workbook") from None
202
+ except ElementTree.ParseError:
203
+ raise CannotCheck(f"{path.name} is not valid xlsx") from None
204
+ if row is None:
205
+ raise CannotCheck(f"{path.name} is not valid xlsx: no header row")
206
+ if row.get("r") not in (None, "1"):
207
+ raise CannotCheck(
208
+ f"{path.name} has no row 1; the header must be the first row of the sheet"
209
+ )
210
+ cells: dict[int, str] = {}
211
+ for cell in (c for c in row if _local(c.tag) == "c"):
212
+ # a cell without a coordinate follows the previous one
213
+ index = _column_index(cell.get("r")) if cell.get("r") else len(cells)
214
+ cells[index] = _cell_text(cell, shared).strip()
215
+ width = max(cells) + 1 if cells else 0
216
+ return [cells.get(i, "") for i in range(width)]
217
+
218
+
219
+ def _first_row(path: Path, stream):
220
+ """The first <row> element, fed to the parser in small pieces so nothing
221
+ past it is ever parsed; never closed, so a later broken row cannot fail
222
+ a good header. A DOCTYPE can only sit before the root element, so the
223
+ bytes fed until the first element opens are the ones checked for it."""
224
+ parser = ElementTree.XMLPullParser(events=("start", "end"))
225
+ preamble, opened, read = b"", False, 0
226
+ while True:
227
+ chunk = stream.read(4096)
228
+ if not chunk:
229
+ return None
230
+ read += len(chunk)
231
+ if read > XML_PART_CAP:
232
+ raise CannotCheck(
233
+ f"{path.name}: no header row in the first {XML_PART_CAP >> 20} MB of the sheet"
234
+ )
235
+ if not opened:
236
+ preamble += chunk
237
+ parser.feed(chunk)
238
+ for event, element in parser.read_events():
239
+ if event == "start" and not opened:
240
+ opened = True
241
+ _guard_xml(path, preamble)
242
+ if event == "end" and _local(element.tag) == "row":
243
+ return element
244
+
245
+
246
+ def _parquet(path: Path) -> list[str]:
247
+ try:
248
+ import pyarrow
249
+ import pyarrow.parquet as pq
250
+ except ImportError:
251
+ raise CannotCheck(
252
+ f"{path.name}: reading a Parquet schema needs pyarrow (pip install pyarrow), "
253
+ "or export the header as csv"
254
+ ) from None
255
+ version = tuple(
256
+ int(p) for p in str(getattr(pyarrow, "__version__", "0")).split(".")[:3] if p.isdigit()
257
+ )
258
+ if version < PYARROW_FLOOR:
259
+ floor = ".".join(str(n) for n in PYARROW_FLOOR)
260
+ raise CannotCheck(
261
+ f"pyarrow {getattr(pyarrow, '__version__', '?')} is older than {floor}, which reads untrusted Parquet "
262
+ "unsafely (CVE-2023-47248); upgrade it or export the header as csv"
263
+ )
264
+ try:
265
+ return list(pq.read_schema(path).names)
266
+ except Exception as e: # pyarrow raises its own hierarchy
267
+ raise CannotCheck(f"{path.name} is not valid Parquet ({e})") from None
268
+
269
+
270
+ def file_map(roots: list[Path]) -> dict[str, str]:
271
+ """File-name patterns to the tables they feed, from the project.
272
+
273
+ A dbt source table carries them as `meta: {ripple: {files: [...]}}`; a
274
+ plain SQL folder keeps a `files:` map in ripple.yml at its root. One
275
+ pattern naming two tables is a conflict, not a choice."""
276
+ import yaml
277
+
278
+ from ripple.loaders.dbt_config import source_tables
279
+
280
+ patterns: dict[str, str] = {}
281
+
282
+ def put(pattern: str, table: str, where: str) -> None:
283
+ if pattern in patterns and patterns[pattern] != table:
284
+ raise CannotCheck(
285
+ f"{where}: pattern {pattern} maps to two tables, {patterns[pattern]} and {table}"
286
+ )
287
+ patterns[pattern] = table
288
+
289
+ for root in dict.fromkeys(Path(r) for r in roots):
290
+ own = root / "ripple.yml"
291
+ if own.is_file():
292
+ try:
293
+ parsed = yaml.safe_load(own.read_text(encoding="utf-8", errors="replace")) or {}
294
+ except yaml.YAMLError as e:
295
+ raise CannotCheck(
296
+ f"ripple.yml is not valid yaml ({str(e).splitlines()[0]})"
297
+ ) from None
298
+ files = parsed.get("files") if isinstance(parsed, dict) else None
299
+ if files is not None and not isinstance(files, dict):
300
+ raise CannotCheck(
301
+ 'ripple.yml: files must map a pattern to a table, like "feed_*.csv": raw.feed'
302
+ )
303
+ for pattern, table in (files or {}).items():
304
+ put(str(pattern), str(table), "ripple.yml")
305
+ for source, table in source_tables(root):
306
+ meta = (table.get("meta") or {}).get("ripple") or {}
307
+ for pattern in meta.get("files") or []:
308
+ put(str(pattern), f"{source}.{table['name']}", "sources yml")
309
+ return patterns
310
+
311
+
312
+ def bind(path: Path, root: Path, patterns: dict[str, str]) -> tuple[str | None, str | None]:
313
+ """(table, how) for a file from the declared patterns, or (None, None).
314
+
315
+ A pattern with a folder in it matches the file's path (relative to the
316
+ project when it is inside it); a bare pattern matches the file name."""
317
+ spellings = [path.as_posix()]
318
+ resolved, base = path.resolve(), Path(root).resolve()
319
+ if base in resolved.parents:
320
+ spellings.append(resolved.relative_to(base).as_posix())
321
+ hits: dict[str, str] = {}
322
+ for pattern, table in patterns.items():
323
+ target = spellings if "/" in pattern else [path.name]
324
+ if any(fnmatch.fnmatch(s, pattern) for s in target):
325
+ hits.setdefault(table, pattern)
326
+ if len(hits) > 1:
327
+ listed = ", ".join(f"{p} for {t}" for t, p in hits.items())
328
+ raise CannotCheck(
329
+ f"{path.name} matches {len(hits)} patterns for different tables: {listed}. Pass --source."
330
+ )
331
+ if hits:
332
+ ((table, pattern),) = hits.items()
333
+ return table, f"files: {pattern}"
334
+ return None, None
335
+
336
+
337
+ def declared_sources(graph) -> list:
338
+ return [s for s in graph.project.sources if s.declared_columns]
339
+
340
+
341
+ def fits(header: list[str], graph) -> list[tuple[object, int]]:
342
+ """Tables ranked by how many of the file's columns are theirs."""
343
+ names = {h.lower() for h in header}
344
+ scored = []
345
+ for source in declared_sources(graph):
346
+ matched = sum(1 for c in source.declared_columns if c.lower() in names)
347
+ if matched:
348
+ scored.append((source, matched))
349
+ scored.sort(key=lambda pair: (-pair[1], pair[0].name))
350
+ return scored
351
+
352
+
353
+ def bind_by_header(header: list[str], graph) -> tuple[object | None, str | None, list]:
354
+ """The one table the file's columns name, or the tables that tie.
355
+
356
+ A file binds when at least two of its columns are one table's, they are
357
+ more than half of the file's columns, and no other table matches as
358
+ many. Anything less is listed for the user to pick from, never picked."""
359
+ ranked = fits(header, graph)
360
+ strong = [(s, n) for s, n in ranked if n >= 2 and n * 2 > len(header)]
361
+ if not strong:
362
+ return None, None, []
363
+ best = strong[0][1]
364
+ tied = [(s, n) for s, n in strong if n == best]
365
+ if len(tied) > 1:
366
+ return None, None, tied
367
+ return tied[0][0], f"matched by {best} of {_plural(len(header), 'column name')}", tied
368
+
369
+
370
+ def find_source(graph, table: str):
371
+ """The source table by its name, an alias, or its bare table name.
372
+
373
+ Case-insensitive. An alias or bare name two schemas' tables share is
374
+ refused with both spellings; an unknown name lists the tables that can
375
+ be checked."""
376
+ wanted = table.lower()
377
+ exact = [s for s in graph.project.sources if s.name.lower() == wanted]
378
+ if len(exact) == 1:
379
+ return exact[0]
380
+ hits = [
381
+ s
382
+ for s in graph.project.sources
383
+ if wanted in {a.lower() for a in s.aliases} or s.name.rsplit(".", 1)[-1].lower() == wanted
384
+ ]
385
+ if len(hits) == 1:
386
+ return hits[0]
387
+ if hits:
388
+ names = ", ".join(sorted(s.name for s in hits))
389
+ raise CannotCheck(
390
+ f"{table} names {len(hits)} tables: {names}. Pass --source with the schema."
391
+ )
392
+ raise CannotCheck(f"no table called {table} in this project. {table_list(graph)}")
393
+
394
+
395
+ def table_list(graph) -> str:
396
+ tables = declared_sources(graph)
397
+ if not tables:
398
+ return "No source table has declared columns yet; add them to its yml or run ripple ingest-schema."
399
+ listed = ", ".join(f"{s.name} ({_plural(len(s.declared_columns), 'column')})" for s in tables)
400
+ return f"Tables with declared columns: {listed}."
401
+
402
+
403
+ def _readers(graph, table: str, column: str) -> dict:
404
+ """Who reads one column of the table: models (a filter counts), dashboards,
405
+ the hardest hit; through select * when the graph cannot tell which
406
+ columns those readers use."""
407
+ try:
408
+ answer = breaks_answer(graph.breaks(table, column, max_depth=MAX_DEPTH), "model")
409
+ except UnknownTarget:
410
+ star = {e.dst_model for e in graph.edges if e.src_model == table and e.src_column == "*"}
411
+ return {
412
+ "models": star,
413
+ "dashboards": set(),
414
+ "filters": set(),
415
+ "hit_hardest": [],
416
+ "via_star": bool(star),
417
+ }
418
+ nodes = [n for n in answer["nodes"] if n["depth"] > 0]
419
+ filters = {r["model"] for r in answer["row_level"]}
420
+ return {
421
+ "models": {n["model"] for n in nodes if n["kind"] == "model"} | filters,
422
+ "dashboards": {n["model"] for n in nodes if n["kind"] == "dashboard"},
423
+ "filters": filters,
424
+ "hit_hardest": hit_hardest(answer, 1),
425
+ "via_star": False,
426
+ }
427
+
428
+
429
+ def check(
430
+ graph, table: str, header: list[str], file_name: str, how: str, noun: str = "model"
431
+ ) -> dict:
432
+ """The report: every column's status, who reads the missing ones, the verdict."""
433
+ source = find_source(graph, table)
434
+ declared = list(source.declared_columns)
435
+ if not declared:
436
+ raise CannotCheck(
437
+ f"{source.name} has no declared columns, so there is nothing to check the file against. "
438
+ "Add them to its yml, or run ripple ingest-schema with your warehouse's column list."
439
+ )
440
+ present = {h.lower() for h in header}
441
+ known = {c.lower() for c in declared}
442
+ missing = [c for c in declared if c.lower() not in present]
443
+ new = [h for h in header if h.lower() not in known]
444
+ columns: list[dict] = []
445
+ who: dict[str, dict] = {}
446
+ for c in declared:
447
+ if c.lower() in present:
448
+ columns.append({"name": c, "status": "unchanged"})
449
+ continue
450
+ r = who[c] = _readers(graph, source.name, c)
451
+ columns.append(
452
+ {
453
+ "name": c,
454
+ "status": "missing",
455
+ "readers": {"models": len(r["models"]), "dashboards": len(r["dashboards"])},
456
+ "filters": len(r["filters"]),
457
+ "via_star": r["via_star"],
458
+ "hit_hardest": [{"model": m, "columns": n} for m, n in r["hit_hardest"]],
459
+ }
460
+ )
461
+ read_missing = [c["name"] for c in columns if c["status"] == "missing" and _count(c)]
462
+ for h in new:
463
+ like = difflib.get_close_matches(
464
+ h.lower(), [m.lower() for m in read_missing], n=1, cutoff=LOOKS_LIKE
465
+ )
466
+ entry = {"name": h, "status": "new"}
467
+ if like:
468
+ entry["looks_like"] = next(m for m in read_missing if m.lower() == like[0])
469
+ columns.append(entry)
470
+ models: set[str] = set()
471
+ dashboards: set[str] = set()
472
+ for name in read_missing:
473
+ models |= who[name]["models"]
474
+ dashboards |= who[name]["dashboards"]
475
+ breaking = len(read_missing)
476
+ if breaking:
477
+ readers = len(models) + len(dashboards)
478
+ verdict = (
479
+ f"Stop: {_plural(breaking, 'column')} that {_where(len(models), len(dashboards), noun)} "
480
+ f"read{'s' if readers == 1 else ''} {'is' if breaking == 1 else 'are'} missing from this file."
481
+ )
482
+ else:
483
+ verdict = "Nothing downstream changes."
484
+ report = {
485
+ "file": file_name,
486
+ "table": source.name,
487
+ "method": how,
488
+ "checked": "names",
489
+ "columns": columns,
490
+ "counts": {
491
+ "file": len(header),
492
+ "unchanged": len(declared) - len(missing),
493
+ "missing": len(missing),
494
+ "new": len(new),
495
+ },
496
+ "verdict": verdict,
497
+ "exit": 1 if breaking else 0,
498
+ "noun": noun,
499
+ }
500
+ report["lines"] = lines(report)
501
+ return report
502
+
503
+
504
+ def failure_report(file_name: str, lines: list[str]) -> dict:
505
+ """The same shape as a finished check, for a check that could not run."""
506
+ return {
507
+ "file": file_name,
508
+ "table": None,
509
+ "method": None,
510
+ "checked": "names",
511
+ "columns": [],
512
+ "counts": None,
513
+ "verdict": lines[0],
514
+ "exit": 2,
515
+ "lines": lines,
516
+ }
517
+
518
+
519
+ def _count(c: dict) -> int:
520
+ return c["readers"]["models"] + c["readers"]["dashboards"]
521
+
522
+
523
+ def _plural(n: int, word: str) -> str:
524
+ return f"{n} {word}" + ("" if n == 1 else "s")
525
+
526
+
527
+ def _where(models: int, dashboards: int, noun: str) -> str:
528
+ parts = []
529
+ if models:
530
+ parts.append(_plural(models, noun))
531
+ if dashboards:
532
+ parts.append(_plural(dashboards, "dashboard"))
533
+ return " and ".join(parts) if parts else "nobody"
534
+
535
+
536
+ def _missing_line(c: dict, noun: str, width: int) -> str:
537
+ who = _where(c["readers"]["models"], c["readers"]["dashboards"], noun)
538
+ line = f"MISSING {c['name']:<{width}}read by {who}"
539
+ if c.get("via_star"):
540
+ return line + " through select * (which columns, Ripple cannot tell)"
541
+ if c.get("filters"):
542
+ line += f"; filters rows in {_plural(c['filters'], noun)}"
543
+ # worth a name only when one reader is hit in more than one column
544
+ if c["hit_hardest"] and c["hit_hardest"][0]["columns"] > 1:
545
+ top = c["hit_hardest"][0]
546
+ line += f"; hit hardest: {top['model']} ({_plural(top['columns'], 'column')})"
547
+ return line
548
+
549
+
550
+ def lines(report: dict) -> list[str]:
551
+ """The terminal text: verdict first and last, damage in between."""
552
+ noun = report.get("noun", "model")
553
+ counts = report["counts"]
554
+ listed = [c for c in report["columns"] if c["status"] == "missing" and _count(c)]
555
+ hinted = [c for c in report["columns"] if c["status"] == "new" and c.get("looks_like")]
556
+ quiet_new = [
557
+ c["name"] for c in report["columns"] if c["status"] == "new" and not c.get("looks_like")
558
+ ]
559
+ quiet_missing = [
560
+ c["name"] for c in report["columns"] if c["status"] == "missing" and not _count(c)
561
+ ]
562
+ width = max((len(c["name"]) for c in listed + hinted), default=0) + 3
563
+ out = [report["verdict"], ""]
564
+ out.append(f"File {report['file']}")
565
+ out.append(f"Table {report['table']} ({report['method']})")
566
+ out.append(HEADER_ONLY)
567
+ out.append("")
568
+ parts = [f"{counts['unchanged']} unchanged"]
569
+ if counts["missing"]:
570
+ parts.append(f"{counts['missing']} missing")
571
+ if counts["new"]:
572
+ parts.append(f"{counts['new']} new")
573
+ out.append(f"{_plural(counts['file'], 'column')}: {', '.join(parts)}")
574
+ out.append("")
575
+ for c in sorted(listed, key=lambda c: (-_count(c), c["name"])):
576
+ out.append(_missing_line(c, noun, width))
577
+ for c in hinted:
578
+ out.append(
579
+ f"NEW {c['name']:<{width}}looks like {c['looks_like']}; "
580
+ "a similar name does not prove the same contents"
581
+ )
582
+ if quiet_new:
583
+ more = "more " if hinted else ""
584
+ out.append(
585
+ f"{len(quiet_new)} {more}new column{'s' if len(quiet_new) != 1 else ''} nobody reads yet: {', '.join(quiet_new)}"
586
+ )
587
+ if quiet_missing:
588
+ out.append(
589
+ f"{len(quiet_missing)} missing column{'s' if len(quiet_missing) != 1 else ''} nobody reads yet: {', '.join(quiet_missing)}"
590
+ )
591
+ if listed:
592
+ out.append("")
593
+ fixes = []
594
+ renamed = {c["looks_like"]: c["name"] for c in hinted}
595
+ for c in listed:
596
+ if c["name"] in renamed:
597
+ fixes.append(
598
+ f"if {renamed[c['name']]} holds the same data as {c['name']}, rename that header to {c['name']}; "
599
+ f"otherwise restore {c['name']}."
600
+ )
601
+ else:
602
+ fixes.append(f"Restore {c['name']}.")
603
+ out.append("To fix: " + " ".join(fixes) + " Then run this check again.")
604
+ out.append(f"Details: ripple breaks {report['table']}.{listed[0]['name']}")
605
+ out.append("")
606
+ out.append(f"{report['verdict']} (exit {report['exit']})")
607
+ return out
608
+
609
+
610
+ def no_binding_lines(file_name: str, header: list[str], graph, tied: list) -> list[str]:
611
+ """What to do when no table claims the file: the choices, never a guess."""
612
+ if tied:
613
+ fits_line = ", ".join(
614
+ f"{s.name} ({n} of {_plural(len(header), 'column name')})" for s, n in tied
615
+ )
616
+ out = [f"Cannot check: {file_name} fits more than one table: {fits_line}."]
617
+ else:
618
+ out = [
619
+ f"Cannot check: no table is mapped to {file_name} and its columns fit none.",
620
+ table_list(graph),
621
+ ]
622
+ out.append(
623
+ "Run again with --source TABLE, or add to that source's yml so the next drop binds itself:"
624
+ )
625
+ out.append(f' meta: {{ripple: {{files: ["{file_name}"]}}}}')
626
+ out.append("A pattern like vendor_feed_*.csv covers every drop of the same feed.")
627
+ return out
@@ -258,6 +258,43 @@ def cmd_breaks(args) -> None:
258
258
  _page_door(args, project, graph, answer, plain_text(breaks_lines(answer, full=True)))
259
259
 
260
260
 
261
+ def cmd_check_file(args) -> None:
262
+ from ripple import check_file
263
+ from ripple.loaders.types import NotAProject
264
+ from ripple.project import find_project_root
265
+
266
+ path = Path(args.file)
267
+
268
+ def stop(lines: list[str]) -> None:
269
+ report = check_file.failure_report(path.name, lines)
270
+ print(json.dumps(report, indent=2) if args.json else "\n".join(lines))
271
+ sys.exit(2)
272
+
273
+ try:
274
+ project, graph = _load_graph(args)
275
+ root = find_project_root(Path(args.path))
276
+ header = check_file.read_header(path, args.encoding)
277
+ if args.source:
278
+ table, how = args.source, "--source"
279
+ else:
280
+ roots = [root, *getattr(project, "dbt_roots", [])]
281
+ table, how = check_file.bind(path, root, check_file.file_map(roots))
282
+ if table is None:
283
+ source, how, tied = check_file.bind_by_header(header, graph)
284
+ if source is None:
285
+ stop(check_file.no_binding_lines(path.name, header, graph, tied))
286
+ table = source.name
287
+ report = check_file.check(graph, table, header, path.name, how, noun_for(project.mode))
288
+ except (check_file.CannotCheck, NotAProject, OSError) as e:
289
+ stop([f"Cannot check: {e}"])
290
+ if args.json:
291
+ print(json.dumps(report, indent=2))
292
+ else:
293
+ print("\n".join(report["lines"]))
294
+ if report["exit"]:
295
+ sys.exit(report["exit"])
296
+
297
+
261
298
  def cmd_trace(args) -> None:
262
299
  project, graph = _load_graph(args)
263
300
  model, column = _parse_target(args.target)
@@ -604,6 +641,19 @@ def main(argv: list[str] | None = None) -> None:
604
641
  )
605
642
  p_columns.add_argument("model")
606
643
 
644
+ p_check = sub.add_parser(
645
+ "check-file",
646
+ help="a file from an external team: what breaks if it is loaded",
647
+ parents=[common],
648
+ )
649
+ p_check.add_argument("file", help="the csv, tsv, xlsx or parquet file that arrived")
650
+ p_check.add_argument(
651
+ "--source",
652
+ help="the table it feeds (raw.vendor_feed, or just vendor_feed); else the project's files map",
653
+ )
654
+ p_check.add_argument(
655
+ "--encoding", help="the file's text encoding when it is not UTF-8 (cp1252, latin-1)"
656
+ )
607
657
  p_breaks = sub.add_parser("breaks", help="what breaks if this column changes", parents=[common])
608
658
  p_breaks.add_argument("target", help="model.column")
609
659
  p_breaks.add_argument(
@@ -716,6 +766,7 @@ def main(argv: list[str] | None = None) -> None:
716
766
  "models": cmd_models,
717
767
  "columns": cmd_columns,
718
768
  "breaks": cmd_breaks,
769
+ "check-file": cmd_check_file,
719
770
  "trace": cmd_trace,
720
771
  "graph": cmd_graph,
721
772
  "ci": cmd_ci,
@@ -5,7 +5,7 @@ from __future__ import annotations
5
5
  import re
6
6
  from pathlib import Path
7
7
 
8
- from ripple.loaders.types import files_under
8
+ from ripple.loaders.types import IGNORE_DIRS, files_under
9
9
 
10
10
 
11
11
  def _project_vars(root: Path) -> dict:
@@ -339,3 +339,56 @@ def _dbt_owned_dirs(project_dir: Path) -> list[Path]:
339
339
  owned += _dbt_model_dirs(project_dir)
340
340
  owned += [project_dir / name for name in _macro_dir_names(project_dir / "dbt_project.yml")]
341
341
  return [d for d in owned if d.is_dir()]
342
+
343
+
344
+ def source_tables(root: Path):
345
+ """Every (source name, table dict) a dbt project declares in its yml.
346
+
347
+ Walks the model paths only (dbt_project.yml's model-paths), so a
348
+ package's or a build folder's yml never counts as this project's."""
349
+ import yaml
350
+
351
+ root = Path(root)
352
+ if not (root / "dbt_project.yml").is_file():
353
+ return
354
+ for model_dir in _dbt_model_dirs(root):
355
+ for yml in [*files_under(model_dir, "*.yml"), *files_under(model_dir, "*.yaml")]:
356
+ # a package's or a build folder's yml is not this project's, even
357
+ # when model-paths is the project root
358
+ if IGNORE_DIRS & set(yml.relative_to(model_dir).parts[:-1]):
359
+ continue
360
+ try:
361
+ parsed = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
362
+ except yaml.YAMLError:
363
+ continue
364
+ if not isinstance(parsed, dict):
365
+ continue
366
+ for source in parsed.get("sources") or []:
367
+ if not isinstance(source, dict) or not source.get("name"):
368
+ continue
369
+ for table in source.get("tables") or []:
370
+ if isinstance(table, dict) and table.get("name"):
371
+ yield str(source["name"]), table
372
+
373
+
374
+ def source_columns(root: Path) -> dict[str, list[str]]:
375
+ """The columns sources.yml declares, as {"schema.table": [names]}.
376
+
377
+ dbt users list a source's columns in yml for docs and tests; those are
378
+ the landing table's contract, and the graph read only ingested schemas
379
+ and seeds before (a dbt source always showed "not declared")."""
380
+ out: dict[str, list[str]] = {}
381
+ spelling: dict[str, str] = {}
382
+ for source, table in source_tables(root):
383
+ key = f"{source}.{table['name']}".lower()
384
+ name = spelling.setdefault(key, f"{source}.{table['name']}")
385
+ # two files may describe one table (docs in one, tests in another);
386
+ # the union is the contract, the first spelling stays
387
+ columns = out.setdefault(name, [])
388
+ known = {c.lower() for c in columns}
389
+ for column in table.get("columns") or []:
390
+ column_name = column.get("name") if isinstance(column, dict) else column
391
+ if column_name and str(column_name).lower() not in known:
392
+ columns.append(str(column_name))
393
+ known.add(str(column_name).lower())
394
+ return {name: columns for name, columns in out.items() if columns}
@@ -3,6 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import re
6
+ from pathlib import Path
6
7
 
7
8
  from ripple.loaders.types import Model, Project
8
9
 
@@ -85,6 +86,45 @@ def _qualifiers_by_bare_name(project: Project) -> dict[str, set[str]]:
85
86
  return seen
86
87
 
87
88
 
89
+ def _apply_declared_sources(project: Project) -> None:
90
+ """Expose sources.yml column lists as the sources' declared columns.
91
+
92
+ By exact name only (schema.table, case-insensitive): a declaration names
93
+ its own table, so it never rides the suffix attachment an ingested schema
94
+ needs, which merged raw.feed and staging.feed into one source. A table
95
+ no SQL references still becomes a source, so a file can be checked
96
+ against it. Every dbt project in a monorepo declares its own."""
97
+ from ripple.loaders.dbt_config import source_columns
98
+
99
+ declared: dict[str, list[str]] = {}
100
+ spelling: dict[str, str] = {}
101
+ for root in project.dbt_roots:
102
+ for name, columns in source_columns(Path(root)).items():
103
+ key = name.lower()
104
+ known = declared.setdefault(spelling.setdefault(key, name), [])
105
+ seen = {c.lower() for c in known}
106
+ known.extend(c for c in columns if c.lower() not in seen)
107
+ if not declared:
108
+ return
109
+ by_name = {s.name.lower(): s for s in project.sources}
110
+ owned = {m.name.lower() for m in project.models if m.sql.strip()}
111
+ for name, columns in declared.items():
112
+ target = by_name.get(name.lower())
113
+ if target is None:
114
+ target = Model(
115
+ name=name,
116
+ sql="",
117
+ path="",
118
+ aliases={a for a in _name_suffixes(name) if a.lower() not in owned},
119
+ is_source=True,
120
+ evidence="declared",
121
+ )
122
+ project.sources.append(target)
123
+ by_name[name.lower()] = target
124
+ seen = {c.lower() for c in target.declared_columns}
125
+ target.declared_columns.extend(c for c in columns if c.lower() not in seen)
126
+
127
+
88
128
  def _apply_ingested_schemas(project: Project, extra_roots: tuple = ()) -> None:
89
129
  """Expose .ripple/schemas.json tables as known external relations.
90
130
 
@@ -122,6 +122,9 @@ class Project:
122
122
  downstream: list = field(default_factory=list)
123
123
  # vars declared in dbt_project.yml, resolved by var() during tier-b renders
124
124
  jinja_vars: dict = field(default_factory=dict)
125
+ # every dbt project directory that was loaded: one for a dbt project, all
126
+ # of them for a monorepo, none for a plain SQL folder
127
+ dbt_roots: list = field(default_factory=list)
125
128
 
126
129
  @property
127
130
  def name_candidates(self) -> dict[str, list[str]]:
@@ -22,7 +22,11 @@ from pathlib import Path
22
22
 
23
23
  from ripple.loaders.dbt import _load_one_dbt, resolve_dbt_dialect
24
24
  from ripple.loaders.dbt_config import _dbt_owned_dirs
25
- from ripple.loaders.identity import _apply_ingested_schemas, _qualify_collisions
25
+ from ripple.loaders.identity import (
26
+ _apply_declared_sources,
27
+ _apply_ingested_schemas,
28
+ _qualify_collisions,
29
+ )
26
30
  from ripple.loaders.sqldir import _load_sql_dir
27
31
  from ripple.loaders.types import (
28
32
  ADAPTER_TO_DIALECT,
@@ -83,7 +87,11 @@ def load_project(path: str | Path = ".", dialect: str | None = None) -> Project:
83
87
  f"{len(stray.models)} plain-SQL models outside the dbt "
84
88
  f"project{'s' if len(dbt_dirs) != 1 else ''} were loaded as well."
85
89
  )
90
+ project.dbt_roots = (
91
+ list(dbt_dirs) if dbt_dirs else ([root] if (root / "dbt_project.yml").exists() else [])
92
+ )
86
93
  _qualify_collisions(project)
94
+ _apply_declared_sources(project)
87
95
  _apply_ingested_schemas(project, extra_roots=(scan_root, start))
88
96
  from ripple.lookml import load_lookml_downstream
89
97
  from ripple.semantic import load_downstream
File without changes
File without changes
File without changes