ripple-sql 0.1.7__tar.gz → 0.1.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/CHANGELOG.md +8 -1
  2. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/PKG-INFO +48 -1
  3. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/README.md +47 -0
  4. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/pyproject.toml +1 -1
  5. ripple_sql-0.1.8/src/ripple/check_file.py +336 -0
  6. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/cli.py +47 -0
  7. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/dbt_config.py +44 -0
  8. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/identity.py +11 -2
  9. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/.gitignore +0 -0
  10. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/LICENSE +0 -0
  11. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/__init__.py +0 -0
  12. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/answer.py +0 -0
  13. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/answer_page.py +0 -0
  14. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/cache.py +0 -0
  15. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/ci.py +0 -0
  16. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/ci_signature.py +0 -0
  17. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/doctor.py +0 -0
  18. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/__init__.py +0 -0
  19. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/budget.py +0 -0
  20. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/column_lineage.py +0 -0
  21. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/column_ref.py +0 -0
  22. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/cte_tracing.py +0 -0
  23. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/dependencies.py +0 -0
  24. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/dialect.py +0 -0
  25. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/dispatch.py +0 -0
  26. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/extraction.py +0 -0
  27. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/generators.py +0 -0
  28. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/jinja.py +0 -0
  29. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/json_sources.py +0 -0
  30. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/macro_source.py +0 -0
  31. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/pipeline.py +0 -0
  32. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/preprocess.py +0 -0
  33. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/safe_gen.py +0 -0
  34. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/schema_qualification.py +0 -0
  35. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/scope.py +0 -0
  36. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/select_sources.py +0 -0
  37. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/sql_script.py +0 -0
  38. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/statement.py +0 -0
  39. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tech_debt.py +0 -0
  40. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_catalog.py +0 -0
  41. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_scalar_vars.py +0 -0
  42. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_tvf.py +0 -0
  43. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/tsql_xml.py +0 -0
  44. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/types.py +0 -0
  45. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/unused_deps.py +0 -0
  46. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/engine/validation.py +0 -0
  47. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/graph.py +0 -0
  48. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/home.py +0 -0
  49. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/__init__.py +0 -0
  50. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/dbt.py +0 -0
  51. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/sidecar.py +0 -0
  52. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/sqldir.py +0 -0
  53. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/loaders/types.py +0 -0
  54. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/lookml.py +0 -0
  55. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/mcp_server.py +0 -0
  56. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/names.py +0 -0
  57. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/project.py +0 -0
  58. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/py.typed +0 -0
  59. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/render.py +0 -0
  60. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/render_shims.py +0 -0
  61. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/schemas.py +0 -0
  62. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/semantic.py +0 -0
  63. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/server.py +0 -0
  64. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/sourcefiles.py +0 -0
  65. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/star_resolution.py +0 -0
  66. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/answer.css +0 -0
  67. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/answer.html +0 -0
  68. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/answer_twin.js +0 -0
  69. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/explore.js +0 -0
  70. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/find.js +0 -0
  71. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/static/focus.js +0 -0
  72. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/__init__.py +0 -0
  73. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/cli.py +0 -0
  74. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/collect.py +0 -0
  75. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/discover.py +0 -0
  76. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/ingest.py +0 -0
  77. {ripple_sql-0.1.7 → ripple_sql-0.1.8}/src/ripple/usage/report.py +0 -0
@@ -6,6 +6,12 @@ Notable changes to Ripple. The format follows
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.8] - 2026-09-15
10
+
11
+ ### Added
12
+ - `ripple check-file FILE [--source TABLE]`: a file arrives from an external team (csv, tsv, xlsx, parquet), and before anyone loads it Ripple diffs its header against the landing table's declared columns and prints who reads every column that is gone: models, dashboards, hit hardest. The verdict is the first line and the last; a similar name is a hint ("looks like customer_name; a similar name does not prove the same contents"), never an assumption; exit 0 when nothing downstream changes (new columns nobody reads included), 1 when something breaks, 2 when the check could not run. The file binds to its table by `--source`, or by a `files:` pattern the project declares (`meta: {ripple: {files: ["vendor_feed_*.csv"]}}` on a dbt source, or `files:` in `ripple.yml` for a plain SQL folder); with neither, the command prints the closest table and the exact line to add and stops. Header only: no data row is read and nothing leaves the machine. Shaped by a two-model debate; the spec's sample output is the command's own.
13
+ - A dbt source's `columns:` in sources.yml are its declared columns. The graph read only ingested schemas and seeds before, so every dbt source showed as "not declared" and a select star through it stayed unresolved even when the yml listed every column.
14
+
9
15
  ## [0.1.7] - 2026-09-10
10
16
 
11
17
  ### Fixed
@@ -264,7 +270,8 @@ First public release.
264
270
  ### Removed
265
271
  - The dark whole-graph canvas that `ripple serve` used to open (`static/index.html`). Every surface now draws one answer.
266
272
 
267
- [Unreleased]: https://github.com/bteh/ripple/compare/v0.1.7...HEAD
273
+ [Unreleased]: https://github.com/bteh/ripple/compare/v0.1.8...HEAD
274
+ [0.1.8]: https://github.com/bteh/ripple/releases/tag/v0.1.8
268
275
  [0.1.7]: https://github.com/bteh/ripple/releases/tag/v0.1.7
269
276
  [0.1.6]: https://github.com/bteh/ripple/releases/tag/v0.1.6
270
277
  [0.1.5]: https://github.com/bteh/ripple/releases/tag/v0.1.5
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: ripple-sql
3
- Version: 0.1.7
3
+ Version: 0.1.8
4
4
  Summary: Offline column-level SQL lineage. See what breaks before you merge.
5
5
  Project-URL: Homepage, https://github.com/bteh/ripple
6
6
  Project-URL: Repository, https://github.com/bteh/ripple
@@ -210,6 +210,53 @@ ripple unresolved # which tables block the most coverage, worst
210
210
  ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
211
211
  ```
212
212
 
213
+ ## A file arrives from another team
214
+
215
+ Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
216
+
217
+ ```bash
218
+ ripple check-file vendor_feed_2026-09.csv
219
+ ```
220
+
221
+ ```
222
+ Stop: 2 columns that 4 models read are missing from this file.
223
+
224
+ File vendor_feed_2026-09.csv
225
+ Table raw.vendor_feed (files: vendor_feed_*.csv)
226
+ Checked: column names only. Not checked: values, types, whether the file loads.
227
+
228
+ 6 columns: 2 unchanged, 2 missing, 4 new
229
+
230
+ MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
231
+ MISSING amount_usd read by 2 models
232
+ NEW cust_nm looks like customer_name; a similar name does not prove the same contents
233
+ 3 more new columns nobody reads yet: region, notes, batch_id
234
+
235
+ To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
236
+ Details: ripple breaks raw.vendor_feed.customer_name
237
+
238
+ Stop: 2 columns that 4 models read are missing from this file. (exit 1)
239
+ ```
240
+
241
+ The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
242
+
243
+ The file finds its table through `--source raw.vendor_feed`, or through a pattern the project declares once:
244
+
245
+ ```yaml
246
+ sources:
247
+ - name: raw
248
+ tables:
249
+ - name: vendor_feed
250
+ meta:
251
+ ripple:
252
+ files: ["vendor_feed_*.csv"]
253
+ columns:
254
+ - name: customer_id
255
+ - name: customer_name
256
+ ```
257
+
258
+ A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow installed). `--json` carries the same facts for an ingest job.
259
+
213
260
  ## Bring your query history (optional)
214
261
 
215
262
  Your SQL files say what is supposed to happen. The warehouse's query log says what
@@ -170,6 +170,53 @@ ripple unresolved # which tables block the most coverage, worst
170
170
  ripple ingest-schema cols.csv # CSV with a table,column header (or JSON, or - for stdin)
171
171
  ```
172
172
 
173
+ ## A file arrives from another team
174
+
175
+ Vendors and business units hand over CSVs and spreadsheets by hand. When a column is renamed or dropped in one of them, the pipeline loads it anyway and dashboards go blank the next morning. Before loading, ask Ripple what this exact file breaks:
176
+
177
+ ```bash
178
+ ripple check-file vendor_feed_2026-09.csv
179
+ ```
180
+
181
+ ```
182
+ Stop: 2 columns that 4 models read are missing from this file.
183
+
184
+ File vendor_feed_2026-09.csv
185
+ Table raw.vendor_feed (files: vendor_feed_*.csv)
186
+ Checked: column names only. Not checked: values, types, whether the file loads.
187
+
188
+ 6 columns: 2 unchanged, 2 missing, 4 new
189
+
190
+ MISSING customer_name read by 3 models; hit hardest: dim_customer (2 columns)
191
+ MISSING amount_usd read by 2 models
192
+ NEW cust_nm looks like customer_name; a similar name does not prove the same contents
193
+ 3 more new columns nobody reads yet: region, notes, batch_id
194
+
195
+ To fix: if cust_nm holds the same data as customer_name, rename that header to customer_name; otherwise restore customer_name. Restore amount_usd. Then run this check again.
196
+ Details: ripple breaks raw.vendor_feed.customer_name
197
+
198
+ Stop: 2 columns that 4 models read are missing from this file. (exit 1)
199
+ ```
200
+
201
+ The first line and the last are the same sentence, so the vendor's screenshot and the engineer's scrolled terminal both end on the verdict. Exit 0 means nothing downstream changes (a new column nobody reads is fine), 1 means something breaks, 2 means the check could not run. Header only: no data row is read and nothing leaves the machine.
202
+
203
+ The file finds its table through `--source raw.vendor_feed`, or through a pattern the project declares once:
204
+
205
+ ```yaml
206
+ sources:
207
+ - name: raw
208
+ tables:
209
+ - name: vendor_feed
210
+ meta:
211
+ ripple:
212
+ files: ["vendor_feed_*.csv"]
213
+ columns:
214
+ - name: customer_id
215
+ - name: customer_name
216
+ ```
217
+
218
+ A plain SQL folder keeps the same map in `ripple.yml` (`files: {"vendor_feed_*.csv": raw.vendor_feed}`). The table's declared columns are the contract: `columns:` in sources.yml, a seed, or a list from `ripple ingest-schema`. Reads csv, tsv, xlsx and parquet (parquet needs pyarrow installed). `--json` carries the same facts for an ingest job.
219
+
173
220
  ## Bring your query history (optional)
174
221
 
175
222
  Your SQL files say what is supposed to happen. The warehouse's query log says what
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "ripple-sql"
3
- version = "0.1.7"
3
+ version = "0.1.8"
4
4
  description = "Offline column-level SQL lineage. See what breaks before you merge."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"
@@ -0,0 +1,336 @@
1
+ """A file arrives from an external team: what breaks if it is loaded.
2
+
3
+ The file feeds one landing table the graph already knows. Its header is
4
+ diffed against the table's declared columns, and every column that is gone
5
+ gets the blast radius `ripple breaks` would give it. Header only: no data
6
+ row is read, nothing leaves the machine, and a similar name is a hint,
7
+ never an assumption.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import csv
13
+ import difflib
14
+ import fnmatch
15
+ import zipfile
16
+ from pathlib import Path
17
+ from xml.etree import ElementTree
18
+
19
+ from ripple.answer import breaks_answer, hit_hardest
20
+ from ripple.graph import UnknownTarget
21
+
22
+ HEADER_ONLY = "Checked: column names only. Not checked: values, types, whether the file loads."
23
+ KINDS = ("csv", "tsv", "xlsx", "parquet")
24
+ LOOKS_LIKE = 0.6
25
+
26
+
27
+ class CannotCheck(Exception):
28
+ """The check could not run; the message says why and what to do."""
29
+
30
+
31
+ def read_header(path: Path) -> list[str]:
32
+ """The column names of a file, from its header only."""
33
+ kind = path.suffix.lower().lstrip(".")
34
+ if kind in ("csv", "tsv", "txt"):
35
+ return _delimited(path, "\t" if kind == "tsv" else ",")
36
+ if kind == "xlsx":
37
+ return _xlsx(path)
38
+ if kind == "parquet":
39
+ return _parquet(path)
40
+ raise CannotCheck(f"{path.name}: Ripple reads a csv, tsv, xlsx or parquet header, not .{kind}")
41
+
42
+
43
+ def _delimited(path: Path, delimiter: str) -> list[str]:
44
+ # newline="" lets the csv module see a quoted header that spans lines
45
+ with open(path, encoding="utf-8-sig", errors="replace", newline="") as f:
46
+ try:
47
+ header = next(csv.reader(f, delimiter=delimiter))
48
+ except StopIteration:
49
+ raise CannotCheck(f"{path.name} is empty; no header row to check") from None
50
+ # a quoted header that spans lines keeps the file's own line ending
51
+ return [h.replace("\r\n", "\n").strip() for h in header]
52
+
53
+
54
+ def _local(tag: str) -> str:
55
+ return tag.rsplit("}", 1)[-1]
56
+
57
+
58
+ def _xlsx(path: Path) -> list[str]:
59
+ """Row 1 of the first sheet, without a spreadsheet library."""
60
+ try:
61
+ with zipfile.ZipFile(path) as z:
62
+ names = z.namelist()
63
+ shared: list[str] = []
64
+ if "xl/sharedStrings.xml" in names:
65
+ for si in ElementTree.fromstring(z.read("xl/sharedStrings.xml")):
66
+ shared.append("".join(t.text or "" for t in si.iter() if _local(t.tag) == "t"))
67
+ sheet = next((n for n in sorted(names) if n.startswith("xl/worksheets/sheet")), None)
68
+ if not sheet:
69
+ raise CannotCheck(f"{path.name} has no worksheet")
70
+ root = ElementTree.fromstring(z.read(sheet))
71
+ except zipfile.BadZipFile:
72
+ raise CannotCheck(f"{path.name} is not an xlsx workbook") from None
73
+ row = next((r for r in root.iter() if _local(r.tag) == "row"), None)
74
+ if row is None:
75
+ raise CannotCheck(f"{path.name} is empty; no header row to check")
76
+ header = []
77
+ for cell in row:
78
+ if _local(cell.tag) != "c":
79
+ continue
80
+ kind = cell.get("t")
81
+ value = next((v.text or "" for v in cell.iter() if _local(v.tag) in ("v", "t")), "")
82
+ if kind == "s" and value.isdigit() and int(value) < len(shared):
83
+ value = shared[int(value)]
84
+ header.append(value.strip())
85
+ return header
86
+
87
+
88
+ def _parquet(path: Path) -> list[str]:
89
+ try:
90
+ import pyarrow.parquet as pq
91
+ except ImportError:
92
+ raise CannotCheck(
93
+ f"{path.name}: reading a Parquet schema needs pyarrow (pip install pyarrow), "
94
+ "or export the header as csv"
95
+ ) from None
96
+ return list(pq.read_schema(path).names)
97
+
98
+
99
+ def file_map(root: Path) -> dict[str, str]:
100
+ """File-name patterns to the tables they feed, from the project.
101
+
102
+ A dbt source table carries them as `meta: {ripple: {files: [...]}}`; a
103
+ plain SQL folder keeps a `files:` map in ripple.yml at its root."""
104
+ import yaml
105
+
106
+ from ripple.loaders.dbt_config import source_tables
107
+
108
+ root = Path(root)
109
+ patterns: dict[str, str] = {}
110
+ own = root / "ripple.yml"
111
+ if own.is_file():
112
+ parsed = yaml.safe_load(own.read_text(encoding="utf-8", errors="replace")) or {}
113
+ for pattern, table in (parsed.get("files") or {}).items():
114
+ patterns[str(pattern)] = str(table)
115
+ for source, table in source_tables(root):
116
+ meta = (table.get("meta") or {}).get("ripple") or {}
117
+ for pattern in meta.get("files") or []:
118
+ patterns[str(pattern)] = f"{source}.{table['name']}"
119
+ return patterns
120
+
121
+
122
+ def bind(
123
+ file_name: str, explicit: str | None, patterns: dict[str, str]
124
+ ) -> tuple[str | None, str | None]:
125
+ """(table, how) for a file: --source, else the first matching pattern."""
126
+ if explicit:
127
+ return explicit, "--source"
128
+ for pattern, table in patterns.items():
129
+ if fnmatch.fnmatch(file_name, pattern):
130
+ return table, f"files: {pattern}"
131
+ return None, None
132
+
133
+
134
+ def find_source(graph, table: str):
135
+ """The source table by its name or an alias, case-insensitive."""
136
+ wanted = table.lower()
137
+ for source in graph.project.sources:
138
+ if source.name.lower() == wanted or wanted in {a.lower() for a in source.aliases}:
139
+ return source
140
+ return None
141
+
142
+
143
+ def closest_source(graph, file_name: str) -> str | None:
144
+ stem = Path(file_name).stem.lower()
145
+ names = [s.name for s in graph.project.sources]
146
+ by_tail = {n.rsplit(".", 1)[-1].lower(): n for n in names}
147
+ match = difflib.get_close_matches(stem, list(by_tail), n=1, cutoff=0.4)
148
+ return by_tail[match[0]] if match else (names[0] if names else None)
149
+
150
+
151
+ def _readers(graph, table: str, column: str) -> dict:
152
+ try:
153
+ answer = breaks_answer(graph.breaks(table, column), "model")
154
+ except UnknownTarget:
155
+ return {"models": set(), "dashboards": set(), "hit_hardest": []}
156
+ nodes = [n for n in answer["nodes"] if n["depth"] > 0]
157
+ return {
158
+ "models": {n["model"] for n in nodes if n["kind"] == "model"},
159
+ "dashboards": {n["model"] for n in nodes if n["kind"] == "dashboard"},
160
+ "hit_hardest": hit_hardest(answer, 1),
161
+ }
162
+
163
+
164
+ def check(
165
+ graph, table: str, header: list[str], file_name: str, how: str, noun: str = "model"
166
+ ) -> dict:
167
+ """The report: every column's status, who reads the missing ones, the verdict."""
168
+ source = find_source(graph, table)
169
+ if source is None:
170
+ names = [s.name for s in graph.project.sources]
171
+ near = difflib.get_close_matches(table, names, n=3, cutoff=0.5)
172
+ hint = f" Did you mean {', '.join(near)}?" if near else ""
173
+ raise CannotCheck(f"no source table called {table} in this project.{hint}")
174
+ declared = list(source.declared_columns)
175
+ if not declared:
176
+ raise CannotCheck(
177
+ f"{source.name} has no declared columns, so there is nothing to check the file against. "
178
+ "Add them to its yml, or run ripple ingest-schema with your warehouse's column list."
179
+ )
180
+ present = {h.lower() for h in header}
181
+ known = {c.lower() for c in declared}
182
+ missing = [c for c in declared if c.lower() not in present]
183
+ new = [h for h in header if h.lower() not in known]
184
+ columns: list[dict] = []
185
+ for c in declared:
186
+ if c.lower() in present:
187
+ columns.append({"name": c, "status": "unchanged"})
188
+ continue
189
+ r = _readers(graph, source.name, c)
190
+ entry = {
191
+ "name": c,
192
+ "status": "missing",
193
+ "readers": {"models": len(r["models"]), "dashboards": len(r["dashboards"])},
194
+ "hit_hardest": [{"model": m, "columns": n} for m, n in r["hit_hardest"]],
195
+ }
196
+ entry["_who"] = r
197
+ columns.append(entry)
198
+ read_missing = [c["name"] for c in columns if c["status"] == "missing" and _count(c)]
199
+ for h in new:
200
+ like = difflib.get_close_matches(
201
+ h.lower(), [m.lower() for m in read_missing], n=1, cutoff=LOOKS_LIKE
202
+ )
203
+ entry = {"name": h, "status": "new"}
204
+ if like:
205
+ entry["looks_like"] = next(m for m in read_missing if m.lower() == like[0])
206
+ columns.append(entry)
207
+ models: set[str] = set()
208
+ dashboards: set[str] = set()
209
+ for c in columns:
210
+ if c["status"] == "missing" and _count(c):
211
+ models |= c["_who"]["models"]
212
+ dashboards |= c["_who"]["dashboards"]
213
+ for c in columns:
214
+ c.pop("_who", None)
215
+ breaking = len(read_missing)
216
+ if breaking:
217
+ verdict = (
218
+ f"Stop: {_plural(breaking, 'column')} that {_where(len(models), len(dashboards), noun)} "
219
+ f"read {'is' if breaking == 1 else 'are'} missing from this file."
220
+ )
221
+ else:
222
+ verdict = "Nothing downstream changes."
223
+ report = {
224
+ "file": file_name,
225
+ "table": source.name,
226
+ "method": how,
227
+ "checked": "names",
228
+ "columns": columns,
229
+ "counts": {
230
+ "file": len(header),
231
+ "unchanged": len(declared) - len(missing),
232
+ "missing": len(missing),
233
+ "new": len(new),
234
+ },
235
+ "verdict": verdict,
236
+ "exit": 1 if breaking else 0,
237
+ "noun": noun,
238
+ }
239
+ report["lines"] = lines(report)
240
+ return report
241
+
242
+
243
+ def _count(c: dict) -> int:
244
+ return c["readers"]["models"] + c["readers"]["dashboards"]
245
+
246
+
247
+ def _plural(n: int, word: str) -> str:
248
+ return f"{n} {word}" + ("" if n == 1 else "s")
249
+
250
+
251
+ def _where(models: int, dashboards: int, noun: str) -> str:
252
+ parts = []
253
+ if models:
254
+ parts.append(_plural(models, noun))
255
+ if dashboards:
256
+ parts.append(_plural(dashboards, "dashboard"))
257
+ return " and ".join(parts) if parts else "nobody"
258
+
259
+
260
+ def lines(report: dict) -> list[str]:
261
+ """The terminal text: verdict first and last, damage in between."""
262
+ noun = report.get("noun", "model")
263
+ counts = report["counts"]
264
+ listed = [c for c in report["columns"] if c["status"] == "missing" and _count(c)]
265
+ hinted = [c for c in report["columns"] if c["status"] == "new" and c.get("looks_like")]
266
+ quiet_new = [
267
+ c["name"] for c in report["columns"] if c["status"] == "new" and not c.get("looks_like")
268
+ ]
269
+ quiet_missing = [
270
+ c["name"] for c in report["columns"] if c["status"] == "missing" and not _count(c)
271
+ ]
272
+ width = max((len(c["name"]) for c in listed + hinted), default=0) + 3
273
+ out = [report["verdict"], ""]
274
+ out.append(f"File {report['file']}")
275
+ out.append(f"Table {report['table']} ({report['method']})")
276
+ out.append(HEADER_ONLY)
277
+ out.append("")
278
+ parts = [f"{counts['unchanged']} unchanged"]
279
+ if counts["missing"]:
280
+ parts.append(f"{counts['missing']} missing")
281
+ if counts["new"]:
282
+ parts.append(f"{counts['new']} new")
283
+ out.append(f"{_plural(counts['file'], 'column')}: {', '.join(parts)}")
284
+ out.append("")
285
+ for c in sorted(listed, key=lambda c: (-_count(c), c["name"])):
286
+ who = _where(c["readers"]["models"], c["readers"]["dashboards"], noun)
287
+ line = f"MISSING {c['name']:<{width}}read by {who}"
288
+ # worth a name only when one reader is hit in more than one column
289
+ if c["hit_hardest"] and c["hit_hardest"][0]["columns"] > 1:
290
+ top = c["hit_hardest"][0]
291
+ line += f"; hit hardest: {top['model']} ({_plural(top['columns'], 'column')})"
292
+ out.append(line)
293
+ for c in hinted:
294
+ out.append(
295
+ f"NEW {c['name']:<{width}}looks like {c['looks_like']}; "
296
+ "a similar name does not prove the same contents"
297
+ )
298
+ if quiet_new:
299
+ more = "more " if hinted else ""
300
+ out.append(
301
+ f"{len(quiet_new)} {more}new column{'s' if len(quiet_new) != 1 else ''} nobody reads yet: {', '.join(quiet_new)}"
302
+ )
303
+ if quiet_missing:
304
+ out.append(
305
+ f"{len(quiet_missing)} missing column{'s' if len(quiet_missing) != 1 else ''} nobody reads yet: {', '.join(quiet_missing)}"
306
+ )
307
+ if listed:
308
+ out.append("")
309
+ fixes = []
310
+ renamed = {c["looks_like"]: c["name"] for c in hinted}
311
+ for c in listed:
312
+ if c["name"] in renamed:
313
+ fixes.append(
314
+ f"if {renamed[c['name']]} holds the same data as {c['name']}, rename that header to {c['name']}; "
315
+ f"otherwise restore {c['name']}."
316
+ )
317
+ else:
318
+ fixes.append(f"Restore {c['name']}.")
319
+ out.append("To fix: " + " ".join(fixes) + " Then run this check again.")
320
+ out.append(f"Details: ripple breaks {report['table']}.{listed[0]['name']}")
321
+ out.append("")
322
+ out.append(f"{report['verdict']} (exit {report['exit']})")
323
+ return out
324
+
325
+
326
+ def no_binding_lines(file_name: str, graph) -> list[str]:
327
+ """What to do when no source claims the file: a hint, never a guess."""
328
+ guess = closest_source(graph, file_name)
329
+ out = [f"Cannot check: no source is mapped to {file_name}."]
330
+ if guess:
331
+ out.append(
332
+ f"It looks like {guess}. Run again with --source {guess}, or add to that source's yml:"
333
+ )
334
+ out.append(f' meta: {{ripple: {{files: ["{file_name}"]}}}}')
335
+ out.append("A pattern like vendor_feed_*.csv covers every drop of the same feed.")
336
+ return out
@@ -258,6 +258,43 @@ def cmd_breaks(args) -> None:
258
258
  _page_door(args, project, graph, answer, plain_text(breaks_lines(answer, full=True)))
259
259
 
260
260
 
261
+ def cmd_check_file(args) -> None:
262
+ from ripple import check_file
263
+ from ripple.project import find_project_root
264
+
265
+ project, graph = _load_graph(args)
266
+ path = Path(args.file)
267
+ root = find_project_root(Path(args.path))
268
+ try:
269
+ header = check_file.read_header(path)
270
+ table, how = check_file.bind(path.name, args.source, check_file.file_map(root))
271
+ if table is None:
272
+ lines = check_file.no_binding_lines(path.name, graph)
273
+ print(
274
+ json.dumps({"file": path.name, "exit": 2, "lines": lines}, indent=2)
275
+ if args.json
276
+ else "\n".join(lines)
277
+ )
278
+ sys.exit(2)
279
+ report = check_file.check(graph, table, header, path.name, how, noun_for(project.mode))
280
+ except check_file.CannotCheck as e:
281
+ print(
282
+ json.dumps({"file": path.name, "exit": 2, "lines": [f"Cannot check: {e}"]}, indent=2)
283
+ if args.json
284
+ else f"Cannot check: {e}"
285
+ )
286
+ sys.exit(2)
287
+ except OSError as e:
288
+ print(f"Cannot check: {e}")
289
+ sys.exit(2)
290
+ if args.json:
291
+ print(json.dumps(report, indent=2))
292
+ else:
293
+ print("\n".join(report["lines"]))
294
+ if report["exit"]:
295
+ sys.exit(report["exit"])
296
+
297
+
261
298
  def cmd_trace(args) -> None:
262
299
  project, graph = _load_graph(args)
263
300
  model, column = _parse_target(args.target)
@@ -604,6 +641,15 @@ def main(argv: list[str] | None = None) -> None:
604
641
  )
605
642
  p_columns.add_argument("model")
606
643
 
644
+ p_check = sub.add_parser(
645
+ "check-file",
646
+ help="a file from an external team: what breaks if it is loaded",
647
+ parents=[common],
648
+ )
649
+ p_check.add_argument("file", help="the csv, tsv, xlsx or parquet file that arrived")
650
+ p_check.add_argument(
651
+ "--source", help="the table it feeds (raw.vendor_feed); else the project's files map"
652
+ )
607
653
  p_breaks = sub.add_parser("breaks", help="what breaks if this column changes", parents=[common])
608
654
  p_breaks.add_argument("target", help="model.column")
609
655
  p_breaks.add_argument(
@@ -716,6 +762,7 @@ def main(argv: list[str] | None = None) -> None:
716
762
  "models": cmd_models,
717
763
  "columns": cmd_columns,
718
764
  "breaks": cmd_breaks,
765
+ "check-file": cmd_check_file,
719
766
  "trace": cmd_trace,
720
767
  "graph": cmd_graph,
721
768
  "ci": cmd_ci,
@@ -339,3 +339,47 @@ def _dbt_owned_dirs(project_dir: Path) -> list[Path]:
339
339
  owned += _dbt_model_dirs(project_dir)
340
340
  owned += [project_dir / name for name in _macro_dir_names(project_dir / "dbt_project.yml")]
341
341
  return [d for d in owned if d.is_dir()]
342
+
343
+
344
+ def source_tables(root: Path):
345
+ """Every (source name, table dict) a dbt project declares in its yml.
346
+
347
+ Walks the model paths only (dbt_project.yml's model-paths), so a
348
+ package's or a build folder's yml never counts as this project's."""
349
+ import yaml
350
+
351
+ root = Path(root)
352
+ if not (root / "dbt_project.yml").is_file():
353
+ return
354
+ for model_dir in _dbt_model_dirs(root):
355
+ for yml in [*files_under(model_dir, "*.yml"), *files_under(model_dir, "*.yaml")]:
356
+ try:
357
+ parsed = yaml.safe_load(yml.read_text(errors="replace", encoding="utf-8"))
358
+ except yaml.YAMLError:
359
+ continue
360
+ if not isinstance(parsed, dict):
361
+ continue
362
+ for source in parsed.get("sources") or []:
363
+ if not isinstance(source, dict) or not source.get("name"):
364
+ continue
365
+ for table in source.get("tables") or []:
366
+ if isinstance(table, dict) and table.get("name"):
367
+ yield str(source["name"]), table
368
+
369
+
370
+ def source_columns(root: Path) -> dict[str, list[str]]:
371
+ """The columns sources.yml declares, as {"schema.table": [names]}.
372
+
373
+ dbt users list a source's columns in yml for docs and tests; those are
374
+ the landing table's contract, and the graph read only ingested schemas
375
+ and seeds before (a dbt source always showed "not declared")."""
376
+ out: dict[str, list[str]] = {}
377
+ for source, table in source_tables(root):
378
+ names = []
379
+ for column in table.get("columns") or []:
380
+ name = column.get("name") if isinstance(column, dict) else column
381
+ if name:
382
+ names.append(str(name))
383
+ if names:
384
+ out[f"{source}.{table['name']}"] = names
385
+ return out
@@ -3,6 +3,7 @@
3
3
  from __future__ import annotations
4
4
 
5
5
  import re
6
+ from pathlib import Path
6
7
 
7
8
  from ripple.loaders.types import Model, Project
8
9
 
@@ -93,14 +94,22 @@ def _apply_ingested_schemas(project: Project, extra_roots: tuple = ()) -> None:
93
94
  a new source with declared columns. A model the project actually derives
94
95
  keeps its own analysis: ingested schemas fill gaps, they never override.
95
96
  """
97
+ from ripple.loaders.dbt_config import source_columns
96
98
  from ripple.schemas import load_schemas
97
99
 
98
100
  # the CLI and MCP write .ripple/schemas.json at the repo root they were
99
101
  # run from; a dbt project in a subfolder (balboa's transform/) has its
100
- # own root, and reading only there made every ingest a silent no-op
102
+ # own root, and reading only there made every ingest a silent no-op.
103
+ # sources.yml columns come first; an ingest adds to them
101
104
  tables: dict = {}
102
105
  for root in dict.fromkeys([project.root, *extra_roots]):
103
- tables.update(load_schemas(root) or {})
106
+ tables.update(source_columns(Path(root)))
107
+ for root in dict.fromkeys([project.root, *extra_roots]):
108
+ for name, columns in (load_schemas(root) or {}).items():
109
+ tables[name] = [
110
+ *tables.get(name, []),
111
+ *[c for c in columns if c not in tables.get(name, [])],
112
+ ]
104
113
  if not tables:
105
114
  return
106
115
  # every relation a spelling names, not the first: a monorepo declares the
File without changes
File without changes
File without changes