graincheck 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
graincheck/__init__.py ADDED
@@ -0,0 +1,5 @@
1
+ """graincheck — find analytical SQL that returns a plausible number and a wrong one."""
2
+ from .scanner import scan_sql, scan_path, to_markdown, to_report, load_uniques # noqa: F401
3
+ from .rules import Finding, ALL_RULES # noqa: F401
4
+
5
+ __version__ = "0.1.0"
graincheck/__main__.py ADDED
@@ -0,0 +1,299 @@
1
+ """graincheck CLI.
2
+
3
+ graincheck scan ./models --manifest target/manifest.json --dialect snowflake --report out.html
4
+ graincheck explain fct_customer_revenue --path ./models --manifest target/manifest.json
5
+ """
6
+ import argparse
7
+ import json
8
+ import os
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ from .scanner import (scan_path, to_markdown, to_report, compile_advice,
13
+ dbt_project_root, load_uniques)
14
+ from .explain import explain
15
+ from .coverage import measure_full, to_text as coverage_text
16
+ from .coverage import FORMULAS as COVERAGE_FORMULAS
17
+ from .exceptions import to_text as exceptions_text
18
+ from . import emit as emit_tests, history as cov_history, incumbent
19
+
20
+
21
+ def _auto_manifest(path: Path, given: str | None) -> Path | None:
22
+ if given:
23
+ return Path(given)
24
+ root = dbt_project_root(path)
25
+ if root is None:
26
+ return None
27
+ candidate = root / "target" / "manifest.json"
28
+ return candidate if candidate.exists() else None
29
+
30
+
31
+ def _make_output_utf8_safe():
32
+ """Stop a legacy console codepage from crashing the whole run.
33
+
34
+ Every message this tool prints contains an em dash, and several contain arrows and middle dots.
35
+ On Windows, `sys.stdout` uses the locale encoding, and the default codepage of `cmd.exe` in the
36
+ US locale is **cp437**, which has no em dash. So:
37
+
38
+ C:\\> graincheck coverage .
39
+ UnicodeEncodeError: 'charmap' codec can't encode character '\\u2014' ...
40
+
41
+ — a traceback and exit 1, on a correct project, before printing a single finding. Reproduced here
42
+ with PYTHONIOENCODING=cp437 and PYTHONIOENCODING=ascii; both crashed at the same character.
43
+
44
+ This is the one class of failure that no amount of checking the SQL would ever have found, because
45
+ every test and every health check has run on Linux with a UTF-8 locale. It is also the one the
46
+ user would have hit first.
47
+
48
+ UTF-8 is the fix where the stream supports it. Where it does not, `errors="replace"` means a
49
+ console that cannot render an em dash prints `?` instead of killing the process — a cosmetic loss
50
+ is always the right trade against losing the output entirely. Wrapped in try/except because a
51
+ stream may be an object with no `reconfigure` (pytest's capture, a pipe someone replaced), and
52
+ failing to improve the console must never itself be the crash.
53
+ """
54
+ for stream in (sys.stdout, sys.stderr):
55
+ try:
56
+ stream.reconfigure(encoding="utf-8", errors="replace")
57
+ except (AttributeError, ValueError, OSError):
58
+ try:
59
+ stream.reconfigure(errors="replace")
60
+ except (AttributeError, ValueError, OSError):
61
+ pass
62
+
63
+
64
+ def _write_output(path_str: str, text: str, what: str) -> bool:
65
+ """Write an output file, or explain why not — never a traceback.
66
+
67
+ A user typing `--report reports/scan.html` before `reports/` exists got a raw
68
+ FileNotFoundError traceback after the scan had already finished, which loses the whole run's work
69
+ to a one-word mistake. The parent directory is created when it can be, and every remaining failure
70
+ (a permission problem, a path that is a directory, a full disk) becomes one sentence on stderr and
71
+ a nonzero exit.
72
+ """
73
+ p = Path(path_str)
74
+ try:
75
+ if p.parent and not p.parent.exists():
76
+ p.parent.mkdir(parents=True, exist_ok=True)
77
+ print(f" created {p.parent}{os.sep}", file=sys.stderr)
78
+ p.write_text(text, encoding="utf-8")
79
+ except OSError as e:
80
+ print(f"could not write the {what} to {path_str}: {e.strerror or e}. "
81
+ f"The scan itself completed; only the file write failed.", file=sys.stderr)
82
+ return False
83
+ print(f"wrote {path_str}")
84
+ return True
85
+
86
+
87
+ def _validate_dialect(name):
88
+ """Reject an unknown dialect before any work happens. Returns an error string, or "".
89
+
90
+ This is the most dangerous bug found in the whole stress pass, and it passed every check that
91
+ existed. `--dialect postgress` — one letter — does not raise. sqlglot is asked for the dialect
92
+ per file, every parse fails, every model lands in "could not be parsed", and the run ends:
93
+
94
+ Scanned **30 models**. Found **0** issues to review: 0 high, 0 medium, 0 low.
95
+ exit 0
96
+
97
+ A CI job with `--fail-on high` goes green on a project that was never read. That is precisely the
98
+ silent pass this tool exists to argue against, produced by the tool itself, and it is far worse
99
+ than any wrong finding: a wrong finding gets argued with, and this gets trusted.
100
+
101
+ sqlglot's own error even carries the suggestion, so there is no reason not to surface it.
102
+ """
103
+ if not name:
104
+ return ""
105
+ try:
106
+ import sqlglot
107
+ sqlglot.parse_one("SELECT 1", read=name)
108
+ return ""
109
+ except Exception as e:
110
+ msg = str(e)
111
+ if "Unknown dialect" not in msg:
112
+ return "" # a parse problem with a valid dialect is not our business here
113
+ try:
114
+ from sqlglot.dialects import Dialects
115
+ known = sorted(d.value for d in Dialects if d.value)
116
+ except Exception: # pragma: no cover
117
+ known = []
118
+ hint = ""
119
+ if known:
120
+ close = [k for k in known if k.startswith(name[:3].lower()) or name.lower().startswith(k)]
121
+ if close:
122
+ hint = f" Did you mean {' or '.join(close)}?"
123
+ return (f"unknown dialect '{name}'.{hint}\n"
124
+ + (f" known dialects: {', '.join(known)}\n" if known else "")
125
+ + " Refusing to continue: an unknown dialect makes every model fail to parse, and the "
126
+ "run would report zero findings on a project it never read.")
127
+
128
+
129
+ def main(argv=None):
130
+ """CLI entry point, and the only place that turns an interrupt into an exit code.
131
+
132
+ A scan of a 2,000-model repository takes about ninety seconds, so pressing Ctrl-C is not an edge
133
+ case — it is something every user does, probably on their first run while they work out which
134
+ directory they meant. Before this wrapper they got a twenty-line traceback ending in
135
+ `KeyboardInterrupt`, which reads exactly like a crash.
136
+
137
+ 128 + SIGINT (130) is the shell convention for "interrupted", and it is what `grep`, `curl` and
138
+ `git` return. A broken pipe gets the same treatment: `graincheck scan . | head -20` closes the
139
+ pipe under a still-printing process, and that must not look like a defect either.
140
+
141
+ Returns the process exit code: 0 normally, 1 when --fail-on is met or an output file could not be
142
+ written, 130 on Ctrl-C.
143
+ """
144
+ try:
145
+ return _main(argv)
146
+ except KeyboardInterrupt:
147
+ print("\ninterrupted.", file=sys.stderr)
148
+ return 130
149
+ except BrokenPipeError:
150
+ # The reader went away (`| head`). Close stdout before the interpreter tries to flush it,
151
+ # or Python prints its own "Exception ignored" noise at shutdown.
152
+ try:
153
+ sys.stdout.close()
154
+ except Exception:
155
+ pass
156
+ return 141
157
+
158
+
159
+ def _main(argv=None):
160
+ _make_output_utf8_safe()
161
+ p = argparse.ArgumentParser(
162
+ prog="graincheck",
163
+ description="Catch metric logic errors before they reach production. Deterministic analytical-SQL integrity: grain, joins and aggregation.")
164
+ sub = p.add_subparsers(dest="cmd", required=True)
165
+
166
+ s = sub.add_parser("scan", help="scan a directory of .sql files")
167
+ s.add_argument("path")
168
+ s.add_argument("--dialect", default=None, help="snowflake | bigquery | databricks | postgres | ...")
169
+ s.add_argument("--manifest", default=None,
170
+ help="path to dbt target/manifest.json; auto-detected inside a dbt project")
171
+ s.add_argument("--json", dest="json_out", default=None)
172
+ s.add_argument("--report", default=None, help="client-facing report (.md or .html by extension)")
173
+ s.add_argument("--name", default=None, help="project name shown in the report header")
174
+ s.add_argument("--raw", action="store_true", help="terse developer listing instead of the report")
175
+ s.add_argument("--fail-on", default=None, choices=["high", "medium", "low"],
176
+ help="exit 1 if any finding at or above this severity")
177
+ s.add_argument("--no-coverage", action="store_true",
178
+ help="skip the join-key coverage measurement")
179
+
180
+ c = sub.add_parser("coverage",
181
+ help="how many of this project's join keys are proven unique by a test")
182
+ c.add_argument("path")
183
+ c.add_argument("--dialect", default=None)
184
+ c.add_argument("--manifest", default=None)
185
+ c.add_argument("--json", dest="json_out", default=None)
186
+ c.add_argument("--emit-tests", dest="emit_tests", default=None,
187
+ help="write a ready-to-paste schema.yml proving every unproven join key")
188
+ c.add_argument("--history", default=None,
189
+ help="append this run to a JSON Lines file and show what moved since last time")
190
+
191
+ x = sub.add_parser("explain", help="walk through one model: grain, join path, risk, remediation")
192
+ x.add_argument("model", help="model name (file stem) or a path to a .sql file")
193
+ x.add_argument("--path", default=".", help="directory to search for the model (default: .)")
194
+ x.add_argument("--dialect", default=None)
195
+ x.add_argument("--manifest", default=None)
196
+
197
+ a = p.parse_args(argv)
198
+
199
+ problem = _validate_dialect(getattr(a, "dialect", None))
200
+ if problem:
201
+ print(problem, file=sys.stderr)
202
+ return 2
203
+
204
+ if a.cmd == "coverage":
205
+ root = Path(a.path)
206
+ rows, paths, summary = measure_full(root, a.dialect, _auto_manifest(root, a.manifest))
207
+ print(coverage_text(rows, summary, paths=paths))
208
+
209
+ project = dbt_project_root(root) or root
210
+ badges = incumbent.primary_key_tested(project)
211
+ fc = incumbent.find(rows, project)
212
+ fc_text = incumbent.to_text(fc, len(badges))
213
+ if fc_text:
214
+ print("\n" + fc_text)
215
+
216
+ if a.history:
217
+ cov_history.append(Path(a.history), summary, project)
218
+ print("\n" + cov_history.to_text(cov_history.read(Path(a.history))))
219
+
220
+ if a.emit_tests:
221
+ n = emit_tests.write(rows, Path(a.emit_tests))
222
+ print(f"\nwrote {a.emit_tests} — {n} relation(s) with a test that would prove the key")
223
+
224
+ if a.json_out and not _write_output(a.json_out, json.dumps(
225
+ {"summary": summary, "formulas": COVERAGE_FORMULAS,
226
+ "joins": [r.to_dict() for r in rows],
227
+ "metric_paths": [p.to_dict() for p in paths],
228
+ "false_comfort": [x.to_dict() for x in fc]}, indent=2), "JSON output"):
229
+ return 1
230
+
231
+ return 0
232
+
233
+ if a.cmd == "explain":
234
+ root = Path(a.path)
235
+ print(explain(root, a.model, a.dialect, _auto_manifest(root, a.manifest)))
236
+ return 0
237
+
238
+ path = Path(a.path)
239
+ manifest = _auto_manifest(path, a.manifest)
240
+ findings, errors, n, uniques = scan_path(path, a.dialect, manifest)
241
+
242
+ advice = compile_advice(path, n, len(errors))
243
+ if advice:
244
+ print(f"\n NOTE: {advice}\n", file=sys.stderr)
245
+ if manifest:
246
+ bad = load_uniques(manifest).get("__unreadable_manifest__")
247
+ if bad:
248
+ print(f" WARNING: the manifest could not be read — {next(iter(next(iter(bad))))}\n"
249
+ " Continuing without it. Declared uniqueness tests are therefore unknown, so "
250
+ "fan-out findings will say the grain could not be verified rather than being "
251
+ "silently downgraded.", file=sys.stderr)
252
+ manifest = None
253
+ if manifest and not a.manifest:
254
+ print(f" using manifest: {manifest}", file=sys.stderr)
255
+ elif not manifest:
256
+ print(" no dbt manifest found — join grain cannot be verified against declared uniqueness "
257
+ "tests. Fan-out findings will say so.", file=sys.stderr)
258
+
259
+ cov_summary = None
260
+ if not a.no_coverage:
261
+ _, _, cov_summary = measure_full(path, a.dialect, manifest)
262
+
263
+ name = a.name or path.resolve().name
264
+ if a.raw:
265
+ report = to_markdown(findings, errors, n, uniques)
266
+ else:
267
+ fmt = "html" if (a.report or "").lower().endswith((".html", ".htm")) else "md"
268
+ report = to_report(findings, errors, n, uniques, name, fmt, cov_summary)
269
+ write_failed = False
270
+ if a.report:
271
+ write_failed |= not _write_output(a.report, report, "report")
272
+ else:
273
+ print(report)
274
+ ex_text = exceptions_text(getattr(findings, "exceptions", None),
275
+ getattr(findings, "suppressed", []))
276
+ if ex_text:
277
+ print("\n" + ex_text)
278
+
279
+ if a.json_out:
280
+ write_failed |= not _write_output(a.json_out, json.dumps(
281
+ {"findings": [f.to_dict() for f in findings],
282
+ "reviewed_exceptions": [f.to_dict()
283
+ for f in getattr(findings, "suppressed", [])],
284
+ "errors": errors,
285
+ "models_scanned": n, "models_parsed": n - len(errors),
286
+ "join_key_coverage": cov_summary,
287
+ "compile_advice": advice}, indent=2), "JSON output")
288
+
289
+ if write_failed:
290
+ return 1
291
+ if a.fail_on:
292
+ order = {"high": 0, "medium": 1, "low": 2}
293
+ if any(order.get(f.severity, 9) <= order[a.fail_on] for f in findings):
294
+ return 1
295
+ return 0
296
+
297
+
298
+ if __name__ == "__main__":
299
+ sys.exit(main())
graincheck/_deps.py ADDED
@@ -0,0 +1,45 @@
1
+ """One warning, about one missing dependency, in a module that imports nothing.
2
+
3
+ PyYAML is not optional. `schema.yml` is graincheck's only source of declared uniqueness when a project
4
+ has not been compiled — which is the case graincheck exists for, since not making the user run
5
+ `dbt compile` first is half the pitch. Three modules import yaml to read those files.
6
+
7
+ It was missing from `pyproject.toml`'s dependency list, and each ImportError was swallowed with
8
+ `return {}`. So a clean `pip install graincheck` produced a tool that read no schema files and said
9
+ nothing about it. That is not a reduced answer, it is a DIFFERENT one:
10
+
11
+ four-line project, schema.yml declares customer_id unique
12
+ with PyYAML: 0 findings
13
+ without PyYAML: 1 finding, JOIN_FANOUT, high impact
14
+
15
+ A correct join reported as a fan-out, at normal confidence, with no indication that the evidence
16
+ proving it safe was never opened. Of the six defects found in the packaging pass this was the only one
17
+ that produced a confident wrong answer rather than a crash, which makes it the worst of them.
18
+
19
+ The dependency is declared now. This module is the belt to that braces: if PyYAML is ever absent
20
+ anyway — an old install, a stripped container, a conda environment assembled by hand — the user is
21
+ told, once per thing it cost, on stderr, in terms of what it does to the findings.
22
+
23
+ It lives in its own module because `incumbent` and `exceptions` both need it and both are imported by
24
+ `scanner`; hanging it off `scanner` created a circular import.
25
+ """
26
+ from __future__ import annotations
27
+
28
+ import sys
29
+
30
+ _WARNED: set[str] = set()
31
+
32
+
33
+ def warn_no_yaml(what: str) -> None:
34
+ """Report that PyYAML is missing and what it cost. Idempotent per `what`."""
35
+ if what in _WARNED:
36
+ return
37
+ _WARNED.add(what)
38
+ print(f" WARNING: PyYAML is not installed, so {what} were not read. Findings will be WRONG, "
39
+ f"not merely fewer — a join your schema.yml proves safe will be reported as a fan-out. "
40
+ f"Fix with: pip install pyyaml", file=sys.stderr)
41
+
42
+
43
+ def reset_for_tests() -> None:
44
+ """Forget what has been warned about, so a test can assert the warning appears."""
45
+ _WARNED.clear()