graincheck 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graincheck/__init__.py +5 -0
- graincheck/__main__.py +299 -0
- graincheck/_deps.py +45 -0
- graincheck/coverage.py +438 -0
- graincheck/emit.py +83 -0
- graincheck/exceptions.py +167 -0
- graincheck/explain.py +403 -0
- graincheck/history.py +118 -0
- graincheck/incumbent.py +151 -0
- graincheck/limits.py +156 -0
- graincheck/project.py +1217 -0
- graincheck/remediation.py +255 -0
- graincheck/render.py +579 -0
- graincheck/report.py +433 -0
- graincheck/rules.py +2620 -0
- graincheck/scanner.py +1035 -0
- graincheck/surrogate.py +308 -0
- graincheck-0.1.0.dist-info/METADATA +838 -0
- graincheck-0.1.0.dist-info/RECORD +23 -0
- graincheck-0.1.0.dist-info/WHEEL +5 -0
- graincheck-0.1.0.dist-info/entry_points.txt +2 -0
- graincheck-0.1.0.dist-info/licenses/LICENSE +97 -0
- graincheck-0.1.0.dist-info/top_level.txt +1 -0
graincheck/__init__.py
ADDED
graincheck/__main__.py
ADDED
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
"""graincheck CLI.
|
|
2
|
+
|
|
3
|
+
graincheck scan ./models --manifest target/manifest.json --dialect snowflake --report out.html
|
|
4
|
+
graincheck explain fct_customer_revenue --path ./models --manifest target/manifest.json
|
|
5
|
+
"""
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from .scanner import (scan_path, to_markdown, to_report, compile_advice,
|
|
13
|
+
dbt_project_root, load_uniques)
|
|
14
|
+
from .explain import explain
|
|
15
|
+
from .coverage import measure_full, to_text as coverage_text
|
|
16
|
+
from .coverage import FORMULAS as COVERAGE_FORMULAS
|
|
17
|
+
from .exceptions import to_text as exceptions_text
|
|
18
|
+
from . import emit as emit_tests, history as cov_history, incumbent
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _auto_manifest(path: Path, given: str | None) -> Path | None:
|
|
22
|
+
if given:
|
|
23
|
+
return Path(given)
|
|
24
|
+
root = dbt_project_root(path)
|
|
25
|
+
if root is None:
|
|
26
|
+
return None
|
|
27
|
+
candidate = root / "target" / "manifest.json"
|
|
28
|
+
return candidate if candidate.exists() else None
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _make_output_utf8_safe():
|
|
32
|
+
"""Stop a legacy console codepage from crashing the whole run.
|
|
33
|
+
|
|
34
|
+
Every message this tool prints contains an em dash, and several contain arrows and middle dots.
|
|
35
|
+
On Windows, `sys.stdout` uses the locale encoding, and the default codepage of `cmd.exe` in the
|
|
36
|
+
US locale is **cp437**, which has no em dash. So:
|
|
37
|
+
|
|
38
|
+
C:\\> graincheck coverage .
|
|
39
|
+
UnicodeEncodeError: 'charmap' codec can't encode character '\\u2014' ...
|
|
40
|
+
|
|
41
|
+
— a traceback and exit 1, on a correct project, before printing a single finding. Reproduced here
|
|
42
|
+
with PYTHONIOENCODING=cp437 and PYTHONIOENCODING=ascii; both crashed at the same character.
|
|
43
|
+
|
|
44
|
+
This is the one class of failure that no amount of checking the SQL would ever have found, because
|
|
45
|
+
every test and every health check has run on Linux with a UTF-8 locale. It is also the one the
|
|
46
|
+
user would have hit first.
|
|
47
|
+
|
|
48
|
+
UTF-8 is the fix where the stream supports it. Where it does not, `errors="replace"` means a
|
|
49
|
+
console that cannot render an em dash prints `?` instead of killing the process — a cosmetic loss
|
|
50
|
+
is always the right trade against losing the output entirely. Wrapped in try/except because a
|
|
51
|
+
stream may be an object with no `reconfigure` (pytest's capture, a pipe someone replaced), and
|
|
52
|
+
failing to improve the console must never itself be the crash.
|
|
53
|
+
"""
|
|
54
|
+
for stream in (sys.stdout, sys.stderr):
|
|
55
|
+
try:
|
|
56
|
+
stream.reconfigure(encoding="utf-8", errors="replace")
|
|
57
|
+
except (AttributeError, ValueError, OSError):
|
|
58
|
+
try:
|
|
59
|
+
stream.reconfigure(errors="replace")
|
|
60
|
+
except (AttributeError, ValueError, OSError):
|
|
61
|
+
pass
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _write_output(path_str: str, text: str, what: str) -> bool:
|
|
65
|
+
"""Write an output file, or explain why not — never a traceback.
|
|
66
|
+
|
|
67
|
+
A user typing `--report reports/scan.html` before `reports/` exists got a raw
|
|
68
|
+
FileNotFoundError traceback after the scan had already finished, which loses the whole run's work
|
|
69
|
+
to a one-word mistake. The parent directory is created when it can be, and every remaining failure
|
|
70
|
+
(a permission problem, a path that is a directory, a full disk) becomes one sentence on stderr and
|
|
71
|
+
a nonzero exit.
|
|
72
|
+
"""
|
|
73
|
+
p = Path(path_str)
|
|
74
|
+
try:
|
|
75
|
+
if p.parent and not p.parent.exists():
|
|
76
|
+
p.parent.mkdir(parents=True, exist_ok=True)
|
|
77
|
+
print(f" created {p.parent}{os.sep}", file=sys.stderr)
|
|
78
|
+
p.write_text(text, encoding="utf-8")
|
|
79
|
+
except OSError as e:
|
|
80
|
+
print(f"could not write the {what} to {path_str}: {e.strerror or e}. "
|
|
81
|
+
f"The scan itself completed; only the file write failed.", file=sys.stderr)
|
|
82
|
+
return False
|
|
83
|
+
print(f"wrote {path_str}")
|
|
84
|
+
return True
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _validate_dialect(name):
|
|
88
|
+
"""Reject an unknown dialect before any work happens. Returns an error string, or "".
|
|
89
|
+
|
|
90
|
+
This is the most dangerous bug found in the whole stress pass, and it passed every check that
|
|
91
|
+
existed. `--dialect postgress` — one letter — does not raise. sqlglot is asked for the dialect
|
|
92
|
+
per file, every parse fails, every model lands in "could not be parsed", and the run ends:
|
|
93
|
+
|
|
94
|
+
Scanned **30 models**. Found **0** issues to review: 0 high, 0 medium, 0 low.
|
|
95
|
+
exit 0
|
|
96
|
+
|
|
97
|
+
A CI job with `--fail-on high` goes green on a project that was never read. That is precisely the
|
|
98
|
+
silent pass this tool exists to argue against, produced by the tool itself, and it is far worse
|
|
99
|
+
than any wrong finding: a wrong finding gets argued with, and this gets trusted.
|
|
100
|
+
|
|
101
|
+
sqlglot's own error even carries the suggestion, so there is no reason not to surface it.
|
|
102
|
+
"""
|
|
103
|
+
if not name:
|
|
104
|
+
return ""
|
|
105
|
+
try:
|
|
106
|
+
import sqlglot
|
|
107
|
+
sqlglot.parse_one("SELECT 1", read=name)
|
|
108
|
+
return ""
|
|
109
|
+
except Exception as e:
|
|
110
|
+
msg = str(e)
|
|
111
|
+
if "Unknown dialect" not in msg:
|
|
112
|
+
return "" # a parse problem with a valid dialect is not our business here
|
|
113
|
+
try:
|
|
114
|
+
from sqlglot.dialects import Dialects
|
|
115
|
+
known = sorted(d.value for d in Dialects if d.value)
|
|
116
|
+
except Exception: # pragma: no cover
|
|
117
|
+
known = []
|
|
118
|
+
hint = ""
|
|
119
|
+
if known:
|
|
120
|
+
close = [k for k in known if k.startswith(name[:3].lower()) or name.lower().startswith(k)]
|
|
121
|
+
if close:
|
|
122
|
+
hint = f" Did you mean {' or '.join(close)}?"
|
|
123
|
+
return (f"unknown dialect '{name}'.{hint}\n"
|
|
124
|
+
+ (f" known dialects: {', '.join(known)}\n" if known else "")
|
|
125
|
+
+ " Refusing to continue: an unknown dialect makes every model fail to parse, and the "
|
|
126
|
+
"run would report zero findings on a project it never read.")
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def main(argv=None):
|
|
130
|
+
"""CLI entry point, and the only place that turns an interrupt into an exit code.
|
|
131
|
+
|
|
132
|
+
A scan of a 2,000-model repository takes about ninety seconds, so pressing Ctrl-C is not an edge
|
|
133
|
+
case — it is something every user does, probably on their first run while they work out which
|
|
134
|
+
directory they meant. Before this wrapper they got a twenty-line traceback ending in
|
|
135
|
+
`KeyboardInterrupt`, which reads exactly like a crash.
|
|
136
|
+
|
|
137
|
+
128 + SIGINT (130) is the shell convention for "interrupted", and it is what `grep`, `curl` and
|
|
138
|
+
`git` return. A broken pipe gets the same treatment: `graincheck scan . | head -20` closes the
|
|
139
|
+
pipe under a still-printing process, and that must not look like a defect either.
|
|
140
|
+
|
|
141
|
+
Returns the process exit code: 0 normally, 1 when --fail-on is met or an output file could not be
|
|
142
|
+
written, 130 on Ctrl-C.
|
|
143
|
+
"""
|
|
144
|
+
try:
|
|
145
|
+
return _main(argv)
|
|
146
|
+
except KeyboardInterrupt:
|
|
147
|
+
print("\ninterrupted.", file=sys.stderr)
|
|
148
|
+
return 130
|
|
149
|
+
except BrokenPipeError:
|
|
150
|
+
# The reader went away (`| head`). Close stdout before the interpreter tries to flush it,
|
|
151
|
+
# or Python prints its own "Exception ignored" noise at shutdown.
|
|
152
|
+
try:
|
|
153
|
+
sys.stdout.close()
|
|
154
|
+
except Exception:
|
|
155
|
+
pass
|
|
156
|
+
return 141
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _main(argv=None):
|
|
160
|
+
_make_output_utf8_safe()
|
|
161
|
+
p = argparse.ArgumentParser(
|
|
162
|
+
prog="graincheck",
|
|
163
|
+
description="Catch metric logic errors before they reach production. Deterministic analytical-SQL integrity: grain, joins and aggregation.")
|
|
164
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
165
|
+
|
|
166
|
+
s = sub.add_parser("scan", help="scan a directory of .sql files")
|
|
167
|
+
s.add_argument("path")
|
|
168
|
+
s.add_argument("--dialect", default=None, help="snowflake | bigquery | databricks | postgres | ...")
|
|
169
|
+
s.add_argument("--manifest", default=None,
|
|
170
|
+
help="path to dbt target/manifest.json; auto-detected inside a dbt project")
|
|
171
|
+
s.add_argument("--json", dest="json_out", default=None)
|
|
172
|
+
s.add_argument("--report", default=None, help="client-facing report (.md or .html by extension)")
|
|
173
|
+
s.add_argument("--name", default=None, help="project name shown in the report header")
|
|
174
|
+
s.add_argument("--raw", action="store_true", help="terse developer listing instead of the report")
|
|
175
|
+
s.add_argument("--fail-on", default=None, choices=["high", "medium", "low"],
|
|
176
|
+
help="exit 1 if any finding at or above this severity")
|
|
177
|
+
s.add_argument("--no-coverage", action="store_true",
|
|
178
|
+
help="skip the join-key coverage measurement")
|
|
179
|
+
|
|
180
|
+
c = sub.add_parser("coverage",
|
|
181
|
+
help="how many of this project's join keys are proven unique by a test")
|
|
182
|
+
c.add_argument("path")
|
|
183
|
+
c.add_argument("--dialect", default=None)
|
|
184
|
+
c.add_argument("--manifest", default=None)
|
|
185
|
+
c.add_argument("--json", dest="json_out", default=None)
|
|
186
|
+
c.add_argument("--emit-tests", dest="emit_tests", default=None,
|
|
187
|
+
help="write a ready-to-paste schema.yml proving every unproven join key")
|
|
188
|
+
c.add_argument("--history", default=None,
|
|
189
|
+
help="append this run to a JSON Lines file and show what moved since last time")
|
|
190
|
+
|
|
191
|
+
x = sub.add_parser("explain", help="walk through one model: grain, join path, risk, remediation")
|
|
192
|
+
x.add_argument("model", help="model name (file stem) or a path to a .sql file")
|
|
193
|
+
x.add_argument("--path", default=".", help="directory to search for the model (default: .)")
|
|
194
|
+
x.add_argument("--dialect", default=None)
|
|
195
|
+
x.add_argument("--manifest", default=None)
|
|
196
|
+
|
|
197
|
+
a = p.parse_args(argv)
|
|
198
|
+
|
|
199
|
+
problem = _validate_dialect(getattr(a, "dialect", None))
|
|
200
|
+
if problem:
|
|
201
|
+
print(problem, file=sys.stderr)
|
|
202
|
+
return 2
|
|
203
|
+
|
|
204
|
+
if a.cmd == "coverage":
|
|
205
|
+
root = Path(a.path)
|
|
206
|
+
rows, paths, summary = measure_full(root, a.dialect, _auto_manifest(root, a.manifest))
|
|
207
|
+
print(coverage_text(rows, summary, paths=paths))
|
|
208
|
+
|
|
209
|
+
project = dbt_project_root(root) or root
|
|
210
|
+
badges = incumbent.primary_key_tested(project)
|
|
211
|
+
fc = incumbent.find(rows, project)
|
|
212
|
+
fc_text = incumbent.to_text(fc, len(badges))
|
|
213
|
+
if fc_text:
|
|
214
|
+
print("\n" + fc_text)
|
|
215
|
+
|
|
216
|
+
if a.history:
|
|
217
|
+
cov_history.append(Path(a.history), summary, project)
|
|
218
|
+
print("\n" + cov_history.to_text(cov_history.read(Path(a.history))))
|
|
219
|
+
|
|
220
|
+
if a.emit_tests:
|
|
221
|
+
n = emit_tests.write(rows, Path(a.emit_tests))
|
|
222
|
+
print(f"\nwrote {a.emit_tests} — {n} relation(s) with a test that would prove the key")
|
|
223
|
+
|
|
224
|
+
if a.json_out and not _write_output(a.json_out, json.dumps(
|
|
225
|
+
{"summary": summary, "formulas": COVERAGE_FORMULAS,
|
|
226
|
+
"joins": [r.to_dict() for r in rows],
|
|
227
|
+
"metric_paths": [p.to_dict() for p in paths],
|
|
228
|
+
"false_comfort": [x.to_dict() for x in fc]}, indent=2), "JSON output"):
|
|
229
|
+
return 1
|
|
230
|
+
|
|
231
|
+
return 0
|
|
232
|
+
|
|
233
|
+
if a.cmd == "explain":
|
|
234
|
+
root = Path(a.path)
|
|
235
|
+
print(explain(root, a.model, a.dialect, _auto_manifest(root, a.manifest)))
|
|
236
|
+
return 0
|
|
237
|
+
|
|
238
|
+
path = Path(a.path)
|
|
239
|
+
manifest = _auto_manifest(path, a.manifest)
|
|
240
|
+
findings, errors, n, uniques = scan_path(path, a.dialect, manifest)
|
|
241
|
+
|
|
242
|
+
advice = compile_advice(path, n, len(errors))
|
|
243
|
+
if advice:
|
|
244
|
+
print(f"\n NOTE: {advice}\n", file=sys.stderr)
|
|
245
|
+
if manifest:
|
|
246
|
+
bad = load_uniques(manifest).get("__unreadable_manifest__")
|
|
247
|
+
if bad:
|
|
248
|
+
print(f" WARNING: the manifest could not be read — {next(iter(next(iter(bad))))}\n"
|
|
249
|
+
" Continuing without it. Declared uniqueness tests are therefore unknown, so "
|
|
250
|
+
"fan-out findings will say the grain could not be verified rather than being "
|
|
251
|
+
"silently downgraded.", file=sys.stderr)
|
|
252
|
+
manifest = None
|
|
253
|
+
if manifest and not a.manifest:
|
|
254
|
+
print(f" using manifest: {manifest}", file=sys.stderr)
|
|
255
|
+
elif not manifest:
|
|
256
|
+
print(" no dbt manifest found — join grain cannot be verified against declared uniqueness "
|
|
257
|
+
"tests. Fan-out findings will say so.", file=sys.stderr)
|
|
258
|
+
|
|
259
|
+
cov_summary = None
|
|
260
|
+
if not a.no_coverage:
|
|
261
|
+
_, _, cov_summary = measure_full(path, a.dialect, manifest)
|
|
262
|
+
|
|
263
|
+
name = a.name or path.resolve().name
|
|
264
|
+
if a.raw:
|
|
265
|
+
report = to_markdown(findings, errors, n, uniques)
|
|
266
|
+
else:
|
|
267
|
+
fmt = "html" if (a.report or "").lower().endswith((".html", ".htm")) else "md"
|
|
268
|
+
report = to_report(findings, errors, n, uniques, name, fmt, cov_summary)
|
|
269
|
+
write_failed = False
|
|
270
|
+
if a.report:
|
|
271
|
+
write_failed |= not _write_output(a.report, report, "report")
|
|
272
|
+
else:
|
|
273
|
+
print(report)
|
|
274
|
+
ex_text = exceptions_text(getattr(findings, "exceptions", None),
|
|
275
|
+
getattr(findings, "suppressed", []))
|
|
276
|
+
if ex_text:
|
|
277
|
+
print("\n" + ex_text)
|
|
278
|
+
|
|
279
|
+
if a.json_out:
|
|
280
|
+
write_failed |= not _write_output(a.json_out, json.dumps(
|
|
281
|
+
{"findings": [f.to_dict() for f in findings],
|
|
282
|
+
"reviewed_exceptions": [f.to_dict()
|
|
283
|
+
for f in getattr(findings, "suppressed", [])],
|
|
284
|
+
"errors": errors,
|
|
285
|
+
"models_scanned": n, "models_parsed": n - len(errors),
|
|
286
|
+
"join_key_coverage": cov_summary,
|
|
287
|
+
"compile_advice": advice}, indent=2), "JSON output")
|
|
288
|
+
|
|
289
|
+
if write_failed:
|
|
290
|
+
return 1
|
|
291
|
+
if a.fail_on:
|
|
292
|
+
order = {"high": 0, "medium": 1, "low": 2}
|
|
293
|
+
if any(order.get(f.severity, 9) <= order[a.fail_on] for f in findings):
|
|
294
|
+
return 1
|
|
295
|
+
return 0
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
if __name__ == "__main__":
|
|
299
|
+
sys.exit(main())
|
graincheck/_deps.py
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""One warning, about one missing dependency, in a module that imports nothing.
|
|
2
|
+
|
|
3
|
+
PyYAML is not optional. `schema.yml` is graincheck's only source of declared uniqueness when a project
|
|
4
|
+
has not been compiled — which is the case graincheck exists for, since not making the user run
|
|
5
|
+
`dbt compile` first is half the pitch. Three modules import yaml to read those files.
|
|
6
|
+
|
|
7
|
+
It was missing from `pyproject.toml`'s dependency list, and each ImportError was swallowed with
|
|
8
|
+
`return {}`. So a clean `pip install graincheck` produced a tool that read no schema files and said
|
|
9
|
+
nothing about it. That is not a reduced answer, it is a DIFFERENT one:
|
|
10
|
+
|
|
11
|
+
four-line project, schema.yml declares customer_id unique
|
|
12
|
+
with PyYAML: 0 findings
|
|
13
|
+
without PyYAML: 1 finding, JOIN_FANOUT, high impact
|
|
14
|
+
|
|
15
|
+
A correct join reported as a fan-out, at normal confidence, with no indication that the evidence
|
|
16
|
+
proving it safe was never opened. Of the six defects found in the packaging pass this was the only one
|
|
17
|
+
that produced a confident wrong answer rather than a crash, which makes it the worst of them.
|
|
18
|
+
|
|
19
|
+
The dependency is declared now. This module is the belt to that braces: if PyYAML is ever absent
|
|
20
|
+
anyway — an old install, a stripped container, a conda environment assembled by hand — the user is
|
|
21
|
+
told, once per thing it cost, on stderr, in terms of what it does to the findings.
|
|
22
|
+
|
|
23
|
+
It lives in its own module because `incumbent` and `exceptions` both need it and both are imported by
|
|
24
|
+
`scanner`; hanging it off `scanner` created a circular import.
|
|
25
|
+
"""
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import sys
|
|
29
|
+
|
|
30
|
+
_WARNED: set[str] = set()
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def warn_no_yaml(what: str) -> None:
|
|
34
|
+
"""Report that PyYAML is missing and what it cost. Idempotent per `what`."""
|
|
35
|
+
if what in _WARNED:
|
|
36
|
+
return
|
|
37
|
+
_WARNED.add(what)
|
|
38
|
+
print(f" WARNING: PyYAML is not installed, so {what} were not read. Findings will be WRONG, "
|
|
39
|
+
f"not merely fewer — a join your schema.yml proves safe will be reported as a fan-out. "
|
|
40
|
+
f"Fix with: pip install pyyaml", file=sys.stderr)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def reset_for_tests() -> None:
|
|
44
|
+
"""Forget what has been warned about, so a test can assert the warning appears."""
|
|
45
|
+
_WARNED.clear()
|