dbt-sentinel 0.6.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dbt_sentinel-0.6.1/PKG-INFO +195 -0
- dbt_sentinel-0.6.1/README.md +167 -0
- dbt_sentinel-0.6.1/pyproject.toml +56 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/__init__.py +43 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/analyze.py +170 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/cli.py +151 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/context.py +110 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/parse.py +158 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/report.py +122 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/store.py +150 -0
- dbt_sentinel-0.6.1/src/dbt_sentinel/warehouse.py +110 -0
|
@@ -0,0 +1,195 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dbt-sentinel
|
|
3
|
+
Version: 0.6.1
|
|
4
|
+
Summary: AI-grounded root-cause analysis for failing dbt tests
|
|
5
|
+
Keywords: dbt,data-quality,observability,analytics-engineering,llm
|
|
6
|
+
Author: Qamar Raza
|
|
7
|
+
Author-email: Qamar Raza <raza.qamar@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Classifier: Development Status :: 4 - Beta
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Classifier: Topic :: Database
|
|
13
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
14
|
+
Requires-Dist: click==8.1.7
|
|
15
|
+
Requires-Dist: duckdb==1.1.3
|
|
16
|
+
Requires-Dist: httpx==0.28.1
|
|
17
|
+
Requires-Dist: python-dotenv>=1.2.2
|
|
18
|
+
Requires-Dist: rich==13.9.4
|
|
19
|
+
Requires-Dist: google-cloud-bigquery>=3.43.0 ; extra == 'bq'
|
|
20
|
+
Requires-Dist: streamlit>=1.40 ; extra == 'ui'
|
|
21
|
+
Requires-Python: >=3.12
|
|
22
|
+
Project-URL: Homepage, https://github.com/qraza/dbt-sentinel
|
|
23
|
+
Project-URL: Repository, https://github.com/qraza/dbt-sentinel
|
|
24
|
+
Project-URL: Issues, https://github.com/qraza/dbt-sentinel/issues
|
|
25
|
+
Provides-Extra: bq
|
|
26
|
+
Provides-Extra: ui
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
|
|
29
|
+
# dbt-sentinel
|
|
30
|
+
|
|
31
|
+
**dbt tells you a test failed. It doesn't tell you why.** `dbt-sentinel` reads dbt's
|
|
32
|
+
run artifacts after a build, pulls the compiled SQL and a sample of the rows that
|
|
33
|
+
actually failed, and asks an LLM — grounded strictly in that evidence — for a root-cause
|
|
34
|
+
explanation and a concrete fix. It flags low-confidence answers instead of bluffing.
|
|
35
|
+
|
|
36
|
+
[](https://github.com/qraza/dbt-sentinel/actions/workflows/ci.yml)
|
|
37
|
+

|
|
38
|
+
[](LICENSE)
|
|
39
|
+
|
|
40
|
+

|
|
41
|
+
|
|
42
|
+
Runs in CI too — on every pull request it posts its grounded diagnosis as a comment:
|
|
43
|
+
|
|
44
|
+

|
|
45
|
+
|
|
46
|
+
## What this is
|
|
47
|
+
|
|
48
|
+
A data-quality companion for dbt. When `dbt build` reports a failing test, you normally
|
|
49
|
+
get a name and a row count — then you go digging. `dbt-sentinel` closes that gap: it joins
|
|
50
|
+
`run_results.json` and `manifest.json` to find each failure, runs the test's own compiled
|
|
51
|
+
SQL against the warehouse to sample the offending rows, and sends that concrete evidence to
|
|
52
|
+
an LLM with a prompt engineered to stay grounded in it. The output is a structured verdict —
|
|
53
|
+
root cause, suggested fix, confidence, and the evidence behind it — in the terminal and,
|
|
54
|
+
optionally, markdown for a PR comment. It reads what dbt already produced; it never re-runs
|
|
55
|
+
your tests or mutates data (the warehouse is opened read-only).
|
|
56
|
+
|
|
57
|
+
It also tracks failures across runs — each one is flagged **new**, **recurring**, or
|
|
58
|
+
**regressed**, and `sentinel history <test-id>` prints a test's full timeline, so you can
|
|
59
|
+
see whether something just broke or has been broken for weeks.
|
|
60
|
+
|
|
61
|
+
## Quickstart
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
uv sync
|
|
65
|
+
export ANTHROPIC_API_KEY="sk-ant-..."
|
|
66
|
+
export ANTHROPIC_MODEL="claude-opus-5"
|
|
67
|
+
|
|
68
|
+
uv run sentinel analyze \
|
|
69
|
+
--target-dir path/to/dbt/target \
|
|
70
|
+
--db path/to/warehouse.duckdb \
|
|
71
|
+
--markdown report.md
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
If all tests pass, it says so and exits. If not, you get a diagnosis per failure.
|
|
75
|
+
|
|
76
|
+
## How it works
|
|
77
|
+
```
|
|
78
|
+
dbt build ─► run_results.json (which tests failed, row counts)
|
|
79
|
+
manifest.json (compiled SQL, guarded model/column, test type)
|
|
80
|
+
│
|
|
81
|
+
▼
|
|
82
|
+
parse.py join artifacts → FailingTest records
|
|
83
|
+
│
|
|
84
|
+
▼
|
|
85
|
+
context.py run the test's compiled SQL against DuckDB,
|
|
86
|
+
│ sample the offending rows (read-only, capped)
|
|
87
|
+
▼
|
|
88
|
+
analyze.py grounded prompt → LLM → {root_cause, fix,
|
|
89
|
+
│ confidence, evidence}
|
|
90
|
+
▼
|
|
91
|
+
report.py Rich terminal panel + markdown
|
|
92
|
+
(via cli.py: sentinel analyze)
|
|
93
|
+
```
|
|
94
|
+
The design choice that keeps it small and robust: **it reads dbt's artifacts rather than
|
|
95
|
+
re-running tests.** Everything it needs — compiled SQL, failing-row count, guarded model —
|
|
96
|
+
is already in the JSON dbt writes on every build.
|
|
97
|
+
|
|
98
|
+
## Grounding is the whole point
|
|
99
|
+
|
|
100
|
+
An LLM asked "why did my dbt test fail?" with no context invents a plausible-sounding
|
|
101
|
+
story. `dbt-sentinel` feeds the model only real evidence — the compiled SQL, the column
|
|
102
|
+
schema, and a sample of the actual failing rows — and tells it to ground every claim in
|
|
103
|
+
that evidence or report low confidence.
|
|
104
|
+
|
|
105
|
+
The difference on a real failure (a taxi-trip speed test where `avg_speed_mph` was computed
|
|
106
|
+
with a `600` multiplier instead of `60`):
|
|
107
|
+
|
|
108
|
+
**Naive prompt — test name + row count only, no evidence:**
|
|
109
|
+
|
|
110
|
+
> Most likely: division by zero or NULL producing infinite/NULL speed. With 1.8M failing
|
|
111
|
+
> rows this suggests a systemic data issue. Run these diagnostic queries to pinpoint the
|
|
112
|
+
> cause… [three SQL queries checking null/zero durations and negative distances].
|
|
113
|
+
|
|
114
|
+
**Grounded (dbt-sentinel):**
|
|
115
|
+
|
|
116
|
+
> The `avg_speed_mph` formula uses a multiplier of 600 instead of 60. A trip of 9.24 miles
|
|
117
|
+
> in 52 minutes yields (9.24/52)×60 = 10.66 mph (plausible), but the bug computes
|
|
118
|
+
> (9.24/52)×600 = 106.62 — which exactly matches the failing row. The same 10× inflation
|
|
119
|
+
> holds across all sampled rows. **Confidence: high.**
|
|
120
|
+
|
|
121
|
+
The naive answer guesses the wrong cause and hands the work back to you as queries to run.
|
|
122
|
+
The grounded answer runs them, finds the real bug, and proves it against a row. See
|
|
123
|
+
[`docs/example-analysis.md`](docs/example-analysis.md) for the full output.
|
|
124
|
+
|
|
125
|
+
## Warehouses
|
|
126
|
+
|
|
127
|
+
Sampling runs behind a small adapter interface, so the engine is a flag, not a rewrite.
|
|
128
|
+
DuckDB works out of the box; BigQuery needs the optional extra and Application Default
|
|
129
|
+
Credentials.
|
|
130
|
+
|
|
131
|
+
```bash
|
|
132
|
+
# DuckDB (default)
|
|
133
|
+
uv run sentinel analyze --target-dir path/to/target --db warehouse.duckdb
|
|
134
|
+
|
|
135
|
+
# BigQuery
|
|
136
|
+
uv sync --group bq
|
|
137
|
+
gcloud auth application-default login
|
|
138
|
+
uv run sentinel analyze --target-dir path/to/target --bq-project my-project --bq-location EU
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
BigQuery works against the free sandbox — no billing account required. Credentials are
|
|
142
|
+
never handled by dbt-sentinel itself; the client reads them from ADC.
|
|
143
|
+
|
|
144
|
+
## Dashboard
|
|
145
|
+
|
|
146
|
+
A single-page Streamlit view over the recorded history — latest run, per-test failure
|
|
147
|
+
trend, and the stored root cause. It reads dbt-sentinel's own history database, so it
|
|
148
|
+
needs no warehouse connection and no API key.
|
|
149
|
+
|
|
150
|
+
```bash
|
|
151
|
+
uv sync --group ui
|
|
152
|
+
uv run --group ui streamlit run app/dashboard.py
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+

|
|
156
|
+
|
|
157
|
+
## Design decisions
|
|
158
|
+
|
|
159
|
+
**Read artifacts, don't re-run tests.** dbt writes the artifacts on every build; parsing
|
|
160
|
+
them needs no dbt invocation and works against any completed run, including CI.
|
|
161
|
+
|
|
162
|
+
**Sample rows via the test's compiled SQL.** Some models are ephemeral and aren't queryable
|
|
163
|
+
as tables. The test's compiled SQL has that logic inlined, so it returns the offending rows
|
|
164
|
+
regardless of materialization.
|
|
165
|
+
|
|
166
|
+
**Ground, then surface confidence.** The prompt contains only evidence; the model is told to
|
|
167
|
+
say "insufficient evidence" rather than speculate, and confidence is shown, not hidden.
|
|
168
|
+
|
|
169
|
+
**Read-only, always.** The warehouse connection inspects; it never mutates.
|
|
170
|
+
|
|
171
|
+
## Limitations
|
|
172
|
+
|
|
173
|
+
- DuckDB and BigQuery supported; other engines need a new Warehouse adapter.
|
|
174
|
+
- One dbt project per run.
|
|
175
|
+
- Models with extended thinking can spend the whole token budget before emitting text,
|
|
176
|
+
so `max_tokens` is set generously; too low a value yields an unparseable (low-confidence)
|
|
177
|
+
result rather than an answer.
|
|
178
|
+
- Diagnosis quality depends on the model and on how much signal the sampled rows carry;
|
|
179
|
+
low-signal failures correctly return low confidence.
|
|
180
|
+
|
|
181
|
+
## Development
|
|
182
|
+
|
|
183
|
+
```bash
|
|
184
|
+
uv sync --group dev
|
|
185
|
+
uv run ruff check .
|
|
186
|
+
uv run pytest -v
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
CI runs the same lint + tests on every push. Tests use committed fixtures and a mocked API —
|
|
190
|
+
no warehouse, no API key needed.
|
|
191
|
+
|
|
192
|
+
---
|
|
193
|
+
|
|
194
|
+
Built by [Qamar Raza](https://github.com/qraza).
|
|
195
|
+
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
# dbt-sentinel
|
|
2
|
+
|
|
3
|
+
**dbt tells you a test failed. It doesn't tell you why.** `dbt-sentinel` reads dbt's
|
|
4
|
+
run artifacts after a build, pulls the compiled SQL and a sample of the rows that
|
|
5
|
+
actually failed, and asks an LLM — grounded strictly in that evidence — for a root-cause
|
|
6
|
+
explanation and a concrete fix. It flags low-confidence answers instead of bluffing.
|
|
7
|
+
|
|
8
|
+
[](https://github.com/qraza/dbt-sentinel/actions/workflows/ci.yml)
|
|
9
|
+

|
|
10
|
+
[](LICENSE)
|
|
11
|
+
|
|
12
|
+

|
|
13
|
+
|
|
14
|
+
Runs in CI too — on every pull request it posts its grounded diagnosis as a comment:
|
|
15
|
+
|
|
16
|
+

|
|
17
|
+
|
|
18
|
+
## What this is
|
|
19
|
+
|
|
20
|
+
A data-quality companion for dbt. When `dbt build` reports a failing test, you normally
|
|
21
|
+
get a name and a row count — then you go digging. `dbt-sentinel` closes that gap: it joins
|
|
22
|
+
`run_results.json` and `manifest.json` to find each failure, runs the test's own compiled
|
|
23
|
+
SQL against the warehouse to sample the offending rows, and sends that concrete evidence to
|
|
24
|
+
an LLM with a prompt engineered to stay grounded in it. The output is a structured verdict —
|
|
25
|
+
root cause, suggested fix, confidence, and the evidence behind it — in the terminal and,
|
|
26
|
+
optionally, markdown for a PR comment. It reads what dbt already produced; it never re-runs
|
|
27
|
+
your tests or mutates data (the warehouse is opened read-only).
|
|
28
|
+
|
|
29
|
+
It also tracks failures across runs — each one is flagged **new**, **recurring**, or
|
|
30
|
+
**regressed**, and `sentinel history <test-id>` prints a test's full timeline, so you can
|
|
31
|
+
see whether something just broke or has been broken for weeks.
|
|
32
|
+
|
|
33
|
+
## Quickstart
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
uv sync
|
|
37
|
+
export ANTHROPIC_API_KEY="sk-ant-..."
|
|
38
|
+
export ANTHROPIC_MODEL="claude-opus-5"
|
|
39
|
+
|
|
40
|
+
uv run sentinel analyze \
|
|
41
|
+
--target-dir path/to/dbt/target \
|
|
42
|
+
--db path/to/warehouse.duckdb \
|
|
43
|
+
--markdown report.md
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
If all tests pass, it says so and exits. If not, you get a diagnosis per failure.
|
|
47
|
+
|
|
48
|
+
## How it works
|
|
49
|
+
```
|
|
50
|
+
dbt build ─► run_results.json (which tests failed, row counts)
|
|
51
|
+
manifest.json (compiled SQL, guarded model/column, test type)
|
|
52
|
+
│
|
|
53
|
+
▼
|
|
54
|
+
parse.py join artifacts → FailingTest records
|
|
55
|
+
│
|
|
56
|
+
▼
|
|
57
|
+
context.py run the test's compiled SQL against DuckDB,
|
|
58
|
+
│ sample the offending rows (read-only, capped)
|
|
59
|
+
▼
|
|
60
|
+
analyze.py grounded prompt → LLM → {root_cause, fix,
|
|
61
|
+
│ confidence, evidence}
|
|
62
|
+
▼
|
|
63
|
+
report.py Rich terminal panel + markdown
|
|
64
|
+
(via cli.py: sentinel analyze)
|
|
65
|
+
```
|
|
66
|
+
The design choice that keeps it small and robust: **it reads dbt's artifacts rather than
|
|
67
|
+
re-running tests.** Everything it needs — compiled SQL, failing-row count, guarded model —
|
|
68
|
+
is already in the JSON dbt writes on every build.
|
|
69
|
+
|
|
70
|
+
## Grounding is the whole point
|
|
71
|
+
|
|
72
|
+
An LLM asked "why did my dbt test fail?" with no context invents a plausible-sounding
|
|
73
|
+
story. `dbt-sentinel` feeds the model only real evidence — the compiled SQL, the column
|
|
74
|
+
schema, and a sample of the actual failing rows — and tells it to ground every claim in
|
|
75
|
+
that evidence or report low confidence.
|
|
76
|
+
|
|
77
|
+
The difference on a real failure (a taxi-trip speed test where `avg_speed_mph` was computed
|
|
78
|
+
with a `600` multiplier instead of `60`):
|
|
79
|
+
|
|
80
|
+
**Naive prompt — test name + row count only, no evidence:**
|
|
81
|
+
|
|
82
|
+
> Most likely: division by zero or NULL producing infinite/NULL speed. With 1.8M failing
|
|
83
|
+
> rows this suggests a systemic data issue. Run these diagnostic queries to pinpoint the
|
|
84
|
+
> cause… [three SQL queries checking null/zero durations and negative distances].
|
|
85
|
+
|
|
86
|
+
**Grounded (dbt-sentinel):**
|
|
87
|
+
|
|
88
|
+
> The `avg_speed_mph` formula uses a multiplier of 600 instead of 60. A trip of 9.24 miles
|
|
89
|
+
> in 52 minutes yields (9.24/52)×60 = 10.66 mph (plausible), but the bug computes
|
|
90
|
+
> (9.24/52)×600 = 106.62 — which exactly matches the failing row. The same 10× inflation
|
|
91
|
+
> holds across all sampled rows. **Confidence: high.**
|
|
92
|
+
|
|
93
|
+
The naive answer guesses the wrong cause and hands the work back to you as queries to run.
|
|
94
|
+
The grounded answer runs them, finds the real bug, and proves it against a row. See
|
|
95
|
+
[`docs/example-analysis.md`](docs/example-analysis.md) for the full output.
|
|
96
|
+
|
|
97
|
+
## Warehouses
|
|
98
|
+
|
|
99
|
+
Sampling runs behind a small adapter interface, so the engine is a flag, not a rewrite.
|
|
100
|
+
DuckDB works out of the box; BigQuery needs the optional extra and Application Default
|
|
101
|
+
Credentials.
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
# DuckDB (default)
|
|
105
|
+
uv run sentinel analyze --target-dir path/to/target --db warehouse.duckdb
|
|
106
|
+
|
|
107
|
+
# BigQuery
|
|
108
|
+
uv sync --group bq
|
|
109
|
+
gcloud auth application-default login
|
|
110
|
+
uv run sentinel analyze --target-dir path/to/target --bq-project my-project --bq-location EU
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
BigQuery works against the free sandbox — no billing account required. Credentials are
|
|
114
|
+
never handled by dbt-sentinel itself; the client reads them from ADC.
|
|
115
|
+
|
|
116
|
+
## Dashboard
|
|
117
|
+
|
|
118
|
+
A single-page Streamlit view over the recorded history — latest run, per-test failure
|
|
119
|
+
trend, and the stored root cause. It reads dbt-sentinel's own history database, so it
|
|
120
|
+
needs no warehouse connection and no API key.
|
|
121
|
+
|
|
122
|
+
```bash
|
|
123
|
+
uv sync --group ui
|
|
124
|
+
uv run --group ui streamlit run app/dashboard.py
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+

|
|
128
|
+
|
|
129
|
+
## Design decisions
|
|
130
|
+
|
|
131
|
+
**Read artifacts, don't re-run tests.** dbt writes the artifacts on every build; parsing
|
|
132
|
+
them needs no dbt invocation and works against any completed run, including CI.
|
|
133
|
+
|
|
134
|
+
**Sample rows via the test's compiled SQL.** Some models are ephemeral and aren't queryable
|
|
135
|
+
as tables. The test's compiled SQL has that logic inlined, so it returns the offending rows
|
|
136
|
+
regardless of materialization.
|
|
137
|
+
|
|
138
|
+
**Ground, then surface confidence.** The prompt contains only evidence; the model is told to
|
|
139
|
+
say "insufficient evidence" rather than speculate, and confidence is shown, not hidden.
|
|
140
|
+
|
|
141
|
+
**Read-only, always.** The warehouse connection inspects; it never mutates.
|
|
142
|
+
|
|
143
|
+
## Limitations
|
|
144
|
+
|
|
145
|
+
- DuckDB and BigQuery supported; other engines need a new Warehouse adapter.
|
|
146
|
+
- One dbt project per run.
|
|
147
|
+
- Models with extended thinking can spend the whole token budget before emitting text,
|
|
148
|
+
so `max_tokens` is set generously; too low a value yields an unparseable (low-confidence)
|
|
149
|
+
result rather than an answer.
|
|
150
|
+
- Diagnosis quality depends on the model and on how much signal the sampled rows carry;
|
|
151
|
+
low-signal failures correctly return low confidence.
|
|
152
|
+
|
|
153
|
+
## Development
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
uv sync --group dev
|
|
157
|
+
uv run ruff check .
|
|
158
|
+
uv run pytest -v
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
CI runs the same lint + tests on every push. Tests use committed fixtures and a mocked API —
|
|
162
|
+
no warehouse, no API key needed.
|
|
163
|
+
|
|
164
|
+
---
|
|
165
|
+
|
|
166
|
+
Built by [Qamar Raza](https://github.com/qraza).
|
|
167
|
+
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "dbt-sentinel"
|
|
3
|
+
version = "0.6.1"
|
|
4
|
+
description = "AI-grounded root-cause analysis for failing dbt tests"
|
|
5
|
+
license = "MIT"
|
|
6
|
+
keywords = ["dbt", "data-quality", "observability", "analytics-engineering", "llm"]
|
|
7
|
+
classifiers = [
|
|
8
|
+
"Development Status :: 4 - Beta",
|
|
9
|
+
"Intended Audience :: Developers",
|
|
10
|
+
"Programming Language :: Python :: 3.12",
|
|
11
|
+
"Topic :: Database",
|
|
12
|
+
"Topic :: Software Development :: Quality Assurance",
|
|
13
|
+
]
|
|
14
|
+
readme = "README.md"
|
|
15
|
+
authors = [
|
|
16
|
+
{ name = "Qamar Raza", email = "raza.qamar@gmail.com" }
|
|
17
|
+
]
|
|
18
|
+
requires-python = ">=3.12"
|
|
19
|
+
dependencies = [
|
|
20
|
+
"click==8.1.7",
|
|
21
|
+
"duckdb==1.1.3",
|
|
22
|
+
"httpx==0.28.1",
|
|
23
|
+
"python-dotenv>=1.2.2",
|
|
24
|
+
"rich==13.9.4",
|
|
25
|
+
]
|
|
26
|
+
|
|
27
|
+
[project.optional-dependencies]
|
|
28
|
+
bq = ["google-cloud-bigquery>=3.43.0"]
|
|
29
|
+
ui = ["streamlit>=1.40"]
|
|
30
|
+
|
|
31
|
+
[project.urls]
|
|
32
|
+
Homepage = "https://github.com/qraza/dbt-sentinel"
|
|
33
|
+
Repository = "https://github.com/qraza/dbt-sentinel"
|
|
34
|
+
Issues = "https://github.com/qraza/dbt-sentinel/issues"
|
|
35
|
+
|
|
36
|
+
[project.scripts]
|
|
37
|
+
sentinel = "dbt_sentinel.cli:main"
|
|
38
|
+
|
|
39
|
+
[build-system]
|
|
40
|
+
requires = ["uv_build>=0.11.3,<0.12.0"]
|
|
41
|
+
build-backend = "uv_build"
|
|
42
|
+
|
|
43
|
+
[dependency-groups]
|
|
44
|
+
bq = [
|
|
45
|
+
"google-cloud-bigquery>=3.43.0",
|
|
46
|
+
]
|
|
47
|
+
dev = [
|
|
48
|
+
"pytest>=9.1.1",
|
|
49
|
+
"ruff>=0.16.0",
|
|
50
|
+
]
|
|
51
|
+
ui = [
|
|
52
|
+
"pandas>=3.0.5",
|
|
53
|
+
"streamlit>=1.61.1",
|
|
54
|
+
]
|
|
55
|
+
|
|
56
|
+
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""dbt-sentinel — AI-grounded root-cause analysis for failing dbt tests.
|
|
2
|
+
|
|
3
|
+
Typical library use:
|
|
4
|
+
|
|
5
|
+
from dbt_sentinel import parse, open_warehouse, gather_context, analyze
|
|
6
|
+
|
|
7
|
+
for test in parse("target"):
|
|
8
|
+
ctx = gather_context(test, open_warehouse(duckdb_path="wh.duckdb"))
|
|
9
|
+
print(analyze(ctx).root_cause)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from .analyze import Analysis, analyze, build_prompt
|
|
13
|
+
from .context import FailureContext, gather_context
|
|
14
|
+
from .parse import FailingTest, parse
|
|
15
|
+
from .report import AnalyzedFailure, build_markdown, render_terminal
|
|
16
|
+
from .warehouse import (
|
|
17
|
+
BigQueryWarehouse,
|
|
18
|
+
DuckDBWarehouse,
|
|
19
|
+
QueryResult,
|
|
20
|
+
Warehouse,
|
|
21
|
+
open_warehouse,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
__version__ = "0.6.1"
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"Analysis",
|
|
28
|
+
"AnalyzedFailure",
|
|
29
|
+
"BigQueryWarehouse",
|
|
30
|
+
"DuckDBWarehouse",
|
|
31
|
+
"FailingTest",
|
|
32
|
+
"FailureContext",
|
|
33
|
+
"QueryResult",
|
|
34
|
+
"Warehouse",
|
|
35
|
+
"__version__",
|
|
36
|
+
"analyze",
|
|
37
|
+
"build_markdown",
|
|
38
|
+
"build_prompt",
|
|
39
|
+
"gather_context",
|
|
40
|
+
"open_warehouse",
|
|
41
|
+
"parse",
|
|
42
|
+
"render_terminal",
|
|
43
|
+
]
|
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""Turn a failing test's gathered context into a grounded, root-cause analysis.
|
|
2
|
+
|
|
3
|
+
This is the part that makes dbt-sentinel more than a wrapper around "ask an LLM
|
|
4
|
+
why my test failed". The model is given *only* concrete evidence -- the test
|
|
5
|
+
definition, the compiled SQL, the column schema, and a capped sample of the
|
|
6
|
+
actual offending rows -- and is instructed to ground every claim in that
|
|
7
|
+
evidence and to report low confidence (rather than invent a cause) when the
|
|
8
|
+
sample doesn't support a conclusion.
|
|
9
|
+
|
|
10
|
+
The LLM is asked to return strict JSON so the result is structured and testable:
|
|
11
|
+
{"root_cause": ..., "suggested_fix": ..., "confidence": "high|medium|low",
|
|
12
|
+
"evidence": ...}
|
|
13
|
+
|
|
14
|
+
The Anthropic call is a thin httpx POST, matching the taxi repo's llm helper, so
|
|
15
|
+
it's trivially mockable in tests -- no network in CI.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import json
|
|
21
|
+
import os
|
|
22
|
+
from dataclasses import dataclass
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
import httpx
|
|
26
|
+
|
|
27
|
+
from .context import FailureContext
|
|
28
|
+
|
|
29
|
+
ANTHROPIC_URL = "https://api.anthropic.com/v1/messages"
|
|
30
|
+
# Set ANTHROPIC_MODEL to whatever model your key can use. Check the exact string
|
|
31
|
+
# your taxi repo's cli/llm.py already uses -- that's the source of truth for your
|
|
32
|
+
# setup -- rather than trusting this default.
|
|
33
|
+
DEFAULT_MODEL = os.environ.get("ANTHROPIC_MODEL", "claude-sonnet-4-5")
|
|
34
|
+
API_VERSION = "2023-06-01"
|
|
35
|
+
|
|
36
|
+
SYSTEM_PROMPT = (
|
|
37
|
+
"You are a senior analytics engineer reviewing a failed dbt data-quality "
|
|
38
|
+
"test. You are given the test definition, the compiled SQL that ran, the "
|
|
39
|
+
"column schema, and a sample of the rows that FAILED the test. "
|
|
40
|
+
"Diagnose the most likely root cause and propose a concrete fix. "
|
|
41
|
+
"Ground every statement in the evidence provided. If the evidence is "
|
|
42
|
+
"insufficient to be sure, say so plainly and set confidence to 'low' rather "
|
|
43
|
+
"than guessing or inventing tables, columns, or values that are not shown. "
|
|
44
|
+
"Respond with ONLY a JSON object with keys: root_cause (string), "
|
|
45
|
+
"suggested_fix (string), confidence (one of 'high', 'medium', 'low'), "
|
|
46
|
+
"evidence (string: which specific columns/values led you to the conclusion)."
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class Analysis:
|
|
52
|
+
"""A structured, grounded verdict on a single failing test."""
|
|
53
|
+
|
|
54
|
+
root_cause: str
|
|
55
|
+
suggested_fix: str
|
|
56
|
+
confidence: str # "high" | "medium" | "low"
|
|
57
|
+
evidence: str
|
|
58
|
+
raw: str = "" # the model's raw text, kept for debugging / display
|
|
59
|
+
|
|
60
|
+
@property
|
|
61
|
+
def is_confident(self) -> bool:
|
|
62
|
+
return self.confidence.lower() in {"high", "medium"}
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def build_prompt(ctx: FailureContext) -> str:
|
|
66
|
+
"""Assemble the grounded user prompt from gathered context.
|
|
67
|
+
|
|
68
|
+
Deliberately includes only evidence -- no external knowledge, no model name
|
|
69
|
+
guessing. Everything here is something dbt or the warehouse actually
|
|
70
|
+
produced for this specific failure.
|
|
71
|
+
"""
|
|
72
|
+
t = ctx.test
|
|
73
|
+
columns = ", ".join(f"{name} ({dtype})" for name, dtype in ctx.columns) or "(unknown)"
|
|
74
|
+
sample = json.dumps(ctx.sample_rows, indent=2, default=str)
|
|
75
|
+
|
|
76
|
+
kind = t.test_type or "singular (hand-written) test"
|
|
77
|
+
kwargs = json.dumps(t.test_kwargs) if t.test_kwargs else "(none)"
|
|
78
|
+
|
|
79
|
+
return (
|
|
80
|
+
f"A dbt test failed.\n\n"
|
|
81
|
+
f"Test name: {t.test_name}\n"
|
|
82
|
+
f"Test type: {kind}\n"
|
|
83
|
+
f"Test arguments: {kwargs}\n"
|
|
84
|
+
f"Guarded model: {t.model_name}\n"
|
|
85
|
+
f"Guarded column: {t.column_name or '(whole-row / singular test)'}\n"
|
|
86
|
+
f"Failing row count: {t.failure_count}\n"
|
|
87
|
+
f"dbt message: {t.message or '(none)'}\n\n"
|
|
88
|
+
f"Columns in the failing rows: {columns}\n\n"
|
|
89
|
+
f"Sample of the rows that FAILED the test (capped):\n{sample}\n\n"
|
|
90
|
+
f"Compiled SQL that dbt executed for this test:\n{t.compiled_sql}\n\n"
|
|
91
|
+
f"Sampling note: {ctx.note}\n"
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _parse_response(text: str) -> Analysis:
|
|
96
|
+
"""Parse the model's JSON reply, tolerating stray prose around the object."""
|
|
97
|
+
raw = text.strip()
|
|
98
|
+
start, end = raw.find("{"), raw.rfind("}")
|
|
99
|
+
if start != -1 and end != -1 and end > start:
|
|
100
|
+
try:
|
|
101
|
+
data: dict[str, Any] = json.loads(raw[start : end + 1])
|
|
102
|
+
return Analysis(
|
|
103
|
+
root_cause=str(data.get("root_cause", "")).strip(),
|
|
104
|
+
suggested_fix=str(data.get("suggested_fix", "")).strip(),
|
|
105
|
+
confidence=str(data.get("confidence", "low")).strip().lower(),
|
|
106
|
+
evidence=str(data.get("evidence", "")).strip(),
|
|
107
|
+
raw=raw,
|
|
108
|
+
)
|
|
109
|
+
except json.JSONDecodeError:
|
|
110
|
+
pass
|
|
111
|
+
# Model didn't return usable JSON -- fail safe as low confidence, not a crash.
|
|
112
|
+
return Analysis(
|
|
113
|
+
root_cause="Could not parse a structured analysis from the model response.",
|
|
114
|
+
suggested_fix="",
|
|
115
|
+
confidence="low",
|
|
116
|
+
evidence="",
|
|
117
|
+
raw=raw,
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def analyze(
|
|
122
|
+
ctx: FailureContext,
|
|
123
|
+
*,
|
|
124
|
+
api_key: str | None = None,
|
|
125
|
+
model: str = DEFAULT_MODEL,
|
|
126
|
+
max_tokens: int = 8192,
|
|
127
|
+
client: httpx.Client | None = None,
|
|
128
|
+
) -> Analysis:
|
|
129
|
+
"""Send grounded context to the Anthropic API and return a structured verdict.
|
|
130
|
+
|
|
131
|
+
Args:
|
|
132
|
+
ctx: the gathered failure context (test + sample offending rows).
|
|
133
|
+
api_key: Anthropic API key; falls back to ANTHROPIC_API_KEY env var.
|
|
134
|
+
model: model string to call.
|
|
135
|
+
client: optional httpx.Client, mainly for injection in tests.
|
|
136
|
+
"""
|
|
137
|
+
key = api_key or os.environ.get("ANTHROPIC_API_KEY")
|
|
138
|
+
if not key:
|
|
139
|
+
raise RuntimeError(
|
|
140
|
+
"No Anthropic API key. Set ANTHROPIC_API_KEY or pass api_key=."
|
|
141
|
+
)
|
|
142
|
+
|
|
143
|
+
payload = {
|
|
144
|
+
"model": model,
|
|
145
|
+
"max_tokens": max_tokens,
|
|
146
|
+
"system": SYSTEM_PROMPT,
|
|
147
|
+
"messages": [{"role": "user", "content": build_prompt(ctx)}],
|
|
148
|
+
}
|
|
149
|
+
headers = {
|
|
150
|
+
"x-api-key": key,
|
|
151
|
+
"anthropic-version": API_VERSION,
|
|
152
|
+
"content-type": "application/json",
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
owns_client = client is None
|
|
156
|
+
client = client or httpx.Client(timeout=60)
|
|
157
|
+
try:
|
|
158
|
+
resp = client.post(ANTHROPIC_URL, headers=headers, json=payload)
|
|
159
|
+
resp.raise_for_status()
|
|
160
|
+
body = resp.json()
|
|
161
|
+
finally:
|
|
162
|
+
if owns_client:
|
|
163
|
+
client.close()
|
|
164
|
+
# Anthropic returns {"content": [{"type": "text", "text": "..."}], ...}
|
|
165
|
+
text = "".join(
|
|
166
|
+
block.get("text", "")
|
|
167
|
+
for block in body.get("content", [])
|
|
168
|
+
if block.get("type") == "text"
|
|
169
|
+
)
|
|
170
|
+
return _parse_response(text)
|