dataguard-ai 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dataguard_ai-0.2.0/LICENSE +17 -0
- dataguard_ai-0.2.0/PKG-INFO +338 -0
- dataguard_ai-0.2.0/README.md +299 -0
- dataguard_ai-0.2.0/pyproject.toml +37 -0
- dataguard_ai-0.2.0/setup.cfg +4 -0
- dataguard_ai-0.2.0/src/dataguard/__init__.py +1 -0
- dataguard_ai-0.2.0/src/dataguard/cli.py +135 -0
- dataguard_ai-0.2.0/src/dataguard/config.py +22 -0
- dataguard_ai-0.2.0/src/dataguard/demo.py +54 -0
- dataguard_ai-0.2.0/src/dataguard/drift.py +34 -0
- dataguard_ai-0.2.0/src/dataguard/explain.py +42 -0
- dataguard_ai-0.2.0/src/dataguard/exporters.py +50 -0
- dataguard_ai-0.2.0/src/dataguard/gx_exporter.py +15 -0
- dataguard_ai-0.2.0/src/dataguard/html_report.py +42 -0
- dataguard_ai-0.2.0/src/dataguard/io.py +22 -0
- dataguard_ai-0.2.0/src/dataguard/models.py +36 -0
- dataguard_ai-0.2.0/src/dataguard/pii.py +47 -0
- dataguard_ai-0.2.0/src/dataguard/rules.py +83 -0
- dataguard_ai-0.2.0/src/dataguard/scanner.py +140 -0
- dataguard_ai-0.2.0/src/dataguard/sources.py +39 -0
- dataguard_ai-0.2.0/src/dataguard_ai.egg-info/PKG-INFO +338 -0
- dataguard_ai-0.2.0/src/dataguard_ai.egg-info/SOURCES.txt +32 -0
- dataguard_ai-0.2.0/src/dataguard_ai.egg-info/dependency_links.txt +1 -0
- dataguard_ai-0.2.0/src/dataguard_ai.egg-info/entry_points.txt +2 -0
- dataguard_ai-0.2.0/src/dataguard_ai.egg-info/requires.txt +34 -0
- dataguard_ai-0.2.0/src/dataguard_ai.egg-info/top_level.txt +1 -0
- dataguard_ai-0.2.0/tests/test_drift.py +12 -0
- dataguard_ai-0.2.0/tests/test_exporters.py +13 -0
- dataguard_ai-0.2.0/tests/test_gx.py +12 -0
- dataguard_ai-0.2.0/tests/test_html.py +10 -0
- dataguard_ai-0.2.0/tests/test_pii.py +12 -0
- dataguard_ai-0.2.0/tests/test_rules.py +11 -0
- dataguard_ai-0.2.0/tests/test_scanner.py +23 -0
- dataguard_ai-0.2.0/tests/test_sources.py +9 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
Apache License
|
|
2
|
+
Version 2.0, January 2004
|
|
3
|
+
http://www.apache.org/licenses/
|
|
4
|
+
|
|
5
|
+
Copyright 2026 Aditya Ranjan
|
|
6
|
+
|
|
7
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
8
|
+
you may not use this file except in compliance with the License.
|
|
9
|
+
You may obtain a copy of the License at
|
|
10
|
+
|
|
11
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
12
|
+
|
|
13
|
+
Unless required by applicable law or agreed to in writing, software
|
|
14
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
15
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
16
|
+
See the License for the specific language governing permissions and
|
|
17
|
+
limitations under the License.
|
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dataguard-ai
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Open-source data quality and governance copilot: scan data, detect risks, explain findings, and generate fixes.
|
|
5
|
+
Author: Aditya Ranjan
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Keywords: data-quality,data-governance,data-contracts,pii,dbt,great-expectations,duckdb,postgresql,ai
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Description-Content-Type: text/markdown
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Dist: pandas>=2.0
|
|
12
|
+
Requires-Dist: typer>=0.12
|
|
13
|
+
Requires-Dist: rich>=13.7
|
|
14
|
+
Requires-Dist: pydantic>=2.7
|
|
15
|
+
Requires-Dist: PyYAML>=6.0
|
|
16
|
+
Provides-Extra: parquet
|
|
17
|
+
Requires-Dist: pyarrow>=15; extra == "parquet"
|
|
18
|
+
Provides-Extra: duckdb
|
|
19
|
+
Requires-Dist: duckdb>=1.0; extra == "duckdb"
|
|
20
|
+
Provides-Extra: postgres
|
|
21
|
+
Requires-Dist: sqlalchemy>=2.0; extra == "postgres"
|
|
22
|
+
Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
|
|
23
|
+
Provides-Extra: gx
|
|
24
|
+
Requires-Dist: great_expectations>=1.0; extra == "gx"
|
|
25
|
+
Provides-Extra: ai
|
|
26
|
+
Requires-Dist: openai>=1.40; extra == "ai"
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: pytest>=8; extra == "dev"
|
|
29
|
+
Requires-Dist: pytest-cov>=5; extra == "dev"
|
|
30
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
31
|
+
Provides-Extra: all
|
|
32
|
+
Requires-Dist: pyarrow>=15; extra == "all"
|
|
33
|
+
Requires-Dist: duckdb>=1.0; extra == "all"
|
|
34
|
+
Requires-Dist: sqlalchemy>=2.0; extra == "all"
|
|
35
|
+
Requires-Dist: psycopg[binary]>=3.1; extra == "all"
|
|
36
|
+
Requires-Dist: great_expectations>=1.0; extra == "all"
|
|
37
|
+
Requires-Dist: openai>=1.40; extra == "all"
|
|
38
|
+
Dynamic: license-file
|
|
39
|
+
|
|
40
|
+
# DataGuard AI
|
|
41
|
+
|
|
42
|
+
[](https://github.com/adiranjan25/dataguard-ai/actions/workflows/ci.yml)
|
|
43
|
+
[](https://www.python.org/)
|
|
44
|
+
[](LICENSE)
|
|
45
|
+
[](#project-status)
|
|
46
|
+
|
|
47
|
+
**Open-source, AI-assisted data quality and governance for modern data platforms.**
|
|
48
|
+
|
|
49
|
+
> **Scan your data. Find quality and governance risks. Understand why they matter. Generate the fix.**
|
|
50
|
+
|
|
51
|
+
DataGuard AI is a developer-first toolkit for **data quality, data governance, PII discovery, schema drift, data contracts, and CI/CD-friendly validation**. Detection is deterministic and does **not** require an LLM. Optional AI assistance can explain structured findings and suggest remediation without making quality detection dependent on a model.
|
|
52
|
+
|
|
53
|
+
**Current version: v0.2.0 public beta.**
|
|
54
|
+
|
|
55
|
+
## Why DataGuard AI?
|
|
56
|
+
|
|
57
|
+
Data teams often manage quality rules, contracts, PII checks, schema drift, metadata, and AI assistants in separate workflows. DataGuard AI provides a lightweight layer developers can run locally or in CI to surface these risks through one interface.
|
|
58
|
+
|
|
59
|
+
### What v0.2 can do
|
|
60
|
+
|
|
61
|
+
- Profile CSV and JSON data, with optional Parquet support
|
|
62
|
+
- Scan **DuckDB** and **PostgreSQL** tables
|
|
63
|
+
- Run reusable **YAML data-quality rules**
|
|
64
|
+
- Detect nulls, duplicate rows, uniqueness issues, range violations, and freshness risks
|
|
65
|
+
- Identify likely PII using value and column-name signals
|
|
66
|
+
- Compute transparent quality and governance scores
|
|
67
|
+
- Generate standalone **HTML** and machine-readable **JSON** reports
|
|
68
|
+
- Snapshot schemas and detect schema drift
|
|
69
|
+
- Generate starter **dbt tests** and YAML data contracts
|
|
70
|
+
- Export portable **Great Expectations** expectation configuration
|
|
71
|
+
- Run in **GitHub Actions**
|
|
72
|
+
- Explain findings locally, with optional AI-assisted remediation guidance
|
|
73
|
+
|
|
74
|
+
## Quick start
|
|
75
|
+
|
|
76
|
+
### Install from source during the public beta
|
|
77
|
+
|
|
78
|
+
Until DataGuard AI is published to PyPI, clone the repository and install it locally:
|
|
79
|
+
|
|
80
|
+
```bash
|
|
81
|
+
git clone https://github.com/adiranjan25/dataguard-ai.git
|
|
82
|
+
cd dataguard-ai
|
|
83
|
+
python3 -m venv .venv
|
|
84
|
+
source .venv/bin/activate
|
|
85
|
+
python -m pip install -e .
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Run the synthetic retail demo:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
dataguard demo --rows 100
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Or scan your own dataset:
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
dataguard scan data/customers.csv
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
Generate JSON and HTML reports:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
dataguard scan data/customers.csv --json-out report.json --html-out report.html
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
> **Coming next:** after the PyPI release, installation will become simply `pip install dataguard-ai`.
|
|
107
|
+
|
|
108
|
+
## Detection philosophy
|
|
109
|
+
|
|
110
|
+
> **AI assists; deterministic and statistical checks detect and verify.**
|
|
111
|
+
|
|
112
|
+
The default scanning path does not require an LLM. This keeps findings reproducible and allows teams to use DataGuard AI without sending raw production datasets to an external model.
|
|
113
|
+
|
|
114
|
+
## YAML rule engine
|
|
115
|
+
|
|
116
|
+
Supported v0.2 custom rule types:
|
|
117
|
+
|
|
118
|
+
- `not_null`
|
|
119
|
+
- `unique`
|
|
120
|
+
- `accepted_values`
|
|
121
|
+
- `between`
|
|
122
|
+
- `regex`
|
|
123
|
+
- `max_null_pct`
|
|
124
|
+
- `row_count_between`
|
|
125
|
+
|
|
126
|
+
Example:
|
|
127
|
+
|
|
128
|
+
```yaml
|
|
129
|
+
quality:
|
|
130
|
+
max_null_pct: 5
|
|
131
|
+
governance:
|
|
132
|
+
owner: data-platform@example.com
|
|
133
|
+
rules:
|
|
134
|
+
- type: not_null
|
|
135
|
+
column: customer_id
|
|
136
|
+
severity: CRITICAL
|
|
137
|
+
- type: unique
|
|
138
|
+
column: customer_id
|
|
139
|
+
- type: accepted_values
|
|
140
|
+
column: state
|
|
141
|
+
values: [TX, CA, NY]
|
|
142
|
+
- type: between
|
|
143
|
+
column: amount
|
|
144
|
+
min: 0
|
|
145
|
+
max: 100000
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
dataguard scan customers.csv --config dataguard.yml --html-out report.html
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## Database scanning
|
|
153
|
+
|
|
154
|
+
### DuckDB
|
|
155
|
+
|
|
156
|
+
```bash
|
|
157
|
+
python -m pip install -e ".[duckdb]"
|
|
158
|
+
dataguard scan-duckdb analytics.duckdb --table customers --html-out report.html
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
### PostgreSQL
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
python -m pip install -e ".[postgres]"
|
|
165
|
+
export DATAGUARD_POSTGRES_URL='postgresql+psycopg://user:password@host/database'
|
|
166
|
+
dataguard scan-postgres --table public.customers --html-out report.html
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
> **Current limitation:** database scans load the selected table/result into memory. Warehouse-scale pushdown profiling is a roadmap item.
|
|
170
|
+
|
|
171
|
+
## Reports
|
|
172
|
+
|
|
173
|
+
```bash
|
|
174
|
+
dataguard scan customers.csv --html-out report.html
|
|
175
|
+
dataguard scan customers.csv --json-out report.json
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
The HTML report includes quality/governance scores, findings, severity, affected columns, suggested remediation, column profiles, and detected PII.
|
|
179
|
+
|
|
180
|
+
## Data contracts and dbt
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
dataguard contract data/customers.csv --out contract.yml
|
|
184
|
+
dataguard generate-dbt data/customers.csv --out schema.yml
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
These outputs are intended as reviewable starting points rather than replacements for domain-specific contract design.
|
|
188
|
+
|
|
189
|
+
## Great Expectations
|
|
190
|
+
|
|
191
|
+
```bash
|
|
192
|
+
dataguard generate-gx data/customers.csv --out gx-expectations.json
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
The exporter deliberately produces reviewable configuration rather than modifying an existing Great Expectations project.
|
|
196
|
+
|
|
197
|
+
## Schema drift
|
|
198
|
+
|
|
199
|
+
```bash
|
|
200
|
+
dataguard snapshot data/customers.csv --out baseline.json
|
|
201
|
+
dataguard drift data/customers_v2.csv --baseline baseline.json
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
## GitHub Actions
|
|
205
|
+
|
|
206
|
+
The repository includes an example workflow at `.github/workflows/dataguard.yml` that demonstrates scanning sample data and uploading JSON/HTML reports as workflow artifacts.
|
|
207
|
+
|
|
208
|
+
Project CI separately runs linting and automated tests against **Python 3.10, 3.11, and 3.12**.
|
|
209
|
+
|
|
210
|
+
## Optional AI explanations
|
|
211
|
+
|
|
212
|
+
Local deterministic explanations are available without an external model.
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
python -m pip install -e ".[ai]"
|
|
216
|
+
export OPENAI_API_KEY=...
|
|
217
|
+
dataguard explain report.json --provider openai
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
The included provider sends structured findings rather than raw dataset rows. Always review your organization's security, privacy, and data-handling requirements before enabling an external provider.
|
|
221
|
+
|
|
222
|
+
## Architecture
|
|
223
|
+
|
|
224
|
+
```text
|
|
225
|
+
Data sources
|
|
226
|
+
|
|
|
227
|
+
+-----------------+-----------------+
|
|
228
|
+
| | |
|
|
229
|
+
CSV / JSON DuckDB PostgreSQL
|
|
230
|
+
/ Parquet | |
|
|
231
|
+
+-----------------+-----------------+
|
|
232
|
+
|
|
|
233
|
+
v
|
|
234
|
+
DataGuard scanner
|
|
235
|
+
|
|
|
236
|
+
+----------------+----------------+
|
|
237
|
+
| | |
|
|
238
|
+
Profiling Rule engine PII detection
|
|
239
|
+
| | |
|
|
240
|
+
+----------------+----------------+
|
|
241
|
+
|
|
|
242
|
+
v
|
|
243
|
+
Quality + governance
|
|
244
|
+
|
|
|
245
|
+
+-----------+-----+------+-----------+
|
|
246
|
+
| | | |
|
|
247
|
+
v v v v
|
|
248
|
+
CLI JSON HTML Contracts / dbt / GX
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
Detection and optional AI explanation are intentionally separated.
|
|
252
|
+
|
|
253
|
+
## Commands
|
|
254
|
+
|
|
255
|
+
| Command | Purpose |
|
|
256
|
+
|---|---|
|
|
257
|
+
| `dataguard scan PATH` | Profile file data and report quality/governance findings |
|
|
258
|
+
| `dataguard demo` | Generate and scan synthetic retail datasets |
|
|
259
|
+
| `dataguard scan-duckdb DATABASE --table TABLE` | Scan a DuckDB table |
|
|
260
|
+
| `dataguard scan-postgres --table TABLE` | Scan a PostgreSQL table |
|
|
261
|
+
| `dataguard contract PATH` | Generate a starter data contract |
|
|
262
|
+
| `dataguard generate-dbt PATH` | Generate starter dbt tests |
|
|
263
|
+
| `dataguard generate-gx PATH` | Generate portable GX expectation configuration |
|
|
264
|
+
| `dataguard snapshot PATH` | Save a schema/profile baseline |
|
|
265
|
+
| `dataguard drift PATH --baseline FILE` | Compare current schema with a baseline |
|
|
266
|
+
| `dataguard explain REPORT.json` | Explain findings locally or with optional AI |
|
|
267
|
+
|
|
268
|
+
Run `dataguard --help` for current CLI options.
|
|
269
|
+
|
|
270
|
+
## Python API
|
|
271
|
+
|
|
272
|
+
```python
|
|
273
|
+
from dataguard.scanner import scan_path
|
|
274
|
+
|
|
275
|
+
report = scan_path("customers.csv")
|
|
276
|
+
print(report.quality_score)
|
|
277
|
+
|
|
278
|
+
for finding in report.findings:
|
|
279
|
+
print(finding.severity, finding.column, finding.message)
|
|
280
|
+
```
|
|
281
|
+
|
|
282
|
+
## Retail demo
|
|
283
|
+
|
|
284
|
+
The bundled synthetic retail demo creates customer, order, and inventory datasets with intentionally injected quality/governance problems so developers can explore DataGuard AI without providing proprietary data.
|
|
285
|
+
|
|
286
|
+
```bash
|
|
287
|
+
dataguard demo --rows 5000
|
|
288
|
+
```
|
|
289
|
+
|
|
290
|
+
## Project status
|
|
291
|
+
|
|
292
|
+
DataGuard AI is currently a **public beta (v0.2.0)**. The API, configuration schema, scoring model, and command behavior may evolve before v1.0.
|
|
293
|
+
|
|
294
|
+
The project is suitable for experimentation, development workflows, demos, and community feedback. Evaluate it against your own requirements before using it as a production control.
|
|
295
|
+
|
|
296
|
+
## Roadmap
|
|
297
|
+
|
|
298
|
+
### v0.3 — Metadata and context
|
|
299
|
+
|
|
300
|
+
- OpenMetadata integration
|
|
301
|
+
- DataHub integration
|
|
302
|
+
- dbt artifact ingestion
|
|
303
|
+
- Dataset-to-dataset referential checks
|
|
304
|
+
- Richer statistical drift and anomaly detection
|
|
305
|
+
|
|
306
|
+
### v0.4 — Agent access
|
|
307
|
+
|
|
308
|
+
- MCP server
|
|
309
|
+
- Agent-accessible quality, contract, and governance tools
|
|
310
|
+
- Ollama/local-model provider
|
|
311
|
+
- Governance context for AI agents
|
|
312
|
+
- Assisted remediation workflows
|
|
313
|
+
|
|
314
|
+
### Toward v1.0
|
|
315
|
+
|
|
316
|
+
- PyPI distribution and automated release workflow
|
|
317
|
+
- Warehouse-scale profiling/pushdown
|
|
318
|
+
- Broader integration tests
|
|
319
|
+
- Benchmark datasets and reproducible evaluation
|
|
320
|
+
- Stable configuration and CLI contracts
|
|
321
|
+
|
|
322
|
+
## Contributing
|
|
323
|
+
|
|
324
|
+
Contributions are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md) for development setup and contribution guidance.
|
|
325
|
+
|
|
326
|
+
Useful first contributions include additional PII detectors, report/export formats, documentation improvements, tests, database adapters, and integration examples.
|
|
327
|
+
|
|
328
|
+
## Security and privacy
|
|
329
|
+
|
|
330
|
+
- Never submit secrets, credentials, or proprietary production datasets in GitHub issues.
|
|
331
|
+
- Use environment variables or an appropriate secrets manager for database and model credentials.
|
|
332
|
+
- Treat detected PII findings as sensitive operational metadata.
|
|
333
|
+
- Review organizational security/privacy requirements before using an external AI provider.
|
|
334
|
+
- See [SECURITY.md](SECURITY.md) for vulnerability-reporting guidance.
|
|
335
|
+
|
|
336
|
+
## License
|
|
337
|
+
|
|
338
|
+
DataGuard AI is licensed under the **Apache License 2.0**. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,299 @@
|
|
|
1
|
+
# DataGuard AI
|
|
2
|
+
|
|
3
|
+
[](https://github.com/adiranjan25/dataguard-ai/actions/workflows/ci.yml)
|
|
4
|
+
[](https://www.python.org/)
|
|
5
|
+
[](LICENSE)
|
|
6
|
+
[](#project-status)
|
|
7
|
+
|
|
8
|
+
**Open-source, AI-assisted data quality and governance for modern data platforms.**
|
|
9
|
+
|
|
10
|
+
> **Scan your data. Find quality and governance risks. Understand why they matter. Generate the fix.**
|
|
11
|
+
|
|
12
|
+
DataGuard AI is a developer-first toolkit for **data quality, data governance, PII discovery, schema drift, data contracts, and CI/CD-friendly validation**. Detection is deterministic and does **not** require an LLM. Optional AI assistance can explain structured findings and suggest remediation without making quality detection dependent on a model.
|
|
13
|
+
|
|
14
|
+
**Current version: v0.2.0 public beta.**
|
|
15
|
+
|
|
16
|
+
## Why DataGuard AI?
|
|
17
|
+
|
|
18
|
+
Data teams often manage quality rules, contracts, PII checks, schema drift, metadata, and AI assistants in separate workflows. DataGuard AI provides a lightweight layer developers can run locally or in CI to surface these risks through one interface.
|
|
19
|
+
|
|
20
|
+
### What v0.2 can do
|
|
21
|
+
|
|
22
|
+
- Profile CSV and JSON data, with optional Parquet support
|
|
23
|
+
- Scan **DuckDB** and **PostgreSQL** tables
|
|
24
|
+
- Run reusable **YAML data-quality rules**
|
|
25
|
+
- Detect nulls, duplicate rows, uniqueness issues, range violations, and freshness risks
|
|
26
|
+
- Identify likely PII using value and column-name signals
|
|
27
|
+
- Compute transparent quality and governance scores
|
|
28
|
+
- Generate standalone **HTML** and machine-readable **JSON** reports
|
|
29
|
+
- Snapshot schemas and detect schema drift
|
|
30
|
+
- Generate starter **dbt tests** and YAML data contracts
|
|
31
|
+
- Export portable **Great Expectations** expectation configuration
|
|
32
|
+
- Run in **GitHub Actions**
|
|
33
|
+
- Explain findings locally, with optional AI-assisted remediation guidance
|
|
34
|
+
|
|
35
|
+
## Quick start
|
|
36
|
+
|
|
37
|
+
### Install from source during the public beta
|
|
38
|
+
|
|
39
|
+
Until DataGuard AI is published to PyPI, clone the repository and install it locally:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
git clone https://github.com/adiranjan25/dataguard-ai.git
|
|
43
|
+
cd dataguard-ai
|
|
44
|
+
python3 -m venv .venv
|
|
45
|
+
source .venv/bin/activate
|
|
46
|
+
python -m pip install -e .
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Run the synthetic retail demo:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
dataguard demo --rows 100
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
Or scan your own dataset:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
dataguard scan data/customers.csv
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
Generate JSON and HTML reports:
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
dataguard scan data/customers.csv --json-out report.json --html-out report.html
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
> **Coming next:** after the PyPI release, installation will become simply `pip install dataguard-ai`.
|
|
68
|
+
|
|
69
|
+
## Detection philosophy
|
|
70
|
+
|
|
71
|
+
> **AI assists; deterministic and statistical checks detect and verify.**
|
|
72
|
+
|
|
73
|
+
The default scanning path does not require an LLM. This keeps findings reproducible and allows teams to use DataGuard AI without sending raw production datasets to an external model.
|
|
74
|
+
|
|
75
|
+
## YAML rule engine
|
|
76
|
+
|
|
77
|
+
Supported v0.2 custom rule types:
|
|
78
|
+
|
|
79
|
+
- `not_null`
|
|
80
|
+
- `unique`
|
|
81
|
+
- `accepted_values`
|
|
82
|
+
- `between`
|
|
83
|
+
- `regex`
|
|
84
|
+
- `max_null_pct`
|
|
85
|
+
- `row_count_between`
|
|
86
|
+
|
|
87
|
+
Example:
|
|
88
|
+
|
|
89
|
+
```yaml
|
|
90
|
+
quality:
|
|
91
|
+
max_null_pct: 5
|
|
92
|
+
governance:
|
|
93
|
+
owner: data-platform@example.com
|
|
94
|
+
rules:
|
|
95
|
+
- type: not_null
|
|
96
|
+
column: customer_id
|
|
97
|
+
severity: CRITICAL
|
|
98
|
+
- type: unique
|
|
99
|
+
column: customer_id
|
|
100
|
+
- type: accepted_values
|
|
101
|
+
column: state
|
|
102
|
+
values: [TX, CA, NY]
|
|
103
|
+
- type: between
|
|
104
|
+
column: amount
|
|
105
|
+
min: 0
|
|
106
|
+
max: 100000
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
```bash
|
|
110
|
+
dataguard scan customers.csv --config dataguard.yml --html-out report.html
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
## Database scanning
|
|
114
|
+
|
|
115
|
+
### DuckDB
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
python -m pip install -e ".[duckdb]"
|
|
119
|
+
dataguard scan-duckdb analytics.duckdb --table customers --html-out report.html
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
### PostgreSQL
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
python -m pip install -e ".[postgres]"
|
|
126
|
+
export DATAGUARD_POSTGRES_URL='postgresql+psycopg://user:password@host/database'
|
|
127
|
+
dataguard scan-postgres --table public.customers --html-out report.html
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
> **Current limitation:** database scans load the selected table/result into memory. Warehouse-scale pushdown profiling is a roadmap item.
|
|
131
|
+
|
|
132
|
+
## Reports
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
dataguard scan customers.csv --html-out report.html
|
|
136
|
+
dataguard scan customers.csv --json-out report.json
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
The HTML report includes quality/governance scores, findings, severity, affected columns, suggested remediation, column profiles, and detected PII.
|
|
140
|
+
|
|
141
|
+
## Data contracts and dbt
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
dataguard contract data/customers.csv --out contract.yml
|
|
145
|
+
dataguard generate-dbt data/customers.csv --out schema.yml
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
These outputs are intended as reviewable starting points rather than replacements for domain-specific contract design.
|
|
149
|
+
|
|
150
|
+
## Great Expectations
|
|
151
|
+
|
|
152
|
+
```bash
|
|
153
|
+
dataguard generate-gx data/customers.csv --out gx-expectations.json
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
The exporter deliberately produces reviewable configuration rather than modifying an existing Great Expectations project.
|
|
157
|
+
|
|
158
|
+
## Schema drift
|
|
159
|
+
|
|
160
|
+
```bash
|
|
161
|
+
dataguard snapshot data/customers.csv --out baseline.json
|
|
162
|
+
dataguard drift data/customers_v2.csv --baseline baseline.json
|
|
163
|
+
```
|
|
164
|
+
|
|
165
|
+
## GitHub Actions
|
|
166
|
+
|
|
167
|
+
The repository includes an example workflow at `.github/workflows/dataguard.yml` that demonstrates scanning sample data and uploading JSON/HTML reports as workflow artifacts.
|
|
168
|
+
|
|
169
|
+
Project CI separately runs linting and automated tests against **Python 3.10, 3.11, and 3.12**.
|
|
170
|
+
|
|
171
|
+
## Optional AI explanations
|
|
172
|
+
|
|
173
|
+
Local deterministic explanations are available without an external model.
|
|
174
|
+
|
|
175
|
+
```bash
|
|
176
|
+
python -m pip install -e ".[ai]"
|
|
177
|
+
export OPENAI_API_KEY=...
|
|
178
|
+
dataguard explain report.json --provider openai
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
The included provider sends structured findings rather than raw dataset rows. Always review your organization's security, privacy, and data-handling requirements before enabling an external provider.
|
|
182
|
+
|
|
183
|
+
## Architecture
|
|
184
|
+
|
|
185
|
+
```text
|
|
186
|
+
Data sources
|
|
187
|
+
|
|
|
188
|
+
+-----------------+-----------------+
|
|
189
|
+
| | |
|
|
190
|
+
CSV / JSON DuckDB PostgreSQL
|
|
191
|
+
/ Parquet | |
|
|
192
|
+
+-----------------+-----------------+
|
|
193
|
+
|
|
|
194
|
+
v
|
|
195
|
+
DataGuard scanner
|
|
196
|
+
|
|
|
197
|
+
+----------------+----------------+
|
|
198
|
+
| | |
|
|
199
|
+
Profiling Rule engine PII detection
|
|
200
|
+
| | |
|
|
201
|
+
+----------------+----------------+
|
|
202
|
+
|
|
|
203
|
+
v
|
|
204
|
+
Quality + governance
|
|
205
|
+
|
|
|
206
|
+
+-----------+-----+------+-----------+
|
|
207
|
+
| | | |
|
|
208
|
+
v v v v
|
|
209
|
+
CLI JSON HTML Contracts / dbt / GX
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
Detection and optional AI explanation are intentionally separated.
|
|
213
|
+
|
|
214
|
+
## Commands
|
|
215
|
+
|
|
216
|
+
| Command | Purpose |
|
|
217
|
+
|---|---|
|
|
218
|
+
| `dataguard scan PATH` | Profile file data and report quality/governance findings |
|
|
219
|
+
| `dataguard demo` | Generate and scan synthetic retail datasets |
|
|
220
|
+
| `dataguard scan-duckdb DATABASE --table TABLE` | Scan a DuckDB table |
|
|
221
|
+
| `dataguard scan-postgres --table TABLE` | Scan a PostgreSQL table |
|
|
222
|
+
| `dataguard contract PATH` | Generate a starter data contract |
|
|
223
|
+
| `dataguard generate-dbt PATH` | Generate starter dbt tests |
|
|
224
|
+
| `dataguard generate-gx PATH` | Generate portable GX expectation configuration |
|
|
225
|
+
| `dataguard snapshot PATH` | Save a schema/profile baseline |
|
|
226
|
+
| `dataguard drift PATH --baseline FILE` | Compare current schema with a baseline |
|
|
227
|
+
| `dataguard explain REPORT.json` | Explain findings locally or with optional AI |
|
|
228
|
+
|
|
229
|
+
Run `dataguard --help` for current CLI options.
|
|
230
|
+
|
|
231
|
+
## Python API
|
|
232
|
+
|
|
233
|
+
```python
|
|
234
|
+
from dataguard.scanner import scan_path
|
|
235
|
+
|
|
236
|
+
report = scan_path("customers.csv")
|
|
237
|
+
print(report.quality_score)
|
|
238
|
+
|
|
239
|
+
for finding in report.findings:
|
|
240
|
+
print(finding.severity, finding.column, finding.message)
|
|
241
|
+
```
|
|
242
|
+
|
|
243
|
+
## Retail demo
|
|
244
|
+
|
|
245
|
+
The bundled synthetic retail demo creates customer, order, and inventory datasets with intentionally injected quality/governance problems so developers can explore DataGuard AI without providing proprietary data.
|
|
246
|
+
|
|
247
|
+
```bash
|
|
248
|
+
dataguard demo --rows 5000
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
## Project status
|
|
252
|
+
|
|
253
|
+
DataGuard AI is currently a **public beta (v0.2.0)**. The API, configuration schema, scoring model, and command behavior may evolve before v1.0.
|
|
254
|
+
|
|
255
|
+
The project is suitable for experimentation, development workflows, demos, and community feedback. Evaluate it against your own requirements before using it as a production control.
|
|
256
|
+
|
|
257
|
+
## Roadmap
|
|
258
|
+
|
|
259
|
+
### v0.3 — Metadata and context
|
|
260
|
+
|
|
261
|
+
- OpenMetadata integration
|
|
262
|
+
- DataHub integration
|
|
263
|
+
- dbt artifact ingestion
|
|
264
|
+
- Dataset-to-dataset referential checks
|
|
265
|
+
- Richer statistical drift and anomaly detection
|
|
266
|
+
|
|
267
|
+
### v0.4 — Agent access
|
|
268
|
+
|
|
269
|
+
- MCP server
|
|
270
|
+
- Agent-accessible quality, contract, and governance tools
|
|
271
|
+
- Ollama/local-model provider
|
|
272
|
+
- Governance context for AI agents
|
|
273
|
+
- Assisted remediation workflows
|
|
274
|
+
|
|
275
|
+
### Toward v1.0
|
|
276
|
+
|
|
277
|
+
- PyPI distribution and automated release workflow
|
|
278
|
+
- Warehouse-scale profiling/pushdown
|
|
279
|
+
- Broader integration tests
|
|
280
|
+
- Benchmark datasets and reproducible evaluation
|
|
281
|
+
- Stable configuration and CLI contracts
|
|
282
|
+
|
|
283
|
+
## Contributing
|
|
284
|
+
|
|
285
|
+
Contributions are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md) for development setup and contribution guidance.
|
|
286
|
+
|
|
287
|
+
Useful first contributions include additional PII detectors, report/export formats, documentation improvements, tests, database adapters, and integration examples.
|
|
288
|
+
|
|
289
|
+
## Security and privacy
|
|
290
|
+
|
|
291
|
+
- Never submit secrets, credentials, or proprietary production datasets in GitHub issues.
|
|
292
|
+
- Use environment variables or an appropriate secrets manager for database and model credentials.
|
|
293
|
+
- Treat detected PII findings as sensitive operational metadata.
|
|
294
|
+
- Review organizational security/privacy requirements before using an external AI provider.
|
|
295
|
+
- See [SECURITY.md](SECURITY.md) for vulnerability-reporting guidance.
|
|
296
|
+
|
|
297
|
+
## License
|
|
298
|
+
|
|
299
|
+
DataGuard AI is licensed under the **Apache License 2.0**. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=69", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dataguard-ai"
|
|
7
|
+
version = "0.2.0"
|
|
8
|
+
description = "Open-source data quality and governance copilot: scan data, detect risks, explain findings, and generate fixes."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = {text = "Apache-2.0"}
|
|
12
|
+
authors = [{name = "Aditya Ranjan"}]
|
|
13
|
+
keywords = ["data-quality", "data-governance", "data-contracts", "pii", "dbt", "great-expectations", "duckdb", "postgresql", "ai"]
|
|
14
|
+
dependencies = ["pandas>=2.0","typer>=0.12","rich>=13.7","pydantic>=2.7","PyYAML>=6.0"]
|
|
15
|
+
|
|
16
|
+
[project.optional-dependencies]
|
|
17
|
+
parquet = ["pyarrow>=15"]
|
|
18
|
+
duckdb = ["duckdb>=1.0"]
|
|
19
|
+
postgres = ["sqlalchemy>=2.0", "psycopg[binary]>=3.1"]
|
|
20
|
+
gx = ["great_expectations>=1.0"]
|
|
21
|
+
ai = ["openai>=1.40"]
|
|
22
|
+
dev = ["pytest>=8", "pytest-cov>=5", "ruff>=0.6"]
|
|
23
|
+
all = ["pyarrow>=15","duckdb>=1.0","sqlalchemy>=2.0","psycopg[binary]>=3.1","great_expectations>=1.0","openai>=1.40"]
|
|
24
|
+
|
|
25
|
+
[project.scripts]
|
|
26
|
+
dataguard = "dataguard.cli:app"
|
|
27
|
+
|
|
28
|
+
[tool.setuptools.packages.find]
|
|
29
|
+
where = ["src"]
|
|
30
|
+
|
|
31
|
+
[tool.pytest.ini_options]
|
|
32
|
+
pythonpath = ["src"]
|
|
33
|
+
testpaths = ["tests"]
|
|
34
|
+
|
|
35
|
+
[tool.ruff]
|
|
36
|
+
line-length = 110
|
|
37
|
+
target-version = "py310"
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.2.0"
|