dataguard-ai 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. dataguard_ai-0.2.0/LICENSE +17 -0
  2. dataguard_ai-0.2.0/PKG-INFO +338 -0
  3. dataguard_ai-0.2.0/README.md +299 -0
  4. dataguard_ai-0.2.0/pyproject.toml +37 -0
  5. dataguard_ai-0.2.0/setup.cfg +4 -0
  6. dataguard_ai-0.2.0/src/dataguard/__init__.py +1 -0
  7. dataguard_ai-0.2.0/src/dataguard/cli.py +135 -0
  8. dataguard_ai-0.2.0/src/dataguard/config.py +22 -0
  9. dataguard_ai-0.2.0/src/dataguard/demo.py +54 -0
  10. dataguard_ai-0.2.0/src/dataguard/drift.py +34 -0
  11. dataguard_ai-0.2.0/src/dataguard/explain.py +42 -0
  12. dataguard_ai-0.2.0/src/dataguard/exporters.py +50 -0
  13. dataguard_ai-0.2.0/src/dataguard/gx_exporter.py +15 -0
  14. dataguard_ai-0.2.0/src/dataguard/html_report.py +42 -0
  15. dataguard_ai-0.2.0/src/dataguard/io.py +22 -0
  16. dataguard_ai-0.2.0/src/dataguard/models.py +36 -0
  17. dataguard_ai-0.2.0/src/dataguard/pii.py +47 -0
  18. dataguard_ai-0.2.0/src/dataguard/rules.py +83 -0
  19. dataguard_ai-0.2.0/src/dataguard/scanner.py +140 -0
  20. dataguard_ai-0.2.0/src/dataguard/sources.py +39 -0
  21. dataguard_ai-0.2.0/src/dataguard_ai.egg-info/PKG-INFO +338 -0
  22. dataguard_ai-0.2.0/src/dataguard_ai.egg-info/SOURCES.txt +32 -0
  23. dataguard_ai-0.2.0/src/dataguard_ai.egg-info/dependency_links.txt +1 -0
  24. dataguard_ai-0.2.0/src/dataguard_ai.egg-info/entry_points.txt +2 -0
  25. dataguard_ai-0.2.0/src/dataguard_ai.egg-info/requires.txt +34 -0
  26. dataguard_ai-0.2.0/src/dataguard_ai.egg-info/top_level.txt +1 -0
  27. dataguard_ai-0.2.0/tests/test_drift.py +12 -0
  28. dataguard_ai-0.2.0/tests/test_exporters.py +13 -0
  29. dataguard_ai-0.2.0/tests/test_gx.py +12 -0
  30. dataguard_ai-0.2.0/tests/test_html.py +10 -0
  31. dataguard_ai-0.2.0/tests/test_pii.py +12 -0
  32. dataguard_ai-0.2.0/tests/test_rules.py +11 -0
  33. dataguard_ai-0.2.0/tests/test_scanner.py +23 -0
  34. dataguard_ai-0.2.0/tests/test_sources.py +9 -0
@@ -0,0 +1,17 @@
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ Copyright 2026 Aditya Ranjan
6
+
7
+ Licensed under the Apache License, Version 2.0 (the "License");
8
+ you may not use this file except in compliance with the License.
9
+ You may obtain a copy of the License at
10
+
11
+ http://www.apache.org/licenses/LICENSE-2.0
12
+
13
+ Unless required by applicable law or agreed to in writing, software
14
+ distributed under the License is distributed on an "AS IS" BASIS,
15
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
16
+ See the License for the specific language governing permissions and
17
+ limitations under the License.
@@ -0,0 +1,338 @@
1
+ Metadata-Version: 2.4
2
+ Name: dataguard-ai
3
+ Version: 0.2.0
4
+ Summary: Open-source data quality and governance copilot: scan data, detect risks, explain findings, and generate fixes.
5
+ Author: Aditya Ranjan
6
+ License: Apache-2.0
7
+ Keywords: data-quality,data-governance,data-contracts,pii,dbt,great-expectations,duckdb,postgresql,ai
8
+ Requires-Python: >=3.10
9
+ Description-Content-Type: text/markdown
10
+ License-File: LICENSE
11
+ Requires-Dist: pandas>=2.0
12
+ Requires-Dist: typer>=0.12
13
+ Requires-Dist: rich>=13.7
14
+ Requires-Dist: pydantic>=2.7
15
+ Requires-Dist: PyYAML>=6.0
16
+ Provides-Extra: parquet
17
+ Requires-Dist: pyarrow>=15; extra == "parquet"
18
+ Provides-Extra: duckdb
19
+ Requires-Dist: duckdb>=1.0; extra == "duckdb"
20
+ Provides-Extra: postgres
21
+ Requires-Dist: sqlalchemy>=2.0; extra == "postgres"
22
+ Requires-Dist: psycopg[binary]>=3.1; extra == "postgres"
23
+ Provides-Extra: gx
24
+ Requires-Dist: great_expectations>=1.0; extra == "gx"
25
+ Provides-Extra: ai
26
+ Requires-Dist: openai>=1.40; extra == "ai"
27
+ Provides-Extra: dev
28
+ Requires-Dist: pytest>=8; extra == "dev"
29
+ Requires-Dist: pytest-cov>=5; extra == "dev"
30
+ Requires-Dist: ruff>=0.6; extra == "dev"
31
+ Provides-Extra: all
32
+ Requires-Dist: pyarrow>=15; extra == "all"
33
+ Requires-Dist: duckdb>=1.0; extra == "all"
34
+ Requires-Dist: sqlalchemy>=2.0; extra == "all"
35
+ Requires-Dist: psycopg[binary]>=3.1; extra == "all"
36
+ Requires-Dist: great_expectations>=1.0; extra == "all"
37
+ Requires-Dist: openai>=1.40; extra == "all"
38
+ Dynamic: license-file
39
+
40
+ # DataGuard AI
41
+
42
+ [![CI](https://github.com/adiranjan25/dataguard-ai/actions/workflows/ci.yml/badge.svg)](https://github.com/adiranjan25/dataguard-ai/actions/workflows/ci.yml)
43
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue.svg)](https://www.python.org/)
44
+ [![License: Apache-2.0](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](LICENSE)
45
+ [![Status: Public Beta](https://img.shields.io/badge/status-public%20beta-orange.svg)](#project-status)
46
+
47
+ **Open-source, AI-assisted data quality and governance for modern data platforms.**
48
+
49
+ > **Scan your data. Find quality and governance risks. Understand why they matter. Generate the fix.**
50
+
51
+ DataGuard AI is a developer-first toolkit for **data quality, data governance, PII discovery, schema drift, data contracts, and CI/CD-friendly validation**. Detection is deterministic and does **not** require an LLM. Optional AI assistance can explain structured findings and suggest remediation without making quality detection dependent on a model.
52
+
53
+ **Current version: v0.2.0 public beta.**
54
+
55
+ ## Why DataGuard AI?
56
+
57
+ Data teams often manage quality rules, contracts, PII checks, schema drift, metadata, and AI assistants in separate workflows. DataGuard AI provides a lightweight layer developers can run locally or in CI to surface these risks through one interface.
58
+
59
+ ### What v0.2 can do
60
+
61
+ - Profile CSV and JSON data, with optional Parquet support
62
+ - Scan **DuckDB** and **PostgreSQL** tables
63
+ - Run reusable **YAML data-quality rules**
64
+ - Detect nulls, duplicate rows, uniqueness issues, range violations, and freshness risks
65
+ - Identify likely PII using value and column-name signals
66
+ - Compute transparent quality and governance scores
67
+ - Generate standalone **HTML** and machine-readable **JSON** reports
68
+ - Snapshot schemas and detect schema drift
69
+ - Generate starter **dbt tests** and YAML data contracts
70
+ - Export portable **Great Expectations** expectation configuration
71
+ - Run in **GitHub Actions**
72
+ - Explain findings locally, with optional AI-assisted remediation guidance
73
+
74
+ ## Quick start
75
+
76
+ ### Install from source during the public beta
77
+
78
+ Until DataGuard AI is published to PyPI, clone the repository and install it locally:
79
+
80
+ ```bash
81
+ git clone https://github.com/adiranjan25/dataguard-ai.git
82
+ cd dataguard-ai
83
+ python3 -m venv .venv
84
+ source .venv/bin/activate
85
+ python -m pip install -e .
86
+ ```
87
+
88
+ Run the synthetic retail demo:
89
+
90
+ ```bash
91
+ dataguard demo --rows 100
92
+ ```
93
+
94
+ Or scan your own dataset:
95
+
96
+ ```bash
97
+ dataguard scan data/customers.csv
98
+ ```
99
+
100
+ Generate JSON and HTML reports:
101
+
102
+ ```bash
103
+ dataguard scan data/customers.csv --json-out report.json --html-out report.html
104
+ ```
105
+
106
+ > **Coming next:** after the PyPI release, installation will become simply `pip install dataguard-ai`.
107
+
108
+ ## Detection philosophy
109
+
110
+ > **AI assists; deterministic and statistical checks detect and verify.**
111
+
112
+ The default scanning path does not require an LLM. This keeps findings reproducible and allows teams to use DataGuard AI without sending raw production datasets to an external model.
113
+
114
+ ## YAML rule engine
115
+
116
+ Supported v0.2 custom rule types:
117
+
118
+ - `not_null`
119
+ - `unique`
120
+ - `accepted_values`
121
+ - `between`
122
+ - `regex`
123
+ - `max_null_pct`
124
+ - `row_count_between`
125
+
126
+ Example:
127
+
128
+ ```yaml
129
+ quality:
130
+ max_null_pct: 5
131
+ governance:
132
+ owner: data-platform@example.com
133
+ rules:
134
+ - type: not_null
135
+ column: customer_id
136
+ severity: CRITICAL
137
+ - type: unique
138
+ column: customer_id
139
+ - type: accepted_values
140
+ column: state
141
+ values: [TX, CA, NY]
142
+ - type: between
143
+ column: amount
144
+ min: 0
145
+ max: 100000
146
+ ```
147
+
148
+ ```bash
149
+ dataguard scan customers.csv --config dataguard.yml --html-out report.html
150
+ ```
151
+
152
+ ## Database scanning
153
+
154
+ ### DuckDB
155
+
156
+ ```bash
157
+ python -m pip install -e ".[duckdb]"
158
+ dataguard scan-duckdb analytics.duckdb --table customers --html-out report.html
159
+ ```
160
+
161
+ ### PostgreSQL
162
+
163
+ ```bash
164
+ python -m pip install -e ".[postgres]"
165
+ export DATAGUARD_POSTGRES_URL='postgresql+psycopg://user:password@host/database'
166
+ dataguard scan-postgres --table public.customers --html-out report.html
167
+ ```
168
+
169
+ > **Current limitation:** database scans load the selected table/result into memory. Warehouse-scale pushdown profiling is a roadmap item.
170
+
171
+ ## Reports
172
+
173
+ ```bash
174
+ dataguard scan customers.csv --html-out report.html
175
+ dataguard scan customers.csv --json-out report.json
176
+ ```
177
+
178
+ The HTML report includes quality/governance scores, findings, severity, affected columns, suggested remediation, column profiles, and detected PII.
179
+
180
+ ## Data contracts and dbt
181
+
182
+ ```bash
183
+ dataguard contract data/customers.csv --out contract.yml
184
+ dataguard generate-dbt data/customers.csv --out schema.yml
185
+ ```
186
+
187
+ These outputs are intended as reviewable starting points rather than replacements for domain-specific contract design.
188
+
189
+ ## Great Expectations
190
+
191
+ ```bash
192
+ dataguard generate-gx data/customers.csv --out gx-expectations.json
193
+ ```
194
+
195
+ The exporter deliberately produces reviewable configuration rather than modifying an existing Great Expectations project.
196
+
197
+ ## Schema drift
198
+
199
+ ```bash
200
+ dataguard snapshot data/customers.csv --out baseline.json
201
+ dataguard drift data/customers_v2.csv --baseline baseline.json
202
+ ```
203
+
204
+ ## GitHub Actions
205
+
206
+ The repository includes an example workflow at `.github/workflows/dataguard.yml` that demonstrates scanning sample data and uploading JSON/HTML reports as workflow artifacts.
207
+
208
+ Project CI separately runs linting and automated tests against **Python 3.10, 3.11, and 3.12**.
209
+
210
+ ## Optional AI explanations
211
+
212
+ Local deterministic explanations are available without an external model.
213
+
214
+ ```bash
215
+ python -m pip install -e ".[ai]"
216
+ export OPENAI_API_KEY=...
217
+ dataguard explain report.json --provider openai
218
+ ```
219
+
220
+ The included provider sends structured findings rather than raw dataset rows. Always review your organization's security, privacy, and data-handling requirements before enabling an external provider.
221
+
222
+ ## Architecture
223
+
224
+ ```text
225
+ Data sources
226
+ |
227
+ +-----------------+-----------------+
228
+ | | |
229
+ CSV / JSON DuckDB PostgreSQL
230
+ / Parquet | |
231
+ +-----------------+-----------------+
232
+ |
233
+ v
234
+ DataGuard scanner
235
+ |
236
+ +----------------+----------------+
237
+ | | |
238
+ Profiling Rule engine PII detection
239
+ | | |
240
+ +----------------+----------------+
241
+ |
242
+ v
243
+ Quality + governance
244
+ |
245
+ +-----------+-----+------+-----------+
246
+ | | | |
247
+ v v v v
248
+ CLI JSON HTML Contracts / dbt / GX
249
+ ```
250
+
251
+ Detection and optional AI explanation are intentionally separated.
252
+
253
+ ## Commands
254
+
255
+ | Command | Purpose |
256
+ |---|---|
257
+ | `dataguard scan PATH` | Profile file data and report quality/governance findings |
258
+ | `dataguard demo` | Generate and scan synthetic retail datasets |
259
+ | `dataguard scan-duckdb DATABASE --table TABLE` | Scan a DuckDB table |
260
+ | `dataguard scan-postgres --table TABLE` | Scan a PostgreSQL table |
261
+ | `dataguard contract PATH` | Generate a starter data contract |
262
+ | `dataguard generate-dbt PATH` | Generate starter dbt tests |
263
+ | `dataguard generate-gx PATH` | Generate portable GX expectation configuration |
264
+ | `dataguard snapshot PATH` | Save a schema/profile baseline |
265
+ | `dataguard drift PATH --baseline FILE` | Compare current schema with a baseline |
266
+ | `dataguard explain REPORT.json` | Explain findings locally or with optional AI |
267
+
268
+ Run `dataguard --help` for current CLI options.
269
+
270
+ ## Python API
271
+
272
+ ```python
273
+ from dataguard.scanner import scan_path
274
+
275
+ report = scan_path("customers.csv")
276
+ print(report.quality_score)
277
+
278
+ for finding in report.findings:
279
+ print(finding.severity, finding.column, finding.message)
280
+ ```
281
+
282
+ ## Retail demo
283
+
284
+ The bundled synthetic retail demo creates customer, order, and inventory datasets with intentionally injected quality/governance problems so developers can explore DataGuard AI without providing proprietary data.
285
+
286
+ ```bash
287
+ dataguard demo --rows 5000
288
+ ```
289
+
290
+ ## Project status
291
+
292
+ DataGuard AI is currently a **public beta (v0.2.0)**. The API, configuration schema, scoring model, and command behavior may evolve before v1.0.
293
+
294
+ The project is suitable for experimentation, development workflows, demos, and community feedback. Evaluate it against your own requirements before using it as a production control.
295
+
296
+ ## Roadmap
297
+
298
+ ### v0.3 — Metadata and context
299
+
300
+ - OpenMetadata integration
301
+ - DataHub integration
302
+ - dbt artifact ingestion
303
+ - Dataset-to-dataset referential checks
304
+ - Richer statistical drift and anomaly detection
305
+
306
+ ### v0.4 — Agent access
307
+
308
+ - MCP server
309
+ - Agent-accessible quality, contract, and governance tools
310
+ - Ollama/local-model provider
311
+ - Governance context for AI agents
312
+ - Assisted remediation workflows
313
+
314
+ ### Toward v1.0
315
+
316
+ - PyPI distribution and automated release workflow
317
+ - Warehouse-scale profiling/pushdown
318
+ - Broader integration tests
319
+ - Benchmark datasets and reproducible evaluation
320
+ - Stable configuration and CLI contracts
321
+
322
+ ## Contributing
323
+
324
+ Contributions are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md) for development setup and contribution guidance.
325
+
326
+ Useful first contributions include additional PII detectors, report/export formats, documentation improvements, tests, database adapters, and integration examples.
327
+
328
+ ## Security and privacy
329
+
330
+ - Never submit secrets, credentials, or proprietary production datasets in GitHub issues.
331
+ - Use environment variables or an appropriate secrets manager for database and model credentials.
332
+ - Treat detected PII findings as sensitive operational metadata.
333
+ - Review organizational security/privacy requirements before using an external AI provider.
334
+ - See [SECURITY.md](SECURITY.md) for vulnerability-reporting guidance.
335
+
336
+ ## License
337
+
338
+ DataGuard AI is licensed under the **Apache License 2.0**. See [LICENSE](LICENSE).
@@ -0,0 +1,299 @@
1
+ # DataGuard AI
2
+
3
+ [![CI](https://github.com/adiranjan25/dataguard-ai/actions/workflows/ci.yml/badge.svg)](https://github.com/adiranjan25/dataguard-ai/actions/workflows/ci.yml)
4
+ [![Python 3.10+](https://img.shields.io/badge/python-3.10%2B-blue.svg)](https://www.python.org/)
5
+ [![License: Apache-2.0](https://img.shields.io/badge/License-Apache%202.0-blue.svg)](LICENSE)
6
+ [![Status: Public Beta](https://img.shields.io/badge/status-public%20beta-orange.svg)](#project-status)
7
+
8
+ **Open-source, AI-assisted data quality and governance for modern data platforms.**
9
+
10
+ > **Scan your data. Find quality and governance risks. Understand why they matter. Generate the fix.**
11
+
12
+ DataGuard AI is a developer-first toolkit for **data quality, data governance, PII discovery, schema drift, data contracts, and CI/CD-friendly validation**. Detection is deterministic and does **not** require an LLM. Optional AI assistance can explain structured findings and suggest remediation without making quality detection dependent on a model.
13
+
14
+ **Current version: v0.2.0 public beta.**
15
+
16
+ ## Why DataGuard AI?
17
+
18
+ Data teams often manage quality rules, contracts, PII checks, schema drift, metadata, and AI assistants in separate workflows. DataGuard AI provides a lightweight layer developers can run locally or in CI to surface these risks through one interface.
19
+
20
+ ### What v0.2 can do
21
+
22
+ - Profile CSV and JSON data, with optional Parquet support
23
+ - Scan **DuckDB** and **PostgreSQL** tables
24
+ - Run reusable **YAML data-quality rules**
25
+ - Detect nulls, duplicate rows, uniqueness issues, range violations, and freshness risks
26
+ - Identify likely PII using value and column-name signals
27
+ - Compute transparent quality and governance scores
28
+ - Generate standalone **HTML** and machine-readable **JSON** reports
29
+ - Snapshot schemas and detect schema drift
30
+ - Generate starter **dbt tests** and YAML data contracts
31
+ - Export portable **Great Expectations** expectation configuration
32
+ - Run in **GitHub Actions**
33
+ - Explain findings locally, with optional AI-assisted remediation guidance
34
+
35
+ ## Quick start
36
+
37
+ ### Install from source during the public beta
38
+
39
+ Until DataGuard AI is published to PyPI, clone the repository and install it locally:
40
+
41
+ ```bash
42
+ git clone https://github.com/adiranjan25/dataguard-ai.git
43
+ cd dataguard-ai
44
+ python3 -m venv .venv
45
+ source .venv/bin/activate
46
+ python -m pip install -e .
47
+ ```
48
+
49
+ Run the synthetic retail demo:
50
+
51
+ ```bash
52
+ dataguard demo --rows 100
53
+ ```
54
+
55
+ Or scan your own dataset:
56
+
57
+ ```bash
58
+ dataguard scan data/customers.csv
59
+ ```
60
+
61
+ Generate JSON and HTML reports:
62
+
63
+ ```bash
64
+ dataguard scan data/customers.csv --json-out report.json --html-out report.html
65
+ ```
66
+
67
+ > **Coming next:** after the PyPI release, installation will become simply `pip install dataguard-ai`.
68
+
69
+ ## Detection philosophy
70
+
71
+ > **AI assists; deterministic and statistical checks detect and verify.**
72
+
73
+ The default scanning path does not require an LLM. This keeps findings reproducible and allows teams to use DataGuard AI without sending raw production datasets to an external model.
74
+
75
+ ## YAML rule engine
76
+
77
+ Supported v0.2 custom rule types:
78
+
79
+ - `not_null`
80
+ - `unique`
81
+ - `accepted_values`
82
+ - `between`
83
+ - `regex`
84
+ - `max_null_pct`
85
+ - `row_count_between`
86
+
87
+ Example:
88
+
89
+ ```yaml
90
+ quality:
91
+ max_null_pct: 5
92
+ governance:
93
+ owner: data-platform@example.com
94
+ rules:
95
+ - type: not_null
96
+ column: customer_id
97
+ severity: CRITICAL
98
+ - type: unique
99
+ column: customer_id
100
+ - type: accepted_values
101
+ column: state
102
+ values: [TX, CA, NY]
103
+ - type: between
104
+ column: amount
105
+ min: 0
106
+ max: 100000
107
+ ```
108
+
109
+ ```bash
110
+ dataguard scan customers.csv --config dataguard.yml --html-out report.html
111
+ ```
112
+
113
+ ## Database scanning
114
+
115
+ ### DuckDB
116
+
117
+ ```bash
118
+ python -m pip install -e ".[duckdb]"
119
+ dataguard scan-duckdb analytics.duckdb --table customers --html-out report.html
120
+ ```
121
+
122
+ ### PostgreSQL
123
+
124
+ ```bash
125
+ python -m pip install -e ".[postgres]"
126
+ export DATAGUARD_POSTGRES_URL='postgresql+psycopg://user:password@host/database'
127
+ dataguard scan-postgres --table public.customers --html-out report.html
128
+ ```
129
+
130
+ > **Current limitation:** database scans load the selected table/result into memory. Warehouse-scale pushdown profiling is a roadmap item.
131
+
132
+ ## Reports
133
+
134
+ ```bash
135
+ dataguard scan customers.csv --html-out report.html
136
+ dataguard scan customers.csv --json-out report.json
137
+ ```
138
+
139
+ The HTML report includes quality/governance scores, findings, severity, affected columns, suggested remediation, column profiles, and detected PII.
140
+
141
+ ## Data contracts and dbt
142
+
143
+ ```bash
144
+ dataguard contract data/customers.csv --out contract.yml
145
+ dataguard generate-dbt data/customers.csv --out schema.yml
146
+ ```
147
+
148
+ These outputs are intended as reviewable starting points rather than replacements for domain-specific contract design.
149
+
150
+ ## Great Expectations
151
+
152
+ ```bash
153
+ dataguard generate-gx data/customers.csv --out gx-expectations.json
154
+ ```
155
+
156
+ The exporter deliberately produces reviewable configuration rather than modifying an existing Great Expectations project.
157
+
158
+ ## Schema drift
159
+
160
+ ```bash
161
+ dataguard snapshot data/customers.csv --out baseline.json
162
+ dataguard drift data/customers_v2.csv --baseline baseline.json
163
+ ```
164
+
165
+ ## GitHub Actions
166
+
167
+ The repository includes an example workflow at `.github/workflows/dataguard.yml` that demonstrates scanning sample data and uploading JSON/HTML reports as workflow artifacts.
168
+
169
+ Project CI separately runs linting and automated tests against **Python 3.10, 3.11, and 3.12**.
170
+
171
+ ## Optional AI explanations
172
+
173
+ Local deterministic explanations are available without an external model.
174
+
175
+ ```bash
176
+ python -m pip install -e ".[ai]"
177
+ export OPENAI_API_KEY=...
178
+ dataguard explain report.json --provider openai
179
+ ```
180
+
181
+ The included provider sends structured findings rather than raw dataset rows. Always review your organization's security, privacy, and data-handling requirements before enabling an external provider.
182
+
183
+ ## Architecture
184
+
185
+ ```text
186
+ Data sources
187
+ |
188
+ +-----------------+-----------------+
189
+ | | |
190
+ CSV / JSON DuckDB PostgreSQL
191
+ / Parquet | |
192
+ +-----------------+-----------------+
193
+ |
194
+ v
195
+ DataGuard scanner
196
+ |
197
+ +----------------+----------------+
198
+ | | |
199
+ Profiling Rule engine PII detection
200
+ | | |
201
+ +----------------+----------------+
202
+ |
203
+ v
204
+ Quality + governance
205
+ |
206
+ +-----------+-----+------+-----------+
207
+ | | | |
208
+ v v v v
209
+ CLI JSON HTML Contracts / dbt / GX
210
+ ```
211
+
212
+ Detection and optional AI explanation are intentionally separated.
213
+
214
+ ## Commands
215
+
216
+ | Command | Purpose |
217
+ |---|---|
218
+ | `dataguard scan PATH` | Profile file data and report quality/governance findings |
219
+ | `dataguard demo` | Generate and scan synthetic retail datasets |
220
+ | `dataguard scan-duckdb DATABASE --table TABLE` | Scan a DuckDB table |
221
+ | `dataguard scan-postgres --table TABLE` | Scan a PostgreSQL table |
222
+ | `dataguard contract PATH` | Generate a starter data contract |
223
+ | `dataguard generate-dbt PATH` | Generate starter dbt tests |
224
+ | `dataguard generate-gx PATH` | Generate portable GX expectation configuration |
225
+ | `dataguard snapshot PATH` | Save a schema/profile baseline |
226
+ | `dataguard drift PATH --baseline FILE` | Compare current schema with a baseline |
227
+ | `dataguard explain REPORT.json` | Explain findings locally or with optional AI |
228
+
229
+ Run `dataguard --help` for current CLI options.
230
+
231
+ ## Python API
232
+
233
+ ```python
234
+ from dataguard.scanner import scan_path
235
+
236
+ report = scan_path("customers.csv")
237
+ print(report.quality_score)
238
+
239
+ for finding in report.findings:
240
+ print(finding.severity, finding.column, finding.message)
241
+ ```
242
+
243
+ ## Retail demo
244
+
245
+ The bundled synthetic retail demo creates customer, order, and inventory datasets with intentionally injected quality/governance problems so developers can explore DataGuard AI without providing proprietary data.
246
+
247
+ ```bash
248
+ dataguard demo --rows 5000
249
+ ```
250
+
251
+ ## Project status
252
+
253
+ DataGuard AI is currently a **public beta (v0.2.0)**. The API, configuration schema, scoring model, and command behavior may evolve before v1.0.
254
+
255
+ The project is suitable for experimentation, development workflows, demos, and community feedback. Evaluate it against your own requirements before using it as a production control.
256
+
257
+ ## Roadmap
258
+
259
+ ### v0.3 — Metadata and context
260
+
261
+ - OpenMetadata integration
262
+ - DataHub integration
263
+ - dbt artifact ingestion
264
+ - Dataset-to-dataset referential checks
265
+ - Richer statistical drift and anomaly detection
266
+
267
+ ### v0.4 — Agent access
268
+
269
+ - MCP server
270
+ - Agent-accessible quality, contract, and governance tools
271
+ - Ollama/local-model provider
272
+ - Governance context for AI agents
273
+ - Assisted remediation workflows
274
+
275
+ ### Toward v1.0
276
+
277
+ - PyPI distribution and automated release workflow
278
+ - Warehouse-scale profiling/pushdown
279
+ - Broader integration tests
280
+ - Benchmark datasets and reproducible evaluation
281
+ - Stable configuration and CLI contracts
282
+
283
+ ## Contributing
284
+
285
+ Contributions are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md) for development setup and contribution guidance.
286
+
287
+ Useful first contributions include additional PII detectors, report/export formats, documentation improvements, tests, database adapters, and integration examples.
288
+
289
+ ## Security and privacy
290
+
291
+ - Never submit secrets, credentials, or proprietary production datasets in GitHub issues.
292
+ - Use environment variables or an appropriate secrets manager for database and model credentials.
293
+ - Treat detected PII findings as sensitive operational metadata.
294
+ - Review organizational security/privacy requirements before using an external AI provider.
295
+ - See [SECURITY.md](SECURITY.md) for vulnerability-reporting guidance.
296
+
297
+ ## License
298
+
299
+ DataGuard AI is licensed under the **Apache License 2.0**. See [LICENSE](LICENSE).
@@ -0,0 +1,37 @@
1
+ [build-system]
2
+ requires = ["setuptools>=69", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "dataguard-ai"
7
+ version = "0.2.0"
8
+ description = "Open-source data quality and governance copilot: scan data, detect risks, explain findings, and generate fixes."
9
+ readme = "README.md"
10
+ requires-python = ">=3.10"
11
+ license = {text = "Apache-2.0"}
12
+ authors = [{name = "Aditya Ranjan"}]
13
+ keywords = ["data-quality", "data-governance", "data-contracts", "pii", "dbt", "great-expectations", "duckdb", "postgresql", "ai"]
14
+ dependencies = ["pandas>=2.0","typer>=0.12","rich>=13.7","pydantic>=2.7","PyYAML>=6.0"]
15
+
16
+ [project.optional-dependencies]
17
+ parquet = ["pyarrow>=15"]
18
+ duckdb = ["duckdb>=1.0"]
19
+ postgres = ["sqlalchemy>=2.0", "psycopg[binary]>=3.1"]
20
+ gx = ["great_expectations>=1.0"]
21
+ ai = ["openai>=1.40"]
22
+ dev = ["pytest>=8", "pytest-cov>=5", "ruff>=0.6"]
23
+ all = ["pyarrow>=15","duckdb>=1.0","sqlalchemy>=2.0","psycopg[binary]>=3.1","great_expectations>=1.0","openai>=1.40"]
24
+
25
+ [project.scripts]
26
+ dataguard = "dataguard.cli:app"
27
+
28
+ [tool.setuptools.packages.find]
29
+ where = ["src"]
30
+
31
+ [tool.pytest.ini_options]
32
+ pythonpath = ["src"]
33
+ testpaths = ["tests"]
34
+
35
+ [tool.ruff]
36
+ line-length = 110
37
+ target-version = "py310"
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1 @@
1
+ __version__ = "0.2.0"