agent-paranoid-android 0.5.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- agent_paranoid_android-0.5.0.dist-info/METADATA +1092 -0
- agent_paranoid_android-0.5.0.dist-info/RECORD +77 -0
- agent_paranoid_android-0.5.0.dist-info/WHEEL +4 -0
- agent_paranoid_android-0.5.0.dist-info/entry_points.txt +4 -0
- agent_paranoid_android-0.5.0.dist-info/licenses/LICENSE +21 -0
- test_data_agent/__init__.py +75 -0
- test_data_agent/adapters/__init__.py +61 -0
- test_data_agent/adapters/csv_file.py +67 -0
- test_data_agent/adapters/csv_folder.py +82 -0
- test_data_agent/adapters/json_profile.py +111 -0
- test_data_agent/adapters/legacy_generation.py +448 -0
- test_data_agent/adapters/legacy_profile.py +257 -0
- test_data_agent/adapters/parquet_dataset.py +95 -0
- test_data_agent/adapters/trino_profile.py +43 -0
- test_data_agent/agent.py +325 -0
- test_data_agent/business_rules.py +33 -0
- test_data_agent/business_validator.py +9 -0
- test_data_agent/cli.py +468 -0
- test_data_agent/compat/__init__.py +61 -0
- test_data_agent/compat/commands.py +56 -0
- test_data_agent/compat/legacy_generation.py +39 -0
- test_data_agent/compat/legacy_outputs.py +53 -0
- test_data_agent/compat/legacy_spec.py +22 -0
- test_data_agent/compat/legacy_workflows.py +85 -0
- test_data_agent/core/__init__.py +76 -0
- test_data_agent/core/constraint.py +66 -0
- test_data_agent/core/dataset.py +126 -0
- test_data_agent/core/distribution.py +178 -0
- test_data_agent/core/entity.py +62 -0
- test_data_agent/core/field.py +73 -0
- test_data_agent/core/limits.py +307 -0
- test_data_agent/core/privacy.py +221 -0
- test_data_agent/core/relationship.py +22 -0
- test_data_agent/core/serialization.py +39 -0
- test_data_agent/core/settings.py +41 -0
- test_data_agent/csv_profiler.py +445 -0
- test_data_agent/generation/__init__.py +7 -0
- test_data_agent/generation/constraint_solver.py +165 -0
- test_data_agent/generation/entity_generator.py +265 -0
- test_data_agent/generation/planner.py +41 -0
- test_data_agent/generator.py +152 -0
- test_data_agent/io/__init__.py +92 -0
- test_data_agent/io/artifacts.py +165 -0
- test_data_agent/io/commands.py +338 -0
- test_data_agent/io/legacy_workflows.py +11 -0
- test_data_agent/io/readers.py +69 -0
- test_data_agent/io/workflows.py +554 -0
- test_data_agent/io/writers.py +156 -0
- test_data_agent/mcp_generator_server.py +403 -0
- test_data_agent/mcp_trino_server.py +1051 -0
- test_data_agent/profiling/__init__.py +49 -0
- test_data_agent/profiling/cache.py +85 -0
- test_data_agent/profiling/constraint_miner.py +165 -0
- test_data_agent/profiling/distribution_profiler.py +83 -0
- test_data_agent/profiling/relationship_profiler.py +53 -0
- test_data_agent/profiling/schema_profiler.py +325 -0
- test_data_agent/py.typed +1 -0
- test_data_agent/rules/__init__.py +33 -0
- test_data_agent/rules/business_config.py +106 -0
- test_data_agent/rules/conditions.py +37 -0
- test_data_agent/rules/contract.py +365 -0
- test_data_agent/rules/engine.py +157 -0
- test_data_agent/rules/expressions.py +148 -0
- test_data_agent/rules/models.py +210 -0
- test_data_agent/rules/scenarios.py +32 -0
- test_data_agent/rules/validation.py +184 -0
- test_data_agent/rules_engine.py +6 -0
- test_data_agent/safety.py +128 -0
- test_data_agent/scenario.py +5 -0
- test_data_agent/spec.py +355 -0
- test_data_agent/validation/__init__.py +5 -0
- test_data_agent/validation/constraint_validator.py +123 -0
- test_data_agent/validation/reconciliation.py +40 -0
- test_data_agent/validation/relationship_validator.py +34 -0
- test_data_agent/validation/schema_validator.py +67 -0
- test_data_agent/validator.py +129 -0
- test_data_agent/version.py +3 -0
|
@@ -0,0 +1,1092 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: agent-paranoid-android
|
|
3
|
+
Version: 0.5.0
|
|
4
|
+
Summary: Safety-first synthetic data generation agent
|
|
5
|
+
Project-URL: Homepage, https://github.com/wa-pis/agent-paranoid-android
|
|
6
|
+
Project-URL: Repository, https://github.com/wa-pis/agent-paranoid-android
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: agent,csv,mcp,pii,synthetic-data,test-data,trino
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Topic :: Database
|
|
17
|
+
Classifier: Topic :: Security
|
|
18
|
+
Classifier: Topic :: Software Development :: Testing
|
|
19
|
+
Classifier: Typing :: Typed
|
|
20
|
+
Requires-Python: >=3.11
|
|
21
|
+
Requires-Dist: faker>=25.0.0
|
|
22
|
+
Requires-Dist: mcp>=1.0.0
|
|
23
|
+
Requires-Dist: pyarrow>=15.0.0
|
|
24
|
+
Requires-Dist: pydantic>=2.7.0
|
|
25
|
+
Requires-Dist: pyyaml>=6.0.0
|
|
26
|
+
Requires-Dist: sqlglot>=30.0.0
|
|
27
|
+
Requires-Dist: trino>=0.330.0
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: hatchling==1.31.0; extra == 'dev'
|
|
30
|
+
Requires-Dist: hypothesis>=6.100.0; extra == 'dev'
|
|
31
|
+
Requires-Dist: mypy>=2.3.0; extra == 'dev'
|
|
32
|
+
Requires-Dist: pip-audit>=2.7.0; extra == 'dev'
|
|
33
|
+
Requires-Dist: pytest-cov>=5.0.0; extra == 'dev'
|
|
34
|
+
Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
35
|
+
Requires-Dist: ruff>=0.15.0; extra == 'dev'
|
|
36
|
+
Requires-Dist: types-pyyaml>=6.0.12; extra == 'dev'
|
|
37
|
+
Description-Content-Type: text/markdown
|
|
38
|
+
|
|
39
|
+
# Agent Paranoid Android
|
|
40
|
+
|
|
41
|
+
[](https://github.com/wa-pis/agent-paranoid-android/actions/workflows/ci.yml)
|
|
42
|
+
[](LICENSE)
|
|
43
|
+
|
|
44
|
+
Safe, deterministic synthetic data generation for database and CSV-driven test
|
|
45
|
+
datasets.
|
|
46
|
+
|
|
47
|
+
The agent profiles schemas and safe aggregate metadata, detects likely
|
|
48
|
+
sensitive fields, builds generation specs, generates synthetic rows from an
|
|
49
|
+
explicit seed, validates the result, and exports data in common formats. It
|
|
50
|
+
never copies source rows into generated output.
|
|
51
|
+
|
|
52
|
+
The project name is a nod to [Radiohead](https://www.radiohead.com/deadairspace/)
|
|
53
|
+
and ["Paranoid Android"](https://music.apple.com/us/song/1097861770). This
|
|
54
|
+
project is unaffiliated with Radiohead, XL Recordings, Parlophone, or any
|
|
55
|
+
related rights holders.
|
|
56
|
+
|
|
57
|
+
## Project Status
|
|
58
|
+
|
|
59
|
+
Current package version: `0.5.0`. The installable distribution is
|
|
60
|
+
`agent-paranoid-android`; the CLI remains `test-data-agent` for compatibility
|
|
61
|
+
and to keep the command focused on the generated-data use case.
|
|
62
|
+
|
|
63
|
+
The domain-agnostic `DatasetSpec` pipeline is the primary path for new work. It
|
|
64
|
+
supports safe CSV folder profiling, spec inference, deterministic multi-table
|
|
65
|
+
generation, constraint reconciliation, validation, and CSV/JSON/Parquet export.
|
|
66
|
+
The generator MCP server now exposes the same profile, infer, generate, validate,
|
|
67
|
+
and export workflow to AI clients without returning dataset rows through MCP.
|
|
68
|
+
|
|
69
|
+
Legacy `GenerationSpec` commands and imports remain available for compatibility,
|
|
70
|
+
but they emit deprecation warnings and should be treated as migration paths.
|
|
71
|
+
|
|
72
|
+
## AI-Assisted Development
|
|
73
|
+
|
|
74
|
+
This project is developed with substantial assistance from AI coding tools.
|
|
75
|
+
All changes remain subject to human review, automated testing, and the same
|
|
76
|
+
security requirements as manually written code. AI assistance does not imply
|
|
77
|
+
endorsement by any AI provider.
|
|
78
|
+
|
|
79
|
+
## Safety Model
|
|
80
|
+
|
|
81
|
+
Generated data must be synthetic. Source data is used only for structure and
|
|
82
|
+
safe metadata:
|
|
83
|
+
|
|
84
|
+
- column names
|
|
85
|
+
- inferred data types
|
|
86
|
+
- null ratios
|
|
87
|
+
- approximate distinct counts
|
|
88
|
+
- enum-like distributions for non-sensitive fields
|
|
89
|
+
- numeric ranges and percentiles
|
|
90
|
+
- date and timestamp ranges
|
|
91
|
+
- masked sensitive patterns
|
|
92
|
+
|
|
93
|
+
Likely PII and secret-like fields are treated as sensitive by default. Sensitive
|
|
94
|
+
CSV columns do not emit raw top values. Trino row-returning queries must be
|
|
95
|
+
read-only, bounded with `LIMIT`, and cannot use unrestricted `SELECT *`.
|
|
96
|
+
Local profile caches store only safe profile JSON: schema metadata,
|
|
97
|
+
aggregates, distributions, inferred rules, and masked patterns. They must not
|
|
98
|
+
store source rows or raw PII.
|
|
99
|
+
|
|
100
|
+
Forbidden behavior includes copying production rows, exposing raw PII, exporting
|
|
101
|
+
real rows, running DDL/DML SQL, and creating unrestricted SQL tools.
|
|
102
|
+
|
|
103
|
+
Generation APIs enforce a configurable per-entity row limit. The default is
|
|
104
|
+
`100000`; override it with `TEST_DATA_AGENT_MAX_GENERATION_COUNT`. Input and
|
|
105
|
+
output paths must be distinct, and folder bundles are assembled atomically.
|
|
106
|
+
Parquet keeps homogeneous numeric and boolean column types; intentionally mixed
|
|
107
|
+
invalid columns are stored as strings so negative datasets remain exportable.
|
|
108
|
+
|
|
109
|
+
Untrusted inputs are bounded before parsing. Defaults are 128 MiB per file,
|
|
110
|
+
512 MiB total, 1,000,000 rows, 1,000 columns, 10,000,000 cells, 100 files,
|
|
111
|
+
and 1,000,000 characters per CSV field. Parquet metadata is checked before
|
|
112
|
+
reading and limits expanded data to 512 MiB. YAML is limited to 50 aliases and
|
|
113
|
+
100 nesting levels.
|
|
114
|
+
Override these with `TEST_DATA_AGENT_MAX_INPUT_FILE_BYTES`,
|
|
115
|
+
`TEST_DATA_AGENT_MAX_TOTAL_INPUT_BYTES`, `TEST_DATA_AGENT_MAX_INPUT_ROWS`,
|
|
116
|
+
`TEST_DATA_AGENT_MAX_INPUT_COLUMNS`, `TEST_DATA_AGENT_MAX_INPUT_CELLS`,
|
|
117
|
+
`TEST_DATA_AGENT_MAX_INPUT_FILES`,
|
|
118
|
+
`TEST_DATA_AGENT_MAX_INPUT_CELL_CHARS`,
|
|
119
|
+
`TEST_DATA_AGENT_MAX_PARQUET_EXPANDED_BYTES`, `TEST_DATA_AGENT_MAX_YAML_ALIASES`,
|
|
120
|
+
and `TEST_DATA_AGENT_MAX_YAML_DEPTH`. Business-rule files and inline payloads
|
|
121
|
+
have a separate 1 MiB default limit controlled by
|
|
122
|
+
`TEST_DATA_AGENT_MAX_BUSINESS_RULES_BYTES`. Estimated business-rule work is
|
|
123
|
+
limited to 5,000,000 row/rule evaluations by default; override it with
|
|
124
|
+
`TEST_DATA_AGENT_MAX_BUSINESS_RULE_EVALUATIONS`.
|
|
125
|
+
|
|
126
|
+
Generated bundles are also bounded before publication. The defaults are
|
|
127
|
+
512 MiB for all generated files, 128 MiB of free disk space left in reserve,
|
|
128
|
+
and five minutes of wall-clock generation time. Override them with
|
|
129
|
+
`TEST_DATA_AGENT_MAX_OUTPUT_BYTES`, `TEST_DATA_AGENT_MIN_FREE_DISK_BYTES`, and
|
|
130
|
+
`TEST_DATA_AGENT_MAX_GENERATION_SECONDS`. DatasetSpec workflows estimate output
|
|
131
|
+
size before allocating rows, check time throughout generation, and publish
|
|
132
|
+
folder bundles only after exact size and validation checks pass.
|
|
133
|
+
|
|
134
|
+
## Install
|
|
135
|
+
|
|
136
|
+
Use Python 3.11 or newer.
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
python3 -m pip install -e ".[dev]"
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
For a reproducible development or CI environment, install the pinned `uv`
|
|
143
|
+
bootstrap version and sync from the committed lock file:
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
python3 -m pip install "uv==0.11.23"
|
|
147
|
+
uv sync --frozen --extra dev --no-install-project
|
|
148
|
+
uv sync --frozen --extra dev --no-editable --no-build-isolation
|
|
149
|
+
uv export --quiet --frozen --extra dev --no-emit-project \
|
|
150
|
+
--output-file /tmp/agent-paranoid-android-audit.txt
|
|
151
|
+
uv run --no-sync python -m pip_audit --require-hashes \
|
|
152
|
+
--requirement /tmp/agent-paranoid-android-audit.txt
|
|
153
|
+
uv run --no-sync python -m mypy
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Version tags trigger a release gate that builds wheel and source archives,
|
|
157
|
+
installs the wheel in an isolated environment, verifies package metadata,
|
|
158
|
+
`py.typed`, CLI entry points, and `doctor --skip-smoke`, exports a CycloneDX
|
|
159
|
+
SBOM, writes SHA-256 checksums, creates GitHub provenance and SBOM attestations,
|
|
160
|
+
and publishes the verified files as a GitHub Release.
|
|
161
|
+
|
|
162
|
+
## Quickstart
|
|
163
|
+
|
|
164
|
+
If you have one CSV file and just want synthetic data, start here:
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
test-data-agent generate-from-csv data/customers.csv \
|
|
168
|
+
--count 100 \
|
|
169
|
+
--seed 12345 \
|
|
170
|
+
--format csv \
|
|
171
|
+
--output out/customers.csv
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
If you have a folder with related CSV files, one file per table, start here:
|
|
175
|
+
|
|
176
|
+
```bash
|
|
177
|
+
test-data-agent generate-from-example data/example_dataset \
|
|
178
|
+
--count 100 \
|
|
179
|
+
--seed 12345 \
|
|
180
|
+
--format csv \
|
|
181
|
+
--output out/generated
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
On success, the CLI prints a short summary with the output location, generated
|
|
185
|
+
row counts, seed, validation status, and the `source rows copied: no` safety
|
|
186
|
+
check. It also writes review artifacts such as `generation_manifest.json` and
|
|
187
|
+
`validation_report.json`.
|
|
188
|
+
|
|
189
|
+
The example path below uses the checked-in fixture data and writes only local
|
|
190
|
+
artifacts under `out/`.
|
|
191
|
+
|
|
192
|
+
1. Install the package and run the test suite:
|
|
193
|
+
|
|
194
|
+
```bash
|
|
195
|
+
python3 -m pip install -e ".[dev]"
|
|
196
|
+
python3 -m pytest
|
|
197
|
+
```
|
|
198
|
+
|
|
199
|
+
After installation, the `test-data-agent` CLI should be available. If your shell
|
|
200
|
+
cannot find it, run the same commands as `python3 -m test_data_agent.cli ...`.
|
|
201
|
+
|
|
202
|
+
2. Generate a synthetic single-table CSV from an example source CSV:
|
|
203
|
+
|
|
204
|
+
```bash
|
|
205
|
+
test-data-agent generate-from-csv tests/fixtures/customers.csv \
|
|
206
|
+
--count 25 \
|
|
207
|
+
--seed 12345 \
|
|
208
|
+
--format csv \
|
|
209
|
+
--output out/customers.csv
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
This prints a summary and creates `out/customers.csv` plus review artifacts next
|
|
213
|
+
to it:
|
|
214
|
+
`csv_profile.json`, `generation_spec.json`, `validation_report.json`, and
|
|
215
|
+
`generation_manifest.json`.
|
|
216
|
+
|
|
217
|
+
3. Generate a related multi-table dataset from an example CSV folder:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
test-data-agent generate-from-example tests/fixtures/example_dataset \
|
|
221
|
+
--count 25 \
|
|
222
|
+
--seed 12345 \
|
|
223
|
+
--format csv \
|
|
224
|
+
--output out/example_dataset
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
This creates synthetic `customers.csv` and `orders.csv` files, a
|
|
228
|
+
`dataset_spec.yaml`, a validation report, and a generation manifest.
|
|
229
|
+
|
|
230
|
+
4. Re-run with the same seed when you need identical output:
|
|
231
|
+
|
|
232
|
+
```bash
|
|
233
|
+
test-data-agent generate-from-example tests/fixtures/example_dataset \
|
|
234
|
+
--count 25 \
|
|
235
|
+
--seed 12345 \
|
|
236
|
+
--format json \
|
|
237
|
+
--output out/example_dataset_json
|
|
238
|
+
```
|
|
239
|
+
|
|
240
|
+
The source CSV files are used only for schema and safe profile metadata. The
|
|
241
|
+
generated rows should not copy source rows or expose raw PII.
|
|
242
|
+
|
|
243
|
+
Run tests:
|
|
244
|
+
|
|
245
|
+
```bash
|
|
246
|
+
python3 -m pytest
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
Run the same quality checks used by CI:
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
python3 -m ruff check src tests
|
|
253
|
+
python3 -m mypy
|
|
254
|
+
python3 -m compileall -q src tests
|
|
255
|
+
python3 -m pytest --cov=test_data_agent --cov-report=term-missing --cov-fail-under=85
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
CI runs these checks on Python 3.11 and 3.12. The security regression suite
|
|
259
|
+
also uses Hypothesis to exercise variations of PII aliases, SQL statement
|
|
260
|
+
tails, duplicate CSV headers, and sensitive-value masking. Pull requests also
|
|
261
|
+
run dependency review and fail when they introduce a dependency with a known
|
|
262
|
+
Moderate-or-higher vulnerability.
|
|
263
|
+
|
|
264
|
+
For a local environment and quickstart smoke check:
|
|
265
|
+
|
|
266
|
+
```bash
|
|
267
|
+
test-data-agent doctor
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
## MVP Readiness Checklist
|
|
271
|
+
|
|
272
|
+
Use this checklist to keep the project scoped to a useful safety-first MVP:
|
|
273
|
+
|
|
274
|
+
- `generate-from-csv tests/fixtures/customers.csv ...` creates synthetic rows,
|
|
275
|
+
`csv_profile.json`, `generation_spec.json`, `validation_report.json`, and
|
|
276
|
+
`generation_manifest.json`.
|
|
277
|
+
- `generate-from-example tests/fixtures/example_dataset ...` creates related
|
|
278
|
+
synthetic tables plus `profile.json`, `dataset_spec.yaml`,
|
|
279
|
+
`validation_report.json`, and `generation_manifest.json`.
|
|
280
|
+
- Re-running the same spec with the same seed is deterministic.
|
|
281
|
+
- Generated manifests report `synthetic: true` and `source_rows_copied: false`.
|
|
282
|
+
- CSV profiles contain schema, safe distributions, ranges, and masked patterns,
|
|
283
|
+
not source rows or raw PII.
|
|
284
|
+
- Dataset validation is performed by deterministic Python code, not by
|
|
285
|
+
free-form LLM reasoning.
|
|
286
|
+
- Trino-facing queries remain read-only, allowlisted, bounded by `LIMIT`, and
|
|
287
|
+
covered by unsafe-SQL regression tests.
|
|
288
|
+
- Legacy `GenerationSpec` support is compatibility-only and remains outside the
|
|
289
|
+
primary quickstart path.
|
|
290
|
+
|
|
291
|
+
## Release Checklist
|
|
292
|
+
|
|
293
|
+
Before merging or cutting a release:
|
|
294
|
+
|
|
295
|
+
- Review the relevant [OpenSpec Baseline](openspec/project.md) capability specs
|
|
296
|
+
and update them when behavior changes.
|
|
297
|
+
- Run `scripts/check_release.sh`; it covers linting, compilation, coverage,
|
|
298
|
+
DatasetSpec schema freshness, and the quickstart fixture smoke flow.
|
|
299
|
+
- Regenerate `schemas/dataset_spec.schema.json` with
|
|
300
|
+
`python3 scripts/export_dataset_schema.py` when `DatasetSpec` changes.
|
|
301
|
+
- Update `CHANGELOG.md`, README examples, and docs for user-visible behavior or
|
|
302
|
+
safety-contract changes.
|
|
303
|
+
- Keep post-MVP features behind explicit OpenSpec changes instead of expanding
|
|
304
|
+
the MVP silently.
|
|
305
|
+
|
|
306
|
+
## Documentation
|
|
307
|
+
|
|
308
|
+
Start here if you want to understand the newer domain-agnostic multi-table
|
|
309
|
+
pipeline:
|
|
310
|
+
|
|
311
|
+
- [Domain-Agnostic Workflow](docs/domain_agnostic_workflow.md)
|
|
312
|
+
- [Dataset Profile And Spec Reference](docs/dataset_profile_and_spec.md)
|
|
313
|
+
- [Agent Design](docs/agent_design.md)
|
|
314
|
+
- [AI Integration](docs/ai_integration.md)
|
|
315
|
+
- [MCP Examples](docs/mcp_examples.md)
|
|
316
|
+
- [Generator MCP Design Rationale](docs/mcp_generator_design.md)
|
|
317
|
+
- [OpenSpec Baseline](openspec/project.md)
|
|
318
|
+
- [Release Process](docs/release.md)
|
|
319
|
+
- [Public Release Checklist](docs/public_release_checklist.md)
|
|
320
|
+
- [Security Policy](SECURITY.md)
|
|
321
|
+
- [Contributing](CONTRIBUTING.md)
|
|
322
|
+
- [License](LICENSE)
|
|
323
|
+
- [DatasetSpec JSON Schema](schemas/dataset_spec.schema.json)
|
|
324
|
+
- [Roadmap](docs/roadmap.md)
|
|
325
|
+
- [Changelog](CHANGELOG.md)
|
|
326
|
+
- [Implementation Map](docs/implementation_map.md)
|
|
327
|
+
- [Architecture Overview Diagram](docs/architecture.puml)
|
|
328
|
+
- [Agent Workflow Diagram](docs/architecture_agent_workflow.puml)
|
|
329
|
+
- [Safety Boundaries Diagram](docs/architecture_safety_boundaries.puml)
|
|
330
|
+
|
|
331
|
+
## How To Use It
|
|
332
|
+
|
|
333
|
+
There are five normal ways to use the project:
|
|
334
|
+
|
|
335
|
+
1. Start from a hand-written generation spec.
|
|
336
|
+
2. Start from a CSV file and let the agent infer a safe profile.
|
|
337
|
+
3. Start from safe Trino/profile metadata and generate from that profile.
|
|
338
|
+
4. Start from an example multi-table CSV folder and infer a dataset spec.
|
|
339
|
+
5. Let an MCP-compatible AI client orchestrate the same safe workflow.
|
|
340
|
+
|
|
341
|
+
Each flow produces synthetic data plus validation artifacts.
|
|
342
|
+
|
|
343
|
+
### 1. Generate From A Spec
|
|
344
|
+
|
|
345
|
+
Create `dataset_spec.yaml`:
|
|
346
|
+
|
|
347
|
+
```yaml
|
|
348
|
+
schema_version: '1.0'
|
|
349
|
+
entities:
|
|
350
|
+
- name: customers
|
|
351
|
+
row_count: 3
|
|
352
|
+
primary_key: customer_id
|
|
353
|
+
fields:
|
|
354
|
+
- name: customer_id
|
|
355
|
+
data_type: integer
|
|
356
|
+
is_identifier: true
|
|
357
|
+
distribution:
|
|
358
|
+
kind: synthetic_identifier
|
|
359
|
+
- name: email
|
|
360
|
+
data_type: string
|
|
361
|
+
sensitive: true
|
|
362
|
+
semantic_type: email
|
|
363
|
+
distribution:
|
|
364
|
+
kind: masked_patterns
|
|
365
|
+
patterns:
|
|
366
|
+
- pattern: email
|
|
367
|
+
count: 1
|
|
368
|
+
- name: status
|
|
369
|
+
data_type: string
|
|
370
|
+
distribution:
|
|
371
|
+
kind: categorical
|
|
372
|
+
categories:
|
|
373
|
+
- {value: new, count: 1}
|
|
374
|
+
- {value: active, count: 2}
|
|
375
|
+
relationships: []
|
|
376
|
+
constraints: []
|
|
377
|
+
generation_settings:
|
|
378
|
+
seed: 42
|
|
379
|
+
output_format: json
|
|
380
|
+
```
|
|
381
|
+
|
|
382
|
+
Generate rows:
|
|
383
|
+
|
|
384
|
+
```bash
|
|
385
|
+
test-data-agent generate dataset_spec.yaml --output out/customers
|
|
386
|
+
```
|
|
387
|
+
|
|
388
|
+
Validate rows against the spec:
|
|
389
|
+
|
|
390
|
+
```bash
|
|
391
|
+
test-data-agent validate out/customers/dataset_spec.yaml out/customers
|
|
392
|
+
```
|
|
393
|
+
|
|
394
|
+
The `generate` command writes:
|
|
395
|
+
|
|
396
|
+
- `out/customers/customers.json`
|
|
397
|
+
- `out/customers/dataset_spec.yaml`
|
|
398
|
+
- `out/customers/validation_report.json`
|
|
399
|
+
- `out/customers/generation_manifest.json`
|
|
400
|
+
|
|
401
|
+
### 2. Generate From A CSV
|
|
402
|
+
|
|
403
|
+
First create a safe CSV profile:
|
|
404
|
+
|
|
405
|
+
```bash
|
|
406
|
+
test-data-agent profile-csv data/customers.csv \
|
|
407
|
+
--output out/customers_profile.json
|
|
408
|
+
```
|
|
409
|
+
|
|
410
|
+
Inspect the profile before using it. It should contain schema, aggregates,
|
|
411
|
+
distributions, ranges, and masked patterns, not raw sensitive values.
|
|
412
|
+
|
|
413
|
+
Generate synthetic data from the CSV:
|
|
414
|
+
|
|
415
|
+
```bash
|
|
416
|
+
test-data-agent generate-from-csv data/customers.csv \
|
|
417
|
+
--count 1000 \
|
|
418
|
+
--mode valid \
|
|
419
|
+
--seed 12345 \
|
|
420
|
+
--format csv \
|
|
421
|
+
--output out/customers.csv
|
|
422
|
+
```
|
|
423
|
+
|
|
424
|
+
For a mixed valid/invalid dataset:
|
|
425
|
+
|
|
426
|
+
```bash
|
|
427
|
+
test-data-agent generate-from-csv data/customers.csv \
|
|
428
|
+
--count 1000 \
|
|
429
|
+
--mode mixed \
|
|
430
|
+
--invalid-ratio 0.02 \
|
|
431
|
+
--seed 12345 \
|
|
432
|
+
--format parquet \
|
|
433
|
+
--output out/customers.parquet
|
|
434
|
+
```
|
|
435
|
+
|
|
436
|
+
`generate-from-csv` writes:
|
|
437
|
+
|
|
438
|
+
- the requested output file
|
|
439
|
+
- `csv_profile.json`
|
|
440
|
+
- `generation_spec.json`
|
|
441
|
+
- `validation_report.json`
|
|
442
|
+
- `generation_manifest.json`
|
|
443
|
+
|
|
444
|
+
Supported output formats are `json`, `csv`, and `parquet`.
|
|
445
|
+
|
|
446
|
+
### 3. Generate From A Profile
|
|
447
|
+
|
|
448
|
+
Use a safe profile such as `examples/orders_profile.json`:
|
|
449
|
+
|
|
450
|
+
```bash
|
|
451
|
+
test-data-agent generate \
|
|
452
|
+
--profile examples/orders_profile.json \
|
|
453
|
+
--count 10000 \
|
|
454
|
+
--mode mixed \
|
|
455
|
+
--invalid-ratio 0.02 \
|
|
456
|
+
--seed 12345 \
|
|
457
|
+
--format csv \
|
|
458
|
+
--output out/orders.csv
|
|
459
|
+
```
|
|
460
|
+
|
|
461
|
+
This writes:
|
|
462
|
+
|
|
463
|
+
- `out/orders.csv`
|
|
464
|
+
- `out/generation_spec.json`
|
|
465
|
+
- `out/validation_report.json`
|
|
466
|
+
- `out/generation_manifest.json`
|
|
467
|
+
|
|
468
|
+
Profile input should contain safe metadata only. Do not include raw production
|
|
469
|
+
samples.
|
|
470
|
+
|
|
471
|
+
### 4. Generate From An Example Dataset
|
|
472
|
+
|
|
473
|
+
For domain-agnostic multi-table generation, place one CSV per table in a folder:
|
|
474
|
+
|
|
475
|
+
```text
|
|
476
|
+
example_dataset/
|
|
477
|
+
customers.csv
|
|
478
|
+
orders.csv
|
|
479
|
+
```
|
|
480
|
+
|
|
481
|
+
Profile the folder without exposing raw PII:
|
|
482
|
+
|
|
483
|
+
```bash
|
|
484
|
+
test-data-agent profile-example example_dataset \
|
|
485
|
+
--output out/profile.json
|
|
486
|
+
```
|
|
487
|
+
|
|
488
|
+
Infer a YAML dataset spec with schema, relationships, distributions, formulas,
|
|
489
|
+
temporal rules, conditional rules, and aggregate mappings:
|
|
490
|
+
|
|
491
|
+
```bash
|
|
492
|
+
test-data-agent infer-spec out/profile.json \
|
|
493
|
+
--count 1000 \
|
|
494
|
+
--output out/dataset_spec.yaml
|
|
495
|
+
```
|
|
496
|
+
|
|
497
|
+
Generate all related tables:
|
|
498
|
+
|
|
499
|
+
```bash
|
|
500
|
+
test-data-agent generate out/dataset_spec.yaml \
|
|
501
|
+
--seed 12345 \
|
|
502
|
+
--format csv \
|
|
503
|
+
--output out/generated
|
|
504
|
+
```
|
|
505
|
+
|
|
506
|
+
Validate the generated folder:
|
|
507
|
+
|
|
508
|
+
```bash
|
|
509
|
+
test-data-agent validate out/generated/dataset_spec.yaml out/generated \
|
|
510
|
+
--output out/generated/validation_report.json
|
|
511
|
+
```
|
|
512
|
+
|
|
513
|
+
Or run the full flow in one command:
|
|
514
|
+
|
|
515
|
+
```bash
|
|
516
|
+
test-data-agent generate-from-example example_dataset \
|
|
517
|
+
--seed 12345 \
|
|
518
|
+
--count 1000 \
|
|
519
|
+
--format parquet \
|
|
520
|
+
--output out/generated
|
|
521
|
+
```
|
|
522
|
+
|
|
523
|
+
All identifiers are regenerated synthetically. Foreign keys are preserved by
|
|
524
|
+
wiring child rows to generated parent IDs, never by reusing source IDs.
|
|
525
|
+
The generated folder also contains the effective `dataset_spec.yaml`,
|
|
526
|
+
`validation_report.json`, and `generation_manifest.json`.
|
|
527
|
+
|
|
528
|
+
Large CSV folders are profiled in a streaming pass, so the profiler does not
|
|
529
|
+
hold every source row in memory. Schema, null ratios, safe distributions, and
|
|
530
|
+
field metadata are computed across the full files. Relationship and constraint
|
|
531
|
+
mining use a bounded local sample because they need row-level comparisons:
|
|
532
|
+
|
|
533
|
+
```bash
|
|
534
|
+
test-data-agent profile-example example_dataset \
|
|
535
|
+
--output out/profile.json \
|
|
536
|
+
--rule-sample-rows 100000
|
|
537
|
+
```
|
|
538
|
+
|
|
539
|
+
Profiles are cached by CSV file names, sizes, modification times, and the
|
|
540
|
+
`--rule-sample-rows` value:
|
|
541
|
+
|
|
542
|
+
```bash
|
|
543
|
+
test-data-agent profile-example example_dataset \
|
|
544
|
+
--output out/profile.json \
|
|
545
|
+
--cache-dir .test_data_agent_cache/profiles
|
|
546
|
+
```
|
|
547
|
+
|
|
548
|
+
Use `--no-cache` when you need to force a fresh profile. The cache contains
|
|
549
|
+
safe profile metadata only, not source rows. Cache writes are atomic, and a
|
|
550
|
+
stale or incomplete cache is treated as a miss.
|
|
551
|
+
|
|
552
|
+
## CLI Reference
|
|
553
|
+
|
|
554
|
+
Profile an example multi-table CSV folder:
|
|
555
|
+
|
|
556
|
+
```bash
|
|
557
|
+
test-data-agent profile-example INPUT_FOLDER --output PROFILE.json
|
|
558
|
+
```
|
|
559
|
+
|
|
560
|
+
Infer a YAML dataset spec:
|
|
561
|
+
|
|
562
|
+
```bash
|
|
563
|
+
test-data-agent infer-spec PROFILE.json --output DATASET_SPEC.yaml
|
|
564
|
+
```
|
|
565
|
+
|
|
566
|
+
Profile a CSV:
|
|
567
|
+
|
|
568
|
+
```bash
|
|
569
|
+
test-data-agent profile-csv INPUT.csv --output PROFILE.json
|
|
570
|
+
```
|
|
571
|
+
|
|
572
|
+
Generate from a spec:
|
|
573
|
+
|
|
574
|
+
```bash
|
|
575
|
+
test-data-agent generate DATASET_SPEC.yaml --format csv --output OUTPUT_FOLDER
|
|
576
|
+
# Deprecated compatibility path:
|
|
577
|
+
test-data-agent generate LEGACY_GENERATION_SPEC.json --output OUTPUT.json
|
|
578
|
+
```
|
|
579
|
+
|
|
580
|
+
Generate from a safe profile:
|
|
581
|
+
|
|
582
|
+
```bash
|
|
583
|
+
test-data-agent generate \
|
|
584
|
+
--profile PROFILE.json \
|
|
585
|
+
--count 1000 \
|
|
586
|
+
--seed 12345 \
|
|
587
|
+
--format json \
|
|
588
|
+
--output OUTPUT.json
|
|
589
|
+
```
|
|
590
|
+
|
|
591
|
+
Generate directly from CSV:
|
|
592
|
+
|
|
593
|
+
```bash
|
|
594
|
+
test-data-agent generate-from-csv INPUT.csv \
|
|
595
|
+
--count 1000 \
|
|
596
|
+
--seed 12345 \
|
|
597
|
+
--format csv \
|
|
598
|
+
--output OUTPUT.csv
|
|
599
|
+
```
|
|
600
|
+
|
|
601
|
+
Generate directly from an example multi-table folder:
|
|
602
|
+
|
|
603
|
+
```bash
|
|
604
|
+
test-data-agent generate-from-example INPUT_FOLDER \
|
|
605
|
+
--count 1000 \
|
|
606
|
+
--seed 12345 \
|
|
607
|
+
--format csv \
|
|
608
|
+
--output OUTPUT_FOLDER
|
|
609
|
+
```
|
|
610
|
+
|
|
611
|
+
Validate generated rows (the first form is the deprecated compatibility path):
|
|
612
|
+
|
|
613
|
+
```bash
|
|
614
|
+
test-data-agent validate SPEC.json ROWS.json
|
|
615
|
+
test-data-agent validate DATASET_SPEC.yaml OUTPUT_FOLDER
|
|
616
|
+
```
|
|
617
|
+
|
|
618
|
+
Plan and approve an AI-agent workflow:
|
|
619
|
+
|
|
620
|
+
```bash
|
|
621
|
+
test-data-agent agent-plan INPUT_FOLDER \
|
|
622
|
+
--source-type csv-folder \
|
|
623
|
+
--workspace out/agent \
|
|
624
|
+
--count 100 \
|
|
625
|
+
--seed 12345 \
|
|
626
|
+
--format csv
|
|
627
|
+
test-data-agent agent-approve out/agent
|
|
628
|
+
```
|
|
629
|
+
|
|
630
|
+
Useful options:
|
|
631
|
+
|
|
632
|
+
- `--count` overrides or supplies row count.
|
|
633
|
+
- `--seed` makes generation reproducible.
|
|
634
|
+
- `--format json|csv|parquet` selects output format.
|
|
635
|
+
- `--mode valid|mixed|negative|edge|load_test` selects the generation mode.
|
|
636
|
+
- `--invalid-ratio 0.02` injects invalid values in `mixed` mode; `negative`
|
|
637
|
+
mode intentionally makes every generated value invalid.
|
|
638
|
+
- `--table NAME` sets the table name for CSV profiling.
|
|
639
|
+
- `--cache-dir PATH` selects the safe profile cache for example-folder
|
|
640
|
+
profiling.
|
|
641
|
+
- `--no-cache` disables profile cache reuse.
|
|
642
|
+
- `--overwrite` allows replacing existing single-file outputs such as generated
|
|
643
|
+
CSV/JSON/Parquet files, profile JSON files, validation reports, and inferred
|
|
644
|
+
specs. Without it, the CLI refuses to overwrite existing files.
|
|
645
|
+
- `--rule-sample-rows N` bounds row-level relationship and constraint mining
|
|
646
|
+
while full-file schema and distribution profiling remains streaming.
|
|
647
|
+
|
|
648
|
+
Check local setup and run a small fixture smoke test:
|
|
649
|
+
|
|
650
|
+
```bash
|
|
651
|
+
test-data-agent doctor
|
|
652
|
+
```
|
|
653
|
+
|
|
654
|
+
Use `test-data-agent doctor --skip-smoke` when you only want dependency and
|
|
655
|
+
Python-version checks.
|
|
656
|
+
|
|
657
|
+
## Architecture
|
|
658
|
+
|
|
659
|
+
The architecture is documented as PlantUML diagrams:
|
|
660
|
+
|
|
661
|
+
- `docs/architecture.puml` shows the high-level application components.
|
|
662
|
+
- `docs/architecture_agent_workflow.puml` shows the review-first agent flow.
|
|
663
|
+
- `docs/architecture_safety_boundaries.puml` shows the trust and safety
|
|
664
|
+
boundaries around AI planning, MCP tools, profiling, generation, and
|
|
665
|
+
validation.
|
|
666
|
+
|
|
667
|
+
## Legacy GenerationSpec Compatibility
|
|
668
|
+
|
|
669
|
+
Legacy single-table specs are Pydantic models serialized as JSON. New
|
|
670
|
+
integrations should use the versioned `DatasetSpec` shown above. Legacy data
|
|
671
|
+
types remain:
|
|
672
|
+
|
|
673
|
+
- `integer`
|
|
674
|
+
- `float`
|
|
675
|
+
- `boolean`
|
|
676
|
+
- `string`
|
|
677
|
+
- `date`
|
|
678
|
+
- `datetime`
|
|
679
|
+
- `email`
|
|
680
|
+
- `phone`
|
|
681
|
+
- `name`
|
|
682
|
+
- `address`
|
|
683
|
+
- `uuid`
|
|
684
|
+
|
|
685
|
+
Supported strategies:
|
|
686
|
+
|
|
687
|
+
- `sequence`
|
|
688
|
+
- `random_int`
|
|
689
|
+
- `random_float`
|
|
690
|
+
- `random_boolean`
|
|
691
|
+
- `faker`
|
|
692
|
+
- `choice`
|
|
693
|
+
- `constant`
|
|
694
|
+
- `date_range`
|
|
695
|
+
- `datetime_range`
|
|
696
|
+
- `uuid`
|
|
697
|
+
|
|
698
|
+
Example with ranges and nullable values:
|
|
699
|
+
|
|
700
|
+
```json
|
|
701
|
+
{
|
|
702
|
+
"seed": 7,
|
|
703
|
+
"output_format": "csv",
|
|
704
|
+
"table": {
|
|
705
|
+
"name": "orders",
|
|
706
|
+
"row_count": 100,
|
|
707
|
+
"columns": [
|
|
708
|
+
{"name": "order_id", "data_type": "integer", "strategy": "sequence"},
|
|
709
|
+
{
|
|
710
|
+
"name": "status",
|
|
711
|
+
"data_type": "string",
|
|
712
|
+
"strategy": "choice",
|
|
713
|
+
"choices": ["new", "paid", "shipped", "cancelled"]
|
|
714
|
+
},
|
|
715
|
+
{
|
|
716
|
+
"name": "order_total",
|
|
717
|
+
"data_type": "float",
|
|
718
|
+
"min_value": 1,
|
|
719
|
+
"max_value": 500
|
|
720
|
+
},
|
|
721
|
+
{
|
|
722
|
+
"name": "created_at",
|
|
723
|
+
"data_type": "datetime",
|
|
724
|
+
"min_datetime": "2024-01-01T00:00:00",
|
|
725
|
+
"max_datetime": "2024-12-31T23:59:59"
|
|
726
|
+
},
|
|
727
|
+
{
|
|
728
|
+
"name": "coupon_code",
|
|
729
|
+
"data_type": "string",
|
|
730
|
+
"nullable": true,
|
|
731
|
+
"null_probability": 0.25
|
|
732
|
+
}
|
|
733
|
+
]
|
|
734
|
+
}
|
|
735
|
+
}
|
|
736
|
+
```
|
|
737
|
+
|
|
738
|
+
## Trino MCP Server
|
|
739
|
+
|
|
740
|
+
The MCP server exposes small safe tools for metadata and profiling:
|
|
741
|
+
|
|
742
|
+
- `list_catalogs`
|
|
743
|
+
- `list_schemas`
|
|
744
|
+
- `list_tables`
|
|
745
|
+
- `describe_table`
|
|
746
|
+
- `profile_table`
|
|
747
|
+
- `profile_table_safe`
|
|
748
|
+
- `profile_column`
|
|
749
|
+
- `profile_foreign_key`
|
|
750
|
+
- `profile_temporal_ordering`
|
|
751
|
+
- `profile_formula_rule`
|
|
752
|
+
- `profile_conditional_required`
|
|
753
|
+
- `profile_conditional_allowed_values`
|
|
754
|
+
- `profile_aggregate_mapping`
|
|
755
|
+
- `sample_rows_masked`
|
|
756
|
+
- `run_safe_select`
|
|
757
|
+
|
|
758
|
+
Configure Trino with environment variables:
|
|
759
|
+
|
|
760
|
+
```bash
|
|
761
|
+
TRINO_HOST=trino.example.internal
|
|
762
|
+
TRINO_PORT=443
|
|
763
|
+
TRINO_USER=your_user
|
|
764
|
+
TRINO_HTTP_SCHEME=https
|
|
765
|
+
TRINO_ALLOWED_CATALOGS=hive,iceberg
|
|
766
|
+
TRINO_ALLOWED_SCHEMAS=dev,test,staging
|
|
767
|
+
TRINO_QUERY_MAX_EXECUTION_TIME=30s
|
|
768
|
+
TRINO_QUERY_MAX_RUN_TIME=45s
|
|
769
|
+
TRINO_QUERY_MAX_SCAN_PHYSICAL_BYTES=1GB
|
|
770
|
+
```
|
|
771
|
+
|
|
772
|
+
Start the server:
|
|
773
|
+
|
|
774
|
+
```bash
|
|
775
|
+
python3 -m test_data_agent.mcp_trino_server
|
|
776
|
+
```
|
|
777
|
+
|
|
778
|
+
`run_safe_select` rejects DDL, DML, executable statements, multiple statements,
|
|
779
|
+
unbounded row-returning queries without a top-level `LIMIT`, unrestricted
|
|
780
|
+
`SELECT *`, and projections of likely PII fields even when they are aliased.
|
|
781
|
+
Both catalog and schema allowlists are required, and arbitrary selects must use
|
|
782
|
+
fully qualified `catalog.schema.table` references that match them. The server
|
|
783
|
+
also defaults to HTTPS and refuses plain HTTP.
|
|
784
|
+
SQL validation uses `sqlglot` AST parsing for Trino syntax. Masked sampling uses
|
|
785
|
+
conservative field-name and value-content detection for likely PII and secrets.
|
|
786
|
+
|
|
787
|
+
For an intentionally unrestricted local environment, set
|
|
788
|
+
`TRINO_ALLOW_UNRESTRICTED=true`. For a local Trino endpoint that cannot use TLS,
|
|
789
|
+
also set `TRINO_ALLOW_INSECURE_HTTP=true`; neither override should be used for
|
|
790
|
+
production access.
|
|
791
|
+
Trino responses are also capped client-side at 10,000 rows; lower this with
|
|
792
|
+
`TRINO_MAX_RESULT_ROWS` when metadata namespaces are small.
|
|
793
|
+
Every connection also applies server-side session budgets for execution time,
|
|
794
|
+
total run time, and physical bytes scanned. The defaults are `30s`, `45s`, and
|
|
795
|
+
`1GB`; lower the corresponding `TRINO_QUERY_MAX_*` variables for narrower
|
|
796
|
+
environments. Values are validated locally and cannot exceed one hour of
|
|
797
|
+
execution, two hours of total run time, or 100 GB scanned.
|
|
798
|
+
|
|
799
|
+
For large Trino tables, use safe profiling instead of downloading rows. The
|
|
800
|
+
safe profile path pushes aggregate work into Trino: row counts, null ratios,
|
|
801
|
+
approximate distinct counts, numeric ranges and percentiles, and timestamp
|
|
802
|
+
ranges are returned as compact metadata. Low-cardinality top values are fetched
|
|
803
|
+
only for non-sensitive string columns and always with a bounded `LIMIT`.
|
|
804
|
+
Sensitive columns never return raw top values. Save the resulting profile JSON
|
|
805
|
+
and reuse it for generation so repeated runs do not re-query the source table.
|
|
806
|
+
|
|
807
|
+
Consistency profiling is also aggregate-only. Use the dedicated rule tools to
|
|
808
|
+
measure whether inferred or proposed rules hold before adding them to a dataset
|
|
809
|
+
spec:
|
|
810
|
+
|
|
811
|
+
- `profile_foreign_key` returns child checked/matched/orphan counts.
|
|
812
|
+
- `profile_temporal_ordering` returns pass/fail counts for timestamp ordering.
|
|
813
|
+
- `profile_formula_rule` returns pass/fail counts and numeric residuals for
|
|
814
|
+
simple arithmetic formulas.
|
|
815
|
+
- `profile_conditional_required` returns scoped present/missing counts without
|
|
816
|
+
echoing condition values.
|
|
817
|
+
- `profile_conditional_allowed_values` returns scoped allowed/violation counts.
|
|
818
|
+
- `profile_aggregate_mapping` compares parent aggregate fields with child
|
|
819
|
+
`sum` or `count` aggregates.
|
|
820
|
+
|
|
821
|
+
Each rule profile includes `confidence` and `status`. The tools do not return
|
|
822
|
+
source rows, identifiers, or raw PII values.
|
|
823
|
+
|
|
824
|
+
## Generator MCP Server
|
|
825
|
+
|
|
826
|
+
The second MCP server exposes the synthetic pipeline to AI clients:
|
|
827
|
+
|
|
828
|
+
- `profile_csv`
|
|
829
|
+
- `infer_dataset_spec`
|
|
830
|
+
- `generate_dataset`
|
|
831
|
+
- `validate_dataset`
|
|
832
|
+
- `export_dataset`
|
|
833
|
+
|
|
834
|
+
Start it from the workspace that contains allowed inputs and outputs:
|
|
835
|
+
|
|
836
|
+
```bash
|
|
837
|
+
TEST_DATA_AGENT_WORKSPACE_ROOT=/path/to/allowed/workspace \
|
|
838
|
+
python3 -m test_data_agent.mcp_generator_server
|
|
839
|
+
```
|
|
840
|
+
|
|
841
|
+
All tool paths are resolved inside `TEST_DATA_AGENT_WORKSPACE_ROOT`; traversal
|
|
842
|
+
and symlink escapes are rejected. MCP responses contain paths, row counts,
|
|
843
|
+
version metadata, and validation results, never generated rows. `export_dataset`
|
|
844
|
+
always generates fresh synthetic data from a `DatasetSpec`; it cannot convert or
|
|
845
|
+
export arbitrary source rows.
|
|
846
|
+
|
|
847
|
+
Generated bundles contain `dataset_spec.yaml`, `validation_report.json`, and
|
|
848
|
+
`generation_manifest.json`. The manifest records the package and schema
|
|
849
|
+
versions, spec fingerprint, seed, output format, row counts, validation status,
|
|
850
|
+
and the explicit provenance flags `synthetic: true` and
|
|
851
|
+
`source_rows_copied: false`. Runs with business rules also record a SHA-256
|
|
852
|
+
fingerprint, rule count, pass/fail counts, truncation status, and overall
|
|
853
|
+
business validity.
|
|
854
|
+
|
|
855
|
+
`infer_dataset_spec` accepts exactly one of a workspace `profile_path` or an
|
|
856
|
+
inline `profile_payload` from the Trino MCP server. MCP tools never overwrite
|
|
857
|
+
existing output files, and generation requires a new or empty output folder.
|
|
858
|
+
`generate_dataset` and `export_dataset` accept at most one of a workspace
|
|
859
|
+
`business_rules_path` or structured `business_rules_payload`. Unknown keys,
|
|
860
|
+
dangling references, unsupported expressions, oversized inputs, and
|
|
861
|
+
raw-looking sensitive literals fail before output is created.
|
|
862
|
+
|
|
863
|
+
Example MCP rule payload:
|
|
864
|
+
|
|
865
|
+
```json
|
|
866
|
+
{
|
|
867
|
+
"field_rules": [
|
|
868
|
+
{
|
|
869
|
+
"table": "orders",
|
|
870
|
+
"field": "status",
|
|
871
|
+
"required": true,
|
|
872
|
+
"allowed_values": ["new", "paid", "cancelled"]
|
|
873
|
+
}
|
|
874
|
+
]
|
|
875
|
+
}
|
|
876
|
+
```
|
|
877
|
+
|
|
878
|
+
The MCP response returns only the compact business-validation summary and
|
|
879
|
+
`business_validation_report_path`; detailed bounded errors remain in the
|
|
880
|
+
workspace artifact and generated rows are never returned inline.
|
|
881
|
+
|
|
882
|
+
Run the local end-to-end example from safe Trino profile metadata:
|
|
883
|
+
|
|
884
|
+
```bash
|
|
885
|
+
python scripts/run_ai_demo.py --output out/ai_demo --count 100 --seed 12345
|
|
886
|
+
```
|
|
887
|
+
|
|
888
|
+
## DatasetSpec Generation
|
|
889
|
+
|
|
890
|
+
Use `DatasetSpec` when synthetic datasets need deterministic relationships such
|
|
891
|
+
as foreign keys:
|
|
892
|
+
|
|
893
|
+
```python
|
|
894
|
+
from test_data_agent.core.dataset import DatasetSpec
|
|
895
|
+
from test_data_agent.core.entity import EntitySpec
|
|
896
|
+
from test_data_agent.core.field import FieldSpec
|
|
897
|
+
from test_data_agent.core.relationship import Relationship
|
|
898
|
+
from test_data_agent.core.settings import GenerationSettings
|
|
899
|
+
from test_data_agent.generation.entity_generator import generate_dataset
|
|
900
|
+
|
|
901
|
+
spec = DatasetSpec(
|
|
902
|
+
entities=[
|
|
903
|
+
EntitySpec(
|
|
904
|
+
name="customers",
|
|
905
|
+
row_count=100,
|
|
906
|
+
primary_key="customer_id",
|
|
907
|
+
fields=[
|
|
908
|
+
FieldSpec(name="customer_id", data_type="integer", is_identifier=True),
|
|
909
|
+
FieldSpec(name="status", data_type="string"),
|
|
910
|
+
],
|
|
911
|
+
),
|
|
912
|
+
EntitySpec(
|
|
913
|
+
name="orders",
|
|
914
|
+
row_count=1000,
|
|
915
|
+
primary_key="order_id",
|
|
916
|
+
fields=[
|
|
917
|
+
FieldSpec(name="order_id", data_type="integer", is_identifier=True),
|
|
918
|
+
FieldSpec(name="customer_id", data_type="integer"),
|
|
919
|
+
],
|
|
920
|
+
),
|
|
921
|
+
],
|
|
922
|
+
relationships=[
|
|
923
|
+
Relationship(
|
|
924
|
+
child_entity="orders",
|
|
925
|
+
child_field="customer_id",
|
|
926
|
+
parent_entity="customers",
|
|
927
|
+
parent_field="customer_id",
|
|
928
|
+
confidence=1.0,
|
|
929
|
+
status="confirmed",
|
|
930
|
+
)
|
|
931
|
+
],
|
|
932
|
+
generation_settings=GenerationSettings(seed=12345),
|
|
933
|
+
)
|
|
934
|
+
|
|
935
|
+
rows_by_entity = generate_dataset(spec, seed=12345)
|
|
936
|
+
```
|
|
937
|
+
|
|
938
|
+
Parent and child rows are generated synthetically from safe CSV-derived or
|
|
939
|
+
Trino-derived profile metadata. Foreign-key values are assigned from generated
|
|
940
|
+
parent rows, never copied from source data. Legacy `MultiTableGenerationSpec`
|
|
941
|
+
support remains available through compatibility adapters while downstream code
|
|
942
|
+
migrates.
|
|
943
|
+
|
|
944
|
+
## Business Rules
|
|
945
|
+
|
|
946
|
+
Business logic is represented as structured YAML or JSON and enforced by code,
|
|
947
|
+
not by free-form LLM reasoning. Current rule models support:
|
|
948
|
+
|
|
949
|
+
- field rules
|
|
950
|
+
- conditional required fields
|
|
951
|
+
- conditional allowed values
|
|
952
|
+
- temporal ordering
|
|
953
|
+
- row formulas
|
|
954
|
+
- foreign keys
|
|
955
|
+
- aggregate formulas
|
|
956
|
+
- scenario distributions
|
|
957
|
+
|
|
958
|
+
Example:
|
|
959
|
+
|
|
960
|
+
```yaml
|
|
961
|
+
field_rules:
|
|
962
|
+
- table: orders
|
|
963
|
+
field: status
|
|
964
|
+
required: true
|
|
965
|
+
allowed_values: [new, paid, shipped, cancelled]
|
|
966
|
+
- table: orders
|
|
967
|
+
field: order_total
|
|
968
|
+
required: true
|
|
969
|
+
min_value: 0
|
|
970
|
+
|
|
971
|
+
row_rules:
|
|
972
|
+
- type: temporal_ordering
|
|
973
|
+
table: orders
|
|
974
|
+
start_field: created_at
|
|
975
|
+
end_field: fulfilled_at
|
|
976
|
+
allow_equal: true
|
|
977
|
+
|
|
978
|
+
scenarios:
|
|
979
|
+
- name: paid_order
|
|
980
|
+
weight: 8
|
|
981
|
+
field_values:
|
|
982
|
+
orders:
|
|
983
|
+
status: paid
|
|
984
|
+
- name: cancelled_order
|
|
985
|
+
weight: 1
|
|
986
|
+
field_values:
|
|
987
|
+
orders:
|
|
988
|
+
status: cancelled
|
|
989
|
+
```
|
|
990
|
+
|
|
991
|
+
Business rules can be applied from the CLI with `--business-rules` on
|
|
992
|
+
`generate`, `generate-from-csv`, and profile-based generation. Generator MCP
|
|
993
|
+
calls use `business_rules_path` or `business_rules_payload`. Rules are strict:
|
|
994
|
+
unknown keys and fields fail closed, formulas use a bounded arithmetic AST,
|
|
995
|
+
and concrete PII or secret values are rejected by both CLI and MCP workflows.
|
|
996
|
+
They are also available from Python:
|
|
997
|
+
|
|
998
|
+
```python
|
|
999
|
+
from pathlib import Path
|
|
1000
|
+
|
|
1001
|
+
from test_data_agent.business_rules import load_business_rules
|
|
1002
|
+
from test_data_agent.business_validator import validate_business_rules
|
|
1003
|
+
from test_data_agent.rules_engine import apply_business_rules
|
|
1004
|
+
|
|
1005
|
+
rules = load_business_rules(Path("rules/orders.yaml"))
|
|
1006
|
+
rows_by_table = {"orders": rows}
|
|
1007
|
+
|
|
1008
|
+
apply_business_rules(
|
|
1009
|
+
rows_by_table,
|
|
1010
|
+
rules,
|
|
1011
|
+
seed=12345,
|
|
1012
|
+
mode="mixed",
|
|
1013
|
+
invalid_ratio=0.02,
|
|
1014
|
+
)
|
|
1015
|
+
|
|
1016
|
+
report = validate_business_rules(rows_by_table, rules)
|
|
1017
|
+
```
|
|
1018
|
+
|
|
1019
|
+
## Python API
|
|
1020
|
+
|
|
1021
|
+
```python
|
|
1022
|
+
from test_data_agent.core.dataset import DatasetSpec
|
|
1023
|
+
from test_data_agent.core.entity import EntitySpec
|
|
1024
|
+
from test_data_agent.core.field import FieldSpec
|
|
1025
|
+
from test_data_agent.core.settings import GenerationSettings
|
|
1026
|
+
from test_data_agent.generation.entity_generator import generate_dataset
|
|
1027
|
+
from test_data_agent.validation import validate_dataset
|
|
1028
|
+
|
|
1029
|
+
spec = DatasetSpec(
|
|
1030
|
+
entities=[
|
|
1031
|
+
EntitySpec(
|
|
1032
|
+
name="events",
|
|
1033
|
+
row_count=100,
|
|
1034
|
+
primary_key="event_id",
|
|
1035
|
+
fields=[
|
|
1036
|
+
FieldSpec(name="event_id", data_type="integer", is_identifier=True),
|
|
1037
|
+
FieldSpec(name="status", data_type="string"),
|
|
1038
|
+
FieldSpec(name="amount", data_type="float"),
|
|
1039
|
+
FieldSpec(name="created_at", data_type="datetime"),
|
|
1040
|
+
],
|
|
1041
|
+
)
|
|
1042
|
+
],
|
|
1043
|
+
generation_settings=GenerationSettings(seed=123),
|
|
1044
|
+
)
|
|
1045
|
+
|
|
1046
|
+
rows_by_entity = generate_dataset(spec, seed=123)
|
|
1047
|
+
report = validate_dataset(rows_by_entity, spec)
|
|
1048
|
+
events = rows_by_entity["events"]
|
|
1049
|
+
```
|
|
1050
|
+
|
|
1051
|
+
Use adapter helpers when converting legacy Trino or CSV profile payloads into a
|
|
1052
|
+
`DatasetSpec`. Legacy `GenerationSpec` APIs remain supported for compatibility,
|
|
1053
|
+
but new integrations should target the domain-agnostic modules.
|
|
1054
|
+
|
|
1055
|
+
## Project Layout
|
|
1056
|
+
|
|
1057
|
+
- `src/test_data_agent/core/` - domain-agnostic dataset, entity, field,
|
|
1058
|
+
constraint, privacy, distribution, and settings models.
|
|
1059
|
+
- `src/test_data_agent/generation/` - deterministic synthetic dataset
|
|
1060
|
+
generation.
|
|
1061
|
+
- `src/test_data_agent/validation/` - dataset validation and rule checks.
|
|
1062
|
+
- `src/test_data_agent/adapters/` - safe adapters from CSV, JSON, Trino, and
|
|
1063
|
+
legacy specs into `DatasetProfile` or `DatasetSpec`.
|
|
1064
|
+
- `src/test_data_agent/csv_profiler.py` - safe single-CSV profiling.
|
|
1065
|
+
- `src/test_data_agent/mcp_trino_server.py` - safe read-only Trino MCP tools.
|
|
1066
|
+
- `src/test_data_agent/mcp_generator_server.py` - workspace-bounded MCP tools
|
|
1067
|
+
for profiling, spec inference, generation, validation, and export.
|
|
1068
|
+
- `src/test_data_agent/agent.py` - review-first agent orchestration over the
|
|
1069
|
+
deterministic pipeline.
|
|
1070
|
+
- `src/test_data_agent/safety.py` - profile and source-row reuse safety checks.
|
|
1071
|
+
- `src/test_data_agent/profiling/` - domain-agnostic CSV-folder profiling,
|
|
1072
|
+
relationship inference, constraint mining, and safe profile caching.
|
|
1073
|
+
- `src/test_data_agent/business_rules.py` - business-rule models and YAML loader.
|
|
1074
|
+
- `src/test_data_agent/rules_engine.py` - scenario application and invalid-case injection.
|
|
1075
|
+
- `src/test_data_agent/business_validator.py` - executable business-rule validation.
|
|
1076
|
+
- `src/test_data_agent/spec.py` - legacy single-table compatibility models.
|
|
1077
|
+
- `src/test_data_agent/generator.py` - legacy row-generation compatibility path.
|
|
1078
|
+
- `src/test_data_agent/validator.py` - legacy row-validation compatibility path.
|
|
1079
|
+
- `src/test_data_agent/cli.py` - local command-line interface.
|
|
1080
|
+
- `examples/` - safe profile examples.
|
|
1081
|
+
- `scripts/run_ai_demo.py` - Trino-profile to synthetic CSV demonstration.
|
|
1082
|
+
- `prompts/` - agent prompt templates.
|
|
1083
|
+
- `tests/` - unit tests with mocked/local inputs.
|
|
1084
|
+
|
|
1085
|
+
## Development Notes
|
|
1086
|
+
|
|
1087
|
+
- Keep generation deterministic with an explicit seed.
|
|
1088
|
+
- Keep Trino tools small, explicit, read-only, and bounded.
|
|
1089
|
+
- Prefer profiles, aggregates, distributions, and masked examples over samples.
|
|
1090
|
+
- Add tests for safety checks, PII masking, schema matching, and generator
|
|
1091
|
+
behavior when changing core logic.
|
|
1092
|
+
- Normal unit tests must not require real Trino access.
|