agent-paranoid-android 0.5.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. agent_paranoid_android-0.5.0.dist-info/METADATA +1092 -0
  2. agent_paranoid_android-0.5.0.dist-info/RECORD +77 -0
  3. agent_paranoid_android-0.5.0.dist-info/WHEEL +4 -0
  4. agent_paranoid_android-0.5.0.dist-info/entry_points.txt +4 -0
  5. agent_paranoid_android-0.5.0.dist-info/licenses/LICENSE +21 -0
  6. test_data_agent/__init__.py +75 -0
  7. test_data_agent/adapters/__init__.py +61 -0
  8. test_data_agent/adapters/csv_file.py +67 -0
  9. test_data_agent/adapters/csv_folder.py +82 -0
  10. test_data_agent/adapters/json_profile.py +111 -0
  11. test_data_agent/adapters/legacy_generation.py +448 -0
  12. test_data_agent/adapters/legacy_profile.py +257 -0
  13. test_data_agent/adapters/parquet_dataset.py +95 -0
  14. test_data_agent/adapters/trino_profile.py +43 -0
  15. test_data_agent/agent.py +325 -0
  16. test_data_agent/business_rules.py +33 -0
  17. test_data_agent/business_validator.py +9 -0
  18. test_data_agent/cli.py +468 -0
  19. test_data_agent/compat/__init__.py +61 -0
  20. test_data_agent/compat/commands.py +56 -0
  21. test_data_agent/compat/legacy_generation.py +39 -0
  22. test_data_agent/compat/legacy_outputs.py +53 -0
  23. test_data_agent/compat/legacy_spec.py +22 -0
  24. test_data_agent/compat/legacy_workflows.py +85 -0
  25. test_data_agent/core/__init__.py +76 -0
  26. test_data_agent/core/constraint.py +66 -0
  27. test_data_agent/core/dataset.py +126 -0
  28. test_data_agent/core/distribution.py +178 -0
  29. test_data_agent/core/entity.py +62 -0
  30. test_data_agent/core/field.py +73 -0
  31. test_data_agent/core/limits.py +307 -0
  32. test_data_agent/core/privacy.py +221 -0
  33. test_data_agent/core/relationship.py +22 -0
  34. test_data_agent/core/serialization.py +39 -0
  35. test_data_agent/core/settings.py +41 -0
  36. test_data_agent/csv_profiler.py +445 -0
  37. test_data_agent/generation/__init__.py +7 -0
  38. test_data_agent/generation/constraint_solver.py +165 -0
  39. test_data_agent/generation/entity_generator.py +265 -0
  40. test_data_agent/generation/planner.py +41 -0
  41. test_data_agent/generator.py +152 -0
  42. test_data_agent/io/__init__.py +92 -0
  43. test_data_agent/io/artifacts.py +165 -0
  44. test_data_agent/io/commands.py +338 -0
  45. test_data_agent/io/legacy_workflows.py +11 -0
  46. test_data_agent/io/readers.py +69 -0
  47. test_data_agent/io/workflows.py +554 -0
  48. test_data_agent/io/writers.py +156 -0
  49. test_data_agent/mcp_generator_server.py +403 -0
  50. test_data_agent/mcp_trino_server.py +1051 -0
  51. test_data_agent/profiling/__init__.py +49 -0
  52. test_data_agent/profiling/cache.py +85 -0
  53. test_data_agent/profiling/constraint_miner.py +165 -0
  54. test_data_agent/profiling/distribution_profiler.py +83 -0
  55. test_data_agent/profiling/relationship_profiler.py +53 -0
  56. test_data_agent/profiling/schema_profiler.py +325 -0
  57. test_data_agent/py.typed +1 -0
  58. test_data_agent/rules/__init__.py +33 -0
  59. test_data_agent/rules/business_config.py +106 -0
  60. test_data_agent/rules/conditions.py +37 -0
  61. test_data_agent/rules/contract.py +365 -0
  62. test_data_agent/rules/engine.py +157 -0
  63. test_data_agent/rules/expressions.py +148 -0
  64. test_data_agent/rules/models.py +210 -0
  65. test_data_agent/rules/scenarios.py +32 -0
  66. test_data_agent/rules/validation.py +184 -0
  67. test_data_agent/rules_engine.py +6 -0
  68. test_data_agent/safety.py +128 -0
  69. test_data_agent/scenario.py +5 -0
  70. test_data_agent/spec.py +355 -0
  71. test_data_agent/validation/__init__.py +5 -0
  72. test_data_agent/validation/constraint_validator.py +123 -0
  73. test_data_agent/validation/reconciliation.py +40 -0
  74. test_data_agent/validation/relationship_validator.py +34 -0
  75. test_data_agent/validation/schema_validator.py +67 -0
  76. test_data_agent/validator.py +129 -0
  77. test_data_agent/version.py +3 -0
@@ -0,0 +1,1092 @@
1
+ Metadata-Version: 2.4
2
+ Name: agent-paranoid-android
3
+ Version: 0.5.0
4
+ Summary: Safety-first synthetic data generation agent
5
+ Project-URL: Homepage, https://github.com/wa-pis/agent-paranoid-android
6
+ Project-URL: Repository, https://github.com/wa-pis/agent-paranoid-android
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Keywords: agent,csv,mcp,pii,synthetic-data,test-data,trino
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Developers
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Topic :: Database
17
+ Classifier: Topic :: Security
18
+ Classifier: Topic :: Software Development :: Testing
19
+ Classifier: Typing :: Typed
20
+ Requires-Python: >=3.11
21
+ Requires-Dist: faker>=25.0.0
22
+ Requires-Dist: mcp>=1.0.0
23
+ Requires-Dist: pyarrow>=15.0.0
24
+ Requires-Dist: pydantic>=2.7.0
25
+ Requires-Dist: pyyaml>=6.0.0
26
+ Requires-Dist: sqlglot>=30.0.0
27
+ Requires-Dist: trino>=0.330.0
28
+ Provides-Extra: dev
29
+ Requires-Dist: hatchling==1.31.0; extra == 'dev'
30
+ Requires-Dist: hypothesis>=6.100.0; extra == 'dev'
31
+ Requires-Dist: mypy>=2.3.0; extra == 'dev'
32
+ Requires-Dist: pip-audit>=2.7.0; extra == 'dev'
33
+ Requires-Dist: pytest-cov>=5.0.0; extra == 'dev'
34
+ Requires-Dist: pytest>=8.0.0; extra == 'dev'
35
+ Requires-Dist: ruff>=0.15.0; extra == 'dev'
36
+ Requires-Dist: types-pyyaml>=6.0.12; extra == 'dev'
37
+ Description-Content-Type: text/markdown
38
+
39
+ # Agent Paranoid Android
40
+
41
+ [![CI](https://github.com/wa-pis/agent-paranoid-android/actions/workflows/ci.yml/badge.svg)](https://github.com/wa-pis/agent-paranoid-android/actions/workflows/ci.yml)
42
+ [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
43
+
44
+ Safe, deterministic synthetic data generation for database and CSV-driven test
45
+ datasets.
46
+
47
+ The agent profiles schemas and safe aggregate metadata, detects likely
48
+ sensitive fields, builds generation specs, generates synthetic rows from an
49
+ explicit seed, validates the result, and exports data in common formats. It
50
+ never copies source rows into generated output.
51
+
52
+ The project name is a nod to [Radiohead](https://www.radiohead.com/deadairspace/)
53
+ and ["Paranoid Android"](https://music.apple.com/us/song/1097861770). This
54
+ project is unaffiliated with Radiohead, XL Recordings, Parlophone, or any
55
+ related rights holders.
56
+
57
+ ## Project Status
58
+
59
+ Current package version: `0.5.0`. The installable distribution is
60
+ `agent-paranoid-android`; the CLI remains `test-data-agent` for compatibility
61
+ and to keep the command focused on the generated-data use case.
62
+
63
+ The domain-agnostic `DatasetSpec` pipeline is the primary path for new work. It
64
+ supports safe CSV folder profiling, spec inference, deterministic multi-table
65
+ generation, constraint reconciliation, validation, and CSV/JSON/Parquet export.
66
+ The generator MCP server now exposes the same profile, infer, generate, validate,
67
+ and export workflow to AI clients without returning dataset rows through MCP.
68
+
69
+ Legacy `GenerationSpec` commands and imports remain available for compatibility,
70
+ but they emit deprecation warnings and should be treated as migration paths.
71
+
72
+ ## AI-Assisted Development
73
+
74
+ This project is developed with substantial assistance from AI coding tools.
75
+ All changes remain subject to human review, automated testing, and the same
76
+ security requirements as manually written code. AI assistance does not imply
77
+ endorsement by any AI provider.
78
+
79
+ ## Safety Model
80
+
81
+ Generated data must be synthetic. Source data is used only for structure and
82
+ safe metadata:
83
+
84
+ - column names
85
+ - inferred data types
86
+ - null ratios
87
+ - approximate distinct counts
88
+ - enum-like distributions for non-sensitive fields
89
+ - numeric ranges and percentiles
90
+ - date and timestamp ranges
91
+ - masked sensitive patterns
92
+
93
+ Likely PII and secret-like fields are treated as sensitive by default. Sensitive
94
+ CSV columns do not emit raw top values. Trino row-returning queries must be
95
+ read-only, bounded with `LIMIT`, and cannot use unrestricted `SELECT *`.
96
+ Local profile caches store only safe profile JSON: schema metadata,
97
+ aggregates, distributions, inferred rules, and masked patterns. They must not
98
+ store source rows or raw PII.
99
+
100
+ Forbidden behavior includes copying production rows, exposing raw PII, exporting
101
+ real rows, running DDL/DML SQL, and creating unrestricted SQL tools.
102
+
103
+ Generation APIs enforce a configurable per-entity row limit. The default is
104
+ `100000`; override it with `TEST_DATA_AGENT_MAX_GENERATION_COUNT`. Input and
105
+ output paths must be distinct, and folder bundles are assembled atomically.
106
+ Parquet keeps homogeneous numeric and boolean column types; intentionally mixed
107
+ invalid columns are stored as strings so negative datasets remain exportable.
108
+
109
+ Untrusted inputs are bounded before parsing. Defaults are 128 MiB per file,
110
+ 512 MiB total, 1,000,000 rows, 1,000 columns, 10,000,000 cells, 100 files,
111
+ and 1,000,000 characters per CSV field. Parquet metadata is checked before
112
+ reading and limits expanded data to 512 MiB. YAML is limited to 50 aliases and
113
+ 100 nesting levels.
114
+ Override these with `TEST_DATA_AGENT_MAX_INPUT_FILE_BYTES`,
115
+ `TEST_DATA_AGENT_MAX_TOTAL_INPUT_BYTES`, `TEST_DATA_AGENT_MAX_INPUT_ROWS`,
116
+ `TEST_DATA_AGENT_MAX_INPUT_COLUMNS`, `TEST_DATA_AGENT_MAX_INPUT_CELLS`,
117
+ `TEST_DATA_AGENT_MAX_INPUT_FILES`,
118
+ `TEST_DATA_AGENT_MAX_INPUT_CELL_CHARS`,
119
+ `TEST_DATA_AGENT_MAX_PARQUET_EXPANDED_BYTES`, `TEST_DATA_AGENT_MAX_YAML_ALIASES`,
120
+ and `TEST_DATA_AGENT_MAX_YAML_DEPTH`. Business-rule files and inline payloads
121
+ have a separate 1 MiB default limit controlled by
122
+ `TEST_DATA_AGENT_MAX_BUSINESS_RULES_BYTES`. Estimated business-rule work is
123
+ limited to 5,000,000 row/rule evaluations by default; override it with
124
+ `TEST_DATA_AGENT_MAX_BUSINESS_RULE_EVALUATIONS`.
125
+
126
+ Generated bundles are also bounded before publication. The defaults are
127
+ 512 MiB for all generated files, 128 MiB of free disk space left in reserve,
128
+ and five minutes of wall-clock generation time. Override them with
129
+ `TEST_DATA_AGENT_MAX_OUTPUT_BYTES`, `TEST_DATA_AGENT_MIN_FREE_DISK_BYTES`, and
130
+ `TEST_DATA_AGENT_MAX_GENERATION_SECONDS`. DatasetSpec workflows estimate output
131
+ size before allocating rows, check time throughout generation, and publish
132
+ folder bundles only after exact size and validation checks pass.
133
+
134
+ ## Install
135
+
136
+ Use Python 3.11 or newer.
137
+
138
+ ```bash
139
+ python3 -m pip install -e ".[dev]"
140
+ ```
141
+
142
+ For a reproducible development or CI environment, install the pinned `uv`
143
+ bootstrap version and sync from the committed lock file:
144
+
145
+ ```bash
146
+ python3 -m pip install "uv==0.11.23"
147
+ uv sync --frozen --extra dev --no-install-project
148
+ uv sync --frozen --extra dev --no-editable --no-build-isolation
149
+ uv export --quiet --frozen --extra dev --no-emit-project \
150
+ --output-file /tmp/agent-paranoid-android-audit.txt
151
+ uv run --no-sync python -m pip_audit --require-hashes \
152
+ --requirement /tmp/agent-paranoid-android-audit.txt
153
+ uv run --no-sync python -m mypy
154
+ ```
155
+
156
+ Version tags trigger a release gate that builds wheel and source archives,
157
+ installs the wheel in an isolated environment, verifies package metadata,
158
+ `py.typed`, CLI entry points, and `doctor --skip-smoke`, exports a CycloneDX
159
+ SBOM, writes SHA-256 checksums, creates GitHub provenance and SBOM attestations,
160
+ and publishes the verified files as a GitHub Release.
161
+
162
+ ## Quickstart
163
+
164
+ If you have one CSV file and just want synthetic data, start here:
165
+
166
+ ```bash
167
+ test-data-agent generate-from-csv data/customers.csv \
168
+ --count 100 \
169
+ --seed 12345 \
170
+ --format csv \
171
+ --output out/customers.csv
172
+ ```
173
+
174
+ If you have a folder with related CSV files, one file per table, start here:
175
+
176
+ ```bash
177
+ test-data-agent generate-from-example data/example_dataset \
178
+ --count 100 \
179
+ --seed 12345 \
180
+ --format csv \
181
+ --output out/generated
182
+ ```
183
+
184
+ On success, the CLI prints a short summary with the output location, generated
185
+ row counts, seed, validation status, and the `source rows copied: no` safety
186
+ check. It also writes review artifacts such as `generation_manifest.json` and
187
+ `validation_report.json`.
188
+
189
+ The example path below uses the checked-in fixture data and writes only local
190
+ artifacts under `out/`.
191
+
192
+ 1. Install the package and run the test suite:
193
+
194
+ ```bash
195
+ python3 -m pip install -e ".[dev]"
196
+ python3 -m pytest
197
+ ```
198
+
199
+ After installation, the `test-data-agent` CLI should be available. If your shell
200
+ cannot find it, run the same commands as `python3 -m test_data_agent.cli ...`.
201
+
202
+ 2. Generate a synthetic single-table CSV from an example source CSV:
203
+
204
+ ```bash
205
+ test-data-agent generate-from-csv tests/fixtures/customers.csv \
206
+ --count 25 \
207
+ --seed 12345 \
208
+ --format csv \
209
+ --output out/customers.csv
210
+ ```
211
+
212
+ This prints a summary and creates `out/customers.csv` plus review artifacts next
213
+ to it:
214
+ `csv_profile.json`, `generation_spec.json`, `validation_report.json`, and
215
+ `generation_manifest.json`.
216
+
217
+ 3. Generate a related multi-table dataset from an example CSV folder:
218
+
219
+ ```bash
220
+ test-data-agent generate-from-example tests/fixtures/example_dataset \
221
+ --count 25 \
222
+ --seed 12345 \
223
+ --format csv \
224
+ --output out/example_dataset
225
+ ```
226
+
227
+ This creates synthetic `customers.csv` and `orders.csv` files, a
228
+ `dataset_spec.yaml`, a validation report, and a generation manifest.
229
+
230
+ 4. Re-run with the same seed when you need identical output:
231
+
232
+ ```bash
233
+ test-data-agent generate-from-example tests/fixtures/example_dataset \
234
+ --count 25 \
235
+ --seed 12345 \
236
+ --format json \
237
+ --output out/example_dataset_json
238
+ ```
239
+
240
+ The source CSV files are used only for schema and safe profile metadata. The
241
+ generated rows should not copy source rows or expose raw PII.
242
+
243
+ Run tests:
244
+
245
+ ```bash
246
+ python3 -m pytest
247
+ ```
248
+
249
+ Run the same quality checks used by CI:
250
+
251
+ ```bash
252
+ python3 -m ruff check src tests
253
+ python3 -m mypy
254
+ python3 -m compileall -q src tests
255
+ python3 -m pytest --cov=test_data_agent --cov-report=term-missing --cov-fail-under=85
256
+ ```
257
+
258
+ CI runs these checks on Python 3.11 and 3.12. The security regression suite
259
+ also uses Hypothesis to exercise variations of PII aliases, SQL statement
260
+ tails, duplicate CSV headers, and sensitive-value masking. Pull requests also
261
+ run dependency review and fail when they introduce a dependency with a known
262
+ Moderate-or-higher vulnerability.
263
+
264
+ For a local environment and quickstart smoke check:
265
+
266
+ ```bash
267
+ test-data-agent doctor
268
+ ```
269
+
270
+ ## MVP Readiness Checklist
271
+
272
+ Use this checklist to keep the project scoped to a useful safety-first MVP:
273
+
274
+ - `generate-from-csv tests/fixtures/customers.csv ...` creates synthetic rows,
275
+ `csv_profile.json`, `generation_spec.json`, `validation_report.json`, and
276
+ `generation_manifest.json`.
277
+ - `generate-from-example tests/fixtures/example_dataset ...` creates related
278
+ synthetic tables plus `profile.json`, `dataset_spec.yaml`,
279
+ `validation_report.json`, and `generation_manifest.json`.
280
+ - Re-running the same spec with the same seed is deterministic.
281
+ - Generated manifests report `synthetic: true` and `source_rows_copied: false`.
282
+ - CSV profiles contain schema, safe distributions, ranges, and masked patterns,
283
+ not source rows or raw PII.
284
+ - Dataset validation is performed by deterministic Python code, not by
285
+ free-form LLM reasoning.
286
+ - Trino-facing queries remain read-only, allowlisted, bounded by `LIMIT`, and
287
+ covered by unsafe-SQL regression tests.
288
+ - Legacy `GenerationSpec` support is compatibility-only and remains outside the
289
+ primary quickstart path.
290
+
291
+ ## Release Checklist
292
+
293
+ Before merging or cutting a release:
294
+
295
+ - Review the relevant [OpenSpec Baseline](openspec/project.md) capability specs
296
+ and update them when behavior changes.
297
+ - Run `scripts/check_release.sh`; it covers linting, compilation, coverage,
298
+ DatasetSpec schema freshness, and the quickstart fixture smoke flow.
299
+ - Regenerate `schemas/dataset_spec.schema.json` with
300
+ `python3 scripts/export_dataset_schema.py` when `DatasetSpec` changes.
301
+ - Update `CHANGELOG.md`, README examples, and docs for user-visible behavior or
302
+ safety-contract changes.
303
+ - Keep post-MVP features behind explicit OpenSpec changes instead of expanding
304
+ the MVP silently.
305
+
306
+ ## Documentation
307
+
308
+ Start here if you want to understand the newer domain-agnostic multi-table
309
+ pipeline:
310
+
311
+ - [Domain-Agnostic Workflow](docs/domain_agnostic_workflow.md)
312
+ - [Dataset Profile And Spec Reference](docs/dataset_profile_and_spec.md)
313
+ - [Agent Design](docs/agent_design.md)
314
+ - [AI Integration](docs/ai_integration.md)
315
+ - [MCP Examples](docs/mcp_examples.md)
316
+ - [Generator MCP Design Rationale](docs/mcp_generator_design.md)
317
+ - [OpenSpec Baseline](openspec/project.md)
318
+ - [Release Process](docs/release.md)
319
+ - [Public Release Checklist](docs/public_release_checklist.md)
320
+ - [Security Policy](SECURITY.md)
321
+ - [Contributing](CONTRIBUTING.md)
322
+ - [License](LICENSE)
323
+ - [DatasetSpec JSON Schema](schemas/dataset_spec.schema.json)
324
+ - [Roadmap](docs/roadmap.md)
325
+ - [Changelog](CHANGELOG.md)
326
+ - [Implementation Map](docs/implementation_map.md)
327
+ - [Architecture Overview Diagram](docs/architecture.puml)
328
+ - [Agent Workflow Diagram](docs/architecture_agent_workflow.puml)
329
+ - [Safety Boundaries Diagram](docs/architecture_safety_boundaries.puml)
330
+
331
+ ## How To Use It
332
+
333
+ There are five normal ways to use the project:
334
+
335
+ 1. Start from a hand-written generation spec.
336
+ 2. Start from a CSV file and let the agent infer a safe profile.
337
+ 3. Start from safe Trino/profile metadata and generate from that profile.
338
+ 4. Start from an example multi-table CSV folder and infer a dataset spec.
339
+ 5. Let an MCP-compatible AI client orchestrate the same safe workflow.
340
+
341
+ Each flow produces synthetic data plus validation artifacts.
342
+
343
+ ### 1. Generate From A Spec
344
+
345
+ Create `dataset_spec.yaml`:
346
+
347
+ ```yaml
348
+ schema_version: '1.0'
349
+ entities:
350
+ - name: customers
351
+ row_count: 3
352
+ primary_key: customer_id
353
+ fields:
354
+ - name: customer_id
355
+ data_type: integer
356
+ is_identifier: true
357
+ distribution:
358
+ kind: synthetic_identifier
359
+ - name: email
360
+ data_type: string
361
+ sensitive: true
362
+ semantic_type: email
363
+ distribution:
364
+ kind: masked_patterns
365
+ patterns:
366
+ - pattern: email
367
+ count: 1
368
+ - name: status
369
+ data_type: string
370
+ distribution:
371
+ kind: categorical
372
+ categories:
373
+ - {value: new, count: 1}
374
+ - {value: active, count: 2}
375
+ relationships: []
376
+ constraints: []
377
+ generation_settings:
378
+ seed: 42
379
+ output_format: json
380
+ ```
381
+
382
+ Generate rows:
383
+
384
+ ```bash
385
+ test-data-agent generate dataset_spec.yaml --output out/customers
386
+ ```
387
+
388
+ Validate rows against the spec:
389
+
390
+ ```bash
391
+ test-data-agent validate out/customers/dataset_spec.yaml out/customers
392
+ ```
393
+
394
+ The `generate` command writes:
395
+
396
+ - `out/customers/customers.json`
397
+ - `out/customers/dataset_spec.yaml`
398
+ - `out/customers/validation_report.json`
399
+ - `out/customers/generation_manifest.json`
400
+
401
+ ### 2. Generate From A CSV
402
+
403
+ First create a safe CSV profile:
404
+
405
+ ```bash
406
+ test-data-agent profile-csv data/customers.csv \
407
+ --output out/customers_profile.json
408
+ ```
409
+
410
+ Inspect the profile before using it. It should contain schema, aggregates,
411
+ distributions, ranges, and masked patterns, not raw sensitive values.
412
+
413
+ Generate synthetic data from the CSV:
414
+
415
+ ```bash
416
+ test-data-agent generate-from-csv data/customers.csv \
417
+ --count 1000 \
418
+ --mode valid \
419
+ --seed 12345 \
420
+ --format csv \
421
+ --output out/customers.csv
422
+ ```
423
+
424
+ For a mixed valid/invalid dataset:
425
+
426
+ ```bash
427
+ test-data-agent generate-from-csv data/customers.csv \
428
+ --count 1000 \
429
+ --mode mixed \
430
+ --invalid-ratio 0.02 \
431
+ --seed 12345 \
432
+ --format parquet \
433
+ --output out/customers.parquet
434
+ ```
435
+
436
+ `generate-from-csv` writes:
437
+
438
+ - the requested output file
439
+ - `csv_profile.json`
440
+ - `generation_spec.json`
441
+ - `validation_report.json`
442
+ - `generation_manifest.json`
443
+
444
+ Supported output formats are `json`, `csv`, and `parquet`.
445
+
446
+ ### 3. Generate From A Profile
447
+
448
+ Use a safe profile such as `examples/orders_profile.json`:
449
+
450
+ ```bash
451
+ test-data-agent generate \
452
+ --profile examples/orders_profile.json \
453
+ --count 10000 \
454
+ --mode mixed \
455
+ --invalid-ratio 0.02 \
456
+ --seed 12345 \
457
+ --format csv \
458
+ --output out/orders.csv
459
+ ```
460
+
461
+ This writes:
462
+
463
+ - `out/orders.csv`
464
+ - `out/generation_spec.json`
465
+ - `out/validation_report.json`
466
+ - `out/generation_manifest.json`
467
+
468
+ Profile input should contain safe metadata only. Do not include raw production
469
+ samples.
470
+
471
+ ### 4. Generate From An Example Dataset
472
+
473
+ For domain-agnostic multi-table generation, place one CSV per table in a folder:
474
+
475
+ ```text
476
+ example_dataset/
477
+ customers.csv
478
+ orders.csv
479
+ ```
480
+
481
+ Profile the folder without exposing raw PII:
482
+
483
+ ```bash
484
+ test-data-agent profile-example example_dataset \
485
+ --output out/profile.json
486
+ ```
487
+
488
+ Infer a YAML dataset spec with schema, relationships, distributions, formulas,
489
+ temporal rules, conditional rules, and aggregate mappings:
490
+
491
+ ```bash
492
+ test-data-agent infer-spec out/profile.json \
493
+ --count 1000 \
494
+ --output out/dataset_spec.yaml
495
+ ```
496
+
497
+ Generate all related tables:
498
+
499
+ ```bash
500
+ test-data-agent generate out/dataset_spec.yaml \
501
+ --seed 12345 \
502
+ --format csv \
503
+ --output out/generated
504
+ ```
505
+
506
+ Validate the generated folder:
507
+
508
+ ```bash
509
+ test-data-agent validate out/generated/dataset_spec.yaml out/generated \
510
+ --output out/generated/validation_report.json
511
+ ```
512
+
513
+ Or run the full flow in one command:
514
+
515
+ ```bash
516
+ test-data-agent generate-from-example example_dataset \
517
+ --seed 12345 \
518
+ --count 1000 \
519
+ --format parquet \
520
+ --output out/generated
521
+ ```
522
+
523
+ All identifiers are regenerated synthetically. Foreign keys are preserved by
524
+ wiring child rows to generated parent IDs, never by reusing source IDs.
525
+ The generated folder also contains the effective `dataset_spec.yaml`,
526
+ `validation_report.json`, and `generation_manifest.json`.
527
+
528
+ Large CSV folders are profiled in a streaming pass, so the profiler does not
529
+ hold every source row in memory. Schema, null ratios, safe distributions, and
530
+ field metadata are computed across the full files. Relationship and constraint
531
+ mining use a bounded local sample because they need row-level comparisons:
532
+
533
+ ```bash
534
+ test-data-agent profile-example example_dataset \
535
+ --output out/profile.json \
536
+ --rule-sample-rows 100000
537
+ ```
538
+
539
+ Profiles are cached by CSV file names, sizes, modification times, and the
540
+ `--rule-sample-rows` value:
541
+
542
+ ```bash
543
+ test-data-agent profile-example example_dataset \
544
+ --output out/profile.json \
545
+ --cache-dir .test_data_agent_cache/profiles
546
+ ```
547
+
548
+ Use `--no-cache` when you need to force a fresh profile. The cache contains
549
+ safe profile metadata only, not source rows. Cache writes are atomic, and a
550
+ stale or incomplete cache is treated as a miss.
551
+
552
+ ## CLI Reference
553
+
554
+ Profile an example multi-table CSV folder:
555
+
556
+ ```bash
557
+ test-data-agent profile-example INPUT_FOLDER --output PROFILE.json
558
+ ```
559
+
560
+ Infer a YAML dataset spec:
561
+
562
+ ```bash
563
+ test-data-agent infer-spec PROFILE.json --output DATASET_SPEC.yaml
564
+ ```
565
+
566
+ Profile a CSV:
567
+
568
+ ```bash
569
+ test-data-agent profile-csv INPUT.csv --output PROFILE.json
570
+ ```
571
+
572
+ Generate from a spec:
573
+
574
+ ```bash
575
+ test-data-agent generate DATASET_SPEC.yaml --format csv --output OUTPUT_FOLDER
576
+ # Deprecated compatibility path:
577
+ test-data-agent generate LEGACY_GENERATION_SPEC.json --output OUTPUT.json
578
+ ```
579
+
580
+ Generate from a safe profile:
581
+
582
+ ```bash
583
+ test-data-agent generate \
584
+ --profile PROFILE.json \
585
+ --count 1000 \
586
+ --seed 12345 \
587
+ --format json \
588
+ --output OUTPUT.json
589
+ ```
590
+
591
+ Generate directly from CSV:
592
+
593
+ ```bash
594
+ test-data-agent generate-from-csv INPUT.csv \
595
+ --count 1000 \
596
+ --seed 12345 \
597
+ --format csv \
598
+ --output OUTPUT.csv
599
+ ```
600
+
601
+ Generate directly from an example multi-table folder:
602
+
603
+ ```bash
604
+ test-data-agent generate-from-example INPUT_FOLDER \
605
+ --count 1000 \
606
+ --seed 12345 \
607
+ --format csv \
608
+ --output OUTPUT_FOLDER
609
+ ```
610
+
611
+ Validate generated rows (the first form is the deprecated compatibility path):
612
+
613
+ ```bash
614
+ test-data-agent validate SPEC.json ROWS.json
615
+ test-data-agent validate DATASET_SPEC.yaml OUTPUT_FOLDER
616
+ ```
617
+
618
+ Plan and approve an AI-agent workflow:
619
+
620
+ ```bash
621
+ test-data-agent agent-plan INPUT_FOLDER \
622
+ --source-type csv-folder \
623
+ --workspace out/agent \
624
+ --count 100 \
625
+ --seed 12345 \
626
+ --format csv
627
+ test-data-agent agent-approve out/agent
628
+ ```
629
+
630
+ Useful options:
631
+
632
+ - `--count` overrides or supplies row count.
633
+ - `--seed` makes generation reproducible.
634
+ - `--format json|csv|parquet` selects output format.
635
+ - `--mode valid|mixed|negative|edge|load_test` selects the generation mode.
636
+ - `--invalid-ratio 0.02` injects invalid values in `mixed` mode; `negative`
637
+ mode intentionally makes every generated value invalid.
638
+ - `--table NAME` sets the table name for CSV profiling.
639
+ - `--cache-dir PATH` selects the safe profile cache for example-folder
640
+ profiling.
641
+ - `--no-cache` disables profile cache reuse.
642
+ - `--overwrite` allows replacing existing single-file outputs such as generated
643
+ CSV/JSON/Parquet files, profile JSON files, validation reports, and inferred
644
+ specs. Without it, the CLI refuses to overwrite existing files.
645
+ - `--rule-sample-rows N` bounds row-level relationship and constraint mining
646
+ while full-file schema and distribution profiling remains streaming.
647
+
648
+ Check local setup and run a small fixture smoke test:
649
+
650
+ ```bash
651
+ test-data-agent doctor
652
+ ```
653
+
654
+ Use `test-data-agent doctor --skip-smoke` when you only want dependency and
655
+ Python-version checks.
656
+
657
+ ## Architecture
658
+
659
+ The architecture is documented as PlantUML diagrams:
660
+
661
+ - `docs/architecture.puml` shows the high-level application components.
662
+ - `docs/architecture_agent_workflow.puml` shows the review-first agent flow.
663
+ - `docs/architecture_safety_boundaries.puml` shows the trust and safety
664
+ boundaries around AI planning, MCP tools, profiling, generation, and
665
+ validation.
666
+
667
+ ## Legacy GenerationSpec Compatibility
668
+
669
+ Legacy single-table specs are Pydantic models serialized as JSON. New
670
+ integrations should use the versioned `DatasetSpec` shown above. Legacy data
671
+ types remain:
672
+
673
+ - `integer`
674
+ - `float`
675
+ - `boolean`
676
+ - `string`
677
+ - `date`
678
+ - `datetime`
679
+ - `email`
680
+ - `phone`
681
+ - `name`
682
+ - `address`
683
+ - `uuid`
684
+
685
+ Supported strategies:
686
+
687
+ - `sequence`
688
+ - `random_int`
689
+ - `random_float`
690
+ - `random_boolean`
691
+ - `faker`
692
+ - `choice`
693
+ - `constant`
694
+ - `date_range`
695
+ - `datetime_range`
696
+ - `uuid`
697
+
698
+ Example with ranges and nullable values:
699
+
700
+ ```json
701
+ {
702
+ "seed": 7,
703
+ "output_format": "csv",
704
+ "table": {
705
+ "name": "orders",
706
+ "row_count": 100,
707
+ "columns": [
708
+ {"name": "order_id", "data_type": "integer", "strategy": "sequence"},
709
+ {
710
+ "name": "status",
711
+ "data_type": "string",
712
+ "strategy": "choice",
713
+ "choices": ["new", "paid", "shipped", "cancelled"]
714
+ },
715
+ {
716
+ "name": "order_total",
717
+ "data_type": "float",
718
+ "min_value": 1,
719
+ "max_value": 500
720
+ },
721
+ {
722
+ "name": "created_at",
723
+ "data_type": "datetime",
724
+ "min_datetime": "2024-01-01T00:00:00",
725
+ "max_datetime": "2024-12-31T23:59:59"
726
+ },
727
+ {
728
+ "name": "coupon_code",
729
+ "data_type": "string",
730
+ "nullable": true,
731
+ "null_probability": 0.25
732
+ }
733
+ ]
734
+ }
735
+ }
736
+ ```
737
+
738
+ ## Trino MCP Server
739
+
740
+ The MCP server exposes small safe tools for metadata and profiling:
741
+
742
+ - `list_catalogs`
743
+ - `list_schemas`
744
+ - `list_tables`
745
+ - `describe_table`
746
+ - `profile_table`
747
+ - `profile_table_safe`
748
+ - `profile_column`
749
+ - `profile_foreign_key`
750
+ - `profile_temporal_ordering`
751
+ - `profile_formula_rule`
752
+ - `profile_conditional_required`
753
+ - `profile_conditional_allowed_values`
754
+ - `profile_aggregate_mapping`
755
+ - `sample_rows_masked`
756
+ - `run_safe_select`
757
+
758
+ Configure Trino with environment variables:
759
+
760
+ ```bash
761
+ TRINO_HOST=trino.example.internal
762
+ TRINO_PORT=443
763
+ TRINO_USER=your_user
764
+ TRINO_HTTP_SCHEME=https
765
+ TRINO_ALLOWED_CATALOGS=hive,iceberg
766
+ TRINO_ALLOWED_SCHEMAS=dev,test,staging
767
+ TRINO_QUERY_MAX_EXECUTION_TIME=30s
768
+ TRINO_QUERY_MAX_RUN_TIME=45s
769
+ TRINO_QUERY_MAX_SCAN_PHYSICAL_BYTES=1GB
770
+ ```
771
+
772
+ Start the server:
773
+
774
+ ```bash
775
+ python3 -m test_data_agent.mcp_trino_server
776
+ ```
777
+
778
+ `run_safe_select` rejects DDL, DML, executable statements, multiple statements,
779
+ unbounded row-returning queries without a top-level `LIMIT`, unrestricted
780
+ `SELECT *`, and projections of likely PII fields even when they are aliased.
781
+ Both catalog and schema allowlists are required, and arbitrary selects must use
782
+ fully qualified `catalog.schema.table` references that match them. The server
783
+ also defaults to HTTPS and refuses plain HTTP.
784
+ SQL validation uses `sqlglot` AST parsing for Trino syntax. Masked sampling uses
785
+ conservative field-name and value-content detection for likely PII and secrets.
786
+
787
+ For an intentionally unrestricted local environment, set
788
+ `TRINO_ALLOW_UNRESTRICTED=true`. For a local Trino endpoint that cannot use TLS,
789
+ also set `TRINO_ALLOW_INSECURE_HTTP=true`; neither override should be used for
790
+ production access.
791
+ Trino responses are also capped client-side at 10,000 rows; lower this with
792
+ `TRINO_MAX_RESULT_ROWS` when metadata namespaces are small.
793
+ Every connection also applies server-side session budgets for execution time,
794
+ total run time, and physical bytes scanned. The defaults are `30s`, `45s`, and
795
+ `1GB`; lower the corresponding `TRINO_QUERY_MAX_*` variables for narrower
796
+ environments. Values are validated locally and cannot exceed one hour of
797
+ execution, two hours of total run time, or 100 GB scanned.
798
+
799
+ For large Trino tables, use safe profiling instead of downloading rows. The
800
+ safe profile path pushes aggregate work into Trino: row counts, null ratios,
801
+ approximate distinct counts, numeric ranges and percentiles, and timestamp
802
+ ranges are returned as compact metadata. Low-cardinality top values are fetched
803
+ only for non-sensitive string columns and always with a bounded `LIMIT`.
804
+ Sensitive columns never return raw top values. Save the resulting profile JSON
805
+ and reuse it for generation so repeated runs do not re-query the source table.
806
+
807
+ Consistency profiling is also aggregate-only. Use the dedicated rule tools to
808
+ measure whether inferred or proposed rules hold before adding them to a dataset
809
+ spec:
810
+
811
+ - `profile_foreign_key` returns child checked/matched/orphan counts.
812
+ - `profile_temporal_ordering` returns pass/fail counts for timestamp ordering.
813
+ - `profile_formula_rule` returns pass/fail counts and numeric residuals for
814
+ simple arithmetic formulas.
815
+ - `profile_conditional_required` returns scoped present/missing counts without
816
+ echoing condition values.
817
+ - `profile_conditional_allowed_values` returns scoped allowed/violation counts.
818
+ - `profile_aggregate_mapping` compares parent aggregate fields with child
819
+ `sum` or `count` aggregates.
820
+
821
+ Each rule profile includes `confidence` and `status`. The tools do not return
822
+ source rows, identifiers, or raw PII values.
823
+
824
+ ## Generator MCP Server
825
+
826
+ The second MCP server exposes the synthetic pipeline to AI clients:
827
+
828
+ - `profile_csv`
829
+ - `infer_dataset_spec`
830
+ - `generate_dataset`
831
+ - `validate_dataset`
832
+ - `export_dataset`
833
+
834
+ Start it from the workspace that contains allowed inputs and outputs:
835
+
836
+ ```bash
837
+ TEST_DATA_AGENT_WORKSPACE_ROOT=/path/to/allowed/workspace \
838
+ python3 -m test_data_agent.mcp_generator_server
839
+ ```
840
+
841
+ All tool paths are resolved inside `TEST_DATA_AGENT_WORKSPACE_ROOT`; traversal
842
+ and symlink escapes are rejected. MCP responses contain paths, row counts,
843
+ version metadata, and validation results, never generated rows. `export_dataset`
844
+ always generates fresh synthetic data from a `DatasetSpec`; it cannot convert or
845
+ export arbitrary source rows.
846
+
847
+ Generated bundles contain `dataset_spec.yaml`, `validation_report.json`, and
848
+ `generation_manifest.json`. The manifest records the package and schema
849
+ versions, spec fingerprint, seed, output format, row counts, validation status,
850
+ and the explicit provenance flags `synthetic: true` and
851
+ `source_rows_copied: false`. Runs with business rules also record a SHA-256
852
+ fingerprint, rule count, pass/fail counts, truncation status, and overall
853
+ business validity.
854
+
855
+ `infer_dataset_spec` accepts exactly one of a workspace `profile_path` or an
856
+ inline `profile_payload` from the Trino MCP server. MCP tools never overwrite
857
+ existing output files, and generation requires a new or empty output folder.
858
+ `generate_dataset` and `export_dataset` accept at most one of a workspace
859
+ `business_rules_path` or structured `business_rules_payload`. Unknown keys,
860
+ dangling references, unsupported expressions, oversized inputs, and
861
+ raw-looking sensitive literals fail before output is created.
862
+
863
+ Example MCP rule payload:
864
+
865
+ ```json
866
+ {
867
+ "field_rules": [
868
+ {
869
+ "table": "orders",
870
+ "field": "status",
871
+ "required": true,
872
+ "allowed_values": ["new", "paid", "cancelled"]
873
+ }
874
+ ]
875
+ }
876
+ ```
877
+
878
+ The MCP response returns only the compact business-validation summary and
879
+ `business_validation_report_path`; detailed bounded errors remain in the
880
+ workspace artifact and generated rows are never returned inline.
881
+
882
+ Run the local end-to-end example from safe Trino profile metadata:
883
+
884
+ ```bash
885
+ python scripts/run_ai_demo.py --output out/ai_demo --count 100 --seed 12345
886
+ ```
887
+
888
+ ## DatasetSpec Generation
889
+
890
+ Use `DatasetSpec` when synthetic datasets need deterministic relationships such
891
+ as foreign keys:
892
+
893
+ ```python
894
+ from test_data_agent.core.dataset import DatasetSpec
895
+ from test_data_agent.core.entity import EntitySpec
896
+ from test_data_agent.core.field import FieldSpec
897
+ from test_data_agent.core.relationship import Relationship
898
+ from test_data_agent.core.settings import GenerationSettings
899
+ from test_data_agent.generation.entity_generator import generate_dataset
900
+
901
+ spec = DatasetSpec(
902
+ entities=[
903
+ EntitySpec(
904
+ name="customers",
905
+ row_count=100,
906
+ primary_key="customer_id",
907
+ fields=[
908
+ FieldSpec(name="customer_id", data_type="integer", is_identifier=True),
909
+ FieldSpec(name="status", data_type="string"),
910
+ ],
911
+ ),
912
+ EntitySpec(
913
+ name="orders",
914
+ row_count=1000,
915
+ primary_key="order_id",
916
+ fields=[
917
+ FieldSpec(name="order_id", data_type="integer", is_identifier=True),
918
+ FieldSpec(name="customer_id", data_type="integer"),
919
+ ],
920
+ ),
921
+ ],
922
+ relationships=[
923
+ Relationship(
924
+ child_entity="orders",
925
+ child_field="customer_id",
926
+ parent_entity="customers",
927
+ parent_field="customer_id",
928
+ confidence=1.0,
929
+ status="confirmed",
930
+ )
931
+ ],
932
+ generation_settings=GenerationSettings(seed=12345),
933
+ )
934
+
935
+ rows_by_entity = generate_dataset(spec, seed=12345)
936
+ ```
937
+
938
+ Parent and child rows are generated synthetically from safe CSV-derived or
939
+ Trino-derived profile metadata. Foreign-key values are assigned from generated
940
+ parent rows, never copied from source data. Legacy `MultiTableGenerationSpec`
941
+ support remains available through compatibility adapters while downstream code
942
+ migrates.
943
+
944
+ ## Business Rules
945
+
946
+ Business logic is represented as structured YAML or JSON and enforced by code,
947
+ not by free-form LLM reasoning. Current rule models support:
948
+
949
+ - field rules
950
+ - conditional required fields
951
+ - conditional allowed values
952
+ - temporal ordering
953
+ - row formulas
954
+ - foreign keys
955
+ - aggregate formulas
956
+ - scenario distributions
957
+
958
+ Example:
959
+
960
+ ```yaml
961
+ field_rules:
962
+ - table: orders
963
+ field: status
964
+ required: true
965
+ allowed_values: [new, paid, shipped, cancelled]
966
+ - table: orders
967
+ field: order_total
968
+ required: true
969
+ min_value: 0
970
+
971
+ row_rules:
972
+ - type: temporal_ordering
973
+ table: orders
974
+ start_field: created_at
975
+ end_field: fulfilled_at
976
+ allow_equal: true
977
+
978
+ scenarios:
979
+ - name: paid_order
980
+ weight: 8
981
+ field_values:
982
+ orders:
983
+ status: paid
984
+ - name: cancelled_order
985
+ weight: 1
986
+ field_values:
987
+ orders:
988
+ status: cancelled
989
+ ```
990
+
991
+ Business rules can be applied from the CLI with `--business-rules` on
992
+ `generate`, `generate-from-csv`, and profile-based generation. Generator MCP
993
+ calls use `business_rules_path` or `business_rules_payload`. Rules are strict:
994
+ unknown keys and fields fail closed, formulas use a bounded arithmetic AST,
995
+ and concrete PII or secret values are rejected by both CLI and MCP workflows.
996
+ They are also available from Python:
997
+
998
+ ```python
999
+ from pathlib import Path
1000
+
1001
+ from test_data_agent.business_rules import load_business_rules
1002
+ from test_data_agent.business_validator import validate_business_rules
1003
+ from test_data_agent.rules_engine import apply_business_rules
1004
+
1005
+ rules = load_business_rules(Path("rules/orders.yaml"))
1006
+ rows_by_table = {"orders": rows}
1007
+
1008
+ apply_business_rules(
1009
+ rows_by_table,
1010
+ rules,
1011
+ seed=12345,
1012
+ mode="mixed",
1013
+ invalid_ratio=0.02,
1014
+ )
1015
+
1016
+ report = validate_business_rules(rows_by_table, rules)
1017
+ ```
1018
+
1019
+ ## Python API
1020
+
1021
+ ```python
1022
+ from test_data_agent.core.dataset import DatasetSpec
1023
+ from test_data_agent.core.entity import EntitySpec
1024
+ from test_data_agent.core.field import FieldSpec
1025
+ from test_data_agent.core.settings import GenerationSettings
1026
+ from test_data_agent.generation.entity_generator import generate_dataset
1027
+ from test_data_agent.validation import validate_dataset
1028
+
1029
+ spec = DatasetSpec(
1030
+ entities=[
1031
+ EntitySpec(
1032
+ name="events",
1033
+ row_count=100,
1034
+ primary_key="event_id",
1035
+ fields=[
1036
+ FieldSpec(name="event_id", data_type="integer", is_identifier=True),
1037
+ FieldSpec(name="status", data_type="string"),
1038
+ FieldSpec(name="amount", data_type="float"),
1039
+ FieldSpec(name="created_at", data_type="datetime"),
1040
+ ],
1041
+ )
1042
+ ],
1043
+ generation_settings=GenerationSettings(seed=123),
1044
+ )
1045
+
1046
+ rows_by_entity = generate_dataset(spec, seed=123)
1047
+ report = validate_dataset(rows_by_entity, spec)
1048
+ events = rows_by_entity["events"]
1049
+ ```
1050
+
1051
+ Use adapter helpers when converting legacy Trino or CSV profile payloads into a
1052
+ `DatasetSpec`. Legacy `GenerationSpec` APIs remain supported for compatibility,
1053
+ but new integrations should target the domain-agnostic modules.
1054
+
1055
+ ## Project Layout
1056
+
1057
+ - `src/test_data_agent/core/` - domain-agnostic dataset, entity, field,
1058
+ constraint, privacy, distribution, and settings models.
1059
+ - `src/test_data_agent/generation/` - deterministic synthetic dataset
1060
+ generation.
1061
+ - `src/test_data_agent/validation/` - dataset validation and rule checks.
1062
+ - `src/test_data_agent/adapters/` - safe adapters from CSV, JSON, Trino, and
1063
+ legacy specs into `DatasetProfile` or `DatasetSpec`.
1064
+ - `src/test_data_agent/csv_profiler.py` - safe single-CSV profiling.
1065
+ - `src/test_data_agent/mcp_trino_server.py` - safe read-only Trino MCP tools.
1066
+ - `src/test_data_agent/mcp_generator_server.py` - workspace-bounded MCP tools
1067
+ for profiling, spec inference, generation, validation, and export.
1068
+ - `src/test_data_agent/agent.py` - review-first agent orchestration over the
1069
+ deterministic pipeline.
1070
+ - `src/test_data_agent/safety.py` - profile and source-row reuse safety checks.
1071
+ - `src/test_data_agent/profiling/` - domain-agnostic CSV-folder profiling,
1072
+ relationship inference, constraint mining, and safe profile caching.
1073
+ - `src/test_data_agent/business_rules.py` - business-rule models and YAML loader.
1074
+ - `src/test_data_agent/rules_engine.py` - scenario application and invalid-case injection.
1075
+ - `src/test_data_agent/business_validator.py` - executable business-rule validation.
1076
+ - `src/test_data_agent/spec.py` - legacy single-table compatibility models.
1077
+ - `src/test_data_agent/generator.py` - legacy row-generation compatibility path.
1078
+ - `src/test_data_agent/validator.py` - legacy row-validation compatibility path.
1079
+ - `src/test_data_agent/cli.py` - local command-line interface.
1080
+ - `examples/` - safe profile examples.
1081
+ - `scripts/run_ai_demo.py` - Trino-profile to synthetic CSV demonstration.
1082
+ - `prompts/` - agent prompt templates.
1083
+ - `tests/` - unit tests with mocked/local inputs.
1084
+
1085
+ ## Development Notes
1086
+
1087
+ - Keep generation deterministic with an explicit seed.
1088
+ - Keep Trino tools small, explicit, read-only, and bounded.
1089
+ - Prefer profiles, aggregates, distributions, and masked examples over samples.
1090
+ - Add tests for safety checks, PII masking, schema matching, and generator
1091
+ behavior when changing core logic.
1092
+ - Normal unit tests must not require real Trino access.