sqlseed-ai 0.2.4__tar.gz → 0.2.5__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/PKG-INFO +58 -10
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/README.md +55 -7
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/README.zh-CN.md +39 -6
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/pyproject.toml +2 -2
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/_client.py +38 -2
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/_json_utils.py +69 -29
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/_prompts.py +19 -3
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/analyzer/_caller.py +67 -15
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/analyzer/_context.py +17 -9
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/analyzer/_json_parser.py +3 -2
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/analyzer/_streaming.py +95 -18
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/analyzer/_tool_calling.py +49 -26
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/cli/ai_commands.py +59 -27
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/errors.py +10 -1
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/level2_column_healer.py +7 -6
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/mcp.py +38 -24
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/refiner.py +69 -11
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/runtime.py +2 -2
- sqlseed_ai-0.2.5/tests/healer/conftest.py +66 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_heal_orchestrator_real.py +2 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_level1_subgraph_healer_real.py +2 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_level2_column_healer_real.py +1 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_level3_compact_healer_real.py +3 -0
- sqlseed_ai-0.2.5/tests/healer/test_llm_availability.py +112 -0
- sqlseed_ai-0.2.5/tests/http_helpers.py +23 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_caller.py +20 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_client.py +3 -3
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_json_utils.py +59 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_plugin.py +9 -6
- sqlseed_ai-0.2.5/tests/test_ai_preserve_names.py +465 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_prompts_tools.py +16 -0
- sqlseed_ai-0.2.5/tests/test_cli_suggestion_diagnostics.py +211 -0
- sqlseed_ai-0.2.5/tests/test_cli_suggestion_identity.py +139 -0
- sqlseed_ai-0.2.5/tests/test_client_loopback.py +170 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_mcp.py +90 -22
- sqlseed_ai-0.2.5/tests/test_mcp_error_redaction.py +29 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_mcp_stdio.py +9 -11
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_nonweb_ai_boundaries.py +23 -3
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_quality_error_boundaries.py +24 -0
- sqlseed_ai-0.2.5/tests/test_real_llm_environment.py +58 -0
- sqlseed_ai-0.2.5/tests/test_refiner_json_recovery.py +200 -0
- sqlseed_ai-0.2.5/tests/test_refiner_table_boundaries.py +238 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_runtime.py +12 -9
- sqlseed_ai-0.2.4/tests/healer/conftest.py +0 -57
- sqlseed_ai-0.2.4/tests/healer/test_llm_availability.py +0 -40
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/.gitignore +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/LICENSE +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/_generator_names.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/_hardware.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/_model_selector.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/_tools.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/ai_mediator.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/analyzer/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/auto_heal/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/auto_heal/_check_inference.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/auto_heal/_cross_column_checks.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/auto_heal/orchestrator.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/auto_heal/time_budget.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/cli/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/config.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/contracts/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/contracts/builtin_violations.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/contracts/matrix.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/contracts/registry.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/examples.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/exceptions.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/_client.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/_llm_call.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/candidate_validation.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/context_detector.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/degrader.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/diff_learner.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/failure_classifier.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/level1_subgraph_healer.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/level3_compact_healer.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/models.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/orchestrator.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/oscillation.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/post_repair.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/healer/subgraph.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/repair/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/repair/executor.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/repair/models.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/repair/pipeline.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/repair/strategies.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/composite_fk.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/cross_column.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/dialect_parser.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/main.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/models.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/schema_snapshot.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/shadow_fk_scan.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/src/sqlseed_ai/validator/single_column.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/.pylintrc +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/conftest.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/scenario_helpers.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_context_detector.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_failure_classifier.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_level2_context_builder.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/healer/test_llm_call_boundary.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/property/__init__.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/property/test_matrix_completeness.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/schema_helpers.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_analyzer_streaming.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_commands.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_config.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_errors.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_hardware.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_mediator.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_model_selector.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_plugin_init.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_ai_tool_calling.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_auto_heal_orchestrator.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_auto_heal_sonar_boundaries.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_auto_heal_time_budget.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_cli_auto_heal.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_cli_input_contract.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_contracts_builtin.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_contracts_matrix.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_contracts_registry.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_healer_candidate_contract.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_healer_degrader.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_healer_diff_learner.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_healer_models.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_healer_oscillation.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_healer_post_repair.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_healer_subgraph.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_prompts_p0_p3.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_refiner.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_repair_executor.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_repair_models.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_repair_pipeline.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_repair_strategies.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_schema_snapshot.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_validator_composite_fk.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_validator_cross_column.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_validator_dialect_parser.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_validator_main.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_validator_models.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_validator_shadow_fk_scan.py +0 -0
- {sqlseed_ai-0.2.4 → sqlseed_ai-0.2.5}/tests/test_validator_single_column.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: sqlseed-ai
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.5
|
|
4
4
|
Summary: Optional LLM schema analysis and contract-driven configuration repair for sqlseed
|
|
5
5
|
Project-URL: Documentation, https://sunbos.github.io/sqlseed/gemma4-integration/
|
|
6
6
|
Project-URL: Homepage, https://github.com/sunbos/sqlseed
|
|
@@ -18,9 +18,9 @@ Classifier: Programming Language :: Python :: 3.13
|
|
|
18
18
|
Requires-Python: >=3.10
|
|
19
19
|
Requires-Dist: httpx>=0.24.0
|
|
20
20
|
Requires-Dist: networkx>=3.0
|
|
21
|
-
Requires-Dist: openai>=1.
|
|
21
|
+
Requires-Dist: openai>=1.55.3
|
|
22
22
|
Requires-Dist: sqlseed-cli<0.3,>=0.2.4.dev0
|
|
23
|
-
Requires-Dist: sqlseed<0.3,>=0.2.
|
|
23
|
+
Requires-Dist: sqlseed<0.3,>=0.2.5.dev0
|
|
24
24
|
Provides-Extra: dev
|
|
25
25
|
Requires-Dist: hypothesis>=6.100; extra == 'dev'
|
|
26
26
|
Requires-Dist: pytest-asyncio>=0.21; extra == 'dev'
|
|
@@ -46,13 +46,15 @@ backend test; installing the plugin does not perform one.
|
|
|
46
46
|
|
|
47
47
|
## Installation
|
|
48
48
|
|
|
49
|
-
|
|
49
|
+
These instructions target version 0.2.5. Check [Releases](https://github.com/sunbos/sqlseed/releases)
|
|
50
|
+
for publication status; use the source installation below to test an unpublished candidate.
|
|
51
|
+
Use a Python 3.10+ virtual environment:
|
|
50
52
|
|
|
51
53
|
```bash
|
|
52
|
-
python -m pip install "sqlseed-ai==0.2.
|
|
54
|
+
python -m pip install "sqlseed-ai==0.2.5"
|
|
53
55
|
```
|
|
54
56
|
|
|
55
|
-
Core 0.2.
|
|
57
|
+
Core 0.2.4 and older lack the shared diagnostic interfaces required by this version.
|
|
56
58
|
For development, install Core and the required local plugins together from the
|
|
57
59
|
repository root:
|
|
58
60
|
|
|
@@ -100,6 +102,28 @@ names fail before output is written. `--merge` requires `--output`; it replaces
|
|
|
100
102
|
selected tables, retains existing dependency and unrelated tables plus root settings,
|
|
101
103
|
and appends missing generated tables.
|
|
102
104
|
|
|
105
|
+
Single-table `ai-suggest` checks that the target exists before contacting the
|
|
106
|
+
model. Suggestions and cached results must name that same table, including when
|
|
107
|
+
using `--no-verify` or `--max-retries 0`. SQLite case aliases remain supported;
|
|
108
|
+
export preserves real table and column names, including leading punctuation.
|
|
109
|
+
Rejected suggestions leave existing output files unchanged.
|
|
110
|
+
|
|
111
|
+
The prompt requests one JSON object configuring only the named table, preserving
|
|
112
|
+
its table and column names. Other table names are reference context, not additional
|
|
113
|
+
output targets; the response still passes target validation.
|
|
114
|
+
|
|
115
|
+
Direct analysis (`--no-verify` or `--max-retries 0`), streaming or non-streaming,
|
|
116
|
+
reports empty replies, invalid JSON, output-limit truncation, and empty configuration
|
|
117
|
+
objects separately, without
|
|
118
|
+
echoing the model response in these diagnostics. It announces a retry only when
|
|
119
|
+
another existing shorter-prompt level remains. Once those levels are exhausted,
|
|
120
|
+
it reports the final cause and exits unsuccessfully; it does not increase the
|
|
121
|
+
request budget or change the existing output YAML or database.
|
|
122
|
+
|
|
123
|
+
Direct Python callers can pass `preserve_names=True` to
|
|
124
|
+
`SchemaAnalyzer.call_llm()` or `call_llm_streaming()` before validating against
|
|
125
|
+
their schema. The default retains the existing leading-punctuation cleanup.
|
|
126
|
+
|
|
103
127
|
`auto-heal --config` reads that document. An explicit `--db` or `--url` selects the
|
|
104
128
|
output connection. Invalid YAML, configuration structure, or unknown input tables
|
|
105
129
|
fail without replacing the output file. Candidate repairs are checked for config
|
|
@@ -112,7 +136,7 @@ actual generated values and database constraints.
|
|
|
112
136
|
The AI MCP entry point requires the `mcp` extra:
|
|
113
137
|
|
|
114
138
|
```bash
|
|
115
|
-
python -m pip install "sqlseed-ai[mcp]==0.2.
|
|
139
|
+
python -m pip install "sqlseed-ai[mcp]==0.2.5"
|
|
116
140
|
mcp-server-sqlseed-ai
|
|
117
141
|
```
|
|
118
142
|
|
|
@@ -145,15 +169,39 @@ explicit backend, then known URL patterns, then OpenAI-compatible behavior. It d
|
|
|
145
169
|
not probe every service as a fallback chain. The `tool_calling_protocol` setting and
|
|
146
170
|
its resolver choose the response protocol; a model name alone is insufficient.
|
|
147
171
|
|
|
172
|
+
AI requests to `localhost` and loopback IP addresses connect directly even when
|
|
173
|
+
an HTTP proxy is configured. Remote services keep the environment proxy settings;
|
|
174
|
+
`SSL_CERT_FILE` and `SSL_CERT_DIR` remain effective for HTTPS certificate validation.
|
|
175
|
+
|
|
176
|
+
For Python callers, `SchemaAnalyzer.call_llm(..., strict_json=True)` and
|
|
177
|
+
`call_llm_streaming(..., strict_json=True)` distinguish empty replies, invalid JSON,
|
|
178
|
+
and output-limit truncation using content-free `JSONResponseError.code` values.
|
|
179
|
+
JSON parsing can complete missing final `}` or `]`
|
|
180
|
+
delimiters, including inside code fences, but never fills missing values or strings.
|
|
181
|
+
An output-limit response is rejected even if its prefix parses, including a streaming
|
|
182
|
+
length marker in a separate empty terminal chunk. Both methods default to
|
|
183
|
+
`strict_json=False`. Strict local calls request JSON sampling constraints: LM Studio
|
|
184
|
+
uses its JSON-schema grammar interface, and Ollama uses JSON object mode. A server
|
|
185
|
+
that explicitly rejects the format gets one text-mode compatibility attempt;
|
|
186
|
+
the same strict parser still rejects invalid or truncated output. The CLI's direct
|
|
187
|
+
path and both refiner modes keep their existing prompt/refinement retry budgets.
|
|
188
|
+
Strict tool calling also rejects arrays, scalars and `null` arguments as `invalid_json`;
|
|
189
|
+
compatibility mode may still fall back to response text. Parsed suggestions still
|
|
190
|
+
require scope and rule validation.
|
|
191
|
+
|
|
148
192
|
AI configuration caches include schema hashes. Schema changes invalidate cached
|
|
149
|
-
suggestions; `--no-cache` bypasses them.
|
|
193
|
+
suggestions; `--no-cache` bypasses them. Malformed cache metadata or configuration
|
|
194
|
+
containers are treated as cache misses. Review model output before writing data.
|
|
150
195
|
|
|
151
196
|
## Requirements
|
|
152
197
|
|
|
198
|
+
These metadata requirements apply to version 0.2.5 and its source candidates.
|
|
199
|
+
Use local Core and plugins together when developing from source.
|
|
200
|
+
|
|
153
201
|
- Python `>=3.10`
|
|
154
|
-
- `sqlseed>=0.2.
|
|
202
|
+
- `sqlseed>=0.2.5.dev0,<0.3`
|
|
155
203
|
- `sqlseed-cli>=0.2.4.dev0,<0.3`
|
|
156
|
-
- `openai>=1.0
|
|
204
|
+
- `openai>=1.55.3` (SDK transport defaults with HTTPX 0.28 compatibility)
|
|
157
205
|
- `httpx>=0.24.0`
|
|
158
206
|
- `networkx>=3.0`
|
|
159
207
|
- Optional `mcp` extra: `mcp>=1.0,<2`
|
|
@@ -14,13 +14,15 @@ backend test; installing the plugin does not perform one.
|
|
|
14
14
|
|
|
15
15
|
## Installation
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
These instructions target version 0.2.5. Check [Releases](https://github.com/sunbos/sqlseed/releases)
|
|
18
|
+
for publication status; use the source installation below to test an unpublished candidate.
|
|
19
|
+
Use a Python 3.10+ virtual environment:
|
|
18
20
|
|
|
19
21
|
```bash
|
|
20
|
-
python -m pip install "sqlseed-ai==0.2.
|
|
22
|
+
python -m pip install "sqlseed-ai==0.2.5"
|
|
21
23
|
```
|
|
22
24
|
|
|
23
|
-
Core 0.2.
|
|
25
|
+
Core 0.2.4 and older lack the shared diagnostic interfaces required by this version.
|
|
24
26
|
For development, install Core and the required local plugins together from the
|
|
25
27
|
repository root:
|
|
26
28
|
|
|
@@ -68,6 +70,28 @@ names fail before output is written. `--merge` requires `--output`; it replaces
|
|
|
68
70
|
selected tables, retains existing dependency and unrelated tables plus root settings,
|
|
69
71
|
and appends missing generated tables.
|
|
70
72
|
|
|
73
|
+
Single-table `ai-suggest` checks that the target exists before contacting the
|
|
74
|
+
model. Suggestions and cached results must name that same table, including when
|
|
75
|
+
using `--no-verify` or `--max-retries 0`. SQLite case aliases remain supported;
|
|
76
|
+
export preserves real table and column names, including leading punctuation.
|
|
77
|
+
Rejected suggestions leave existing output files unchanged.
|
|
78
|
+
|
|
79
|
+
The prompt requests one JSON object configuring only the named table, preserving
|
|
80
|
+
its table and column names. Other table names are reference context, not additional
|
|
81
|
+
output targets; the response still passes target validation.
|
|
82
|
+
|
|
83
|
+
Direct analysis (`--no-verify` or `--max-retries 0`), streaming or non-streaming,
|
|
84
|
+
reports empty replies, invalid JSON, output-limit truncation, and empty configuration
|
|
85
|
+
objects separately, without
|
|
86
|
+
echoing the model response in these diagnostics. It announces a retry only when
|
|
87
|
+
another existing shorter-prompt level remains. Once those levels are exhausted,
|
|
88
|
+
it reports the final cause and exits unsuccessfully; it does not increase the
|
|
89
|
+
request budget or change the existing output YAML or database.
|
|
90
|
+
|
|
91
|
+
Direct Python callers can pass `preserve_names=True` to
|
|
92
|
+
`SchemaAnalyzer.call_llm()` or `call_llm_streaming()` before validating against
|
|
93
|
+
their schema. The default retains the existing leading-punctuation cleanup.
|
|
94
|
+
|
|
71
95
|
`auto-heal --config` reads that document. An explicit `--db` or `--url` selects the
|
|
72
96
|
output connection. Invalid YAML, configuration structure, or unknown input tables
|
|
73
97
|
fail without replacing the output file. Candidate repairs are checked for config
|
|
@@ -80,7 +104,7 @@ actual generated values and database constraints.
|
|
|
80
104
|
The AI MCP entry point requires the `mcp` extra:
|
|
81
105
|
|
|
82
106
|
```bash
|
|
83
|
-
python -m pip install "sqlseed-ai[mcp]==0.2.
|
|
107
|
+
python -m pip install "sqlseed-ai[mcp]==0.2.5"
|
|
84
108
|
mcp-server-sqlseed-ai
|
|
85
109
|
```
|
|
86
110
|
|
|
@@ -113,15 +137,39 @@ explicit backend, then known URL patterns, then OpenAI-compatible behavior. It d
|
|
|
113
137
|
not probe every service as a fallback chain. The `tool_calling_protocol` setting and
|
|
114
138
|
its resolver choose the response protocol; a model name alone is insufficient.
|
|
115
139
|
|
|
140
|
+
AI requests to `localhost` and loopback IP addresses connect directly even when
|
|
141
|
+
an HTTP proxy is configured. Remote services keep the environment proxy settings;
|
|
142
|
+
`SSL_CERT_FILE` and `SSL_CERT_DIR` remain effective for HTTPS certificate validation.
|
|
143
|
+
|
|
144
|
+
For Python callers, `SchemaAnalyzer.call_llm(..., strict_json=True)` and
|
|
145
|
+
`call_llm_streaming(..., strict_json=True)` distinguish empty replies, invalid JSON,
|
|
146
|
+
and output-limit truncation using content-free `JSONResponseError.code` values.
|
|
147
|
+
JSON parsing can complete missing final `}` or `]`
|
|
148
|
+
delimiters, including inside code fences, but never fills missing values or strings.
|
|
149
|
+
An output-limit response is rejected even if its prefix parses, including a streaming
|
|
150
|
+
length marker in a separate empty terminal chunk. Both methods default to
|
|
151
|
+
`strict_json=False`. Strict local calls request JSON sampling constraints: LM Studio
|
|
152
|
+
uses its JSON-schema grammar interface, and Ollama uses JSON object mode. A server
|
|
153
|
+
that explicitly rejects the format gets one text-mode compatibility attempt;
|
|
154
|
+
the same strict parser still rejects invalid or truncated output. The CLI's direct
|
|
155
|
+
path and both refiner modes keep their existing prompt/refinement retry budgets.
|
|
156
|
+
Strict tool calling also rejects arrays, scalars and `null` arguments as `invalid_json`;
|
|
157
|
+
compatibility mode may still fall back to response text. Parsed suggestions still
|
|
158
|
+
require scope and rule validation.
|
|
159
|
+
|
|
116
160
|
AI configuration caches include schema hashes. Schema changes invalidate cached
|
|
117
|
-
suggestions; `--no-cache` bypasses them.
|
|
161
|
+
suggestions; `--no-cache` bypasses them. Malformed cache metadata or configuration
|
|
162
|
+
containers are treated as cache misses. Review model output before writing data.
|
|
118
163
|
|
|
119
164
|
## Requirements
|
|
120
165
|
|
|
166
|
+
These metadata requirements apply to version 0.2.5 and its source candidates.
|
|
167
|
+
Use local Core and plugins together when developing from source.
|
|
168
|
+
|
|
121
169
|
- Python `>=3.10`
|
|
122
|
-
- `sqlseed>=0.2.
|
|
170
|
+
- `sqlseed>=0.2.5.dev0,<0.3`
|
|
123
171
|
- `sqlseed-cli>=0.2.4.dev0,<0.3`
|
|
124
|
-
- `openai>=1.0
|
|
172
|
+
- `openai>=1.55.3` (SDK transport defaults with HTTPX 0.28 compatibility)
|
|
125
173
|
- `httpx>=0.24.0`
|
|
126
174
|
- `networkx>=3.0`
|
|
127
175
|
- Optional `mcp` extra: `mcp>=1.0,<2`
|
|
@@ -11,13 +11,13 @@
|
|
|
11
11
|
|
|
12
12
|
## 安装
|
|
13
13
|
|
|
14
|
-
|
|
14
|
+
本文安装说明对应 0.2.5 版本;发布状态以 [Releases](https://github.com/sunbos/sqlseed/releases) 为准。测试尚未发布的候选版本时,使用下方的源码安装方式。请先创建并激活 Python 3.10+ 虚拟环境:
|
|
15
15
|
|
|
16
16
|
```bash
|
|
17
|
-
python -m pip install "sqlseed-ai==0.2.
|
|
17
|
+
python -m pip install "sqlseed-ai==0.2.5"
|
|
18
18
|
```
|
|
19
19
|
|
|
20
|
-
Core 0.2.
|
|
20
|
+
Core 0.2.4 及更早版本缺少本版本使用的共享诊断接口。
|
|
21
21
|
开发源码时,从仓库根一次安装本地 Core、CLI 和 AI:
|
|
22
22
|
|
|
23
23
|
```bash
|
|
@@ -91,7 +91,7 @@ native/custom 方法、实际生成值及依赖数据库状态的约束需另行
|
|
|
91
91
|
AI MCP 入口要求本包的 `mcp` extra:
|
|
92
92
|
|
|
93
93
|
```bash
|
|
94
|
-
python -m pip install "sqlseed-ai[mcp]==0.2.
|
|
94
|
+
python -m pip install "sqlseed-ai[mcp]==0.2.5"
|
|
95
95
|
mcp-server-sqlseed-ai
|
|
96
96
|
```
|
|
97
97
|
|
|
@@ -114,6 +114,9 @@ mcp-server-sqlseed-ai
|
|
|
114
114
|
后端解析顺序是显式 `SQLSEED_AI_BACKEND`、已知 URL 模式、最后 `openai_compat`。
|
|
115
115
|
这不是逐个探测所有服务的 fallback 链。
|
|
116
116
|
|
|
117
|
+
AI 请求访问 `localhost` 或回环 IP 地址时直接连接,不经过环境代理;远程服务仍使用
|
|
118
|
+
原有代理配置。HTTPS 证书校验继续遵循 `SSL_CERT_FILE` 和 `SSL_CERT_DIR`。
|
|
119
|
+
|
|
117
120
|
| 变量 | 用途 |
|
|
118
121
|
| --- | --- |
|
|
119
122
|
| `SQLSEED_AI_BACKEND` | `google_ai_studio`、`lm_studio`、`ollama` 或 `openai_compat` |
|
|
@@ -148,6 +151,33 @@ Gemma 26B ID。注册的模型名称不保证服务当前提供该模型,请
|
|
|
148
151
|
修正;默认最多重试 3 次,仍失败时报告 `AISuggestionFailedError`。
|
|
149
152
|
`ai-analyze` 和 `auto-heal` 则使用 v4 `AutoHealOrchestrator` 的契约驱动路径。
|
|
150
153
|
|
|
154
|
+
单表 `ai-suggest` 在请求模型前检查目标表是否存在;建议与缓存必须指向同一张表。
|
|
155
|
+
`--no-verify` 和 `--max-retries 0` 只跳过生成校验,仍保留目标保护。
|
|
156
|
+
SQLite 表名大小写别名继续可用;导出保留真实表名和列名,包括前导 `.` 或 `:`。
|
|
157
|
+
拒绝的建议不会覆盖已有输出文件。
|
|
158
|
+
|
|
159
|
+
Prompt 明确要求只为指定表返回一个 JSON 对象,并保留表名和列名原文。上下文中的
|
|
160
|
+
其他表名只供参考,不是额外输出目标;这一提示不能替代响应后的目标校验。
|
|
161
|
+
|
|
162
|
+
直接分析(`--no-verify` 或 `--max-retries 0`)的流式和非流式路径均分别说明空回答、无效 JSON、输出长度
|
|
163
|
+
截断和空配置对象,这些诊断不回显模型原文。仅在既定流程中还有更短提示层时才提示
|
|
164
|
+
并继续重试;最后一层失败会报告具体原因并以失败退出,不再声称正在重试,也不增加
|
|
165
|
+
请求预算。拒绝响应时,已有输出 YAML 与数据库均保持不变。
|
|
166
|
+
|
|
167
|
+
Python 调用可使用 `SchemaAnalyzer.call_llm(..., strict_json=True)` 或
|
|
168
|
+
`call_llm_streaming(..., strict_json=True)`,通过不含原文的
|
|
169
|
+
`JSONResponseError.code` 区分空回答、无效 JSON 和输出长度截断。解析器可补齐末尾缺失的
|
|
170
|
+
`}` / `]`,包括代码围栏内的 JSON,但不会补值或字符串;达到输出长度上限时,即使前缀
|
|
171
|
+
可解析也会拒绝,包括流式独立空终止帧中的长度截断标记。两种 Python 方法默认均为
|
|
172
|
+
`strict_json=False`。严格本地调用请求 JSON 采样约束:LM Studio 使用 JSON Schema 语法接口,
|
|
173
|
+
Ollama 使用 JSON 对象模式。服务明确拒绝该格式时,只进行一次文本模式兼容请求,仍由同一
|
|
174
|
+
严格解析器拒绝无效或截断输出。CLI 直接分析与 refiner 保留各自现有提示与自纠正重试预算。
|
|
175
|
+
严格工具调用还会拒绝数组、标量和 `null` 参数,
|
|
176
|
+
返回 `invalid_json`;兼容模式仍可回退到响应正文。解析后的建议仍需验证范围和业务规则。
|
|
177
|
+
|
|
178
|
+
直接 Python 调用可给 `SchemaAnalyzer.call_llm()` 或 `call_llm_streaming()` 传入
|
|
179
|
+
`preserve_names=True`,保留标识符后再按实际 schema 校验;默认仍保留既有的前导标点清理。
|
|
180
|
+
|
|
151
181
|
开启 AI 生成路径时,`sqlseed_pre_generate_templates` 可为符合条件的未匹配字符串列
|
|
152
182
|
准备候选值。用户明确配置、UNIQUE、默认值或主键等条件会影响是否使用模板池,
|
|
153
183
|
不保证每个复杂字段都会调用模型。
|
|
@@ -165,6 +195,7 @@ Gemma 26B ID。注册的模型名称不保证服务当前提供该模型,请
|
|
|
165
195
|
### 文件缓存
|
|
166
196
|
|
|
167
197
|
AI 配置缓存包含 schema hash,结构变化会使旧建议失效;`--no-cache` 跳过缓存。
|
|
198
|
+
缓存元数据或配置容器类型无效时,按缓存未命中重新分析。
|
|
168
199
|
默认路径为 macOS 的 `~/Library/Caches/sqlseed/ai_configs/`、Linux 的
|
|
169
200
|
`$XDG_CACHE_HOME/sqlseed/ai_configs/`(未设置时为 `~/.cache/sqlseed/ai_configs/`),
|
|
170
201
|
以及 Windows 的 `%LOCALAPPDATA%/sqlseed/ai_configs/`。
|
|
@@ -186,10 +217,12 @@ column-mapper 注册 hooks,也不要求 Core 导入 AI 实现。
|
|
|
186
217
|
|
|
187
218
|
## 依赖
|
|
188
219
|
|
|
220
|
+
以下为 0.2.5 版本及其源码候选的依赖要求。源码开发时请在同一次解析中安装本地 Core 和插件。
|
|
221
|
+
|
|
189
222
|
- Python `>=3.10`
|
|
190
|
-
- `sqlseed>=0.2.
|
|
223
|
+
- `sqlseed>=0.2.5.dev0,<0.3`
|
|
191
224
|
- `sqlseed-cli>=0.2.4.dev0,<0.3`
|
|
192
|
-
- `openai>=1.0
|
|
225
|
+
- `openai>=1.55.3`(保留 SDK 传输默认值并兼容 HTTPX 0.28)
|
|
193
226
|
- `httpx>=0.24.0`
|
|
194
227
|
- `networkx>=3.0`
|
|
195
228
|
- 可选 `mcp` extra:`mcp>=1.0,<2`
|
|
@@ -23,9 +23,9 @@ classifiers = [
|
|
|
23
23
|
"Programming Language :: Python :: 3.13",
|
|
24
24
|
]
|
|
25
25
|
dependencies = [
|
|
26
|
-
"sqlseed>=0.2.
|
|
26
|
+
"sqlseed>=0.2.5.dev0,<0.3",
|
|
27
27
|
"sqlseed-cli>=0.2.4.dev0,<0.3",
|
|
28
|
-
"openai>=1.
|
|
28
|
+
"openai>=1.55.3",
|
|
29
29
|
"httpx>=0.24.0",
|
|
30
30
|
"networkx>=3.0",
|
|
31
31
|
]
|
|
@@ -7,18 +7,25 @@ unified httpx timeouts suitable for both cloud and local (GPU) inference.
|
|
|
7
7
|
|
|
8
8
|
from __future__ import annotations
|
|
9
9
|
|
|
10
|
+
from ipaddress import ip_address
|
|
11
|
+
from typing import TYPE_CHECKING
|
|
12
|
+
|
|
10
13
|
import httpx
|
|
11
|
-
from openai import APIConnectionError, APIError, APITimeoutError
|
|
14
|
+
from openai import APIConnectionError, APIError, APITimeoutError
|
|
12
15
|
from sqlseed_ai.config import AIConfig
|
|
13
16
|
|
|
14
17
|
from sqlseed._utils.logger import get_logger
|
|
15
18
|
|
|
19
|
+
if TYPE_CHECKING:
|
|
20
|
+
from openai import OpenAI
|
|
21
|
+
|
|
16
22
|
logger = get_logger(__name__)
|
|
17
23
|
|
|
18
24
|
__all__ = [
|
|
19
25
|
"APIConnectionError",
|
|
20
26
|
"APIError",
|
|
21
27
|
"APITimeoutError",
|
|
28
|
+
"build_openai_client",
|
|
22
29
|
"get_openai_client",
|
|
23
30
|
"httpx_timeout",
|
|
24
31
|
]
|
|
@@ -49,7 +56,36 @@ def get_openai_client(config: AIConfig | None = None) -> OpenAI:
|
|
|
49
56
|
# - pool=10s: connection-pool acquisition timeout
|
|
50
57
|
kwargs["timeout"] = httpx_timeout(config.resolve_timeout())
|
|
51
58
|
logger.info("Creating OpenAI client", **{"backend": config.backend.value, "base_url": kwargs["base_url"]})
|
|
52
|
-
return
|
|
59
|
+
return build_openai_client(**kwargs)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def build_openai_client(*, api_key: str, base_url: str, timeout: float | httpx.Timeout | None) -> OpenAI:
|
|
63
|
+
"""Create a client with direct loopback routing and SDK transport defaults.
|
|
64
|
+
|
|
65
|
+
Exact loopback hosts bypass environment proxies. TLS certificate settings,
|
|
66
|
+
remote proxy routes, timeouts, redirects, and pool limits remain unchanged.
|
|
67
|
+
The returned SDK client owns its optional HTTP client.
|
|
68
|
+
"""
|
|
69
|
+
from openai import DefaultHttpxClient, OpenAI, Timeout
|
|
70
|
+
|
|
71
|
+
resolved_timeout = Timeout(**timeout.as_dict()) if isinstance(timeout, httpx.Timeout) else timeout
|
|
72
|
+
if (host := httpx.URL(base_url).host) != "localhost":
|
|
73
|
+
try:
|
|
74
|
+
is_loopback = ip_address(host).is_loopback
|
|
75
|
+
except ValueError:
|
|
76
|
+
is_loopback = False
|
|
77
|
+
else:
|
|
78
|
+
is_loopback = True
|
|
79
|
+
if not is_loopback:
|
|
80
|
+
return OpenAI(api_key=api_key, base_url=base_url, timeout=resolved_timeout)
|
|
81
|
+
|
|
82
|
+
authority = f"[{host}]" if ":" in host else host
|
|
83
|
+
transport = DefaultHttpxClient(mounts={f"all://{authority}": None})
|
|
84
|
+
try:
|
|
85
|
+
return OpenAI(api_key=api_key, base_url=base_url, timeout=resolved_timeout, http_client=transport)
|
|
86
|
+
except BaseException:
|
|
87
|
+
transport.close()
|
|
88
|
+
raise
|
|
53
89
|
|
|
54
90
|
|
|
55
91
|
def httpx_timeout(total: float) -> httpx.Timeout:
|
|
@@ -34,11 +34,31 @@ from typing import Any
|
|
|
34
34
|
_CHANNEL_END_MARKER = "<channel|>"
|
|
35
35
|
|
|
36
36
|
|
|
37
|
-
|
|
38
|
-
"""
|
|
37
|
+
class JSONResponseError(ValueError):
|
|
38
|
+
"""A safe, content-free diagnostic for an unusable model response."""
|
|
39
|
+
|
|
40
|
+
def __init__(self, code: str) -> None:
|
|
41
|
+
self.code = code
|
|
42
|
+
super().__init__(code)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def parse_json_response(content: str, *, strict: bool = False, preserve_names: bool = False) -> dict[str, Any]:
|
|
46
|
+
"""Parse JSON, optionally preserving identifiers for schema-aware validation.
|
|
47
|
+
|
|
48
|
+
Strict mode diagnoses failures without inventing missing values. Name
|
|
49
|
+
preservation leaves leading punctuation untouched; it does not relax JSON
|
|
50
|
+
parsing or the caller's responsibility to validate the returned document.
|
|
51
|
+
"""
|
|
39
52
|
cleaned = _strip_channel_prefix(content.strip())
|
|
40
53
|
|
|
41
|
-
|
|
54
|
+
if strict and not cleaned:
|
|
55
|
+
raise JSONResponseError("empty_response")
|
|
56
|
+
for parser in (_try_direct_parse, _try_markdown_fence_parse, _try_raw_decode):
|
|
57
|
+
if (result := parser(cleaned, preserve_names=preserve_names)) is not None:
|
|
58
|
+
return result
|
|
59
|
+
if strict:
|
|
60
|
+
raise JSONResponseError("invalid_json")
|
|
61
|
+
return {}
|
|
42
62
|
|
|
43
63
|
|
|
44
64
|
def _strip_channel_prefix(content: str) -> str:
|
|
@@ -52,19 +72,20 @@ def _strip_channel_prefix(content: str) -> str:
|
|
|
52
72
|
return content[idx + len(_CHANNEL_END_MARKER) :].strip()
|
|
53
73
|
|
|
54
74
|
|
|
55
|
-
def _try_direct_parse(content: str) -> dict[str, Any] | None:
|
|
75
|
+
def _try_direct_parse(content: str, *, preserve_names: bool = False) -> dict[str, Any] | None:
|
|
56
76
|
"""Strategy 1: Direct parse (ideal case — model outputs raw JSON)."""
|
|
57
77
|
try:
|
|
58
78
|
result = json.loads(content)
|
|
59
79
|
if isinstance(result, dict):
|
|
60
|
-
|
|
80
|
+
if not preserve_names:
|
|
81
|
+
_sanitize_names(result)
|
|
61
82
|
return result
|
|
62
83
|
except json.JSONDecodeError:
|
|
63
84
|
pass
|
|
64
85
|
return None
|
|
65
86
|
|
|
66
87
|
|
|
67
|
-
def _try_markdown_fence_parse(content: str) -> dict[str, Any] | None:
|
|
88
|
+
def _try_markdown_fence_parse(content: str, *, preserve_names: bool = False) -> dict[str, Any] | None:
|
|
68
89
|
"""Strategy 2: Strip markdown code fences (```json\n{...}\n```)."""
|
|
69
90
|
if (open_idx := content.find("```")) < 0:
|
|
70
91
|
return None
|
|
@@ -75,17 +96,10 @@ def _try_markdown_fence_parse(content: str) -> dict[str, Any] | None:
|
|
|
75
96
|
if (close_idx := after_open.find("```", content_start)) < 0:
|
|
76
97
|
return None
|
|
77
98
|
fence_content = after_open[content_start:close_idx].strip()
|
|
78
|
-
|
|
79
|
-
result = json.loads(fence_content)
|
|
80
|
-
if isinstance(result, dict):
|
|
81
|
-
_sanitize_names(result)
|
|
82
|
-
return result
|
|
83
|
-
except json.JSONDecodeError:
|
|
84
|
-
pass
|
|
85
|
-
return None
|
|
99
|
+
return _try_raw_decode(fence_content, preserve_names=preserve_names)
|
|
86
100
|
|
|
87
101
|
|
|
88
|
-
def _try_raw_decode(content: str) -> dict[str, Any] | None:
|
|
102
|
+
def _try_raw_decode(content: str, *, preserve_names: bool = False) -> dict[str, Any] | None:
|
|
89
103
|
"""Strategy 3: Find first '{' and use json.JSONDecoder.raw_decode().
|
|
90
104
|
|
|
91
105
|
Handles explanatory text before/after JSON without code fences.
|
|
@@ -94,7 +108,7 @@ def _try_raw_decode(content: str) -> dict[str, Any] | None:
|
|
|
94
108
|
Also repairs truncated JSON by attempting to add missing closing
|
|
95
109
|
brackets/braces. Small LLMs (e.g., Gemma 4 E2B) sometimes emit JSON
|
|
96
110
|
missing the final ``}`` or ``]`` characters even when stopReason is
|
|
97
|
-
"eosFound".
|
|
111
|
+
"eosFound". Recover only the delimiters determined by the JSON nesting.
|
|
98
112
|
"""
|
|
99
113
|
if (first_brace := content.find("{")) < 0:
|
|
100
114
|
return None
|
|
@@ -103,24 +117,45 @@ def _try_raw_decode(content: str) -> dict[str, Any] | None:
|
|
|
103
117
|
try:
|
|
104
118
|
result, _ = decoder.raw_decode(content, idx=first_brace)
|
|
105
119
|
if isinstance(result, dict):
|
|
106
|
-
|
|
120
|
+
if not preserve_names:
|
|
121
|
+
_sanitize_names(result)
|
|
107
122
|
return result
|
|
108
123
|
except json.JSONDecodeError:
|
|
109
124
|
pass
|
|
110
|
-
#
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
try:
|
|
115
|
-
result, _ = decoder.raw_decode(content + suffix, idx=first_brace)
|
|
116
|
-
if isinstance(result, dict):
|
|
117
|
-
_sanitize_names(result)
|
|
118
|
-
return result
|
|
119
|
-
except json.JSONDecodeError:
|
|
120
|
-
continue
|
|
125
|
+
# Never invent missing strings, values or separators.
|
|
126
|
+
candidate = content[first_brace:].strip()
|
|
127
|
+
if closers := _missing_closers(candidate):
|
|
128
|
+
return _try_direct_parse(candidate + closers, preserve_names=preserve_names)
|
|
121
129
|
return None
|
|
122
130
|
|
|
123
131
|
|
|
132
|
+
def _advance_quoted_string(char: str, escaped: bool) -> tuple[bool, bool]:
|
|
133
|
+
"""Return the quoted/escaped state after one character inside a JSON string."""
|
|
134
|
+
if escaped:
|
|
135
|
+
return True, False
|
|
136
|
+
if char == "\\":
|
|
137
|
+
return True, True
|
|
138
|
+
return char != '"', False
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _missing_closers(content: str) -> str:
|
|
142
|
+
"""Complete delimiters only, never strings, keys, commas or business values."""
|
|
143
|
+
stack: list[str] = []
|
|
144
|
+
quoted = escaped = False
|
|
145
|
+
for char in content:
|
|
146
|
+
if quoted:
|
|
147
|
+
quoted, escaped = _advance_quoted_string(char, escaped)
|
|
148
|
+
elif char == '"':
|
|
149
|
+
quoted = True
|
|
150
|
+
elif char in "{[":
|
|
151
|
+
stack.append("}" if char == "{" else "]")
|
|
152
|
+
elif char in "}]" and (not stack or stack.pop() != char):
|
|
153
|
+
return ""
|
|
154
|
+
if quoted or not stack or len(stack) > 8:
|
|
155
|
+
return ""
|
|
156
|
+
return "".join(reversed(stack))
|
|
157
|
+
|
|
158
|
+
|
|
124
159
|
def _sanitize_names(data: dict[str, Any]) -> None:
|
|
125
160
|
"""Strip leading colons/dots from table and column name fields.
|
|
126
161
|
|
|
@@ -132,7 +167,12 @@ def _sanitize_names(data: dict[str, Any]) -> None:
|
|
|
132
167
|
if isinstance(name, str):
|
|
133
168
|
data["name"] = re.sub(r"^[:.]+", "", name)
|
|
134
169
|
|
|
135
|
-
|
|
170
|
+
columns = data.get("columns")
|
|
171
|
+
if not isinstance(columns, list):
|
|
172
|
+
# Preserve malformed containers for the configuration validator. Name
|
|
173
|
+
# normalization must not turn valid JSON into an incidental TypeError.
|
|
174
|
+
return
|
|
175
|
+
for col in columns:
|
|
136
176
|
if isinstance(col, dict):
|
|
137
177
|
col_name = col.get("name")
|
|
138
178
|
if isinstance(col_name, str):
|
|
@@ -8,7 +8,16 @@ across the three verbosity tiers (full, compact, ultra-compact).
|
|
|
8
8
|
|
|
9
9
|
from __future__ import annotations
|
|
10
10
|
|
|
11
|
-
|
|
11
|
+
_TEMPORAL_PARAM_RULE = (
|
|
12
|
+
"Date/time bounds: start_date/end_date must be ISO YYYY-MM-DD strings; "
|
|
13
|
+
"start_time/end_time must be HH:MM or HH:MM:SS strings. "
|
|
14
|
+
"Use explicit dates/times or omit optional bounds to use generator defaults. "
|
|
15
|
+
"Never use 'now', 'today', or relative dates in these parameters.\n\n"
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
SYSTEM_PROMPT = (
|
|
19
|
+
_TEMPORAL_PARAM_RULE
|
|
20
|
+
+ """You are an expert database test data engineer.
|
|
12
21
|
You analyze database table schemas and recommend data generation configurations for the sqlseed toolkit.
|
|
13
22
|
|
|
14
23
|
The schema may come from SQLite, PostgreSQL, or other databases.
|
|
@@ -217,8 +226,11 @@ The JSON object must have this exact structure:
|
|
|
217
226
|
IMPORTANT: Do NOT include columns that are auto-incrementing primary keys or have DEFAULT values.
|
|
218
227
|
IMPORTANT: Output ONLY the JSON object, nothing else.
|
|
219
228
|
IMPORTANT: Do NOT wrap output in markdown code blocks (no ```json```). Output raw JSON only."""
|
|
229
|
+
)
|
|
220
230
|
|
|
221
|
-
_COMPACT_SYSTEM_PROMPT =
|
|
231
|
+
_COMPACT_SYSTEM_PROMPT = (
|
|
232
|
+
_TEMPORAL_PARAM_RULE
|
|
233
|
+
+ """Output a JSON config for test data generation.
|
|
222
234
|
|
|
223
235
|
Generators and key params:
|
|
224
236
|
- string (min_length, max_length, charset)
|
|
@@ -267,8 +279,11 @@ Format: {"name":"t","count":1000,"columns":[
|
|
|
267
279
|
]}
|
|
268
280
|
|
|
269
281
|
Output ONLY raw JSON. No markdown, no ```json```, no explanation, no whitespace."""
|
|
282
|
+
)
|
|
270
283
|
|
|
271
|
-
_ULTRA_COMPACT_SYSTEM_PROMPT =
|
|
284
|
+
_ULTRA_COMPACT_SYSTEM_PROMPT = (
|
|
285
|
+
_TEMPORAL_PARAM_RULE
|
|
286
|
+
+ """Output JSON test data config.
|
|
272
287
|
Skip PRIMARY KEY AUTOINCREMENT, DEFAULT, GENERATED, and foreign-key cols (auto-handled by core).
|
|
273
288
|
UNIQUE col → add "constraints":{"unique":true} (do NOT skip).
|
|
274
289
|
Enum CHECK (col IN ('a','b')) → weighted_choice with weighted_choices:{a:80,b:15,c:5} (realistic, NOT uniform).
|
|
@@ -296,6 +311,7 @@ lookup(table,column,key) — cross-table value fetch for derive_from expressions
|
|
|
296
311
|
Expr funcs ONLY: random_float/random_int/random_choice/timedelta/lookup/int/float/str/abs/min/max/round/len/
|
|
297
312
|
upper/lower/substr/concat/replace/zfill/lpad/rpad. NO random_uniform (use random_float).
|
|
298
313
|
Output ONLY raw JSON. No markdown, no explanation."""
|
|
314
|
+
)
|
|
299
315
|
|
|
300
316
|
TEMPLATE_SYSTEM_PROMPT = (
|
|
301
317
|
"You are a data generation assistant. Generate realistic sample values "
|