google-cloud-db-context-engineering 0.5.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- google/cloud/db_context_enrichment/__init__.py +1 -0
- google/cloud/db_context_enrichment/bootstrap/__init__.py +1 -0
- google/cloud/db_context_enrichment/bootstrap/bootstrap_generator.py +74 -0
- google/cloud/db_context_enrichment/common/__init__.py +0 -0
- google/cloud/db_context_enrichment/common/config.py +10 -0
- google/cloud/db_context_enrichment/common/context_mutator.py +132 -0
- google/cloud/db_context_enrichment/common/parameterizer.py +192 -0
- google/cloud/db_context_enrichment/dataset/__init__.py +0 -0
- google/cloud/db_context_enrichment/dataset/dataset_generator.py +44 -0
- google/cloud/db_context_enrichment/evaluate/__init__.py +3 -0
- google/cloud/db_context_enrichment/evaluate/db_generators/__init__.py +0 -0
- google/cloud/db_context_enrichment/evaluate/db_generators/alloydb.py +72 -0
- google/cloud/db_context_enrichment/evaluate/db_generators/base.py +76 -0
- google/cloud/db_context_enrichment/evaluate/db_generators/mysql.py +69 -0
- google/cloud/db_context_enrichment/evaluate/db_generators/postgres.py +69 -0
- google/cloud/db_context_enrichment/evaluate/db_generators/spanner.py +62 -0
- google/cloud/db_context_enrichment/evaluate/evaluate_generator.py +239 -0
- google/cloud/db_context_enrichment/evaluate/result_reader.py +181 -0
- google/cloud/db_context_enrichment/facet/__init__.py +0 -0
- google/cloud/db_context_enrichment/facet/facet_generator.py +68 -0
- google/cloud/db_context_enrichment/main.py +450 -0
- google/cloud/db_context_enrichment/model/__init__.py +0 -0
- google/cloud/db_context_enrichment/model/context.py +86 -0
- google/cloud/db_context_enrichment/prompts/__init__.py +9 -0
- google/cloud/db_context_enrichment/prompts/targeted_facets.py +56 -0
- google/cloud/db_context_enrichment/prompts/targeted_templates.py +57 -0
- google/cloud/db_context_enrichment/prompts/targeted_value_search.py +93 -0
- google/cloud/db_context_enrichment/template/__init__.py +0 -0
- google/cloud/db_context_enrichment/template/template_generator.py +67 -0
- google/cloud/db_context_enrichment/value_search/__init__.py +0 -0
- google/cloud/db_context_enrichment/value_search/generator.py +91 -0
- google/cloud/db_context_enrichment/value_search/match_templates.py +267 -0
- google_cloud_db_context_engineering-0.5.1.dist-info/METADATA +160 -0
- google_cloud_db_context_engineering-0.5.1.dist-info/RECORD +38 -0
- google_cloud_db_context_engineering-0.5.1.dist-info/WHEEL +5 -0
- google_cloud_db_context_engineering-0.5.1.dist-info/entry_points.txt +2 -0
- google_cloud_db_context_engineering-0.5.1.dist-info/licenses/LICENSE +202 -0
- google_cloud_db_context_engineering-0.5.1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,450 @@
|
|
|
1
|
+
import datetime
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
|
|
5
|
+
from fastmcp import FastMCP
|
|
6
|
+
|
|
7
|
+
import google.cloud.db_context_enrichment.prompts as prompts
|
|
8
|
+
from google.cloud.db_context_enrichment.bootstrap import bootstrap_generator
|
|
9
|
+
from google.cloud.db_context_enrichment.common import context_mutator
|
|
10
|
+
from google.cloud.db_context_enrichment.dataset import dataset_generator
|
|
11
|
+
from google.cloud.db_context_enrichment.evaluate import (
|
|
12
|
+
evaluate_generator,
|
|
13
|
+
result_reader,
|
|
14
|
+
)
|
|
15
|
+
from google.cloud.db_context_enrichment.facet import facet_generator
|
|
16
|
+
from google.cloud.db_context_enrichment.model import context
|
|
17
|
+
from google.cloud.db_context_enrichment.template import template_generator
|
|
18
|
+
from google.cloud.db_context_enrichment.value_search import generator as vi_generator
|
|
19
|
+
from google.cloud.db_context_enrichment.value_search import match_templates
|
|
20
|
+
|
|
21
|
+
mcp = FastMCP("Context Engineering Agent MCP")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@mcp.tool
|
|
25
|
+
async def generate_templates(
|
|
26
|
+
template_inputs_json: str, sql_dialect: str = "postgresql"
|
|
27
|
+
) -> str:
|
|
28
|
+
"""
|
|
29
|
+
Generates final templates from a list of user-approved template question, template SQL statement, and optional template intent.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
template_inputs_json: A JSON string representing a list of dictionaries (template inputs),
|
|
33
|
+
where each dictionary has "question", "sql", and optional "intent" keys.
|
|
34
|
+
Example (with intent): '[{"question": "How many users?", "sql": "SELECT count(*) FROM users", "intent": "Count total users"}]'
|
|
35
|
+
Example (default intent): '[{"question": "List all items", "sql": "SELECT * FROM items"}]'
|
|
36
|
+
sql_dialect: The SQL dialect to use for parameterization. Accepted
|
|
37
|
+
values are 'postgresql' (default), 'mysql', or 'googlesql'.
|
|
38
|
+
|
|
39
|
+
Returns:
|
|
40
|
+
A JSON string representing a ContextSet object.
|
|
41
|
+
"""
|
|
42
|
+
return await template_generator.generate_templates(
|
|
43
|
+
template_inputs_json, sql_dialect
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@mcp.tool
|
|
48
|
+
async def generate_facets(
|
|
49
|
+
facet_inputs_json: str, sql_dialect: str = "postgresql"
|
|
50
|
+
) -> str:
|
|
51
|
+
"""
|
|
52
|
+
Generates final facets from a list of user-approved facet intent and facet SQL snippet.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
facet_inputs_json: A JSON string representing a list of dictionaries (facet inputs),
|
|
56
|
+
where each dictionary has "intent" and "sql_snippet".
|
|
57
|
+
Example: '[{"intent": "high price", "sql_snippet": "price > 1000"}]'
|
|
58
|
+
sql_dialect: The SQL dialect to use for parameterization. Accepted
|
|
59
|
+
values are 'postgresql' (default), 'mysql', or 'googlesql'.
|
|
60
|
+
|
|
61
|
+
Returns:
|
|
62
|
+
A JSON string representing a ContextSet object.
|
|
63
|
+
"""
|
|
64
|
+
return await facet_generator.generate_facets(facet_inputs_json, sql_dialect)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@mcp.tool
|
|
68
|
+
async def generate_bootstrap_context(
|
|
69
|
+
output_file_path: str,
|
|
70
|
+
template_inputs_json: str | None = None,
|
|
71
|
+
facet_inputs_json: str | None = None,
|
|
72
|
+
sql_dialect: str = "postgresql",
|
|
73
|
+
) -> str:
|
|
74
|
+
"""
|
|
75
|
+
Generates a single unified ContextSet from key information and saves it to a file.
|
|
76
|
+
|
|
77
|
+
Args:
|
|
78
|
+
output_file_path: The absolute path where the JSON ContextSet file should be saved.
|
|
79
|
+
template_inputs_json: A JSON string representing a list of extracted seed information used to generate full templates.
|
|
80
|
+
Each item in the list should be a dictionary with keys:
|
|
81
|
+
- "question": The natural language question.
|
|
82
|
+
- "sql": The corresponding SQL query to answer the question.
|
|
83
|
+
- "intent": (Optional) A brief description of the intent.
|
|
84
|
+
|
|
85
|
+
Example:
|
|
86
|
+
'[{"question": "How many users?", "sql": "SELECT COUNT(*) FROM users", "intent": "Count total users"}]'
|
|
87
|
+
|
|
88
|
+
facet_inputs_json: A JSON string representing a list of extracted seed information used to generate full facets.
|
|
89
|
+
Each item in the list should be a dictionary with keys:
|
|
90
|
+
- "intent": A brief description of the facet intent.
|
|
91
|
+
- "sql_snippet": A specific SQL fragment (such as a filter condition) representing the intent.
|
|
92
|
+
|
|
93
|
+
Example:
|
|
94
|
+
'[{"intent": "high price", "sql_snippet": "price > 1000"}]'
|
|
95
|
+
sql_dialect: SQL engine dialect.
|
|
96
|
+
|
|
97
|
+
Returns:
|
|
98
|
+
The absolute file path pointing to the generated and saved ContextSet JSON.
|
|
99
|
+
"""
|
|
100
|
+
return await bootstrap_generator.generate_context(
|
|
101
|
+
output_file_path, sql_dialect, template_inputs_json, facet_inputs_json
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
@mcp.tool
|
|
106
|
+
async def generate_dataset(
|
|
107
|
+
dataset_entries_json: str,
|
|
108
|
+
output_file_path: str,
|
|
109
|
+
) -> str:
|
|
110
|
+
"""
|
|
111
|
+
Validates a list of evaluation dataset entries and saves them to a JSON file.
|
|
112
|
+
|
|
113
|
+
Args:
|
|
114
|
+
dataset_entries_json: A JSON string representing a list of dataset items.
|
|
115
|
+
Each item should have "id", "database", "nlq", and "golden_sql" keys.
|
|
116
|
+
Example: '[{"id": "eval_001", "database": "my_db", "nlq": "Count users", "golden_sql": "SELECT COUNT(*) FROM users"}]'
|
|
117
|
+
output_file_path: The absolute path where the dataset JSON file should be saved.
|
|
118
|
+
|
|
119
|
+
Returns:
|
|
120
|
+
The absolute file path where the dataset was saved.
|
|
121
|
+
"""
|
|
122
|
+
return await dataset_generator.generate_dataset(
|
|
123
|
+
dataset_entries_json, output_file_path
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@mcp.tool
|
|
128
|
+
def generate_evalbench_configs(
|
|
129
|
+
experiment_name: str,
|
|
130
|
+
dataset_path: str,
|
|
131
|
+
context_set_id: str,
|
|
132
|
+
toolbox_config_path: str,
|
|
133
|
+
toolbox_source_name: str,
|
|
134
|
+
) -> str:
|
|
135
|
+
"""
|
|
136
|
+
Generates Evalbench YAML configurations and converts the user-facing golden dataset to be compatible for evaluation, saving all files directly to disk.
|
|
137
|
+
|
|
138
|
+
This tool writes the following files inside `experiments/<experiment_name>/eval_configs/`:
|
|
139
|
+
- `db_config.yaml`
|
|
140
|
+
- `model_config.yaml`
|
|
141
|
+
- `run_config.yaml`
|
|
142
|
+
- `llmrater_config.yaml`
|
|
143
|
+
- `golden_queries.json` (converted to EvalBench internal format)
|
|
144
|
+
|
|
145
|
+
Args:
|
|
146
|
+
experiment_name: The name of the target experiment folder.
|
|
147
|
+
dataset_path: The absolute path to the golden dataset file in the simplified user-facing format (JSON list of objects with keys: "id", "database", "nlq", "golden_sql").
|
|
148
|
+
context_set_id: The specific context_set_id inside the experiment.
|
|
149
|
+
toolbox_config_path: The absolute path to the tools.yaml configuration file.
|
|
150
|
+
toolbox_source_name: The name of the database source to use inside tools.yaml. The underlying source block must use a supported 'type' (cloud-sql-postgres, cloud-sql-mysql, spanner, alloydb-postgres).
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
A message indicating that the configuration files were successfully created.
|
|
154
|
+
"""
|
|
155
|
+
evaluate_generator.generate_evalbench_configs(
|
|
156
|
+
experiment_name,
|
|
157
|
+
dataset_path,
|
|
158
|
+
context_set_id,
|
|
159
|
+
toolbox_config_path,
|
|
160
|
+
toolbox_source_name,
|
|
161
|
+
)
|
|
162
|
+
return f"Successfully generated all configs for evaluation in experiments/{experiment_name}/eval_configs/"
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
@mcp.tool
|
|
166
|
+
async def generate_value_searches(
|
|
167
|
+
value_search_inputs_json: str,
|
|
168
|
+
dialect: str,
|
|
169
|
+
db_version: str | None = None,
|
|
170
|
+
) -> str:
|
|
171
|
+
"""
|
|
172
|
+
Generates final value searches from a list of user-approved value search definitions.
|
|
173
|
+
|
|
174
|
+
Args:
|
|
175
|
+
value_search_inputs_json: A JSON string representing a list of value search definitions.
|
|
176
|
+
Each item in the list should be a dictionary with keys:
|
|
177
|
+
- "table_name": The name of the table.
|
|
178
|
+
- "column_name": The name of the column.
|
|
179
|
+
- "concept_type": The semantic type (e.g., 'City').
|
|
180
|
+
- "match_function": The match function to use (e.g., 'EXACT_MATCH_STRINGS').
|
|
181
|
+
- "description": (Optional) A description of the value search.
|
|
182
|
+
|
|
183
|
+
Example:
|
|
184
|
+
'[
|
|
185
|
+
{"table_name": "users", "column_name": "city", "concept_type": "City", "match_function": "EXACT_MATCH_STRINGS"},
|
|
186
|
+
{"table_name": "products", "column_name": "name", "concept_type": "Product", "match_function": "FUZZY_MATCH_STRINGS"}
|
|
187
|
+
]'
|
|
188
|
+
|
|
189
|
+
dialect: The database dialect (postgresql, mysql, etc.).
|
|
190
|
+
db_version: The database version (optional).
|
|
191
|
+
|
|
192
|
+
Returns:
|
|
193
|
+
A JSON string representing a ContextSet object containing all the new value searches.
|
|
194
|
+
"""
|
|
195
|
+
if db_version and not db_version.strip():
|
|
196
|
+
db_version = None
|
|
197
|
+
|
|
198
|
+
return vi_generator.generate_value_searches(
|
|
199
|
+
value_search_inputs_json, dialect, db_version
|
|
200
|
+
)
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
@mcp.tool
|
|
204
|
+
def list_match_functions(dialect: str, db_version: str | None = None) -> str:
|
|
205
|
+
"""
|
|
206
|
+
Lists the valid match template functions with their descriptions and examples for a specific database dialect.
|
|
207
|
+
Use this to show the user what 'match_function' options are available, along with their details.
|
|
208
|
+
|
|
209
|
+
If the dialect or version is not supported, this will return an error message
|
|
210
|
+
listing the valid options.
|
|
211
|
+
|
|
212
|
+
Args:
|
|
213
|
+
dialect: The database dialect (e.g., 'postgresql').
|
|
214
|
+
db_version: The specific database version (optional).
|
|
215
|
+
|
|
216
|
+
Returns:
|
|
217
|
+
A JSON string containing a dictionary of available function names mapped to their descriptions and examples,
|
|
218
|
+
or an error message if validation fails.
|
|
219
|
+
"""
|
|
220
|
+
try:
|
|
221
|
+
functions = match_templates.get_available_functions(dialect, db_version)
|
|
222
|
+
return json.dumps(functions)
|
|
223
|
+
except ValueError as e:
|
|
224
|
+
return f"Error: {str(e)}"
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
@mcp.tool
|
|
228
|
+
def save_context_set(
|
|
229
|
+
context_set_json: str,
|
|
230
|
+
db_instance: str,
|
|
231
|
+
db_name: str,
|
|
232
|
+
output_dir: str,
|
|
233
|
+
) -> str:
|
|
234
|
+
"""
|
|
235
|
+
Saves a ContextSet to a new JSON file with a generated timestamp.
|
|
236
|
+
|
|
237
|
+
Args:
|
|
238
|
+
context_set_json: The JSON string of the ContextSet.
|
|
239
|
+
db_instance: The database instance name.
|
|
240
|
+
db_name: The database name.
|
|
241
|
+
output_dir: The directory to save the file in. The root of where the
|
|
242
|
+
Gemini CLI is running.
|
|
243
|
+
|
|
244
|
+
Returns:
|
|
245
|
+
A confirmation message with the path to the newly created file.
|
|
246
|
+
"""
|
|
247
|
+
timestamp = datetime.datetime.now().strftime("%Y%m%d%H%M%S")
|
|
248
|
+
filename = f"{db_instance}_{db_name}_context_set_{timestamp}.json"
|
|
249
|
+
filepath = os.path.join(output_dir, filename)
|
|
250
|
+
|
|
251
|
+
try:
|
|
252
|
+
data = json.loads(context_set_json)
|
|
253
|
+
with open(filepath, "w") as f:
|
|
254
|
+
json.dump(data, f, indent=2)
|
|
255
|
+
return f"Successfully saved context set to {filepath}"
|
|
256
|
+
except (OSError, json.JSONDecodeError) as e:
|
|
257
|
+
return f"Error saving file: {e}"
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
@mcp.tool
|
|
261
|
+
def attach_context_set(
|
|
262
|
+
context_set_json: str,
|
|
263
|
+
file_path: str,
|
|
264
|
+
) -> str:
|
|
265
|
+
"""
|
|
266
|
+
Attaches a ContextSet to an existing JSON file.
|
|
267
|
+
|
|
268
|
+
This tool reads an existing JSON file containing a ContextSet,
|
|
269
|
+
appends new templates/facets/value_searches to it, and writes the updated ContextSet
|
|
270
|
+
back to the file. Exceptions are propagated to the caller.
|
|
271
|
+
|
|
272
|
+
Args:
|
|
273
|
+
context_set_json: The JSON string output from the generation tools.
|
|
274
|
+
file_path: The **absolute path** to the existing template file.
|
|
275
|
+
|
|
276
|
+
Returns:
|
|
277
|
+
A confirmation message with the path to the updated file.
|
|
278
|
+
"""
|
|
279
|
+
|
|
280
|
+
existing_content_dict = {"templates": [], "facets": [], "value_searches": []}
|
|
281
|
+
if os.path.exists(file_path) and os.path.getsize(file_path) > 0:
|
|
282
|
+
with open(file_path) as f:
|
|
283
|
+
existing_content_dict = json.load(f)
|
|
284
|
+
|
|
285
|
+
existing_context = context.ContextSet(**existing_content_dict)
|
|
286
|
+
|
|
287
|
+
new_context = context.ContextSet(**json.loads(context_set_json))
|
|
288
|
+
|
|
289
|
+
if existing_context.templates is None:
|
|
290
|
+
existing_context.templates = []
|
|
291
|
+
if new_context.templates:
|
|
292
|
+
existing_context.templates.extend(new_context.templates)
|
|
293
|
+
|
|
294
|
+
if existing_context.facets is None:
|
|
295
|
+
existing_context.facets = []
|
|
296
|
+
if new_context.facets:
|
|
297
|
+
existing_context.facets.extend(new_context.facets)
|
|
298
|
+
|
|
299
|
+
if existing_context.value_searches is None:
|
|
300
|
+
existing_context.value_searches = []
|
|
301
|
+
if new_context.value_searches:
|
|
302
|
+
existing_context.value_searches.extend(new_context.value_searches)
|
|
303
|
+
|
|
304
|
+
with open(file_path, "w") as f:
|
|
305
|
+
json.dump(existing_context.model_dump(), f, indent=2)
|
|
306
|
+
|
|
307
|
+
return f"Successfully attached context to {file_path}"
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
@mcp.tool
|
|
311
|
+
def generate_upload_url(
|
|
312
|
+
db_engine: str,
|
|
313
|
+
project_id: str,
|
|
314
|
+
location: str | None = None,
|
|
315
|
+
cluster_id: str | None = None,
|
|
316
|
+
instance_id: str | None = None,
|
|
317
|
+
database_id: str | None = None,
|
|
318
|
+
) -> str:
|
|
319
|
+
"""
|
|
320
|
+
Generates a URL for uploading the template file based on the database engine.
|
|
321
|
+
|
|
322
|
+
Args:
|
|
323
|
+
db_engine: The database engine. Accepted values are 'alloydb',
|
|
324
|
+
'cloudsql', or 'spanner'. This can be derived from the 'kind'
|
|
325
|
+
field in the tools.yaml file. For example, 'alloydb-postgres'
|
|
326
|
+
becomes 'alloydb', and 'cloud-sql-postgres' becomes 'cloudsql'.
|
|
327
|
+
project_id: The Google Cloud project ID.
|
|
328
|
+
location: The location of the AlloyDB cluster.
|
|
329
|
+
cluster_id: The ID of the AlloyDB cluster.
|
|
330
|
+
instance_id: The ID of the Cloud SQL or Spanner instance.
|
|
331
|
+
database_id: The ID of the Spanner database.
|
|
332
|
+
|
|
333
|
+
Returns:
|
|
334
|
+
The generated URL as a string, or an error message if the source kind is invalid.
|
|
335
|
+
"""
|
|
336
|
+
if db_engine == "alloydb":
|
|
337
|
+
if location and cluster_id and project_id:
|
|
338
|
+
return f"https://console.cloud.google.com/alloydb/locations/{location}/clusters/{cluster_id}/studio?project={project_id}"
|
|
339
|
+
else:
|
|
340
|
+
return "Error: Missing location, cluster_id, or project_id for alloydb."
|
|
341
|
+
elif db_engine == "cloudsql":
|
|
342
|
+
if instance_id and project_id:
|
|
343
|
+
return f"https://console.cloud.google.com/sql/instances/{instance_id}/studio?project={project_id}"
|
|
344
|
+
else:
|
|
345
|
+
return "Error: Missing instance_id or project_id for cloudsql."
|
|
346
|
+
elif db_engine == "spanner":
|
|
347
|
+
if instance_id and database_id and project_id:
|
|
348
|
+
return f"https://console.cloud.google.com/spanner/instances/{instance_id}/databases/{database_id}/details/query?project={project_id}"
|
|
349
|
+
else:
|
|
350
|
+
return "Error: Missing instance_id, database_id, or project_id for spanner."
|
|
351
|
+
else:
|
|
352
|
+
return "Error: Invalid db_engine. Must be one of 'alloydb', 'cloudsql', or 'spanner'."
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
@mcp.prompt
|
|
356
|
+
def generate_targeted_templates() -> str:
|
|
357
|
+
"""Initiates a guided workflow to generate specific templates based on the user's input."""
|
|
358
|
+
return prompts.GENERATE_TARGETED_TEMPLATES_PROMPT
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
@mcp.prompt
|
|
362
|
+
def generate_targeted_facets() -> str:
|
|
363
|
+
"""Initiates a guided workflow to generate specific facets based on the user's input."""
|
|
364
|
+
return prompts.GENERATE_TARGETED_FACETS_PROMPT
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
@mcp.prompt
|
|
368
|
+
def generate_targeted_value_searches() -> str:
|
|
369
|
+
"""Initiates a guided workflow to generate specific Value Search configurations."""
|
|
370
|
+
return prompts.GENERATE_TARGETED_VALUE_SEARCH_PROMPT
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
@mcp.tool
|
|
374
|
+
def mutate_context_set(
|
|
375
|
+
file_path: str,
|
|
376
|
+
mutations_json: str,
|
|
377
|
+
) -> str:
|
|
378
|
+
"""
|
|
379
|
+
Apply structural mutations to an existing ContextSet JSON file.
|
|
380
|
+
|
|
381
|
+
Parameters:
|
|
382
|
+
- file_path (str): The absolute path to the ContextSet file.
|
|
383
|
+
- mutations_json (str): A JSON string representing a list of mutations.
|
|
384
|
+
Each mutation must contain:
|
|
385
|
+
- 'operation': "add", "delete", or "update"
|
|
386
|
+
- 'type': "template", "facet", or "value_search"
|
|
387
|
+
- 'identifier' (dict): Required for "delete" and "update" to find the target item (e.g., {"nl_query": "What are all users?"}).
|
|
388
|
+
- 'value' (dict): Required for "add" and "update".
|
|
389
|
+
- For "add": Must be the FULL item body. Rely on specialized generation tools (like `generate_templates`) to produce this content deterministically.
|
|
390
|
+
- For "update": Can be a PARTIAL body containing only the fields to change (it will be merged with the existing item).
|
|
391
|
+
|
|
392
|
+
Example 'mutations_json':
|
|
393
|
+
'[
|
|
394
|
+
{
|
|
395
|
+
"operation": "add",
|
|
396
|
+
"type": "template",
|
|
397
|
+
"value": {
|
|
398
|
+
"nl_query": "How many users registered in 2023?",
|
|
399
|
+
"sql": "SELECT count(*) FROM users WHERE year = 2023",
|
|
400
|
+
"intent": "Count users registered in 2023",
|
|
401
|
+
"manifest": "Count users registered in a given year",
|
|
402
|
+
"parameterized": {
|
|
403
|
+
"parameterized_sql": "SELECT count(*) FROM users WHERE year = $1",
|
|
404
|
+
"parameterized_intent": "Count users registered in $1"
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
},
|
|
408
|
+
{
|
|
409
|
+
"operation": "delete",
|
|
410
|
+
"type": "facet",
|
|
411
|
+
"identifier": {"intent": "high price"}
|
|
412
|
+
},
|
|
413
|
+
{
|
|
414
|
+
"operation": "update",
|
|
415
|
+
"type": "facet",
|
|
416
|
+
"identifier": {"intent": "high price"},
|
|
417
|
+
"value": {"sql_snippet": "price > 2000", "intent": "very high price"}
|
|
418
|
+
}
|
|
419
|
+
]'
|
|
420
|
+
"""
|
|
421
|
+
try:
|
|
422
|
+
mutations_data = json.loads(mutations_json)
|
|
423
|
+
if not isinstance(mutations_data, list):
|
|
424
|
+
return "Error applying mutations: mutations_json must be a JSON list."
|
|
425
|
+
mutations = [context_mutator.Mutation(**mut) for mut in mutations_data]
|
|
426
|
+
context_mutator.mutate_context_set(file_path, mutations)
|
|
427
|
+
return f"Successfully applied {len(mutations)} mutations to {file_path}"
|
|
428
|
+
except Exception as e:
|
|
429
|
+
return f"Error applying mutations: {str(e)}"
|
|
430
|
+
|
|
431
|
+
|
|
432
|
+
@mcp.tool
|
|
433
|
+
async def read_evaluation_result(
|
|
434
|
+
run_folder_path: str, offset: int = 0, batch_size: int = 10
|
|
435
|
+
) -> str:
|
|
436
|
+
"""Reads evaluation results from a folder and produces a markdown summary.
|
|
437
|
+
|
|
438
|
+
Args:
|
|
439
|
+
run_folder_path: The absolute path to the evaluation run result folder, which ends with the eval run job id.
|
|
440
|
+
offset: Offset to start reading failure cases from (default: 0).
|
|
441
|
+
batch_size: Number of failure cases to show in the report (default: 10).
|
|
442
|
+
|
|
443
|
+
Returns:
|
|
444
|
+
A string in markdown format containing the summary and failure cases.
|
|
445
|
+
"""
|
|
446
|
+
return result_reader.read_eval_results(run_folder_path, offset, batch_size)
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
if __name__ == "__main__":
|
|
450
|
+
mcp.run() # Uses STDIO transport by default
|
|
File without changes
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
from pydantic import AliasChoices, BaseModel, Field
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class ParameterizedTemplate(BaseModel):
|
|
5
|
+
"""Defines the parameterized version of a SQL query and intent."""
|
|
6
|
+
|
|
7
|
+
parameterized_sql: str = Field(
|
|
8
|
+
..., description="The SQL query with placeholders (eg., )."
|
|
9
|
+
)
|
|
10
|
+
parameterized_intent: str = Field(
|
|
11
|
+
..., description="The natural language intent with placeholders."
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class Template(BaseModel):
|
|
16
|
+
"""Represents a single, complete template."""
|
|
17
|
+
|
|
18
|
+
nl_query: str = Field(
|
|
19
|
+
..., description="A natural language question about the data."
|
|
20
|
+
)
|
|
21
|
+
sql: str = Field(..., description="The corresponding, complete SQL query.")
|
|
22
|
+
intent: str = Field(..., description="The user's specific intent.")
|
|
23
|
+
manifest: str = Field(
|
|
24
|
+
..., description="A general description of what the template does."
|
|
25
|
+
)
|
|
26
|
+
parameterized: ParameterizedTemplate
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class ParameterizedFacet(BaseModel):
|
|
30
|
+
"""Defines the parameterized version of a SQL facet and intent."""
|
|
31
|
+
|
|
32
|
+
parameterized_sql_snippet: str = Field(
|
|
33
|
+
...,
|
|
34
|
+
description="The SQL facet with placeholders (eg., ).",
|
|
35
|
+
# "fragment" is deprecated, keep alias for backward compatibility
|
|
36
|
+
validation_alias=AliasChoices(
|
|
37
|
+
"parameterized_sql_snippet", "parameterized_fragment"
|
|
38
|
+
),
|
|
39
|
+
)
|
|
40
|
+
parameterized_intent: str = Field(
|
|
41
|
+
..., description="The natural language intent with placeholders."
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Facet(BaseModel):
|
|
46
|
+
"""Represents a single, complete facet."""
|
|
47
|
+
|
|
48
|
+
sql_snippet: str = Field(
|
|
49
|
+
...,
|
|
50
|
+
description="The corresponding, complete SQL facet.",
|
|
51
|
+
# "fragment" is deprecated, keep alias for backward compatibility
|
|
52
|
+
validation_alias=AliasChoices("sql_snippet", "fragment"),
|
|
53
|
+
)
|
|
54
|
+
intent: str = Field(..., description="The user's specific intent.")
|
|
55
|
+
manifest: str = Field(
|
|
56
|
+
..., description="A general description of what the facet does."
|
|
57
|
+
)
|
|
58
|
+
parameterized: ParameterizedFacet
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class ValueSearch(BaseModel):
|
|
62
|
+
"""Represents a single, complete value search."""
|
|
63
|
+
|
|
64
|
+
query: str = Field(..., description="The parameterized SQL query (using $value).")
|
|
65
|
+
concept_type: str = Field(
|
|
66
|
+
..., description="The semantic type (e.g., 'City', 'Product ID')."
|
|
67
|
+
)
|
|
68
|
+
description: str | None = Field(None, description="Optional description.")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class ContextSet(BaseModel):
|
|
72
|
+
"""A set of templates, facets and value searches."""
|
|
73
|
+
|
|
74
|
+
templates: list[Template] | None = Field(
|
|
75
|
+
None, description="A list of complete templates."
|
|
76
|
+
)
|
|
77
|
+
facets: list[Facet] | None = Field(
|
|
78
|
+
None,
|
|
79
|
+
description="A list of SQL facets.",
|
|
80
|
+
# "fragments" is deprecated, keep alias for backward compatibility
|
|
81
|
+
validation_alias=AliasChoices("facets", "fragments"),
|
|
82
|
+
)
|
|
83
|
+
value_searches: list[ValueSearch] | None = Field(
|
|
84
|
+
None,
|
|
85
|
+
description="A list of value searches.",
|
|
86
|
+
)
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
from .targeted_facets import GENERATE_TARGETED_FACETS_PROMPT
|
|
2
|
+
from .targeted_templates import GENERATE_TARGETED_TEMPLATES_PROMPT
|
|
3
|
+
from .targeted_value_search import GENERATE_TARGETED_VALUE_SEARCH_PROMPT
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"GENERATE_TARGETED_TEMPLATES_PROMPT",
|
|
7
|
+
"GENERATE_TARGETED_FACETS_PROMPT",
|
|
8
|
+
"GENERATE_TARGETED_VALUE_SEARCH_PROMPT",
|
|
9
|
+
]
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
import textwrap
|
|
2
|
+
|
|
3
|
+
GENERATE_TARGETED_FACETS_PROMPT = textwrap.dedent(
|
|
4
|
+
"""
|
|
5
|
+
**Workflow for Generating Targeted Facets**
|
|
6
|
+
|
|
7
|
+
1. **User Input Loop:**
|
|
8
|
+
- Ask the user to provide an intent and its corresponding SQL snippet.
|
|
9
|
+
- **Important:** Do not infer the intent or SQL snippet. Wait for the user to provide them.
|
|
10
|
+
- **Note:** Remind the user to use table-qualified column names (e.g., `table.column`) in the SQL snippet to avoid ambiguity.
|
|
11
|
+
- After capturing the intent and SQL snippet pair, ask the user if they would like to add another one.
|
|
12
|
+
- Continue this loop until the user indicates they have no more pairs to add.
|
|
13
|
+
|
|
14
|
+
2. **Review and Confirmation:**
|
|
15
|
+
- Present the complete list of user-provided Intent/SQL snippet pairs for confirmation.
|
|
16
|
+
- **Use the following format for each facet:**
|
|
17
|
+
**Facet [Number]**
|
|
18
|
+
**Intent:** [The intent]
|
|
19
|
+
**SQL snippet:**
|
|
20
|
+
```sql
|
|
21
|
+
[The SQL snippet, properly formatted]
|
|
22
|
+
```
|
|
23
|
+
- Ask if any modifications are needed. If so, work with the user to refine the pairs.
|
|
24
|
+
|
|
25
|
+
3. **Final Facet Generation:**
|
|
26
|
+
- Once approved, call the `generate_facets` tool with the approved pairs.
|
|
27
|
+
- **Note:** If the number of approved pairs is very large (e.g., over 50), break the list into smaller chunks and call the `generate_facets` tool for each chunk.
|
|
28
|
+
- The tool will return the final JSON content as a string.
|
|
29
|
+
|
|
30
|
+
4. **Save Facets:**
|
|
31
|
+
- Ask the user to choose one of the following options:
|
|
32
|
+
1. Create a new context set file.
|
|
33
|
+
2. Append facets to an existing context set file.
|
|
34
|
+
|
|
35
|
+
- **If creating a new file:**
|
|
36
|
+
- You will need to ask the user for the database instance and database name to create the filename.
|
|
37
|
+
- Call the `save_context_set` tool. You will need to provide the database instance, database name, the JSON content from the previous step, and the root directory where the Gemini CLI is running.
|
|
38
|
+
|
|
39
|
+
- **If appending to an existing file:**
|
|
40
|
+
- Ask the user to provide the path to the existing context set file.
|
|
41
|
+
- Call the `attach_context_set` tool with the JSON content and the absolute file path.
|
|
42
|
+
|
|
43
|
+
5. **Generate Upload URL (Optional):**
|
|
44
|
+
- After the file is saved, ask the user if they want to generate a URL to upload the context set file.
|
|
45
|
+
- If the user confirms, you must collect the necessary database context from them. This includes:
|
|
46
|
+
- **Database Type:** 'alloydb', 'cloudsql', or 'spanner'.
|
|
47
|
+
- **Project ID:** The Google Cloud project ID.
|
|
48
|
+
- **And depending on the database type:**
|
|
49
|
+
- For 'alloydb': Location and Cluster ID.
|
|
50
|
+
- For 'cloudsql': Instance ID.
|
|
51
|
+
- For 'spanner': Instance ID and Database ID.
|
|
52
|
+
- Once you have the required information, call the `generate_upload_url` tool to provide the upload URL to the user.
|
|
53
|
+
|
|
54
|
+
Start the workflow.
|
|
55
|
+
"""
|
|
56
|
+
)
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import textwrap
|
|
2
|
+
|
|
3
|
+
GENERATE_TARGETED_TEMPLATES_PROMPT = textwrap.dedent(
|
|
4
|
+
"""
|
|
5
|
+
**Workflow for Generating Targeted Templates**
|
|
6
|
+
|
|
7
|
+
1. **User Input Loop:**
|
|
8
|
+
- Ask the user to provide a natural language question and its corresponding SQL query.
|
|
9
|
+
- **Optionally**, ask if they want to provide a specific "intent" for this pair. If not provided, the question will be used as the intent.
|
|
10
|
+
- **Important:** Do not infer the question or SQL query. Wait for the user to provide them.
|
|
11
|
+
- After capturing the inputs for a template, ask the user if they would like to add another one.
|
|
12
|
+
- Continue this loop until the user indicates they have no more to add.
|
|
13
|
+
|
|
14
|
+
2. **Review and Confirmation:**
|
|
15
|
+
- Present the complete list of user-provided Question/SQL pairs for confirmation.
|
|
16
|
+
- **Use the following format for each pair:**
|
|
17
|
+
**Template [Number]**
|
|
18
|
+
**Question:** [The natural language question]
|
|
19
|
+
**SQL:**
|
|
20
|
+
```sql
|
|
21
|
+
[The SQL query, properly formatted]
|
|
22
|
+
```
|
|
23
|
+
**Intent:** [The intent, if provided.]
|
|
24
|
+
- Ask if any modifications are needed. If so, work with the user to refine the pairs.
|
|
25
|
+
|
|
26
|
+
3. **Final Template Generation:**
|
|
27
|
+
- Once approved, call the `generate_templates` tool with the approved pairs.
|
|
28
|
+
- **Note:** If the number of approved pairs is very large (e.g., over 50), break the list into smaller chunks and call the `generate_templates` tool for each chunk.
|
|
29
|
+
- The tool will return the final JSON content as a string.
|
|
30
|
+
|
|
31
|
+
4. **Save Templates:**
|
|
32
|
+
- Ask the user to choose one of the following options:
|
|
33
|
+
1. Create a new context set file.
|
|
34
|
+
2. Append templates to an existing context set file.
|
|
35
|
+
|
|
36
|
+
- **If creating a new file:**
|
|
37
|
+
- You will need to ask the user for the database instance and database name to create the filename.
|
|
38
|
+
- Call the `save_context_set` tool. You will need to provide the database instance, database name, the JSON content from the previous step, and the root directory where the Gemini CLI is running.
|
|
39
|
+
|
|
40
|
+
- **If appending to an existing file:**
|
|
41
|
+
- Ask the user to provide the path to the existing context set file.
|
|
42
|
+
- Call the `attach_context_set` tool with the JSON content and the absolute file path.
|
|
43
|
+
|
|
44
|
+
5. **Generate Upload URL (Optional):**
|
|
45
|
+
- After the file is saved, ask the user if they want to generate a URL to upload the context set file.
|
|
46
|
+
- If the user confirms, you must collect the necessary database context from them. This includes:
|
|
47
|
+
- **Database Type:** 'alloydb', 'cloudsql', or 'spanner'.
|
|
48
|
+
- **Project ID:** The Google Cloud project ID.
|
|
49
|
+
- **And depending on the database type:**
|
|
50
|
+
- For 'alloydb': Location and Cluster ID.
|
|
51
|
+
- For 'cloudsql': Instance ID.
|
|
52
|
+
- For 'spanner': Instance ID and Database ID.
|
|
53
|
+
- Once you have the required information, call the `generate_upload_url` tool to provide the upload URL to the user.
|
|
54
|
+
|
|
55
|
+
Start the workflow.
|
|
56
|
+
"""
|
|
57
|
+
)
|