google-cloud-db-context-engineering 0.7.2__tar.gz → 0.7.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {google_cloud_db_context_engineering-0.7.2/src/google_cloud_db_context_engineering.egg-info → google_cloud_db_context_engineering-0.7.4}/PKG-INFO +8 -11
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/README.md +7 -10
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/pyproject.toml +3 -3
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/context_mutator.py +2 -2
- google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/common/context_validator.py +188 -0
- google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/__init__.py +15 -0
- google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/bigtable.py +44 -0
- google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/custom.py +104 -0
- google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/firestore.py +60 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/spanner.py +40 -2
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/evaluate_generator.py +171 -7
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/main.py +37 -6
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4/src/google_cloud_db_context_engineering.egg-info}/PKG-INFO +8 -11
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/SOURCES.txt +4 -0
- google_cloud_db_context_engineering-0.7.2/src/google/cloud/db_context_enrichment/model/__init__.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/LICENSE +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/setup.cfg +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/__init__.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/__init__.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/config.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/context_store_client.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/dataset/__init__.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/dataset/dataset_generator.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/__init__.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/alloydb.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/base.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/mysql.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/postgres.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/result_reader.py +0 -0
- {google_cloud_db_context_engineering-0.7.2/src/google/cloud/db_context_enrichment/evaluate/db_generators → google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/model}/__init__.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/model/context.py +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/dependency_links.txt +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/entry_points.txt +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/requires.txt +0 -0
- {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: google-cloud-db-context-engineering
|
|
3
|
-
Version: 0.7.
|
|
3
|
+
Version: 0.7.4
|
|
4
4
|
Summary: A FastMCP server for generating natural language to SQL templates from database schemas.
|
|
5
5
|
Requires-Python: >=3.12
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -16,19 +16,17 @@ Requires-Dist: pytest; extra == "test"
|
|
|
16
16
|
Requires-Dist: pytest-asyncio; extra == "test"
|
|
17
17
|
Dynamic: license-file
|
|
18
18
|
|
|
19
|
-
This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
|
|
20
|
-
|
|
21
19
|
# Context Engineering Agent
|
|
22
20
|
|
|
23
|
-
The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL
|
|
21
|
+
The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL, Spanner Graph, and PostgreSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
|
|
24
22
|
|
|
25
23
|
---
|
|
26
24
|
|
|
27
25
|
## Why Context Engineering?
|
|
28
26
|
|
|
29
|
-
When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL, pure GQL, or hybrid graph queries—is critical.
|
|
27
|
+
When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL (PostgreSQL, GoogleSQL, MySQL), pure GQL, or hybrid graph queries—is critical.
|
|
30
28
|
|
|
31
|
-
As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner
|
|
29
|
+
As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
|
|
32
30
|
|
|
33
31
|
---
|
|
34
32
|
|
|
@@ -40,7 +38,7 @@ A `ContextSet` is the central artifact generated and managed by the agent, conta
|
|
|
40
38
|
* **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses, specialized join filters, or graph `MATCH` traversal patterns) linked to domain vocabulary.
|
|
41
39
|
* **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or trigram search on relational and graph property tables.
|
|
42
40
|
|
|
43
|
-
For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner
|
|
41
|
+
For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
|
|
44
42
|
|
|
45
43
|
---
|
|
46
44
|
|
|
@@ -49,13 +47,13 @@ For full schema details, structure specifications, and dialect-specific JSON rep
|
|
|
49
47
|
Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
|
|
50
48
|
|
|
51
49
|
Follow the step-by-step setup guide in the official documentation:
|
|
52
|
-
👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner
|
|
50
|
+
👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
|
|
53
51
|
|
|
54
52
|
---
|
|
55
53
|
|
|
56
54
|
## Primary Workflow Phases
|
|
57
55
|
|
|
58
|
-
The agent enables you to craft an optimized context for QueryData API through three primary phases
|
|
56
|
+
The agent enables you to craft an optimized context for QueryData API through three primary phases. Depending on your needs, you may also toggle the agent to skip phases. For example, if you have a dataset already, you can skip directly to context optimization "optimize context using dataset in \<file\>."
|
|
59
57
|
|
|
60
58
|
### Phase 1: Artifact Ingestion
|
|
61
59
|
*Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
|
|
@@ -80,8 +78,7 @@ The optimization loop creates an initial `ContextSet` and then iteratively refin
|
|
|
80
78
|
1. **Bootstrap**: Generate an initial baseline context.
|
|
81
79
|
2. **Evaluate**: Measure context effectiveness against a golden dataset.
|
|
82
80
|
3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
|
|
83
|
-
4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
|
|
84
|
-
5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
|
|
81
|
+
4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality until we reach an optimal point.
|
|
85
82
|
|
|
86
83
|
*Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
|
|
87
84
|
|
{google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/README.md
RENAMED
|
@@ -1,16 +1,14 @@
|
|
|
1
|
-
This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
|
|
2
|
-
|
|
3
1
|
# Context Engineering Agent
|
|
4
2
|
|
|
5
|
-
The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL
|
|
3
|
+
The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL, Spanner Graph, and PostgreSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
|
|
6
4
|
|
|
7
5
|
---
|
|
8
6
|
|
|
9
7
|
## Why Context Engineering?
|
|
10
8
|
|
|
11
|
-
When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL, pure GQL, or hybrid graph queries—is critical.
|
|
9
|
+
When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL (PostgreSQL, GoogleSQL, MySQL), pure GQL, or hybrid graph queries—is critical.
|
|
12
10
|
|
|
13
|
-
As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner
|
|
11
|
+
As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
|
|
14
12
|
|
|
15
13
|
---
|
|
16
14
|
|
|
@@ -22,7 +20,7 @@ A `ContextSet` is the central artifact generated and managed by the agent, conta
|
|
|
22
20
|
* **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses, specialized join filters, or graph `MATCH` traversal patterns) linked to domain vocabulary.
|
|
23
21
|
* **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or trigram search on relational and graph property tables.
|
|
24
22
|
|
|
25
|
-
For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner
|
|
23
|
+
For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
|
|
26
24
|
|
|
27
25
|
---
|
|
28
26
|
|
|
@@ -31,13 +29,13 @@ For full schema details, structure specifications, and dialect-specific JSON rep
|
|
|
31
29
|
Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
|
|
32
30
|
|
|
33
31
|
Follow the step-by-step setup guide in the official documentation:
|
|
34
|
-
👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner
|
|
32
|
+
👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
|
|
35
33
|
|
|
36
34
|
---
|
|
37
35
|
|
|
38
36
|
## Primary Workflow Phases
|
|
39
37
|
|
|
40
|
-
The agent enables you to craft an optimized context for QueryData API through three primary phases
|
|
38
|
+
The agent enables you to craft an optimized context for QueryData API through three primary phases. Depending on your needs, you may also toggle the agent to skip phases. For example, if you have a dataset already, you can skip directly to context optimization "optimize context using dataset in \<file\>."
|
|
41
39
|
|
|
42
40
|
### Phase 1: Artifact Ingestion
|
|
43
41
|
*Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
|
|
@@ -62,8 +60,7 @@ The optimization loop creates an initial `ContextSet` and then iteratively refin
|
|
|
62
60
|
1. **Bootstrap**: Generate an initial baseline context.
|
|
63
61
|
2. **Evaluate**: Measure context effectiveness against a golden dataset.
|
|
64
62
|
3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
|
|
65
|
-
4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
|
|
66
|
-
5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
|
|
63
|
+
4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality until we reach an optimal point.
|
|
67
64
|
|
|
68
65
|
*Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
|
|
69
66
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "google-cloud-db-context-engineering"
|
|
3
|
-
version = "0.7.
|
|
3
|
+
version = "0.7.4"
|
|
4
4
|
description = "A FastMCP server for generating natural language to SQL templates from database schemas."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.12"
|
|
@@ -45,8 +45,8 @@ dev = [
|
|
|
45
45
|
]
|
|
46
46
|
|
|
47
47
|
[tool.db-context-engineering]
|
|
48
|
-
toolbox_version = "1.
|
|
49
|
-
evalbench_version = "1.
|
|
48
|
+
toolbox_version = "1.10.0"
|
|
49
|
+
evalbench_version = "1.17.0"
|
|
50
50
|
|
|
51
51
|
[tool.ruff]
|
|
52
52
|
line-length = 88
|
|
@@ -48,7 +48,7 @@ def mutate_context_set(file_path: str, mutations: list[Mutation]) -> None:
|
|
|
48
48
|
context_set = context.ContextSet()
|
|
49
49
|
else:
|
|
50
50
|
try:
|
|
51
|
-
with open(file_path) as f:
|
|
51
|
+
with open(file_path, encoding="utf-8") as f:
|
|
52
52
|
raw_data = json.load(f)
|
|
53
53
|
context_set = context.ContextSet.model_validate(raw_data)
|
|
54
54
|
except ValidationError as e:
|
|
@@ -126,7 +126,7 @@ def mutate_context_set(file_path: str, mutations: list[Mutation]) -> None:
|
|
|
126
126
|
# 4. Save validated ContextSet
|
|
127
127
|
try:
|
|
128
128
|
os.makedirs(os.path.dirname(os.path.abspath(file_path)), exist_ok=True)
|
|
129
|
-
with open(file_path, "w") as f:
|
|
129
|
+
with open(file_path, "w", encoding="utf-8") as f:
|
|
130
130
|
f.write(context_set.model_dump_json(indent=2, exclude_none=True))
|
|
131
131
|
except OSError as e:
|
|
132
132
|
raise RuntimeError(f"Error saving ContextSet to {file_path}: {e}") from e
|
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
import json
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
from pydantic import ValidationError
|
|
5
|
+
|
|
6
|
+
from google.cloud.db_context_enrichment.model import context
|
|
7
|
+
|
|
8
|
+
_ATTR_TO_TYPE = {
|
|
9
|
+
"templates": "template",
|
|
10
|
+
"facets": "facet",
|
|
11
|
+
"value_searches": "value_search",
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
# ContextSet accepts camelCase aliases (Context Store returns them on download)
|
|
15
|
+
# and the deprecated "fragments" name. Map them back to the canonical attribute
|
|
16
|
+
# so every check below only has to handle one spelling.
|
|
17
|
+
_ALIAS_TO_ATTR = {
|
|
18
|
+
"valueSearches": "value_searches",
|
|
19
|
+
"fragments": "facets",
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def validate_context_set(file_path: str) -> dict[str, Any]:
|
|
24
|
+
"""Validate a ContextSet file and return a structured report of issues.
|
|
25
|
+
|
|
26
|
+
Always returns a dict of the shape:
|
|
27
|
+
{
|
|
28
|
+
"valid": bool,
|
|
29
|
+
"issues": [
|
|
30
|
+
{
|
|
31
|
+
"location": {"type": str, "index": int} | None,
|
|
32
|
+
"message": str,
|
|
33
|
+
},
|
|
34
|
+
...
|
|
35
|
+
],
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
File access failures (missing path, permission denied, path is a directory,
|
|
39
|
+
etc.) are surfaced as a single issue rather than raised.
|
|
40
|
+
"""
|
|
41
|
+
try:
|
|
42
|
+
with open(file_path, encoding="utf-8") as f:
|
|
43
|
+
text = f.read()
|
|
44
|
+
except OSError as e:
|
|
45
|
+
return {
|
|
46
|
+
"valid": False,
|
|
47
|
+
"issues": [
|
|
48
|
+
_make_issue(f"Could not read file {file_path}: {type(e).__name__}: {e}")
|
|
49
|
+
],
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
if text.strip() == "":
|
|
53
|
+
return {"valid": True, "issues": []}
|
|
54
|
+
|
|
55
|
+
try:
|
|
56
|
+
raw = json.loads(text)
|
|
57
|
+
except json.JSONDecodeError as e:
|
|
58
|
+
snippet = e.doc[max(0, e.pos - 30) : e.pos + 30]
|
|
59
|
+
return {
|
|
60
|
+
"valid": False,
|
|
61
|
+
"issues": [_make_issue(f"File is not valid JSON: {e}. Near: {snippet!r}")],
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
if not isinstance(raw, dict):
|
|
65
|
+
return {
|
|
66
|
+
"valid": False,
|
|
67
|
+
"issues": [_make_issue("Top-level value must be a JSON object")],
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
raw = _normalize_aliases(raw)
|
|
71
|
+
|
|
72
|
+
issues: list[dict[str, Any]] = []
|
|
73
|
+
issues.extend(_check_pydantic(raw))
|
|
74
|
+
issues.extend(_check_duplicates(raw))
|
|
75
|
+
issues.extend(_check_value_search_value_param(raw))
|
|
76
|
+
|
|
77
|
+
return {"valid": len(issues) == 0, "issues": issues}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _make_issue(message: str, location: dict[str, Any] | None = None) -> dict[str, Any]:
|
|
81
|
+
return {"location": location, "message": message}
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _normalize_aliases(raw: dict[str, Any]) -> dict[str, Any]:
|
|
85
|
+
"""Rewrite alias keys to their canonical ContextSet attribute names."""
|
|
86
|
+
return {_ALIAS_TO_ATTR.get(key, key): value for key, value in raw.items()}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _check_pydantic(raw: dict[str, Any]) -> list[dict[str, Any]]:
|
|
90
|
+
issues: list[dict[str, Any]] = []
|
|
91
|
+
try:
|
|
92
|
+
context.ContextSet.model_validate(raw)
|
|
93
|
+
return issues
|
|
94
|
+
except ValidationError as e:
|
|
95
|
+
for err in e.errors():
|
|
96
|
+
loc = err.get("loc", ())
|
|
97
|
+
msg = err.get("msg", "validation error")
|
|
98
|
+
location: dict[str, Any] | None = None
|
|
99
|
+
descriptor = ""
|
|
100
|
+
|
|
101
|
+
if len(loc) >= 1 and loc[0] in _ATTR_TO_TYPE:
|
|
102
|
+
item_type = _ATTR_TO_TYPE[loc[0]]
|
|
103
|
+
if len(loc) >= 2 and isinstance(loc[1], int):
|
|
104
|
+
item_index = loc[1]
|
|
105
|
+
location = {"type": item_type, "index": item_index}
|
|
106
|
+
items = raw.get(loc[0])
|
|
107
|
+
if isinstance(items, list) and 0 <= item_index < len(items):
|
|
108
|
+
descriptor = _describe(item_type, items[item_index])
|
|
109
|
+
|
|
110
|
+
field_path = ".".join(str(p) for p in loc) if loc else ""
|
|
111
|
+
parts = [msg]
|
|
112
|
+
if field_path:
|
|
113
|
+
parts.append(f"at {field_path}")
|
|
114
|
+
if descriptor:
|
|
115
|
+
parts.append(descriptor)
|
|
116
|
+
full_msg = parts[0] + (
|
|
117
|
+
" (" + "; ".join(parts[1:]) + ")" if len(parts) > 1 else ""
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
issues.append(_make_issue(full_msg, location=location))
|
|
121
|
+
return issues
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _check_duplicates(raw: dict[str, Any]) -> list[dict[str, Any]]:
|
|
125
|
+
issues: list[dict[str, Any]] = []
|
|
126
|
+
for attr, item_type in _ATTR_TO_TYPE.items():
|
|
127
|
+
items = raw.get(attr)
|
|
128
|
+
if not isinstance(items, list):
|
|
129
|
+
continue
|
|
130
|
+
seen: dict[str, int] = {}
|
|
131
|
+
for idx, item in enumerate(items):
|
|
132
|
+
try:
|
|
133
|
+
key = _canonical(item)
|
|
134
|
+
except (TypeError, ValueError):
|
|
135
|
+
continue
|
|
136
|
+
if key in seen:
|
|
137
|
+
descriptor = _describe(item_type, item)
|
|
138
|
+
suffix = f" ({descriptor})" if descriptor else ""
|
|
139
|
+
issues.append(
|
|
140
|
+
_make_issue(
|
|
141
|
+
f"Exact duplicate of {item_type} at index {seen[key]}{suffix}",
|
|
142
|
+
location={"type": item_type, "index": idx},
|
|
143
|
+
)
|
|
144
|
+
)
|
|
145
|
+
else:
|
|
146
|
+
seen[key] = idx
|
|
147
|
+
return issues
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _check_value_search_value_param(raw: dict[str, Any]) -> list[dict[str, Any]]:
|
|
151
|
+
issues: list[dict[str, Any]] = []
|
|
152
|
+
items = raw.get("value_searches")
|
|
153
|
+
if not isinstance(items, list):
|
|
154
|
+
return issues
|
|
155
|
+
for idx, item in enumerate(items):
|
|
156
|
+
if not isinstance(item, dict):
|
|
157
|
+
continue
|
|
158
|
+
query = item.get("query")
|
|
159
|
+
if isinstance(query, str) and "$value" not in query:
|
|
160
|
+
descriptor = _describe("value_search", item)
|
|
161
|
+
suffix = f" ({descriptor})" if descriptor else ""
|
|
162
|
+
issues.append(
|
|
163
|
+
_make_issue(
|
|
164
|
+
f"value_search query must reference the $value parameter{suffix}",
|
|
165
|
+
location={"type": "value_search", "index": idx},
|
|
166
|
+
)
|
|
167
|
+
)
|
|
168
|
+
return issues
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _describe(item_type: str, item: Any) -> str:
|
|
172
|
+
"""Short human-readable identifier for an item, e.g. "intent: 'active users'"."""
|
|
173
|
+
if not isinstance(item, dict):
|
|
174
|
+
return ""
|
|
175
|
+
if item_type == "template":
|
|
176
|
+
nl = item.get("nl_query")
|
|
177
|
+
return f"nl_query: {nl!r}" if nl else ""
|
|
178
|
+
if item_type == "facet":
|
|
179
|
+
intent = item.get("intent")
|
|
180
|
+
return f"intent: {intent!r}" if intent else ""
|
|
181
|
+
if item_type == "value_search":
|
|
182
|
+
ct = item.get("concept_type")
|
|
183
|
+
return f"concept_type: {ct!r}" if ct else ""
|
|
184
|
+
return ""
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _canonical(obj: Any) -> str:
|
|
188
|
+
return json.dumps(obj, sort_keys=True, ensure_ascii=False)
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from .alloydb import AlloyDBConfigGenerator
|
|
2
|
+
from .base import BaseDBConfigGenerator
|
|
3
|
+
from .custom import CustomDBConfigGenerator
|
|
4
|
+
from .mysql import MySQLConfigGenerator
|
|
5
|
+
from .postgres import PostgresConfigGenerator
|
|
6
|
+
from .spanner import SpannerConfigGenerator
|
|
7
|
+
|
|
8
|
+
__all__ = [
|
|
9
|
+
"BaseDBConfigGenerator",
|
|
10
|
+
"AlloyDBConfigGenerator",
|
|
11
|
+
"CustomDBConfigGenerator",
|
|
12
|
+
"MySQLConfigGenerator",
|
|
13
|
+
"PostgresConfigGenerator",
|
|
14
|
+
"SpannerConfigGenerator",
|
|
15
|
+
]
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
3
|
+
import yaml
|
|
4
|
+
|
|
5
|
+
from .base import BaseDBConfigGenerator
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class BigtableConfigGenerator(BaseDBConfigGenerator):
|
|
9
|
+
"""
|
|
10
|
+
Dedicated generator mapping properties to explicit Bigtable configuration
|
|
11
|
+
topologies. Bypasses the GDA Python SDK to construct the model configuration
|
|
12
|
+
directly, as the SDK may lack native BigtableReference definitions.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
SOURCE_TYPE = "bigtable"
|
|
16
|
+
DIALECT = "bigtable"
|
|
17
|
+
REQUIRED_FIELDS = {"project", "instance"}
|
|
18
|
+
|
|
19
|
+
def generate_db_config(self) -> str:
|
|
20
|
+
db_config = {
|
|
21
|
+
"db_type": "bigtable",
|
|
22
|
+
"dialect": self.DIALECT,
|
|
23
|
+
"database_name": self.params.get("instance"),
|
|
24
|
+
"database_path": f"projects/{self.params.get('project')}/instances/{self.params.get('instance')}",
|
|
25
|
+
"instance_id": self.params.get("instance"),
|
|
26
|
+
"gcp_project_id": self.params.get("project"),
|
|
27
|
+
"max_executions_per_minute": 100,
|
|
28
|
+
}
|
|
29
|
+
return yaml.safe_dump(
|
|
30
|
+
db_config, sort_keys=False, default_flow_style=False
|
|
31
|
+
).strip()
|
|
32
|
+
|
|
33
|
+
def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
|
|
34
|
+
return {
|
|
35
|
+
"bigtable_reference": {
|
|
36
|
+
"database_reference": {
|
|
37
|
+
"project_id": self.params.get("project"),
|
|
38
|
+
"instance_id": self.params.get("instance"),
|
|
39
|
+
},
|
|
40
|
+
"agent_context_reference": {
|
|
41
|
+
"context_set_id": context_set_id,
|
|
42
|
+
},
|
|
43
|
+
}
|
|
44
|
+
}
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
import os
|
|
2
|
+
from typing import Any
|
|
3
|
+
|
|
4
|
+
import yaml
|
|
5
|
+
|
|
6
|
+
from .base import BaseDBConfigGenerator
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class CustomDBConfigGenerator(BaseDBConfigGenerator):
|
|
10
|
+
"""
|
|
11
|
+
Pluggable generator mapping properties to custom evaluation configurations.
|
|
12
|
+
Enables arbitrary third-party or proprietary internal engines to be plugged
|
|
13
|
+
into the evaluation framework via dynamic connector and generator SPIs
|
|
14
|
+
without exposing engine-specific code or configuration schemas.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
SOURCE_TYPE = "custom"
|
|
18
|
+
DIALECT = "sql"
|
|
19
|
+
REQUIRED_FIELDS = set()
|
|
20
|
+
|
|
21
|
+
def __init__(self, params: dict[str, Any]):
|
|
22
|
+
self.params = params
|
|
23
|
+
if "project" not in self.params:
|
|
24
|
+
project_id = os.environ.get("GOOGLE_CLOUD_PROJECT") or os.environ.get(
|
|
25
|
+
"GCP_PROJECT"
|
|
26
|
+
)
|
|
27
|
+
if project_id:
|
|
28
|
+
self.params["project"] = project_id
|
|
29
|
+
self.connector_class = params.get("connector_class", "")
|
|
30
|
+
self.generator_class = params.get("generator_class", "")
|
|
31
|
+
self.DIALECT = params.get("dialect", self.DIALECT)
|
|
32
|
+
self.validate()
|
|
33
|
+
|
|
34
|
+
def validate(self) -> None:
|
|
35
|
+
"""
|
|
36
|
+
Validates that either connector_class or generator_class is specified.
|
|
37
|
+
"""
|
|
38
|
+
if not self.connector_class and not self.generator_class:
|
|
39
|
+
raise ValueError(
|
|
40
|
+
"Custom source configuration must specify at least 'connector_class' or 'generator_class'."
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
def generate_db_config(self) -> str:
|
|
44
|
+
"""
|
|
45
|
+
Generates the db_config.yaml payload for custom connectors.
|
|
46
|
+
"""
|
|
47
|
+
db_config: dict[str, Any] = {
|
|
48
|
+
"db_type": self.params.get("db_type", "custom"),
|
|
49
|
+
"dialect": self.DIALECT,
|
|
50
|
+
}
|
|
51
|
+
if self.connector_class:
|
|
52
|
+
db_config["connector_class"] = self.connector_class
|
|
53
|
+
|
|
54
|
+
# Forward all non-meta parameters from tools.yaml source block
|
|
55
|
+
excluded_keys = {
|
|
56
|
+
"kind",
|
|
57
|
+
"name",
|
|
58
|
+
"type",
|
|
59
|
+
"connector_class",
|
|
60
|
+
"generator_class",
|
|
61
|
+
"dialect",
|
|
62
|
+
"db_type",
|
|
63
|
+
}
|
|
64
|
+
for key, value in self.params.items():
|
|
65
|
+
if key not in excluded_keys:
|
|
66
|
+
db_config[key] = value
|
|
67
|
+
|
|
68
|
+
return yaml.safe_dump(
|
|
69
|
+
db_config, sort_keys=False, default_flow_style=False
|
|
70
|
+
).strip()
|
|
71
|
+
|
|
72
|
+
def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
|
|
73
|
+
"""
|
|
74
|
+
Datasource reference dictionary. Custom engines manage their own
|
|
75
|
+
schema references or context binding.
|
|
76
|
+
"""
|
|
77
|
+
return {}
|
|
78
|
+
|
|
79
|
+
def generate_model_config(self, context_set_id: str) -> str:
|
|
80
|
+
"""
|
|
81
|
+
Generates model_config.yaml. If generator_class is specified, produces
|
|
82
|
+
a custom generator configuration for dynamic instantiation.
|
|
83
|
+
"""
|
|
84
|
+
if self.generator_class:
|
|
85
|
+
model_config: dict[str, Any] = {
|
|
86
|
+
"generator": "custom",
|
|
87
|
+
"generator_class": self.generator_class,
|
|
88
|
+
"context_set_id": context_set_id,
|
|
89
|
+
}
|
|
90
|
+
excluded_keys = {
|
|
91
|
+
"kind",
|
|
92
|
+
"name",
|
|
93
|
+
"type",
|
|
94
|
+
"connector_class",
|
|
95
|
+
"generator_class",
|
|
96
|
+
}
|
|
97
|
+
for key, value in self.params.items():
|
|
98
|
+
if key not in excluded_keys:
|
|
99
|
+
model_config[key] = value
|
|
100
|
+
return yaml.safe_dump(
|
|
101
|
+
model_config, sort_keys=False, default_flow_style=False
|
|
102
|
+
).strip()
|
|
103
|
+
|
|
104
|
+
return super().generate_model_config(context_set_id)
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
from typing import Any
|
|
2
|
+
|
|
3
|
+
import yaml
|
|
4
|
+
|
|
5
|
+
from .base import BaseDBConfigGenerator
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class FirestoreConfigGenerator(BaseDBConfigGenerator):
|
|
9
|
+
"""
|
|
10
|
+
Dedicated generator mapping properties to Firestore (MongoDB query dialect)
|
|
11
|
+
topologies utilized by both EvalBench binaries and GDA REST API.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
SOURCE_TYPE = "firestore"
|
|
15
|
+
DIALECT = "mongodb"
|
|
16
|
+
REQUIRED_FIELDS = BaseDBConfigGenerator.REQUIRED_FIELDS | {
|
|
17
|
+
"project",
|
|
18
|
+
"database",
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
def __init__(self, params: dict[str, Any]):
|
|
22
|
+
super().__init__(params)
|
|
23
|
+
self.project = params.get("project")
|
|
24
|
+
self.database = params.get("database")
|
|
25
|
+
self.connection_string = params.get("connection_string")
|
|
26
|
+
self.collection_ids = params.get("collection_ids") or params.get("table_ids")
|
|
27
|
+
|
|
28
|
+
def generate_db_config(self) -> str:
|
|
29
|
+
db_type = "mongodb"
|
|
30
|
+
|
|
31
|
+
db_config = {
|
|
32
|
+
"db_type": db_type,
|
|
33
|
+
"dialect": self.DIALECT,
|
|
34
|
+
"database_name": self.database,
|
|
35
|
+
"database_path": "",
|
|
36
|
+
"firestore_database": f"projects/{self.project}/databases/{self.database}",
|
|
37
|
+
"max_executions_per_minute": 120,
|
|
38
|
+
}
|
|
39
|
+
if self.connection_string:
|
|
40
|
+
db_config["connection_string"] = self.connection_string
|
|
41
|
+
|
|
42
|
+
return yaml.safe_dump(
|
|
43
|
+
db_config, sort_keys=False, default_flow_style=False
|
|
44
|
+
).strip()
|
|
45
|
+
|
|
46
|
+
def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
|
|
47
|
+
db_ref: dict[str, Any] = {
|
|
48
|
+
"project_id": self.project,
|
|
49
|
+
"database_id": self.database,
|
|
50
|
+
}
|
|
51
|
+
if self.collection_ids:
|
|
52
|
+
db_ref["collection_ids"] = self.collection_ids
|
|
53
|
+
|
|
54
|
+
ref: dict[str, Any] = {"firestore_reference": {"database_reference": db_ref}}
|
|
55
|
+
if context_set_id:
|
|
56
|
+
ref["firestore_reference"]["agent_context_reference"] = {
|
|
57
|
+
"context_set_id": context_set_id
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
return ref
|
|
@@ -9,10 +9,10 @@ class SpannerConfigGenerator(BaseDBConfigGenerator):
|
|
|
9
9
|
"""
|
|
10
10
|
Dedicated generator mapping properties to explicit Spanner configuration
|
|
11
11
|
topologies utilized by both EvalBench binaries and GDA Context objects.
|
|
12
|
+
Supports both GoogleSQL and PostgreSQL dialects.
|
|
12
13
|
"""
|
|
13
14
|
|
|
14
15
|
SOURCE_TYPE = "spanner"
|
|
15
|
-
DIALECT = "spanner_gsql"
|
|
16
16
|
REQUIRED_FIELDS = BaseDBConfigGenerator.REQUIRED_FIELDS | {
|
|
17
17
|
"project",
|
|
18
18
|
"instance",
|
|
@@ -25,6 +25,40 @@ class SpannerConfigGenerator(BaseDBConfigGenerator):
|
|
|
25
25
|
self.instance = params.get("instance")
|
|
26
26
|
self.database = params.get("database")
|
|
27
27
|
|
|
28
|
+
raw_dialect = (
|
|
29
|
+
params.get("dialect")
|
|
30
|
+
or params.get("engine")
|
|
31
|
+
or params.get("database_dialect")
|
|
32
|
+
)
|
|
33
|
+
if not raw_dialect and params.get("type") in ("spanner-postgres", "spanner-pg"):
|
|
34
|
+
raw_dialect = "POSTGRESQL"
|
|
35
|
+
|
|
36
|
+
if raw_dialect:
|
|
37
|
+
normalized = str(raw_dialect).strip().lower().replace("-", "_")
|
|
38
|
+
if normalized in (
|
|
39
|
+
"postgresql",
|
|
40
|
+
"postgres",
|
|
41
|
+
"spanner_pg",
|
|
42
|
+
"pg",
|
|
43
|
+
"spanner_postgres",
|
|
44
|
+
):
|
|
45
|
+
self.engine = "POSTGRESQL"
|
|
46
|
+
self._dialect = "spanner_pg"
|
|
47
|
+
elif normalized in ("google_sql", "googlesql", "spanner_gsql", "gsql"):
|
|
48
|
+
self.engine = "GOOGLE_SQL"
|
|
49
|
+
self._dialect = "spanner_gsql"
|
|
50
|
+
else:
|
|
51
|
+
raise ValueError(
|
|
52
|
+
f"Unsupported Spanner dialect/engine: '{raw_dialect}'. Must be 'GOOGLE_SQL' or 'POSTGRESQL'."
|
|
53
|
+
)
|
|
54
|
+
else:
|
|
55
|
+
self.engine = "GOOGLE_SQL"
|
|
56
|
+
self._dialect = "spanner_gsql"
|
|
57
|
+
|
|
58
|
+
@property
|
|
59
|
+
def DIALECT(self) -> str:
|
|
60
|
+
return self._dialect
|
|
61
|
+
|
|
28
62
|
def generate_db_config(self) -> str:
|
|
29
63
|
db_type = "spanner"
|
|
30
64
|
db_path = f"projects/{self.project}/instances/{self.instance}/databases/{self.database}"
|
|
@@ -44,12 +78,16 @@ class SpannerConfigGenerator(BaseDBConfigGenerator):
|
|
|
44
78
|
|
|
45
79
|
def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
|
|
46
80
|
database_ref: dict[str, Any] = {
|
|
47
|
-
"engine":
|
|
81
|
+
"engine": self.engine,
|
|
48
82
|
"project_id": self.project,
|
|
49
83
|
"instance_id": self.instance,
|
|
50
84
|
"database_id": self.database,
|
|
51
85
|
}
|
|
52
86
|
if graph_ids := self.params.get("graph_ids"):
|
|
87
|
+
if self.engine == "POSTGRESQL":
|
|
88
|
+
raise ValueError(
|
|
89
|
+
"graph_ids is not supported for Spanner PostgreSQL dialect"
|
|
90
|
+
)
|
|
53
91
|
if not isinstance(graph_ids, list) or not all(
|
|
54
92
|
isinstance(g, str) for g in graph_ids
|
|
55
93
|
):
|
|
@@ -1,7 +1,11 @@
|
|
|
1
|
+
import importlib
|
|
1
2
|
import json
|
|
3
|
+
import logging
|
|
2
4
|
import os
|
|
3
5
|
import re
|
|
6
|
+
import sys
|
|
4
7
|
import textwrap
|
|
8
|
+
import threading
|
|
5
9
|
from typing import Any
|
|
6
10
|
|
|
7
11
|
import yaml
|
|
@@ -10,10 +14,15 @@ from google.cloud.db_context_enrichment.common import config
|
|
|
10
14
|
|
|
11
15
|
from .db_generators.alloydb import AlloyDBConfigGenerator
|
|
12
16
|
from .db_generators.base import BaseDBConfigGenerator
|
|
17
|
+
from .db_generators.bigtable import BigtableConfigGenerator
|
|
18
|
+
from .db_generators.custom import CustomDBConfigGenerator
|
|
19
|
+
from .db_generators.firestore import FirestoreConfigGenerator
|
|
13
20
|
from .db_generators.mysql import MySQLConfigGenerator
|
|
14
21
|
from .db_generators.postgres import PostgresConfigGenerator
|
|
15
22
|
from .db_generators.spanner import SpannerConfigGenerator
|
|
16
23
|
|
|
24
|
+
logger = logging.getLogger(__name__)
|
|
25
|
+
|
|
17
26
|
# Constants for EvalBench configuration filenames
|
|
18
27
|
DB_CONFIG_NAME = "db_config.yaml"
|
|
19
28
|
MODEL_CONFIG_NAME = "model_config.yaml"
|
|
@@ -76,10 +85,10 @@ def _extract_toolbox_params(
|
|
|
76
85
|
with open(toolbox_config_path) as f:
|
|
77
86
|
content = f.read()
|
|
78
87
|
interpolated = _interpolate_env_vars(content)
|
|
79
|
-
docs = yaml.safe_load_all(interpolated)
|
|
88
|
+
docs = [doc for doc in yaml.safe_load_all(interpolated) if doc]
|
|
89
|
+
|
|
90
|
+
source_doc = None
|
|
80
91
|
for doc in docs:
|
|
81
|
-
if not doc:
|
|
82
|
-
continue
|
|
83
92
|
if (
|
|
84
93
|
doc.get("kind") == "source"
|
|
85
94
|
and doc.get("name") == toolbox_source_name
|
|
@@ -88,11 +97,24 @@ def _extract_toolbox_params(
|
|
|
88
97
|
raise ValueError(
|
|
89
98
|
f"Selected source '{toolbox_source_name}' is missing the 'type' field."
|
|
90
99
|
)
|
|
91
|
-
|
|
100
|
+
source_doc = doc
|
|
101
|
+
break
|
|
92
102
|
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
103
|
+
if not source_doc:
|
|
104
|
+
raise ValueError(
|
|
105
|
+
f"Could not find a 'kind: source' named '{toolbox_source_name}' in {toolbox_config_path}"
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
# For Spanner sources, state.md is the authoritative single source of truth for graph_ids
|
|
109
|
+
# (QueryData API requires explicit graph_ids in model_config.yaml, whereas tools.yaml
|
|
110
|
+
# only configures MCP Toolbox runtime tools and parameters).
|
|
111
|
+
if source_doc.get("type") == "spanner":
|
|
112
|
+
state_md_dir = os.path.dirname(toolbox_config_path)
|
|
113
|
+
state_md_path = os.path.join(state_md_dir, "state.md")
|
|
114
|
+
if graph_ids := _parse_graph_ids_from_state_md(state_md_path):
|
|
115
|
+
source_doc["graph_ids"] = graph_ids
|
|
116
|
+
|
|
117
|
+
return source_doc
|
|
96
118
|
|
|
97
119
|
except FileNotFoundError:
|
|
98
120
|
raise ValueError(f"Config file not found: {toolbox_config_path}")
|
|
@@ -104,6 +126,72 @@ def _extract_toolbox_params(
|
|
|
104
126
|
raise ValueError(f"Failed to parse {toolbox_config_path} as YAML: {e}")
|
|
105
127
|
|
|
106
128
|
|
|
129
|
+
def _parse_graph_ids_from_state_md(state_md_path: str) -> list[str] | None:
|
|
130
|
+
"""Parses graph_ids from state.md.
|
|
131
|
+
|
|
132
|
+
state.md is the authoritative source of truth for the database and graph scope
|
|
133
|
+
because QueryData API requires explicit graph_ids in model_config.yaml to evaluate
|
|
134
|
+
property graphs, whereas tools.yaml only configures MCP Toolbox runtime tools.
|
|
135
|
+
"""
|
|
136
|
+
if not os.path.exists(state_md_path):
|
|
137
|
+
return None
|
|
138
|
+
with open(state_md_path, encoding="utf-8") as f:
|
|
139
|
+
content = f.read()
|
|
140
|
+
match = re.search(
|
|
141
|
+
r"(?:^[ \t]*[-*][ \t]*)?\*\*Graph\s+Ids?:?\*\*:?[ \t]*([^\n]*)",
|
|
142
|
+
content,
|
|
143
|
+
re.MULTILINE | re.IGNORECASE,
|
|
144
|
+
)
|
|
145
|
+
if not match:
|
|
146
|
+
return None
|
|
147
|
+
|
|
148
|
+
val_str = match.group(1).split("#")[0].strip()
|
|
149
|
+
|
|
150
|
+
# If empty on the same line, check for multiline sub-bullets
|
|
151
|
+
if not val_str:
|
|
152
|
+
after_match = content[match.end() :]
|
|
153
|
+
bullet_items = []
|
|
154
|
+
for line in after_match.splitlines():
|
|
155
|
+
line_stripped = line.strip()
|
|
156
|
+
if not line_stripped:
|
|
157
|
+
continue
|
|
158
|
+
if line_stripped.startswith(("-", "*")) and not re.match(
|
|
159
|
+
r"^[-*]\s*\*\*", line_stripped
|
|
160
|
+
):
|
|
161
|
+
item = line_stripped.lstrip("-* ").split("#")[0].strip().strip("'\"`")
|
|
162
|
+
if item and item.lower() not in (
|
|
163
|
+
"none",
|
|
164
|
+
"n/a",
|
|
165
|
+
"null",
|
|
166
|
+
"nil",
|
|
167
|
+
"-",
|
|
168
|
+
):
|
|
169
|
+
bullet_items.append(item)
|
|
170
|
+
elif (
|
|
171
|
+
line_stripped.startswith("#")
|
|
172
|
+
or line_stripped.startswith("- **")
|
|
173
|
+
or line_stripped.startswith("* **")
|
|
174
|
+
):
|
|
175
|
+
break
|
|
176
|
+
else:
|
|
177
|
+
break
|
|
178
|
+
return bullet_items if bullet_items else None
|
|
179
|
+
|
|
180
|
+
# Handle explicit empty / none indicators
|
|
181
|
+
if val_str.lower() in ("none", "n/a", "null", "nil", "-", "[]", ""):
|
|
182
|
+
return None
|
|
183
|
+
|
|
184
|
+
if val_str.startswith("[") and val_str.endswith("]"):
|
|
185
|
+
val_str = val_str[1:-1]
|
|
186
|
+
graphs = [
|
|
187
|
+
g.strip().strip("'\"`")
|
|
188
|
+
for g in val_str.split(",")
|
|
189
|
+
if g.strip().strip("'\"`")
|
|
190
|
+
and g.strip().strip("'\"`").lower() not in ("none", "n/a", "null", "nil", "-")
|
|
191
|
+
]
|
|
192
|
+
return graphs if graphs else None
|
|
193
|
+
|
|
194
|
+
|
|
107
195
|
def _interpolate_env_vars(raw_yaml: str) -> str:
|
|
108
196
|
"""Replaces ${ENV_NAME} or ${ENV_NAME:default_value} with environment variables."""
|
|
109
197
|
# Matches ${VAR_NAME} or ${VAR_NAME:fallback}
|
|
@@ -124,17 +212,93 @@ def _interpolate_env_vars(raw_yaml: str) -> str:
|
|
|
124
212
|
return pattern.sub(replacer, raw_yaml)
|
|
125
213
|
|
|
126
214
|
|
|
215
|
+
_SYS_PATH_LOCK = threading.Lock()
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _import_with_cwd_fallback(mod_name: str):
|
|
219
|
+
"""Imports mod_name, falling back to appending os.getcwd() to sys.path."""
|
|
220
|
+
try:
|
|
221
|
+
return importlib.import_module(mod_name)
|
|
222
|
+
except ModuleNotFoundError as e:
|
|
223
|
+
if e.name is None or not (
|
|
224
|
+
mod_name == e.name or mod_name.startswith(e.name + ".")
|
|
225
|
+
):
|
|
226
|
+
raise
|
|
227
|
+
cwd = os.path.abspath(os.getcwd())
|
|
228
|
+
with _SYS_PATH_LOCK:
|
|
229
|
+
if not any(os.path.abspath(p or cwd) == cwd for p in sys.path):
|
|
230
|
+
# Keep cwd at the end of sys.path so lazy runtime imports work
|
|
231
|
+
# without shadowing standard library or installed packages.
|
|
232
|
+
sys.path.append(cwd)
|
|
233
|
+
importlib.invalidate_caches()
|
|
234
|
+
return importlib.import_module(mod_name)
|
|
235
|
+
|
|
236
|
+
|
|
127
237
|
def _get_db_generator(params: dict[str, Any]) -> BaseDBConfigGenerator:
|
|
128
238
|
"""Factory function to build the correct Evaluation Generator."""
|
|
129
239
|
source_type = params.get("type", "").lower()
|
|
130
240
|
|
|
131
241
|
generators = {
|
|
132
242
|
AlloyDBConfigGenerator.SOURCE_TYPE: AlloyDBConfigGenerator,
|
|
243
|
+
BigtableConfigGenerator.SOURCE_TYPE: BigtableConfigGenerator,
|
|
133
244
|
PostgresConfigGenerator.SOURCE_TYPE: PostgresConfigGenerator,
|
|
134
245
|
MySQLConfigGenerator.SOURCE_TYPE: MySQLConfigGenerator,
|
|
135
246
|
SpannerConfigGenerator.SOURCE_TYPE: SpannerConfigGenerator,
|
|
247
|
+
"spanner-postgres": SpannerConfigGenerator,
|
|
248
|
+
"spanner-pg": SpannerConfigGenerator,
|
|
249
|
+
FirestoreConfigGenerator.SOURCE_TYPE: FirestoreConfigGenerator,
|
|
250
|
+
CustomDBConfigGenerator.SOURCE_TYPE: CustomDBConfigGenerator,
|
|
136
251
|
}
|
|
137
252
|
|
|
253
|
+
# Dynamically register external custom database configuration generators.
|
|
254
|
+
# AUTOCTX_CUSTOM_GENERATORS holds a dotted Python module import path
|
|
255
|
+
# (e.g., "my_package.custom_generators") that exposes a
|
|
256
|
+
# CUSTOM_GENERATORS: dict[str, type[BaseDBConfigGenerator]] mapping
|
|
257
|
+
# custom tools.yaml source types to BaseDBConfigGenerator subclasses.
|
|
258
|
+
custom_gens = {}
|
|
259
|
+
custom_plugin = os.environ.get("AUTOCTX_CUSTOM_GENERATORS")
|
|
260
|
+
if custom_plugin:
|
|
261
|
+
try:
|
|
262
|
+
mod = _import_with_cwd_fallback(custom_plugin)
|
|
263
|
+
custom_gens = getattr(mod, "CUSTOM_GENERATORS", None)
|
|
264
|
+
if custom_gens is None:
|
|
265
|
+
raise AttributeError(
|
|
266
|
+
f"Custom generator module '{custom_plugin}' must define 'CUSTOM_GENERATORS' dict."
|
|
267
|
+
)
|
|
268
|
+
if not isinstance(custom_gens, dict):
|
|
269
|
+
raise TypeError(
|
|
270
|
+
f"CUSTOM_GENERATORS in '{custom_plugin}' must be a dictionary, "
|
|
271
|
+
f"got {type(custom_gens).__name__}."
|
|
272
|
+
)
|
|
273
|
+
generators.update(custom_gens)
|
|
274
|
+
except Exception as e:
|
|
275
|
+
logger.error(
|
|
276
|
+
"Failed to load custom generators from plugin module '%s': %s",
|
|
277
|
+
custom_plugin,
|
|
278
|
+
e,
|
|
279
|
+
)
|
|
280
|
+
raise RuntimeError(
|
|
281
|
+
f"Failed to load custom generators from plugin module '{custom_plugin}': {e}"
|
|
282
|
+
) from e
|
|
283
|
+
|
|
284
|
+
# If the source type is not handled by an external AUTOCTX_CUSTOM_GENERATORS
|
|
285
|
+
# plugin, allow inline connector_class / generator_class in tools.yaml to
|
|
286
|
+
# route directly to CustomDBConfigGenerator.
|
|
287
|
+
if source_type not in custom_gens and (
|
|
288
|
+
"connector_class" in params or "generator_class" in params
|
|
289
|
+
):
|
|
290
|
+
if (
|
|
291
|
+
source_type in generators
|
|
292
|
+
and source_type != CustomDBConfigGenerator.SOURCE_TYPE
|
|
293
|
+
):
|
|
294
|
+
logger.warning(
|
|
295
|
+
"Source type '%s' is being overridden by custom connector_class '%s' / generator_class '%s'.",
|
|
296
|
+
source_type,
|
|
297
|
+
params.get("connector_class"),
|
|
298
|
+
params.get("generator_class"),
|
|
299
|
+
)
|
|
300
|
+
return CustomDBConfigGenerator(params)
|
|
301
|
+
|
|
138
302
|
if source_type not in generators:
|
|
139
303
|
supported = ", ".join(generators.keys())
|
|
140
304
|
raise ValueError(
|
|
@@ -6,6 +6,7 @@ from fastmcp import FastMCP
|
|
|
6
6
|
from google.cloud.db_context_enrichment.common import (
|
|
7
7
|
context_mutator,
|
|
8
8
|
context_store_client,
|
|
9
|
+
context_validator,
|
|
9
10
|
)
|
|
10
11
|
from google.cloud.db_context_enrichment.dataset import dataset_generator
|
|
11
12
|
from google.cloud.db_context_enrichment.evaluate import (
|
|
@@ -73,7 +74,7 @@ def generate_evalbench_configs(
|
|
|
73
74
|
dataset_path: The absolute path to the golden dataset file in the simplified user-facing format (JSON list of objects with keys: "id", "database", "nlq", "golden_sql").
|
|
74
75
|
context_set_id: Full ContextSet resource name to evaluate against.
|
|
75
76
|
toolbox_config_path: The absolute path to the tools.yaml configuration file.
|
|
76
|
-
toolbox_source_name: The name of the database source to use inside tools.yaml. The underlying source block must use a supported 'type' (cloud-sql-postgres, cloud-sql-mysql, spanner, alloydb-postgres).
|
|
77
|
+
toolbox_source_name: The name of the database source to use inside tools.yaml. The underlying source block must use a supported 'type' (cloud-sql-postgres, cloud-sql-mysql, spanner, alloydb-postgres, or custom).
|
|
77
78
|
|
|
78
79
|
Returns:
|
|
79
80
|
A message indicating that the configuration files were successfully created.
|
|
@@ -102,13 +103,14 @@ def generate_upload_url(
|
|
|
102
103
|
|
|
103
104
|
Args:
|
|
104
105
|
db_engine: The database engine. Accepted values are 'alloydb',
|
|
105
|
-
'cloudsql', or '
|
|
106
|
-
field in the tools.yaml file. For example,
|
|
107
|
-
becomes 'alloydb',
|
|
106
|
+
'cloudsql', 'spanner', or 'bigtable'. This can be derived from
|
|
107
|
+
the 'kind' field in the tools.yaml file. For example,
|
|
108
|
+
'alloydb-postgres' becomes 'alloydb', 'cloud-sql-postgres'
|
|
109
|
+
becomes 'cloudsql', and 'bigtable' becomes 'bigtable'.
|
|
108
110
|
project_id: The Google Cloud project ID.
|
|
109
111
|
location: The location of the AlloyDB cluster.
|
|
110
112
|
cluster_id: The ID of the AlloyDB cluster.
|
|
111
|
-
instance_id: The ID of the Cloud SQL or
|
|
113
|
+
instance_id: The ID of the Cloud SQL, Spanner, or Bigtable instance.
|
|
112
114
|
database_id: The ID of the Spanner database.
|
|
113
115
|
|
|
114
116
|
Returns:
|
|
@@ -129,8 +131,13 @@ def generate_upload_url(
|
|
|
129
131
|
return f"https://console.cloud.google.com/spanner/instances/{instance_id}/databases/{database_id}/details/query?project={project_id}"
|
|
130
132
|
else:
|
|
131
133
|
return "Error: Missing instance_id, database_id, or project_id for spanner."
|
|
134
|
+
elif db_engine == "bigtable":
|
|
135
|
+
if instance_id and project_id:
|
|
136
|
+
return f"https://console.cloud.google.com/bigtable/instances/{instance_id}/overview?project={project_id}"
|
|
137
|
+
else:
|
|
138
|
+
return "Error: Missing instance_id or project_id for bigtable."
|
|
132
139
|
else:
|
|
133
|
-
return "Error: Invalid db_engine. Must be one of 'alloydb', 'cloudsql', or '
|
|
140
|
+
return "Error: Invalid db_engine. Must be one of 'alloydb', 'cloudsql', 'spanner', or 'bigtable'."
|
|
134
141
|
|
|
135
142
|
|
|
136
143
|
# NOTE: `@mcp.tool` is intentionally NOT applied to upload_context_set /
|
|
@@ -258,6 +265,30 @@ def mutate_context_set(
|
|
|
258
265
|
return f"Error applying mutations: {str(e)}"
|
|
259
266
|
|
|
260
267
|
|
|
268
|
+
@mcp.tool
|
|
269
|
+
def validate_context_set(file_path: str) -> str:
|
|
270
|
+
"""
|
|
271
|
+
Validate a ContextSet JSON file for structural and convention issues. Reports issues only — does not fix them. The caller (agent) is expected to apply fixes via `mutate_context_set`, then re-run validation until `valid` is true.
|
|
272
|
+
|
|
273
|
+
Args:
|
|
274
|
+
file_path: Absolute path to the ContextSet file.
|
|
275
|
+
|
|
276
|
+
Returns:
|
|
277
|
+
A JSON string of the shape:
|
|
278
|
+
{
|
|
279
|
+
"valid": bool,
|
|
280
|
+
"issues": [
|
|
281
|
+
{
|
|
282
|
+
"location": {"type": "template" | "facet" | "value_search", "index": int} | null,
|
|
283
|
+
"message": str
|
|
284
|
+
},
|
|
285
|
+
...
|
|
286
|
+
]
|
|
287
|
+
}
|
|
288
|
+
"""
|
|
289
|
+
return json.dumps(context_validator.validate_context_set(file_path), indent=2)
|
|
290
|
+
|
|
291
|
+
|
|
261
292
|
@mcp.tool
|
|
262
293
|
async def read_evaluation_result(
|
|
263
294
|
run_folder_path: str, offset: int = 0, batch_size: int = 10
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: google-cloud-db-context-engineering
|
|
3
|
-
Version: 0.7.
|
|
3
|
+
Version: 0.7.4
|
|
4
4
|
Summary: A FastMCP server for generating natural language to SQL templates from database schemas.
|
|
5
5
|
Requires-Python: >=3.12
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -16,19 +16,17 @@ Requires-Dist: pytest; extra == "test"
|
|
|
16
16
|
Requires-Dist: pytest-asyncio; extra == "test"
|
|
17
17
|
Dynamic: license-file
|
|
18
18
|
|
|
19
|
-
This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
|
|
20
|
-
|
|
21
19
|
# Context Engineering Agent
|
|
22
20
|
|
|
23
|
-
The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL
|
|
21
|
+
The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL, Spanner Graph, and PostgreSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
|
|
24
22
|
|
|
25
23
|
---
|
|
26
24
|
|
|
27
25
|
## Why Context Engineering?
|
|
28
26
|
|
|
29
|
-
When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL, pure GQL, or hybrid graph queries—is critical.
|
|
27
|
+
When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL (PostgreSQL, GoogleSQL, MySQL), pure GQL, or hybrid graph queries—is critical.
|
|
30
28
|
|
|
31
|
-
As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner
|
|
29
|
+
As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
|
|
32
30
|
|
|
33
31
|
---
|
|
34
32
|
|
|
@@ -40,7 +38,7 @@ A `ContextSet` is the central artifact generated and managed by the agent, conta
|
|
|
40
38
|
* **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses, specialized join filters, or graph `MATCH` traversal patterns) linked to domain vocabulary.
|
|
41
39
|
* **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or trigram search on relational and graph property tables.
|
|
42
40
|
|
|
43
|
-
For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner
|
|
41
|
+
For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
|
|
44
42
|
|
|
45
43
|
---
|
|
46
44
|
|
|
@@ -49,13 +47,13 @@ For full schema details, structure specifications, and dialect-specific JSON rep
|
|
|
49
47
|
Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
|
|
50
48
|
|
|
51
49
|
Follow the step-by-step setup guide in the official documentation:
|
|
52
|
-
👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner
|
|
50
|
+
👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
|
|
53
51
|
|
|
54
52
|
---
|
|
55
53
|
|
|
56
54
|
## Primary Workflow Phases
|
|
57
55
|
|
|
58
|
-
The agent enables you to craft an optimized context for QueryData API through three primary phases
|
|
56
|
+
The agent enables you to craft an optimized context for QueryData API through three primary phases. Depending on your needs, you may also toggle the agent to skip phases. For example, if you have a dataset already, you can skip directly to context optimization "optimize context using dataset in \<file\>."
|
|
59
57
|
|
|
60
58
|
### Phase 1: Artifact Ingestion
|
|
61
59
|
*Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
|
|
@@ -80,8 +78,7 @@ The optimization loop creates an initial `ContextSet` and then iteratively refin
|
|
|
80
78
|
1. **Bootstrap**: Generate an initial baseline context.
|
|
81
79
|
2. **Evaluate**: Measure context effectiveness against a golden dataset.
|
|
82
80
|
3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
|
|
83
|
-
4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
|
|
84
|
-
5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
|
|
81
|
+
4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality until we reach an optimal point.
|
|
85
82
|
|
|
86
83
|
*Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
|
|
87
84
|
|
|
@@ -7,6 +7,7 @@ src/google/cloud/db_context_enrichment/common/__init__.py
|
|
|
7
7
|
src/google/cloud/db_context_enrichment/common/config.py
|
|
8
8
|
src/google/cloud/db_context_enrichment/common/context_mutator.py
|
|
9
9
|
src/google/cloud/db_context_enrichment/common/context_store_client.py
|
|
10
|
+
src/google/cloud/db_context_enrichment/common/context_validator.py
|
|
10
11
|
src/google/cloud/db_context_enrichment/dataset/__init__.py
|
|
11
12
|
src/google/cloud/db_context_enrichment/dataset/dataset_generator.py
|
|
12
13
|
src/google/cloud/db_context_enrichment/evaluate/__init__.py
|
|
@@ -15,6 +16,9 @@ src/google/cloud/db_context_enrichment/evaluate/result_reader.py
|
|
|
15
16
|
src/google/cloud/db_context_enrichment/evaluate/db_generators/__init__.py
|
|
16
17
|
src/google/cloud/db_context_enrichment/evaluate/db_generators/alloydb.py
|
|
17
18
|
src/google/cloud/db_context_enrichment/evaluate/db_generators/base.py
|
|
19
|
+
src/google/cloud/db_context_enrichment/evaluate/db_generators/bigtable.py
|
|
20
|
+
src/google/cloud/db_context_enrichment/evaluate/db_generators/custom.py
|
|
21
|
+
src/google/cloud/db_context_enrichment/evaluate/db_generators/firestore.py
|
|
18
22
|
src/google/cloud/db_context_enrichment/evaluate/db_generators/mysql.py
|
|
19
23
|
src/google/cloud/db_context_enrichment/evaluate/db_generators/postgres.py
|
|
20
24
|
src/google/cloud/db_context_enrichment/evaluate/db_generators/spanner.py
|
google_cloud_db_context_engineering-0.7.2/src/google/cloud/db_context_enrichment/model/__init__.py
DELETED
|
File without changes
|
{google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/LICENSE
RENAMED
|
File without changes
|
{google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/setup.cfg
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|