google-cloud-db-context-engineering 0.7.2__tar.gz → 0.7.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {google_cloud_db_context_engineering-0.7.2/src/google_cloud_db_context_engineering.egg-info → google_cloud_db_context_engineering-0.7.4}/PKG-INFO +8 -11
  2. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/README.md +7 -10
  3. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/pyproject.toml +3 -3
  4. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/context_mutator.py +2 -2
  5. google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/common/context_validator.py +188 -0
  6. google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/__init__.py +15 -0
  7. google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/bigtable.py +44 -0
  8. google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/custom.py +104 -0
  9. google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/evaluate/db_generators/firestore.py +60 -0
  10. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/spanner.py +40 -2
  11. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/evaluate_generator.py +171 -7
  12. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/main.py +37 -6
  13. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4/src/google_cloud_db_context_engineering.egg-info}/PKG-INFO +8 -11
  14. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/SOURCES.txt +4 -0
  15. google_cloud_db_context_engineering-0.7.2/src/google/cloud/db_context_enrichment/model/__init__.py +0 -0
  16. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/LICENSE +0 -0
  17. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/setup.cfg +0 -0
  18. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/__init__.py +0 -0
  19. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/__init__.py +0 -0
  20. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/config.py +0 -0
  21. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/common/context_store_client.py +0 -0
  22. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/dataset/__init__.py +0 -0
  23. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/dataset/dataset_generator.py +0 -0
  24. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/__init__.py +0 -0
  25. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/alloydb.py +0 -0
  26. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/base.py +0 -0
  27. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/mysql.py +0 -0
  28. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/db_generators/postgres.py +0 -0
  29. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/evaluate/result_reader.py +0 -0
  30. {google_cloud_db_context_engineering-0.7.2/src/google/cloud/db_context_enrichment/evaluate/db_generators → google_cloud_db_context_engineering-0.7.4/src/google/cloud/db_context_enrichment/model}/__init__.py +0 -0
  31. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google/cloud/db_context_enrichment/model/context.py +0 -0
  32. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/dependency_links.txt +0 -0
  33. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/entry_points.txt +0 -0
  34. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/requires.txt +0 -0
  35. {google_cloud_db_context_engineering-0.7.2 → google_cloud_db_context_engineering-0.7.4}/src/google_cloud_db_context_engineering.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: google-cloud-db-context-engineering
3
- Version: 0.7.2
3
+ Version: 0.7.4
4
4
  Summary: A FastMCP server for generating natural language to SQL templates from database schemas.
5
5
  Requires-Python: >=3.12
6
6
  Description-Content-Type: text/markdown
@@ -16,19 +16,17 @@ Requires-Dist: pytest; extra == "test"
16
16
  Requires-Dist: pytest-asyncio; extra == "test"
17
17
  Dynamic: license-file
18
18
 
19
- This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
20
-
21
19
  # Context Engineering Agent
22
20
 
23
- The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL & Spanner Graph)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
21
+ The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL, Spanner Graph, and PostgreSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
24
22
 
25
23
  ---
26
24
 
27
25
  ## Why Context Engineering?
28
26
 
29
- When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL, pure GQL, or hybrid graph queries—is critical.
27
+ When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL (PostgreSQL, GoogleSQL, MySQL), pure GQL, or hybrid graph queries—is critical.
30
28
 
31
- As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
29
+ As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
32
30
 
33
31
  ---
34
32
 
@@ -40,7 +38,7 @@ A `ContextSet` is the central artifact generated and managed by the agent, conta
40
38
  * **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses, specialized join filters, or graph `MATCH` traversal patterns) linked to domain vocabulary.
41
39
  * **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or trigram search on relational and graph property tables.
42
40
 
43
- For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
41
+ For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
44
42
 
45
43
  ---
46
44
 
@@ -49,13 +47,13 @@ For full schema details, structure specifications, and dialect-specific JSON rep
49
47
  Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
50
48
 
51
49
  Follow the step-by-step setup guide in the official documentation:
52
- 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
50
+ 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
53
51
 
54
52
  ---
55
53
 
56
54
  ## Primary Workflow Phases
57
55
 
58
- The agent enables you to craft an optimized context for QueryData API through three primary phases:
56
+ The agent enables you to craft an optimized context for QueryData API through three primary phases. Depending on your needs, you may also toggle the agent to skip phases. For example, if you have a dataset already, you can skip directly to context optimization "optimize context using dataset in \<file\>."
59
57
 
60
58
  ### Phase 1: Artifact Ingestion
61
59
  *Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
@@ -80,8 +78,7 @@ The optimization loop creates an initial `ContextSet` and then iteratively refin
80
78
  1. **Bootstrap**: Generate an initial baseline context.
81
79
  2. **Evaluate**: Measure context effectiveness against a golden dataset.
82
80
  3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
83
- 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
84
- 5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
81
+ 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality until we reach an optimal point.
85
82
 
86
83
  *Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
87
84
 
@@ -1,16 +1,14 @@
1
- This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
2
-
3
1
  # Context Engineering Agent
4
2
 
5
- The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL & Spanner Graph)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
3
+ The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL, Spanner Graph, and PostgreSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
6
4
 
7
5
  ---
8
6
 
9
7
  ## Why Context Engineering?
10
8
 
11
- When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL, pure GQL, or hybrid graph queries—is critical.
9
+ When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL (PostgreSQL, GoogleSQL, MySQL), pure GQL, or hybrid graph queries—is critical.
12
10
 
13
- As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
11
+ As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
14
12
 
15
13
  ---
16
14
 
@@ -22,7 +20,7 @@ A `ContextSet` is the central artifact generated and managed by the agent, conta
22
20
  * **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses, specialized join filters, or graph `MATCH` traversal patterns) linked to domain vocabulary.
23
21
  * **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or trigram search on relational and graph property tables.
24
22
 
25
- For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
23
+ For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
26
24
 
27
25
  ---
28
26
 
@@ -31,13 +29,13 @@ For full schema details, structure specifications, and dialect-specific JSON rep
31
29
  Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
32
30
 
33
31
  Follow the step-by-step setup guide in the official documentation:
34
- 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
32
+ 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
35
33
 
36
34
  ---
37
35
 
38
36
  ## Primary Workflow Phases
39
37
 
40
- The agent enables you to craft an optimized context for QueryData API through three primary phases:
38
+ The agent enables you to craft an optimized context for QueryData API through three primary phases. Depending on your needs, you may also toggle the agent to skip phases. For example, if you have a dataset already, you can skip directly to context optimization "optimize context using dataset in \<file\>."
41
39
 
42
40
  ### Phase 1: Artifact Ingestion
43
41
  *Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
@@ -62,8 +60,7 @@ The optimization loop creates an initial `ContextSet` and then iteratively refin
62
60
  1. **Bootstrap**: Generate an initial baseline context.
63
61
  2. **Evaluate**: Measure context effectiveness against a golden dataset.
64
62
  3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
65
- 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
66
- 5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
63
+ 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality until we reach an optimal point.
67
64
 
68
65
  *Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
69
66
 
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "google-cloud-db-context-engineering"
3
- version = "0.7.2"
3
+ version = "0.7.4"
4
4
  description = "A FastMCP server for generating natural language to SQL templates from database schemas."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
@@ -45,8 +45,8 @@ dev = [
45
45
  ]
46
46
 
47
47
  [tool.db-context-engineering]
48
- toolbox_version = "1.4.0"
49
- evalbench_version = "1.12.0"
48
+ toolbox_version = "1.10.0"
49
+ evalbench_version = "1.17.0"
50
50
 
51
51
  [tool.ruff]
52
52
  line-length = 88
@@ -48,7 +48,7 @@ def mutate_context_set(file_path: str, mutations: list[Mutation]) -> None:
48
48
  context_set = context.ContextSet()
49
49
  else:
50
50
  try:
51
- with open(file_path) as f:
51
+ with open(file_path, encoding="utf-8") as f:
52
52
  raw_data = json.load(f)
53
53
  context_set = context.ContextSet.model_validate(raw_data)
54
54
  except ValidationError as e:
@@ -126,7 +126,7 @@ def mutate_context_set(file_path: str, mutations: list[Mutation]) -> None:
126
126
  # 4. Save validated ContextSet
127
127
  try:
128
128
  os.makedirs(os.path.dirname(os.path.abspath(file_path)), exist_ok=True)
129
- with open(file_path, "w") as f:
129
+ with open(file_path, "w", encoding="utf-8") as f:
130
130
  f.write(context_set.model_dump_json(indent=2, exclude_none=True))
131
131
  except OSError as e:
132
132
  raise RuntimeError(f"Error saving ContextSet to {file_path}: {e}") from e
@@ -0,0 +1,188 @@
1
+ import json
2
+ from typing import Any
3
+
4
+ from pydantic import ValidationError
5
+
6
+ from google.cloud.db_context_enrichment.model import context
7
+
8
+ _ATTR_TO_TYPE = {
9
+ "templates": "template",
10
+ "facets": "facet",
11
+ "value_searches": "value_search",
12
+ }
13
+
14
+ # ContextSet accepts camelCase aliases (Context Store returns them on download)
15
+ # and the deprecated "fragments" name. Map them back to the canonical attribute
16
+ # so every check below only has to handle one spelling.
17
+ _ALIAS_TO_ATTR = {
18
+ "valueSearches": "value_searches",
19
+ "fragments": "facets",
20
+ }
21
+
22
+
23
+ def validate_context_set(file_path: str) -> dict[str, Any]:
24
+ """Validate a ContextSet file and return a structured report of issues.
25
+
26
+ Always returns a dict of the shape:
27
+ {
28
+ "valid": bool,
29
+ "issues": [
30
+ {
31
+ "location": {"type": str, "index": int} | None,
32
+ "message": str,
33
+ },
34
+ ...
35
+ ],
36
+ }
37
+
38
+ File access failures (missing path, permission denied, path is a directory,
39
+ etc.) are surfaced as a single issue rather than raised.
40
+ """
41
+ try:
42
+ with open(file_path, encoding="utf-8") as f:
43
+ text = f.read()
44
+ except OSError as e:
45
+ return {
46
+ "valid": False,
47
+ "issues": [
48
+ _make_issue(f"Could not read file {file_path}: {type(e).__name__}: {e}")
49
+ ],
50
+ }
51
+
52
+ if text.strip() == "":
53
+ return {"valid": True, "issues": []}
54
+
55
+ try:
56
+ raw = json.loads(text)
57
+ except json.JSONDecodeError as e:
58
+ snippet = e.doc[max(0, e.pos - 30) : e.pos + 30]
59
+ return {
60
+ "valid": False,
61
+ "issues": [_make_issue(f"File is not valid JSON: {e}. Near: {snippet!r}")],
62
+ }
63
+
64
+ if not isinstance(raw, dict):
65
+ return {
66
+ "valid": False,
67
+ "issues": [_make_issue("Top-level value must be a JSON object")],
68
+ }
69
+
70
+ raw = _normalize_aliases(raw)
71
+
72
+ issues: list[dict[str, Any]] = []
73
+ issues.extend(_check_pydantic(raw))
74
+ issues.extend(_check_duplicates(raw))
75
+ issues.extend(_check_value_search_value_param(raw))
76
+
77
+ return {"valid": len(issues) == 0, "issues": issues}
78
+
79
+
80
+ def _make_issue(message: str, location: dict[str, Any] | None = None) -> dict[str, Any]:
81
+ return {"location": location, "message": message}
82
+
83
+
84
+ def _normalize_aliases(raw: dict[str, Any]) -> dict[str, Any]:
85
+ """Rewrite alias keys to their canonical ContextSet attribute names."""
86
+ return {_ALIAS_TO_ATTR.get(key, key): value for key, value in raw.items()}
87
+
88
+
89
+ def _check_pydantic(raw: dict[str, Any]) -> list[dict[str, Any]]:
90
+ issues: list[dict[str, Any]] = []
91
+ try:
92
+ context.ContextSet.model_validate(raw)
93
+ return issues
94
+ except ValidationError as e:
95
+ for err in e.errors():
96
+ loc = err.get("loc", ())
97
+ msg = err.get("msg", "validation error")
98
+ location: dict[str, Any] | None = None
99
+ descriptor = ""
100
+
101
+ if len(loc) >= 1 and loc[0] in _ATTR_TO_TYPE:
102
+ item_type = _ATTR_TO_TYPE[loc[0]]
103
+ if len(loc) >= 2 and isinstance(loc[1], int):
104
+ item_index = loc[1]
105
+ location = {"type": item_type, "index": item_index}
106
+ items = raw.get(loc[0])
107
+ if isinstance(items, list) and 0 <= item_index < len(items):
108
+ descriptor = _describe(item_type, items[item_index])
109
+
110
+ field_path = ".".join(str(p) for p in loc) if loc else ""
111
+ parts = [msg]
112
+ if field_path:
113
+ parts.append(f"at {field_path}")
114
+ if descriptor:
115
+ parts.append(descriptor)
116
+ full_msg = parts[0] + (
117
+ " (" + "; ".join(parts[1:]) + ")" if len(parts) > 1 else ""
118
+ )
119
+
120
+ issues.append(_make_issue(full_msg, location=location))
121
+ return issues
122
+
123
+
124
+ def _check_duplicates(raw: dict[str, Any]) -> list[dict[str, Any]]:
125
+ issues: list[dict[str, Any]] = []
126
+ for attr, item_type in _ATTR_TO_TYPE.items():
127
+ items = raw.get(attr)
128
+ if not isinstance(items, list):
129
+ continue
130
+ seen: dict[str, int] = {}
131
+ for idx, item in enumerate(items):
132
+ try:
133
+ key = _canonical(item)
134
+ except (TypeError, ValueError):
135
+ continue
136
+ if key in seen:
137
+ descriptor = _describe(item_type, item)
138
+ suffix = f" ({descriptor})" if descriptor else ""
139
+ issues.append(
140
+ _make_issue(
141
+ f"Exact duplicate of {item_type} at index {seen[key]}{suffix}",
142
+ location={"type": item_type, "index": idx},
143
+ )
144
+ )
145
+ else:
146
+ seen[key] = idx
147
+ return issues
148
+
149
+
150
+ def _check_value_search_value_param(raw: dict[str, Any]) -> list[dict[str, Any]]:
151
+ issues: list[dict[str, Any]] = []
152
+ items = raw.get("value_searches")
153
+ if not isinstance(items, list):
154
+ return issues
155
+ for idx, item in enumerate(items):
156
+ if not isinstance(item, dict):
157
+ continue
158
+ query = item.get("query")
159
+ if isinstance(query, str) and "$value" not in query:
160
+ descriptor = _describe("value_search", item)
161
+ suffix = f" ({descriptor})" if descriptor else ""
162
+ issues.append(
163
+ _make_issue(
164
+ f"value_search query must reference the $value parameter{suffix}",
165
+ location={"type": "value_search", "index": idx},
166
+ )
167
+ )
168
+ return issues
169
+
170
+
171
+ def _describe(item_type: str, item: Any) -> str:
172
+ """Short human-readable identifier for an item, e.g. "intent: 'active users'"."""
173
+ if not isinstance(item, dict):
174
+ return ""
175
+ if item_type == "template":
176
+ nl = item.get("nl_query")
177
+ return f"nl_query: {nl!r}" if nl else ""
178
+ if item_type == "facet":
179
+ intent = item.get("intent")
180
+ return f"intent: {intent!r}" if intent else ""
181
+ if item_type == "value_search":
182
+ ct = item.get("concept_type")
183
+ return f"concept_type: {ct!r}" if ct else ""
184
+ return ""
185
+
186
+
187
+ def _canonical(obj: Any) -> str:
188
+ return json.dumps(obj, sort_keys=True, ensure_ascii=False)
@@ -0,0 +1,15 @@
1
+ from .alloydb import AlloyDBConfigGenerator
2
+ from .base import BaseDBConfigGenerator
3
+ from .custom import CustomDBConfigGenerator
4
+ from .mysql import MySQLConfigGenerator
5
+ from .postgres import PostgresConfigGenerator
6
+ from .spanner import SpannerConfigGenerator
7
+
8
+ __all__ = [
9
+ "BaseDBConfigGenerator",
10
+ "AlloyDBConfigGenerator",
11
+ "CustomDBConfigGenerator",
12
+ "MySQLConfigGenerator",
13
+ "PostgresConfigGenerator",
14
+ "SpannerConfigGenerator",
15
+ ]
@@ -0,0 +1,44 @@
1
+ from typing import Any
2
+
3
+ import yaml
4
+
5
+ from .base import BaseDBConfigGenerator
6
+
7
+
8
+ class BigtableConfigGenerator(BaseDBConfigGenerator):
9
+ """
10
+ Dedicated generator mapping properties to explicit Bigtable configuration
11
+ topologies. Bypasses the GDA Python SDK to construct the model configuration
12
+ directly, as the SDK may lack native BigtableReference definitions.
13
+ """
14
+
15
+ SOURCE_TYPE = "bigtable"
16
+ DIALECT = "bigtable"
17
+ REQUIRED_FIELDS = {"project", "instance"}
18
+
19
+ def generate_db_config(self) -> str:
20
+ db_config = {
21
+ "db_type": "bigtable",
22
+ "dialect": self.DIALECT,
23
+ "database_name": self.params.get("instance"),
24
+ "database_path": f"projects/{self.params.get('project')}/instances/{self.params.get('instance')}",
25
+ "instance_id": self.params.get("instance"),
26
+ "gcp_project_id": self.params.get("project"),
27
+ "max_executions_per_minute": 100,
28
+ }
29
+ return yaml.safe_dump(
30
+ db_config, sort_keys=False, default_flow_style=False
31
+ ).strip()
32
+
33
+ def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
34
+ return {
35
+ "bigtable_reference": {
36
+ "database_reference": {
37
+ "project_id": self.params.get("project"),
38
+ "instance_id": self.params.get("instance"),
39
+ },
40
+ "agent_context_reference": {
41
+ "context_set_id": context_set_id,
42
+ },
43
+ }
44
+ }
@@ -0,0 +1,104 @@
1
+ import os
2
+ from typing import Any
3
+
4
+ import yaml
5
+
6
+ from .base import BaseDBConfigGenerator
7
+
8
+
9
+ class CustomDBConfigGenerator(BaseDBConfigGenerator):
10
+ """
11
+ Pluggable generator mapping properties to custom evaluation configurations.
12
+ Enables arbitrary third-party or proprietary internal engines to be plugged
13
+ into the evaluation framework via dynamic connector and generator SPIs
14
+ without exposing engine-specific code or configuration schemas.
15
+ """
16
+
17
+ SOURCE_TYPE = "custom"
18
+ DIALECT = "sql"
19
+ REQUIRED_FIELDS = set()
20
+
21
+ def __init__(self, params: dict[str, Any]):
22
+ self.params = params
23
+ if "project" not in self.params:
24
+ project_id = os.environ.get("GOOGLE_CLOUD_PROJECT") or os.environ.get(
25
+ "GCP_PROJECT"
26
+ )
27
+ if project_id:
28
+ self.params["project"] = project_id
29
+ self.connector_class = params.get("connector_class", "")
30
+ self.generator_class = params.get("generator_class", "")
31
+ self.DIALECT = params.get("dialect", self.DIALECT)
32
+ self.validate()
33
+
34
+ def validate(self) -> None:
35
+ """
36
+ Validates that either connector_class or generator_class is specified.
37
+ """
38
+ if not self.connector_class and not self.generator_class:
39
+ raise ValueError(
40
+ "Custom source configuration must specify at least 'connector_class' or 'generator_class'."
41
+ )
42
+
43
+ def generate_db_config(self) -> str:
44
+ """
45
+ Generates the db_config.yaml payload for custom connectors.
46
+ """
47
+ db_config: dict[str, Any] = {
48
+ "db_type": self.params.get("db_type", "custom"),
49
+ "dialect": self.DIALECT,
50
+ }
51
+ if self.connector_class:
52
+ db_config["connector_class"] = self.connector_class
53
+
54
+ # Forward all non-meta parameters from tools.yaml source block
55
+ excluded_keys = {
56
+ "kind",
57
+ "name",
58
+ "type",
59
+ "connector_class",
60
+ "generator_class",
61
+ "dialect",
62
+ "db_type",
63
+ }
64
+ for key, value in self.params.items():
65
+ if key not in excluded_keys:
66
+ db_config[key] = value
67
+
68
+ return yaml.safe_dump(
69
+ db_config, sort_keys=False, default_flow_style=False
70
+ ).strip()
71
+
72
+ def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
73
+ """
74
+ Datasource reference dictionary. Custom engines manage their own
75
+ schema references or context binding.
76
+ """
77
+ return {}
78
+
79
+ def generate_model_config(self, context_set_id: str) -> str:
80
+ """
81
+ Generates model_config.yaml. If generator_class is specified, produces
82
+ a custom generator configuration for dynamic instantiation.
83
+ """
84
+ if self.generator_class:
85
+ model_config: dict[str, Any] = {
86
+ "generator": "custom",
87
+ "generator_class": self.generator_class,
88
+ "context_set_id": context_set_id,
89
+ }
90
+ excluded_keys = {
91
+ "kind",
92
+ "name",
93
+ "type",
94
+ "connector_class",
95
+ "generator_class",
96
+ }
97
+ for key, value in self.params.items():
98
+ if key not in excluded_keys:
99
+ model_config[key] = value
100
+ return yaml.safe_dump(
101
+ model_config, sort_keys=False, default_flow_style=False
102
+ ).strip()
103
+
104
+ return super().generate_model_config(context_set_id)
@@ -0,0 +1,60 @@
1
+ from typing import Any
2
+
3
+ import yaml
4
+
5
+ from .base import BaseDBConfigGenerator
6
+
7
+
8
+ class FirestoreConfigGenerator(BaseDBConfigGenerator):
9
+ """
10
+ Dedicated generator mapping properties to Firestore (MongoDB query dialect)
11
+ topologies utilized by both EvalBench binaries and GDA REST API.
12
+ """
13
+
14
+ SOURCE_TYPE = "firestore"
15
+ DIALECT = "mongodb"
16
+ REQUIRED_FIELDS = BaseDBConfigGenerator.REQUIRED_FIELDS | {
17
+ "project",
18
+ "database",
19
+ }
20
+
21
+ def __init__(self, params: dict[str, Any]):
22
+ super().__init__(params)
23
+ self.project = params.get("project")
24
+ self.database = params.get("database")
25
+ self.connection_string = params.get("connection_string")
26
+ self.collection_ids = params.get("collection_ids") or params.get("table_ids")
27
+
28
+ def generate_db_config(self) -> str:
29
+ db_type = "mongodb"
30
+
31
+ db_config = {
32
+ "db_type": db_type,
33
+ "dialect": self.DIALECT,
34
+ "database_name": self.database,
35
+ "database_path": "",
36
+ "firestore_database": f"projects/{self.project}/databases/{self.database}",
37
+ "max_executions_per_minute": 120,
38
+ }
39
+ if self.connection_string:
40
+ db_config["connection_string"] = self.connection_string
41
+
42
+ return yaml.safe_dump(
43
+ db_config, sort_keys=False, default_flow_style=False
44
+ ).strip()
45
+
46
+ def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
47
+ db_ref: dict[str, Any] = {
48
+ "project_id": self.project,
49
+ "database_id": self.database,
50
+ }
51
+ if self.collection_ids:
52
+ db_ref["collection_ids"] = self.collection_ids
53
+
54
+ ref: dict[str, Any] = {"firestore_reference": {"database_reference": db_ref}}
55
+ if context_set_id:
56
+ ref["firestore_reference"]["agent_context_reference"] = {
57
+ "context_set_id": context_set_id
58
+ }
59
+
60
+ return ref
@@ -9,10 +9,10 @@ class SpannerConfigGenerator(BaseDBConfigGenerator):
9
9
  """
10
10
  Dedicated generator mapping properties to explicit Spanner configuration
11
11
  topologies utilized by both EvalBench binaries and GDA Context objects.
12
+ Supports both GoogleSQL and PostgreSQL dialects.
12
13
  """
13
14
 
14
15
  SOURCE_TYPE = "spanner"
15
- DIALECT = "spanner_gsql"
16
16
  REQUIRED_FIELDS = BaseDBConfigGenerator.REQUIRED_FIELDS | {
17
17
  "project",
18
18
  "instance",
@@ -25,6 +25,40 @@ class SpannerConfigGenerator(BaseDBConfigGenerator):
25
25
  self.instance = params.get("instance")
26
26
  self.database = params.get("database")
27
27
 
28
+ raw_dialect = (
29
+ params.get("dialect")
30
+ or params.get("engine")
31
+ or params.get("database_dialect")
32
+ )
33
+ if not raw_dialect and params.get("type") in ("spanner-postgres", "spanner-pg"):
34
+ raw_dialect = "POSTGRESQL"
35
+
36
+ if raw_dialect:
37
+ normalized = str(raw_dialect).strip().lower().replace("-", "_")
38
+ if normalized in (
39
+ "postgresql",
40
+ "postgres",
41
+ "spanner_pg",
42
+ "pg",
43
+ "spanner_postgres",
44
+ ):
45
+ self.engine = "POSTGRESQL"
46
+ self._dialect = "spanner_pg"
47
+ elif normalized in ("google_sql", "googlesql", "spanner_gsql", "gsql"):
48
+ self.engine = "GOOGLE_SQL"
49
+ self._dialect = "spanner_gsql"
50
+ else:
51
+ raise ValueError(
52
+ f"Unsupported Spanner dialect/engine: '{raw_dialect}'. Must be 'GOOGLE_SQL' or 'POSTGRESQL'."
53
+ )
54
+ else:
55
+ self.engine = "GOOGLE_SQL"
56
+ self._dialect = "spanner_gsql"
57
+
58
+ @property
59
+ def DIALECT(self) -> str:
60
+ return self._dialect
61
+
28
62
  def generate_db_config(self) -> str:
29
63
  db_type = "spanner"
30
64
  db_path = f"projects/{self.project}/instances/{self.instance}/databases/{self.database}"
@@ -44,12 +78,16 @@ class SpannerConfigGenerator(BaseDBConfigGenerator):
44
78
 
45
79
  def build_datasource_reference(self, context_set_id: str) -> dict[str, Any]:
46
80
  database_ref: dict[str, Any] = {
47
- "engine": "GOOGLE_SQL",
81
+ "engine": self.engine,
48
82
  "project_id": self.project,
49
83
  "instance_id": self.instance,
50
84
  "database_id": self.database,
51
85
  }
52
86
  if graph_ids := self.params.get("graph_ids"):
87
+ if self.engine == "POSTGRESQL":
88
+ raise ValueError(
89
+ "graph_ids is not supported for Spanner PostgreSQL dialect"
90
+ )
53
91
  if not isinstance(graph_ids, list) or not all(
54
92
  isinstance(g, str) for g in graph_ids
55
93
  ):
@@ -1,7 +1,11 @@
1
+ import importlib
1
2
  import json
3
+ import logging
2
4
  import os
3
5
  import re
6
+ import sys
4
7
  import textwrap
8
+ import threading
5
9
  from typing import Any
6
10
 
7
11
  import yaml
@@ -10,10 +14,15 @@ from google.cloud.db_context_enrichment.common import config
10
14
 
11
15
  from .db_generators.alloydb import AlloyDBConfigGenerator
12
16
  from .db_generators.base import BaseDBConfigGenerator
17
+ from .db_generators.bigtable import BigtableConfigGenerator
18
+ from .db_generators.custom import CustomDBConfigGenerator
19
+ from .db_generators.firestore import FirestoreConfigGenerator
13
20
  from .db_generators.mysql import MySQLConfigGenerator
14
21
  from .db_generators.postgres import PostgresConfigGenerator
15
22
  from .db_generators.spanner import SpannerConfigGenerator
16
23
 
24
+ logger = logging.getLogger(__name__)
25
+
17
26
  # Constants for EvalBench configuration filenames
18
27
  DB_CONFIG_NAME = "db_config.yaml"
19
28
  MODEL_CONFIG_NAME = "model_config.yaml"
@@ -76,10 +85,10 @@ def _extract_toolbox_params(
76
85
  with open(toolbox_config_path) as f:
77
86
  content = f.read()
78
87
  interpolated = _interpolate_env_vars(content)
79
- docs = yaml.safe_load_all(interpolated)
88
+ docs = [doc for doc in yaml.safe_load_all(interpolated) if doc]
89
+
90
+ source_doc = None
80
91
  for doc in docs:
81
- if not doc:
82
- continue
83
92
  if (
84
93
  doc.get("kind") == "source"
85
94
  and doc.get("name") == toolbox_source_name
@@ -88,11 +97,24 @@ def _extract_toolbox_params(
88
97
  raise ValueError(
89
98
  f"Selected source '{toolbox_source_name}' is missing the 'type' field."
90
99
  )
91
- return doc
100
+ source_doc = doc
101
+ break
92
102
 
93
- raise ValueError(
94
- f"Could not find a 'kind: source' named '{toolbox_source_name}' in {toolbox_config_path}"
95
- )
103
+ if not source_doc:
104
+ raise ValueError(
105
+ f"Could not find a 'kind: source' named '{toolbox_source_name}' in {toolbox_config_path}"
106
+ )
107
+
108
+ # For Spanner sources, state.md is the authoritative single source of truth for graph_ids
109
+ # (QueryData API requires explicit graph_ids in model_config.yaml, whereas tools.yaml
110
+ # only configures MCP Toolbox runtime tools and parameters).
111
+ if source_doc.get("type") == "spanner":
112
+ state_md_dir = os.path.dirname(toolbox_config_path)
113
+ state_md_path = os.path.join(state_md_dir, "state.md")
114
+ if graph_ids := _parse_graph_ids_from_state_md(state_md_path):
115
+ source_doc["graph_ids"] = graph_ids
116
+
117
+ return source_doc
96
118
 
97
119
  except FileNotFoundError:
98
120
  raise ValueError(f"Config file not found: {toolbox_config_path}")
@@ -104,6 +126,72 @@ def _extract_toolbox_params(
104
126
  raise ValueError(f"Failed to parse {toolbox_config_path} as YAML: {e}")
105
127
 
106
128
 
129
+ def _parse_graph_ids_from_state_md(state_md_path: str) -> list[str] | None:
130
+ """Parses graph_ids from state.md.
131
+
132
+ state.md is the authoritative source of truth for the database and graph scope
133
+ because QueryData API requires explicit graph_ids in model_config.yaml to evaluate
134
+ property graphs, whereas tools.yaml only configures MCP Toolbox runtime tools.
135
+ """
136
+ if not os.path.exists(state_md_path):
137
+ return None
138
+ with open(state_md_path, encoding="utf-8") as f:
139
+ content = f.read()
140
+ match = re.search(
141
+ r"(?:^[ \t]*[-*][ \t]*)?\*\*Graph\s+Ids?:?\*\*:?[ \t]*([^\n]*)",
142
+ content,
143
+ re.MULTILINE | re.IGNORECASE,
144
+ )
145
+ if not match:
146
+ return None
147
+
148
+ val_str = match.group(1).split("#")[0].strip()
149
+
150
+ # If empty on the same line, check for multiline sub-bullets
151
+ if not val_str:
152
+ after_match = content[match.end() :]
153
+ bullet_items = []
154
+ for line in after_match.splitlines():
155
+ line_stripped = line.strip()
156
+ if not line_stripped:
157
+ continue
158
+ if line_stripped.startswith(("-", "*")) and not re.match(
159
+ r"^[-*]\s*\*\*", line_stripped
160
+ ):
161
+ item = line_stripped.lstrip("-* ").split("#")[0].strip().strip("'\"`")
162
+ if item and item.lower() not in (
163
+ "none",
164
+ "n/a",
165
+ "null",
166
+ "nil",
167
+ "-",
168
+ ):
169
+ bullet_items.append(item)
170
+ elif (
171
+ line_stripped.startswith("#")
172
+ or line_stripped.startswith("- **")
173
+ or line_stripped.startswith("* **")
174
+ ):
175
+ break
176
+ else:
177
+ break
178
+ return bullet_items if bullet_items else None
179
+
180
+ # Handle explicit empty / none indicators
181
+ if val_str.lower() in ("none", "n/a", "null", "nil", "-", "[]", ""):
182
+ return None
183
+
184
+ if val_str.startswith("[") and val_str.endswith("]"):
185
+ val_str = val_str[1:-1]
186
+ graphs = [
187
+ g.strip().strip("'\"`")
188
+ for g in val_str.split(",")
189
+ if g.strip().strip("'\"`")
190
+ and g.strip().strip("'\"`").lower() not in ("none", "n/a", "null", "nil", "-")
191
+ ]
192
+ return graphs if graphs else None
193
+
194
+
107
195
  def _interpolate_env_vars(raw_yaml: str) -> str:
108
196
  """Replaces ${ENV_NAME} or ${ENV_NAME:default_value} with environment variables."""
109
197
  # Matches ${VAR_NAME} or ${VAR_NAME:fallback}
@@ -124,17 +212,93 @@ def _interpolate_env_vars(raw_yaml: str) -> str:
124
212
  return pattern.sub(replacer, raw_yaml)
125
213
 
126
214
 
215
+ _SYS_PATH_LOCK = threading.Lock()
216
+
217
+
218
+ def _import_with_cwd_fallback(mod_name: str):
219
+ """Imports mod_name, falling back to appending os.getcwd() to sys.path."""
220
+ try:
221
+ return importlib.import_module(mod_name)
222
+ except ModuleNotFoundError as e:
223
+ if e.name is None or not (
224
+ mod_name == e.name or mod_name.startswith(e.name + ".")
225
+ ):
226
+ raise
227
+ cwd = os.path.abspath(os.getcwd())
228
+ with _SYS_PATH_LOCK:
229
+ if not any(os.path.abspath(p or cwd) == cwd for p in sys.path):
230
+ # Keep cwd at the end of sys.path so lazy runtime imports work
231
+ # without shadowing standard library or installed packages.
232
+ sys.path.append(cwd)
233
+ importlib.invalidate_caches()
234
+ return importlib.import_module(mod_name)
235
+
236
+
127
237
  def _get_db_generator(params: dict[str, Any]) -> BaseDBConfigGenerator:
128
238
  """Factory function to build the correct Evaluation Generator."""
129
239
  source_type = params.get("type", "").lower()
130
240
 
131
241
  generators = {
132
242
  AlloyDBConfigGenerator.SOURCE_TYPE: AlloyDBConfigGenerator,
243
+ BigtableConfigGenerator.SOURCE_TYPE: BigtableConfigGenerator,
133
244
  PostgresConfigGenerator.SOURCE_TYPE: PostgresConfigGenerator,
134
245
  MySQLConfigGenerator.SOURCE_TYPE: MySQLConfigGenerator,
135
246
  SpannerConfigGenerator.SOURCE_TYPE: SpannerConfigGenerator,
247
+ "spanner-postgres": SpannerConfigGenerator,
248
+ "spanner-pg": SpannerConfigGenerator,
249
+ FirestoreConfigGenerator.SOURCE_TYPE: FirestoreConfigGenerator,
250
+ CustomDBConfigGenerator.SOURCE_TYPE: CustomDBConfigGenerator,
136
251
  }
137
252
 
253
+ # Dynamically register external custom database configuration generators.
254
+ # AUTOCTX_CUSTOM_GENERATORS holds a dotted Python module import path
255
+ # (e.g., "my_package.custom_generators") that exposes a
256
+ # CUSTOM_GENERATORS: dict[str, type[BaseDBConfigGenerator]] mapping
257
+ # custom tools.yaml source types to BaseDBConfigGenerator subclasses.
258
+ custom_gens = {}
259
+ custom_plugin = os.environ.get("AUTOCTX_CUSTOM_GENERATORS")
260
+ if custom_plugin:
261
+ try:
262
+ mod = _import_with_cwd_fallback(custom_plugin)
263
+ custom_gens = getattr(mod, "CUSTOM_GENERATORS", None)
264
+ if custom_gens is None:
265
+ raise AttributeError(
266
+ f"Custom generator module '{custom_plugin}' must define 'CUSTOM_GENERATORS' dict."
267
+ )
268
+ if not isinstance(custom_gens, dict):
269
+ raise TypeError(
270
+ f"CUSTOM_GENERATORS in '{custom_plugin}' must be a dictionary, "
271
+ f"got {type(custom_gens).__name__}."
272
+ )
273
+ generators.update(custom_gens)
274
+ except Exception as e:
275
+ logger.error(
276
+ "Failed to load custom generators from plugin module '%s': %s",
277
+ custom_plugin,
278
+ e,
279
+ )
280
+ raise RuntimeError(
281
+ f"Failed to load custom generators from plugin module '{custom_plugin}': {e}"
282
+ ) from e
283
+
284
+ # If the source type is not handled by an external AUTOCTX_CUSTOM_GENERATORS
285
+ # plugin, allow inline connector_class / generator_class in tools.yaml to
286
+ # route directly to CustomDBConfigGenerator.
287
+ if source_type not in custom_gens and (
288
+ "connector_class" in params or "generator_class" in params
289
+ ):
290
+ if (
291
+ source_type in generators
292
+ and source_type != CustomDBConfigGenerator.SOURCE_TYPE
293
+ ):
294
+ logger.warning(
295
+ "Source type '%s' is being overridden by custom connector_class '%s' / generator_class '%s'.",
296
+ source_type,
297
+ params.get("connector_class"),
298
+ params.get("generator_class"),
299
+ )
300
+ return CustomDBConfigGenerator(params)
301
+
138
302
  if source_type not in generators:
139
303
  supported = ", ".join(generators.keys())
140
304
  raise ValueError(
@@ -6,6 +6,7 @@ from fastmcp import FastMCP
6
6
  from google.cloud.db_context_enrichment.common import (
7
7
  context_mutator,
8
8
  context_store_client,
9
+ context_validator,
9
10
  )
10
11
  from google.cloud.db_context_enrichment.dataset import dataset_generator
11
12
  from google.cloud.db_context_enrichment.evaluate import (
@@ -73,7 +74,7 @@ def generate_evalbench_configs(
73
74
  dataset_path: The absolute path to the golden dataset file in the simplified user-facing format (JSON list of objects with keys: "id", "database", "nlq", "golden_sql").
74
75
  context_set_id: Full ContextSet resource name to evaluate against.
75
76
  toolbox_config_path: The absolute path to the tools.yaml configuration file.
76
- toolbox_source_name: The name of the database source to use inside tools.yaml. The underlying source block must use a supported 'type' (cloud-sql-postgres, cloud-sql-mysql, spanner, alloydb-postgres).
77
+ toolbox_source_name: The name of the database source to use inside tools.yaml. The underlying source block must use a supported 'type' (cloud-sql-postgres, cloud-sql-mysql, spanner, alloydb-postgres, or custom).
77
78
 
78
79
  Returns:
79
80
  A message indicating that the configuration files were successfully created.
@@ -102,13 +103,14 @@ def generate_upload_url(
102
103
 
103
104
  Args:
104
105
  db_engine: The database engine. Accepted values are 'alloydb',
105
- 'cloudsql', or 'spanner'. This can be derived from the 'kind'
106
- field in the tools.yaml file. For example, 'alloydb-postgres'
107
- becomes 'alloydb', and 'cloud-sql-postgres' becomes 'cloudsql'.
106
+ 'cloudsql', 'spanner', or 'bigtable'. This can be derived from
107
+ the 'kind' field in the tools.yaml file. For example,
108
+ 'alloydb-postgres' becomes 'alloydb', 'cloud-sql-postgres'
109
+ becomes 'cloudsql', and 'bigtable' becomes 'bigtable'.
108
110
  project_id: The Google Cloud project ID.
109
111
  location: The location of the AlloyDB cluster.
110
112
  cluster_id: The ID of the AlloyDB cluster.
111
- instance_id: The ID of the Cloud SQL or Spanner instance.
113
+ instance_id: The ID of the Cloud SQL, Spanner, or Bigtable instance.
112
114
  database_id: The ID of the Spanner database.
113
115
 
114
116
  Returns:
@@ -129,8 +131,13 @@ def generate_upload_url(
129
131
  return f"https://console.cloud.google.com/spanner/instances/{instance_id}/databases/{database_id}/details/query?project={project_id}"
130
132
  else:
131
133
  return "Error: Missing instance_id, database_id, or project_id for spanner."
134
+ elif db_engine == "bigtable":
135
+ if instance_id and project_id:
136
+ return f"https://console.cloud.google.com/bigtable/instances/{instance_id}/overview?project={project_id}"
137
+ else:
138
+ return "Error: Missing instance_id or project_id for bigtable."
132
139
  else:
133
- return "Error: Invalid db_engine. Must be one of 'alloydb', 'cloudsql', or 'spanner'."
140
+ return "Error: Invalid db_engine. Must be one of 'alloydb', 'cloudsql', 'spanner', or 'bigtable'."
134
141
 
135
142
 
136
143
  # NOTE: `@mcp.tool` is intentionally NOT applied to upload_context_set /
@@ -258,6 +265,30 @@ def mutate_context_set(
258
265
  return f"Error applying mutations: {str(e)}"
259
266
 
260
267
 
268
+ @mcp.tool
269
+ def validate_context_set(file_path: str) -> str:
270
+ """
271
+ Validate a ContextSet JSON file for structural and convention issues. Reports issues only — does not fix them. The caller (agent) is expected to apply fixes via `mutate_context_set`, then re-run validation until `valid` is true.
272
+
273
+ Args:
274
+ file_path: Absolute path to the ContextSet file.
275
+
276
+ Returns:
277
+ A JSON string of the shape:
278
+ {
279
+ "valid": bool,
280
+ "issues": [
281
+ {
282
+ "location": {"type": "template" | "facet" | "value_search", "index": int} | null,
283
+ "message": str
284
+ },
285
+ ...
286
+ ]
287
+ }
288
+ """
289
+ return json.dumps(context_validator.validate_context_set(file_path), indent=2)
290
+
291
+
261
292
  @mcp.tool
262
293
  async def read_evaluation_result(
263
294
  run_folder_path: str, offset: int = 0, batch_size: int = 10
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: google-cloud-db-context-engineering
3
- Version: 0.7.2
3
+ Version: 0.7.4
4
4
  Summary: A FastMCP server for generating natural language to SQL templates from database schemas.
5
5
  Requires-Python: >=3.12
6
6
  Description-Content-Type: text/markdown
@@ -16,19 +16,17 @@ Requires-Dist: pytest; extra == "test"
16
16
  Requires-Dist: pytest-asyncio; extra == "test"
17
17
  Dynamic: license-file
18
18
 
19
- This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
20
-
21
19
  # Context Engineering Agent
22
20
 
23
- The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL & Spanner Graph)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
21
+ The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, supporting both **relational SQL** and **Graph Query Language (GQL)** across [AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview), Cloud SQL ([PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview)), and [Cloud Spanner (GoogleSQL, Spanner Graph, and PostgreSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview).
24
22
 
25
23
  ---
26
24
 
27
25
  ## Why Context Engineering?
28
26
 
29
- When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL, pure GQL, or hybrid graph queries—is critical.
27
+ When building data agents and natural language analytics interfaces, accurately translating user intent into database queries—whether relational SQL (PostgreSQL, GoogleSQL, MySQL), pure GQL, or hybrid graph queries—is critical.
30
28
 
31
- As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
29
+ As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL/GQL translation accuracy with low latency**.
32
30
 
33
31
  ---
34
32
 
@@ -40,7 +38,7 @@ A `ContextSet` is the central artifact generated and managed by the agent, conta
40
38
  * **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses, specialized join filters, or graph `MATCH` traversal patterns) linked to domain vocabulary.
41
39
  * **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or trigram search on relational and graph property tables.
42
40
 
43
- For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
41
+ For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
44
42
 
45
43
  ---
46
44
 
@@ -49,13 +47,13 @@ For full schema details, structure specifications, and dialect-specific JSON rep
49
47
  Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
50
48
 
51
49
  Follow the step-by-step setup guide in the official documentation:
52
- 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
50
+ 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
53
51
 
54
52
  ---
55
53
 
56
54
  ## Primary Workflow Phases
57
55
 
58
- The agent enables you to craft an optimized context for QueryData API through three primary phases:
56
+ The agent enables you to craft an optimized context for QueryData API through three primary phases. Depending on your needs, you may also toggle the agent to skip phases. For example, if you have a dataset already, you can skip directly to context optimization "optimize context using dataset in \<file\>."
59
57
 
60
58
  ### Phase 1: Artifact Ingestion
61
59
  *Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
@@ -80,8 +78,7 @@ The optimization loop creates an initial `ContextSet` and then iteratively refin
80
78
  1. **Bootstrap**: Generate an initial baseline context.
81
79
  2. **Evaluate**: Measure context effectiveness against a golden dataset.
82
80
  3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
83
- 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
84
- 5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
81
+ 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality until we reach an optimal point.
85
82
 
86
83
  *Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
87
84
 
@@ -7,6 +7,7 @@ src/google/cloud/db_context_enrichment/common/__init__.py
7
7
  src/google/cloud/db_context_enrichment/common/config.py
8
8
  src/google/cloud/db_context_enrichment/common/context_mutator.py
9
9
  src/google/cloud/db_context_enrichment/common/context_store_client.py
10
+ src/google/cloud/db_context_enrichment/common/context_validator.py
10
11
  src/google/cloud/db_context_enrichment/dataset/__init__.py
11
12
  src/google/cloud/db_context_enrichment/dataset/dataset_generator.py
12
13
  src/google/cloud/db_context_enrichment/evaluate/__init__.py
@@ -15,6 +16,9 @@ src/google/cloud/db_context_enrichment/evaluate/result_reader.py
15
16
  src/google/cloud/db_context_enrichment/evaluate/db_generators/__init__.py
16
17
  src/google/cloud/db_context_enrichment/evaluate/db_generators/alloydb.py
17
18
  src/google/cloud/db_context_enrichment/evaluate/db_generators/base.py
19
+ src/google/cloud/db_context_enrichment/evaluate/db_generators/bigtable.py
20
+ src/google/cloud/db_context_enrichment/evaluate/db_generators/custom.py
21
+ src/google/cloud/db_context_enrichment/evaluate/db_generators/firestore.py
18
22
  src/google/cloud/db_context_enrichment/evaluate/db_generators/mysql.py
19
23
  src/google/cloud/db_context_enrichment/evaluate/db_generators/postgres.py
20
24
  src/google/cloud/db_context_enrichment/evaluate/db_generators/spanner.py