google-cloud-db-context-engineering 0.6.0__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. google_cloud_db_context_engineering-0.7.0/PKG-INFO +110 -0
  2. google_cloud_db_context_engineering-0.7.0/README.md +92 -0
  3. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/pyproject.toml +3 -1
  4. google_cloud_db_context_engineering-0.7.0/src/google/cloud/db_context_enrichment/common/context_store_client.py +190 -0
  5. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/evaluate_generator.py +18 -7
  6. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/main.py +90 -8
  7. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/model/context.py +32 -12
  8. google_cloud_db_context_engineering-0.7.0/src/google_cloud_db_context_engineering.egg-info/PKG-INFO +110 -0
  9. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google_cloud_db_context_engineering.egg-info/SOURCES.txt +1 -0
  10. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google_cloud_db_context_engineering.egg-info/requires.txt +2 -0
  11. google_cloud_db_context_engineering-0.6.0/PKG-INFO +0 -158
  12. google_cloud_db_context_engineering-0.6.0/README.md +0 -142
  13. google_cloud_db_context_engineering-0.6.0/src/google_cloud_db_context_engineering.egg-info/PKG-INFO +0 -158
  14. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/LICENSE +0 -0
  15. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/setup.cfg +0 -0
  16. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/__init__.py +0 -0
  17. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/common/__init__.py +0 -0
  18. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/common/config.py +0 -0
  19. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/common/context_mutator.py +0 -0
  20. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/dataset/__init__.py +0 -0
  21. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/dataset/dataset_generator.py +0 -0
  22. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/__init__.py +0 -0
  23. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/db_generators/__init__.py +0 -0
  24. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/db_generators/alloydb.py +0 -0
  25. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/db_generators/base.py +0 -0
  26. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/db_generators/mysql.py +0 -0
  27. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/db_generators/postgres.py +0 -0
  28. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/db_generators/spanner.py +0 -0
  29. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/evaluate/result_reader.py +0 -0
  30. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google/cloud/db_context_enrichment/model/__init__.py +0 -0
  31. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google_cloud_db_context_engineering.egg-info/dependency_links.txt +0 -0
  32. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google_cloud_db_context_engineering.egg-info/entry_points.txt +0 -0
  33. {google_cloud_db_context_engineering-0.6.0 → google_cloud_db_context_engineering-0.7.0}/src/google_cloud_db_context_engineering.egg-info/top_level.txt +0 -0
@@ -0,0 +1,110 @@
1
+ Metadata-Version: 2.4
2
+ Name: google-cloud-db-context-engineering
3
+ Version: 0.7.0
4
+ Summary: A FastMCP server for generating natural language to SQL templates from database schemas.
5
+ Requires-Python: >=3.12
6
+ Description-Content-Type: text/markdown
7
+ License-File: LICENSE
8
+ Requires-Dist: fastmcp==3.3.1
9
+ Requires-Dist: google-genai==1.46.0
10
+ Requires-Dist: toolbox-core==0.5.2
11
+ Requires-Dist: google-cloud-geminidataanalytics==0.11.0
12
+ Requires-Dist: google-auth==2.41.1
13
+ Requires-Dist: requests==2.33.0
14
+ Provides-Extra: test
15
+ Requires-Dist: pytest; extra == "test"
16
+ Requires-Dist: pytest-asyncio; extra == "test"
17
+ Dynamic: license-file
18
+
19
+ This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
20
+
21
+ # Context Engineering Agent
22
+
23
+ The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, such as QueryData ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview)).
24
+
25
+ ---
26
+
27
+ ## Why Context Engineering?
28
+
29
+ When building data agents and natural language analytics interfaces, accurately translating user intent into database queries is critical.
30
+
31
+ As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL translation accuracy with low latency**.
32
+
33
+ ---
34
+
35
+ ## Core Concepts
36
+
37
+ A `ContextSet` is the central artifact generated and managed by the agent, containing structured knowledge in three primary forms:
38
+
39
+ * **Templates**: Link natural language query patterns to complete query statements.
40
+ * **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses or specialized join filters) linked to domain vocabulary.
41
+ * **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or simple trigram search.
42
+
43
+ For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
44
+
45
+ ---
46
+
47
+ ## Prerequisites & Environment Setup
48
+
49
+ Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
50
+
51
+ Follow the step-by-step setup guide in the official documentation:
52
+ 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
53
+
54
+ ---
55
+
56
+ ## Primary Workflow Phases
57
+
58
+ The agent enables you to craft an optimized context for QueryData API through three primary phases:
59
+
60
+ ### Phase 1: Artifact Ingestion
61
+ *Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
62
+
63
+ 1. **Connect & Discover**: Inspect any resources shared with the agent that may be relevant to your application and its domain.
64
+ 2. **Domain Concept Extraction**: Extract key terminology, domain rules, jargon, and common query patterns.
65
+ 3. **Schema Mapping**: Map extracted concepts directly to their corresponding underlying database tables and columns.
66
+
67
+ ### Phase 2: Dataset Creation
68
+ *Why it matters: A realistic golden dataset establishes your benchmark for accuracy, ensuring your data agent is evaluated against messy human language rather than artificially simple database queries.*
69
+
70
+ 1. **Dataset Planning**: Set goals for your dataset, deciding on question difficulty, business topics to cover, and total dataset size.
71
+ 2. **Initial Question Generation**: Create a core set of natural language questions grounded in your business documents, verifying that each question's corresponding query runs accurately on your database.
72
+ 3. **Dataset Expansion**: Scale up the question set by adding real-world variations, such as different human phrasing, complex combinations of joins and measures, and realistic values—to cover the ambiguity of natural language.
73
+ 4. **Dataset Validation**: Audit the full dataset against your original plan and ask for your approval before proceeding.
74
+
75
+ ### Phase 3: Context Optimization via Recursive Hill-Climbing
76
+ *Why it matters: Iterative evaluation and gap analysis systematically fix query translation failures, driving accuracy toward ~100% while ensuring fixes don't break previously working queries.*
77
+
78
+ The optimization loop creates an initial `ContextSet` and then iteratively refines it using evaluation feedback:
79
+
80
+ 1. **Bootstrap**: Generate an initial baseline context.
81
+ 2. **Evaluate**: Measure context effectiveness against a golden dataset.
82
+ 3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
83
+ 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
84
+ 5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
85
+
86
+ *Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
87
+
88
+ ---
89
+
90
+ ## Bring Artifacts via Filesystem or MCP
91
+
92
+ To prevent the AI from generating trivial schema-only questions (e.g., *"What is the count of users?"*), the agent bridges the semantic gap by grounding context generation in real-world business documents—such as **product glossaries**, **business wikis**, **SOPs**, **ORM data models**, **emails**, or **application or database logs**.
93
+
94
+ You can provide business artifacts directly to the agent from local filesystems or via **Model Context Protocol (MCP) Servers**.
95
+
96
+ ---
97
+
98
+ ## How to Use
99
+
100
+ Launch your agent harness (Gemini CLI, Claude Code, or Antigravity) in your workspace directory and interact in natural language:
101
+
102
+ ### Example Natural Language Prompts
103
+ * **End-to-End**: (From the app directory) *"Optimize context for my app."*
104
+ * **Dataset Curation**: *"Expand my dataset with app changes in `<PULL_REQUEST_LINK>`."*
105
+ * **Ad-hoc Evaluation**: *"Evaluate accuracy of QueryData with `ContextSet` `<context_set_id>` on dataset.json."*
106
+ * **Targeted Authoring**: *"Add a facet for active premium subscriptions."*
107
+
108
+ ---
109
+
110
+ > 🛠️ **Developer Note**: For developer setup instructions, local CLI linking, unit testing (`pytest`), linting (`ruff`), and fork release testing, see [docs/development.md](docs/development.md).
@@ -0,0 +1,92 @@
1
+ This is not an officially supported Google product. This project is not eligible for the [Google Open Source Software Vulnerability Rewards Program](https://bughunters.google.com/open-source-security), [Google Cloud Platform/SecOps Terms of Service](https://cloud.google.com/terms), [How Gemini for Google Cloud uses your data](https://cloud.google.com/gemini/docs/discover/data-governance). This tool is provided "as is" without warranty of any kind. Users are solely responsible for understanding and managing the tool's interaction with their databases. Use of this tool constitutes acceptance of all risks associated with database access, reading, usage, and modifications.
2
+
3
+ # Context Engineering Agent
4
+
5
+ The **Context Engineering Agent** is an AI coding agent plugin designed to run in developer agent harnesses (such as Claude Code, Antigravity, or Gemini CLI). It generates, evaluates, and iteratively tunes tailored context artifacts (`ContextSets` comprising `Templates`, `Facets`, and `Value Searches`) to enrich database schemas for **Gemini Data Analytics's data agent developer platform tools**, such as QueryData ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/data-agent-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/data-agent-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/data-agent-overview) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/data-agent-overview)).
6
+
7
+ ---
8
+
9
+ ## Why Context Engineering?
10
+
11
+ When building data agents and natural language analytics interfaces, accurately translating user intent into database queries is critical.
12
+
13
+ As outlined in **Build Context with Context Engineering Agent** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli)), by optimizing a `ContextSet` to match your application's expected query stream, the **QueryData API** acts as a data agent tool capable of achieving **~100% NL-to-SQL translation accuracy with low latency**.
14
+
15
+ ---
16
+
17
+ ## Core Concepts
18
+
19
+ A `ContextSet` is the central artifact generated and managed by the agent, containing structured knowledge in three primary forms:
20
+
21
+ * **Templates**: Link natural language query patterns to complete query statements.
22
+ * **Facets**: Reusable, modular query fragments (e.g., parameterized `WHERE` clauses or specialized join filters) linked to domain vocabulary.
23
+ * **Value Searches**: Specialized mapping queries that dynamically resolve user-supplied values (e.g., *"Lndn"*) to database records (*"London"*) via the capabilities of the underlying database, such as embedding search, AI operators, or simple trigram search.
24
+
25
+ For full schema details, structure specifications, and dialect-specific JSON representations of `ContextSets`, see the official **Context Sets Overview** ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/context-sets-overview) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/context-sets-overview) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/context-sets-overview) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/context-sets-overview)).
26
+
27
+ ---
28
+
29
+ ## Prerequisites & Environment Setup
30
+
31
+ Before getting started, prepare your GCP environment, required APIs (Data Analytics API, Gemini for Google Cloud API, Dataplex Universal Catalog API), IAM permissions, and database Data API settings.
32
+
33
+ Follow the step-by-step setup guide in the official documentation:
34
+ 👉 **Prepare Your Environment**: ([AlloyDB](https://docs.cloud.google.com/gemini/data-agents/querydata/alloydb/build-context-gemini-cli#prepare-your-environment) | Cloud SQL: [PostgreSQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-postgres/build-context-gemini-cli#prepare-your-environment) / [MySQL](https://docs.cloud.google.com/gemini/data-agents/querydata/sql-mysql/build-context-gemini-cli#prepare-your-environment) | [Spanner (GoogleSQL)](https://docs.cloud.google.com/gemini/data-agents/querydata/spanner/build-context-gemini-cli#prepare-your-environment))
35
+
36
+ ---
37
+
38
+ ## Primary Workflow Phases
39
+
40
+ The agent enables you to craft an optimized context for QueryData API through three primary phases:
41
+
42
+ ### Phase 1: Artifact Ingestion
43
+ *Why it matters: Without broader context on the application's goals and scope, AI models generate sterile queries based solely on database column names, missing how your users actually ask for information.*
44
+
45
+ 1. **Connect & Discover**: Inspect any resources shared with the agent that may be relevant to your application and its domain.
46
+ 2. **Domain Concept Extraction**: Extract key terminology, domain rules, jargon, and common query patterns.
47
+ 3. **Schema Mapping**: Map extracted concepts directly to their corresponding underlying database tables and columns.
48
+
49
+ ### Phase 2: Dataset Creation
50
+ *Why it matters: A realistic golden dataset establishes your benchmark for accuracy, ensuring your data agent is evaluated against messy human language rather than artificially simple database queries.*
51
+
52
+ 1. **Dataset Planning**: Set goals for your dataset, deciding on question difficulty, business topics to cover, and total dataset size.
53
+ 2. **Initial Question Generation**: Create a core set of natural language questions grounded in your business documents, verifying that each question's corresponding query runs accurately on your database.
54
+ 3. **Dataset Expansion**: Scale up the question set by adding real-world variations, such as different human phrasing, complex combinations of joins and measures, and realistic values—to cover the ambiguity of natural language.
55
+ 4. **Dataset Validation**: Audit the full dataset against your original plan and ask for your approval before proceeding.
56
+
57
+ ### Phase 3: Context Optimization via Recursive Hill-Climbing
58
+ *Why it matters: Iterative evaluation and gap analysis systematically fix query translation failures, driving accuracy toward ~100% while ensuring fixes don't break previously working queries.*
59
+
60
+ The optimization loop creates an initial `ContextSet` and then iteratively refines it using evaluation feedback:
61
+
62
+ 1. **Bootstrap**: Generate an initial baseline context.
63
+ 2. **Evaluate**: Measure context effectiveness against a golden dataset.
64
+ 3. **Hill-Climbing**: Perform gap analysis on failures and generate automated fixes.
65
+ 4. **Iterate**: Apply the improved context and re-run evaluation to continuously improve quality.
66
+ 5. **Final Validation** (Optional): Verify mutations against a separated test set to ensure generalization and prevent overfitting.
67
+
68
+ *Note: While there is a typical ordering for these CUJs, the agent is flexible in how you want to execute. You can run the full pipeline end-to-end, trigger any individual phase, or ask for targeted changes to the `ContextSet`.*
69
+
70
+ ---
71
+
72
+ ## Bring Artifacts via Filesystem or MCP
73
+
74
+ To prevent the AI from generating trivial schema-only questions (e.g., *"What is the count of users?"*), the agent bridges the semantic gap by grounding context generation in real-world business documents—such as **product glossaries**, **business wikis**, **SOPs**, **ORM data models**, **emails**, or **application or database logs**.
75
+
76
+ You can provide business artifacts directly to the agent from local filesystems or via **Model Context Protocol (MCP) Servers**.
77
+
78
+ ---
79
+
80
+ ## How to Use
81
+
82
+ Launch your agent harness (Gemini CLI, Claude Code, or Antigravity) in your workspace directory and interact in natural language:
83
+
84
+ ### Example Natural Language Prompts
85
+ * **End-to-End**: (From the app directory) *"Optimize context for my app."*
86
+ * **Dataset Curation**: *"Expand my dataset with app changes in `<PULL_REQUEST_LINK>`."*
87
+ * **Ad-hoc Evaluation**: *"Evaluate accuracy of QueryData with `ContextSet` `<context_set_id>` on dataset.json."*
88
+ * **Targeted Authoring**: *"Add a facet for active premium subscriptions."*
89
+
90
+ ---
91
+
92
+ > 🛠️ **Developer Note**: For developer setup instructions, local CLI linking, unit testing (`pytest`), linting (`ruff`), and fork release testing, see [docs/development.md](docs/development.md).
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "google-cloud-db-context-engineering"
3
- version = "0.6.0"
3
+ version = "0.7.0"
4
4
  description = "A FastMCP server for generating natural language to SQL templates from database schemas."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.12"
@@ -10,6 +10,8 @@ dependencies = [
10
10
  "google-genai==1.46.0",
11
11
  "toolbox-core==0.5.2",
12
12
  "google-cloud-geminidataanalytics==0.11.0",
13
+ "google-auth==2.41.1",
14
+ "requests==2.33.0",
13
15
  ]
14
16
 
15
17
  [project.scripts]
@@ -0,0 +1,190 @@
1
+ """REST client for the Context Store API.
2
+
3
+ Hand-rolled because the service is GOOGLE_INTERNAL and has no public SDK.
4
+ All calls target the autopush sandbox endpoint pinned in this module.
5
+ """
6
+
7
+ import json
8
+ import time
9
+ from typing import Any
10
+
11
+ import google.auth
12
+ import google.auth.exceptions
13
+ import requests
14
+ from google.auth.transport import requests as auth_requests
15
+
16
+ from google.cloud.db_context_enrichment.model import context
17
+
18
+ CONTEXT_STORE_ENDPOINT = "https://autopush-dataplex.sandbox.googleapis.com"
19
+ CONTEXT_STORE_LOCATION = "us-central1"
20
+ API_VERSION = "v1"
21
+ DEFAULT_OAUTH_SCOPES = ("https://www.googleapis.com/auth/cloud-platform",)
22
+
23
+ # Per-request HTTP timeout + LRO poll schedule. Transient-error retry is
24
+ # intentionally NOT done here — the MCP agent driving these tools retries
25
+ # failed tool calls on its own, and the Cloud SDK (once available) will bring
26
+ # proper retry back. Only the per-call timeout and op polling stay in this
27
+ # hand-rolled client.
28
+ _LRO_POLL_INTERVALS_SECONDS = (2.0, 4.0, 8.0, 16.0, 16.0, 16.0)
29
+ _REQUEST_TIMEOUT_SECONDS = 30
30
+
31
+
32
+ class ContextStoreClient:
33
+ """REST client for the Context Store API (autopush sandbox).
34
+
35
+ The client is stateless w.r.t. project. Operations that build a resource
36
+ path from IDs (`ensure_*`) take `project_id` as an explicit argument;
37
+ operations that take a fully-qualified resource name don't need it.
38
+
39
+ The quota project (billing / quota) is always taken from ADC's
40
+ `quota_project_id` and sent as `X-Goog-User-Project` on every request.
41
+ """
42
+
43
+ def __init__(self):
44
+ try:
45
+ credentials, _ = google.auth.default(scopes=DEFAULT_OAUTH_SCOPES)
46
+ except google.auth.exceptions.DefaultCredentialsError as e:
47
+ raise RuntimeError(
48
+ "No Application Default Credentials found. Run "
49
+ "'gcloud auth application-default login' first."
50
+ ) from e
51
+ if not credentials.quota_project_id:
52
+ raise RuntimeError(
53
+ "No quota project set. Run "
54
+ "'gcloud auth application-default set-quota-project <PROJECT_ID>'."
55
+ )
56
+ self._session = auth_requests.AuthorizedSession(credentials)
57
+ self._session.headers["X-Goog-User-Project"] = credentials.quota_project_id
58
+
59
+ def ensure_context_set_group(self, project_id: str, csg_id: str) -> str:
60
+ """Return a CSG's full resource name, creating it if absent.
61
+
62
+ Idempotent: a 409 (already exists, whether pre-existing or from a
63
+ concurrent create) is treated as success.
64
+ """
65
+ parent = f"projects/{project_id}/locations/{CONTEXT_STORE_LOCATION}"
66
+ create_url = (
67
+ f"{CONTEXT_STORE_ENDPOINT}/{API_VERSION}/{parent}/contextSetGroups"
68
+ f"?context_set_group_id={csg_id}"
69
+ )
70
+ try:
71
+ self._request_lro("POST", create_url, json_body={})
72
+ except requests.HTTPError as e:
73
+ if e.response.status_code != 409:
74
+ raise
75
+ return f"{parent}/contextSetGroups/{csg_id}"
76
+
77
+ def delete_context_set_group(self, csg_resource_name: str) -> None:
78
+ """Delete a ContextSetGroup by full resource name. Blocks on LRO.
79
+
80
+ Idempotent: a 404 (already deleted or never existed) is treated as
81
+ success. Cascades: all ContextSets inside the group are also deleted.
82
+ """
83
+ try:
84
+ self._request_lro(
85
+ "DELETE", f"{CONTEXT_STORE_ENDPOINT}/{API_VERSION}/{csg_resource_name}"
86
+ )
87
+ except requests.HTTPError as e:
88
+ if e.response.status_code != 404:
89
+ raise
90
+
91
+ def ensure_context_set(
92
+ self, project_id: str, csg_id: str, cs_id: str, version: str
93
+ ) -> str:
94
+ """Return a CS's full resource name, creating it (and the parent CSG)
95
+ if absent.
96
+
97
+ Idempotent: a 409 on the CS create is treated as success. Body is not
98
+ touched — the caller still needs `upload_context_set` to overwrite it.
99
+ """
100
+ csg_resource_name = self.ensure_context_set_group(project_id, csg_id)
101
+ url = (
102
+ f"{CONTEXT_STORE_ENDPOINT}/{API_VERSION}/{csg_resource_name}/contextSets"
103
+ f"?context_set_id={cs_id}"
104
+ )
105
+ body = {"context_set_id": cs_id, "version": version}
106
+ try:
107
+ self._request("POST", url, json_body=body)
108
+ except requests.HTTPError as e:
109
+ if e.response.status_code != 409:
110
+ raise
111
+ return f"{csg_resource_name}/contextSets/{cs_id}@{version}"
112
+
113
+ def upload_context_set(
114
+ self, cs_resource_name: str, ctx: context.ContextSet
115
+ ) -> None:
116
+ """Populate an existing ContextSet's body.
117
+
118
+ The resource must already exist (call `ensure_context_set` first).
119
+ """
120
+ url = f"{CONTEXT_STORE_ENDPOINT}/{API_VERSION}/{cs_resource_name}:upload"
121
+ body = {"context_json": ctx.model_dump_json(exclude_none=True)}
122
+ self._request("POST", url, json_body=body)
123
+
124
+ def download_context_set(self, cs_resource_name: str) -> context.ContextSet:
125
+ """Download and parse a ContextSet by full resource name."""
126
+ url = f"{CONTEXT_STORE_ENDPOINT}/{API_VERSION}/{cs_resource_name}:download"
127
+ resp = self._request("POST", url, json_body={})
128
+ raw = resp.get("contextJson")
129
+ if raw is None:
130
+ raise ValueError(f"DownloadContextSet response missing contextJson: {resp}")
131
+ # ContextSet models accept camelCase aliases too (see
132
+ # `_BaseContextModel`), so the server's mixed casing validates
133
+ # without conversion.
134
+ return context.ContextSet.model_validate(json.loads(raw))
135
+
136
+ def _request(
137
+ self,
138
+ method: str,
139
+ url: str,
140
+ json_body: dict[str, Any] | None = None,
141
+ ) -> dict[str, Any]:
142
+ """Send one HTTP request and return parsed JSON.
143
+
144
+ Returns an empty dict if the response body is empty (eg. 204 No
145
+ Content). Raises `requests.HTTPError` on 4xx/5xx,
146
+ `requests.RequestException` on network failure.
147
+ """
148
+ response = self._session.request(
149
+ method, url, json=json_body, timeout=_REQUEST_TIMEOUT_SECONDS
150
+ )
151
+ response.raise_for_status()
152
+ return response.json() if response.content else {}
153
+
154
+ def _request_lro(
155
+ self,
156
+ method: str,
157
+ url: str,
158
+ json_body: dict[str, Any] | None = None,
159
+ ) -> dict[str, Any]:
160
+ """Issue a request that returns an LRO and poll until it completes.
161
+
162
+ - HTTP layer (`_request`): each call returns 200; raises on 4xx/5xx.
163
+ - Op layer: `done` / `response` / `error` live in the JSON body, not
164
+ the HTTP status.
165
+ - Sleeps per `_LRO_POLL_INTERVALS_SECONDS`; raises `TimeoutError`
166
+ on budget exhaustion or `RuntimeError` on op-reported error.
167
+ """
168
+ op = self._request(method, url, json_body=json_body)
169
+ op_name = op.get("name")
170
+ if not op_name:
171
+ raise RuntimeError(f"{method} {url} returned no operation name: {op}")
172
+
173
+ # Skip the poll cycle if the initial op is already done (some LROs
174
+ # complete synchronously). Otherwise poll until it does or the
175
+ # budget is exhausted.
176
+ poll_url = f"{CONTEXT_STORE_ENDPOINT}/{API_VERSION}/{op_name}"
177
+ for delay in _LRO_POLL_INTERVALS_SECONDS:
178
+ if op.get("done"):
179
+ break
180
+ time.sleep(delay)
181
+ op = self._request("GET", poll_url)
182
+ if not op.get("done"):
183
+ raise TimeoutError(f"LRO {op_name} did not complete within the poll budget")
184
+ if "error" in op:
185
+ err = op["error"]
186
+ raise RuntimeError(
187
+ f"LRO {op_name} failed: [{err.get('code')}] "
188
+ f"{err.get('message', 'unknown LRO error')}"
189
+ )
190
+ return op.get("response", {})
@@ -23,7 +23,7 @@ GOLDEN_QUERIES_NAME = "golden_queries.json"
23
23
 
24
24
 
25
25
  def generate_evalbench_configs(
26
- experiment_name: str,
26
+ output_dir: str,
27
27
  dataset_path: str,
28
28
  context_set_id: str,
29
29
  toolbox_config_path: str,
@@ -32,6 +32,10 @@ def generate_evalbench_configs(
32
32
  """
33
33
  Main entrypoint: Generates Evalbench-compatible YAML configurations natively using
34
34
  private DB format converters and the google-cloud-geminidataanalytics API validations.
35
+
36
+ All output files land under `<output_dir>/eval_configs/`. The generated
37
+ `run_config.yaml` also points evalbench at `<output_dir>/eval_reports/`
38
+ for its results.
35
39
  """
36
40
  params = _extract_toolbox_params(toolbox_config_path, toolbox_source_name)
37
41
  generator = _get_db_generator(params)
@@ -39,13 +43,13 @@ def generate_evalbench_configs(
39
43
  db_config_yaml = generator.generate_db_config()
40
44
  model_config_yaml = generator.generate_model_config(context_set_id)
41
45
  llmrater_config_yaml = _generate_llmrater_config(params.get("project"))
42
- run_config_yaml = _generate_run_config(experiment_name, generator.DIALECT)
46
+ run_config_yaml = _generate_run_config(output_dir, generator.DIALECT)
43
47
 
44
48
  # Convert simplified dataset to EvalBench standard format
45
49
  golden_queries_json = _convert_dataset(dataset_path, generator.DIALECT)
46
50
 
47
51
  # Write all files directly
48
- eval_configs_dir = f"autoctx/experiments/{experiment_name}/eval_configs"
52
+ eval_configs_dir = os.path.join(output_dir, "eval_configs")
49
53
  os.makedirs(eval_configs_dir, exist_ok=True)
50
54
 
51
55
  with open(os.path.join(eval_configs_dir, DB_CONFIG_NAME), "w") as f:
@@ -140,10 +144,17 @@ def _get_db_generator(params: dict[str, Any]) -> BaseDBConfigGenerator:
140
144
  return generators[source_type](params)
141
145
 
142
146
 
143
- def _generate_run_config(experiment_name: str, dialect: str) -> str:
144
- """Generates the main EvalBench Run Experiment scaffolding."""
145
- configs_dir = f"autoctx/experiments/{experiment_name}/eval_configs"
146
- reports_dir = f"autoctx/experiments/{experiment_name}/eval_reports"
147
+ def _generate_run_config(output_dir: str, dialect: str) -> str:
148
+ """Generates the main EvalBench Run Experiment scaffolding.
149
+
150
+ Path values embedded in the YAML are normalized to POSIX separators so
151
+ the generated file is consistent across platforms (avoids mixed
152
+ `C:\\out\\eval_configs/db_config.yaml` on Windows and keeps string
153
+ assertions in tests portable).
154
+ """
155
+ output_dir_posix = output_dir.replace(os.sep, "/")
156
+ configs_dir = f"{output_dir_posix}/eval_configs"
157
+ reports_dir = f"{output_dir_posix}/eval_reports"
147
158
 
148
159
  return textwrap.dedent(f"""\
149
160
  ############################################################
@@ -1,13 +1,18 @@
1
1
  import json
2
+ import pathlib
2
3
 
3
4
  from fastmcp import FastMCP
4
5
 
5
- from google.cloud.db_context_enrichment.common import context_mutator
6
+ from google.cloud.db_context_enrichment.common import (
7
+ context_mutator,
8
+ context_store_client,
9
+ )
6
10
  from google.cloud.db_context_enrichment.dataset import dataset_generator
7
11
  from google.cloud.db_context_enrichment.evaluate import (
8
12
  evaluate_generator,
9
13
  result_reader,
10
14
  )
15
+ from google.cloud.db_context_enrichment.model import context
11
16
 
12
17
  mcp = FastMCP("Context Engineering Agent MCP")
13
18
 
@@ -18,7 +23,15 @@ async def generate_dataset(
18
23
  output_file_path: str,
19
24
  ) -> str:
20
25
  """
21
- Validates a list of evaluation dataset entries and saves them to a JSON file.
26
+ Validates a list of evaluation dataset entries and saves them to a JSON file. This is the REQUIRED tool for generating 'golden' datasets.
27
+
28
+ CRITICAL: Any request to generate an evaluation dataset MUST follow the "context-engineering-workflow" skill.
29
+ Do NOT use generic file-writing tools to save the golden dataset. Using this tool is the final step of the dataset generation workflow.
30
+
31
+ The workflow REQUIRES the following steps BEFORE calling this tool:
32
+ 1. Activate the `context-engineering-workflow` skill.
33
+ 2. Read the `context-engineering-dataset-generation` skill's `SKILL.md` for the full procedure.
34
+ 3. Generate mandatory audit reports (evalset_environment_inputs.md, evalset_gen_plan.md, evalset_report_pair_level.md, evalset_report_dataset_level.md) as specified in the workflow. These files are required for the verification process to pass.
22
35
 
23
36
  Args:
24
37
  dataset_entries_json: A JSON string representing a list of dataset items.
@@ -36,7 +49,7 @@ async def generate_dataset(
36
49
 
37
50
  @mcp.tool
38
51
  def generate_evalbench_configs(
39
- experiment_name: str,
52
+ output_dir: str,
40
53
  dataset_path: str,
41
54
  context_set_id: str,
42
55
  toolbox_config_path: str,
@@ -45,17 +58,20 @@ def generate_evalbench_configs(
45
58
  """
46
59
  Generates Evalbench YAML configurations and converts the user-facing golden dataset to be compatible for evaluation, saving all files directly to disk.
47
60
 
48
- This tool writes the following files inside `experiments/<experiment_name>/eval_configs/`:
61
+ This tool writes the following files inside `<output_dir>/eval_configs/`:
49
62
  - `db_config.yaml`
50
63
  - `model_config.yaml`
51
64
  - `run_config.yaml`
52
65
  - `llmrater_config.yaml`
53
66
  - `golden_queries.json` (converted to EvalBench internal format)
54
67
 
68
+ The generated `run_config.yaml` also points evalbench at
69
+ `<output_dir>/eval_reports/` for its results.
70
+
55
71
  Args:
56
- experiment_name: The name of the target experiment folder.
72
+ output_dir: Directory (absolute or workspace-relative) where the eval configs and reports should live. Created if missing.
57
73
  dataset_path: The absolute path to the golden dataset file in the simplified user-facing format (JSON list of objects with keys: "id", "database", "nlq", "golden_sql").
58
- context_set_id: The specific context_set_id inside the experiment.
74
+ context_set_id: Full ContextSet resource name to evaluate against.
59
75
  toolbox_config_path: The absolute path to the tools.yaml configuration file.
60
76
  toolbox_source_name: The name of the database source to use inside tools.yaml. The underlying source block must use a supported 'type' (cloud-sql-postgres, cloud-sql-mysql, spanner, alloydb-postgres).
61
77
 
@@ -63,13 +79,13 @@ def generate_evalbench_configs(
63
79
  A message indicating that the configuration files were successfully created.
64
80
  """
65
81
  evaluate_generator.generate_evalbench_configs(
66
- experiment_name,
82
+ output_dir,
67
83
  dataset_path,
68
84
  context_set_id,
69
85
  toolbox_config_path,
70
86
  toolbox_source_name,
71
87
  )
72
- return f"Successfully generated all configs for evaluation in experiments/{experiment_name}/eval_configs/"
88
+ return f"Successfully generated all configs for evaluation in {output_dir}/eval_configs/"
73
89
 
74
90
 
75
91
  @mcp.tool
@@ -117,6 +133,72 @@ def generate_upload_url(
117
133
  return "Error: Invalid db_engine. Must be one of 'alloydb', 'cloudsql', or 'spanner'."
118
134
 
119
135
 
136
+ # NOTE: `@mcp.tool` is intentionally NOT applied to upload_context_set /
137
+ # download_context_set. The Context Store client library ships in this
138
+ # release, but the MCP tool wrappers are held back until the Context Store
139
+ # API is stable in production. Re-add the decorator to expose these to
140
+ # agents when ready.
141
+ def upload_context_set(
142
+ local_file_path: str,
143
+ project_id: str,
144
+ csg_id: str,
145
+ cs_id: str,
146
+ version: str,
147
+ ) -> str:
148
+ """
149
+ Upload a local ContextSet JSON file to the Context Store.
150
+
151
+ Resource hierarchy: a ContextSetGroup (CSG) is a logical container that
152
+ holds versioned ContextSets — typically one CSG per experiment, one
153
+ cs_id per lineage (eg. "autoctx"), and versions like "v0", "v1", "v2".
154
+
155
+ The CSG and the (cs_id, version) ContextSet resource are created if they
156
+ don't already exist; then the file contents are written as the
157
+ ContextSet body. Re-uploading the same (csg_id, cs_id, version)
158
+ overwrites the body.
159
+
160
+ Args:
161
+ local_file_path: Absolute path to a ContextSet JSON file.
162
+ project_id: GCP project where the CSG / CS should live. Typically
163
+ the same project the target DB lives in.
164
+ csg_id: ContextSetGroup ID (eg. an experiment name).
165
+ cs_id: ContextSet ID (eg. "autoctx"). Stable across versions.
166
+ version: Version label (eg. "baseline", "v1").
167
+
168
+ Returns:
169
+ Full ContextSet resource name, eg.
170
+ `projects/<p>/locations/<l>/contextSetGroups/<csg_id>/contextSets/<cs_id>@<version>`.
171
+ """
172
+ text = pathlib.Path(local_file_path).read_text()
173
+ ctx = context.ContextSet.model_validate_json(text)
174
+ client = context_store_client.ContextStoreClient()
175
+ cs_resource_name = client.ensure_context_set(project_id, csg_id, cs_id, version)
176
+ client.upload_context_set(cs_resource_name, ctx)
177
+ return cs_resource_name
178
+
179
+
180
+ def download_context_set(cs_resource_name: str, output_file_path: str) -> str:
181
+ """
182
+ Download a ContextSet from the Context Store and write it to a local
183
+ JSON file. Parent directories are created as needed; the output file
184
+ is overwritten if it exists.
185
+
186
+ Args:
187
+ cs_resource_name: Full ContextSet resource name, eg.
188
+ `projects/<p>/locations/<l>/contextSetGroups/<csg_id>/contextSets/<cs_id>@<version>`.
189
+ output_file_path: Absolute path where the JSON file should be written.
190
+
191
+ Returns:
192
+ The output file path.
193
+ """
194
+ client = context_store_client.ContextStoreClient()
195
+ ctx = client.download_context_set(cs_resource_name)
196
+ out = pathlib.Path(output_file_path)
197
+ out.parent.mkdir(parents=True, exist_ok=True)
198
+ out.write_text(ctx.model_dump_json(exclude_none=True, indent=2))
199
+ return output_file_path
200
+
201
+
120
202
  @mcp.tool
121
203
  def mutate_context_set(
122
204
  file_path: str,