tdsql-mcp 1.4.3__tar.gz → 1.4.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/PKG-INFO +2 -2
  2. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/pyproject.toml +1 -1
  3. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/authorization-objects.md +70 -4
  4. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/catalog-views.md +78 -6
  5. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/guidelines.md +68 -29
  6. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/index.md +8 -4
  7. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/open-table-format.md +17 -5
  8. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/path-analysis.md +30 -0
  9. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/query-tuning.md +1 -1
  10. tdsql_mcp-1.4.4/skills/teradata-sql-analytics/syntax/string-functions.md +344 -0
  11. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/text-analytics.md +48 -0
  12. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/vector-search.md +236 -0
  13. tdsql_mcp-1.4.3/skills/teradata-sql-analytics/syntax/string-functions.md +0 -74
  14. tdsql_mcp-1.4.3/uv.lock +0 -797
  15. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/.claude-plugin/plugin.json +0 -0
  16. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/.env.example +0 -0
  17. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/.gitignore +0 -0
  18. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/CLAUDE.md +0 -0
  19. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/README.md +0 -0
  20. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/docs/architecture.md +0 -0
  21. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/requirements.txt +0 -0
  22. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/README.md +0 -0
  23. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/SKILL.md +0 -0
  24. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/aggregate-functions.md +0 -0
  25. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/ai-text-analytics.md +0 -0
  26. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/association-analysis.md +0 -0
  27. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/bit-byte-functions.md +0 -0
  28. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/byom-model-loading.md +0 -0
  29. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/byom-scoring.md +0 -0
  30. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/conditional.md +0 -0
  31. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/data-cleaning.md +0 -0
  32. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/data-exploration.md +0 -0
  33. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/data-prep.md +0 -0
  34. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/data-types-casting.md +0 -0
  35. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/date-time.md +0 -0
  36. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/embeddings.md +0 -0
  37. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/fit-transform-pattern.md +0 -0
  38. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/geospatial.md +0 -0
  39. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/hypothesis-testing.md +0 -0
  40. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/json-functions.md +0 -0
  41. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/llm-providers.md +0 -0
  42. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/ml-functions.md +0 -0
  43. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/ml-patterns.md +0 -0
  44. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/model-evaluation.md +0 -0
  45. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/numeric-functions.md +0 -0
  46. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/object-store.md +0 -0
  47. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/sql-basics.md +0 -0
  48. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-concepts.md +0 -0
  49. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-data-prep.md +0 -0
  50. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-diagnostics.md +0 -0
  51. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-dsp.md +0 -0
  52. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-estimation.md +0 -0
  53. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-forecasting.md +0 -0
  54. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-formula-rules.md +0 -0
  55. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/uaf-utility.md +0 -0
  56. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/utility-functions.md +0 -0
  57. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/skills/teradata-sql-analytics/syntax/window-functions.md +0 -0
  58. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/src/tdsql_mcp/__init__.py +0 -0
  59. {tdsql_mcp-1.4.3 → tdsql_mcp-1.4.4}/src/tdsql_mcp/server.py +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: tdsql-mcp
3
- Version: 1.4.3
3
+ Version: 1.4.4
4
4
  Summary: MCP server for Teradata Vantage — SQL execution and native analytics function reference for AI agents
5
5
  Project-URL: Homepage, https://github.com/ksturgeon-td/tdsql-mcp
6
6
  Project-URL: Repository, https://github.com/ksturgeon-td/tdsql-mcp
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "tdsql-mcp"
7
- version = "1.4.3"
7
+ version = "1.4.4"
8
8
  description = "MCP server for Teradata Vantage — SQL execution and native analytics function reference for AI agents"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -8,10 +8,11 @@ Rather than embedding API keys or access tokens directly in a function call, you
8
8
 
9
9
  ## CREATE / REPLACE AUTHORIZATION
10
10
 
11
- Two forms are supported: standard user/password credentials, and IAM role assumption (AWS only).
11
+ Three forms are supported: standard user/password credentials, credentials with a DEFINER or INVOKER execution context, and IAM role assumption (AWS only).
12
12
 
13
13
  ```sql
14
14
  { CREATE | REPLACE } AUTHORIZATION [DatabaseName.]authorization_name
15
+ [ AS { DEFINER | INVOKER } TRUSTED ]
15
16
  { user_password_auth | extended_auth }
16
17
 
17
18
  -- Form 1: user_password_auth (all providers)
@@ -26,6 +27,32 @@ EXTERNALID 'external_id_value'
26
27
  [ DURATION_SECONDS 'duration_in_seconds' ]
27
28
  ```
28
29
 
30
+ **`AS DEFINER` vs `AS INVOKER` vs `TRUSTED`:**
31
+
32
+ | Clause | Meaning |
33
+ |--------|---------|
34
+ | `AS DEFINER` | Shared access — usable by multiple users of the database. Can be created in any database. |
35
+ | `AS INVOKER` | Exclusive access by the creating user. Typically must be created in the current user's own database; exact privilege requirements may vary. |
36
+ | `TRUSTED` | Required when the auth object is referenced in an `EXTERNAL SECURITY` clause (CREATE FOREIGN TABLE, CREATE FUNCTION MAPPING). |
37
+
38
+ > **EXTERNAL SECURITY — the reference must match how the auth object was created:**
39
+ >
40
+ > | Auth created as | `EXTERNAL SECURITY` reference | Qualified `db.auth` allowed? |
41
+ > |---|---|---|
42
+ > | Plain (no `AS` clause) | `EXTERNAL SECURITY <db>.<auth>` (no keywords) | **Yes** — verified for DATALAKE and FOREIGN TABLE |
43
+ > | `AS DEFINER TRUSTED` | `EXTERNAL SECURITY DEFINER TRUSTED <auth>` | **No** — Error 3706; auth must be in the same DB as the object |
44
+ > | `AS INVOKER TRUSTED` | `EXTERNAL SECURITY INVOKER TRUSTED <auth>` | No (per syntax docs; not verified on EC) |
45
+ >
46
+ > A mismatch between the auth creation form and the EXTERNAL SECURITY reference raises
47
+ > Error 6953 (`authorization definition does not match`).
48
+ >
49
+ > **Elastic Compute:** For DATALAKE objects, the auth object must be in a **GLOBAL database**
50
+ > so it replicates across all CE instances alongside the datalake. For NOS foreign tables,
51
+ > the auth object can be in the same local or global database as the table.
52
+ >
53
+ > **GRANT EXECUTE:** After creating an auth object, grant `EXECUTE` at the database level
54
+ > so roles can use it. See the [Permissions](#permissions) section below.
55
+
29
56
  **Form 1 — user_password_auth:**
30
57
  - **`CREATE`** — creates a new authorization object; fails if it already exists
31
58
  - **`REPLACE`** — creates or replaces; use this to rotate credentials without dropping first
@@ -47,7 +74,7 @@ The three fields (`USER`, `PASSWORD`, `SESSION_TOKEN`) map to different provider
47
74
 
48
75
  | Provider | USER | PASSWORD | SESSION_TOKEN |
49
76
  |----------|------|----------|---------------|
50
- | **AWS Bedrock** | AccessKey | SecretKey | SessionKey *(optional)* |
77
+ | **AWS Bedrock / S3** | AccessKey | SecretKey | SessionToken *(optional — for STS temporary credentials only)* |
51
78
  | **Azure** | ApiBase (endpoint URL) | ApiKey | ApiVersion *(required)* |
52
79
  | **Google Cloud (GCP)** | Project | Region | AccessToken *(required)* |
53
80
  | **NVIDIA NIM** | ApiBase (endpoint URL) | ApiKey | *(not used)* |
@@ -66,11 +93,50 @@ The three fields (`USER`, `PASSWORD`, `SESSION_TOKEN`) map to different provider
66
93
  ## Examples
67
94
 
68
95
  ```sql
69
- -- AWS Bedrock (SessionKey optional — omit for long-term credentials)
96
+ -- NOS / OTF: DEFINER TRUSTED (shared service account — all users get same access)
97
+ CREATE AUTHORIZATION mydb.s3_definer_auth
98
+ AS DEFINER TRUSTED
99
+ USER '{AWS_ACCESS_KEY}'
100
+ PASSWORD '{AWS_SECRET_KEY}';
101
+ -- Optional: add SESSION_TOKEN for temporary AWS credentials (STS-issued)
102
+ -- SESSION_TOKEN '{AWS_SESSION_TOKEN}'
103
+
104
+ -- NOS / OTF: INVOKER TRUSTED (per-invoker runtime check)
105
+ CREATE AUTHORIZATION mydb.s3_invoker_auth
106
+ AS INVOKER TRUSTED
107
+ USER '{AWS_ACCESS_KEY}'
108
+ PASSWORD '{AWS_SECRET_KEY}';
109
+
110
+ -- How they appear in EXTERNAL SECURITY clauses:
111
+ -- (NOS foreign table — auth must be in the SAME database as the table;
112
+ -- reference by UNQUALIFIED name only — qualified db.auth raises Error 3706)
113
+ CREATE AUTHORIZATION mydb.s3_definer_auth -- same db as the table
114
+ AS DEFINER TRUSTED
115
+ USER '{AWS_ACCESS_KEY}' PASSWORD '{AWS_SECRET_KEY}';
116
+
117
+ CREATE MULTISET FOREIGN TABLE mydb.orders_ft,
118
+ EXTERNAL SECURITY DEFINER TRUSTED s3_definer_auth -- unqualified name
119
+ USING ( LOCATION ('/S3/s3.amazonaws.com/my-bucket/') STOREDAS ('PARQUET') )
120
+ NO PRIMARY INDEX;
121
+
122
+ -- (OTF DATALAKE — plain auth in a GLOBAL database; no keywords in EXTERNAL SECURITY)
123
+ CREATE AUTHORIZATION global_db.s3_auth
124
+ USER '{AWS_ACCESS_KEY}' PASSWORD '{AWS_SECRET_KEY}';
125
+
126
+ CREATE DATALAKE my_lake
127
+ EXTERNAL SECURITY CATALOG global_db.s3_auth, -- no DEFINER/TRUSTED keywords
128
+ EXTERNAL SECURITY STORAGE global_db.s3_auth
129
+ USING catalog_type ('glue') ... TABLE FORMAT iceberg;
130
+ ```
131
+
132
+ ### AI Function Authorization (no AS DEFINER/INVOKER)
133
+
134
+ ```sql
135
+ -- AWS Bedrock (SESSION_TOKEN optional — include only for STS-issued temporary credentials)
70
136
  CREATE AUTHORIZATION db.td_gen_aws_auth
71
137
  USER '{AWS_ACCESS_KEY}'
72
138
  PASSWORD '{AWS_SECRET_KEY}'
73
- SESSION_TOKEN '{AWS_SESSION_TOKEN}';
139
+ SESSION_TOKEN '{AWS_SESSION_TOKEN}'; -- omit for long-term IAM user credentials
74
140
 
75
141
  -- Azure (SESSION_TOKEN = ApiVersion — required)
76
142
  CREATE AUTHORIZATION db.td_gen_azure_auth
@@ -107,6 +107,16 @@ WHERE UserName = USER
107
107
  ORDER BY DatabaseName, TableName;
108
108
  ```
109
109
 
110
+ ## Database Hierarchy
111
+ ```sql
112
+ -- List direct children of a parent database (Elastic Compute: global databases)
113
+ -- DBC.ChildrenV columns are Child and Parent (NOT DatabaseName / ParentName)
114
+ SELECT Child AS DatabaseName FROM DBC.ChildrenV WHERE Parent = 'TD_GLOBAL';
115
+
116
+ -- Find all databases that are children of TD_PARENT (local/user databases)
117
+ SELECT Child AS DatabaseName FROM DBC.ChildrenV WHERE Parent = 'TD_PARENT';
118
+ ```
119
+
110
120
  ## Common Lookup Patterns
111
121
  ```sql
112
122
  -- Fully qualified table info in one query
@@ -136,23 +146,45 @@ These views supplement the HELP commands — see `open-table-format` topic for f
136
146
 
137
147
  ### DBC.DatalakeInfoV — Registered Datalakes (Iceberg / Delta Lake catalogs)
138
148
 
149
+ > **Access note:** On some systems, `DBC.DatalakeInfoV` raises **Error 3523** (not accessible)
150
+ > even for `TD_CREATOR` / `TD_ADMIN` users. If you get Error 3523 or an empty result set,
151
+ > fall back to `DBC.ServerV` and `SHOW DATALAKE <name>` for discovery.
152
+
139
153
  ```sql
140
- SELECT DatalakeName, CatalogType, CatalogURL,
141
- ObjectStoragePlatform, AuthorizationName, CommentString
154
+ SELECT DatalakeName, OTFTableFormat, CatalogType, CatalogLocation,
155
+ StorageLocation, StorageEndPoint, StorageRegion,
156
+ UnityCatalogName, StorageAccountName
142
157
  FROM DBC.DatalakeInfoV
143
158
  ORDER BY DatalakeName;
159
+ -- String values include embedded single quotes, e.g. CatalogType returns 'glue' not glue
144
160
  ```
145
161
 
146
- Columns include: `DatalakeName`, `CatalogType` (hive/glue/unity/rest/fabric), `CatalogURL`, `ObjectStoragePlatform` (S3/Azure/GCS), `AuthorizationName`, `CreateTimeStamp`, `CommentString`.
162
+ Columns include: `DatalakeName`, `OTFTableFormat` (ICEBERG/DELTA), `CatalogType`
163
+ (hive/glue/unity/rest/fabric/biglake), `CatalogLocation` (catalog URL / AWS region),
164
+ `StorageLocation`, `StorageEndPoint`, `StorageRegion`, `UnityCatalogName`, `StorageAccountName`.
165
+
166
+ Once a datalake name is known, get its full DDL (auth names, catalog config):
167
+ ```sql
168
+ SHOW DATALAKE <datalake_name>;
169
+ ```
147
170
 
148
- ### DBC.ServerV — External Servers
171
+ ### DBC.ServerV — External Servers (primary discovery path)
172
+
173
+ Use `DBC.ServerV` as the primary fallback when `DBC.DatalakeInfoV` is inaccessible or
174
+ returns no rows. It lists all registered foreign servers, including OTF datalakes.
149
175
 
150
176
  ```sql
151
- SELECT ServerName, ServerType, AuthorizationName, CommentString
177
+ SELECT ServerName, DataBaseName, AuthorizationName, AuthorizationType,
178
+ CatalogAuthName, TableFormat
152
179
  FROM DBC.ServerV
153
- ORDER BY ServerType, ServerName;
180
+ ORDER BY ServerName;
181
+ -- DataBaseName: the database this server is registered in (usually TD_SERVER_DB)
182
+ -- TableFormat: ICEBERG, DELTA, or blank for non-OTF servers
183
+ -- AuthorizationType: may appear blank on some deployments
154
184
  ```
155
185
 
186
+ > **Note:** `DBC.ServerV` does not have `ServerType` or `CommentString` columns.
187
+
156
188
  ### DBC.ManagedOTFTablesV — Managed OTF Tables
157
189
 
158
190
  ```sql
@@ -183,3 +215,43 @@ ORDER BY TableName;
183
215
  ```
184
216
 
185
217
  `DBC.StatsV` covers only relational tables. Use `DBC.AllStatsV` when you need a single view across both table types.
218
+
219
+ ---
220
+
221
+ ## Alternate OTF Discovery Path — TD_SERVER_DB and HELP FOREIGN SERVER
222
+
223
+ `TD_SERVER_DB` is a system database that holds all foreign server registrations, including DATALAKE objects and QueryGrid foreign servers. Use this path when `DBC.DatalakeInfoV` returns no rows or you need to enumerate all foreign servers to find which ones are OTF datalakes.
224
+
225
+ ### Step 1 — List all foreign servers
226
+
227
+ ```sql
228
+ HELP DATABASE TD_SERVER_DB;
229
+ ```
230
+
231
+ Returns a row per object. Rows with `Kind = 'K'` are foreign servers. These can be either QueryGrid foreign servers or DATALAKE objects — the next step distinguishes them.
232
+
233
+ ### Step 2 — Inspect a foreign server
234
+
235
+ ```sql
236
+ HELP FOREIGN SERVER TD_SERVER_DB.<foreign_server_name>;
237
+ ```
238
+
239
+ - If the object is an **OTF DATALAKE**, this returns the databases inside it — equivalent to `HELP DATALAKE <datalake_name>`.
240
+ - If it is a QueryGrid foreign server, the output will reflect that server's structure instead.
241
+
242
+ ### Full discovery workflow
243
+
244
+ ```sql
245
+ -- Step 1: find all foreign server entries
246
+ HELP DATABASE TD_SERVER_DB;
247
+ -- Look for rows where Kind = 'K'
248
+
249
+ -- Step 2: for each Kind='K' entry, inspect it
250
+ HELP FOREIGN SERVER TD_SERVER_DB.my_lake_name;
251
+ -- If OTF: returns list of databases inside the datalake
252
+ -- Then use HELP DATABASE and HELP TABLE to drill down:
253
+ HELP DATABASE my_lake_name.my_otf_database;
254
+ HELP TABLE my_lake_name.my_otf_database.my_table;
255
+ ```
256
+
257
+ > **When to use this path:** Use `TD_SERVER_DB` + `HELP FOREIGN SERVER` when the user asks about available external data, Iceberg catalogs, or datalake objects and you do not already know the datalake name. Start here to enumerate what exists, then use `HELP DATALAKE` / `HELP DATABASE` / `HELP TABLE` to drill into specifics.
@@ -6,6 +6,11 @@ Teradata Vantage has built-in distributed table operators for most analytics, ML
6
6
 
7
7
  **Before writing any SQL for analytics, transformation, or ML: check this guide and the relevant syntax topic.**
8
8
 
9
+ > **This server is reference only — it does not execute SQL.** It serves syntax and native-function
10
+ > documentation; running a query, an EXPLAIN, or any DDL/DML is the job of whatever SQL client or other
11
+ > MCP server you have connected. So the guidance below says *what to run and why*, and never names a tool
12
+ > on this server to run it with.
13
+
9
14
  ---
10
15
 
11
16
  ## Minimize Data Movement — Critical at Teradata Scale
@@ -27,13 +32,19 @@ SELECT * FROM TD_ColumnSummary(
27
32
  ) AS t;
28
33
  ```
29
34
 
30
- **2. Use SAMPLE or TOP N when you need to see rows**
35
+ **2. Use SHOW TABLE / SHOW SELECT to inspect structure; TOP N only for sample rows**
31
36
 
32
37
  ```sql
33
- -- Explore structure and values without touching the full table
38
+ -- Confirm column names and types — no data scan
39
+ SHOW TABLE db.large_table;
40
+ SHOW SELECT * FROM db.my_view;
41
+
42
+ -- Sample rows when you need actual data values (structure already known)
34
43
  SELECT TOP 100 * FROM db.large_table SAMPLE 0.001;
35
44
  ```
36
45
 
46
+ `TOP N` on a large table can perform badly even when N is small. Use `SHOW TABLE` or `SHOW SELECT *` for structure checks.
47
+
37
48
  **3. Filter before you function — push predicates as deep as possible**
38
49
 
39
50
  ```sql
@@ -87,9 +98,9 @@ SELECT * FROM TD_XGBoost(
87
98
  ) AS t;
88
99
  ```
89
100
 
90
- **6. Use execute_query with small max_rows for validation only**
101
+ **6. Return rows only to validate, never to process**
91
102
 
92
- `execute_query` is for checking results, previewing schemas, and validating output — not for analytics. Default `max_rows=100` exists for this reason. If you find yourself wanting to increase `max_rows` significantly to "process" data, that is a signal to use a native function instead.
103
+ Fetching rows is for checking results, previewing a schema, and validating output — not for analytics. Cap it small; a hundred rows is plenty. If you find yourself wanting to raise that cap substantially in order to "process" the data, that is the signal to use a native function instead and let the database do the work.
93
104
 
94
105
  **7. Aggregate before returning — let the database do the grouping**
95
106
 
@@ -117,6 +128,49 @@ Native functions distribute across all AMPs. The result set returned to the agen
117
128
 
118
129
  ## Common Operations → Native Function Mapping
119
130
 
131
+ ### Schema & Data Discovery
132
+
133
+ When users ask about available data — especially external, Iceberg, or datalake sources — use in-database discovery commands and DBC views rather than querying data directly. The table below maps common user intents to the right discovery path and topic.
134
+
135
+ | User intent | Discovery approach | Topic |
136
+ |-------------|-------------------|-------|
137
+ | "What databases / schemas exist?" | `SELECT DatabaseName FROM DBC.DatabasesV` | `catalog-views` |
138
+ | "What tables are in database X?" | `SELECT TableName, TableKind FROM DBC.TablesV WHERE DatabaseName = 'X'` | `catalog-views` |
139
+ | "What columns does table X have?" | `SELECT ColumnName, ColumnType FROM DBC.ColumnsV WHERE ...` | `catalog-views` |
140
+ | "What Iceberg / Delta Lake / OTF sources are registered?" | `SELECT * FROM DBC.DatalakeInfoV` — if empty, fall back to `HELP DATABASE TD_SERVER_DB` + `HELP FOREIGN SERVER TD_SERVER_DB.<name>` | `catalog-views`, `open-table-format` |
141
+ | "What datalakes are available?" | Same as above — `DBC.DatalakeInfoV` first, then TD_SERVER_DB path | `catalog-views`, `open-table-format` |
142
+ | "What external / foreign servers exist?" | `HELP DATABASE TD_SERVER_DB` — rows with `Kind = 'K'` are foreign servers | `catalog-views` |
143
+ | "What Iceberg tables are in datalake X?" | `HELP DATALAKE X` (known name) — or `HELP FOREIGN SERVER TD_SERVER_DB.X` (unknown name) | `open-table-format`, `catalog-views` |
144
+ | "What tables are in OTF database X.Y?" | `HELP DATABASE X.Y` | `open-table-format` |
145
+ | "What columns does OTF table X.Y.Z have?" | `HELP TABLE X.Y.Z` | `open-table-format` |
146
+ | "What NOS / object-store foreign tables exist?" | `SELECT TableName FROM DBC.TablesV WHERE DatabaseName = 'X' AND TableKind = 'O'` | `catalog-views`, `object-store` |
147
+ | "What QueryGrid foreign servers are there?" | `HELP DATABASE TD_SERVER_DB` — rows with `Kind = 'K'` that are not OTF datalakes | `catalog-views` |
148
+
149
+ **Schema inspection — use SHOW, not SELECT TOP:**
150
+
151
+ ```sql
152
+ -- WRONG: can perform very badly on large tables; avoid for structure inspection
153
+ SELECT TOP 1 * FROM db.large_table;
154
+
155
+ -- RIGHT: SHOW TABLE returns DDL (column names, types, constraints) — no data scan
156
+ SHOW TABLE db.large_table;
157
+
158
+ -- RIGHT: SHOW SELECT returns the resolved column list for a view or derived query
159
+ SHOW SELECT * FROM db.my_view;
160
+ SHOW SELECT * FROM db.large_table;
161
+ ```
162
+
163
+ Use `SHOW TABLE` or `SHOW SELECT *` whenever the goal is to confirm structure, not data. Reserve `TOP N` for when you actually need sample rows.
164
+
165
+ **OTF discovery decision tree:**
166
+ 1. Try `DBC.DatalakeInfoV` — lists all registered datalakes when the user has view access.
167
+ 2. If empty or access denied, try `HELP DATABASE TD_SERVER_DB` → find `Kind = 'K'` rows → `HELP FOREIGN SERVER TD_SERVER_DB.<name>` for each.
168
+ 3. Once a datalake name is known, use `HELP DATALAKE <name>` / `HELP DATABASE <datalake>.<db>` / `HELP TABLE <datalake>.<db>.<table>` to drill in.
169
+
170
+ Load `get_syntax_help(topic="catalog-views")` for DBC view syntax and the full TD_SERVER_DB workflow. Load `get_syntax_help(topic="open-table-format")` for HELP DATALAKE syntax and OTF DDL/DML.
171
+
172
+ ---
173
+
120
174
  ### Data Exploration & Statistics
121
175
 
122
176
  | Instead of this (manual SQL) | Use this (native function) | Topic |
@@ -203,6 +257,12 @@ Native functions distribute across all AMPs. The result set returned to the agen
203
257
  | External approximate nearest neighbor index | `TD_HNSW` / `TD_HNSWPredict` | `vector-search` |
204
258
  | External embedding storage type | `VECTOR` / `Vector32` data type | `data-types-casting` |
205
259
 
260
+ **Inline NL Query → Embedding → Vector Search:** For RAG retrieval, use the full CTE pipeline pattern — embed the query inline with `AI_TEXTEMBEDDINGS`, normalize with `TD_VectorNormalize(Approach('UNITVECTOR'))`, and search with `TD_VectorDistance` against a pre-built corpus embedding table, all in a single SQL statement. See `vector-search` topic, "Inline NL Query → Embedding → Vector Search Pipeline" section.
261
+
262
+ **Building a corpus embedding table:** Use the full workflow: source text table → `AI_TEXTEMBEDDINGS` → `TD_VectorNormalize(Approach('UNITVECTOR'))` → CTAS. See `vector-search` topic, "Full Corpus Build Workflow" section.
263
+
264
+ **Vector dimension introspection:** Use `embedding.LENGTH()` to get the number of dimensions from a VECTOR column — never infer from UDT byte size. `SELECT embedding.LENGTH() AS dims FROM db.table SAMPLE 1;`
265
+
206
266
  ### Statistical Testing
207
267
 
208
268
  | Instead of this | Use this (native function) | Topic |
@@ -278,16 +338,16 @@ See the `ml-patterns` topic for complete end-to-end ML pipeline examples. See `u
278
338
 
279
339
  ## Query Validation and Optimization with EXPLAIN
280
340
 
281
- For any non-trivial query, run `explain_query` before executing. Read the plan and optimize if needed — do not just use EXPLAIN for syntax checking.
341
+ For any non-trivial query, run `EXPLAIN` on it before executing. Read the plan and optimize if needed — do not use EXPLAIN merely as a syntax check.
282
342
 
283
343
  **When to always EXPLAIN first:**
284
344
  - Queries joining two or more large tables
285
345
  - Queries without a PI-aligned filter (likely full table scan)
286
346
  - Queries with subqueries, correlated subqueries, or complex predicates
287
- - Any query you're about to run with `execute_statement` that modifies data
347
+ - Any statement you are about to run that modifies data (INSERT, UPDATE, DELETE, MERGE, DDL)
288
348
 
289
349
  **Decision loop:**
290
- 1. Run `explain_query(sql=...)`
350
+ 1. Run `EXPLAIN <your statement>`
291
351
  2. Scan for red flags (see below) — if found, fix and re-EXPLAIN before executing
292
352
  3. Only execute once the plan looks reasonable
293
353
 
@@ -312,27 +372,6 @@ For full EXPLAIN interpretation guidance, optimization playbook, and stats colle
312
372
 
313
373
  ---
314
374
 
315
- ## External Data Access
316
-
317
- ### Open Table Format (OTF) — Iceberg and Delta Lake
318
-
319
- For working with Apache Iceberg or Delta Lake tables in external catalogs (Hive, Glue, Unity, REST):
320
- - Load `get_syntax_help(topic='open-table-format')` before writing any OTF DDL or DML
321
- - OTF tables use three-tier dot notation: `datalake.database.table`
322
- - **HELP TABLE, not describe_table:** `DBC.ColumnsV` does not cover OTF tables. Use `HELP TABLE my_lake.db.table;` passed through `execute_query` to inspect OTF table columns
323
- - **HELP commands are first-class statements:** `HELP DATALAKE`, `HELP DATABASE`, `HELP TABLE` query the external catalog directly — do not rewrite them as SELECT queries against DBC views
324
- - For CREATE/ALTER/DROP TABLE and DML (INSERT/UPDATE/DELETE) on OTF tables, follow the syntax exactly — OTF has significant restrictions vs. relational SQL
325
-
326
- ### Native Object Store (NOS) — S3, Azure, GCS
327
-
328
- For ad-hoc object store access (READ_NOS), persistent access (CREATE FOREIGN TABLE), or export (WRITE_NOS):
329
- - Load `get_syntax_help(topic='object-store')` before writing any NOS SQL
330
- - **HELP TABLE, not describe_table:** `DBC.ColumnsV` does not cover foreign tables. Use `HELP TABLE mydb.foreign_table;` or `READ_NOS(USING ... RETURNTYPE('NOSREAD_SCHEMA'))` to discover schema
331
- - For import workflows, use the foreign table → CAST view → permanent table pattern documented in the `object-store` topic
332
- - Collect statistics on payload attributes used in joins or filters to improve query plans against foreign tables
333
-
334
- ---
335
-
336
375
  ## When Manual SQL Is Appropriate
337
376
 
338
377
  Native functions do not cover everything. Use hand-written SQL for:
@@ -341,7 +380,7 @@ Native functions do not cover everything. Use hand-written SQL for:
341
380
  - Date/time arithmetic (`date-time`)
342
381
  - CASE expressions and NULL handling (`conditional`)
343
382
  - Window functions for lag/lead features, running totals (`window-functions`)
344
- - Schema discovery via MCP tools first — `list_databases`, `list_tables`, `describe_table` cover the common cases; fall back to manual DBC.* queries only for capabilities not covered by those tools (`catalog-views`)
383
+ - Schema discovery queries against DBC.* views (`catalog-views`)
345
384
  - Bit/byte manipulation — `BITAND`, `BITOR`, `BITXOR`, `BITNOT`, `SHIFTLEFT`/`SHIFTRIGHT`, `ROTATELEFT`/`ROTATERIGHT`, `GETBIT`, `SETBIT`, `COUNTSET`, `SUBBITSTR`, `TO_BYTE` — no ANSI equivalents; do not use `&`, `|`, `^`, `~` operators (`bit-byte-functions`)
346
385
  - JSON data — native `JSON` type with BSON/UBJSON binary formats; JSONPath extraction; shredding and publishing — all in-database (`json-functions`)
347
386
  - One-off computations not covered by any native function
@@ -38,10 +38,10 @@ Use `get_syntax_help(topic="<name>")` to load any topic below.
38
38
  | `data-cleaning` | NULL handling, deduplication, string cleaning, outlier detection, type validation |
39
39
  | `data-prep` | Feature engineering: binning, encoding, scaling, pivoting, polynomial features, dimensionality reduction, Fit/Transform pairs, TD_SMOTE oversampling |
40
40
  | `utility-functions` | TD_FillRowID, TD_NumApply, TD_RoundColumns, TD_StrApply |
41
- | `text-analytics` | Text tokenization, classification, and entity extraction: TD_NgramSplitter, TD_NaiveBayesTextClassifier, TD_NERExtractor |
41
+ | `text-analytics` | Text tokenization, classification, tagging, and entity extraction: TD_NgramSplitter, TD_TextParser, TD_TextTagger, TD_TFIDF, TD_SentimentExtractor, TD_WordEmbeddings, TD_NaiveBayesTextClassifier, TD_TextMorph, TD_POSTagger, TD_NERExtractor |
42
42
  | `hypothesis-testing` | Statistical hypothesis tests: TD_ANOVA, TD_ChiSq, TD_FTest, TD_ZTest |
43
43
  | `association-analysis` | Frequent itemset mining and collaborative filtering: TD_Apriori, TD_CFilter |
44
- | `path-analysis` | Event sequence analysis: Attribution, Sessionize, nPath |
44
+ | `path-analysis` | Event sequence analysis: Attribution, Sessionize, nPath, TD_PathSummarizer |
45
45
  | `model-evaluation` | Model evaluation and explainability: TD_TrainTestSplit, TD_ClassificationEvaluator, TD_RegressionEvaluator, TD_ROC, TD_Silhouette, TD_SHAP |
46
46
  | `ml-patterns` | End-to-end ML pipeline patterns: CTE prediction pipeline, elbow method, train/evaluate/retrain loop, class imbalance workflow, micromodeling |
47
47
  | `vector-search` | Vector similarity search: TD_VectorDistance (exact), TD_HNSW/TD_HNSWPredict (approximate), KMeans IVF pattern |
@@ -100,6 +100,7 @@ Recommended topic reading order for common end-to-end tasks. Load these topics i
100
100
  | **Micromodeling (per-segment models)** | `ml-functions` (TD_GLM) → `ml-patterns` (micromodeling) |
101
101
  | **Semantic search / RAG embeddings** | `authorization-objects` → `llm-providers` → `embeddings` → `vector-search` |
102
102
  | **In-database ONNX inference** | `byom-model-loading` → `embeddings` (ONNXEmbeddings) → `vector-search` |
103
+ | **Discover available Iceberg / Delta Lake / OTF sources** | `catalog-views` (DBC.DatalakeInfoV + TD_SERVER_DB section) → `open-table-format` |
103
104
  | **Query Iceberg / Delta Lake tables** | `open-table-format` |
104
105
  | **Import NOS data into Vantage** | `object-store` (foreign table → CAST view → permanent table) |
105
106
  | **Export Vantage data to S3/Azure/GCS** | `object-store` (WRITE_NOS) |
@@ -109,5 +110,8 @@ Recommended topic reading order for common end-to-end tasks. Load these topics i
109
110
  | **Digital signal filtering** | `uaf-concepts` → `uaf-utility` (TD_FILTERFACTORY1D) → `uaf-dsp` (TD_CONVOLVE) |
110
111
 
111
112
  ---
112
- > **Adding topics:** Drop a new `.md` file into `src/tdsql_mcp/syntax/` and it appears here
113
- > automatically — no code changes needed.
113
+ > **Adding topics:** Drop a new `.md` file into the matching category subfolder
114
+ > (`core/`, `functions/`, `analytics/`, `uaf/`, `geospatial/`, `reference/`). The
115
+ > filename (without `.md`) is the topic slug — it **must be globally unique
116
+ > across subfolders**. Run `bash scripts/check-unique-slugs.sh` before pushing.
117
+ > Add a row to the matching table above so the LLM can discover it.
@@ -60,6 +60,8 @@ HELP TABLE my_lake.sales_db.orders;
60
60
  | AWS Glue | `glue` | Yes/Yes | N/A | N/A |
61
61
  | Databricks Unity | `unity` | Yes/Yes | Yes/Yes | Yes/Yes |
62
62
  | REST (Polaris/Gravitino) | `rest` | Yes/Yes | Yes/Yes | No/No |
63
+ | Microsoft Fabric / OneLake | `fabric` | N/A | Yes/Yes | N/A |
64
+ | GCP BigLake | `biglake` | N/A | N/A | Yes/Yes |
63
65
  | Object Store (direct) | — | Yes/No | Yes/No | Yes/No |
64
66
 
65
67
  ### Delta Lake — Catalog Support (Read/Write)
@@ -68,6 +70,7 @@ HELP TABLE my_lake.sales_db.orders;
68
70
  |---|---|---|---|---|
69
71
  | AWS Glue | `glue` | Yes/Yes | N/A | N/A |
70
72
  | Databricks Unity | `unity` | Yes/Yes | Yes/Yes | Yes/Yes |
73
+ | Microsoft Fabric / OneLake | `fabric` | N/A | Yes/Yes | N/A |
71
74
 
72
75
  ### Object Storage
73
76
  - Amazon S3
@@ -100,8 +103,11 @@ CREATE AUTHORIZATION user1.my_auth
100
103
  USER '<access_key_id_or_client_id>'
101
104
  PASSWORD '<secret_key_or_client_secret>';
102
105
 
103
- -- Grant execute privilege to other users
106
+ -- Grant execute to a specific user
104
107
  GRANT EXECUTE ON user1.my_auth TO user2 WITH GRANT OPTION;
108
+
109
+ -- Grant execute at the database level (covers all auth objects in the DB — preferred for roles)
110
+ GRANT EXECUTE ON <db> TO <role_name>;
105
111
  ```
106
112
 
107
113
  ### AWS Assume Role
@@ -111,7 +117,7 @@ CREATE AUTHORIZATION assume_role_auth
111
117
  USING
112
118
  AUTHSERVICETYPE 'ASSUME_ROLE'
113
119
  ROLENAME '<IAM_role_ARN>'
114
- EXTERNAL_ID '<external_id_from_trust_policy>';
120
+ EXTERNALID '<external_id_from_trust_policy>';
115
121
  ```
116
122
 
117
123
  ### Azure Active Directory Service Principal
@@ -131,12 +137,18 @@ All DATALAKE objects are created in `TD_SERVER_DB`. `TABLE FORMAT` is specified
131
137
 
132
138
  ### Syntax
133
139
 
140
+ > **EXTERNAL SECURITY matching rule:** The keywords in `EXTERNAL SECURITY` must match how
141
+ > the auth object was created. Plain auth (no `AS` clause) → no keywords in the reference,
142
+ > qualified `db.auth` name allowed. `AS DEFINER TRUSTED` → `DEFINER TRUSTED` keyword required,
143
+ > unqualified name only. Mismatch raises Error 6953 or Error 3706. Plain auth with qualified
144
+ > reference is the verified production pattern for DATALAKE objects.
145
+
134
146
  ```sql
135
147
  CREATE DATALAKE <datalake_name>
136
148
  EXTERNAL SECURITY [DEFINER TRUSTED | [INVOKER] TRUSTED] CATALOG <auth_name>,
137
149
  EXTERNAL SECURITY [DEFINER TRUSTED | [INVOKER] TRUSTED] STORAGE <auth_name>
138
150
  USING
139
- catalog_type ('<type>') -- Required: hive|glue|unity|rest|fabric
151
+ catalog_type ('<type>') -- Required: hive|glue|unity|rest|fabric|biglake
140
152
  [catalog_location ('<uri>')] -- Required: hive (thrift://...), unity (https://...), rest
141
153
  [storage_location ('<uri>')] -- Required: hive, glue; optional: unity on Azure
142
154
  [storage_region ('<region>')] -- Required: glue, hive; e.g. 'us-west-2'
@@ -174,7 +186,7 @@ TABLE FORMAT iceberg;
174
186
  CREATE AUTHORIZATION delta_assume_role
175
187
  USING AUTHSERVICETYPE 'ASSUME_ROLE'
176
188
  ROLENAME '<IAM_role_ARN>'
177
- EXTERNAL_ID '<external_id>';
189
+ EXTERNALID '<external_id>';
178
190
 
179
191
  CREATE DATALAKE my_delta_lake
180
192
  EXTERNAL SECURITY CATALOG delta_assume_role,
@@ -328,7 +340,7 @@ Note: `TABLE FORMAT` cannot be changed with `ALTER DATALAKE`.
328
340
  -- List all registered datalakes
329
341
  SELECT * FROM DBC.DatalakeInfoV;
330
342
  -- Columns: DatalakeName, OTFTableFormat, CatalogType, CatalogLocation,
331
- -- StorageLocation, StorageRegion, UnityCatalogName, StorageAccountName
343
+ -- StorageLocation, StorageEndPoint, StorageRegion, UnityCatalogName, StorageAccountName
332
344
 
333
345
  -- Server-level view (includes auth metadata)
334
346
  SELECT * FROM DBC.ServerV WHERE TableFormat = 'iceberg';
@@ -356,3 +356,33 @@ FROM nPath(
356
356
  ) AS dt
357
357
  GROUP BY 1 ORDER BY 2 DESC;
358
358
  ```
359
+
360
+ ---
361
+
362
+ ## TD_PathSummarizer — Path Summary Generation
363
+
364
+ Builds a hierarchical path summary from ordered path-generation data by partition and sequence. Commonly used to materialize a connected tree representation of traversed states, prefixes, and parent-child relationships for downstream analysis.
365
+
366
+ ```sql
367
+ TD_PathSummarizer (
368
+ ON { table | view | (query) } AS InputTable
369
+ PARTITION BY partition_column [,...]
370
+ USING
371
+ [ CountColumn('count_column') ]
372
+ [ Delimiter('delimiter') ]
373
+ SeqColumn('sequence_column')
374
+ PartitionNames('partition_column' [,...])
375
+ [ HashCode({ 'true' | 't' | 'yes' | 'y' | '1' |
376
+ 'false' | 'f' | 'no' | 'n' | '0' }) ]
377
+ PrefixColumn('prefix_column')
378
+ )
379
+ ```
380
+
381
+ > **Notes:**
382
+ > - The input must be aliased as `InputTable`
383
+ > - `SeqColumn`, `PartitionNames`, and `PrefixColumn` are required
384
+ > - `PartitionNames` must match the partition columns in the `PARTITION BY` clause
385
+ > - `CountColumn`, `Delimiter`, and `HashCode` are optional
386
+ > - CLOB columns are not supported in partition columns
387
+ > - The function expects path-generator-shaped sequence and prefix data
388
+ > - Output fields include `node`, `parent`, `children`, `cnt`, `depth`, and `prefix`
@@ -10,7 +10,7 @@ EXPLAIN SELECT col1, col2 FROM db.table WHERE id = 1;
10
10
  DYNAMIC EXPLAIN SELECT col1, col2 FROM db.table WHERE id = 1;
11
11
  ```
12
12
 
13
- Use `explain_query` before executing any non-trivial query. If the plan shows critical issues (see below), fix and re-EXPLAIN before running.
13
+ Run `EXPLAIN` before executing any non-trivial query. If the plan shows critical issues (see below), fix and re-EXPLAIN before running.
14
14
 
15
15
  ### EXPLAIN Phrase Glossary
16
16