tdsql-mcp 1.4.0__tar.gz → 1.4.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tdsql_mcp-1.4.2/.claude-plugin/plugin.json +5 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/PKG-INFO +1 -1
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/pyproject.toml +1 -1
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/SKILL.md +6 -1
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/catalog-views.md +4 -31
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/data-prep.md +2 -4
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/guidelines.md +3 -92
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/index.md +1 -1
- tdsql_mcp-1.4.2/skills/teradata-sql-analytics/syntax/query-tuning.md +152 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/sql-basics.md +12 -97
- tdsql_mcp-1.4.2/skills/teradata-sql-analytics/syntax/string-functions.md +74 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/vector-search.md +0 -262
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/src/tdsql_mcp/server.py +16 -0
- tdsql_mcp-1.4.0/skills/teradata-sql-analytics/syntax/query-tuning.md +0 -321
- tdsql_mcp-1.4.0/skills/teradata-sql-analytics/syntax/string-functions.md +0 -344
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/.env.example +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/.gitignore +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/CLAUDE.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/README.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/docs/architecture.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/requirements.txt +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/README.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/aggregate-functions.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/ai-text-analytics.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/association-analysis.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/authorization-objects.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/bit-byte-functions.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/byom-model-loading.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/byom-scoring.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/conditional.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/data-cleaning.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/data-exploration.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/data-types-casting.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/date-time.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/embeddings.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/fit-transform-pattern.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/geospatial.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/hypothesis-testing.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/json-functions.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/llm-providers.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/ml-functions.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/ml-patterns.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/model-evaluation.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/numeric-functions.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/object-store.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/open-table-format.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/path-analysis.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/text-analytics.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-concepts.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-data-prep.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-diagnostics.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-dsp.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-estimation.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-forecasting.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-formula-rules.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/uaf-utility.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/utility-functions.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/skills/teradata-sql-analytics/syntax/window-functions.md +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/src/tdsql_mcp/__init__.py +0 -0
- {tdsql_mcp-1.4.0 → tdsql_mcp-1.4.2}/uv.lock +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: tdsql-mcp
|
|
3
|
-
Version: 1.4.
|
|
3
|
+
Version: 1.4.2
|
|
4
4
|
Summary: MCP server for Teradata Vantage — SQL execution and native analytics function reference for AI agents
|
|
5
5
|
Project-URL: Homepage, https://github.com/ksturgeon-td/tdsql-mcp
|
|
6
6
|
Project-URL: Repository, https://github.com/ksturgeon-td/tdsql-mcp
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "tdsql-mcp"
|
|
7
|
-
version = "1.4.
|
|
7
|
+
version = "1.4.2"
|
|
8
8
|
description = "MCP server for Teradata Vantage — SQL execution and native analytics function reference for AI agents"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: teradata-sql-analytics
|
|
3
|
-
description: Load at the start of any Teradata
|
|
3
|
+
description: Load at the start of any Teradata analytics session. Injects native function guidelines and syntax.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
You are working with a Teradata Vantage database.
|
|
@@ -29,6 +29,11 @@ Apply these principles throughout the session:
|
|
|
29
29
|
**1. Don't assume. Surface uncertainty and tradeoffs.**
|
|
30
30
|
If you know a native function exists but haven't loaded its syntax topic, say so — don't write syntax from training knowledge. When multiple approaches fit (exact vs. approximate vector search, ARIMA vs. Holt-Winters, TD_XGBoost vs. TD_GLM), state the tradeoff and let the user decide. If the schema or task is ambiguous, ask before writing SQL.
|
|
31
31
|
|
|
32
|
+
**Topic triggers — load these immediately when the task involves:**
|
|
33
|
+
- Apache Iceberg, Delta Lake, Open Table Format, OTF tables, DATALAKE objects, three-tier notation (`datalake.db.table`) → read [syntax/open-table-format.md](syntax/open-table-format.md)
|
|
34
|
+
- S3, Azure Blob, GCS, object store, READ_NOS, WRITE_NOS, CREATE FOREIGN TABLE, NOS foreign tables → read [syntax/object-store.md](syntax/object-store.md)
|
|
35
|
+
- `describe_table` does not work for OTF or foreign tables — use `HELP TABLE` passed through `execute_query` instead
|
|
36
|
+
|
|
32
37
|
**2. Minimum SQL that solves the problem. Nothing speculative.**
|
|
33
38
|
Don't add columns, CTEs, or transformations that weren't requested. At Teradata scale, unnecessary work has real cost. Load only the syntax topics needed for the current task.
|
|
34
39
|
|
|
@@ -98,31 +98,15 @@ WHERE DatabaseName = 'mydb' AND TableName = 'mytable'
|
|
|
98
98
|
ORDER BY ColumnName;
|
|
99
99
|
```
|
|
100
100
|
|
|
101
|
-
##
|
|
102
|
-
|
|
103
|
-
Prefer the `list_databases` MCP tool — it calls `DBC.DatabasesV` which already filters to databases visible to the current session.
|
|
104
|
-
|
|
105
|
-
`DBC.DatabasesV` answers "what databases exist and are visible to me?"
|
|
106
|
-
`DBC.AllRightsV` answers "what databases do I have explicit rights on?" (includes role-inherited grants)
|
|
107
|
-
|
|
108
|
-
Use `DBC.AllRightsV` when you need to know what rights a specific user holds:
|
|
109
|
-
|
|
101
|
+
## Access Rights
|
|
110
102
|
```sql
|
|
111
|
-
--
|
|
112
|
-
SELECT DISTINCT DatabaseName
|
|
113
|
-
FROM DBC.AllRightsV
|
|
114
|
-
WHERE UserName = 'some_user'
|
|
115
|
-
ORDER BY DatabaseName;
|
|
116
|
-
|
|
117
|
-
-- Full rights detail for a user
|
|
103
|
+
-- What access does the current user have on a database?
|
|
118
104
|
SELECT AccessRight, DatabaseName, TableName
|
|
119
|
-
FROM DBC.
|
|
120
|
-
WHERE UserName =
|
|
105
|
+
FROM DBC.UserRightsV
|
|
106
|
+
WHERE UserName = USER
|
|
121
107
|
ORDER BY DatabaseName, TableName;
|
|
122
108
|
```
|
|
123
109
|
|
|
124
|
-
> **Do not use `DBC.UserRightsV` for database enumeration** — it shows only directly granted rights and misses role-inherited access. Use `DBC.AllRightsV` for a complete picture.
|
|
125
|
-
|
|
126
110
|
## Common Lookup Patterns
|
|
127
111
|
```sql
|
|
128
112
|
-- Fully qualified table info in one query
|
|
@@ -149,7 +133,6 @@ These views cover registered datalakes, external servers, and OTF statistics. Th
|
|
|
149
133
|
### DBC.DatalakeInfoV — Registered Datalakes
|
|
150
134
|
|
|
151
135
|
```sql
|
|
152
|
-
-- All registered datalakes and their properties
|
|
153
136
|
SELECT DatalakeName, CatalogType, CatalogURL,
|
|
154
137
|
ObjectStoragePlatform, AuthorizationName, CommentString
|
|
155
138
|
FROM DBC.DatalakeInfoV
|
|
@@ -161,7 +144,6 @@ Columns include: `DatalakeName`, `CatalogType` (hive/glue/unity/rest/fabric), `C
|
|
|
161
144
|
### DBC.ServerV — External Servers
|
|
162
145
|
|
|
163
146
|
```sql
|
|
164
|
-
-- All external server objects (includes datalakes)
|
|
165
147
|
SELECT ServerName, ServerType, AuthorizationName, CommentString
|
|
166
148
|
FROM DBC.ServerV
|
|
167
149
|
ORDER BY ServerType, ServerName;
|
|
@@ -169,8 +151,6 @@ ORDER BY ServerType, ServerName;
|
|
|
169
151
|
|
|
170
152
|
### DBC.ManagedOTFTablesV — Managed OTF Tables
|
|
171
153
|
|
|
172
|
-
Lists OTF tables that Teradata has registered as managed (created via Vantage DDL rather than discovered from the catalog):
|
|
173
|
-
|
|
174
154
|
```sql
|
|
175
155
|
SELECT DatalakeName, DatabaseName, TableName, TableFormat,
|
|
176
156
|
CreateTimeStamp, LastAlterTimeStamp
|
|
@@ -180,24 +160,17 @@ ORDER BY DatalakeName, DatabaseName, TableName;
|
|
|
180
160
|
|
|
181
161
|
### DBC.OtfStatsV — OTF Statistics
|
|
182
162
|
|
|
183
|
-
Statistics collected on OTF table columns (via COLLECT STATISTICS with three-tier notation):
|
|
184
|
-
|
|
185
163
|
```sql
|
|
186
|
-
-- Statistics collected on OTF tables
|
|
187
164
|
SELECT DatabaseName, TableName, ColumnName, StatsType,
|
|
188
165
|
LastCollectTimeStamp, SampleSize
|
|
189
166
|
FROM DBC.OtfStatsV
|
|
190
167
|
WHERE DatabaseName = 'my_lake'
|
|
191
168
|
ORDER BY TableName, ColumnName;
|
|
192
|
-
-- DatabaseName here is the datalake name
|
|
193
169
|
```
|
|
194
170
|
|
|
195
171
|
### DBC.AllStatsV — All Statistics (Teradata + OTF)
|
|
196
172
|
|
|
197
|
-
Combined view of statistics across both relational Teradata tables and OTF tables:
|
|
198
|
-
|
|
199
173
|
```sql
|
|
200
|
-
-- All stats (TD + OTF) for a database
|
|
201
174
|
SELECT DatabaseName, TableName, ColumnName, StatsType,
|
|
202
175
|
LastCollectTimeStamp
|
|
203
176
|
FROM DBC.AllStatsV
|
|
@@ -329,8 +329,6 @@ FROM db.campaign;
|
|
|
329
329
|
|
|
330
330
|
Generates synthetic minority-class samples to address class imbalance in training data. Supports four oversampling strategies: standard SMOTE, ADASYN, Borderline-SMOTE, and SMOTE-NC (for mixed numeric + categorical features).
|
|
331
331
|
|
|
332
|
-
> **Required:** The `ON` clause must be written as `ON ... AS InputTable PARTITION BY ANY`. Both `AS InputTable` and `PARTITION BY ANY` are required — omitting either will cause an error.
|
|
333
|
-
|
|
334
332
|
> **Output contains synthetic samples only** — `UNION ALL` with the original training table to produce the full augmented dataset before training.
|
|
335
333
|
|
|
336
334
|
```sql
|
|
@@ -359,7 +357,7 @@ SELECT * FROM TD_SMOTE(
|
|
|
359
357
|
SELECT * FROM db.training_table
|
|
360
358
|
UNION ALL
|
|
361
359
|
SELECT * FROM TD_SMOTE(
|
|
362
|
-
ON db.training_table
|
|
360
|
+
ON db.training_table PARTITION BY ANY
|
|
363
361
|
USING
|
|
364
362
|
IDColumn('id')
|
|
365
363
|
ResponseColumn('label')
|
|
@@ -409,7 +407,7 @@ SELECT StatValue AS MedianValue FROM TD_UnivariateStatistics(
|
|
|
409
407
|
|
|
410
408
|
-- Step 3: run TD_SMOTE with smotenc
|
|
411
409
|
SELECT * FROM TD_SMOTE(
|
|
412
|
-
ON db.training_table
|
|
410
|
+
ON db.training_table PARTITION BY ANY
|
|
413
411
|
ON db.smote_encodings AS EncodingsTable DIMENSION
|
|
414
412
|
USING
|
|
415
413
|
IDColumn('id')
|
|
@@ -6,8 +6,6 @@ Teradata Vantage has built-in distributed table operators for most analytics, ML
|
|
|
6
6
|
|
|
7
7
|
**Before writing any SQL for analytics, transformation, or ML: check this guide and the relevant syntax topic.**
|
|
8
8
|
|
|
9
|
-
> **Never write native table operator syntax from training knowledge.** Functions like `nPath`, `TD_XGBoost`, `TD_HNSW`, UAF functions, and `AI_*` functions have complex, version-specific parameter names, clause ordering, and required options. Training data knowledge of these functions is frequently incomplete or incorrect. This guide tells you **which** function to use — you must call `get_syntax_help(topic='<name>')` to load the authoritative syntax before writing any native function call.
|
|
10
|
-
|
|
11
9
|
---
|
|
12
10
|
|
|
13
11
|
## Minimize Data Movement — Critical at Teradata Scale
|
|
@@ -197,31 +195,6 @@ Native functions distribute across all AMPs. The result set returned to the agen
|
|
|
197
195
|
| External TF-IDF | `TD_TFIDF` | `text-analytics` |
|
|
198
196
|
| External word embeddings | `TD_WordEmbeddings` | `text-analytics` |
|
|
199
197
|
|
|
200
|
-
### LLM-Powered Text Analytics (AI_* Functions)
|
|
201
|
-
|
|
202
|
-
> **Prerequisites:** Requires an authorization object and LLM provider configuration before use. See `authorization-objects` and `llm-providers` topics. All functions use `TD_SYSFNLIB.<FunctionName>(ON ...)` — do **not** add a `PARTITION BY` clause.
|
|
203
|
-
|
|
204
|
-
| Instead of this | Use this (native function) | Topic |
|
|
205
|
-
|-----------------|---------------------------|-------|
|
|
206
|
-
| External/post-hoc sentiment scoring | `AI_AnalyzeSentiment` | `ai-text-analytics` |
|
|
207
|
-
| Per-row LLM API calls with question + context data | `AI_AskLLM` (two-table: InputTable + ContextTable, co-partitioned by key) | `ai-text-analytics` |
|
|
208
|
-
| External language detection | `AI_DetectLanguage` | `ai-text-analytics` |
|
|
209
|
-
| External key phrase extraction | `AI_ExtractKeyPhrases` | `ai-text-analytics` |
|
|
210
|
-
| External PII masking | `AI_MaskPII` (detects PII + returns `Masked_Phrase` with `*` replacement) | `ai-text-analytics` |
|
|
211
|
-
| External NER (general named entities — people, places, orgs, dates) | `AI_RecognizeEntities` | `ai-text-analytics` |
|
|
212
|
-
| External PII entity detection (structured metadata, no masking) | `AI_RecognizePIIEntities` | `ai-text-analytics` |
|
|
213
|
-
| External text classification with custom label set | `AI_TextClassifier` — supports single-label and multi-label | `ai-text-analytics` |
|
|
214
|
-
| External summarization | `AI_TextSummarize` — supports 1–5 compression levels | `ai-text-analytics` |
|
|
215
|
-
| External translation | `AI_TextTranslate` | `ai-text-analytics` |
|
|
216
|
-
|
|
217
|
-
**NER/PII function selection:**
|
|
218
|
-
|
|
219
|
-
| Need | Function |
|
|
220
|
-
|------|----------|
|
|
221
|
-
| General named entities (people, places, orgs, dates) | `AI_RecognizeEntities` |
|
|
222
|
-
| PII detection + masked text output | `AI_MaskPII` |
|
|
223
|
-
| PII detection + structured metadata only (no masking) | `AI_RecognizePIIEntities` |
|
|
224
|
-
|
|
225
198
|
### Vector Search
|
|
226
199
|
|
|
227
200
|
| Instead of this | Use this (native function) | Topic |
|
|
@@ -230,68 +203,6 @@ Native functions distribute across all AMPs. The result set returned to the agen
|
|
|
230
203
|
| External approximate nearest neighbor index | `TD_HNSW` / `TD_HNSWPredict` | `vector-search` |
|
|
231
204
|
| External embedding storage type | `VECTOR` / `Vector32` data type | `data-types-casting` |
|
|
232
205
|
|
|
233
|
-
**Inline NL Query → Embedding → Vector Search:** For RAG retrieval, use the full CTE pipeline pattern — embed the query inline with `AI_TEXTEMBEDDINGS`, normalize with `TD_VectorNormalize(Approach('UNITVECTOR'))`, and search with `TD_VectorDistance` against a pre-built corpus embedding table, all in a single SQL statement. See `vector-search` topic, "Inline NL Query → Embedding → Vector Search Pipeline" section.
|
|
234
|
-
|
|
235
|
-
**Building a corpus embedding table:** Use the full workflow: source text table → `AI_TEXTEMBEDDINGS` → `TD_VectorNormalize(Approach('UNITVECTOR'))` → CTAS. See `vector-search` topic, "Full Corpus Build Workflow" section.
|
|
236
|
-
|
|
237
|
-
**Vector dimension introspection:** Use `embedding.LENGTH()` to get the number of dimensions from a VECTOR column — never infer from UDT byte size. `SELECT embedding.LENGTH() AS dims FROM db.table SAMPLE 1;`
|
|
238
|
-
|
|
239
|
-
**Finding the embedding model for an existing corpus:** Query `TD_SYSAI.TD_CollectionsV` or `TD_SYSAI.TD_VectorStores` to discover the model name, provider, and embedding size used to build a corpus. The query pipeline must use the exact same model — mismatched embeddings produce meaningless scores. See `vector-search` topic, "Discovering Existing Vector Stores" section.
|
|
240
|
-
|
|
241
|
-
### Embeddings
|
|
242
|
-
|
|
243
|
-
| Scenario | Use this | Topic |
|
|
244
|
-
|----------|---------|-------|
|
|
245
|
-
| REST-based embedding API (Azure, AWS Bedrock, GCP, NVIDIA NIM, LiteLLM) | `AI_TextEmbeddings` with `OutputFormat('VECTOR')` | `embeddings` |
|
|
246
|
-
| In-database inference — no external API (air-gapped or latency-sensitive) | `ONNXEmbeddings` — model stored as BLOB in Vantage; requires tokenizer table | `embeddings`, `byom-model-loading` |
|
|
247
|
-
| Classical word/document embeddings (GloVe-style) | `TD_WordEmbeddings` | `text-analytics` |
|
|
248
|
-
| Store embeddings for reuse | CTAS with `VECTOR` column, then `TD_VectorNormalize(Approach('UNITVECTOR'))` at storage time | `embeddings`, `vector-search` |
|
|
249
|
-
| Build fast approximate search index | `TD_HNSW` on normalized VECTOR column | `vector-search` |
|
|
250
|
-
|
|
251
|
-
> **Always use `OutputFormat('VECTOR')`** when embeddings will be stored, normalized, or used with `TD_VectorDistance` / `TD_HNSW` / `TD_HNSWPredict`. The `VECTOR` type integrates directly with all vector search functions. See `data-types-casting` for VECTOR sizing (bytes, not dimensions).
|
|
252
|
-
|
|
253
|
-
### JSON Data
|
|
254
|
-
|
|
255
|
-
| Operation | Use this | Topic |
|
|
256
|
-
|-----------|---------|-------|
|
|
257
|
-
| Store JSON in a column | Native `JSON` type — `JSON(n)`, `STORAGE FORMAT BSON\|UBJSON` | `json-functions` |
|
|
258
|
-
| Extract a scalar value from JSON | `JSONExtractValue('$.path')` or dot notation `j.field` | `json-functions` |
|
|
259
|
-
| Extract multiple values / array | `JSONExtract('$..field')` → JSON array | `json-functions` |
|
|
260
|
-
| Type-safe extraction (returns NULL on failure) | `JSONGETVALUE(j, '$.age' AS INTEGER)` | `json-functions` |
|
|
261
|
-
| Check if a JSON path exists | `j.ExistValue('$.path')` → 1/0 | `json-functions` |
|
|
262
|
-
| List all key paths in a document | `JSON_KEYS(ON (...) USING QUOTES('N'))` | `json-functions` |
|
|
263
|
-
| Validate JSON string before loading | `JSON_CHECK('...')` → 'OK' or 'INVALID: reason' | `json-functions` |
|
|
264
|
-
| Validate BSON binary before loading | `BSON_CHECK(bytes_col)` → 'OK' or 'INVALID: reason' | `json-functions` |
|
|
265
|
-
| Convert binary JSON to text | `j.AsJSONText()` | `json-functions` |
|
|
266
|
-
| Convert JSON to BSON | `j.AsBSON()` or `CAST(j AS JSON STORAGE FORMAT BSON)` | `json-functions` |
|
|
267
|
-
| Estimate storage size | `j.StorageSize('BSON')` | `json-functions` |
|
|
268
|
-
| SQL rows → JSON doc (simple) | `SELECT AS JSON col1, col2 FROM t` | `json-functions` |
|
|
269
|
-
| SQL rows → JSON doc (hierarchical) | `JSON_COMPOSE(col, JSON_AGG(...) AS nested)` | `json-functions` |
|
|
270
|
-
| SQL rows → JSON doc (any format, >64K) | `JSON_PUBLISH` table operator | `json-functions` |
|
|
271
|
-
| JSON → relational rows (JSONPath) | `JSON_TABLE(ON (...) USING ROWEXPR(...) COLEXPR(...))` | `json-functions` |
|
|
272
|
-
| JSON → relational rows (fast, CLOB) | `TD_JSONSHRED(ON (...) USING ROWEXPR(...) COLEXPR(...) RETURNTYPES(...))` | `json-functions` |
|
|
273
|
-
| JSON → existing relational tables | `CALL SYSLIB.JSON_SHRED_BATCH(...)` | `json-functions` |
|
|
274
|
-
| NVP string → JSON | `NVP2JSON('k=v&k2=v2')` | `json-functions` |
|
|
275
|
-
| Vantage ARRAY → JSON | `ARRAY_TO_JSON(arr_col)` | `json-functions` |
|
|
276
|
-
| ST_Geometry ↔ GeoJSON | `GeoJSONFromGeom(geom)` / `GeomFromGeoJSON(json, srid)` | `json-functions` |
|
|
277
|
-
|
|
278
|
-
### BYOM — Bring Your Own Model
|
|
279
|
-
|
|
280
|
-
Apply externally trained models to in-database data without moving data out of Teradata. All scoring functions share the same two-table pattern: `InputTable` + `ModelTable DIMENSION`. See `byom-model-loading` topic for model ingestion; see `byom-scoring` for full syntax.
|
|
281
|
-
|
|
282
|
-
| Instead of this | Use this (native function) | Topic |
|
|
283
|
-
|-----------------|---------------------------|-------|
|
|
284
|
-
| Running PMML model inference externally | `PMMLPredict` | `byom-scoring` |
|
|
285
|
-
| Running H2O MOJO or Driverless AI model externally | `H2OPredict` (supports contributions, stage probabilities, leaf node assignments) | `byom-scoring` |
|
|
286
|
-
| Running ONNX tabular model externally | `ONNXPredict` (use `ShowModelInputFieldsMap('true')` to inspect tensor mapping) | `byom-scoring` |
|
|
287
|
-
| Running Dataiku Thin JAR externally | `DataikuPredict` (model_id = fully qualified Java class name) | `byom-scoring` |
|
|
288
|
-
| Running DataRobot Scoring Code externally | `DataRobotPredict` (cast DATE/TIMESTAMP to VARCHAR before scoring) | `byom-scoring` |
|
|
289
|
-
| Running MLeap model externally | `MLeapPredict` | `byom-scoring` |
|
|
290
|
-
| In-database text generation / seq-to-seq (translation, summarization) | `ONNXSeq2Seq` — ONNX transformer, no external API | `byom-scoring` |
|
|
291
|
-
| In-database text classification (transformer) | `ONNXClassification` — supports softmax, argmax, custom output column mapping | `byom-scoring` |
|
|
292
|
-
|
|
293
|
-
> **Architecture distinction:** `ONNXSeq2Seq` and `ONNXClassification` run Hugging Face ONNX transformer models entirely in-database. For REST-based LLM text tasks (sentiment, PII, translation, summarization), use the `AI_*` functions in `ai-text-analytics` instead.
|
|
294
|
-
|
|
295
206
|
### Statistical Testing
|
|
296
207
|
|
|
297
208
|
| Instead of this | Use this (native function) | Topic |
|
|
@@ -397,7 +308,7 @@ For any non-trivial query, run `explain_query` before executing. Read the plan a
|
|
|
397
308
|
- `duplicated on all AMPs` on a small table — correct broadcast strategy
|
|
398
309
|
- `execute the following steps in parallel` — independent steps dispatched concurrently
|
|
399
310
|
|
|
400
|
-
For full EXPLAIN interpretation guidance, optimization playbook, stats collection patterns
|
|
311
|
+
For full EXPLAIN interpretation guidance, optimization playbook, and stats collection patterns: `get_syntax_help(topic="query-tuning")`.
|
|
401
312
|
|
|
402
313
|
---
|
|
403
314
|
|
|
@@ -431,8 +342,8 @@ Native functions do not cover everything. Use hand-written SQL for:
|
|
|
431
342
|
- CASE expressions and NULL handling (`conditional`)
|
|
432
343
|
- Window functions for lag/lead features, running totals (`window-functions`)
|
|
433
344
|
- Schema discovery via MCP tools first — `list_databases`, `list_tables`, `describe_table` cover the common cases; fall back to manual DBC.* queries only for capabilities not covered by those tools (`catalog-views`)
|
|
434
|
-
- Bit/byte manipulation — `BITAND`, `BITOR`, `BITXOR`, `BITNOT`, `SHIFTLEFT`/`SHIFTRIGHT`, `ROTATELEFT`/`ROTATERIGHT`, `GETBIT`, `SETBIT`, `COUNTSET`, `SUBBITSTR`, `TO_BYTE` —
|
|
435
|
-
- JSON data — native `JSON` type with BSON/UBJSON binary formats; JSONPath extraction; shredding
|
|
345
|
+
- Bit/byte manipulation — `BITAND`, `BITOR`, `BITXOR`, `BITNOT`, `SHIFTLEFT`/`SHIFTRIGHT`, `ROTATELEFT`/`ROTATERIGHT`, `GETBIT`, `SETBIT`, `COUNTSET`, `SUBBITSTR`, `TO_BYTE` — no ANSI equivalents; do not use `&`, `|`, `^`, `~` operators (`bit-byte-functions`)
|
|
346
|
+
- JSON data — native `JSON` type with BSON/UBJSON binary formats; JSONPath extraction; shredding and publishing — all in-database (`json-functions`)
|
|
436
347
|
- One-off computations not covered by any native function
|
|
437
348
|
|
|
438
349
|
If you are unsure whether a native function exists for an operation, call `get_syntax_help(topic='index')` and check.
|
|
@@ -75,7 +75,7 @@ Use `get_syntax_help(topic="<name>")` to load any topic below.
|
|
|
75
75
|
| Topic | Description |
|
|
76
76
|
|-------|-------------|
|
|
77
77
|
| `catalog-views` | DBC.* system views for schema discovery |
|
|
78
|
-
| `query-tuning` | EXPLAIN, PI design,
|
|
78
|
+
| `query-tuning` | EXPLAIN, PI design, collect stats, query rewrite tips |
|
|
79
79
|
| `authorization-objects` | CREATE/REPLACE/GRANT authorization objects for external service credentials (AI functions, external procedures) |
|
|
80
80
|
| `llm-providers` | LLM provider argument blocks for AI functions — Azure, AWS Bedrock, GCP, NVIDIA NIM, LiteLLM |
|
|
81
81
|
| `byom-model-loading` | Loading PMML, H2O MOJO, ONNX, Dataiku, DataRobot, and MLeap models into Teradata BYOM tables; conversion workflow for ONNX embedding models |
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
# Teradata Query Tuning & EXPLAIN
|
|
2
|
+
|
|
3
|
+
## EXPLAIN
|
|
4
|
+
Validate syntax and preview the execution plan without running the query:
|
|
5
|
+
```sql
|
|
6
|
+
EXPLAIN SELECT col1, col2 FROM db.table WHERE id = 1;
|
|
7
|
+
|
|
8
|
+
-- Dynamic EXPLAIN: better for correlated subqueries, multi-level nesting,
|
|
9
|
+
-- or queries where intermediate sizes dramatically change the plan
|
|
10
|
+
DYNAMIC EXPLAIN SELECT col1, col2 FROM db.table WHERE id = 1;
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
Use `explain_query` before executing any non-trivial query. If the plan shows critical issues (see below), fix and re-EXPLAIN before running.
|
|
14
|
+
|
|
15
|
+
### EXPLAIN Phrase Glossary
|
|
16
|
+
|
|
17
|
+
| Phrase | Optimizer Intent |
|
|
18
|
+
|--------|-----------------|
|
|
19
|
+
| `single-AMP RETRIEVE by way of the unique primary index` | Best-case tactical access — no spool, no redistribution |
|
|
20
|
+
| `by way of an all-rows scan` | Full table scan — examine predicates, stats, and partitioning |
|
|
21
|
+
| `redistributed by hash code` | Row movement to co-locate join keys — check skew risk |
|
|
22
|
+
| `duplicated on all AMPs` | Broadcast small input to all AMPs — verify it's truly small |
|
|
23
|
+
| `(group_amps)` | Spool built on subset of AMPs — potential skew signal |
|
|
24
|
+
| `(all_amps)` | Spool built on every AMP — expected for large tables |
|
|
25
|
+
| `SORT to order Spool n by row hash` | Preparing for merge/hash join |
|
|
26
|
+
| `estimated with no confidence` | Missing statistics — unreliable cardinality estimate |
|
|
27
|
+
| `estimated with low confidence` | Partial or stale statistics — conservative planning |
|
|
28
|
+
| `estimated with high confidence` | Good statistics — optimizer has reliable estimates |
|
|
29
|
+
| `index join confidence` | Optimizer used index stats — reliable |
|
|
30
|
+
| `execute the following steps in parallel` | Independent sub-steps dispatched concurrently |
|
|
31
|
+
| `RowHash match scan` | Join driven by rowhash ordering |
|
|
32
|
+
| `eliminating duplicate rows` | DISTINCT or duplicate removal in spool |
|
|
33
|
+
| `hash join` | Standard large-table join — efficient when well-sized |
|
|
34
|
+
| `merge join` | Sorted join — efficient when inputs already sorted by join key |
|
|
35
|
+
| `product join` | Cartesian join — almost always a problem on large tables |
|
|
36
|
+
| `nested join` | Driven by index — efficient for small outer inputs |
|
|
37
|
+
|
|
38
|
+
### Severity Classification
|
|
39
|
+
|
|
40
|
+
**Critical — fix before executing:**
|
|
41
|
+
- `no confidence` on large tables
|
|
42
|
+
- `product join` on large tables (potential Cartesian explosion)
|
|
43
|
+
- `all-rows scan` on very large tables (>1M rows) without a clear reason
|
|
44
|
+
- Steps consuming >15% of total estimated time
|
|
45
|
+
- `(group_amps)` materializing on very few AMPs (severe skew)
|
|
46
|
+
- Massive intermediate spool (>1 GB estimated)
|
|
47
|
+
|
|
48
|
+
**Warning — should investigate:**
|
|
49
|
+
- `low confidence` on join or filter columns
|
|
50
|
+
- Steps consuming 5–15% of total estimated time
|
|
51
|
+
- Large redistributions (>500 MB spool)
|
|
52
|
+
- Secondary index scans with large result sets
|
|
53
|
+
- Multiple sequential redistributions
|
|
54
|
+
- Missing partition elimination on a partitioned table
|
|
55
|
+
|
|
56
|
+
**Good — plan is efficient:**
|
|
57
|
+
- `single-AMP RETRIEVE by way of the unique primary index`
|
|
58
|
+
- `high confidence` or `index join confidence` estimates
|
|
59
|
+
- `duplicated on all AMPs` on a provably small table
|
|
60
|
+
- `execute the following steps in parallel`
|
|
61
|
+
- Spool `(Last Use)` markers correctly placed (spool freed promptly)
|
|
62
|
+
- Local aggregation (no redistribution needed)
|
|
63
|
+
|
|
64
|
+
## Collect Statistics
|
|
65
|
+
Missing stats = bad plans. Collect on PI columns, join columns, and WHERE-clause columns:
|
|
66
|
+
```sql
|
|
67
|
+
COLLECT STATISTICS ON db.table COLUMN (id);
|
|
68
|
+
COLLECT STATISTICS ON db.table COLUMN (customer_id, order_date); -- composite
|
|
69
|
+
COLLECT STATISTICS ON db.table INDEX (primary_index_col);
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
## Primary Index (PI) Design Principles
|
|
73
|
+
- The PI determines how rows are distributed across AMPs
|
|
74
|
+
- Good PI: high cardinality, frequently used in joins/filters
|
|
75
|
+
- Bad PI: low cardinality (e.g., boolean, status) → AMP skew
|
|
76
|
+
- Check for skew:
|
|
77
|
+
```sql
|
|
78
|
+
SELECT Hashamp() + 1 AS amp, COUNT(*) AS row_count
|
|
79
|
+
FROM db.table
|
|
80
|
+
GROUP BY Hashamp()
|
|
81
|
+
ORDER BY row_count DESC;
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
## NoPI Tables (Staging / Load)
|
|
85
|
+
```sql
|
|
86
|
+
CREATE MULTISET TABLE db.staging_table, NO PRIMARY INDEX AS
|
|
87
|
+
(SELECT * FROM db.source WHERE 1=0);
|
|
88
|
+
```
|
|
89
|
+
Use NoPI for temp/staging tables where you don't know the access pattern yet.
|
|
90
|
+
|
|
91
|
+
## Volatile Tables (Session-Scoped Temp Tables)
|
|
92
|
+
```sql
|
|
93
|
+
CREATE VOLATILE TABLE tmp_results AS (
|
|
94
|
+
SELECT customer_id, SUM(amount) AS total
|
|
95
|
+
FROM db.orders
|
|
96
|
+
GROUP BY customer_id
|
|
97
|
+
) WITH DATA
|
|
98
|
+
ON COMMIT PRESERVE ROWS;
|
|
99
|
+
|
|
100
|
+
-- Use in subsequent queries
|
|
101
|
+
SELECT * FROM tmp_results WHERE total > 1000;
|
|
102
|
+
```
|
|
103
|
+
Volatile tables exist only for the session duration — no cleanup needed.
|
|
104
|
+
|
|
105
|
+
## Derived Tables vs CTEs
|
|
106
|
+
Both are equivalent in Teradata. CTEs are generally more readable:
|
|
107
|
+
```sql
|
|
108
|
+
-- CTE (preferred for multi-step logic)
|
|
109
|
+
WITH base AS (SELECT ... FROM db.t WHERE ...),
|
|
110
|
+
agg AS (SELECT id, SUM(val) FROM base GROUP BY id)
|
|
111
|
+
SELECT * FROM agg;
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
## Filtering Early
|
|
115
|
+
Push filters as close to the base table as possible:
|
|
116
|
+
```sql
|
|
117
|
+
-- Better: filter before join
|
|
118
|
+
SELECT a.*, b.name
|
|
119
|
+
FROM (SELECT * FROM db.orders WHERE order_date >= CURRENT_DATE - 30) a
|
|
120
|
+
JOIN db.customers b ON a.customer_id = b.id;
|
|
121
|
+
|
|
122
|
+
-- Worse: filter after join
|
|
123
|
+
SELECT a.*, b.name
|
|
124
|
+
FROM db.orders a JOIN db.customers b ON a.customer_id = b.id
|
|
125
|
+
WHERE a.order_date >= CURRENT_DATE - 30;
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
## Avoiding Common Anti-Patterns
|
|
129
|
+
```sql
|
|
130
|
+
-- Avoid functions on indexed columns in WHERE (prevents PI lookup)
|
|
131
|
+
-- Bad:
|
|
132
|
+
WHERE EXTRACT(YEAR FROM order_date) = 2024
|
|
133
|
+
-- Better:
|
|
134
|
+
WHERE order_date BETWEEN DATE '2024-01-01' AND DATE '2024-12-31'
|
|
135
|
+
|
|
136
|
+
-- Avoid implicit type conversions in joins
|
|
137
|
+
-- Bad (CHAR vs VARCHAR mismatch):
|
|
138
|
+
WHERE char_col = varchar_col
|
|
139
|
+
-- Better: explicit CAST
|
|
140
|
+
WHERE CAST(char_col AS VARCHAR(50)) = varchar_col
|
|
141
|
+
|
|
142
|
+
-- Avoid SELECT * in production queries — enumerate needed columns
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## Session-Level Tuning
|
|
146
|
+
```sql
|
|
147
|
+
-- Show current session info
|
|
148
|
+
SELECT * FROM DBC.SessionInfoV WHERE SessionNo = SESSION;
|
|
149
|
+
|
|
150
|
+
-- Increase spool space limit for a session (if you have rights)
|
|
151
|
+
-- Usually set by DBA at user/profile level
|
|
152
|
+
```
|
|
@@ -118,21 +118,6 @@ CREATE TABLE db.my_table AS (
|
|
|
118
118
|
-- CREATE OR REPLACE TABLE does not exist — to replace: DROP then CREATE, or use a staging pattern
|
|
119
119
|
```
|
|
120
120
|
|
|
121
|
-
### CREATE USER — DEFAULT CHARACTER SET
|
|
122
|
-
|
|
123
|
-
```sql
|
|
124
|
-
-- WRONG: CHARACTER SET is not valid at the user level
|
|
125
|
-
CREATE USER myuser AS PASSWORD = 'secret' CHARACTER SET UNICODE;
|
|
126
|
-
|
|
127
|
-
-- RIGHT: the clause is DEFAULT CHARACTER SET
|
|
128
|
-
CREATE USER myuser AS
|
|
129
|
-
PASSWORD = 'secret'
|
|
130
|
-
DEFAULT DATABASE = mydb
|
|
131
|
-
DEFAULT CHARACTER SET UNICODE;
|
|
132
|
-
```
|
|
133
|
-
|
|
134
|
-
> **`DEFAULT CHARACTER SET`**, not `CHARACTER SET`, is the correct syntax in `CREATE USER` and `MODIFY USER`. Using `CHARACTER SET` alone will fail.
|
|
135
|
-
|
|
136
121
|
### Other DDL reminders
|
|
137
122
|
```sql
|
|
138
123
|
-- Teradata uses MINUS, not EXCEPT (already noted in Set Operations above)
|
|
@@ -140,75 +125,15 @@ CREATE USER myuser AS
|
|
|
140
125
|
-- Semicolons: required in BTEQ; optional in most client tools
|
|
141
126
|
```
|
|
142
127
|
|
|
143
|
-
## Reserved Words
|
|
128
|
+
## Reserved Words as Column Names in Table Operator Clauses
|
|
144
129
|
|
|
145
|
-
|
|
130
|
+
When a column name is a Teradata reserved word (e.g. `type`, `date`, `time`, `value`, `name`, `format`, `title`), it must be double-quoted. In regular SQL projections this looks normal:
|
|
146
131
|
|
|
147
132
|
```sql
|
|
148
|
-
SELECT
|
|
149
|
-
|
|
150
|
-
CREATE TABLE db.events (
|
|
151
|
-
id INTEGER,
|
|
152
|
-
"type" VARCHAR(30), -- reserved word — must quote
|
|
153
|
-
label VARCHAR(100)
|
|
154
|
-
) PRIMARY INDEX (id);
|
|
133
|
+
SELECT "type", "date" FROM db.my_table;
|
|
155
134
|
```
|
|
156
135
|
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
### Teradata-Specific Reserved Words — Common Identifier Conflicts
|
|
160
|
-
|
|
161
|
-
The ANSI SQL reserved words are well-known. The words below are **Teradata-only** — not in the ANSI SQL-99 standard — so agents may not recognize them as reserved. Always quote these when using them as column, table, or alias names.
|
|
162
|
-
|
|
163
|
-
**Words that frequently appear as column or table names:**
|
|
164
|
-
|
|
165
|
-
| Reserved word | Commonly appears as | TD since |
|
|
166
|
-
|--------------|---------------------|----------|
|
|
167
|
-
| `TYPE` | transaction type, event type, record type | V2R3 |
|
|
168
|
-
| `FORMAT` | file format, output format, date format | V2R3 |
|
|
169
|
-
| `TITLE` | document title; also controls the TD column display header | V2R3 |
|
|
170
|
-
| `MODE` | processing mode, run mode, lock mode | V2R3 |
|
|
171
|
-
| `ACCOUNT` | account_id, account-related tables | V2R3 |
|
|
172
|
-
| `LOG` | log tables, audit logs, log level | V2R3 |
|
|
173
|
-
| `LOCK` | lock status, concurrency tables | V2R3 |
|
|
174
|
-
| `HASH` | hash keys, checksums, deduplication columns | V2R3 |
|
|
175
|
-
| `REQUEST` | request_id, API and service event tables | V2R3 |
|
|
176
|
-
| `STATISTICS` | monitoring tables, collected stats columns | V2R3 |
|
|
177
|
-
| `JOURNAL` | financial journals, transaction audit logs | V2R3 |
|
|
178
|
-
| `CLUSTER` | cluster_id, partition or segment label | V2R3 |
|
|
179
|
-
| `NAMED` | TD column alias syntax (`expr (NAMED alias)`) — risky as a column name | V2R3 |
|
|
180
|
-
| `RANK` | search rank, priority rank, competition rank | V2R3 |
|
|
181
|
-
| `DATABASE` | database name columns in catalog/metadata tables | V2R3 |
|
|
182
|
-
| `PERCENT` | percent_change, completion_percent, analytics columns | V2R3 |
|
|
183
|
-
| `ENABLED` / `DISABLED` | feature flag columns, configuration status | V2R3 |
|
|
184
|
-
| `CLASS` | object class, classification, CSS class | V2R5 |
|
|
185
|
-
| `PROFILE` | user profiles, configuration profiles | V2R5 |
|
|
186
|
-
| `SUMMARY` | summary text columns, reporting tables | V2R5 |
|
|
187
|
-
| `THRESHOLD` | alert thresholds, monitoring limit columns | V2R5 |
|
|
188
|
-
| `TRACE` | trace_id, debug or telemetry columns | V2R5 |
|
|
189
|
-
|
|
190
|
-
**Teradata SQL extension keywords — these are clause keywords, not identifiers:**
|
|
191
|
-
|
|
192
|
-
| Keyword | Purpose |
|
|
193
|
-
|---------|---------|
|
|
194
|
-
| `QUALIFY` | Filters window function results — like WHERE for OVER clauses; not in ANSI SQL |
|
|
195
|
-
| `SAMPLE` | Random row sampling: `SELECT * FROM t SAMPLE 100` or `SAMPLE .05` |
|
|
196
|
-
| `VOLATILE` | Session-scoped temp table: `CREATE VOLATILE TABLE ...` |
|
|
197
|
-
| `LOCKING` | Lock modifier: `LOCKING TABLE t FOR ACCESS SELECT ...` |
|
|
198
|
-
| `REPLACE` | TD DDL: `REPLACE VIEW` — not `CREATE OR REPLACE` |
|
|
199
|
-
| `EXPLAIN` | Execution plan: `EXPLAIN SELECT ...` |
|
|
200
|
-
| `FALLBACK` | Table-level data protection option at `CREATE TABLE` time |
|
|
201
|
-
| `MULTISET` | Table type allowing duplicate rows: `CREATE MULTISET TABLE ...` |
|
|
202
|
-
| `MACRO` | Stored parameterized query: `CREATE MACRO ...` |
|
|
203
|
-
| `COLLECT` | Statistics collection: `COLLECT STATISTICS ON db.t COLUMN (col)` |
|
|
204
|
-
| `BT` / `ET` | Begin Transaction / End Transaction |
|
|
205
|
-
| `SEL` | Shorthand for `SELECT` |
|
|
206
|
-
| `DEL` / `INS` / `UPD` | Shorthands for DELETE / INSERT / UPDATE — `DEL` and `INS` can appear as column aliases in CDC/audit schemas |
|
|
207
|
-
| `CM` / `CT` / `CD` / `CS` / `CV` / `SS` / `UC` | Additional 2-letter Teradata BTEQ abbreviations — all reserved |
|
|
208
|
-
|
|
209
|
-
### Reserved Words in Table Operator String Arguments
|
|
210
|
-
|
|
211
|
-
In table operator clauses that take column names as **string arguments** (`Accumulate`, `IDColumn`, `TargetColumns`, `ResponseColumn`, etc.), double-quotes must be embedded **inside** the single-quoted string:
|
|
136
|
+
In table operator string arguments (`ACCUMULATE`, `IDColumn`, `TargetColumns`, `Accumulate`, etc.), the double-quotes must be embedded **inside** the single-quoted string:
|
|
212
137
|
|
|
213
138
|
```sql
|
|
214
139
|
-- WRONG: 'type' is a reserved word — Teradata will reject or misparse this
|
|
@@ -218,16 +143,19 @@ USING IDColumn('id') Accumulate('type', 'value')
|
|
|
218
143
|
USING IDColumn('id') Accumulate('"type"', '"value"')
|
|
219
144
|
```
|
|
220
145
|
|
|
221
|
-
This applies to any table operator clause that takes column names as string arguments:
|
|
146
|
+
This applies to **any** table operator clause that takes column names as string arguments:
|
|
222
147
|
|
|
223
148
|
```sql
|
|
149
|
+
-- All of these follow the same rule
|
|
224
150
|
IDColumn('"type"')
|
|
225
151
|
TargetColumns('"value"', '"date"', 'non_reserved_col')
|
|
226
152
|
Accumulate('"type"', 'amount', '"date"')
|
|
227
153
|
ResponseColumn('"value"')
|
|
228
154
|
```
|
|
229
155
|
|
|
230
|
-
**When in doubt, quote it.** Double-quoting a non-reserved word
|
|
156
|
+
**When in doubt, quote it.** Double-quoting a non-reserved word in a string argument is harmless; leaving a reserved word unquoted will cause a parse error.
|
|
157
|
+
|
|
158
|
+
Common Teradata reserved words that appear as column names: `type`, `date`, `time`, `timestamp`, `value`, `name`, `format`, `title`, `level`, `mode`, `status`, `class`, `key`, `index`, `year`, `month`, `day`, `hour`, `minute`, `second`.
|
|
231
159
|
|
|
232
160
|
## Teradata Operator Differences
|
|
233
161
|
|
|
@@ -251,29 +179,20 @@ WHERE status <> 'active'
|
|
|
251
179
|
|
|
252
180
|
### Three-Tier Dot Notation (OTF Tables)
|
|
253
181
|
|
|
254
|
-
Open Table Format tables (Iceberg, Delta Lake) use three-tier notation: `datalake.database.table`.
|
|
182
|
+
Open Table Format tables (Iceberg, Delta Lake) use three-tier notation: `datalake.database.table`.
|
|
255
183
|
|
|
256
184
|
```sql
|
|
257
|
-
-- Query OTF table
|
|
258
185
|
SELECT * FROM my_lake.sales_db.orders WHERE order_date >= DATE '2024-01-01';
|
|
259
186
|
|
|
260
187
|
-- Join OTF table with a relational table
|
|
261
188
|
SELECT o.order_id, c.name
|
|
262
189
|
FROM my_lake.sales_db.orders o
|
|
263
190
|
JOIN mydb.customers c ON o.customer_id = c.customer_id;
|
|
264
|
-
|
|
265
|
-
-- CTAS from OTF into a relational table
|
|
266
|
-
CREATE TABLE mydb.orders_local AS (
|
|
267
|
-
SELECT * FROM my_lake.sales_db.orders
|
|
268
|
-
) WITH DATA;
|
|
269
|
-
|
|
270
|
-
-- Reference OTF database
|
|
271
|
-
HELP DATABASE my_lake.sales_db;
|
|
272
191
|
```
|
|
273
192
|
|
|
274
193
|
### HELP Commands — Pass Through as-is
|
|
275
194
|
|
|
276
|
-
`HELP DATALAKE`, `HELP DATABASE`, and `HELP TABLE` are first-class Teradata statements for OTF and foreign table metadata. Pass them through `execute_query` exactly as written — do NOT rewrite as SELECT queries
|
|
195
|
+
`HELP DATALAKE`, `HELP DATABASE`, and `HELP TABLE` are first-class Teradata statements for OTF and foreign table metadata. Pass them through `execute_query` exactly as written — do NOT rewrite as SELECT queries.
|
|
277
196
|
|
|
278
197
|
```sql
|
|
279
198
|
HELP DATALAKE my_lake; -- list databases in datalake
|
|
@@ -282,12 +201,10 @@ HELP TABLE my_lake.sales_db.orders; -- describe OTF table columns
|
|
|
282
201
|
HELP TABLE mydb.my_foreign_table; -- describe foreign table columns
|
|
283
202
|
```
|
|
284
203
|
|
|
285
|
-
`describe_table` (which queries `DBC.ColumnsV`) does not work for OTF tables or NOS foreign tables.
|
|
204
|
+
`describe_table` (which queries `DBC.ColumnsV`) does not work for OTF tables or NOS foreign tables.
|
|
286
205
|
|
|
287
206
|
### READ_NOS — Two Equivalent Calling Forms
|
|
288
207
|
|
|
289
|
-
You may encounter READ_NOS written in two forms — both are valid and produce identical results:
|
|
290
|
-
|
|
291
208
|
```sql
|
|
292
209
|
-- Explicit form — generate this when writing new SQL
|
|
293
210
|
SELECT * FROM READ_NOS (
|
|
@@ -304,5 +221,3 @@ SELECT * FROM (
|
|
|
304
221
|
RETURNTYPE='NOSREAD_KEYS'
|
|
305
222
|
) AS d;
|
|
306
223
|
```
|
|
307
|
-
|
|
308
|
-
Always generate the explicit `READ_NOS(USING(...))` form when writing new SQL.
|