zuplo 7.9.4 → 7.9.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/ai-gateway/apps.mdx +28 -16
- package/docs/ai-gateway/custom-policies.mdx +7 -7
- package/docs/ai-gateway/custom-providers.mdx +1 -1
- package/docs/ai-gateway/fallback.mdx +5 -5
- package/docs/ai-gateway/integrations/ai-sdk.mdx +2 -2
- package/docs/ai-gateway/integrations/claude-code.mdx +41 -33
- package/docs/ai-gateway/integrations/claude-desktop.mdx +37 -34
- package/docs/ai-gateway/integrations/codex.mdx +42 -31
- package/docs/ai-gateway/integrations/github-copilot.mdx +39 -37
- package/docs/ai-gateway/integrations/goose.mdx +27 -30
- package/docs/ai-gateway/integrations/langchain.mdx +2 -2
- package/docs/ai-gateway/integrations/openai.mdx +2 -2
- package/docs/ai-gateway/jev.mdx +344 -0
- package/docs/ai-gateway/managing-apps.mdx +22 -22
- package/docs/ai-gateway/managing-pools.mdx +87 -0
- package/docs/ai-gateway/managing-providers.mdx +4 -4
- package/docs/ai-gateway/overview.mdx +38 -26
- package/docs/ai-gateway/policy-chains.mdx +7 -7
- package/docs/ai-gateway/policy-templates.mdx +22 -22
- package/docs/ai-gateway/pools.mdx +53 -0
- package/docs/ai-gateway/providers.mdx +32 -2
- package/docs/ai-gateway/source-control.mdx +2 -2
- package/docs/ai-gateway/universal-api.mdx +4 -0
- package/docs/ai-gateway/usage-limits.mdx +49 -47
- package/docs/ai-gateway/user-apps.mdx +166 -0
- package/docs/articles/accounts/roles-and-permissions.mdx +92 -63
- package/docs/concepts/ai-gateway.mdx +24 -20
- package/docs/policies/ai-gateway-akamai-firewall-inbound/doc.md +1 -1
- package/docs/policies/ai-gateway-metering-inbound/schema.json +16 -2
- package/docs/policies/ai-gateway-smart-router-inbound/doc.md +172 -29
- package/docs/policies/ai-gateway-smart-router-inbound/intro.md +4 -3
- package/docs/policies/ai-gateway-smart-router-inbound/schema.json +177 -30
- package/package.json +5 -5
- package/docs/ai-gateway/managing-teams.mdx +0 -87
- package/docs/ai-gateway/teams.mdx +0 -49
|
@@ -2,46 +2,50 @@
|
|
|
2
2
|
title: AI Gateway Concepts
|
|
3
3
|
sidebar_label: AI Gateway
|
|
4
4
|
description:
|
|
5
|
-
The mental model behind the Zuplo AI Gateway — providers,
|
|
6
|
-
|
|
5
|
+
The mental model behind the Zuplo AI Gateway — providers, pools, apps, User
|
|
6
|
+
Apps, and policy chains — and how it all relates to the rest of the platform.
|
|
7
7
|
---
|
|
8
8
|
|
|
9
|
-
The Zuplo AI Gateway is a proxy that sits between your apps
|
|
10
|
-
like OpenAI, Anthropic, Google, Mistral, and xAI. Instead of each
|
|
11
|
-
provider API keys and calling providers directly, every request
|
|
12
|
-
the gateway, which applies policies, controls, and monitoring.
|
|
13
|
-
explains the concepts behind it. For the full feature tour, see the
|
|
9
|
+
The Zuplo AI Gateway is a proxy that sits between your apps, your people, and
|
|
10
|
+
LLM providers like OpenAI, Anthropic, Google, Mistral, and xAI. Instead of each
|
|
11
|
+
app holding provider API keys and calling providers directly, every request
|
|
12
|
+
flows through the gateway, which applies policies, controls, and monitoring.
|
|
13
|
+
This page explains the concepts behind it. For the full feature tour, see the
|
|
14
14
|
[AI Gateway overview](../ai-gateway/overview.mdx).
|
|
15
15
|
|
|
16
|
-
## Providers,
|
|
16
|
+
## Providers, pools, and apps
|
|
17
17
|
|
|
18
|
-
|
|
18
|
+
Four building blocks organize an AI Gateway:
|
|
19
19
|
|
|
20
20
|
- **[Providers](../ai-gateway/providers.mdx)** are the LLM vendors your gateway
|
|
21
21
|
can route to. Administrators configure providers once, including any
|
|
22
22
|
[custom OpenAI-compatible providers](../ai-gateway/custom-providers.mdx), and
|
|
23
23
|
consumers never see the underlying credentials.
|
|
24
|
-
- **[
|
|
25
|
-
|
|
26
|
-
the gateway, ancestor
|
|
24
|
+
- **[Pools](../ai-gateway/pools.mdx)** group apps and carry budget limits. Each
|
|
25
|
+
pool totals usage across its sub-pools and apps. A request must remain within
|
|
26
|
+
the gateway, ancestor pool, and app limits. See
|
|
27
27
|
[Usage Limits](../ai-gateway/usage-limits.mdx).
|
|
28
28
|
- **[Apps](../ai-gateway/apps.mdx)** are the pieces of software that call the
|
|
29
29
|
gateway — a support chatbot is one app, an internal coding agent is another.
|
|
30
30
|
Each app gets its own gateway URL, its own Zuplo-managed API key, and its own
|
|
31
31
|
policy chain.
|
|
32
|
+
- **The [User App](../ai-gateway/user-apps.mdx)** is how people call the gateway
|
|
33
|
+
from their own tools. Each project has one. People call its URL with their own
|
|
34
|
+
personal API key, and each person has their own budget.
|
|
32
35
|
|
|
33
36
|
## Policy chains
|
|
34
37
|
|
|
35
|
-
Every app runs its own ordered
|
|
36
|
-
model filtering, fallback
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
minute with
|
|
38
|
+
Every app, and the User App, runs its own ordered
|
|
39
|
+
[policy chain](../ai-gateway/policy-chains.mdx) — model filtering, fallback
|
|
40
|
+
models, budgets, semantic caching, guardrails, tracing, and any
|
|
41
|
+
[custom policies](../ai-gateway/custom-policies.mdx) written in TypeScript. Pool
|
|
42
|
+
[policy templates](../ai-gateway/policy-templates.mdx) give new apps a
|
|
43
|
+
consistent starting pipeline, and chain changes apply within about a minute with
|
|
44
|
+
no redeploy.
|
|
41
45
|
|
|
42
46
|
App-level controls are opt-in: an app with an empty policy chain applies no
|
|
43
47
|
model restrictions, budgets, guardrails, or caching of its own — though gateway
|
|
44
|
-
and
|
|
48
|
+
and pool usage limits still apply.
|
|
45
49
|
|
|
46
50
|
## The Universal API
|
|
47
51
|
|
|
@@ -58,7 +62,7 @@ code change.
|
|
|
58
62
|
An AI Gateway is a type of Zuplo project. A new project deploys without a Git
|
|
59
63
|
repository; you can [connect a repository](../ai-gateway/source-control.mdx)
|
|
60
64
|
later so gateway changes go through the same review workflow as the rest of your
|
|
61
|
-
Zuplo configuration. The providers,
|
|
65
|
+
Zuplo configuration. The providers, pools, and apps you configure belong to that
|
|
62
66
|
project. The platform primitives — deployment model, environments, analytics —
|
|
63
67
|
are the same ones described in [How Zuplo Works](./how-zuplo-works.mdx).
|
|
64
68
|
|
|
@@ -36,7 +36,7 @@ object; they do not merge with it.
|
|
|
36
36
|
|
|
37
37
|
Do not add this policy to the inbound chain of an app used as a
|
|
38
38
|
[Smart Router](/docs/policies/ai-gateway-smart-router-inbound) classifier
|
|
39
|
-
(`
|
|
39
|
+
(`classifier.app`). The firewall would inspect Smart Router's internal
|
|
40
40
|
classification call as if it were a completion request, and Smart Router fails
|
|
41
41
|
open on a resulting denial — so this combination is rejected with a
|
|
42
42
|
`ConfigurationError` instead of silently degrading classification.
|
|
@@ -73,6 +73,21 @@
|
|
|
73
73
|
"type": "object",
|
|
74
74
|
"additionalProperties": false,
|
|
75
75
|
"required": ["budgetBy", "meters"],
|
|
76
|
+
"if": {
|
|
77
|
+
"required": ["budgetBy"],
|
|
78
|
+
"properties": {
|
|
79
|
+
"budgetBy": {
|
|
80
|
+
"const": "app"
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
},
|
|
84
|
+
"then": {
|
|
85
|
+
"properties": {
|
|
86
|
+
"meters": {
|
|
87
|
+
"minItems": 1
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
},
|
|
76
91
|
"properties": {
|
|
77
92
|
"budgetBy": {
|
|
78
93
|
"type": "string",
|
|
@@ -86,9 +101,8 @@
|
|
|
86
101
|
},
|
|
87
102
|
"meters": {
|
|
88
103
|
"type": "array",
|
|
89
|
-
"minItems": 1,
|
|
90
104
|
"maxItems": 24,
|
|
91
|
-
"description": "Thresholds this rule enforces. Add one entry per meter, period, and action you want.",
|
|
105
|
+
"description": "Thresholds this rule enforces. Add one entry per meter, period, and action you want. An \"app\" rule needs at least one entry. An \"expression\" rule may leave this empty to track usage per value without enforcing a limit.",
|
|
92
106
|
"items": {
|
|
93
107
|
"type": "object",
|
|
94
108
|
"additionalProperties": false,
|
|
@@ -12,24 +12,125 @@ classifier prompt.
|
|
|
12
12
|
|
|
13
13
|
## Required options
|
|
14
14
|
|
|
15
|
-
- `
|
|
16
|
-
|
|
17
|
-
- `classifierModel` — `providerName/model` sent on the classifier request.
|
|
15
|
+
- `classifier` — where the prompt is classified. Set exactly one of `app`,
|
|
16
|
+
`chatCompletions`, or `jev`; see [Choose a classifier](#choose-a-classifier).
|
|
18
17
|
- `modelsByComplexity` — `providerName/model` for each of `low`, `medium`, and
|
|
19
18
|
`high`. Used for routing unless `smartRoutingEnabled` is set to `false`.
|
|
20
19
|
|
|
20
|
+
Omit `intents` and `classifierPrompt` to use the built-in dictionary (code,
|
|
21
|
+
summarization, translation, qa, conversation, classification, creative_writing,
|
|
22
|
+
agentic, document_qa, other) and the built-in classification prompt.
|
|
23
|
+
|
|
24
|
+
## Choose a classifier
|
|
25
|
+
|
|
26
|
+
| Classifier | Set | Use it when |
|
|
27
|
+
| ---------------------------- | ----------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
28
|
+
| `classifier.app` | `appId`, `model` | You want the classifier to use one of your providers, through an AI Gateway app in this project. The call stays in-process. |
|
|
29
|
+
| `classifier.chatCompletions` | `apiKey`, `model` | You want to call an OpenAI-compatible model directly with its own key, without a second AI Gateway app. |
|
|
30
|
+
| `classifier.jev` | `apiKey` | You want a model built for classification. Jev, from TypeSafe, answers typed questions instead of generating text. TypeSafe and OpenRouter both serve it. |
|
|
31
|
+
|
|
32
|
+
The three are mutually exclusive. Setting more than one is rejected with a
|
|
33
|
+
`ConfigurationError` rather than guessed at, because routing on a classifier you
|
|
34
|
+
didn't mean to use would look like it was working.
|
|
35
|
+
|
|
36
|
+
### AI Gateway app
|
|
37
|
+
|
|
38
|
+
```json
|
|
39
|
+
"classifier": {
|
|
40
|
+
"app": {
|
|
41
|
+
"appId": "$env(CLASSIFIER_APP_ID)",
|
|
42
|
+
"model": "openai/gpt-4o-mini"
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
- `appId` — the AI Gateway app whose Chat Completions endpoint runs the
|
|
48
|
+
classifier.
|
|
49
|
+
- `model` — the `providerName/model` that app routes the classifier request to.
|
|
50
|
+
- `apiKey` — optional. See
|
|
51
|
+
[Authenticating the classifier app](#authenticating-the-classifier-app).
|
|
52
|
+
|
|
21
53
|
Give the classifier its own application rather than pointing this at the app the
|
|
22
54
|
policy runs on. The classifier request never leaves the gateway, so that
|
|
23
55
|
application needs no API key checked at all — put the Ensure Gateway Internal
|
|
24
56
|
Invocation Only policy on it in place of API Key Authentication.
|
|
25
57
|
|
|
26
|
-
|
|
27
|
-
bearer token for it, typically `$env(CLASSIFIER_APP_API_KEY)`. Omit it, or leave
|
|
28
|
-
it empty, and the classifier request carries no `Authorization` header.
|
|
58
|
+
### Chat Completions API
|
|
29
59
|
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
60
|
+
```json
|
|
61
|
+
"classifier": {
|
|
62
|
+
"chatCompletions": {
|
|
63
|
+
"apiKey": "$env(OPENAI_API_KEY)",
|
|
64
|
+
"model": "gpt-4o-mini"
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
- `apiKey` — sent as `Authorization: Bearer`.
|
|
70
|
+
- `model` — the model id the service expects, such as `gpt-4o-mini`, not a
|
|
71
|
+
`providerName/model` reference. It must support structured outputs: the
|
|
72
|
+
classifier request asks for a strict JSON schema in `response_format`.
|
|
73
|
+
- `baseUrl` — optional, `https://api.openai.com/v1` by default. Set it to call
|
|
74
|
+
another OpenAI-compatible service. The request goes to
|
|
75
|
+
`{baseUrl}/chat/completions`.
|
|
76
|
+
|
|
77
|
+
To classify with an app in this project, use `classifier.app` rather than
|
|
78
|
+
pointing `baseUrl` at the app's URL. `classifier.app` invokes the app
|
|
79
|
+
in-process, and it rejects a classifier app that would call back into this
|
|
80
|
+
policy.
|
|
81
|
+
|
|
82
|
+
### Jev
|
|
83
|
+
|
|
84
|
+
```json
|
|
85
|
+
"classifier": {
|
|
86
|
+
"jev": {
|
|
87
|
+
"apiKey": "$env(TYPESAFE_API_KEY)"
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
- `apiKey` — your TypeSafe API key, from the
|
|
93
|
+
[TypeSafe dashboard](https://console.typesafe.ai/keys), sent as
|
|
94
|
+
`Authorization: Bearer`. Through OpenRouter, your OpenRouter key.
|
|
95
|
+
- `model` — optional, `jev-latest` by default, which follows the newest stable
|
|
96
|
+
release. After you tune `minConfidenceForRouting`, pin a versioned model so a
|
|
97
|
+
new release can't move answers under your threshold: `jev-1.13.0` on TypeSafe,
|
|
98
|
+
or `jev-1.13` on OpenRouter.
|
|
99
|
+
- `baseUrl` — optional, `https://api.typesafe.ai` by default. It's the API root,
|
|
100
|
+
the same value the TypeSafe SDK's `baseURL` takes, and the request goes to
|
|
101
|
+
`{baseUrl}/v1/systemone`.
|
|
102
|
+
|
|
103
|
+
To call Jev through OpenRouter, set `baseUrl` to OpenRouter's API root and use
|
|
104
|
+
an OpenRouter key. OpenRouter serves the same System One API and bills your
|
|
105
|
+
OpenRouter account:
|
|
106
|
+
|
|
107
|
+
```json
|
|
108
|
+
"classifier": {
|
|
109
|
+
"jev": {
|
|
110
|
+
"apiKey": "$env(OPENROUTER_API_KEY)",
|
|
111
|
+
"baseUrl": "https://openrouter.ai/api"
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
The policy sends the prompt and two questions to `{baseUrl}/v1/systemone`:
|
|
117
|
+
|
|
118
|
+
- **Intent** — a Choice over your intents. Each `id` is an option, and its
|
|
119
|
+
`description` tells Jev what the option covers.
|
|
120
|
+
- **Complexity** — a Score over three levels, low to high. The policy rounds the
|
|
121
|
+
score to the nearest level.
|
|
122
|
+
|
|
123
|
+
`profile.confidence` is the lower of Jev's confidence in the two answers, so
|
|
124
|
+
routing applies only when Jev is sure of both. Jev writes no reasons, so
|
|
125
|
+
`profile.reasons` is empty.
|
|
126
|
+
|
|
127
|
+
Jev takes no system prompt, so setting `classifierPrompt` with `classifier.jev`
|
|
128
|
+
is rejected with a `ConfigurationError`.
|
|
129
|
+
|
|
130
|
+
`chatCompletions` and `jev` call a service outside the gateway with the key you
|
|
131
|
+
configure, and send it the prompt. When that service answers with an error, the
|
|
132
|
+
policy logs the status and its most likely cause but not the response body,
|
|
133
|
+
which can echo the request back, prompt included.
|
|
33
134
|
|
|
34
135
|
## Authenticating the classifier app
|
|
35
136
|
|
|
@@ -39,18 +140,21 @@ the
|
|
|
39
140
|
[Ensure Gateway Internal Invocation Only](/docs/policies/ai-gateway-internal-only-inbound)
|
|
40
141
|
policy instead of AI Gateway Authentication: it accepts requests this gateway
|
|
41
142
|
made itself and rejects everything that arrives over the network. Then omit
|
|
42
|
-
`
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
143
|
+
`classifier.app.apiKey` — nothing is sent, and there is no key to rotate or
|
|
144
|
+
leak.
|
|
145
|
+
|
|
146
|
+
Set `classifier.app.apiKey` only when the classifier application authenticates
|
|
147
|
+
callers with API keys. It is sent as `Authorization: Bearer`, typically from
|
|
148
|
+
`$env(CLASSIFIER_APP_API_KEY)`. Omit it, or leave it empty, and the classifier
|
|
149
|
+
request carries no `Authorization` header. If you leave it unset while the
|
|
150
|
+
classifier app still checks keys, that app answers `401` and the gateway logs
|
|
151
|
+
which policy to add.
|
|
152
|
+
|
|
153
|
+
`classifier.app.appId` must not be this app's own id, and the classifier
|
|
154
|
+
application must not run this policy itself. Both are rejected with a
|
|
155
|
+
`ConfigurationError` instead of silently skipping or failing open, so a
|
|
156
|
+
misconfigured routing loop — within one app, or between two — is never mistaken
|
|
157
|
+
for smart routing that is simply doing nothing.
|
|
54
158
|
|
|
55
159
|
The classifier application must not run
|
|
56
160
|
[Akamai AI Firewall](/docs/policies/ai-gateway-akamai-firewall-inbound) either.
|
|
@@ -70,8 +174,12 @@ nothing. This is rejected with a `ConfigurationError` for the same reason.
|
|
|
70
174
|
"export": "AIGatewaySmartRouterInboundPolicy",
|
|
71
175
|
"module": "$import(@zuplo/runtime)",
|
|
72
176
|
"options": {
|
|
73
|
-
"
|
|
74
|
-
|
|
177
|
+
"classifier": {
|
|
178
|
+
"app": {
|
|
179
|
+
"appId": "$env(CLASSIFIER_APP_ID)",
|
|
180
|
+
"model": "openai/gpt-4o-mini"
|
|
181
|
+
}
|
|
182
|
+
},
|
|
75
183
|
"smartRoutingEnabled": true,
|
|
76
184
|
"modelsByComplexity": {
|
|
77
185
|
"low": "openai/gpt-4o-mini",
|
|
@@ -83,6 +191,20 @@ nothing. This is rejected with a `ConfigurationError` for the same reason.
|
|
|
83
191
|
}
|
|
84
192
|
```
|
|
85
193
|
|
|
194
|
+
## Migrating from `classifierAppID`
|
|
195
|
+
|
|
196
|
+
Earlier versions configured the classifier app with three top-level options.
|
|
197
|
+
They're deprecated, and still honored when `classifier` isn't set:
|
|
198
|
+
|
|
199
|
+
| Deprecated option | Replacement |
|
|
200
|
+
| --------------------- | ----------------------- |
|
|
201
|
+
| `classifierAppID` | `classifier.app.appId` |
|
|
202
|
+
| `classifierModel` | `classifier.app.model` |
|
|
203
|
+
| `classifierAppApiKey` | `classifier.app.apiKey` |
|
|
204
|
+
|
|
205
|
+
Move all three at once. Setting `classifier` together with any of them is
|
|
206
|
+
rejected with a `ConfigurationError`.
|
|
207
|
+
|
|
86
208
|
## Policy order
|
|
87
209
|
|
|
88
210
|
```text
|
|
@@ -149,14 +271,19 @@ classifier calls a strict JSON-schema chat completion, so it can only return one
|
|
|
149
271
|
of your configured ids. The `description` values are only shown to the
|
|
150
272
|
classifier if your prompt includes `{{intents}}` (see below).
|
|
151
273
|
|
|
274
|
+
With `classifier.jev`, the `id` values are the options of Jev's intent question,
|
|
275
|
+
and every `description` is shown to Jev with its option. There is no prompt, so
|
|
276
|
+
`{{intents}}` doesn't apply.
|
|
277
|
+
|
|
152
278
|
If the classifier returns an `id` outside this list, Smart Router keeps it as an
|
|
153
279
|
"unknown intent" and caps its confidence just below `minConfidenceForRouting`,
|
|
154
280
|
so it's never eligible for routing.
|
|
155
281
|
|
|
156
282
|
### Custom classifier prompt
|
|
157
283
|
|
|
158
|
-
`classifierPrompt` replaces the built-in classifier prompt
|
|
159
|
-
|
|
284
|
+
`classifierPrompt` replaces the built-in classifier prompt for `classifier.app`
|
|
285
|
+
and `classifier.chatCompletions`. Jev takes no prompt, so it can't be combined
|
|
286
|
+
with `classifier.jev`. It accepts either a string or an array of lines:
|
|
160
287
|
|
|
161
288
|
```json
|
|
162
289
|
{
|
|
@@ -225,6 +352,11 @@ interface AIGatewaySmartRouterResult {
|
|
|
225
352
|
}
|
|
226
353
|
```
|
|
227
354
|
|
|
355
|
+
With `classifier.jev`, `classifierModel` is the versioned model that answered,
|
|
356
|
+
in the host's naming: `jev-1.13.0` on TypeSafe, or `typesafe/jev-1.13-20260917`
|
|
357
|
+
on OpenRouter. `usage` reports the input and output tokens that the Jev API
|
|
358
|
+
returns as `promptTokens` and `completionTokens`.
|
|
359
|
+
|
|
228
360
|
Read it from any policy placed after Smart Router in the chain. Common uses:
|
|
229
361
|
|
|
230
362
|
- **Branch on intent or complexity** — apply a stricter rate limit, a different
|
|
@@ -300,10 +432,21 @@ complexity. Embeddings, tool messages and other non-AI paths are skipped.
|
|
|
300
432
|
## Fail-open behavior
|
|
301
433
|
|
|
302
434
|
The policy never 500s the user request for an internal classifier problem.
|
|
303
|
-
Invalid options, classifier timeouts, empty classifier
|
|
304
|
-
routing catalog errors are logged and the original request
|
|
305
|
-
Completions, Responses, and Anthropic Messages are classified
|
|
306
|
-
Embeddings and other non-AI paths are skipped.
|
|
435
|
+
Invalid or incomplete options, classifier errors and timeouts, empty classifier
|
|
436
|
+
responses, and smart routing catalog errors are logged and the original request
|
|
437
|
+
continues. Chat Completions, Responses, and Anthropic Messages are classified
|
|
438
|
+
automatically. Embeddings and other non-AI paths are skipped.
|
|
439
|
+
|
|
440
|
+
These misconfigurations throw a `ConfigurationError` instead, which fails the
|
|
441
|
+
request with a `500` that names the problem. Failing open on them would look
|
|
442
|
+
like smart routing quietly doing nothing:
|
|
443
|
+
|
|
444
|
+
- More than one section in `classifier`.
|
|
445
|
+
- A deprecated `classifierAppID`, `classifierModel`, or `classifierAppApiKey`
|
|
446
|
+
next to `classifier`.
|
|
447
|
+
- `classifierPrompt` with `classifier.jev`.
|
|
448
|
+
- A `classifier.app.appId` that names this app.
|
|
449
|
+
- A classifier app that runs this policy or Akamai AI Firewall.
|
|
307
450
|
|
|
308
451
|
One shape is refused rather than skipped. A Bedrock Runtime request
|
|
309
452
|
(`/model/{modelId}/{operation}`) carries a model-native body that the gateway
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
-
Classifies the last user prompt
|
|
2
|
-
|
|
3
|
-
optionally routes completions by
|
|
1
|
+
Classifies the last user prompt with an AI Gateway app, an OpenAI-compatible
|
|
2
|
+
Chat Completions API, or Jev from TypeSafe, stores the result on
|
|
3
|
+
`AIGatewaySmartRouter` for later policies, and optionally routes completions by
|
|
4
|
+
classified complexity.
|
|
@@ -32,17 +32,39 @@
|
|
|
32
32
|
"options": {
|
|
33
33
|
"type": "object",
|
|
34
34
|
"title": "AIGatewaySmartRouterInboundPolicyOptions",
|
|
35
|
-
"description": "Options for the Smart Router policy: classify the last user prompt with
|
|
35
|
+
"description": "Options for the Smart Router policy: classify the last user prompt with an AI Gateway app, a Chat Completions API, or Jev, then optionally route by complexity.",
|
|
36
36
|
"additionalProperties": false,
|
|
37
|
-
"required": [
|
|
38
|
-
|
|
39
|
-
"
|
|
40
|
-
|
|
41
|
-
|
|
37
|
+
"required": ["modelsByComplexity"],
|
|
38
|
+
"if": {
|
|
39
|
+
"required": ["classifierAppID"]
|
|
40
|
+
},
|
|
41
|
+
"then": {
|
|
42
|
+
"required": ["classifierModel"]
|
|
43
|
+
},
|
|
44
|
+
"else": {
|
|
45
|
+
"required": ["classifier"]
|
|
46
|
+
},
|
|
42
47
|
"examples": [
|
|
43
48
|
{
|
|
44
|
-
"
|
|
45
|
-
|
|
49
|
+
"classifier": {
|
|
50
|
+
"app": {
|
|
51
|
+
"appId": "$env(CLASSIFIER_APP_ID)",
|
|
52
|
+
"model": "openai/gpt-4o-mini"
|
|
53
|
+
}
|
|
54
|
+
},
|
|
55
|
+
"smartRoutingEnabled": true,
|
|
56
|
+
"modelsByComplexity": {
|
|
57
|
+
"low": "openai/gpt-4o-mini",
|
|
58
|
+
"medium": "openai/gpt-4o",
|
|
59
|
+
"high": "openai/gpt-5"
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
{
|
|
63
|
+
"classifier": {
|
|
64
|
+
"jev": {
|
|
65
|
+
"apiKey": "$env(TYPESAFE_API_KEY)"
|
|
66
|
+
}
|
|
67
|
+
},
|
|
46
68
|
"smartRoutingEnabled": true,
|
|
47
69
|
"modelsByComplexity": {
|
|
48
70
|
"low": "openai/gpt-4o-mini",
|
|
@@ -52,18 +74,101 @@
|
|
|
52
74
|
}
|
|
53
75
|
],
|
|
54
76
|
"properties": {
|
|
55
|
-
"
|
|
56
|
-
"type": "
|
|
57
|
-
"title": "Classifier
|
|
58
|
-
"description": "
|
|
59
|
-
"
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
"
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
77
|
+
"classifier": {
|
|
78
|
+
"type": "object",
|
|
79
|
+
"title": "Classifier",
|
|
80
|
+
"description": "Where the user's prompt is classified. Required unless the deprecated `classifierAppID` and `classifierModel` are set. Set exactly one of `app`, `chatCompletions`, or `jev`; setting more than one is a configuration error.",
|
|
81
|
+
"additionalProperties": false,
|
|
82
|
+
"minProperties": 1,
|
|
83
|
+
"maxProperties": 1,
|
|
84
|
+
"properties": {
|
|
85
|
+
"app": {
|
|
86
|
+
"type": "object",
|
|
87
|
+
"title": "Classifier App",
|
|
88
|
+
"description": "Classify with another AI Gateway app in this project. The call runs in-process and never leaves the gateway, so the classifier uses that app's providers, routing, and quotas.",
|
|
89
|
+
"additionalProperties": false,
|
|
90
|
+
"required": ["appId", "model"],
|
|
91
|
+
"properties": {
|
|
92
|
+
"appId": {
|
|
93
|
+
"type": "string",
|
|
94
|
+
"title": "Classifier App ID",
|
|
95
|
+
"description": "The id of the AI Gateway app that runs the classifier. It must not be this app, and that app must not run this policy.",
|
|
96
|
+
"examples": ["$env(CLASSIFIER_APP_ID)"]
|
|
97
|
+
},
|
|
98
|
+
"model": {
|
|
99
|
+
"type": "string",
|
|
100
|
+
"title": "Classifier Model",
|
|
101
|
+
"description": "The model (`providerName/model`) the classifier app uses to evaluate the user's request. E.g. `openai/gpt-4o-mini`.",
|
|
102
|
+
"pattern": "^[^/\\s]+/.+$",
|
|
103
|
+
"examples": ["openai/gpt-4o-mini"]
|
|
104
|
+
},
|
|
105
|
+
"apiKey": {
|
|
106
|
+
"type": "string",
|
|
107
|
+
"title": "Classifier App API Key",
|
|
108
|
+
"description": "API key sent as `Authorization: Bearer` when invoking the classifier app. Omit it when the classifier app runs the Ensure Gateway Internal Invocation Only policy: the classifier call never leaves the gateway, so that policy accepts it and rejects everything arriving over the network, and no credential is sent. Set it only when the classifier app authenticates callers with API keys.",
|
|
109
|
+
"examples": ["$env(CLASSIFIER_APP_API_KEY)"],
|
|
110
|
+
"x-advanced": true
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
},
|
|
114
|
+
"chatCompletions": {
|
|
115
|
+
"type": "object",
|
|
116
|
+
"title": "Chat Completions API",
|
|
117
|
+
"description": "Classify by calling an OpenAI-compatible Chat Completions API directly, outside the gateway. The model must support structured outputs (`response_format` with a JSON schema).",
|
|
118
|
+
"additionalProperties": false,
|
|
119
|
+
"required": ["apiKey", "model"],
|
|
120
|
+
"properties": {
|
|
121
|
+
"apiKey": {
|
|
122
|
+
"type": "string",
|
|
123
|
+
"title": "Chat Completions API Key",
|
|
124
|
+
"description": "API key for the service at `baseUrl`, sent as `Authorization: Bearer`.",
|
|
125
|
+
"examples": ["$env(OPENAI_API_KEY)"]
|
|
126
|
+
},
|
|
127
|
+
"model": {
|
|
128
|
+
"type": "string",
|
|
129
|
+
"title": "Chat Completions Model",
|
|
130
|
+
"description": "The model id the service expects, such as `gpt-4o-mini`. This is sent to the service as-is, not as a `providerName/model` reference.",
|
|
131
|
+
"examples": ["gpt-4o-mini"]
|
|
132
|
+
},
|
|
133
|
+
"baseUrl": {
|
|
134
|
+
"type": "string",
|
|
135
|
+
"title": "Chat Completions Base URL",
|
|
136
|
+
"description": "Base URL of the OpenAI-compatible API. The classifier request is sent to `{baseUrl}/chat/completions`.",
|
|
137
|
+
"default": "https://api.openai.com/v1",
|
|
138
|
+
"x-advanced": true
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
},
|
|
142
|
+
"jev": {
|
|
143
|
+
"type": "object",
|
|
144
|
+
"title": "Jev (TypeSafe)",
|
|
145
|
+
"description": "Classify with Jev, TypeSafe's classification model, by calling a System One API directly: TypeSafe's by default, or another host such as OpenRouter through `baseUrl`. Jev answers typed questions instead of generating text, so it takes no `classifierPrompt`.",
|
|
146
|
+
"additionalProperties": false,
|
|
147
|
+
"required": ["apiKey"],
|
|
148
|
+
"properties": {
|
|
149
|
+
"apiKey": {
|
|
150
|
+
"type": "string",
|
|
151
|
+
"title": "Jev API Key",
|
|
152
|
+
"description": "API key for the service at `baseUrl`, sent as `Authorization: Bearer`: a TypeSafe key by default, or an OpenRouter key when `baseUrl` is OpenRouter.",
|
|
153
|
+
"examples": ["$env(TYPESAFE_API_KEY)"]
|
|
154
|
+
},
|
|
155
|
+
"model": {
|
|
156
|
+
"type": "string",
|
|
157
|
+
"title": "Jev Model",
|
|
158
|
+
"description": "The Jev model to call. `jev-latest` follows the newest stable release on both TypeSafe and OpenRouter. Pin a versioned id to keep answers stable after tuning `minConfidenceForRouting`: `jev-1.13.0` on TypeSafe, or `jev-1.13` on OpenRouter.",
|
|
159
|
+
"default": "jev-latest",
|
|
160
|
+
"x-advanced": true
|
|
161
|
+
},
|
|
162
|
+
"baseUrl": {
|
|
163
|
+
"type": "string",
|
|
164
|
+
"title": "Jev Base URL",
|
|
165
|
+
"description": "API root of the System One API that serves Jev, the same value the TypeSafe SDK's `baseURL` takes. The classifier request is sent to `{baseUrl}/v1/systemone`. Set it to `https://openrouter.ai/api` to call Jev through OpenRouter.",
|
|
166
|
+
"default": "https://api.typesafe.ai",
|
|
167
|
+
"x-advanced": true
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
}
|
|
67
172
|
},
|
|
68
173
|
"modelsByComplexity": {
|
|
69
174
|
"type": "object",
|
|
@@ -95,13 +200,6 @@
|
|
|
95
200
|
}
|
|
96
201
|
}
|
|
97
202
|
},
|
|
98
|
-
"classifierAppApiKey": {
|
|
99
|
-
"type": "string",
|
|
100
|
-
"title": "Classifier App API Key",
|
|
101
|
-
"description": "API key sent as `Authorization: Bearer` when invoking the classifier app. Omit it when the classifier app runs the Ensure Gateway Internal Invocation Only policy: the classifier call never leaves the gateway, so that policy accepts it and rejects everything arriving over the network, and no credential is sent. Set it only when the classifier app authenticates callers with API keys.",
|
|
102
|
-
"examples": ["$env(CLASSIFIER_APP_API_KEY)"],
|
|
103
|
-
"x-advanced": true
|
|
104
|
-
},
|
|
105
203
|
"smartRoutingEnabled": {
|
|
106
204
|
"type": "boolean",
|
|
107
205
|
"title": "Smart Routing Enabled",
|
|
@@ -140,7 +238,7 @@
|
|
|
140
238
|
},
|
|
141
239
|
"classifierPrompt": {
|
|
142
240
|
"title": "Classifier Prompt",
|
|
143
|
-
"description": "Prompt used to analyze and classify the user message. Omit to use the built-in classifier prompt.",
|
|
241
|
+
"description": "Prompt used to analyze and classify the user message with `classifier.app` or `classifier.chatCompletions`. Omit to use the built-in classifier prompt. Jev takes no prompt, so setting this with `classifier.jev` is a configuration error.",
|
|
144
242
|
"oneOf": [
|
|
145
243
|
{
|
|
146
244
|
"type": "string",
|
|
@@ -181,6 +279,34 @@
|
|
|
181
279
|
"minimum": 1,
|
|
182
280
|
"default": 8000,
|
|
183
281
|
"x-advanced": true
|
|
282
|
+
},
|
|
283
|
+
"classifierAppID": {
|
|
284
|
+
"type": "string",
|
|
285
|
+
"title": "Classifier App ID (Deprecated)",
|
|
286
|
+
"description": "\\*\\*Deprecated\\*\\*: use `classifier.app.appId` instead. Still honored when `classifier` is not set; setting both is a configuration error.",
|
|
287
|
+
"deprecated": true,
|
|
288
|
+
"doNotSuggest": true,
|
|
289
|
+
"x-show-example": false,
|
|
290
|
+
"x-advanced": true
|
|
291
|
+
},
|
|
292
|
+
"classifierModel": {
|
|
293
|
+
"type": "string",
|
|
294
|
+
"title": "Classifier Model (Deprecated)",
|
|
295
|
+
"description": "\\*\\*Deprecated\\*\\*: use `classifier.app.model` instead. Still honored when `classifier` is not set; setting both is a configuration error.",
|
|
296
|
+
"pattern": "^[^/\\s]+/.+$",
|
|
297
|
+
"deprecated": true,
|
|
298
|
+
"doNotSuggest": true,
|
|
299
|
+
"x-show-example": false,
|
|
300
|
+
"x-advanced": true
|
|
301
|
+
},
|
|
302
|
+
"classifierAppApiKey": {
|
|
303
|
+
"type": "string",
|
|
304
|
+
"title": "Classifier App API Key (Deprecated)",
|
|
305
|
+
"description": "\\*\\*Deprecated\\*\\*: use `classifier.app.apiKey` instead. Still honored when `classifier` is not set; setting both is a configuration error.",
|
|
306
|
+
"deprecated": true,
|
|
307
|
+
"doNotSuggest": true,
|
|
308
|
+
"x-show-example": false,
|
|
309
|
+
"x-advanced": true
|
|
184
310
|
}
|
|
185
311
|
}
|
|
186
312
|
}
|
|
@@ -190,8 +316,29 @@
|
|
|
190
316
|
"export": "AIGatewaySmartRouterInboundPolicy",
|
|
191
317
|
"module": "$import(@zuplo/runtime)",
|
|
192
318
|
"options": {
|
|
193
|
-
"
|
|
194
|
-
|
|
319
|
+
"classifier": {
|
|
320
|
+
"app": {
|
|
321
|
+
"appId": "$env(CLASSIFIER_APP_ID)",
|
|
322
|
+
"model": "openai/gpt-4o-mini"
|
|
323
|
+
}
|
|
324
|
+
},
|
|
325
|
+
"smartRoutingEnabled": true,
|
|
326
|
+
"modelsByComplexity": {
|
|
327
|
+
"low": "openai/gpt-4o-mini",
|
|
328
|
+
"medium": "openai/gpt-4o",
|
|
329
|
+
"high": "openai/gpt-5"
|
|
330
|
+
}
|
|
331
|
+
}
|
|
332
|
+
},
|
|
333
|
+
{
|
|
334
|
+
"export": "AIGatewaySmartRouterInboundPolicy",
|
|
335
|
+
"module": "$import(@zuplo/runtime)",
|
|
336
|
+
"options": {
|
|
337
|
+
"classifier": {
|
|
338
|
+
"jev": {
|
|
339
|
+
"apiKey": "$env(TYPESAFE_API_KEY)"
|
|
340
|
+
}
|
|
341
|
+
},
|
|
195
342
|
"smartRoutingEnabled": true,
|
|
196
343
|
"modelsByComplexity": {
|
|
197
344
|
"low": "openai/gpt-4o-mini",
|