@oxy.so/contracts 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +202 -0
- package/NOTICE +16 -0
- package/dist/cjs/.tsbuildinfo +1 -0
- package/dist/cjs/accountGraph.js +489 -0
- package/dist/cjs/agency.js +439 -0
- package/dist/cjs/browserHub.js +215 -0
- package/dist/cjs/civic.js +163 -0
- package/dist/cjs/commonsSignIn.js +59 -0
- package/dist/cjs/deviceBoot.js +50 -0
- package/dist/cjs/deviceDirectory.js +189 -0
- package/dist/cjs/devicePairing.js +138 -0
- package/dist/cjs/deviceSession.js +164 -0
- package/dist/cjs/emailAgentContext.js +32 -0
- package/dist/cjs/followGraph.js +28 -0
- package/dist/cjs/identity.js +258 -0
- package/dist/cjs/inboxPush.js +24 -0
- package/dist/cjs/index.js +618 -0
- package/dist/cjs/inference/accountBilling.js +334 -0
- package/dist/cjs/inference/aliaModelRelease.js +262 -0
- package/dist/cjs/inference/attribution.js +106 -0
- package/dist/cjs/inference/catalogue.js +487 -0
- package/dist/cjs/inference/entitlement.js +217 -0
- package/dist/cjs/inference/errors.js +309 -0
- package/dist/cjs/inference/identifiers.js +224 -0
- package/dist/cjs/inference/inbox.js +105 -0
- package/dist/cjs/inference/modelDocumentation.js +433 -0
- package/dist/cjs/inference/money.js +188 -0
- package/dist/cjs/inference/priceVersion.js +110 -0
- package/dist/cjs/inference/providerConnection.js +455 -0
- package/dist/cjs/inference/request.js +477 -0
- package/dist/cjs/inference/routingPolicy.js +318 -0
- package/dist/cjs/inference/streamEvents.js +258 -0
- package/dist/cjs/inference/usage.js +329 -0
- package/dist/cjs/inference/version.js +105 -0
- package/dist/cjs/keyRecovery.js +91 -0
- package/dist/cjs/keyRotation.js +75 -0
- package/dist/cjs/links.js +68 -0
- package/dist/cjs/moderationReputation.js +298 -0
- package/dist/cjs/oauth.js +66 -0
- package/dist/cjs/oxyRecordTypes.js +71 -0
- package/dist/cjs/protocol.js +53 -0
- package/dist/cjs/recommendations.js +168 -0
- package/dist/cjs/reputation.js +297 -0
- package/dist/cjs/sessionStatus.js +121 -0
- package/dist/cjs/transparency.js +89 -0
- package/dist/cjs/updates.js +252 -0
- package/dist/cjs/userInvalidation.js +89 -0
- package/dist/cjs/userResponse.js +245 -0
- package/dist/cjs/username.js +290 -0
- package/dist/cjs/webauthn.js +71 -0
- package/dist/esm/.tsbuildinfo +1 -0
- package/dist/esm/accountGraph.js +480 -0
- package/dist/esm/agency.js +436 -0
- package/dist/esm/browserHub.js +212 -0
- package/dist/esm/civic.js +160 -0
- package/dist/esm/commonsSignIn.js +56 -0
- package/dist/esm/deviceBoot.js +47 -0
- package/dist/esm/deviceDirectory.js +186 -0
- package/dist/esm/devicePairing.js +135 -0
- package/dist/esm/deviceSession.js +161 -0
- package/dist/esm/emailAgentContext.js +29 -0
- package/dist/esm/followGraph.js +27 -0
- package/dist/esm/identity.js +255 -0
- package/dist/esm/inboxPush.js +21 -0
- package/dist/esm/index.js +172 -0
- package/dist/esm/inference/accountBilling.js +331 -0
- package/dist/esm/inference/aliaModelRelease.js +259 -0
- package/dist/esm/inference/attribution.js +103 -0
- package/dist/esm/inference/catalogue.js +484 -0
- package/dist/esm/inference/entitlement.js +214 -0
- package/dist/esm/inference/errors.js +306 -0
- package/dist/esm/inference/identifiers.js +221 -0
- package/dist/esm/inference/inbox.js +102 -0
- package/dist/esm/inference/modelDocumentation.js +430 -0
- package/dist/esm/inference/money.js +185 -0
- package/dist/esm/inference/priceVersion.js +107 -0
- package/dist/esm/inference/providerConnection.js +452 -0
- package/dist/esm/inference/request.js +474 -0
- package/dist/esm/inference/routingPolicy.js +315 -0
- package/dist/esm/inference/streamEvents.js +255 -0
- package/dist/esm/inference/usage.js +326 -0
- package/dist/esm/inference/version.js +102 -0
- package/dist/esm/keyRecovery.js +88 -0
- package/dist/esm/keyRotation.js +72 -0
- package/dist/esm/links.js +65 -0
- package/dist/esm/moderationReputation.js +295 -0
- package/dist/esm/oauth.js +63 -0
- package/dist/esm/oxyRecordTypes.js +68 -0
- package/dist/esm/protocol.js +50 -0
- package/dist/esm/recommendations.js +165 -0
- package/dist/esm/reputation.js +293 -0
- package/dist/esm/sessionStatus.js +118 -0
- package/dist/esm/transparency.js +86 -0
- package/dist/esm/updates.js +249 -0
- package/dist/esm/userInvalidation.js +85 -0
- package/dist/esm/userResponse.js +240 -0
- package/dist/esm/username.js +283 -0
- package/dist/esm/webauthn.js +68 -0
- package/dist/types/.tsbuildinfo +1 -0
- package/dist/types/accountGraph.d.ts +378 -0
- package/dist/types/agency.d.ts +2162 -0
- package/dist/types/browserHub.d.ts +856 -0
- package/dist/types/civic.d.ts +338 -0
- package/dist/types/commonsSignIn.d.ts +58 -0
- package/dist/types/deviceBoot.d.ts +74 -0
- package/dist/types/deviceDirectory.d.ts +1317 -0
- package/dist/types/devicePairing.d.ts +130 -0
- package/dist/types/deviceSession.d.ts +411 -0
- package/dist/types/emailAgentContext.d.ts +248 -0
- package/dist/types/followGraph.d.ts +150 -0
- package/dist/types/identity.d.ts +402 -0
- package/dist/types/inboxPush.d.ts +30 -0
- package/dist/types/index.d.ts +100 -0
- package/dist/types/inference/accountBilling.d.ts +738 -0
- package/dist/types/inference/aliaModelRelease.d.ts +609 -0
- package/dist/types/inference/attribution.d.ts +176 -0
- package/dist/types/inference/catalogue.d.ts +1618 -0
- package/dist/types/inference/entitlement.d.ts +519 -0
- package/dist/types/inference/errors.d.ts +242 -0
- package/dist/types/inference/identifiers.d.ts +182 -0
- package/dist/types/inference/inbox.d.ts +374 -0
- package/dist/types/inference/modelDocumentation.d.ts +1603 -0
- package/dist/types/inference/money.d.ts +185 -0
- package/dist/types/inference/priceVersion.d.ts +182 -0
- package/dist/types/inference/providerConnection.d.ts +968 -0
- package/dist/types/inference/request.d.ts +2800 -0
- package/dist/types/inference/routingPolicy.d.ts +616 -0
- package/dist/types/inference/streamEvents.d.ts +950 -0
- package/dist/types/inference/usage.d.ts +1164 -0
- package/dist/types/inference/version.d.ts +102 -0
- package/dist/types/keyRecovery.d.ts +138 -0
- package/dist/types/keyRotation.d.ts +103 -0
- package/dist/types/links.d.ts +96 -0
- package/dist/types/moderationReputation.d.ts +487 -0
- package/dist/types/oauth.d.ts +86 -0
- package/dist/types/oxyRecordTypes.d.ts +62 -0
- package/dist/types/protocol.d.ts +86 -0
- package/dist/types/recommendations.d.ts +542 -0
- package/dist/types/reputation.d.ts +457 -0
- package/dist/types/sessionStatus.d.ts +231 -0
- package/dist/types/transparency.d.ts +392 -0
- package/dist/types/updates.d.ts +545 -0
- package/dist/types/userInvalidation.d.ts +94 -0
- package/dist/types/userResponse.d.ts +1706 -0
- package/dist/types/username.d.ts +265 -0
- package/dist/types/webauthn.d.ts +77 -0
- package/package.json +87 -0
|
@@ -0,0 +1,474 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The normalized inference request — the canonical internal envelope Oxy's
|
|
3
|
+
* public edge forwards to the data plane.
|
|
4
|
+
*
|
|
5
|
+
* The public surface speaks several dialects (`/v1/responses`,
|
|
6
|
+
* `/v1/chat/completions`, embeddings, images, audio). Exactly one of them is
|
|
7
|
+
* normalized here, at the edge, so that routing, metering, policy enforcement
|
|
8
|
+
* and settlement are written once against one shape instead of once per dialect.
|
|
9
|
+
* `client.apiFormat` records which dialect the customer used, because the
|
|
10
|
+
* response has to be rendered back in it.
|
|
11
|
+
*
|
|
12
|
+
* What the envelope carries that a provider request does not: the resolved
|
|
13
|
+
* attribution block (who pays, which application, which credential, which
|
|
14
|
+
* delegated user), the exact routing policy reference, the routes that policy
|
|
15
|
+
* has already authorized, and the customer's idempotency key. Those are the
|
|
16
|
+
* fields that make a request billable and explainable, and they are resolved
|
|
17
|
+
* BEFORE the request enters the data plane.
|
|
18
|
+
*
|
|
19
|
+
* The policy VALUES never travel. `routingPolicy` is a reference — provenance
|
|
20
|
+
* for the receipt — and `authorizedRoutes` is the result of applying the policy,
|
|
21
|
+
* in preference order, so the data plane needs no policy semantics to fail over.
|
|
22
|
+
*
|
|
23
|
+
* Decided in: docs/adr/0010-public-api-compatibility.md,
|
|
24
|
+
* docs/adr/0017-authorized-routes-in-the-envelope.md.
|
|
25
|
+
*/
|
|
26
|
+
import { z } from "zod";
|
|
27
|
+
import { inferenceAttributionSchema } from "./attribution.js";
|
|
28
|
+
import { inferenceModalitySchema } from "./catalogue.js";
|
|
29
|
+
import { idempotencyKeySchema, inferenceTimestampSchema } from "./identifiers.js";
|
|
30
|
+
import { authorizedRouteSchema, routingPolicyReferenceSchema, routingTargetSchema, } from "./routingPolicy.js";
|
|
31
|
+
/* -------------------------------------------------------------------------- */
|
|
32
|
+
/* Input */
|
|
33
|
+
/* -------------------------------------------------------------------------- */
|
|
34
|
+
/**
|
|
35
|
+
* Where binary or remote content comes from.
|
|
36
|
+
*
|
|
37
|
+
* `url` is fetched by the data plane; `inline` carries base64 the customer sent
|
|
38
|
+
* with the request. Both are transient: neither is persisted by default, and
|
|
39
|
+
* neither appears in a receipt, a log line or a telemetry event.
|
|
40
|
+
*/
|
|
41
|
+
export const inferenceContentSourceSchema = z.discriminatedUnion("kind", [
|
|
42
|
+
z
|
|
43
|
+
.object({ kind: z.literal("url"), url: z.string().min(1).max(4096) })
|
|
44
|
+
.strict(),
|
|
45
|
+
z
|
|
46
|
+
.object({
|
|
47
|
+
kind: z.literal("inline"),
|
|
48
|
+
mediaType: z.string().min(1).max(255),
|
|
49
|
+
data: z.string().min(1),
|
|
50
|
+
})
|
|
51
|
+
.strict(),
|
|
52
|
+
]);
|
|
53
|
+
/**
|
|
54
|
+
* One part of a message's content. A message is always a list of parts.
|
|
55
|
+
*
|
|
56
|
+
* ## `refusal` is a member and `reasoning` is not, and the asymmetry is the point
|
|
57
|
+
*
|
|
58
|
+
* A model that DECLINES says why, and those words are meant for the customer:
|
|
59
|
+
* they are the difference between "rephrase this" and "stop asking". Before this
|
|
60
|
+
* member existed there was nowhere in an {@link InferenceMessage} to put them, so
|
|
61
|
+
* a non-streaming fold kept `finishReason: 'refusal'` and dropped the sentence —
|
|
62
|
+
* the customer learned they were refused and not why, while a streaming caller of
|
|
63
|
+
* the same request got the whole explanation on the `refusal` delta channel.
|
|
64
|
+
*
|
|
65
|
+
* Reasoning is the opposite case and stays absent. It is the model's private
|
|
66
|
+
* working, and a `text` part is where somebody would put it — which renders
|
|
67
|
+
* private reasoning to the customer AS the answer, the product bug the delta
|
|
68
|
+
* channels exist to prevent. An opaque per-block reasoning blob crossing this
|
|
69
|
+
* boundary needs a home nobody has chosen yet, and inventing one here would
|
|
70
|
+
* choose it by accident.
|
|
71
|
+
*
|
|
72
|
+
* Both public dialects can carry a refusal, which is what separates the two
|
|
73
|
+
* cases at the boundary as well as in principle: `refusal` is OpenAI's OWN field
|
|
74
|
+
* in both of its shapes (`delta.refusal` streaming, `message.refusal`
|
|
75
|
+
* non-streaming), while reasoning has no OpenAI field at all — the
|
|
76
|
+
* `reasoning_content`/`reasoning` spellings an OpenAI-compatible provider emits
|
|
77
|
+
* are provider extensions. So carrying the refusal costs no dialect its
|
|
78
|
+
* standard-client parseability, and carrying reasoning would.
|
|
79
|
+
*
|
|
80
|
+
* The part is a MEMBER rather than a field because a refusal need not have text:
|
|
81
|
+
* an Anthropic `stop_reason: "refusal"` maps to the finish reason and separates
|
|
82
|
+
* no words from the answer, while an OpenAI-compatible `refusal` does. A required
|
|
83
|
+
* field would force the first provider to invent a sentence.
|
|
84
|
+
*/
|
|
85
|
+
export const inferenceContentPartSchema = z.discriminatedUnion("type", [
|
|
86
|
+
z.object({ type: z.literal("text"), text: z.string() }).strict(),
|
|
87
|
+
z
|
|
88
|
+
.object({
|
|
89
|
+
type: z.literal("image"),
|
|
90
|
+
source: inferenceContentSourceSchema,
|
|
91
|
+
/** Provider-independent hint; providers that ignore it are unaffected. */
|
|
92
|
+
detail: z.enum(["auto", "low", "high"]).optional(),
|
|
93
|
+
})
|
|
94
|
+
.strict(),
|
|
95
|
+
z
|
|
96
|
+
.object({ type: z.literal("audio"), source: inferenceContentSourceSchema })
|
|
97
|
+
.strict(),
|
|
98
|
+
z
|
|
99
|
+
.object({
|
|
100
|
+
type: z.literal("file"),
|
|
101
|
+
source: inferenceContentSourceSchema,
|
|
102
|
+
filename: z.string().max(255).optional(),
|
|
103
|
+
})
|
|
104
|
+
.strict(),
|
|
105
|
+
/**
|
|
106
|
+
* A model's explanation for declining. Its own part, never a `text` one, so no
|
|
107
|
+
* renderer can present a refusal as the answer that was asked for.
|
|
108
|
+
*/
|
|
109
|
+
z.object({ type: z.literal("refusal"), text: z.string() }).strict(),
|
|
110
|
+
]);
|
|
111
|
+
/**
|
|
112
|
+
* A tool call an assistant made, in the normalized form.
|
|
113
|
+
*
|
|
114
|
+
* `arguments` is the JSON TEXT the model emitted, not a parsed object: models
|
|
115
|
+
* emit invalid JSON often enough that parsing it here would turn a recoverable
|
|
116
|
+
* model mistake into a rejected message.
|
|
117
|
+
*/
|
|
118
|
+
export const inferenceToolCallSchema = z
|
|
119
|
+
.object({
|
|
120
|
+
id: z.string().min(1).max(128),
|
|
121
|
+
name: z.string().min(1).max(128),
|
|
122
|
+
arguments: z.string(),
|
|
123
|
+
})
|
|
124
|
+
.strict();
|
|
125
|
+
export const inferenceMessageRoleSchema = z.enum([
|
|
126
|
+
"system",
|
|
127
|
+
"developer",
|
|
128
|
+
"user",
|
|
129
|
+
"assistant",
|
|
130
|
+
"tool",
|
|
131
|
+
]);
|
|
132
|
+
/**
|
|
133
|
+
* One normalized message.
|
|
134
|
+
*
|
|
135
|
+
* The role-specific fields are refined rather than modelled as a discriminated
|
|
136
|
+
* union so the shape stays the one a caller recognises from the OpenAI-style
|
|
137
|
+
* dialects; the refinement is what stops `toolCallId` from riding on a user
|
|
138
|
+
* message, where every provider would silently ignore it.
|
|
139
|
+
*/
|
|
140
|
+
export const inferenceMessageSchema = z
|
|
141
|
+
.object({
|
|
142
|
+
role: inferenceMessageRoleSchema,
|
|
143
|
+
content: z.array(inferenceContentPartSchema),
|
|
144
|
+
/** Participant name, where the dialect supports naming participants. */
|
|
145
|
+
name: z.string().max(128).optional(),
|
|
146
|
+
/** The tool call this message answers. Required on, and only on, `tool`. */
|
|
147
|
+
toolCallId: z.string().min(1).max(128).optional(),
|
|
148
|
+
/** Tool calls the assistant made. Only on `assistant`. */
|
|
149
|
+
toolCalls: z.array(inferenceToolCallSchema).optional(),
|
|
150
|
+
})
|
|
151
|
+
.strict()
|
|
152
|
+
.superRefine((message, ctx) => {
|
|
153
|
+
if (message.role === "tool" && message.toolCallId === undefined) {
|
|
154
|
+
ctx.addIssue({
|
|
155
|
+
code: z.ZodIssueCode.custom,
|
|
156
|
+
path: ["toolCallId"],
|
|
157
|
+
message: "a tool message must name the tool call it answers",
|
|
158
|
+
});
|
|
159
|
+
}
|
|
160
|
+
if (message.role !== "tool" && message.toolCallId !== undefined) {
|
|
161
|
+
ctx.addIssue({
|
|
162
|
+
code: z.ZodIssueCode.custom,
|
|
163
|
+
path: ["toolCallId"],
|
|
164
|
+
message: "only a tool message answers a tool call",
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
if (message.role !== "assistant" && message.toolCalls !== undefined) {
|
|
168
|
+
ctx.addIssue({
|
|
169
|
+
code: z.ZodIssueCode.custom,
|
|
170
|
+
path: ["toolCalls"],
|
|
171
|
+
message: "only an assistant message makes tool calls",
|
|
172
|
+
});
|
|
173
|
+
}
|
|
174
|
+
// Same rule as `toolCalls`, for the same reason: only the assistant can
|
|
175
|
+
// decline, so a refusal on any other role is a part every provider would
|
|
176
|
+
// silently ignore — and one a renderer might not.
|
|
177
|
+
if (message.role !== "assistant") {
|
|
178
|
+
for (const [index, part] of message.content.entries()) {
|
|
179
|
+
if (part.type === "refusal") {
|
|
180
|
+
ctx.addIssue({
|
|
181
|
+
code: z.ZodIssueCode.custom,
|
|
182
|
+
path: ["content", index, "type"],
|
|
183
|
+
message: "only an assistant message carries a refusal",
|
|
184
|
+
});
|
|
185
|
+
}
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
});
|
|
189
|
+
/**
|
|
190
|
+
* The request's input.
|
|
191
|
+
*
|
|
192
|
+
* Three formats, because the modalities genuinely differ: a chat request is a
|
|
193
|
+
* conversation, an embedding request is a string or a batch of strings, and
|
|
194
|
+
* pretending the latter is a one-message conversation loses the batch boundary
|
|
195
|
+
* that both metering and provider translation depend on.
|
|
196
|
+
*/
|
|
197
|
+
export const inferenceInputSchema = z.discriminatedUnion("format", [
|
|
198
|
+
z
|
|
199
|
+
.object({
|
|
200
|
+
format: z.literal("messages"),
|
|
201
|
+
messages: z.array(inferenceMessageSchema).min(1),
|
|
202
|
+
})
|
|
203
|
+
.strict(),
|
|
204
|
+
z.object({ format: z.literal("text"), text: z.string() }).strict(),
|
|
205
|
+
z
|
|
206
|
+
.object({
|
|
207
|
+
format: z.literal("text_batch"),
|
|
208
|
+
texts: z.array(z.string()).min(1).max(2048),
|
|
209
|
+
})
|
|
210
|
+
.strict(),
|
|
211
|
+
]);
|
|
212
|
+
/* -------------------------------------------------------------------------- */
|
|
213
|
+
/* Generation controls */
|
|
214
|
+
/* -------------------------------------------------------------------------- */
|
|
215
|
+
/** Sampling parameters, all optional: absent means the route's own default. */
|
|
216
|
+
export const samplingParametersSchema = z
|
|
217
|
+
.object({
|
|
218
|
+
temperature: z.number().min(0).max(2).optional(),
|
|
219
|
+
topP: z.number().min(0).max(1).optional(),
|
|
220
|
+
topK: z.number().int().positive().safe().optional(),
|
|
221
|
+
frequencyPenalty: z.number().min(-2).max(2).optional(),
|
|
222
|
+
presencePenalty: z.number().min(-2).max(2).optional(),
|
|
223
|
+
/** A seed makes a request reproducible on providers that honour one. */
|
|
224
|
+
seed: z.number().int().safe().optional(),
|
|
225
|
+
stopSequences: z.array(z.string().min(1).max(256)).max(8).optional(),
|
|
226
|
+
})
|
|
227
|
+
.strict();
|
|
228
|
+
/**
|
|
229
|
+
* A tool the model may call. `parameters` is a JSON Schema document, carried
|
|
230
|
+
* as an opaque object: validating the customer's JSON Schema against a meta
|
|
231
|
+
* schema here would reject documents providers accept.
|
|
232
|
+
*/
|
|
233
|
+
export const toolDefinitionSchema = z
|
|
234
|
+
.object({
|
|
235
|
+
type: z.literal("function"),
|
|
236
|
+
name: z.string().min(1).max(128),
|
|
237
|
+
description: z.string().max(2000).optional(),
|
|
238
|
+
parameters: z.record(z.unknown()),
|
|
239
|
+
/** Ask the provider to enforce the schema, where it supports enforcement. */
|
|
240
|
+
strict: z.boolean().optional(),
|
|
241
|
+
})
|
|
242
|
+
.strict();
|
|
243
|
+
/** Whether, and which, tool the model must call. */
|
|
244
|
+
export const toolChoiceSchema = z.union([
|
|
245
|
+
z.enum(["auto", "none", "required"]),
|
|
246
|
+
z
|
|
247
|
+
.object({ type: z.literal("function"), name: z.string().min(1).max(128) })
|
|
248
|
+
.strict(),
|
|
249
|
+
]);
|
|
250
|
+
/** Structured-output request: free text, any JSON object, or a named schema. */
|
|
251
|
+
export const responseFormatSchema = z.discriminatedUnion("type", [
|
|
252
|
+
z.object({ type: z.literal("text") }).strict(),
|
|
253
|
+
z.object({ type: z.literal("json_object") }).strict(),
|
|
254
|
+
z
|
|
255
|
+
.object({
|
|
256
|
+
type: z.literal("json_schema"),
|
|
257
|
+
name: z.string().min(1).max(128),
|
|
258
|
+
schema: z.record(z.unknown()),
|
|
259
|
+
strict: z.boolean(),
|
|
260
|
+
})
|
|
261
|
+
.strict(),
|
|
262
|
+
]);
|
|
263
|
+
/* -------------------------------------------------------------------------- */
|
|
264
|
+
/* Client metadata */
|
|
265
|
+
/* -------------------------------------------------------------------------- */
|
|
266
|
+
/**
|
|
267
|
+
* What the edge records about the CALL, as opposed to its content.
|
|
268
|
+
*
|
|
269
|
+
* `.strict()` is a privacy control, not tidiness. Oxy never persists a user IP
|
|
270
|
+
* — raw, hashed or geo-derived — and this object is the natural place somebody
|
|
271
|
+
* would add one "for security". Strict means a producer that attaches `ip`,
|
|
272
|
+
* `country`, `userAgent` or `forwardedFor` fails the parse instead of quietly
|
|
273
|
+
* shipping it into the data plane and the telemetry stream behind it.
|
|
274
|
+
*
|
|
275
|
+
* `labels` is customer-supplied cost-attribution metadata (a team name, a
|
|
276
|
+
* feature flag). It is echoed on the receipt, so it must never be used for
|
|
277
|
+
* anything the customer would not want to read back to themselves.
|
|
278
|
+
*/
|
|
279
|
+
export const clientRequestMetadataSchema = z
|
|
280
|
+
.object({
|
|
281
|
+
/** The public dialect the customer called. The response is rendered in it. */
|
|
282
|
+
apiFormat: z.enum([
|
|
283
|
+
"responses",
|
|
284
|
+
"chat_completions",
|
|
285
|
+
"embeddings",
|
|
286
|
+
"images_generations",
|
|
287
|
+
"audio_transcriptions",
|
|
288
|
+
"audio_speech",
|
|
289
|
+
"rerank",
|
|
290
|
+
"batches",
|
|
291
|
+
]),
|
|
292
|
+
/** The public path, e.g. `/v1/responses`. */
|
|
293
|
+
endpoint: z.string().min(1).max(256),
|
|
294
|
+
/** The customer's own correlation id, when they sent one. */
|
|
295
|
+
clientRequestId: z.string().min(1).max(128).optional(),
|
|
296
|
+
receivedAt: inferenceTimestampSchema,
|
|
297
|
+
labels: z.record(z.string().max(256)).optional(),
|
|
298
|
+
})
|
|
299
|
+
.strict();
|
|
300
|
+
/* -------------------------------------------------------------------------- */
|
|
301
|
+
/* The envelope */
|
|
302
|
+
/* -------------------------------------------------------------------------- */
|
|
303
|
+
/**
|
|
304
|
+
* The model LINE a reference names, with any pinned revision dropped.
|
|
305
|
+
*
|
|
306
|
+
* Substitution is a question about the line — `anthropic/claude-opus-5` — never
|
|
307
|
+
* about the revision, so the comparisons below have to be made on it. Same split
|
|
308
|
+
* `resolveEdgeRoute` makes on the way in.
|
|
309
|
+
*/
|
|
310
|
+
const modelLineOf = (reference) => {
|
|
311
|
+
const at = reference.indexOf("@");
|
|
312
|
+
return at === -1 ? reference : reference.slice(0, at);
|
|
313
|
+
};
|
|
314
|
+
/**
|
|
315
|
+
* The canonical internal request Oxy forwards to the data plane.
|
|
316
|
+
*
|
|
317
|
+
* `target` distinguishes the two questions a caller can ask — "serve THIS
|
|
318
|
+
* model" versus "choose one for me" — structurally. Everything downstream that
|
|
319
|
+
* must not silently substitute a model reads that discriminant rather than
|
|
320
|
+
* inferring intent from a string.
|
|
321
|
+
*/
|
|
322
|
+
export const inferenceRequestSchema = z
|
|
323
|
+
.object({
|
|
324
|
+
/** See `version.ts`: this is the Oxy→data-plane request envelope. */
|
|
325
|
+
schemaVersion: z.literal(2),
|
|
326
|
+
attribution: inferenceAttributionSchema,
|
|
327
|
+
target: routingTargetSchema,
|
|
328
|
+
modality: inferenceModalitySchema,
|
|
329
|
+
input: inferenceInputSchema,
|
|
330
|
+
stream: z.boolean(),
|
|
331
|
+
maxOutputTokens: z.number().int().positive().safe().optional(),
|
|
332
|
+
sampling: samplingParametersSchema,
|
|
333
|
+
tools: z.array(toolDefinitionSchema).default([]),
|
|
334
|
+
toolChoice: toolChoiceSchema.optional(),
|
|
335
|
+
responseFormat: responseFormatSchema.optional(),
|
|
336
|
+
client: clientRequestMetadataSchema,
|
|
337
|
+
/** Present when the operation is safe to deduplicate on retry. */
|
|
338
|
+
idempotencyKey: idempotencyKeySchema.optional(),
|
|
339
|
+
/** The exact policy revision this request is served under. */
|
|
340
|
+
routingPolicy: routingPolicyReferenceSchema,
|
|
341
|
+
/**
|
|
342
|
+
* The routes the control plane has already authorized for this request, in
|
|
343
|
+
* PREFERENCE ORDER. The first entry is the primary route Oxy resolved; the
|
|
344
|
+
* data plane fails over by taking the next one.
|
|
345
|
+
*
|
|
346
|
+
* This is what closes the gap ADR 0010's amendment left open. That amendment
|
|
347
|
+
* assigns the data plane "failover within the destinations the policy
|
|
348
|
+
* authorized" — and the envelope named no destinations, so a data plane could
|
|
349
|
+
* only fail over by re-deriving the customer's policy from values it does not
|
|
350
|
+
* have. Enumerating the survivors instead means a route switch outside the
|
|
351
|
+
* policy is impossible BY CONSTRUCTION rather than by two enforcement engines
|
|
352
|
+
* agreeing in two languages.
|
|
353
|
+
*
|
|
354
|
+
* **Absent means no failover is authorized, never "choose freely."** It is
|
|
355
|
+
* the state every envelope built before this field existed is in, and the
|
|
356
|
+
* behaviour a data plane that reads no list must already have: resolve the
|
|
357
|
+
* `target` and serve it or fail. Permission is granted by an ENTRY, so its
|
|
358
|
+
* absence can only ever narrow, and there is no reading of an absent list
|
|
359
|
+
* that widens what may be served.
|
|
360
|
+
*
|
|
361
|
+
* An EMPTY list is refused rather than treated as that state. `[]` would say
|
|
362
|
+
* "no route is authorized at all", which contradicts an envelope that was
|
|
363
|
+
* built to be served, and it is exactly the "permission granted, destination
|
|
364
|
+
* unnamed" shape `authorizedRouteSchema` exists to make unrepresentable.
|
|
365
|
+
*
|
|
366
|
+
* No price rides here. Oxy sized the hold against the most expensive route
|
|
367
|
+
* the policy permits (`usageReservationRequestSchema.ceilingPriceVersionId`),
|
|
368
|
+
* and every entry is one that policy permitted, so no failover among them can
|
|
369
|
+
* exceed it. A per-route price would also be a second authority for ranking,
|
|
370
|
+
* beside the order this list already carries.
|
|
371
|
+
*/
|
|
372
|
+
authorizedRoutes: z.array(authorizedRouteSchema).min(1).optional(),
|
|
373
|
+
})
|
|
374
|
+
.superRefine((request, ctx) => {
|
|
375
|
+
if (request.toolChoice !== undefined && request.tools.length === 0) {
|
|
376
|
+
ctx.addIssue({
|
|
377
|
+
code: z.ZodIssueCode.custom,
|
|
378
|
+
path: ["toolChoice"],
|
|
379
|
+
message: "a tool choice requires at least one tool definition",
|
|
380
|
+
});
|
|
381
|
+
}
|
|
382
|
+
const toolNames = request.tools.map((tool) => tool.name);
|
|
383
|
+
if (new Set(toolNames).size !== toolNames.length) {
|
|
384
|
+
ctx.addIssue({
|
|
385
|
+
code: z.ZodIssueCode.custom,
|
|
386
|
+
path: ["tools"],
|
|
387
|
+
message: "tool names must be unique within one request",
|
|
388
|
+
});
|
|
389
|
+
}
|
|
390
|
+
const routes = request.authorizedRoutes;
|
|
391
|
+
// The emptiness is RE-CHECKED rather than assumed away by `.min(1)`. A
|
|
392
|
+
// failed `.min()` marks the parse dirty rather than aborting it, so zod runs
|
|
393
|
+
// this refinement with the empty array still in hand; `[]` is already
|
|
394
|
+
// refused above, and reading `routes[0]` here would throw instead.
|
|
395
|
+
if (routes === undefined || routes.length === 0)
|
|
396
|
+
return;
|
|
397
|
+
const primary = routes[0];
|
|
398
|
+
// The primary is not a substitution for itself, and every `substitution`
|
|
399
|
+
// value is read RELATIVE to it. A list whose first entry claims to be a
|
|
400
|
+
// cross-model substitute names no original to have substituted for.
|
|
401
|
+
if (primary.substitution !== "same_model") {
|
|
402
|
+
ctx.addIssue({
|
|
403
|
+
code: z.ZodIssueCode.custom,
|
|
404
|
+
path: ["authorizedRoutes", 0, "substitution"],
|
|
405
|
+
message: "the first authorized route is the primary and cannot be a substitution",
|
|
406
|
+
});
|
|
407
|
+
}
|
|
408
|
+
const primaryLine = modelLineOf(primary.modelReference);
|
|
409
|
+
// A request that named a concrete model is served or refused, never
|
|
410
|
+
// substituted, and one that PINNED a revision is served on exactly those
|
|
411
|
+
// weights. Both checks are on the primary, because a primary that already
|
|
412
|
+
// drifted makes every entry after it a substitution nobody labelled.
|
|
413
|
+
if (request.target.kind === "model") {
|
|
414
|
+
const targetReference = request.target.modelReference;
|
|
415
|
+
const targetIsPinned = targetReference.includes("@");
|
|
416
|
+
if (targetIsPinned && primary.modelReference !== targetReference) {
|
|
417
|
+
ctx.addIssue({
|
|
418
|
+
code: z.ZodIssueCode.custom,
|
|
419
|
+
path: ["authorizedRoutes", 0, "modelReference"],
|
|
420
|
+
message: "a pinned request is served on exactly the revision it pinned",
|
|
421
|
+
});
|
|
422
|
+
}
|
|
423
|
+
if (!targetIsPinned && primaryLine !== targetReference) {
|
|
424
|
+
ctx.addIssue({
|
|
425
|
+
code: z.ZodIssueCode.custom,
|
|
426
|
+
path: ["authorizedRoutes", 0, "modelReference"],
|
|
427
|
+
message: "the primary authorized route must serve the model the request named",
|
|
428
|
+
});
|
|
429
|
+
}
|
|
430
|
+
if (targetIsPinned) {
|
|
431
|
+
for (const [index, route] of routes.entries()) {
|
|
432
|
+
if (route.substitution === "cross_model") {
|
|
433
|
+
ctx.addIssue({
|
|
434
|
+
code: z.ZodIssueCode.custom,
|
|
435
|
+
path: ["authorizedRoutes", index, "substitution"],
|
|
436
|
+
message: "a request that pinned a revision authorizes no cross-model substitute",
|
|
437
|
+
});
|
|
438
|
+
}
|
|
439
|
+
}
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
for (const [index, route] of routes.entries()) {
|
|
443
|
+
const line = modelLineOf(route.modelReference);
|
|
444
|
+
// A mislabelled entry is the whole failure mode: `same_model` on a
|
|
445
|
+
// different model line is a substitution wearing the label that needs no
|
|
446
|
+
// authorization, and `cross_model` on the same line claims an
|
|
447
|
+
// authorization the customer never had to give.
|
|
448
|
+
if (route.substitution === "same_model" && line !== primaryLine) {
|
|
449
|
+
ctx.addIssue({
|
|
450
|
+
code: z.ZodIssueCode.custom,
|
|
451
|
+
path: ["authorizedRoutes", index, "substitution"],
|
|
452
|
+
message: `route ${index} serves ${line}, not ${primaryLine}, so it is a cross-model substitute`,
|
|
453
|
+
});
|
|
454
|
+
}
|
|
455
|
+
if (route.substitution === "cross_model" && line === primaryLine) {
|
|
456
|
+
ctx.addIssue({
|
|
457
|
+
code: z.ZodIssueCode.custom,
|
|
458
|
+
path: ["authorizedRoutes", index, "substitution"],
|
|
459
|
+
message: `route ${index} serves ${primaryLine}, so it is same-model failover`,
|
|
460
|
+
});
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
// Failing over to the deployment that just failed is not failover. The
|
|
464
|
+
// duplicate would also make `routeSwitches` count a switch that changed
|
|
465
|
+
// nothing.
|
|
466
|
+
const deployments = routes.map((route) => route.deploymentId);
|
|
467
|
+
if (new Set(deployments).size !== deployments.length) {
|
|
468
|
+
ctx.addIssue({
|
|
469
|
+
code: z.ZodIssueCode.custom,
|
|
470
|
+
path: ["authorizedRoutes"],
|
|
471
|
+
message: "each deployment appears at most once in the authorized route list",
|
|
472
|
+
});
|
|
473
|
+
}
|
|
474
|
+
});
|