@velum-labs/routekit-gateway 1.0.7 → 1.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/responses-codec.js +11 -2
- package/dist/adapters/responses-stream.js +27 -9
- package/dist/observability/provenance.js +53 -13
- package/dist/providers/anthropic-codec.js +127 -4
- package/dist/test/provenance.test.js +62 -0
- package/dist/test/provider-validation-anthropic.test.js +171 -0
- package/dist/test/responses-streaming-catalog.test.js +32 -1
- package/package.json +7 -7
|
@@ -740,17 +740,26 @@ function buildOutput(message, toolRegistry) {
|
|
|
740
740
|
return output;
|
|
741
741
|
}
|
|
742
742
|
export function chatToResponses(openai, model, toolRegistry = EMPTY_TOOL_REGISTRY, searches = []) {
|
|
743
|
-
const
|
|
743
|
+
const choice = openai.choices?.[0];
|
|
744
|
+
const message = choice?.message;
|
|
745
|
+
const incompleteReason = choice?.finish_reason === "length" ||
|
|
746
|
+
choice?.finish_reason === "max_tokens" ||
|
|
747
|
+
choice?.finish_reason === "max_output_tokens"
|
|
748
|
+
? "max_output_tokens"
|
|
749
|
+
: choice?.finish_reason === "content_filter"
|
|
750
|
+
? "content_filter"
|
|
751
|
+
: undefined;
|
|
744
752
|
// Gateway-executed searches happened before the terminal step's output.
|
|
745
753
|
const output = [...searches.map(executedSearchItem), ...buildOutput(message, toolRegistry)];
|
|
746
754
|
return {
|
|
747
755
|
id: `resp_${openai.id ?? randomId()}`,
|
|
748
756
|
object: "response",
|
|
749
757
|
created_at: Math.floor(Date.now() / 1000),
|
|
750
|
-
status: "completed",
|
|
758
|
+
status: incompleteReason === undefined ? "completed" : "incomplete",
|
|
751
759
|
model,
|
|
752
760
|
output,
|
|
753
761
|
usage: chatUsageToResponses(openai.usage),
|
|
762
|
+
...(incompleteReason === undefined ? {} : { incomplete_details: { reason: incompleteReason } }),
|
|
754
763
|
...(openai.provider_cost !== undefined ? { provider_cost: openai.provider_cost } : {})
|
|
755
764
|
};
|
|
756
765
|
}
|
|
@@ -123,7 +123,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
123
123
|
let nextOutputIndex = 0;
|
|
124
124
|
let messageOutputIndex = -1;
|
|
125
125
|
let finished = false;
|
|
126
|
-
let
|
|
126
|
+
let finishReason;
|
|
127
127
|
let usage;
|
|
128
128
|
let providerCost;
|
|
129
129
|
let sequenceNumber = 0;
|
|
@@ -136,7 +136,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
136
136
|
// search, completed items collected for the terminal response payload.
|
|
137
137
|
const openSearches = new Map();
|
|
138
138
|
const completedSearchItems = [];
|
|
139
|
-
const baseResponse = (status, output) => ({
|
|
139
|
+
const baseResponse = (status, output, incompleteReason) => ({
|
|
140
140
|
id: responseId,
|
|
141
141
|
object: "response",
|
|
142
142
|
created_at: Math.floor(Date.now() / 1000),
|
|
@@ -144,6 +144,9 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
144
144
|
model,
|
|
145
145
|
output,
|
|
146
146
|
usage: status === "completed" ? chatUsageToResponses(usage) : null,
|
|
147
|
+
...(status === "incomplete" && incompleteReason !== undefined
|
|
148
|
+
? { incomplete_details: { reason: incompleteReason } }
|
|
149
|
+
: {}),
|
|
147
150
|
...(status === "completed" && providerCost !== undefined ? { provider_cost: providerCost } : {})
|
|
148
151
|
});
|
|
149
152
|
const ensureCreated = (controller) => {
|
|
@@ -341,7 +344,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
341
344
|
}
|
|
342
345
|
return indexed.sort((a, b) => a.outputIndex - b.outputIndex).map(({ item }) => item);
|
|
343
346
|
};
|
|
344
|
-
const finalize = (controller, terminal = "completed") => {
|
|
347
|
+
const finalize = (controller, terminal = "completed", incompleteReason) => {
|
|
345
348
|
if (finished)
|
|
346
349
|
return;
|
|
347
350
|
finished = true;
|
|
@@ -410,7 +413,22 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
410
413
|
// as incomplete.
|
|
411
414
|
controller.enqueue(terminal === "completed"
|
|
412
415
|
? emit("response.completed", { response: baseResponse("completed", assembleOutput()) })
|
|
413
|
-
: emit("response.incomplete", {
|
|
416
|
+
: emit("response.incomplete", {
|
|
417
|
+
response: baseResponse("incomplete", assembleOutput(), incompleteReason)
|
|
418
|
+
}));
|
|
419
|
+
};
|
|
420
|
+
const finalizeFromFinishReason = (controller) => {
|
|
421
|
+
if (finishReason === "length" ||
|
|
422
|
+
finishReason === "max_tokens" ||
|
|
423
|
+
finishReason === "max_output_tokens") {
|
|
424
|
+
finalize(controller, "incomplete", "max_output_tokens");
|
|
425
|
+
return;
|
|
426
|
+
}
|
|
427
|
+
if (finishReason === "content_filter") {
|
|
428
|
+
finalize(controller, "incomplete", "content_filter");
|
|
429
|
+
return;
|
|
430
|
+
}
|
|
431
|
+
finalize(controller, finishReason === undefined ? "incomplete" : "completed");
|
|
414
432
|
};
|
|
415
433
|
// A mid-stream provider failure (`data: {"error": {...}}` — e.g. the
|
|
416
434
|
// router's classified provider_error) becomes a `response.failed` event
|
|
@@ -590,9 +608,9 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
590
608
|
}
|
|
591
609
|
if (choice.finish_reason !== null && choice.finish_reason !== undefined) {
|
|
592
610
|
// OpenAI can emit token usage/provider cost in a later choices:[] chunk.
|
|
593
|
-
// Record
|
|
594
|
-
//
|
|
595
|
-
|
|
611
|
+
// Record how the turn ended, but wait for [DONE] or EOF before emitting
|
|
612
|
+
// the one terminal Responses event.
|
|
613
|
+
finishReason = choice.finish_reason;
|
|
596
614
|
}
|
|
597
615
|
};
|
|
598
616
|
const handleEvent = (controller, data) => {
|
|
@@ -601,7 +619,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
601
619
|
if (data === "[DONE]") {
|
|
602
620
|
// A `[DONE]` with no prior finish_reason is truncation, not a clean stop.
|
|
603
621
|
if (!finished)
|
|
604
|
-
|
|
622
|
+
finalizeFromFinishReason(controller);
|
|
605
623
|
return;
|
|
606
624
|
}
|
|
607
625
|
let chunk;
|
|
@@ -639,7 +657,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
|
|
|
639
657
|
},
|
|
640
658
|
onEnd(controller) {
|
|
641
659
|
if (!finished)
|
|
642
|
-
|
|
660
|
+
finalizeFromFinishReason(controller);
|
|
643
661
|
}
|
|
644
662
|
});
|
|
645
663
|
}
|
|
@@ -165,19 +165,55 @@ function streamedProviderError(body) {
|
|
|
165
165
|
const response = asRecord(payload?.response);
|
|
166
166
|
return (asRecord(payload?.error) ??
|
|
167
167
|
asRecord(response?.error) ??
|
|
168
|
-
asRecord(response?.incomplete_details)
|
|
168
|
+
asRecord(response?.incomplete_details) ?? {
|
|
169
|
+
type: eventType === "response.incomplete" ? "response_incomplete" : "response_failed"
|
|
170
|
+
});
|
|
169
171
|
}
|
|
170
172
|
return undefined;
|
|
171
173
|
}
|
|
172
|
-
function
|
|
173
|
-
const
|
|
174
|
+
function bufferedProviderError(body) {
|
|
175
|
+
const response = asRecord(parseJson(body));
|
|
176
|
+
if (response?.status === "incomplete") {
|
|
177
|
+
return asRecord(response.incomplete_details) ?? { type: "response_incomplete" };
|
|
178
|
+
}
|
|
179
|
+
if (response?.status === "failed") {
|
|
180
|
+
return asRecord(response.error) ?? { type: "response_failed" };
|
|
181
|
+
}
|
|
182
|
+
return undefined;
|
|
183
|
+
}
|
|
184
|
+
function requestedMaximumOutputTokens(body) {
|
|
185
|
+
const request = asRecord(body);
|
|
186
|
+
for (const field of ["max_output_tokens", "max_completion_tokens", "max_tokens"]) {
|
|
187
|
+
const value = request?.[field];
|
|
188
|
+
if (typeof value === "number" && Number.isInteger(value) && value > 0)
|
|
189
|
+
return value;
|
|
190
|
+
}
|
|
191
|
+
return undefined;
|
|
192
|
+
}
|
|
193
|
+
function terminalStopReason(context, result, usage) {
|
|
194
|
+
const terminalError = bufferedProviderError(result.responseBody) ?? streamedProviderError(result.responseBody);
|
|
195
|
+
const reason = terminalError?.reason ?? terminalError?.code;
|
|
196
|
+
if (typeof reason === "string" && reason.length > 0)
|
|
197
|
+
return reason;
|
|
198
|
+
const maximumOutputTokens = requestedMaximumOutputTokens(context.requestBody);
|
|
199
|
+
return maximumOutputTokens !== undefined &&
|
|
200
|
+
usage?.completion_tokens !== undefined &&
|
|
201
|
+
usage.completion_tokens >= maximumOutputTokens
|
|
202
|
+
? "max_output_tokens"
|
|
203
|
+
: undefined;
|
|
204
|
+
}
|
|
205
|
+
function providerError(result, stopReason) {
|
|
206
|
+
const terminalError = bufferedProviderError(result.responseBody) ?? streamedProviderError(result.responseBody);
|
|
174
207
|
if (result.error === undefined &&
|
|
175
208
|
result.statusCode >= 200 &&
|
|
176
209
|
result.statusCode < 400 &&
|
|
177
|
-
|
|
210
|
+
terminalError === undefined &&
|
|
211
|
+
stopReason === undefined) {
|
|
178
212
|
return undefined;
|
|
179
213
|
}
|
|
180
|
-
const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ??
|
|
214
|
+
const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ??
|
|
215
|
+
terminalError ??
|
|
216
|
+
(stopReason === undefined ? undefined : { reason: stopReason });
|
|
181
217
|
const noModelAvailable = result.statusCode === 503 &&
|
|
182
218
|
responseError?.type === "unavailable" &&
|
|
183
219
|
responseError.message === "no model is available; configure a provider";
|
|
@@ -196,13 +232,15 @@ function providerError(result) {
|
|
|
196
232
|
: "provider_error";
|
|
197
233
|
const message = kind === "capability_missing"
|
|
198
234
|
? "no model route is configured"
|
|
199
|
-
:
|
|
200
|
-
? "provider
|
|
201
|
-
: kind === "
|
|
202
|
-
? "provider
|
|
203
|
-
: kind === "
|
|
204
|
-
? "provider
|
|
205
|
-
:
|
|
235
|
+
: responseError?.reason === "max_output_tokens"
|
|
236
|
+
? "provider response reached the maximum output token limit"
|
|
237
|
+
: kind === "timeout"
|
|
238
|
+
? "provider request timed out"
|
|
239
|
+
: kind === "rate_limited"
|
|
240
|
+
? "provider rate limited the request"
|
|
241
|
+
: kind === "validation_error"
|
|
242
|
+
? "provider rejected the request"
|
|
243
|
+
: "provider request failed";
|
|
206
244
|
return {
|
|
207
245
|
kind,
|
|
208
246
|
message,
|
|
@@ -225,13 +263,15 @@ export function buildModelCallRecord(context, result) {
|
|
|
225
263
|
totalTokens: usage.total_tokens
|
|
226
264
|
}
|
|
227
265
|
});
|
|
228
|
-
const
|
|
266
|
+
const stopReason = terminalStopReason(context, result, usage);
|
|
267
|
+
const error = providerError(result, stopReason);
|
|
229
268
|
const metadata = {
|
|
230
269
|
dialect: context.dialect,
|
|
231
270
|
stream: context.stream,
|
|
232
271
|
http_status: result.statusCode,
|
|
233
272
|
duration_ms: result.durationMs,
|
|
234
273
|
requested_model: context.requestedModel ?? null,
|
|
274
|
+
...(stopReason === undefined ? {} : { stop_reason: stopReason }),
|
|
235
275
|
unknown_usage: callCost.unknownUsage,
|
|
236
276
|
unknown_cost: callCost.unknownCost,
|
|
237
277
|
...(context.attribution !== undefined
|
|
@@ -135,6 +135,114 @@ function anthropicToolChoice(choice, parallelToolCalls) {
|
|
|
135
135
|
? { type: "tool", name: fn.name, ...disableParallel }
|
|
136
136
|
: undefined;
|
|
137
137
|
}
|
|
138
|
+
const ANTHROPIC_SUPPORTED_SCHEMA_KEYWORDS = new Set([
|
|
139
|
+
"$defs",
|
|
140
|
+
"$ref",
|
|
141
|
+
"additionalProperties",
|
|
142
|
+
"allOf",
|
|
143
|
+
"anyOf",
|
|
144
|
+
"const",
|
|
145
|
+
"description",
|
|
146
|
+
"enum",
|
|
147
|
+
"format",
|
|
148
|
+
"items",
|
|
149
|
+
"minItems",
|
|
150
|
+
"oneOf",
|
|
151
|
+
"pattern",
|
|
152
|
+
"properties",
|
|
153
|
+
"required",
|
|
154
|
+
"title",
|
|
155
|
+
"type"
|
|
156
|
+
]);
|
|
157
|
+
const ANTHROPIC_SUPPORTED_STRING_FORMATS = new Set([
|
|
158
|
+
"date",
|
|
159
|
+
"date-time",
|
|
160
|
+
"duration",
|
|
161
|
+
"email",
|
|
162
|
+
"hostname",
|
|
163
|
+
"ipv4",
|
|
164
|
+
"ipv6",
|
|
165
|
+
"time",
|
|
166
|
+
"uri",
|
|
167
|
+
"uuid"
|
|
168
|
+
]);
|
|
169
|
+
function isRecord(value) {
|
|
170
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
171
|
+
}
|
|
172
|
+
function cloneSchemaValue(value) {
|
|
173
|
+
if (Array.isArray(value))
|
|
174
|
+
return value.map(cloneSchemaValue);
|
|
175
|
+
if (!isRecord(value))
|
|
176
|
+
return value;
|
|
177
|
+
return Object.fromEntries(Object.entries(value).map(([key, child]) => [key, cloneSchemaValue(child)]));
|
|
178
|
+
}
|
|
179
|
+
function sanitizedSchemaMap(value) {
|
|
180
|
+
if (!isRecord(value))
|
|
181
|
+
return cloneSchemaValue(value);
|
|
182
|
+
return Object.fromEntries(Object.entries(value).map(([key, schema]) => [key, anthropicStructuredOutputSchema(schema)]));
|
|
183
|
+
}
|
|
184
|
+
/**
|
|
185
|
+
* Anthropic structured outputs accept only a subset of JSON Schema. Match the
|
|
186
|
+
* documented SDK behavior by retaining that subset, converting `oneOf` to
|
|
187
|
+
* `anyOf`, and moving omitted constraints into descriptions for model guidance.
|
|
188
|
+
* Callers remain responsible for validating the parsed response against the
|
|
189
|
+
* original schema.
|
|
190
|
+
*/
|
|
191
|
+
function anthropicStructuredOutputSchema(schema) {
|
|
192
|
+
if (!isRecord(schema))
|
|
193
|
+
return cloneSchemaValue(schema);
|
|
194
|
+
if (typeof schema.$ref === "string")
|
|
195
|
+
return { $ref: schema.$ref };
|
|
196
|
+
const sanitized = {};
|
|
197
|
+
const deferredConstraints = [];
|
|
198
|
+
let oneOf;
|
|
199
|
+
for (const [key, value] of Object.entries(schema)) {
|
|
200
|
+
if (!ANTHROPIC_SUPPORTED_SCHEMA_KEYWORDS.has(key) ||
|
|
201
|
+
(key === "minItems" && value !== 0 && value !== 1) ||
|
|
202
|
+
(key === "format" &&
|
|
203
|
+
(typeof value !== "string" || !ANTHROPIC_SUPPORTED_STRING_FORMATS.has(value)))) {
|
|
204
|
+
deferredConstraints.push([key, cloneSchemaValue(value)]);
|
|
205
|
+
continue;
|
|
206
|
+
}
|
|
207
|
+
if (key === "$defs" || key === "properties") {
|
|
208
|
+
sanitized[key] = sanitizedSchemaMap(value);
|
|
209
|
+
continue;
|
|
210
|
+
}
|
|
211
|
+
if ((key === "allOf" || key === "anyOf") && Array.isArray(value)) {
|
|
212
|
+
sanitized[key] = value.map(anthropicStructuredOutputSchema);
|
|
213
|
+
continue;
|
|
214
|
+
}
|
|
215
|
+
if (key === "oneOf") {
|
|
216
|
+
oneOf = value;
|
|
217
|
+
continue;
|
|
218
|
+
}
|
|
219
|
+
if (key === "additionalProperties" || key === "items") {
|
|
220
|
+
sanitized[key] = Array.isArray(value)
|
|
221
|
+
? value.map(anthropicStructuredOutputSchema)
|
|
222
|
+
: anthropicStructuredOutputSchema(value);
|
|
223
|
+
continue;
|
|
224
|
+
}
|
|
225
|
+
sanitized[key] = cloneSchemaValue(value);
|
|
226
|
+
}
|
|
227
|
+
if (oneOf !== undefined) {
|
|
228
|
+
if (!Object.hasOwn(sanitized, "anyOf") && Array.isArray(oneOf)) {
|
|
229
|
+
sanitized.anyOf = oneOf.map(anthropicStructuredOutputSchema);
|
|
230
|
+
}
|
|
231
|
+
else {
|
|
232
|
+
deferredConstraints.push(["oneOf", cloneSchemaValue(oneOf)]);
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
if (deferredConstraints.length > 0) {
|
|
236
|
+
const constraintDescription = `{${deferredConstraints
|
|
237
|
+
.map(([key, value]) => `${key}: ${JSON.stringify(value)}`)
|
|
238
|
+
.join(", ")}}`;
|
|
239
|
+
sanitized.description =
|
|
240
|
+
typeof sanitized.description === "string" && sanitized.description.length > 0
|
|
241
|
+
? `${sanitized.description}\n\n${constraintDescription}`
|
|
242
|
+
: constraintDescription;
|
|
243
|
+
}
|
|
244
|
+
return sanitized;
|
|
245
|
+
}
|
|
138
246
|
function anthropicStructuredOutputFormat(responseFormat) {
|
|
139
247
|
if (responseFormat?.type !== "json_schema" ||
|
|
140
248
|
typeof responseFormat.json_schema?.schema !== "object" ||
|
|
@@ -143,7 +251,22 @@ function anthropicStructuredOutputFormat(responseFormat) {
|
|
|
143
251
|
}
|
|
144
252
|
return {
|
|
145
253
|
type: "json_schema",
|
|
146
|
-
schema: responseFormat.json_schema.schema
|
|
254
|
+
schema: anthropicStructuredOutputSchema(responseFormat.json_schema.schema)
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
function sanitizedAnthropicOutputConfig(outputConfig) {
|
|
258
|
+
if (outputConfig === undefined)
|
|
259
|
+
return undefined;
|
|
260
|
+
const format = outputConfig.format;
|
|
261
|
+
if (!isRecord(format) || format.type !== "json_schema" || !isRecord(format.schema)) {
|
|
262
|
+
return outputConfig;
|
|
263
|
+
}
|
|
264
|
+
return {
|
|
265
|
+
...outputConfig,
|
|
266
|
+
format: {
|
|
267
|
+
...format,
|
|
268
|
+
schema: anthropicStructuredOutputSchema(format.schema)
|
|
269
|
+
}
|
|
147
270
|
};
|
|
148
271
|
}
|
|
149
272
|
export function anthropicMessages(body, model) {
|
|
@@ -211,7 +334,7 @@ export function anthropicMessages(body, model) {
|
|
|
211
334
|
const translatedOutput = selection.mode === "effort" ? { effort: selection.effort } : undefined;
|
|
212
335
|
const thinking = metadata?.thinking ?? translatedThinking;
|
|
213
336
|
const structuredOutputFormat = anthropicStructuredOutputFormat(body.response_format);
|
|
214
|
-
const outputConfig = metadata?.output_config != null
|
|
337
|
+
const outputConfig = sanitizedAnthropicOutputConfig(metadata?.output_config != null
|
|
215
338
|
? {
|
|
216
339
|
...metadata.output_config,
|
|
217
340
|
...(structuredOutputFormat !== undefined &&
|
|
@@ -224,7 +347,7 @@ export function anthropicMessages(body, model) {
|
|
|
224
347
|
...translatedOutput,
|
|
225
348
|
...(structuredOutputFormat !== undefined ? { format: structuredOutputFormat } : {})
|
|
226
349
|
}
|
|
227
|
-
: undefined;
|
|
350
|
+
: undefined);
|
|
228
351
|
const toolChoice = anthropicToolChoice(body.tool_choice, body.parallel_tool_calls);
|
|
229
352
|
return {
|
|
230
353
|
model,
|
|
@@ -246,7 +369,7 @@ export function anthropicMessages(body, model) {
|
|
|
246
369
|
{
|
|
247
370
|
name: tool.function.name,
|
|
248
371
|
description: tool.function.description,
|
|
249
|
-
input_schema: tool.function.parameters ?? { type: "object" }
|
|
372
|
+
input_schema: anthropicStructuredOutputSchema(tool.function.parameters ?? { type: "object" })
|
|
250
373
|
}
|
|
251
374
|
])
|
|
252
375
|
}
|
|
@@ -261,3 +261,65 @@ test("HTTP 200 Codex terminal SSE quota failure is rate-limited provenance", ()
|
|
|
261
261
|
assert.equal(record.error?.kind, "rate_limited");
|
|
262
262
|
assert.equal(record.error?.retryable, true);
|
|
263
263
|
});
|
|
264
|
+
test("HTTP 200 incomplete Responses output is failed provenance", () => {
|
|
265
|
+
const record = buildModelCallRecord({
|
|
266
|
+
callId: "call_response_incomplete",
|
|
267
|
+
dialect: "openai-responses",
|
|
268
|
+
requestedModel: "claude-code/claude-opus-5",
|
|
269
|
+
model: "claude-code/claude-opus-5",
|
|
270
|
+
stream: false,
|
|
271
|
+
requestBody: { input: "author twenty evaluation cases" },
|
|
272
|
+
startedAt: "2026-08-20T13:00:00.000Z"
|
|
273
|
+
}, {
|
|
274
|
+
statusCode: 200,
|
|
275
|
+
durationMs: 5,
|
|
276
|
+
responseBody: Buffer.from(JSON.stringify({
|
|
277
|
+
status: "incomplete",
|
|
278
|
+
incomplete_details: { reason: "max_output_tokens" },
|
|
279
|
+
usage: { input_tokens: 31_533, output_tokens: 16_384 }
|
|
280
|
+
}))
|
|
281
|
+
});
|
|
282
|
+
assert.equal(record.status, "failed");
|
|
283
|
+
assert.equal(record.error?.message, "provider response reached the maximum output token limit");
|
|
284
|
+
assert.equal(record.usage?.completion_tokens, 16_384);
|
|
285
|
+
assert.equal(record.metadata?.stop_reason, "max_output_tokens");
|
|
286
|
+
});
|
|
287
|
+
test("an exact requested output-token cap is failed provenance", () => {
|
|
288
|
+
const record = buildModelCallRecord({
|
|
289
|
+
callId: "call_response_exact_cap",
|
|
290
|
+
dialect: "openai-responses",
|
|
291
|
+
requestedModel: "claude-code/claude-opus-5",
|
|
292
|
+
model: "claude-code/claude-opus-5",
|
|
293
|
+
stream: false,
|
|
294
|
+
requestBody: {
|
|
295
|
+
input: "author twenty evaluation cases",
|
|
296
|
+
max_output_tokens: 32_768
|
|
297
|
+
},
|
|
298
|
+
startedAt: "2026-08-20T13:00:00.000Z"
|
|
299
|
+
}, {
|
|
300
|
+
statusCode: 200,
|
|
301
|
+
durationMs: 5,
|
|
302
|
+
responseBody: Buffer.from(JSON.stringify({
|
|
303
|
+
status: "completed",
|
|
304
|
+
usage: { input_tokens: 31_533, output_tokens: 32_768 }
|
|
305
|
+
}))
|
|
306
|
+
});
|
|
307
|
+
assert.equal(record.status, "failed");
|
|
308
|
+
assert.equal(record.metadata?.stop_reason, "max_output_tokens");
|
|
309
|
+
});
|
|
310
|
+
test("HTTP 200 Responses stream without a finish reason is failed provenance", () => {
|
|
311
|
+
const record = buildModelCallRecord({
|
|
312
|
+
callId: "call_response_stream_truncated",
|
|
313
|
+
dialect: "openai-responses",
|
|
314
|
+
requestedModel: "openai/model",
|
|
315
|
+
model: "openai/model",
|
|
316
|
+
stream: true,
|
|
317
|
+
requestBody: { input: "answer", stream: true },
|
|
318
|
+
startedAt: "2026-08-20T13:00:00.000Z"
|
|
319
|
+
}, {
|
|
320
|
+
statusCode: 200,
|
|
321
|
+
durationMs: 5,
|
|
322
|
+
responseBody: Buffer.from('event: response.incomplete\ndata: {"type":"response.incomplete","response":{"status":"incomplete"}}\n\n')
|
|
323
|
+
});
|
|
324
|
+
assert.equal(record.status, "failed");
|
|
325
|
+
});
|
|
@@ -35,6 +35,177 @@ test("Responses JSON schemas reach Anthropic as native structured output formats
|
|
|
35
35
|
}
|
|
36
36
|
});
|
|
37
37
|
});
|
|
38
|
+
test("Anthropic structured outputs defer unsupported JSON Schema constraints", () => {
|
|
39
|
+
const schema = {
|
|
40
|
+
type: "object",
|
|
41
|
+
additionalProperties: false,
|
|
42
|
+
required: ["minimum", "scores", "labels"],
|
|
43
|
+
properties: {
|
|
44
|
+
minimum: { type: "integer", minimum: 1, maximum: 16_384 },
|
|
45
|
+
scores: {
|
|
46
|
+
type: "array",
|
|
47
|
+
minItems: 20,
|
|
48
|
+
maxItems: 20,
|
|
49
|
+
items: { type: "number", minimum: 0, maximum: 1 }
|
|
50
|
+
},
|
|
51
|
+
labels: {
|
|
52
|
+
type: "array",
|
|
53
|
+
minItems: 1,
|
|
54
|
+
items: { type: "string", minLength: 1, maxLength: 128 }
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
};
|
|
58
|
+
const original = structuredClone(schema);
|
|
59
|
+
const outbound = anthropicMessages({
|
|
60
|
+
model: "claude-opus-5",
|
|
61
|
+
messages: [{ role: "user", content: "author evaluations" }],
|
|
62
|
+
response_format: {
|
|
63
|
+
type: "json_schema",
|
|
64
|
+
json_schema: { name: "routekit_evaluations", schema, strict: true }
|
|
65
|
+
}
|
|
66
|
+
}, "claude-opus-5");
|
|
67
|
+
assert.deepEqual(schema, original);
|
|
68
|
+
assert.deepEqual(outbound.output_config, {
|
|
69
|
+
format: {
|
|
70
|
+
type: "json_schema",
|
|
71
|
+
schema: {
|
|
72
|
+
type: "object",
|
|
73
|
+
additionalProperties: false,
|
|
74
|
+
required: ["minimum", "scores", "labels"],
|
|
75
|
+
properties: {
|
|
76
|
+
minimum: {
|
|
77
|
+
type: "integer",
|
|
78
|
+
description: "{minimum: 1, maximum: 16384}"
|
|
79
|
+
},
|
|
80
|
+
scores: {
|
|
81
|
+
type: "array",
|
|
82
|
+
items: {
|
|
83
|
+
type: "number",
|
|
84
|
+
description: "{minimum: 0, maximum: 1}"
|
|
85
|
+
},
|
|
86
|
+
description: "{minItems: 20, maxItems: 20}"
|
|
87
|
+
},
|
|
88
|
+
labels: {
|
|
89
|
+
type: "array",
|
|
90
|
+
minItems: 1,
|
|
91
|
+
items: {
|
|
92
|
+
type: "string",
|
|
93
|
+
description: "{minLength: 1, maxLength: 128}"
|
|
94
|
+
}
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
});
|
|
100
|
+
const nativeOutbound = anthropicMessages(anthropicToChat({
|
|
101
|
+
model: "claude-opus-5",
|
|
102
|
+
messages: [{ role: "user", content: "author evaluations" }],
|
|
103
|
+
output_config: {
|
|
104
|
+
format: { type: "json_schema", schema }
|
|
105
|
+
}
|
|
106
|
+
}, "claude-opus-5"), "claude-opus-5");
|
|
107
|
+
assert.deepEqual(nativeOutbound.output_config, outbound.output_config);
|
|
108
|
+
});
|
|
109
|
+
test("Anthropic tool input schemas defer unsupported JSON Schema constraints", () => {
|
|
110
|
+
const parameters = {
|
|
111
|
+
type: "object",
|
|
112
|
+
additionalProperties: false,
|
|
113
|
+
required: ["limit", "scores"],
|
|
114
|
+
properties: {
|
|
115
|
+
limit: { type: "integer", minimum: 1, maximum: 16_384 },
|
|
116
|
+
scores: {
|
|
117
|
+
type: "array",
|
|
118
|
+
minItems: 20,
|
|
119
|
+
maxItems: 20,
|
|
120
|
+
items: { type: "number", minimum: 0, maximum: 1 }
|
|
121
|
+
}
|
|
122
|
+
}
|
|
123
|
+
};
|
|
124
|
+
const original = structuredClone(parameters);
|
|
125
|
+
const outbound = anthropicMessages({
|
|
126
|
+
model: "claude-opus-5",
|
|
127
|
+
messages: [{ role: "user", content: "score the inputs" }],
|
|
128
|
+
tools: [
|
|
129
|
+
{
|
|
130
|
+
type: "function",
|
|
131
|
+
function: {
|
|
132
|
+
name: "score_inputs",
|
|
133
|
+
description: "score structured inputs",
|
|
134
|
+
parameters
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
]
|
|
138
|
+
}, "claude-opus-5");
|
|
139
|
+
assert.deepEqual(parameters, original);
|
|
140
|
+
assert.deepEqual(outbound.tools, [
|
|
141
|
+
{
|
|
142
|
+
name: "score_inputs",
|
|
143
|
+
description: "score structured inputs",
|
|
144
|
+
input_schema: {
|
|
145
|
+
type: "object",
|
|
146
|
+
additionalProperties: false,
|
|
147
|
+
required: ["limit", "scores"],
|
|
148
|
+
properties: {
|
|
149
|
+
limit: {
|
|
150
|
+
type: "integer",
|
|
151
|
+
description: "{minimum: 1, maximum: 16384}"
|
|
152
|
+
},
|
|
153
|
+
scores: {
|
|
154
|
+
type: "array",
|
|
155
|
+
items: {
|
|
156
|
+
type: "number",
|
|
157
|
+
description: "{minimum: 0, maximum: 1}"
|
|
158
|
+
},
|
|
159
|
+
description: "{minItems: 20, maxItems: 20}"
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
]);
|
|
165
|
+
});
|
|
166
|
+
test("Anthropic structured outputs keep only supported schema keywords and formats", () => {
|
|
167
|
+
const outbound = anthropicMessages({
|
|
168
|
+
model: "claude-opus-5",
|
|
169
|
+
messages: [{ role: "user", content: "return a structured value" }],
|
|
170
|
+
response_format: {
|
|
171
|
+
type: "json_schema",
|
|
172
|
+
json_schema: {
|
|
173
|
+
name: "routekit_supported_schema",
|
|
174
|
+
strict: true,
|
|
175
|
+
schema: {
|
|
176
|
+
type: "object",
|
|
177
|
+
minProperties: 1,
|
|
178
|
+
properties: {
|
|
179
|
+
identifier: {
|
|
180
|
+
type: "string",
|
|
181
|
+
pattern: "^[a-z]+$",
|
|
182
|
+
format: "regex"
|
|
183
|
+
},
|
|
184
|
+
createdAt: {
|
|
185
|
+
type: "string",
|
|
186
|
+
format: "date-time"
|
|
187
|
+
},
|
|
188
|
+
choice: {
|
|
189
|
+
oneOf: [{ const: "one" }, { const: "two" }]
|
|
190
|
+
}
|
|
191
|
+
},
|
|
192
|
+
required: ["identifier", "createdAt", "choice"],
|
|
193
|
+
additionalProperties: false
|
|
194
|
+
}
|
|
195
|
+
}
|
|
196
|
+
}
|
|
197
|
+
}, "claude-opus-5");
|
|
198
|
+
const schema = outbound.output_config.format.schema;
|
|
199
|
+
const properties = schema.properties;
|
|
200
|
+
assert.equal(schema.minProperties, undefined);
|
|
201
|
+
assert.match(schema.description, /minProperties/u);
|
|
202
|
+
assert.equal(properties.identifier.format, undefined);
|
|
203
|
+
assert.equal(properties.identifier.pattern, "^[a-z]+$");
|
|
204
|
+
assert.match(properties.identifier.description, /format/u);
|
|
205
|
+
assert.equal(properties.createdAt.format, "date-time");
|
|
206
|
+
assert.equal(properties.choice.oneOf, undefined);
|
|
207
|
+
assert.deepEqual(properties.choice.anyOf, [{ const: "one" }, { const: "two" }]);
|
|
208
|
+
});
|
|
38
209
|
test("direct provider backends reject malformed reasoning controls before transport", async () => {
|
|
39
210
|
const cases = [
|
|
40
211
|
{
|
|
@@ -4,7 +4,7 @@ import { runRouteKitEffect } from "@velum-labs/routekit-runtime/effect";
|
|
|
4
4
|
import { Effect } from "effect";
|
|
5
5
|
import { reasoningSelectionOf } from "../adapters/openai-chat-wire.js";
|
|
6
6
|
import { parseResponsesEncryptedContent, wrapResponsesEncryptedContent } from "../adapters/openai-responses-wire.js";
|
|
7
|
-
import { openAiSseToResponses } from "../adapters/responses.js";
|
|
7
|
+
import { chatToResponses, openAiSseToResponses } from "../adapters/responses.js";
|
|
8
8
|
import { RoutingBackend } from "../routing/router.js";
|
|
9
9
|
import { startGateway } from "../gateway-service.js";
|
|
10
10
|
import { testProviderSource } from "./provider-source-fixture.js";
|
|
@@ -26,6 +26,37 @@ test("a mid-stream provider error event becomes response.failed with the upstrea
|
|
|
26
26
|
assert.ok(text.includes("openrouter call failed (unknown)"));
|
|
27
27
|
assert.ok(!text.includes("event: response.completed"));
|
|
28
28
|
});
|
|
29
|
+
test("max-token Chat completion becomes an incomplete Responses result", () => {
|
|
30
|
+
for (const finishReason of ["length", "max_tokens"]) {
|
|
31
|
+
const response = chatToResponses({
|
|
32
|
+
choices: [
|
|
33
|
+
{
|
|
34
|
+
message: { role: "assistant", content: '{"cases":[{"id":"truncated' },
|
|
35
|
+
finish_reason: finishReason
|
|
36
|
+
}
|
|
37
|
+
]
|
|
38
|
+
}, "claude-code/claude-opus-5");
|
|
39
|
+
assert.equal(response.status, "incomplete");
|
|
40
|
+
assert.deepEqual(response.incomplete_details, { reason: "max_output_tokens" });
|
|
41
|
+
}
|
|
42
|
+
});
|
|
43
|
+
test("max-token Chat stream becomes response.incomplete with its reason", async () => {
|
|
44
|
+
const stream = openAiSseToResponses(sseStream(`data: ${JSON.stringify({
|
|
45
|
+
choices: [
|
|
46
|
+
{
|
|
47
|
+
index: 0,
|
|
48
|
+
delta: { content: '{"cases":[{"id":"truncated' },
|
|
49
|
+
finish_reason: null
|
|
50
|
+
}
|
|
51
|
+
]
|
|
52
|
+
})}\n\n`, `data: ${JSON.stringify({
|
|
53
|
+
choices: [{ index: 0, delta: {}, finish_reason: "length" }]
|
|
54
|
+
})}\n\n`, "data: [DONE]\n\n"), "claude-code/claude-opus-5");
|
|
55
|
+
const text = await new Response(stream).text();
|
|
56
|
+
assert.match(text, /event: response\.incomplete/u);
|
|
57
|
+
assert.match(text, /"incomplete_details":\{"reason":"max_output_tokens"\}/u);
|
|
58
|
+
assert.doesNotMatch(text, /event: response\.completed/u);
|
|
59
|
+
});
|
|
29
60
|
test("translates a streamed Responses event sequence", async () => {
|
|
30
61
|
const mock = await startMock();
|
|
31
62
|
const gateway = await startGateway({
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@velum-labs/routekit-gateway",
|
|
3
3
|
"private": false,
|
|
4
|
-
"version": "1.0.
|
|
4
|
+
"version": "1.0.9",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
7
7
|
"url": "git+https://github.com/velum-labs/routekit.git",
|
|
@@ -45,12 +45,12 @@
|
|
|
45
45
|
"@aws-sdk/client-bedrock": "3.1095.0",
|
|
46
46
|
"@aws-sdk/client-bedrock-runtime": "3.1095.0",
|
|
47
47
|
"effect": "4.0.0-rc.108",
|
|
48
|
-
"@velum-labs/routekit-config-core": "1.0.
|
|
49
|
-
"@velum-labs/routekit-contracts": "1.0.
|
|
50
|
-
"@velum-labs/routekit-eval-contracts": "1.0.
|
|
51
|
-
"@velum-labs/routekit-eval-core": "1.0.
|
|
52
|
-
"@velum-labs/routekit-registry": "1.0.
|
|
53
|
-
"@velum-labs/routekit-runtime": "1.0.
|
|
48
|
+
"@velum-labs/routekit-config-core": "1.0.9",
|
|
49
|
+
"@velum-labs/routekit-contracts": "1.0.9",
|
|
50
|
+
"@velum-labs/routekit-eval-contracts": "1.0.9",
|
|
51
|
+
"@velum-labs/routekit-eval-core": "1.0.9",
|
|
52
|
+
"@velum-labs/routekit-registry": "1.0.9",
|
|
53
|
+
"@velum-labs/routekit-runtime": "1.0.9"
|
|
54
54
|
},
|
|
55
55
|
"keywords": [
|
|
56
56
|
"routekit",
|