@velum-labs/routekit-gateway 1.0.7 → 1.0.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -740,17 +740,26 @@ function buildOutput(message, toolRegistry) {
740
740
  return output;
741
741
  }
742
742
  export function chatToResponses(openai, model, toolRegistry = EMPTY_TOOL_REGISTRY, searches = []) {
743
- const message = openai.choices?.[0]?.message;
743
+ const choice = openai.choices?.[0];
744
+ const message = choice?.message;
745
+ const incompleteReason = choice?.finish_reason === "length" ||
746
+ choice?.finish_reason === "max_tokens" ||
747
+ choice?.finish_reason === "max_output_tokens"
748
+ ? "max_output_tokens"
749
+ : choice?.finish_reason === "content_filter"
750
+ ? "content_filter"
751
+ : undefined;
744
752
  // Gateway-executed searches happened before the terminal step's output.
745
753
  const output = [...searches.map(executedSearchItem), ...buildOutput(message, toolRegistry)];
746
754
  return {
747
755
  id: `resp_${openai.id ?? randomId()}`,
748
756
  object: "response",
749
757
  created_at: Math.floor(Date.now() / 1000),
750
- status: "completed",
758
+ status: incompleteReason === undefined ? "completed" : "incomplete",
751
759
  model,
752
760
  output,
753
761
  usage: chatUsageToResponses(openai.usage),
762
+ ...(incompleteReason === undefined ? {} : { incomplete_details: { reason: incompleteReason } }),
754
763
  ...(openai.provider_cost !== undefined ? { provider_cost: openai.provider_cost } : {})
755
764
  };
756
765
  }
@@ -123,7 +123,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
123
123
  let nextOutputIndex = 0;
124
124
  let messageOutputIndex = -1;
125
125
  let finished = false;
126
- let sawFinishReason = false;
126
+ let finishReason;
127
127
  let usage;
128
128
  let providerCost;
129
129
  let sequenceNumber = 0;
@@ -136,7 +136,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
136
136
  // search, completed items collected for the terminal response payload.
137
137
  const openSearches = new Map();
138
138
  const completedSearchItems = [];
139
- const baseResponse = (status, output) => ({
139
+ const baseResponse = (status, output, incompleteReason) => ({
140
140
  id: responseId,
141
141
  object: "response",
142
142
  created_at: Math.floor(Date.now() / 1000),
@@ -144,6 +144,9 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
144
144
  model,
145
145
  output,
146
146
  usage: status === "completed" ? chatUsageToResponses(usage) : null,
147
+ ...(status === "incomplete" && incompleteReason !== undefined
148
+ ? { incomplete_details: { reason: incompleteReason } }
149
+ : {}),
147
150
  ...(status === "completed" && providerCost !== undefined ? { provider_cost: providerCost } : {})
148
151
  });
149
152
  const ensureCreated = (controller) => {
@@ -341,7 +344,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
341
344
  }
342
345
  return indexed.sort((a, b) => a.outputIndex - b.outputIndex).map(({ item }) => item);
343
346
  };
344
- const finalize = (controller, terminal = "completed") => {
347
+ const finalize = (controller, terminal = "completed", incompleteReason) => {
345
348
  if (finished)
346
349
  return;
347
350
  finished = true;
@@ -410,7 +413,22 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
410
413
  // as incomplete.
411
414
  controller.enqueue(terminal === "completed"
412
415
  ? emit("response.completed", { response: baseResponse("completed", assembleOutput()) })
413
- : emit("response.incomplete", { response: baseResponse("incomplete", assembleOutput()) }));
416
+ : emit("response.incomplete", {
417
+ response: baseResponse("incomplete", assembleOutput(), incompleteReason)
418
+ }));
419
+ };
420
+ const finalizeFromFinishReason = (controller) => {
421
+ if (finishReason === "length" ||
422
+ finishReason === "max_tokens" ||
423
+ finishReason === "max_output_tokens") {
424
+ finalize(controller, "incomplete", "max_output_tokens");
425
+ return;
426
+ }
427
+ if (finishReason === "content_filter") {
428
+ finalize(controller, "incomplete", "content_filter");
429
+ return;
430
+ }
431
+ finalize(controller, finishReason === undefined ? "incomplete" : "completed");
414
432
  };
415
433
  // A mid-stream provider failure (`data: {"error": {...}}` — e.g. the
416
434
  // router's classified provider_error) becomes a `response.failed` event
@@ -590,9 +608,9 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
590
608
  }
591
609
  if (choice.finish_reason !== null && choice.finish_reason !== undefined) {
592
610
  // OpenAI can emit token usage/provider cost in a later choices:[] chunk.
593
- // Record that the turn ended cleanly, but wait for [DONE] or EOF before
594
- // emitting the one terminal Responses event.
595
- sawFinishReason = true;
611
+ // Record how the turn ended, but wait for [DONE] or EOF before emitting
612
+ // the one terminal Responses event.
613
+ finishReason = choice.finish_reason;
596
614
  }
597
615
  };
598
616
  const handleEvent = (controller, data) => {
@@ -601,7 +619,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
601
619
  if (data === "[DONE]") {
602
620
  // A `[DONE]` with no prior finish_reason is truncation, not a clean stop.
603
621
  if (!finished)
604
- finalize(controller, sawFinishReason ? "completed" : "incomplete");
622
+ finalizeFromFinishReason(controller);
605
623
  return;
606
624
  }
607
625
  let chunk;
@@ -639,7 +657,7 @@ export function openAiSseToResponses(upstream, model, toolRegistry = new Map())
639
657
  },
640
658
  onEnd(controller) {
641
659
  if (!finished)
642
- finalize(controller, sawFinishReason ? "completed" : "incomplete");
660
+ finalizeFromFinishReason(controller);
643
661
  }
644
662
  });
645
663
  }
@@ -165,19 +165,55 @@ function streamedProviderError(body) {
165
165
  const response = asRecord(payload?.response);
166
166
  return (asRecord(payload?.error) ??
167
167
  asRecord(response?.error) ??
168
- asRecord(response?.incomplete_details));
168
+ asRecord(response?.incomplete_details) ?? {
169
+ type: eventType === "response.incomplete" ? "response_incomplete" : "response_failed"
170
+ });
169
171
  }
170
172
  return undefined;
171
173
  }
172
- function providerError(result) {
173
- const streamError = streamedProviderError(result.responseBody);
174
+ function bufferedProviderError(body) {
175
+ const response = asRecord(parseJson(body));
176
+ if (response?.status === "incomplete") {
177
+ return asRecord(response.incomplete_details) ?? { type: "response_incomplete" };
178
+ }
179
+ if (response?.status === "failed") {
180
+ return asRecord(response.error) ?? { type: "response_failed" };
181
+ }
182
+ return undefined;
183
+ }
184
+ function requestedMaximumOutputTokens(body) {
185
+ const request = asRecord(body);
186
+ for (const field of ["max_output_tokens", "max_completion_tokens", "max_tokens"]) {
187
+ const value = request?.[field];
188
+ if (typeof value === "number" && Number.isInteger(value) && value > 0)
189
+ return value;
190
+ }
191
+ return undefined;
192
+ }
193
+ function terminalStopReason(context, result, usage) {
194
+ const terminalError = bufferedProviderError(result.responseBody) ?? streamedProviderError(result.responseBody);
195
+ const reason = terminalError?.reason ?? terminalError?.code;
196
+ if (typeof reason === "string" && reason.length > 0)
197
+ return reason;
198
+ const maximumOutputTokens = requestedMaximumOutputTokens(context.requestBody);
199
+ return maximumOutputTokens !== undefined &&
200
+ usage?.completion_tokens !== undefined &&
201
+ usage.completion_tokens >= maximumOutputTokens
202
+ ? "max_output_tokens"
203
+ : undefined;
204
+ }
205
+ function providerError(result, stopReason) {
206
+ const terminalError = bufferedProviderError(result.responseBody) ?? streamedProviderError(result.responseBody);
174
207
  if (result.error === undefined &&
175
208
  result.statusCode >= 200 &&
176
209
  result.statusCode < 400 &&
177
- streamError === undefined) {
210
+ terminalError === undefined &&
211
+ stopReason === undefined) {
178
212
  return undefined;
179
213
  }
180
- const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ?? streamError;
214
+ const responseError = asRecord(asRecord(parseJson(result.responseBody))?.error) ??
215
+ terminalError ??
216
+ (stopReason === undefined ? undefined : { reason: stopReason });
181
217
  const noModelAvailable = result.statusCode === 503 &&
182
218
  responseError?.type === "unavailable" &&
183
219
  responseError.message === "no model is available; configure a provider";
@@ -196,13 +232,15 @@ function providerError(result) {
196
232
  : "provider_error";
197
233
  const message = kind === "capability_missing"
198
234
  ? "no model route is configured"
199
- : kind === "timeout"
200
- ? "provider request timed out"
201
- : kind === "rate_limited"
202
- ? "provider rate limited the request"
203
- : kind === "validation_error"
204
- ? "provider rejected the request"
205
- : "provider request failed";
235
+ : responseError?.reason === "max_output_tokens"
236
+ ? "provider response reached the maximum output token limit"
237
+ : kind === "timeout"
238
+ ? "provider request timed out"
239
+ : kind === "rate_limited"
240
+ ? "provider rate limited the request"
241
+ : kind === "validation_error"
242
+ ? "provider rejected the request"
243
+ : "provider request failed";
206
244
  return {
207
245
  kind,
208
246
  message,
@@ -225,13 +263,15 @@ export function buildModelCallRecord(context, result) {
225
263
  totalTokens: usage.total_tokens
226
264
  }
227
265
  });
228
- const error = providerError(result);
266
+ const stopReason = terminalStopReason(context, result, usage);
267
+ const error = providerError(result, stopReason);
229
268
  const metadata = {
230
269
  dialect: context.dialect,
231
270
  stream: context.stream,
232
271
  http_status: result.statusCode,
233
272
  duration_ms: result.durationMs,
234
273
  requested_model: context.requestedModel ?? null,
274
+ ...(stopReason === undefined ? {} : { stop_reason: stopReason }),
235
275
  unknown_usage: callCost.unknownUsage,
236
276
  unknown_cost: callCost.unknownCost,
237
277
  ...(context.attribution !== undefined
@@ -135,6 +135,114 @@ function anthropicToolChoice(choice, parallelToolCalls) {
135
135
  ? { type: "tool", name: fn.name, ...disableParallel }
136
136
  : undefined;
137
137
  }
138
+ const ANTHROPIC_SUPPORTED_SCHEMA_KEYWORDS = new Set([
139
+ "$defs",
140
+ "$ref",
141
+ "additionalProperties",
142
+ "allOf",
143
+ "anyOf",
144
+ "const",
145
+ "description",
146
+ "enum",
147
+ "format",
148
+ "items",
149
+ "minItems",
150
+ "oneOf",
151
+ "pattern",
152
+ "properties",
153
+ "required",
154
+ "title",
155
+ "type"
156
+ ]);
157
+ const ANTHROPIC_SUPPORTED_STRING_FORMATS = new Set([
158
+ "date",
159
+ "date-time",
160
+ "duration",
161
+ "email",
162
+ "hostname",
163
+ "ipv4",
164
+ "ipv6",
165
+ "time",
166
+ "uri",
167
+ "uuid"
168
+ ]);
169
+ function isRecord(value) {
170
+ return typeof value === "object" && value !== null && !Array.isArray(value);
171
+ }
172
+ function cloneSchemaValue(value) {
173
+ if (Array.isArray(value))
174
+ return value.map(cloneSchemaValue);
175
+ if (!isRecord(value))
176
+ return value;
177
+ return Object.fromEntries(Object.entries(value).map(([key, child]) => [key, cloneSchemaValue(child)]));
178
+ }
179
+ function sanitizedSchemaMap(value) {
180
+ if (!isRecord(value))
181
+ return cloneSchemaValue(value);
182
+ return Object.fromEntries(Object.entries(value).map(([key, schema]) => [key, anthropicStructuredOutputSchema(schema)]));
183
+ }
184
+ /**
185
+ * Anthropic structured outputs accept only a subset of JSON Schema. Match the
186
+ * documented SDK behavior by retaining that subset, converting `oneOf` to
187
+ * `anyOf`, and moving omitted constraints into descriptions for model guidance.
188
+ * Callers remain responsible for validating the parsed response against the
189
+ * original schema.
190
+ */
191
+ function anthropicStructuredOutputSchema(schema) {
192
+ if (!isRecord(schema))
193
+ return cloneSchemaValue(schema);
194
+ if (typeof schema.$ref === "string")
195
+ return { $ref: schema.$ref };
196
+ const sanitized = {};
197
+ const deferredConstraints = [];
198
+ let oneOf;
199
+ for (const [key, value] of Object.entries(schema)) {
200
+ if (!ANTHROPIC_SUPPORTED_SCHEMA_KEYWORDS.has(key) ||
201
+ (key === "minItems" && value !== 0 && value !== 1) ||
202
+ (key === "format" &&
203
+ (typeof value !== "string" || !ANTHROPIC_SUPPORTED_STRING_FORMATS.has(value)))) {
204
+ deferredConstraints.push([key, cloneSchemaValue(value)]);
205
+ continue;
206
+ }
207
+ if (key === "$defs" || key === "properties") {
208
+ sanitized[key] = sanitizedSchemaMap(value);
209
+ continue;
210
+ }
211
+ if ((key === "allOf" || key === "anyOf") && Array.isArray(value)) {
212
+ sanitized[key] = value.map(anthropicStructuredOutputSchema);
213
+ continue;
214
+ }
215
+ if (key === "oneOf") {
216
+ oneOf = value;
217
+ continue;
218
+ }
219
+ if (key === "additionalProperties" || key === "items") {
220
+ sanitized[key] = Array.isArray(value)
221
+ ? value.map(anthropicStructuredOutputSchema)
222
+ : anthropicStructuredOutputSchema(value);
223
+ continue;
224
+ }
225
+ sanitized[key] = cloneSchemaValue(value);
226
+ }
227
+ if (oneOf !== undefined) {
228
+ if (!Object.hasOwn(sanitized, "anyOf") && Array.isArray(oneOf)) {
229
+ sanitized.anyOf = oneOf.map(anthropicStructuredOutputSchema);
230
+ }
231
+ else {
232
+ deferredConstraints.push(["oneOf", cloneSchemaValue(oneOf)]);
233
+ }
234
+ }
235
+ if (deferredConstraints.length > 0) {
236
+ const constraintDescription = `{${deferredConstraints
237
+ .map(([key, value]) => `${key}: ${JSON.stringify(value)}`)
238
+ .join(", ")}}`;
239
+ sanitized.description =
240
+ typeof sanitized.description === "string" && sanitized.description.length > 0
241
+ ? `${sanitized.description}\n\n${constraintDescription}`
242
+ : constraintDescription;
243
+ }
244
+ return sanitized;
245
+ }
138
246
  function anthropicStructuredOutputFormat(responseFormat) {
139
247
  if (responseFormat?.type !== "json_schema" ||
140
248
  typeof responseFormat.json_schema?.schema !== "object" ||
@@ -143,7 +251,22 @@ function anthropicStructuredOutputFormat(responseFormat) {
143
251
  }
144
252
  return {
145
253
  type: "json_schema",
146
- schema: responseFormat.json_schema.schema
254
+ schema: anthropicStructuredOutputSchema(responseFormat.json_schema.schema)
255
+ };
256
+ }
257
+ function sanitizedAnthropicOutputConfig(outputConfig) {
258
+ if (outputConfig === undefined)
259
+ return undefined;
260
+ const format = outputConfig.format;
261
+ if (!isRecord(format) || format.type !== "json_schema" || !isRecord(format.schema)) {
262
+ return outputConfig;
263
+ }
264
+ return {
265
+ ...outputConfig,
266
+ format: {
267
+ ...format,
268
+ schema: anthropicStructuredOutputSchema(format.schema)
269
+ }
147
270
  };
148
271
  }
149
272
  export function anthropicMessages(body, model) {
@@ -211,7 +334,7 @@ export function anthropicMessages(body, model) {
211
334
  const translatedOutput = selection.mode === "effort" ? { effort: selection.effort } : undefined;
212
335
  const thinking = metadata?.thinking ?? translatedThinking;
213
336
  const structuredOutputFormat = anthropicStructuredOutputFormat(body.response_format);
214
- const outputConfig = metadata?.output_config != null
337
+ const outputConfig = sanitizedAnthropicOutputConfig(metadata?.output_config != null
215
338
  ? {
216
339
  ...metadata.output_config,
217
340
  ...(structuredOutputFormat !== undefined &&
@@ -224,7 +347,7 @@ export function anthropicMessages(body, model) {
224
347
  ...translatedOutput,
225
348
  ...(structuredOutputFormat !== undefined ? { format: structuredOutputFormat } : {})
226
349
  }
227
- : undefined;
350
+ : undefined);
228
351
  const toolChoice = anthropicToolChoice(body.tool_choice, body.parallel_tool_calls);
229
352
  return {
230
353
  model,
@@ -246,7 +369,7 @@ export function anthropicMessages(body, model) {
246
369
  {
247
370
  name: tool.function.name,
248
371
  description: tool.function.description,
249
- input_schema: tool.function.parameters ?? { type: "object" }
372
+ input_schema: anthropicStructuredOutputSchema(tool.function.parameters ?? { type: "object" })
250
373
  }
251
374
  ])
252
375
  }
@@ -261,3 +261,65 @@ test("HTTP 200 Codex terminal SSE quota failure is rate-limited provenance", ()
261
261
  assert.equal(record.error?.kind, "rate_limited");
262
262
  assert.equal(record.error?.retryable, true);
263
263
  });
264
+ test("HTTP 200 incomplete Responses output is failed provenance", () => {
265
+ const record = buildModelCallRecord({
266
+ callId: "call_response_incomplete",
267
+ dialect: "openai-responses",
268
+ requestedModel: "claude-code/claude-opus-5",
269
+ model: "claude-code/claude-opus-5",
270
+ stream: false,
271
+ requestBody: { input: "author twenty evaluation cases" },
272
+ startedAt: "2026-08-20T13:00:00.000Z"
273
+ }, {
274
+ statusCode: 200,
275
+ durationMs: 5,
276
+ responseBody: Buffer.from(JSON.stringify({
277
+ status: "incomplete",
278
+ incomplete_details: { reason: "max_output_tokens" },
279
+ usage: { input_tokens: 31_533, output_tokens: 16_384 }
280
+ }))
281
+ });
282
+ assert.equal(record.status, "failed");
283
+ assert.equal(record.error?.message, "provider response reached the maximum output token limit");
284
+ assert.equal(record.usage?.completion_tokens, 16_384);
285
+ assert.equal(record.metadata?.stop_reason, "max_output_tokens");
286
+ });
287
+ test("an exact requested output-token cap is failed provenance", () => {
288
+ const record = buildModelCallRecord({
289
+ callId: "call_response_exact_cap",
290
+ dialect: "openai-responses",
291
+ requestedModel: "claude-code/claude-opus-5",
292
+ model: "claude-code/claude-opus-5",
293
+ stream: false,
294
+ requestBody: {
295
+ input: "author twenty evaluation cases",
296
+ max_output_tokens: 32_768
297
+ },
298
+ startedAt: "2026-08-20T13:00:00.000Z"
299
+ }, {
300
+ statusCode: 200,
301
+ durationMs: 5,
302
+ responseBody: Buffer.from(JSON.stringify({
303
+ status: "completed",
304
+ usage: { input_tokens: 31_533, output_tokens: 32_768 }
305
+ }))
306
+ });
307
+ assert.equal(record.status, "failed");
308
+ assert.equal(record.metadata?.stop_reason, "max_output_tokens");
309
+ });
310
+ test("HTTP 200 Responses stream without a finish reason is failed provenance", () => {
311
+ const record = buildModelCallRecord({
312
+ callId: "call_response_stream_truncated",
313
+ dialect: "openai-responses",
314
+ requestedModel: "openai/model",
315
+ model: "openai/model",
316
+ stream: true,
317
+ requestBody: { input: "answer", stream: true },
318
+ startedAt: "2026-08-20T13:00:00.000Z"
319
+ }, {
320
+ statusCode: 200,
321
+ durationMs: 5,
322
+ responseBody: Buffer.from('event: response.incomplete\ndata: {"type":"response.incomplete","response":{"status":"incomplete"}}\n\n')
323
+ });
324
+ assert.equal(record.status, "failed");
325
+ });
@@ -35,6 +35,177 @@ test("Responses JSON schemas reach Anthropic as native structured output formats
35
35
  }
36
36
  });
37
37
  });
38
+ test("Anthropic structured outputs defer unsupported JSON Schema constraints", () => {
39
+ const schema = {
40
+ type: "object",
41
+ additionalProperties: false,
42
+ required: ["minimum", "scores", "labels"],
43
+ properties: {
44
+ minimum: { type: "integer", minimum: 1, maximum: 16_384 },
45
+ scores: {
46
+ type: "array",
47
+ minItems: 20,
48
+ maxItems: 20,
49
+ items: { type: "number", minimum: 0, maximum: 1 }
50
+ },
51
+ labels: {
52
+ type: "array",
53
+ minItems: 1,
54
+ items: { type: "string", minLength: 1, maxLength: 128 }
55
+ }
56
+ }
57
+ };
58
+ const original = structuredClone(schema);
59
+ const outbound = anthropicMessages({
60
+ model: "claude-opus-5",
61
+ messages: [{ role: "user", content: "author evaluations" }],
62
+ response_format: {
63
+ type: "json_schema",
64
+ json_schema: { name: "routekit_evaluations", schema, strict: true }
65
+ }
66
+ }, "claude-opus-5");
67
+ assert.deepEqual(schema, original);
68
+ assert.deepEqual(outbound.output_config, {
69
+ format: {
70
+ type: "json_schema",
71
+ schema: {
72
+ type: "object",
73
+ additionalProperties: false,
74
+ required: ["minimum", "scores", "labels"],
75
+ properties: {
76
+ minimum: {
77
+ type: "integer",
78
+ description: "{minimum: 1, maximum: 16384}"
79
+ },
80
+ scores: {
81
+ type: "array",
82
+ items: {
83
+ type: "number",
84
+ description: "{minimum: 0, maximum: 1}"
85
+ },
86
+ description: "{minItems: 20, maxItems: 20}"
87
+ },
88
+ labels: {
89
+ type: "array",
90
+ minItems: 1,
91
+ items: {
92
+ type: "string",
93
+ description: "{minLength: 1, maxLength: 128}"
94
+ }
95
+ }
96
+ }
97
+ }
98
+ }
99
+ });
100
+ const nativeOutbound = anthropicMessages(anthropicToChat({
101
+ model: "claude-opus-5",
102
+ messages: [{ role: "user", content: "author evaluations" }],
103
+ output_config: {
104
+ format: { type: "json_schema", schema }
105
+ }
106
+ }, "claude-opus-5"), "claude-opus-5");
107
+ assert.deepEqual(nativeOutbound.output_config, outbound.output_config);
108
+ });
109
+ test("Anthropic tool input schemas defer unsupported JSON Schema constraints", () => {
110
+ const parameters = {
111
+ type: "object",
112
+ additionalProperties: false,
113
+ required: ["limit", "scores"],
114
+ properties: {
115
+ limit: { type: "integer", minimum: 1, maximum: 16_384 },
116
+ scores: {
117
+ type: "array",
118
+ minItems: 20,
119
+ maxItems: 20,
120
+ items: { type: "number", minimum: 0, maximum: 1 }
121
+ }
122
+ }
123
+ };
124
+ const original = structuredClone(parameters);
125
+ const outbound = anthropicMessages({
126
+ model: "claude-opus-5",
127
+ messages: [{ role: "user", content: "score the inputs" }],
128
+ tools: [
129
+ {
130
+ type: "function",
131
+ function: {
132
+ name: "score_inputs",
133
+ description: "score structured inputs",
134
+ parameters
135
+ }
136
+ }
137
+ ]
138
+ }, "claude-opus-5");
139
+ assert.deepEqual(parameters, original);
140
+ assert.deepEqual(outbound.tools, [
141
+ {
142
+ name: "score_inputs",
143
+ description: "score structured inputs",
144
+ input_schema: {
145
+ type: "object",
146
+ additionalProperties: false,
147
+ required: ["limit", "scores"],
148
+ properties: {
149
+ limit: {
150
+ type: "integer",
151
+ description: "{minimum: 1, maximum: 16384}"
152
+ },
153
+ scores: {
154
+ type: "array",
155
+ items: {
156
+ type: "number",
157
+ description: "{minimum: 0, maximum: 1}"
158
+ },
159
+ description: "{minItems: 20, maxItems: 20}"
160
+ }
161
+ }
162
+ }
163
+ }
164
+ ]);
165
+ });
166
+ test("Anthropic structured outputs keep only supported schema keywords and formats", () => {
167
+ const outbound = anthropicMessages({
168
+ model: "claude-opus-5",
169
+ messages: [{ role: "user", content: "return a structured value" }],
170
+ response_format: {
171
+ type: "json_schema",
172
+ json_schema: {
173
+ name: "routekit_supported_schema",
174
+ strict: true,
175
+ schema: {
176
+ type: "object",
177
+ minProperties: 1,
178
+ properties: {
179
+ identifier: {
180
+ type: "string",
181
+ pattern: "^[a-z]+$",
182
+ format: "regex"
183
+ },
184
+ createdAt: {
185
+ type: "string",
186
+ format: "date-time"
187
+ },
188
+ choice: {
189
+ oneOf: [{ const: "one" }, { const: "two" }]
190
+ }
191
+ },
192
+ required: ["identifier", "createdAt", "choice"],
193
+ additionalProperties: false
194
+ }
195
+ }
196
+ }
197
+ }, "claude-opus-5");
198
+ const schema = outbound.output_config.format.schema;
199
+ const properties = schema.properties;
200
+ assert.equal(schema.minProperties, undefined);
201
+ assert.match(schema.description, /minProperties/u);
202
+ assert.equal(properties.identifier.format, undefined);
203
+ assert.equal(properties.identifier.pattern, "^[a-z]+$");
204
+ assert.match(properties.identifier.description, /format/u);
205
+ assert.equal(properties.createdAt.format, "date-time");
206
+ assert.equal(properties.choice.oneOf, undefined);
207
+ assert.deepEqual(properties.choice.anyOf, [{ const: "one" }, { const: "two" }]);
208
+ });
38
209
  test("direct provider backends reject malformed reasoning controls before transport", async () => {
39
210
  const cases = [
40
211
  {
@@ -4,7 +4,7 @@ import { runRouteKitEffect } from "@velum-labs/routekit-runtime/effect";
4
4
  import { Effect } from "effect";
5
5
  import { reasoningSelectionOf } from "../adapters/openai-chat-wire.js";
6
6
  import { parseResponsesEncryptedContent, wrapResponsesEncryptedContent } from "../adapters/openai-responses-wire.js";
7
- import { openAiSseToResponses } from "../adapters/responses.js";
7
+ import { chatToResponses, openAiSseToResponses } from "../adapters/responses.js";
8
8
  import { RoutingBackend } from "../routing/router.js";
9
9
  import { startGateway } from "../gateway-service.js";
10
10
  import { testProviderSource } from "./provider-source-fixture.js";
@@ -26,6 +26,37 @@ test("a mid-stream provider error event becomes response.failed with the upstrea
26
26
  assert.ok(text.includes("openrouter call failed (unknown)"));
27
27
  assert.ok(!text.includes("event: response.completed"));
28
28
  });
29
+ test("max-token Chat completion becomes an incomplete Responses result", () => {
30
+ for (const finishReason of ["length", "max_tokens"]) {
31
+ const response = chatToResponses({
32
+ choices: [
33
+ {
34
+ message: { role: "assistant", content: '{"cases":[{"id":"truncated' },
35
+ finish_reason: finishReason
36
+ }
37
+ ]
38
+ }, "claude-code/claude-opus-5");
39
+ assert.equal(response.status, "incomplete");
40
+ assert.deepEqual(response.incomplete_details, { reason: "max_output_tokens" });
41
+ }
42
+ });
43
+ test("max-token Chat stream becomes response.incomplete with its reason", async () => {
44
+ const stream = openAiSseToResponses(sseStream(`data: ${JSON.stringify({
45
+ choices: [
46
+ {
47
+ index: 0,
48
+ delta: { content: '{"cases":[{"id":"truncated' },
49
+ finish_reason: null
50
+ }
51
+ ]
52
+ })}\n\n`, `data: ${JSON.stringify({
53
+ choices: [{ index: 0, delta: {}, finish_reason: "length" }]
54
+ })}\n\n`, "data: [DONE]\n\n"), "claude-code/claude-opus-5");
55
+ const text = await new Response(stream).text();
56
+ assert.match(text, /event: response\.incomplete/u);
57
+ assert.match(text, /"incomplete_details":\{"reason":"max_output_tokens"\}/u);
58
+ assert.doesNotMatch(text, /event: response\.completed/u);
59
+ });
29
60
  test("translates a streamed Responses event sequence", async () => {
30
61
  const mock = await startMock();
31
62
  const gateway = await startGateway({
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@velum-labs/routekit-gateway",
3
3
  "private": false,
4
- "version": "1.0.7",
4
+ "version": "1.0.9",
5
5
  "repository": {
6
6
  "type": "git",
7
7
  "url": "git+https://github.com/velum-labs/routekit.git",
@@ -45,12 +45,12 @@
45
45
  "@aws-sdk/client-bedrock": "3.1095.0",
46
46
  "@aws-sdk/client-bedrock-runtime": "3.1095.0",
47
47
  "effect": "4.0.0-rc.108",
48
- "@velum-labs/routekit-config-core": "1.0.7",
49
- "@velum-labs/routekit-contracts": "1.0.7",
50
- "@velum-labs/routekit-eval-contracts": "1.0.7",
51
- "@velum-labs/routekit-eval-core": "1.0.7",
52
- "@velum-labs/routekit-registry": "1.0.7",
53
- "@velum-labs/routekit-runtime": "1.0.7"
48
+ "@velum-labs/routekit-config-core": "1.0.9",
49
+ "@velum-labs/routekit-contracts": "1.0.9",
50
+ "@velum-labs/routekit-eval-contracts": "1.0.9",
51
+ "@velum-labs/routekit-eval-core": "1.0.9",
52
+ "@velum-labs/routekit-registry": "1.0.9",
53
+ "@velum-labs/routekit-runtime": "1.0.9"
54
54
  },
55
55
  "keywords": [
56
56
  "routekit",