@f5-sales-demo/pi-ai 19.90.0 → 19.91.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +4 -4
- package/src/model-thinking.ts +63 -4
- package/src/models.json +2 -2
- package/src/providers/amazon-bedrock.ts +1 -0
- package/src/providers/anthropic.ts +108 -11
- package/src/providers/gitlab-duo.ts +4 -2
- package/src/providers/kimi.ts +2 -1
- package/src/providers/openai-codex/request-transformer.ts +6 -2
- package/src/providers/synthetic.ts +2 -1
- package/src/stream.ts +11 -2
- package/src/types.ts +49 -0
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@f5-sales-demo/pi-ai",
|
|
4
|
-
"version": "19.
|
|
4
|
+
"version": "19.91.0",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://github.com/f5-sales-demo/xcsh",
|
|
7
7
|
"author": "Can Boluk",
|
|
@@ -41,11 +41,11 @@
|
|
|
41
41
|
"generate-models": "bun scripts/generate-models.ts"
|
|
42
42
|
},
|
|
43
43
|
"dependencies": {
|
|
44
|
-
"@anthropic-ai/sdk": "^0.
|
|
44
|
+
"@anthropic-ai/sdk": "^0.115",
|
|
45
45
|
"@aws-sdk/client-bedrock-runtime": "^3",
|
|
46
46
|
"@bufbuild/protobuf": "^2.11",
|
|
47
|
-
"@google/genai": "^
|
|
48
|
-
"@f5-sales-demo/pi-utils": "19.
|
|
47
|
+
"@google/genai": "^2.13",
|
|
48
|
+
"@f5-sales-demo/pi-utils": "19.91.0",
|
|
49
49
|
"@sinclair/typebox": "^0.34",
|
|
50
50
|
"@smithy/node-http-handler": "^4.4",
|
|
51
51
|
"ajv": "^8.20",
|
package/src/model-thinking.ts
CHANGED
|
@@ -1,21 +1,51 @@
|
|
|
1
1
|
import { resolveOpenAICompat } from "./providers/openai-completions-compat";
|
|
2
2
|
import type { Api, Model as ApiModel, ThinkingConfig } from "./types";
|
|
3
3
|
|
|
4
|
-
/**
|
|
4
|
+
/**
|
|
5
|
+
* User-facing thinking levels, ordered least to most intensive.
|
|
6
|
+
*
|
|
7
|
+
* `Minimal` is xcsh-only — no provider accepts it on the wire (Anthropic rejects
|
|
8
|
+
* it with `output_config.effort: Input should be 'low', 'medium', 'high', 'xhigh'
|
|
9
|
+
* or 'max'`), so every mapper must translate it down.
|
|
10
|
+
*/
|
|
5
11
|
export const enum Effort {
|
|
6
12
|
Minimal = "minimal",
|
|
7
13
|
Low = "low",
|
|
8
14
|
Medium = "medium",
|
|
9
15
|
High = "high",
|
|
10
16
|
XHigh = "xhigh",
|
|
17
|
+
Max = "max",
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Anthropic's `output_config.effort` enum, verbatim. Authority is the API's own
|
|
22
|
+
* validation error: `Input should be 'low', 'medium', 'high', 'xhigh' or 'max'`.
|
|
23
|
+
* Deliberately excludes xcsh's `minimal`, which the API rejects.
|
|
24
|
+
*/
|
|
25
|
+
export type AnthropicAdaptiveEffort = "low" | "medium" | "high" | "xhigh" | "max";
|
|
26
|
+
|
|
27
|
+
/** Effort levels for providers whose wire enum stops at `xhigh` (no `max`). */
|
|
28
|
+
export type EffortThroughXHigh = "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Clamp an effort to a provider whose wire enum has no `max` (OpenAI-compat
|
|
32
|
+
* reasoning_effort, GitLab Duo, Kimi, Synthetic, …). `requireSupportedEffort`
|
|
33
|
+
* already keeps `Max` out of those providers at runtime — because their
|
|
34
|
+
* supported-effort lists exclude it — but the wire types can't prove that, and
|
|
35
|
+
* clamping is the safe direction if a caller bypasses the range check.
|
|
36
|
+
*/
|
|
37
|
+
export function clampEffortThroughXHigh(effort: Effort): EffortThroughXHigh {
|
|
38
|
+
return effort === Effort.Max ? Effort.XHigh : effort;
|
|
11
39
|
}
|
|
12
40
|
|
|
41
|
+
/** Order is load-bearing: `indexOf` drives expandEffortRange/requireSupportedEffort. */
|
|
13
42
|
export const THINKING_EFFORTS: readonly Effort[] = [
|
|
14
43
|
Effort.Minimal,
|
|
15
44
|
Effort.Low,
|
|
16
45
|
Effort.Medium,
|
|
17
46
|
Effort.High,
|
|
18
47
|
Effort.XHigh,
|
|
48
|
+
Effort.Max,
|
|
19
49
|
];
|
|
20
50
|
|
|
21
51
|
const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
|
|
@@ -26,6 +56,18 @@ const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [
|
|
|
26
56
|
Effort.High,
|
|
27
57
|
Effort.XHigh,
|
|
28
58
|
];
|
|
59
|
+
/**
|
|
60
|
+
* Anthropic 4.6+/5-era range. These models accept the full API enum, including
|
|
61
|
+
* `xhigh` (Anthropic's recommended setting for coding/agentic work) and `max`.
|
|
62
|
+
*/
|
|
63
|
+
const ANTHROPIC_ADAPTIVE_EFFORTS: readonly Effort[] = [
|
|
64
|
+
Effort.Minimal,
|
|
65
|
+
Effort.Low,
|
|
66
|
+
Effort.Medium,
|
|
67
|
+
Effort.High,
|
|
68
|
+
Effort.XHigh,
|
|
69
|
+
Effort.Max,
|
|
70
|
+
];
|
|
29
71
|
const GEMINI_3_PRO_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High];
|
|
30
72
|
const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
|
|
31
73
|
const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
|
|
@@ -250,15 +292,22 @@ export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
|
|
|
250
292
|
return "MEDIUM";
|
|
251
293
|
case Effort.High:
|
|
252
294
|
case Effort.XHigh:
|
|
295
|
+
case Effort.Max:
|
|
253
296
|
return "HIGH";
|
|
254
297
|
}
|
|
255
298
|
}
|
|
256
299
|
|
|
257
|
-
/**
|
|
300
|
+
/**
|
|
301
|
+
* Maps a normalized thinking effort to Anthropic adaptive effort values.
|
|
302
|
+
*
|
|
303
|
+
* The API enum is `low | medium | high | xhigh | max` (per its own validation
|
|
304
|
+
* error). `Minimal` has no wire equivalent and is translated down to `low` —
|
|
305
|
+
* sending `"minimal"` is a 400.
|
|
306
|
+
*/
|
|
258
307
|
export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
|
|
259
308
|
model: ApiModel<TApi>,
|
|
260
309
|
effort: Effort,
|
|
261
|
-
):
|
|
310
|
+
): AnthropicAdaptiveEffort {
|
|
262
311
|
switch (requireSupportedEffort(model, effort)) {
|
|
263
312
|
case Effort.Minimal:
|
|
264
313
|
case Effort.Low:
|
|
@@ -268,6 +317,8 @@ export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
|
|
|
268
317
|
case Effort.High:
|
|
269
318
|
return "high";
|
|
270
319
|
case Effort.XHigh:
|
|
320
|
+
return "xhigh";
|
|
321
|
+
case Effort.Max:
|
|
271
322
|
return "max";
|
|
272
323
|
}
|
|
273
324
|
}
|
|
@@ -398,7 +449,15 @@ function inferAnthropicSupportedEfforts<TApi extends Api>(
|
|
|
398
449
|
(model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") &&
|
|
399
450
|
semverGte(parsedModel.version, "4.6")
|
|
400
451
|
) {
|
|
401
|
-
|
|
452
|
+
// Opus 4.6+ and Sonnet 5+ accept the extended range. Sonnet 4.6 tops out at
|
|
453
|
+
// `high` — the reason this can't key on version alone (#2341).
|
|
454
|
+
const extended = parsedModel.kind === "opus" || semverGte(parsedModel.version, "5.0");
|
|
455
|
+
if (!extended) return DEFAULT_REASONING_EFFORTS;
|
|
456
|
+
// `max` is only claimed for the first-party Messages API, where the enum was
|
|
457
|
+
// verified against the live gateway. Bedrock's effort support is not verified
|
|
458
|
+
// here, so it keeps the pre-existing `xhigh` ceiling rather than being widened
|
|
459
|
+
// on an assumption.
|
|
460
|
+
return model.api === "anthropic-messages" ? ANTHROPIC_ADAPTIVE_EFFORTS : DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
|
402
461
|
}
|
|
403
462
|
return inferFallbackEfforts(model);
|
|
404
463
|
}
|
package/src/models.json
CHANGED
|
@@ -2647,7 +2647,7 @@
|
|
|
2647
2647
|
"thinking": {
|
|
2648
2648
|
"mode": "anthropic-adaptive",
|
|
2649
2649
|
"minLevel": "minimal",
|
|
2650
|
-
"maxLevel": "
|
|
2650
|
+
"maxLevel": "max"
|
|
2651
2651
|
}
|
|
2652
2652
|
},
|
|
2653
2653
|
"claude-sonnet-4-0": {
|
|
@@ -2800,7 +2800,7 @@
|
|
|
2800
2800
|
"thinking": {
|
|
2801
2801
|
"mode": "anthropic-adaptive",
|
|
2802
2802
|
"minLevel": "minimal",
|
|
2803
|
-
"maxLevel": "
|
|
2803
|
+
"maxLevel": "max"
|
|
2804
2804
|
}
|
|
2805
2805
|
}
|
|
2806
2806
|
},
|
|
@@ -3,12 +3,13 @@ import * as fs from "node:fs";
|
|
|
3
3
|
import * as tls from "node:tls";
|
|
4
4
|
import Anthropic, { type ClientOptions as AnthropicSdkClientOptions } from "@anthropic-ai/sdk";
|
|
5
5
|
import type {
|
|
6
|
+
CitationsDelta,
|
|
6
7
|
ContentBlockParam,
|
|
7
8
|
MessageCreateParamsStreaming,
|
|
8
9
|
MessageParam,
|
|
9
10
|
} from "@anthropic-ai/sdk/resources/messages";
|
|
10
11
|
import { $env, abortableSleep, isEnoent } from "@f5-sales-demo/pi-utils";
|
|
11
|
-
import { mapEffortToAnthropicAdaptiveEffort } from "../model-thinking";
|
|
12
|
+
import { type AnthropicAdaptiveEffort, mapEffortToAnthropicAdaptiveEffort } from "../model-thinking";
|
|
12
13
|
import { calculateCost } from "../models";
|
|
13
14
|
import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
|
|
14
15
|
import type {
|
|
@@ -30,6 +31,7 @@ import type {
|
|
|
30
31
|
ToolCall,
|
|
31
32
|
ToolResultMessage,
|
|
32
33
|
Usage,
|
|
34
|
+
WebCitation,
|
|
33
35
|
} from "../types";
|
|
34
36
|
import { isAnthropicOAuthToken, normalizeToolCallId, resolveCacheRetention } from "../utils";
|
|
35
37
|
import { createAbortSourceTracker } from "../utils/abort";
|
|
@@ -355,7 +357,11 @@ function convertContentBlocks(content: (TextContent | ImageContent)[]):
|
|
|
355
357
|
return blocks;
|
|
356
358
|
}
|
|
357
359
|
|
|
358
|
-
|
|
360
|
+
/**
|
|
361
|
+
* Anthropic's `output_config.effort` enum. Single source of truth lives in
|
|
362
|
+
* model-thinking so the mapper and this raw-passthrough path cannot drift.
|
|
363
|
+
*/
|
|
364
|
+
export type AnthropicEffort = AnthropicAdaptiveEffort;
|
|
359
365
|
|
|
360
366
|
export interface AnthropicOptions extends StreamOptions {
|
|
361
367
|
/**
|
|
@@ -625,6 +631,24 @@ export function isProviderRetryableError(error: unknown): boolean {
|
|
|
625
631
|
);
|
|
626
632
|
}
|
|
627
633
|
|
|
634
|
+
/**
|
|
635
|
+
* Map an Anthropic `citations_delta` citation to a `WebCitation`.
|
|
636
|
+
*
|
|
637
|
+
* Only web-search citations are carried: the char/page/content-block locations belong to the
|
|
638
|
+
* document-citations API, which cites user-supplied documents rather than search results and has
|
|
639
|
+
* no source URL to render as a chip.
|
|
640
|
+
*/
|
|
641
|
+
function mapAnthropicWebCitation(citation: CitationsDelta["citation"]): WebCitation | undefined {
|
|
642
|
+
if (citation.type !== "web_search_result_location") return undefined;
|
|
643
|
+
return {
|
|
644
|
+
type: "web_search_result_location",
|
|
645
|
+
url: citation.url,
|
|
646
|
+
...(citation.title ? { title: citation.title } : {}),
|
|
647
|
+
...(citation.cited_text ? { citedText: citation.cited_text } : {}),
|
|
648
|
+
...(citation.encrypted_index ? { encryptedIndex: citation.encrypted_index } : {}),
|
|
649
|
+
};
|
|
650
|
+
}
|
|
651
|
+
|
|
628
652
|
function createEmptyUsage(premiumRequests?: number): Usage {
|
|
629
653
|
return {
|
|
630
654
|
input: 0,
|
|
@@ -733,6 +757,19 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
733
757
|
const anthropicRequest = client.messages.create({ ...params, stream: true }, { signal: requestSignal });
|
|
734
758
|
let streamedReplayUnsafeContent = false;
|
|
735
759
|
|
|
760
|
+
// SERVER-SIDE tools (Anthropic's built-in web search) are executed by the provider,
|
|
761
|
+
// so their blocks are deliberately NOT pushed into `output.content`: there is nothing
|
|
762
|
+
// for the agent loop to dispatch, and keeping their `encrypted_content` out of history
|
|
763
|
+
// avoids the byte-exact echo requirement that would 400 a follow-up turn. They are
|
|
764
|
+
// tracked here only long enough to report progress — keyed by the stream's block index
|
|
765
|
+
// (to accumulate the query) and by tool id (to name the matching result block).
|
|
766
|
+
// Declared inside the retry loop so a replayed attempt starts clean.
|
|
767
|
+
const serverToolUses = new Map<
|
|
768
|
+
number,
|
|
769
|
+
{ id: string; name: string; partialJson: string; inlineInput: unknown }
|
|
770
|
+
>();
|
|
771
|
+
const serverToolNames = new Map<string, string>();
|
|
772
|
+
|
|
736
773
|
try {
|
|
737
774
|
const { data: anthropicStream } = await anthropicRequest.withResponse();
|
|
738
775
|
const firstEventWatchdog = createFirstEventWatchdog(firstEventTimeoutMs, () =>
|
|
@@ -821,6 +858,30 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
821
858
|
contentIndex: output.content.length - 1,
|
|
822
859
|
partial: output,
|
|
823
860
|
});
|
|
861
|
+
} else if (event.content_block.type === "server_tool_use") {
|
|
862
|
+
// Progress-only: remember the call so its query can be reported once the
|
|
863
|
+
// input JSON has finished streaming (see content_block_stop).
|
|
864
|
+
serverToolUses.set(event.index, {
|
|
865
|
+
id: event.content_block.id,
|
|
866
|
+
name: event.content_block.name,
|
|
867
|
+
partialJson: "",
|
|
868
|
+
// Nothing obliges the provider to stream the input as deltas — it may
|
|
869
|
+
// arrive whole right here. Kept as the fallback for the query.
|
|
870
|
+
inlineInput: event.content_block.input,
|
|
871
|
+
});
|
|
872
|
+
serverToolNames.set(event.content_block.id, event.content_block.name);
|
|
873
|
+
} else if (event.content_block.type === "web_search_tool_result") {
|
|
874
|
+
const toolId = event.content_block.tool_use_id;
|
|
875
|
+
const content = event.content_block.content;
|
|
876
|
+
stream.push({
|
|
877
|
+
type: "server_tool_end",
|
|
878
|
+
toolName: serverToolNames.get(toolId) ?? "web_search",
|
|
879
|
+
toolId,
|
|
880
|
+
...(Array.isArray(content)
|
|
881
|
+
? { resultCount: content.length }
|
|
882
|
+
: { errorCode: content.error_code }),
|
|
883
|
+
partial: output,
|
|
884
|
+
});
|
|
824
885
|
}
|
|
825
886
|
} else if (event.type === "content_block_delta") {
|
|
826
887
|
if (event.delta.type === "text_delta") {
|
|
@@ -848,17 +909,31 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
848
909
|
});
|
|
849
910
|
}
|
|
850
911
|
} else if (event.delta.type === "input_json_delta") {
|
|
912
|
+
const serverToolUse = serverToolUses.get(event.index);
|
|
913
|
+
if (serverToolUse) {
|
|
914
|
+
// A server-side tool's input (the search query). Accumulated but not
|
|
915
|
+
// streamed as a toolcall_delta — there is no toolCall block to attach to.
|
|
916
|
+
serverToolUse.partialJson += event.delta.partial_json;
|
|
917
|
+
} else {
|
|
918
|
+
const index = blocks.findIndex(b => b.index === event.index);
|
|
919
|
+
const block = blocks[index];
|
|
920
|
+
if (block && block.type === "toolCall") {
|
|
921
|
+
block.partialJson += event.delta.partial_json;
|
|
922
|
+
block.arguments = parseStreamingJson(block.partialJson);
|
|
923
|
+
stream.push({
|
|
924
|
+
type: "toolcall_delta",
|
|
925
|
+
contentIndex: index,
|
|
926
|
+
delta: event.delta.partial_json,
|
|
927
|
+
partial: output,
|
|
928
|
+
});
|
|
929
|
+
}
|
|
930
|
+
}
|
|
931
|
+
} else if (event.delta.type === "citations_delta") {
|
|
851
932
|
const index = blocks.findIndex(b => b.index === event.index);
|
|
852
933
|
const block = blocks[index];
|
|
853
|
-
|
|
854
|
-
|
|
855
|
-
block.
|
|
856
|
-
stream.push({
|
|
857
|
-
type: "toolcall_delta",
|
|
858
|
-
contentIndex: index,
|
|
859
|
-
delta: event.delta.partial_json,
|
|
860
|
-
partial: output,
|
|
861
|
-
});
|
|
934
|
+
const citation = mapAnthropicWebCitation(event.delta.citation);
|
|
935
|
+
if (block && block.type === "text" && citation) {
|
|
936
|
+
block.citations = [...(block.citations ?? []), citation];
|
|
862
937
|
}
|
|
863
938
|
} else if (event.delta.type === "signature_delta") {
|
|
864
939
|
const index = blocks.findIndex(b => b.index === event.index);
|
|
@@ -869,6 +944,28 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
869
944
|
}
|
|
870
945
|
}
|
|
871
946
|
} else if (event.type === "content_block_stop") {
|
|
947
|
+
const serverToolUse = serverToolUses.get(event.index);
|
|
948
|
+
if (serverToolUse) {
|
|
949
|
+
// The query has finished streaming, so the activity row can name what is
|
|
950
|
+
// being searched for. Reported here rather than at content_block_start
|
|
951
|
+
// because that is the first moment the query is known — and it still lands
|
|
952
|
+
// well before the provider's search completes.
|
|
953
|
+
serverToolUses.delete(event.index);
|
|
954
|
+
const streamed = parseStreamingJson(serverToolUse.partialJson) as
|
|
955
|
+
| { query?: unknown }
|
|
956
|
+
| undefined;
|
|
957
|
+
const inline = serverToolUse.inlineInput as { query?: unknown } | undefined;
|
|
958
|
+
const rawQuery = typeof streamed?.query === "string" ? streamed.query : inline?.query;
|
|
959
|
+
const query = typeof rawQuery === "string" && rawQuery.length > 0 ? rawQuery : undefined;
|
|
960
|
+
stream.push({
|
|
961
|
+
type: "server_tool_start",
|
|
962
|
+
toolName: serverToolUse.name,
|
|
963
|
+
toolId: serverToolUse.id,
|
|
964
|
+
...(query === undefined ? {} : { query }),
|
|
965
|
+
partial: output,
|
|
966
|
+
});
|
|
967
|
+
continue;
|
|
968
|
+
}
|
|
872
969
|
const index = blocks.findIndex(b => b.index === event.index);
|
|
873
970
|
const block = blocks[index];
|
|
874
971
|
if (block) {
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { clampEffortThroughXHigh } from "../model-thinking";
|
|
1
2
|
import { ANTHROPIC_THINKING, mapAnthropicToolChoice } from "../stream";
|
|
2
3
|
import type { Api, Context, Model, SimpleStreamOptions } from "../types";
|
|
3
4
|
import { AssistantMessageEventStream } from "../utils/event-stream";
|
|
@@ -301,6 +302,7 @@ export function streamGitLabDuo(
|
|
|
301
302
|
thinkingBudgetTokens: reasoningEffort
|
|
302
303
|
? (options.thinkingBudgets?.[reasoningEffort] ?? ANTHROPIC_THINKING[reasoningEffort])
|
|
303
304
|
: undefined,
|
|
305
|
+
// Anthropic path — takes the full ladder, no clamp.
|
|
304
306
|
reasoning: reasoningEffort,
|
|
305
307
|
toolChoice: mapAnthropicToolChoice(options.toolChoice),
|
|
306
308
|
},
|
|
@@ -331,7 +333,7 @@ export function streamGitLabDuo(
|
|
|
331
333
|
sessionId: options.sessionId,
|
|
332
334
|
providerSessionState: options.providerSessionState,
|
|
333
335
|
onPayload: options.onPayload,
|
|
334
|
-
reasoning: reasoningEffort,
|
|
336
|
+
reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
|
|
335
337
|
toolChoice: options.toolChoice,
|
|
336
338
|
} satisfies OpenAIResponsesOptions,
|
|
337
339
|
)
|
|
@@ -360,7 +362,7 @@ export function streamGitLabDuo(
|
|
|
360
362
|
sessionId: options.sessionId,
|
|
361
363
|
providerSessionState: options.providerSessionState,
|
|
362
364
|
onPayload: options.onPayload,
|
|
363
|
-
reasoning: reasoningEffort,
|
|
365
|
+
reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
|
|
364
366
|
toolChoice: options.toolChoice,
|
|
365
367
|
} satisfies OpenAICompletionsOptions,
|
|
366
368
|
);
|
package/src/providers/kimi.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { clampEffortThroughXHigh } from "../model-thinking";
|
|
1
2
|
/**
|
|
2
3
|
* Kimi Code provider - wraps OpenAI or Anthropic API based on format setting.
|
|
3
4
|
*
|
|
@@ -104,7 +105,7 @@ export function streamKimi(
|
|
|
104
105
|
headers: mergedHeaders,
|
|
105
106
|
sessionId: options?.sessionId,
|
|
106
107
|
onPayload: options?.onPayload,
|
|
107
|
-
reasoning: reasoningEffort,
|
|
108
|
+
reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
|
|
108
109
|
});
|
|
109
110
|
|
|
110
111
|
for await (const event of innerStream) {
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { Effort } from "../../model-thinking";
|
|
2
|
-
import { requireSupportedEffort } from "../../model-thinking";
|
|
2
|
+
import { clampEffortThroughXHigh, requireSupportedEffort } from "../../model-thinking";
|
|
3
3
|
import type { Api, Model } from "../../types";
|
|
4
4
|
|
|
5
5
|
export interface ReasoningConfig {
|
|
@@ -53,8 +53,12 @@ export interface RequestBody {
|
|
|
53
53
|
|
|
54
54
|
function getReasoningConfig(model: Model<Api>, options: CodexRequestOptions): ReasoningConfig {
|
|
55
55
|
return {
|
|
56
|
+
// Codex's reasoning effort has no `max`; clamp so the ladder's top level
|
|
57
|
+
// degrades to `xhigh` instead of emitting a value Codex would reject.
|
|
56
58
|
effort:
|
|
57
|
-
options.reasoningEffort === "none"
|
|
59
|
+
options.reasoningEffort === "none"
|
|
60
|
+
? "none"
|
|
61
|
+
: clampEffortThroughXHigh(requireSupportedEffort(model, options.reasoningEffort as Effort)),
|
|
58
62
|
summary: options.reasoningSummary ?? "detailed",
|
|
59
63
|
};
|
|
60
64
|
}
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { clampEffortThroughXHigh } from "../model-thinking";
|
|
1
2
|
/**
|
|
2
3
|
* Synthetic provider - wraps OpenAI or Anthropic API based on format setting.
|
|
3
4
|
*
|
|
@@ -107,7 +108,7 @@ export function streamSynthetic(
|
|
|
107
108
|
headers: mergedHeaders,
|
|
108
109
|
sessionId: options?.sessionId,
|
|
109
110
|
onPayload: options?.onPayload,
|
|
110
|
-
reasoning: reasoningEffort,
|
|
111
|
+
reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
|
|
111
112
|
});
|
|
112
113
|
|
|
113
114
|
for await (const event of innerStream) {
|
package/src/stream.ts
CHANGED
|
@@ -5,6 +5,8 @@ import { $env, $pickenv } from "@f5-sales-demo/pi-utils";
|
|
|
5
5
|
import { getCustomApi } from "./api-registry";
|
|
6
6
|
import type { Effort } from "./model-thinking";
|
|
7
7
|
import {
|
|
8
|
+
clampEffortThroughXHigh,
|
|
9
|
+
type EffortThroughXHigh,
|
|
8
10
|
mapEffortToAnthropicAdaptiveEffort,
|
|
9
11
|
mapEffortToGoogleThinkingLevel,
|
|
10
12
|
requireSupportedEffort,
|
|
@@ -317,6 +319,7 @@ export const ANTHROPIC_THINKING: Record<Effort, number> = {
|
|
|
317
319
|
medium: 8192,
|
|
318
320
|
high: 16384,
|
|
319
321
|
xhigh: 32768,
|
|
322
|
+
max: 65536,
|
|
320
323
|
};
|
|
321
324
|
|
|
322
325
|
const GOOGLE_THINKING: Record<Effort, number> = {
|
|
@@ -325,6 +328,8 @@ const GOOGLE_THINKING: Record<Effort, number> = {
|
|
|
325
328
|
medium: 8192,
|
|
326
329
|
high: 16384,
|
|
327
330
|
xhigh: 24575,
|
|
331
|
+
// Google caps the thinking budget here; `max` cannot exceed it.
|
|
332
|
+
max: 24575,
|
|
328
333
|
};
|
|
329
334
|
|
|
330
335
|
const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
|
|
@@ -333,6 +338,8 @@ const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
|
|
|
333
338
|
medium: 8192,
|
|
334
339
|
high: 16384,
|
|
335
340
|
xhigh: 16384,
|
|
341
|
+
// Bedrock Claude caps the thinking budget here; `max` cannot exceed it.
|
|
342
|
+
max: 16384,
|
|
336
343
|
};
|
|
337
344
|
|
|
338
345
|
function resolveBedrockThinkingBudget(
|
|
@@ -394,10 +401,12 @@ function mapOpenAiToolChoice(choice?: ToolChoice): OpenAICompletionsOptions["too
|
|
|
394
401
|
function resolveOpenAiReasoningEffort<TApi extends Api>(
|
|
395
402
|
model: Model<TApi>,
|
|
396
403
|
options?: SimpleStreamOptions,
|
|
397
|
-
):
|
|
404
|
+
): EffortThroughXHigh | undefined {
|
|
398
405
|
const reasoning = options?.reasoning;
|
|
399
406
|
if (!reasoning || !model.reasoning) return undefined;
|
|
400
|
-
|
|
407
|
+
// OpenAI-compat `reasoning_effort` has no `max`; clamp rather than emit a
|
|
408
|
+
// value the provider would reject.
|
|
409
|
+
return clampEffortThroughXHigh(requireSupportedEffort(model, reasoning));
|
|
401
410
|
}
|
|
402
411
|
|
|
403
412
|
const castApi = <TApi extends Api>(api: OptionsForApi<TApi>): OptionsForApi<Api> => api as OptionsForApi<Api>;
|
package/src/types.ts
CHANGED
|
@@ -252,10 +252,33 @@ export interface TextSignatureV1 {
|
|
|
252
252
|
phase?: "commentary" | "final_answer";
|
|
253
253
|
}
|
|
254
254
|
|
|
255
|
+
/**
|
|
256
|
+
* A structured citation annotating a span of assistant text, produced by a provider's
|
|
257
|
+
* SERVER-SIDE search tool (Anthropic `citations_delta` → `web_search_result_location`).
|
|
258
|
+
*
|
|
259
|
+
* Carries the source title/url so hosts can render precise "Sources" chips instead of
|
|
260
|
+
* regex-scraping URLs out of the prose.
|
|
261
|
+
*/
|
|
262
|
+
export interface WebCitation {
|
|
263
|
+
type: "web_search_result_location";
|
|
264
|
+
url: string;
|
|
265
|
+
title?: string;
|
|
266
|
+
/** The span of source material the assistant drew on. */
|
|
267
|
+
citedText?: string;
|
|
268
|
+
/**
|
|
269
|
+
* Opaque provider index for the cited result. Retained for fidelity only — it is never
|
|
270
|
+
* re-sent, because echoing a provider's encrypted citation fields back on a later turn
|
|
271
|
+
* must be byte-exact or the request is rejected.
|
|
272
|
+
*/
|
|
273
|
+
encryptedIndex?: string;
|
|
274
|
+
}
|
|
275
|
+
|
|
255
276
|
export interface TextContent {
|
|
256
277
|
type: "text";
|
|
257
278
|
text: string;
|
|
258
279
|
textSignature?: string; // e.g., for OpenAI responses, message metadata (legacy id string or TextSignatureV1 JSON)
|
|
280
|
+
/** Structured citations for this span, when the provider ran a server-side search. */
|
|
281
|
+
citations?: WebCitation[];
|
|
259
282
|
}
|
|
260
283
|
|
|
261
284
|
export interface ThinkingContent {
|
|
@@ -429,6 +452,32 @@ export type AssistantMessageEvent =
|
|
|
429
452
|
| { type: "toolcall_start"; contentIndex: number; partial: AssistantMessage }
|
|
430
453
|
| { type: "toolcall_delta"; contentIndex: number; delta: string; partial: AssistantMessage }
|
|
431
454
|
| { type: "toolcall_end"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage }
|
|
455
|
+
/**
|
|
456
|
+
* A provider-SIDE tool (e.g. Anthropic's built-in web search) started running.
|
|
457
|
+
*
|
|
458
|
+
* Purely a progress signal so hosts can show a live activity row — the provider executes
|
|
459
|
+
* these itself, so unlike `toolcall_*` there is nothing for the agent loop to dispatch and
|
|
460
|
+
* no content block is added to the message.
|
|
461
|
+
*/
|
|
462
|
+
| {
|
|
463
|
+
type: "server_tool_start";
|
|
464
|
+
contentIndex?: undefined;
|
|
465
|
+
toolName: string;
|
|
466
|
+
toolId: string;
|
|
467
|
+
/** The search query, when the provider streamed one. */
|
|
468
|
+
query?: string;
|
|
469
|
+
partial: AssistantMessage;
|
|
470
|
+
}
|
|
471
|
+
/** A provider-side tool finished: either it returned results, or it failed. */
|
|
472
|
+
| {
|
|
473
|
+
type: "server_tool_end";
|
|
474
|
+
contentIndex?: undefined;
|
|
475
|
+
toolName: string;
|
|
476
|
+
toolId: string;
|
|
477
|
+
resultCount?: number;
|
|
478
|
+
errorCode?: string;
|
|
479
|
+
partial: AssistantMessage;
|
|
480
|
+
}
|
|
432
481
|
| {
|
|
433
482
|
type: "done";
|
|
434
483
|
contentIndex?: undefined;
|