@f5-sales-demo/pi-ai 19.90.0 → 19.91.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@f5-sales-demo/pi-ai",
4
- "version": "19.90.0",
4
+ "version": "19.91.0",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://github.com/f5-sales-demo/xcsh",
7
7
  "author": "Can Boluk",
@@ -41,11 +41,11 @@
41
41
  "generate-models": "bun scripts/generate-models.ts"
42
42
  },
43
43
  "dependencies": {
44
- "@anthropic-ai/sdk": "^0.78",
44
+ "@anthropic-ai/sdk": "^0.115",
45
45
  "@aws-sdk/client-bedrock-runtime": "^3",
46
46
  "@bufbuild/protobuf": "^2.11",
47
- "@google/genai": "^1.43",
48
- "@f5-sales-demo/pi-utils": "19.90.0",
47
+ "@google/genai": "^2.13",
48
+ "@f5-sales-demo/pi-utils": "19.91.0",
49
49
  "@sinclair/typebox": "^0.34",
50
50
  "@smithy/node-http-handler": "^4.4",
51
51
  "ajv": "^8.20",
@@ -1,21 +1,51 @@
1
1
  import { resolveOpenAICompat } from "./providers/openai-completions-compat";
2
2
  import type { Api, Model as ApiModel, ThinkingConfig } from "./types";
3
3
 
4
- /** User-facing thinking levels, ordered least to most intensive. */
4
+ /**
5
+ * User-facing thinking levels, ordered least to most intensive.
6
+ *
7
+ * `Minimal` is xcsh-only — no provider accepts it on the wire (Anthropic rejects
8
+ * it with `output_config.effort: Input should be 'low', 'medium', 'high', 'xhigh'
9
+ * or 'max'`), so every mapper must translate it down.
10
+ */
5
11
  export const enum Effort {
6
12
  Minimal = "minimal",
7
13
  Low = "low",
8
14
  Medium = "medium",
9
15
  High = "high",
10
16
  XHigh = "xhigh",
17
+ Max = "max",
18
+ }
19
+
20
+ /**
21
+ * Anthropic's `output_config.effort` enum, verbatim. Authority is the API's own
22
+ * validation error: `Input should be 'low', 'medium', 'high', 'xhigh' or 'max'`.
23
+ * Deliberately excludes xcsh's `minimal`, which the API rejects.
24
+ */
25
+ export type AnthropicAdaptiveEffort = "low" | "medium" | "high" | "xhigh" | "max";
26
+
27
+ /** Effort levels for providers whose wire enum stops at `xhigh` (no `max`). */
28
+ export type EffortThroughXHigh = "minimal" | "low" | "medium" | "high" | "xhigh";
29
+
30
+ /**
31
+ * Clamp an effort to a provider whose wire enum has no `max` (OpenAI-compat
32
+ * reasoning_effort, GitLab Duo, Kimi, Synthetic, …). `requireSupportedEffort`
33
+ * already keeps `Max` out of those providers at runtime — because their
34
+ * supported-effort lists exclude it — but the wire types can't prove that, and
35
+ * clamping is the safe direction if a caller bypasses the range check.
36
+ */
37
+ export function clampEffortThroughXHigh(effort: Effort): EffortThroughXHigh {
38
+ return effort === Effort.Max ? Effort.XHigh : effort;
11
39
  }
12
40
 
41
+ /** Order is load-bearing: `indexOf` drives expandEffortRange/requireSupportedEffort. */
13
42
  export const THINKING_EFFORTS: readonly Effort[] = [
14
43
  Effort.Minimal,
15
44
  Effort.Low,
16
45
  Effort.Medium,
17
46
  Effort.High,
18
47
  Effort.XHigh,
48
+ Effort.Max,
19
49
  ];
20
50
 
21
51
  const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
@@ -26,6 +56,18 @@ const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [
26
56
  Effort.High,
27
57
  Effort.XHigh,
28
58
  ];
59
+ /**
60
+ * Anthropic 4.6+/5-era range. These models accept the full API enum, including
61
+ * `xhigh` (Anthropic's recommended setting for coding/agentic work) and `max`.
62
+ */
63
+ const ANTHROPIC_ADAPTIVE_EFFORTS: readonly Effort[] = [
64
+ Effort.Minimal,
65
+ Effort.Low,
66
+ Effort.Medium,
67
+ Effort.High,
68
+ Effort.XHigh,
69
+ Effort.Max,
70
+ ];
29
71
  const GEMINI_3_PRO_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High];
30
72
  const GEMINI_3_FLASH_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High];
31
73
  const GPT_5_2_PLUS_EFFORTS: readonly Effort[] = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
@@ -250,15 +292,22 @@ export function mapEffortToGoogleThinkingLevel<TApi extends Api>(
250
292
  return "MEDIUM";
251
293
  case Effort.High:
252
294
  case Effort.XHigh:
295
+ case Effort.Max:
253
296
  return "HIGH";
254
297
  }
255
298
  }
256
299
 
257
- /** Maps a normalized thinking effort to Anthropic adaptive effort values. */
300
+ /**
301
+ * Maps a normalized thinking effort to Anthropic adaptive effort values.
302
+ *
303
+ * The API enum is `low | medium | high | xhigh | max` (per its own validation
304
+ * error). `Minimal` has no wire equivalent and is translated down to `low` —
305
+ * sending `"minimal"` is a 400.
306
+ */
258
307
  export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
259
308
  model: ApiModel<TApi>,
260
309
  effort: Effort,
261
- ): "low" | "medium" | "high" | "max" {
310
+ ): AnthropicAdaptiveEffort {
262
311
  switch (requireSupportedEffort(model, effort)) {
263
312
  case Effort.Minimal:
264
313
  case Effort.Low:
@@ -268,6 +317,8 @@ export function mapEffortToAnthropicAdaptiveEffort<TApi extends Api>(
268
317
  case Effort.High:
269
318
  return "high";
270
319
  case Effort.XHigh:
320
+ return "xhigh";
321
+ case Effort.Max:
271
322
  return "max";
272
323
  }
273
324
  }
@@ -398,7 +449,15 @@ function inferAnthropicSupportedEfforts<TApi extends Api>(
398
449
  (model.api === "anthropic-messages" || model.api === "bedrock-converse-stream") &&
399
450
  semverGte(parsedModel.version, "4.6")
400
451
  ) {
401
- return parsedModel.kind === "opus" ? DEFAULT_REASONING_EFFORTS_WITH_XHIGH : DEFAULT_REASONING_EFFORTS;
452
+ // Opus 4.6+ and Sonnet 5+ accept the extended range. Sonnet 4.6 tops out at
453
+ // `high` — the reason this can't key on version alone (#2341).
454
+ const extended = parsedModel.kind === "opus" || semverGte(parsedModel.version, "5.0");
455
+ if (!extended) return DEFAULT_REASONING_EFFORTS;
456
+ // `max` is only claimed for the first-party Messages API, where the enum was
457
+ // verified against the live gateway. Bedrock's effort support is not verified
458
+ // here, so it keeps the pre-existing `xhigh` ceiling rather than being widened
459
+ // on an assumption.
460
+ return model.api === "anthropic-messages" ? ANTHROPIC_ADAPTIVE_EFFORTS : DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
402
461
  }
403
462
  return inferFallbackEfforts(model);
404
463
  }
package/src/models.json CHANGED
@@ -2647,7 +2647,7 @@
2647
2647
  "thinking": {
2648
2648
  "mode": "anthropic-adaptive",
2649
2649
  "minLevel": "minimal",
2650
- "maxLevel": "xhigh"
2650
+ "maxLevel": "max"
2651
2651
  }
2652
2652
  },
2653
2653
  "claude-sonnet-4-0": {
@@ -2800,7 +2800,7 @@
2800
2800
  "thinking": {
2801
2801
  "mode": "anthropic-adaptive",
2802
2802
  "minLevel": "minimal",
2803
- "maxLevel": "xhigh"
2803
+ "maxLevel": "max"
2804
2804
  }
2805
2805
  }
2806
2806
  },
@@ -659,6 +659,7 @@ function buildAdditionalModelRequestFields(
659
659
  medium: 8192,
660
660
  high: 16384,
661
661
  xhigh: 32768,
662
+ max: 65536,
662
663
  };
663
664
  const budget = options.thinkingBudgets?.[level] ?? defaultBudgets[level];
664
665
 
@@ -3,12 +3,13 @@ import * as fs from "node:fs";
3
3
  import * as tls from "node:tls";
4
4
  import Anthropic, { type ClientOptions as AnthropicSdkClientOptions } from "@anthropic-ai/sdk";
5
5
  import type {
6
+ CitationsDelta,
6
7
  ContentBlockParam,
7
8
  MessageCreateParamsStreaming,
8
9
  MessageParam,
9
10
  } from "@anthropic-ai/sdk/resources/messages";
10
11
  import { $env, abortableSleep, isEnoent } from "@f5-sales-demo/pi-utils";
11
- import { mapEffortToAnthropicAdaptiveEffort } from "../model-thinking";
12
+ import { type AnthropicAdaptiveEffort, mapEffortToAnthropicAdaptiveEffort } from "../model-thinking";
12
13
  import { calculateCost } from "../models";
13
14
  import { getEnvApiKey, OUTPUT_FALLBACK_BUFFER } from "../stream";
14
15
  import type {
@@ -30,6 +31,7 @@ import type {
30
31
  ToolCall,
31
32
  ToolResultMessage,
32
33
  Usage,
34
+ WebCitation,
33
35
  } from "../types";
34
36
  import { isAnthropicOAuthToken, normalizeToolCallId, resolveCacheRetention } from "../utils";
35
37
  import { createAbortSourceTracker } from "../utils/abort";
@@ -355,7 +357,11 @@ function convertContentBlocks(content: (TextContent | ImageContent)[]):
355
357
  return blocks;
356
358
  }
357
359
 
358
- export type AnthropicEffort = "low" | "medium" | "high" | "max";
360
+ /**
361
+ * Anthropic's `output_config.effort` enum. Single source of truth lives in
362
+ * model-thinking so the mapper and this raw-passthrough path cannot drift.
363
+ */
364
+ export type AnthropicEffort = AnthropicAdaptiveEffort;
359
365
 
360
366
  export interface AnthropicOptions extends StreamOptions {
361
367
  /**
@@ -625,6 +631,24 @@ export function isProviderRetryableError(error: unknown): boolean {
625
631
  );
626
632
  }
627
633
 
634
+ /**
635
+ * Map an Anthropic `citations_delta` citation to a `WebCitation`.
636
+ *
637
+ * Only web-search citations are carried: the char/page/content-block locations belong to the
638
+ * document-citations API, which cites user-supplied documents rather than search results and has
639
+ * no source URL to render as a chip.
640
+ */
641
+ function mapAnthropicWebCitation(citation: CitationsDelta["citation"]): WebCitation | undefined {
642
+ if (citation.type !== "web_search_result_location") return undefined;
643
+ return {
644
+ type: "web_search_result_location",
645
+ url: citation.url,
646
+ ...(citation.title ? { title: citation.title } : {}),
647
+ ...(citation.cited_text ? { citedText: citation.cited_text } : {}),
648
+ ...(citation.encrypted_index ? { encryptedIndex: citation.encrypted_index } : {}),
649
+ };
650
+ }
651
+
628
652
  function createEmptyUsage(premiumRequests?: number): Usage {
629
653
  return {
630
654
  input: 0,
@@ -733,6 +757,19 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
733
757
  const anthropicRequest = client.messages.create({ ...params, stream: true }, { signal: requestSignal });
734
758
  let streamedReplayUnsafeContent = false;
735
759
 
760
+ // SERVER-SIDE tools (Anthropic's built-in web search) are executed by the provider,
761
+ // so their blocks are deliberately NOT pushed into `output.content`: there is nothing
762
+ // for the agent loop to dispatch, and keeping their `encrypted_content` out of history
763
+ // avoids the byte-exact echo requirement that would 400 a follow-up turn. They are
764
+ // tracked here only long enough to report progress — keyed by the stream's block index
765
+ // (to accumulate the query) and by tool id (to name the matching result block).
766
+ // Declared inside the retry loop so a replayed attempt starts clean.
767
+ const serverToolUses = new Map<
768
+ number,
769
+ { id: string; name: string; partialJson: string; inlineInput: unknown }
770
+ >();
771
+ const serverToolNames = new Map<string, string>();
772
+
736
773
  try {
737
774
  const { data: anthropicStream } = await anthropicRequest.withResponse();
738
775
  const firstEventWatchdog = createFirstEventWatchdog(firstEventTimeoutMs, () =>
@@ -821,6 +858,30 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
821
858
  contentIndex: output.content.length - 1,
822
859
  partial: output,
823
860
  });
861
+ } else if (event.content_block.type === "server_tool_use") {
862
+ // Progress-only: remember the call so its query can be reported once the
863
+ // input JSON has finished streaming (see content_block_stop).
864
+ serverToolUses.set(event.index, {
865
+ id: event.content_block.id,
866
+ name: event.content_block.name,
867
+ partialJson: "",
868
+ // Nothing obliges the provider to stream the input as deltas — it may
869
+ // arrive whole right here. Kept as the fallback for the query.
870
+ inlineInput: event.content_block.input,
871
+ });
872
+ serverToolNames.set(event.content_block.id, event.content_block.name);
873
+ } else if (event.content_block.type === "web_search_tool_result") {
874
+ const toolId = event.content_block.tool_use_id;
875
+ const content = event.content_block.content;
876
+ stream.push({
877
+ type: "server_tool_end",
878
+ toolName: serverToolNames.get(toolId) ?? "web_search",
879
+ toolId,
880
+ ...(Array.isArray(content)
881
+ ? { resultCount: content.length }
882
+ : { errorCode: content.error_code }),
883
+ partial: output,
884
+ });
824
885
  }
825
886
  } else if (event.type === "content_block_delta") {
826
887
  if (event.delta.type === "text_delta") {
@@ -848,17 +909,31 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
848
909
  });
849
910
  }
850
911
  } else if (event.delta.type === "input_json_delta") {
912
+ const serverToolUse = serverToolUses.get(event.index);
913
+ if (serverToolUse) {
914
+ // A server-side tool's input (the search query). Accumulated but not
915
+ // streamed as a toolcall_delta — there is no toolCall block to attach to.
916
+ serverToolUse.partialJson += event.delta.partial_json;
917
+ } else {
918
+ const index = blocks.findIndex(b => b.index === event.index);
919
+ const block = blocks[index];
920
+ if (block && block.type === "toolCall") {
921
+ block.partialJson += event.delta.partial_json;
922
+ block.arguments = parseStreamingJson(block.partialJson);
923
+ stream.push({
924
+ type: "toolcall_delta",
925
+ contentIndex: index,
926
+ delta: event.delta.partial_json,
927
+ partial: output,
928
+ });
929
+ }
930
+ }
931
+ } else if (event.delta.type === "citations_delta") {
851
932
  const index = blocks.findIndex(b => b.index === event.index);
852
933
  const block = blocks[index];
853
- if (block && block.type === "toolCall") {
854
- block.partialJson += event.delta.partial_json;
855
- block.arguments = parseStreamingJson(block.partialJson);
856
- stream.push({
857
- type: "toolcall_delta",
858
- contentIndex: index,
859
- delta: event.delta.partial_json,
860
- partial: output,
861
- });
934
+ const citation = mapAnthropicWebCitation(event.delta.citation);
935
+ if (block && block.type === "text" && citation) {
936
+ block.citations = [...(block.citations ?? []), citation];
862
937
  }
863
938
  } else if (event.delta.type === "signature_delta") {
864
939
  const index = blocks.findIndex(b => b.index === event.index);
@@ -869,6 +944,28 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
869
944
  }
870
945
  }
871
946
  } else if (event.type === "content_block_stop") {
947
+ const serverToolUse = serverToolUses.get(event.index);
948
+ if (serverToolUse) {
949
+ // The query has finished streaming, so the activity row can name what is
950
+ // being searched for. Reported here rather than at content_block_start
951
+ // because that is the first moment the query is known — and it still lands
952
+ // well before the provider's search completes.
953
+ serverToolUses.delete(event.index);
954
+ const streamed = parseStreamingJson(serverToolUse.partialJson) as
955
+ | { query?: unknown }
956
+ | undefined;
957
+ const inline = serverToolUse.inlineInput as { query?: unknown } | undefined;
958
+ const rawQuery = typeof streamed?.query === "string" ? streamed.query : inline?.query;
959
+ const query = typeof rawQuery === "string" && rawQuery.length > 0 ? rawQuery : undefined;
960
+ stream.push({
961
+ type: "server_tool_start",
962
+ toolName: serverToolUse.name,
963
+ toolId: serverToolUse.id,
964
+ ...(query === undefined ? {} : { query }),
965
+ partial: output,
966
+ });
967
+ continue;
968
+ }
872
969
  const index = blocks.findIndex(b => b.index === event.index);
873
970
  const block = blocks[index];
874
971
  if (block) {
@@ -1,3 +1,4 @@
1
+ import { clampEffortThroughXHigh } from "../model-thinking";
1
2
  import { ANTHROPIC_THINKING, mapAnthropicToolChoice } from "../stream";
2
3
  import type { Api, Context, Model, SimpleStreamOptions } from "../types";
3
4
  import { AssistantMessageEventStream } from "../utils/event-stream";
@@ -301,6 +302,7 @@ export function streamGitLabDuo(
301
302
  thinkingBudgetTokens: reasoningEffort
302
303
  ? (options.thinkingBudgets?.[reasoningEffort] ?? ANTHROPIC_THINKING[reasoningEffort])
303
304
  : undefined,
305
+ // Anthropic path — takes the full ladder, no clamp.
304
306
  reasoning: reasoningEffort,
305
307
  toolChoice: mapAnthropicToolChoice(options.toolChoice),
306
308
  },
@@ -331,7 +333,7 @@ export function streamGitLabDuo(
331
333
  sessionId: options.sessionId,
332
334
  providerSessionState: options.providerSessionState,
333
335
  onPayload: options.onPayload,
334
- reasoning: reasoningEffort,
336
+ reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
335
337
  toolChoice: options.toolChoice,
336
338
  } satisfies OpenAIResponsesOptions,
337
339
  )
@@ -360,7 +362,7 @@ export function streamGitLabDuo(
360
362
  sessionId: options.sessionId,
361
363
  providerSessionState: options.providerSessionState,
362
364
  onPayload: options.onPayload,
363
- reasoning: reasoningEffort,
365
+ reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
364
366
  toolChoice: options.toolChoice,
365
367
  } satisfies OpenAICompletionsOptions,
366
368
  );
@@ -1,3 +1,4 @@
1
+ import { clampEffortThroughXHigh } from "../model-thinking";
1
2
  /**
2
3
  * Kimi Code provider - wraps OpenAI or Anthropic API based on format setting.
3
4
  *
@@ -104,7 +105,7 @@ export function streamKimi(
104
105
  headers: mergedHeaders,
105
106
  sessionId: options?.sessionId,
106
107
  onPayload: options?.onPayload,
107
- reasoning: reasoningEffort,
108
+ reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
108
109
  });
109
110
 
110
111
  for await (const event of innerStream) {
@@ -1,5 +1,5 @@
1
1
  import type { Effort } from "../../model-thinking";
2
- import { requireSupportedEffort } from "../../model-thinking";
2
+ import { clampEffortThroughXHigh, requireSupportedEffort } from "../../model-thinking";
3
3
  import type { Api, Model } from "../../types";
4
4
 
5
5
  export interface ReasoningConfig {
@@ -53,8 +53,12 @@ export interface RequestBody {
53
53
 
54
54
  function getReasoningConfig(model: Model<Api>, options: CodexRequestOptions): ReasoningConfig {
55
55
  return {
56
+ // Codex's reasoning effort has no `max`; clamp so the ladder's top level
57
+ // degrades to `xhigh` instead of emitting a value Codex would reject.
56
58
  effort:
57
- options.reasoningEffort === "none" ? "none" : requireSupportedEffort(model, options.reasoningEffort as Effort),
59
+ options.reasoningEffort === "none"
60
+ ? "none"
61
+ : clampEffortThroughXHigh(requireSupportedEffort(model, options.reasoningEffort as Effort)),
58
62
  summary: options.reasoningSummary ?? "detailed",
59
63
  };
60
64
  }
@@ -1,3 +1,4 @@
1
+ import { clampEffortThroughXHigh } from "../model-thinking";
1
2
  /**
2
3
  * Synthetic provider - wraps OpenAI or Anthropic API based on format setting.
3
4
  *
@@ -107,7 +108,7 @@ export function streamSynthetic(
107
108
  headers: mergedHeaders,
108
109
  sessionId: options?.sessionId,
109
110
  onPayload: options?.onPayload,
110
- reasoning: reasoningEffort,
111
+ reasoning: reasoningEffort && clampEffortThroughXHigh(reasoningEffort),
111
112
  });
112
113
 
113
114
  for await (const event of innerStream) {
package/src/stream.ts CHANGED
@@ -5,6 +5,8 @@ import { $env, $pickenv } from "@f5-sales-demo/pi-utils";
5
5
  import { getCustomApi } from "./api-registry";
6
6
  import type { Effort } from "./model-thinking";
7
7
  import {
8
+ clampEffortThroughXHigh,
9
+ type EffortThroughXHigh,
8
10
  mapEffortToAnthropicAdaptiveEffort,
9
11
  mapEffortToGoogleThinkingLevel,
10
12
  requireSupportedEffort,
@@ -317,6 +319,7 @@ export const ANTHROPIC_THINKING: Record<Effort, number> = {
317
319
  medium: 8192,
318
320
  high: 16384,
319
321
  xhigh: 32768,
322
+ max: 65536,
320
323
  };
321
324
 
322
325
  const GOOGLE_THINKING: Record<Effort, number> = {
@@ -325,6 +328,8 @@ const GOOGLE_THINKING: Record<Effort, number> = {
325
328
  medium: 8192,
326
329
  high: 16384,
327
330
  xhigh: 24575,
331
+ // Google caps the thinking budget here; `max` cannot exceed it.
332
+ max: 24575,
328
333
  };
329
334
 
330
335
  const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
@@ -333,6 +338,8 @@ const BEDROCK_CLAUDE_THINKING: Record<Effort, number> = {
333
338
  medium: 8192,
334
339
  high: 16384,
335
340
  xhigh: 16384,
341
+ // Bedrock Claude caps the thinking budget here; `max` cannot exceed it.
342
+ max: 16384,
336
343
  };
337
344
 
338
345
  function resolveBedrockThinkingBudget(
@@ -394,10 +401,12 @@ function mapOpenAiToolChoice(choice?: ToolChoice): OpenAICompletionsOptions["too
394
401
  function resolveOpenAiReasoningEffort<TApi extends Api>(
395
402
  model: Model<TApi>,
396
403
  options?: SimpleStreamOptions,
397
- ): Effort | undefined {
404
+ ): EffortThroughXHigh | undefined {
398
405
  const reasoning = options?.reasoning;
399
406
  if (!reasoning || !model.reasoning) return undefined;
400
- return requireSupportedEffort(model, reasoning);
407
+ // OpenAI-compat `reasoning_effort` has no `max`; clamp rather than emit a
408
+ // value the provider would reject.
409
+ return clampEffortThroughXHigh(requireSupportedEffort(model, reasoning));
401
410
  }
402
411
 
403
412
  const castApi = <TApi extends Api>(api: OptionsForApi<TApi>): OptionsForApi<Api> => api as OptionsForApi<Api>;
package/src/types.ts CHANGED
@@ -252,10 +252,33 @@ export interface TextSignatureV1 {
252
252
  phase?: "commentary" | "final_answer";
253
253
  }
254
254
 
255
+ /**
256
+ * A structured citation annotating a span of assistant text, produced by a provider's
257
+ * SERVER-SIDE search tool (Anthropic `citations_delta` → `web_search_result_location`).
258
+ *
259
+ * Carries the source title/url so hosts can render precise "Sources" chips instead of
260
+ * regex-scraping URLs out of the prose.
261
+ */
262
+ export interface WebCitation {
263
+ type: "web_search_result_location";
264
+ url: string;
265
+ title?: string;
266
+ /** The span of source material the assistant drew on. */
267
+ citedText?: string;
268
+ /**
269
+ * Opaque provider index for the cited result. Retained for fidelity only — it is never
270
+ * re-sent, because echoing a provider's encrypted citation fields back on a later turn
271
+ * must be byte-exact or the request is rejected.
272
+ */
273
+ encryptedIndex?: string;
274
+ }
275
+
255
276
  export interface TextContent {
256
277
  type: "text";
257
278
  text: string;
258
279
  textSignature?: string; // e.g., for OpenAI responses, message metadata (legacy id string or TextSignatureV1 JSON)
280
+ /** Structured citations for this span, when the provider ran a server-side search. */
281
+ citations?: WebCitation[];
259
282
  }
260
283
 
261
284
  export interface ThinkingContent {
@@ -429,6 +452,32 @@ export type AssistantMessageEvent =
429
452
  | { type: "toolcall_start"; contentIndex: number; partial: AssistantMessage }
430
453
  | { type: "toolcall_delta"; contentIndex: number; delta: string; partial: AssistantMessage }
431
454
  | { type: "toolcall_end"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage }
455
+ /**
456
+ * A provider-SIDE tool (e.g. Anthropic's built-in web search) started running.
457
+ *
458
+ * Purely a progress signal so hosts can show a live activity row — the provider executes
459
+ * these itself, so unlike `toolcall_*` there is nothing for the agent loop to dispatch and
460
+ * no content block is added to the message.
461
+ */
462
+ | {
463
+ type: "server_tool_start";
464
+ contentIndex?: undefined;
465
+ toolName: string;
466
+ toolId: string;
467
+ /** The search query, when the provider streamed one. */
468
+ query?: string;
469
+ partial: AssistantMessage;
470
+ }
471
+ /** A provider-side tool finished: either it returned results, or it failed. */
472
+ | {
473
+ type: "server_tool_end";
474
+ contentIndex?: undefined;
475
+ toolName: string;
476
+ toolId: string;
477
+ resultCount?: number;
478
+ errorCode?: string;
479
+ partial: AssistantMessage;
480
+ }
432
481
  | {
433
482
  type: "done";
434
483
  contentIndex?: undefined;