dsh-lcx-codex 0.4.3-pre.4 → 0.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +68 -126
  2. package/THIRD_PARTY_NOTICES.md +64 -0
  3. package/lib/auxiliary-usage.js +63 -0
  4. package/lib/client.js +1363 -295
  5. package/lib/dsh-compat.js +80 -18
  6. package/lib/dsh-responses.js +75 -8
  7. package/lib/grok-native-search.js +151 -79
  8. package/lib/index.js +159 -101
  9. package/lib/invocation-policy-scope.js +261 -0
  10. package/lib/json-store.js +11 -4
  11. package/lib/pi-responses-runtime.js +1571 -0
  12. package/lib/responses-request.js +1 -1
  13. package/lib/responses-stream.js +152 -84
  14. package/lib/route.js +17 -3
  15. package/lib/search-accounting.js +86 -0
  16. package/lib/search-usage.js +86 -0
  17. package/lib/transport.js +38 -10
  18. package/lib/types/client/index.d.ts +12 -0
  19. package/lib/types/client/search-media.d.ts +16 -0
  20. package/lib/web-run-output.js +22 -0
  21. package/lib/web-search-alpha.js +230 -28
  22. package/lib/web-search-capability.js +26 -1
  23. package/lib/web-search-hosted.js +119 -9
  24. package/lib/web-search-ref-store.js +88 -7
  25. package/package.json +45 -24
  26. package/lib/types/compact-v2.d.ts +0 -104
  27. package/lib/types/dsh-compat.d.ts +0 -78
  28. package/lib/types/dsh-responses.d.ts +0 -82
  29. package/lib/types/grok-native-search.d.ts +0 -39
  30. package/lib/types/json-store.d.ts +0 -10
  31. package/lib/types/native-checkpoint.d.ts +0 -213
  32. package/lib/types/responses-request.d.ts +0 -59
  33. package/lib/types/responses-stream.d.ts +0 -64
  34. package/lib/types/route.d.ts +0 -153
  35. package/lib/types/service-mutex.d.ts +0 -14
  36. package/lib/types/token-budget.d.ts +0 -50
  37. package/lib/types/transport.d.ts +0 -21
  38. package/lib/types/web-run-output.d.ts +0 -29
  39. package/lib/types/web-search-alpha.d.ts +0 -286
  40. package/lib/types/web-search-capability.d.ts +0 -26
  41. package/lib/types/web-search-hosted.d.ts +0 -246
  42. package/lib/types/web-search-ref-store.d.ts +0 -22
@@ -1,5 +1,5 @@
1
1
  // @ts-check
2
- import { clampOpenAIPromptCacheKey } from "@earendil-works/pi-ai/api/openai-prompt-cache";
2
+ import { clampOpenAIPromptCacheKey } from "./pi-responses-runtime.js";
3
3
  import { responsesTools } from "./dsh-responses.js";
4
4
  const OPENAI_RESPONSES_MIN_OUTPUT_TOKENS = 16;
5
5
  /** @param {unknown} value */
@@ -1,9 +1,8 @@
1
1
  // @ts-check
2
- import { createAssistantMessageEventStream, } from "@earendil-works/pi-ai";
3
- import { processResponsesStream } from "@earendil-works/pi-ai/api/openai-responses-shared";
2
+ import { createAssistantMessageEventStream, processResponsesStream, } from "./pi-responses-runtime.js";
4
3
  import { ToolCallId } from "@deepseek-ai/dsh-llm";
5
4
  import { fetchSseWithRetry } from "./transport.js";
6
- import { createGrokNativeReplayEnvelope, } from "./grok-native-search.js";
5
+ import { createGrokNativeReplayEnvelope, grokPendingSourcesStart, sanitizeGrokVisibleText, } from "./grok-native-search.js";
7
6
  function emptyUsage() {
8
7
  return {
9
8
  input: 0,
@@ -92,9 +91,15 @@ export function managedFailure(error, signal) {
92
91
  code = "ABORTED";
93
92
  else if (sourceCodes.has("LCX_RESPONSES_UNSUPPORTED_OPTION"))
94
93
  code = "UNSUPPORTED_OPTION";
94
+ else if (sourceCodes.has("LCX_CHECKPOINT_PORTABLE_UNSUPPORTED_CONTENT"))
95
+ code = "UNSUPPORTED_CONTENT";
96
+ else if (sourceCodes.has("LCX_GROK_NATIVE_PROTOCOL_ERROR"))
97
+ code = "GROK_NATIVE_PROTOCOL_ERROR";
95
98
  else if (sourceCodes.has("LCX_RESPONSES_ROUTE_UNAVAILABLE") ||
96
99
  sourceCodes.has("LCX_RESPONSES_MODEL_UNAVAILABLE"))
97
100
  code = "NO_ADAPTER";
101
+ else if (sourceCodes.has("LCX_INSUFFICIENT_QUOTA"))
102
+ code = "INSUFFICIENT_QUOTA";
98
103
  else if (facts.status === 401 ||
99
104
  facts.status === 403 ||
100
105
  sourceCodes.has("AUTH") ||
@@ -128,6 +133,7 @@ export function managedFailure(error, signal) {
128
133
  const messages = /** @type {Record<string, string>} */ {
129
134
  ABORTED: "Responses request was aborted",
130
135
  AUTH: "Responses request was rejected by authentication",
136
+ INSUFFICIENT_QUOTA: "服务商额度不足或未满足请求预留额度,请检查中转额度(并非 API 密钥无效) / Provider quota is insufficient for this request",
131
137
  UNSUPPORTED_OPTION: "LCX Responses does not support this request option",
132
138
  NO_ADAPTER: "LCX could not resolve the selected Responses route",
133
139
  RATE_LIMIT: "Responses provider rate limit was reached",
@@ -136,6 +142,8 @@ export function managedFailure(error, signal) {
136
142
  ? "Unsupported or invalid LCX checkpoint; start a new session"
137
143
  : "Responses provider rejected the request",
138
144
  CONTEXT_WINDOW_EXCEEDED: "Responses request exceeded the model context window",
145
+ UNSUPPORTED_CONTENT: "LCX cannot serialize a DSH content block for this Responses route",
146
+ GROK_NATIVE_PROTOCOL_ERROR: "Grok returned an unrecognized client tool call while native search was enabled",
139
147
  TIMEOUT: "Responses request timed out",
140
148
  TRANSPORT: "Responses transport failed",
141
149
  RESPONSES_ERROR: "Responses request failed",
@@ -484,6 +492,27 @@ function itemText(item) {
484
492
  return String(item.input ?? "");
485
493
  return "";
486
494
  }
495
+ function itemAnnotations(item) {
496
+ if (item.type !== "message" || !Array.isArray(item.content))
497
+ return [];
498
+ return item.content.flatMap((part) => isObject(part) && Array.isArray(part.annotations) ? part.annotations : []);
499
+ }
500
+ function sanitizeGrokTerminalItem(item, index) {
501
+ const normalized = normalizedTerminalItem(item, index);
502
+ if (normalized.type !== "message" || !Array.isArray(normalized.content))
503
+ return normalized;
504
+ return {
505
+ ...normalized,
506
+ content: normalized.content.map((part) => {
507
+ if (!isObject(part) || (part.type !== "output_text" && part.type !== "refusal"))
508
+ return structuredClone(part);
509
+ const text = sanitizeGrokVisibleText(String(part.text ?? part.refusal ?? ""), Array.isArray(part.annotations) ? part.annotations : []);
510
+ return part.type === "output_text"
511
+ ? { ...structuredClone(part), text }
512
+ : { ...structuredClone(part), refusal: text };
513
+ }),
514
+ };
515
+ }
487
516
  /** @param {UnknownRecord} item @param {number} index */
488
517
  function normalizedTerminalItem(item, index) {
489
518
  if (item.type === "message")
@@ -574,42 +603,60 @@ function recordMatches(record, item) {
574
603
  */
575
604
  async function* normalizedResponseEvents(source, meta = {}, options = {}) {
576
605
  const open = new Map();
577
- const completed = new Set();
578
- const completedWireItems = new Map();
579
- const serverToolIds = new Map();
580
- const citations = new Map();
581
- const observeCitation = (candidate) => {
582
- if (!isObject(candidate))
583
- return;
584
- try {
585
- const url = new URL(String(candidate.url ?? ""));
586
- if (url.protocol !== "http:" && url.protocol !== "https:")
587
- return;
588
- const title = typeof candidate.title === "string"
589
- ? candidate.title.replace(/[\r\n]+/gu, " ").trim()
590
- : "";
591
- if (!citations.has(url.href))
592
- citations.set(url.href, title);
606
+ const grokTextProjection = new Map();
607
+ const hiddenPrefixTokens = ["sources:", "render_inline_citation", "{render_inline_citation", "render_inline_citation", "stateless_invoke", "[[", "<|eos|>"];
608
+ const safeGrokRawPrefix = (raw, final) => {
609
+ if (final)
610
+ return raw;
611
+ const lower = raw.toLowerCase();
612
+ let holdStart = raw.length;
613
+ const pendingSource = grokPendingSourcesStart(raw);
614
+ if (pendingSource !== undefined)
615
+ holdStart = Math.min(holdStart, pendingSource);
616
+ for (const token of hiddenPrefixTokens) {
617
+ const complete = lower.lastIndexOf(token);
618
+ if (complete >= 0) {
619
+ const tail = raw.slice(complete);
620
+ if (sanitizeGrokVisibleText(tail).toLowerCase().includes(token.replace(/^\{/u, "")))
621
+ holdStart = Math.min(holdStart, complete);
622
+ }
623
+ const max = Math.min(token.length - 1, lower.length);
624
+ for (let size = max; size > 0; size -= 1) {
625
+ if (lower.endsWith(token.slice(0, size))) {
626
+ holdStart = Math.min(holdStart, raw.length - size);
627
+ break;
628
+ }
629
+ }
593
630
  }
594
- catch { }
631
+ return raw.slice(0, holdStart);
595
632
  };
596
- const observeCitations = (item) => {
597
- if (!isObject(item))
598
- return;
599
- if (item.type === "message" && Array.isArray(item.content)) {
600
- for (const part of item.content) {
601
- if (!isObject(part) || part.type !== "output_text" || !Array.isArray(part.annotations))
602
- continue;
603
- for (const annotation of part.annotations)
604
- if (isObject(annotation) && annotation.type === "url_citation")
605
- observeCitation(annotation);
606
- }
633
+ const grokProjectionDelta = (index, rawText, final = false, annotations = []) => {
634
+ let state = grokTextProjection.get(index);
635
+ if (!state) {
636
+ state = { raw: "", emitted: "" };
637
+ grokTextProjection.set(index, state);
607
638
  }
608
- // Search candidates are not answer citations. Only output_text annotations qualify.
639
+ state.raw = rawText;
640
+ const target = sanitizeGrokVisibleText(safeGrokRawPrefix(rawText, final), annotations);
641
+ if (!target.startsWith(state.emitted))
642
+ return "";
643
+ const delta = target.slice(state.emitted.length);
644
+ state.emitted = target;
645
+ if (final)
646
+ grokTextProjection.delete(index);
647
+ return delta;
609
648
  };
610
- const isServerToolItem = (value) => isObject(value) && options.serverToolTypes?.has(String(value.type ?? "")) === true;
649
+ const completed = new Set();
650
+ const completedWireItems = new Map();
651
+ const hiddenServerToolIndexes = new Set();
652
+ const serverToolIds = new Map();
653
+ const nativeServerSearchEchoCallIds = new Set();
654
+ const isServerToolItem = (value) => isObject(value) && (options.serverToolTypes?.has(String(value.type ?? "")) === true ||
655
+ options.isServerToolItem?.(value) === true);
611
656
  const observeServerTool = (item) => {
612
657
  const type = String(item.type);
658
+ if (type === "custom_tool_call" && typeof item.call_id === "string")
659
+ nativeServerSearchEchoCallIds.add(item.call_id);
613
660
  let ids = serverToolIds.get(type);
614
661
  if (!ids) {
615
662
  ids = new Set();
@@ -645,8 +692,18 @@ async function* normalizedResponseEvents(source, meta = {}, options = {}) {
645
692
  event.type === "response.output_item.done") &&
646
693
  isServerToolItem(event.item)) {
647
694
  observeServerTool(event.item);
695
+ if (event.type === "response.output_item.added")
696
+ hiddenServerToolIndexes.add(index);
697
+ else
698
+ hiddenServerToolIndexes.delete(index);
648
699
  continue;
649
700
  }
701
+ if (hiddenServerToolIndexes.has(index) &&
702
+ (event.type === "response.custom_tool_call_input.delta" ||
703
+ event.type === "response.custom_tool_call_input.done" ||
704
+ event.type === "response.function_call_arguments.delta" ||
705
+ event.type === "response.function_call_arguments.done"))
706
+ continue;
650
707
  // Grok may send many encrypted reasoning items with no visible summary.
651
708
  // Keep them in completedWireItems/nativeOutput, but do not open empty UI blocks.
652
709
  if (options.serverToolTypes &&
@@ -655,9 +712,10 @@ async function* normalizedResponseEvents(source, meta = {}, options = {}) {
655
712
  !itemText(event.item).trim() && !open.get(index)?.text.trim())
656
713
  continue;
657
714
  if (event.type === "response.output_item.added" && isObject(event.item)) {
658
- const item = normalizedTerminalItem(event.item, index);
659
- open.set(index, { kind: itemKind(item), item, text: itemText(item) });
660
- yield { ...event, item };
715
+ const rawItem = normalizedTerminalItem(event.item, index);
716
+ const item = options.serverToolTypes ? sanitizeGrokTerminalItem(rawItem, index) : rawItem;
717
+ open.set(index, { kind: itemKind(item), item: rawItem, text: itemText(rawItem) });
718
+ yield { ...event, item: options.serverToolTypes ? addedShell(item, index) : item };
661
719
  continue;
662
720
  }
663
721
  if (event.type === "response.output_text.delta") {
@@ -678,7 +736,13 @@ async function* normalizedResponseEvents(source, meta = {}, options = {}) {
678
736
  record = next.value;
679
737
  }
680
738
  record.text = String(record.text ?? "") + String(event.delta ?? "");
681
- yield event;
739
+ if (!options.serverToolTypes)
740
+ yield event;
741
+ else {
742
+ const delta = grokProjectionDelta(index, record.text);
743
+ if (delta)
744
+ yield { ...event, delta };
745
+ }
682
746
  continue;
683
747
  }
684
748
  if (event.type === "response.reasoning_summary_text.delta" ||
@@ -756,7 +820,13 @@ async function* normalizedResponseEvents(source, meta = {}, options = {}) {
756
820
  continue;
757
821
  }
758
822
  if (event.type === "response.output_item.done" && isObject(event.item)) {
759
- const item = normalizedTerminalItem(event.item, index);
823
+ const rawItem = normalizedTerminalItem(event.item, index);
824
+ const item = options.serverToolTypes ? sanitizeGrokTerminalItem(rawItem, index) : rawItem;
825
+ if (options.serverToolTypes && item.type === "message") {
826
+ const delta = grokProjectionDelta(index, itemText(rawItem), true, itemAnnotations(rawItem));
827
+ if (delta)
828
+ yield { type: "response.output_text.delta", output_index: index, content_index: 0, item_id: item.id, delta };
829
+ }
760
830
  const identity = itemIdentity(item);
761
831
  if (identity)
762
832
  completed.add(identity);
@@ -776,52 +846,36 @@ async function* normalizedResponseEvents(source, meta = {}, options = {}) {
776
846
  .sort(([left], [right]) => left - right)
777
847
  .map(([, item]) => item);
778
848
  const nativeOutput = terminalOutput.map((item) => structuredClone(item));
779
- const nativeVisibleAdditions = [];
780
849
  for (const item of terminalOutput) {
781
- observeCitations(item);
850
+ if (isServerToolItem(item) && isObject(item) && item.type === "custom_tool_call" &&
851
+ item.status !== "completed")
852
+ throw Object.assign(new Error("Grok native custom search item did not complete on the server"), { code: "LCX_GROK_NATIVE_PROTOCOL_ERROR" });
782
853
  if (isServerToolItem(item))
783
854
  observeServerTool(item);
855
+ else if (options.serverToolTypes && isObject(item) && item.type === "custom_tool_call" &&
856
+ typeof item.name === "string" && options.declaredToolNames?.has(item.name) !== true)
857
+ throw Object.assign(new Error(`Grok returned undeclared client custom tool "${item.name}"`), { code: "LCX_GROK_NATIVE_PROTOCOL_ERROR" });
784
858
  }
785
859
  const output = terminalOutput
786
860
  .filter((item) => !isServerToolItem(item))
787
861
  .filter((item) => !(options.serverToolTypes && isObject(item) &&
788
862
  item.type === "reasoning" && !itemText(item).trim()))
789
863
  .map((item, terminalIndex) => isObject(item)
790
- ? normalizedTerminalItem(item, terminalIndex)
864
+ ? (options.serverToolTypes
865
+ ? sanitizeGrokTerminalItem(item, terminalIndex)
866
+ : normalizedTerminalItem(item, terminalIndex))
791
867
  : item);
792
- if (options.serverToolTypes && citations.size > 0) {
793
- const visibleText = output
794
- .filter(isObject)
795
- .map((item) => itemText(item))
796
- .join("\n");
797
- const missing = [...citations].filter(([url]) => !visibleText.includes(url));
798
- if (missing.length > 0) {
799
- const suffix = missing
800
- .map(([url, title]) => `- ${title ? `${title}: ` : ""}${url}`)
801
- .join("\n");
802
- const responseId = String(response.id ?? "native")
803
- .replace(/[^A-Za-z0-9_-]/gu, "_")
804
- .slice(0, 40);
805
- const fallbackItem = {
806
- type: "message",
807
- id: `msg_lcx_sources_${responseId}`.slice(0, 64),
808
- role: "assistant",
809
- status: "completed",
810
- content: [{
811
- type: "output_text",
812
- text: `\n\nSources:\n${suffix}`,
813
- annotations: [],
814
- }],
815
- };
816
- output.push(fallbackItem);
817
- nativeVisibleAdditions.push(structuredClone(fallbackItem));
818
- }
819
- }
820
868
  if (event.type === "response.completed" &&
821
869
  response.status === "completed" &&
822
870
  options.serverToolTypes) {
823
871
  meta.nativeOutput = nativeOutput;
824
- meta.nativeVisibleAdditions = nativeVisibleAdditions;
872
+ meta.nativeServerSearchEchoCallIds = [...nativeServerSearchEchoCallIds];
873
+ const usage = isObject(response.usage) ? response.usage : {};
874
+ const details = isObject(usage.server_side_tool_usage_details) ? usage.server_side_tool_usage_details : {};
875
+ meta.inputTokenScope = serverToolIds.size > 0 || nativeOutput.some(isServerToolItem)
876
+ || Number(usage.num_server_side_tools_used ?? 0) > 0
877
+ || Number(details.web_search_calls ?? 0) > 0 || Number(details.x_search_calls ?? 0) > 0
878
+ ? "aggregate" : "request";
825
879
  }
826
880
  const used = new Set();
827
881
  for (const [streamIndex, record] of open) {
@@ -838,6 +892,11 @@ async function* normalizedResponseEvents(source, meta = {}, options = {}) {
838
892
  : normalizedTerminalItem(record.item, streamIndex);
839
893
  if (terminalIndex >= 0)
840
894
  used.add(terminalIndex);
895
+ if (options.serverToolTypes && item.type === "message") {
896
+ const delta = grokProjectionDelta(streamIndex, itemText(item), true, itemAnnotations(item));
897
+ if (delta)
898
+ yield { type: "response.output_text.delta", output_index: streamIndex, content_index: 0, item_id: item.id, delta };
899
+ }
841
900
  yield {
842
901
  type: "response.output_item.done",
843
902
  output_index: streamIndex,
@@ -848,6 +907,7 @@ async function* normalizedResponseEvents(source, meta = {}, options = {}) {
848
907
  completed.add(identity);
849
908
  }
850
909
  open.clear();
910
+ grokTextProjection.clear();
851
911
  for (const [terminalIndex, candidate] of output.entries()) {
852
912
  if (!isObject(candidate) ||
853
913
  !itemKind(candidate) ||
@@ -939,15 +999,19 @@ function isManagedFailure(value) {
939
999
  }
940
1000
  function dshUsage(usage) {
941
1001
  const value = /** @type {UnknownRecord} */ isObject(usage) ? usage : {};
1002
+ const inputTokens = Number(value.input ?? 0);
1003
+ const outputTokens = Number(value.output ?? 0);
1004
+ const cacheReadTokens = Number(value.cacheRead ?? 0);
1005
+ const cacheWriteTokens = Number(value.cacheWrite ?? 0);
1006
+ const reportedTotal = Number(value.totalTokens);
942
1007
  return {
943
- inputTokens: Number(value.input ?? 0),
944
- outputTokens: Number(value.output ?? 0),
945
- ...(Number(value.cacheRead ?? 0) > 0
946
- ? { cacheReadTokens: Number(value.cacheRead) }
947
- : {}),
948
- ...(Number(value.cacheWrite ?? 0) > 0
949
- ? { cacheWriteTokens: Number(value.cacheWrite) }
950
- : {}),
1008
+ inputTokens,
1009
+ outputTokens,
1010
+ totalTokens: Number.isSafeInteger(reportedTotal) && reportedTotal >= 0
1011
+ ? reportedTotal
1012
+ : inputTokens + outputTokens + cacheReadTokens + cacheWriteTokens,
1013
+ cacheReadTokens,
1014
+ cacheWriteTokens,
951
1015
  ...(Number(value.reasoning ?? 0) > 0
952
1016
  ? { reasoningTokens: Number(value.reasoning) }
953
1017
  : {}),
@@ -963,7 +1027,7 @@ function rawArguments(value) {
963
1027
  }
964
1028
  }
965
1029
  /** @param {unknown} message */
966
- function replayState(message, nativeOutput, nativeVisibleAdditions, nativeReplayRoute) {
1030
+ function replayState(message, nativeOutput, nativeReplayRoute, nativeServerSearchEchoCallIds = []) {
967
1031
  if (!isObject(message))
968
1032
  return undefined;
969
1033
  const content = Array.isArray(message.content)
@@ -974,7 +1038,7 @@ function replayState(message, nativeOutput, nativeVisibleAdditions, nativeReplay
974
1038
  const api = typeof message.api === "string" ? message.api : undefined;
975
1039
  if (!provider || !model || !api)
976
1040
  return undefined;
977
- const grokNative = createGrokNativeReplayEnvelope(nativeOutput, nativeReplayRoute, nativeVisibleAdditions);
1041
+ const grokNative = createGrokNativeReplayEnvelope(nativeOutput, nativeReplayRoute, nativeServerSearchEchoCallIds);
978
1042
  return {
979
1043
  response: {
980
1044
  kind: "pi-ai",
@@ -1152,11 +1216,14 @@ async function* toDshChunks(events, signal, wireMeta, nativeReplayRoute) {
1152
1216
  const reason = successfulFinish(message);
1153
1217
  const replay = reason.kind === "error"
1154
1218
  ? undefined
1155
- : replayState(message, wireMeta?.nativeOutput, wireMeta?.nativeVisibleAdditions, nativeReplayRoute);
1219
+ : replayState(message, wireMeta?.nativeOutput, nativeReplayRoute, wireMeta?.nativeServerSearchEchoCallIds);
1156
1220
  yield {
1157
1221
  type: "finish",
1158
1222
  reason,
1159
- ...(replay === undefined ? {} : { replayState: replay }),
1223
+ ...(replay === undefined && !nativeReplayRoute ? {} : { replayState: {
1224
+ ...replay,
1225
+ response: { ...replay?.response, ...(nativeReplayRoute ? { lcxUsage: { version: 1, inputTokenScope: wireMeta?.inputTokenScope ?? "aggregate" } } : {}) },
1226
+ } }),
1160
1227
  };
1161
1228
  return;
1162
1229
  }
@@ -1185,6 +1252,7 @@ async function* toDshChunks(events, signal, wireMeta, nativeReplayRoute) {
1185
1252
  reason: normalizedFailure.code === "ABORTED"
1186
1253
  ? { kind: "aborted", failure: normalizedFailure }
1187
1254
  : { kind: "error", failure: normalizedFailure },
1255
+ ...(nativeReplayRoute ? { replayState: { response: { lcxUsage: { version: 1, inputTokenScope: "aggregate" } } } } : {}),
1188
1256
  };
1189
1257
  return;
1190
1258
  }
@@ -1212,7 +1280,7 @@ async function* toDshChunks(events, signal, wireMeta, nativeReplayRoute) {
1212
1280
  * @param {ReadonlySet<string>} [options.serverToolTypes]
1213
1281
  * @param {(usage: ServerToolUsage) => void} [options.onServerToolUsage]
1214
1282
  */
1215
- export async function* streamResponsesRequest({ baseURL, provider, model, piModel, body, grammarToolInputProperties, headers, signal, timeoutMs, applyDefaultTimeout = true, streamIdleTimeoutMs, maxAttempts = 1, maxResponseBytes, serverToolTypes, nativeReplayRoute, onServerToolUsage, }) {
1283
+ export async function* streamResponsesRequest({ baseURL, provider, model, piModel, body, grammarToolInputProperties, headers, signal, timeoutMs, applyDefaultTimeout = true, streamIdleTimeoutMs, maxAttempts = 1, maxResponseBytes, serverToolTypes, isServerToolItem, declaredToolNames, nativeReplayRoute, onServerToolUsage, }) {
1216
1284
  const effectiveTimeoutMs = timeoutMs ?? (applyDefaultTimeout ? 300_000 : undefined);
1217
1285
  const deadline = requestDeadline(signal, effectiveTimeoutMs);
1218
1286
  const watchdog = streamIdleWatchdog(deadline.signal, streamIdleTimeoutMs);
@@ -1239,7 +1307,7 @@ export async function* streamResponsesRequest({ baseURL, provider, model, piMode
1239
1307
  await processResponsesStream(validatedResponseEvents(normalizedResponseEvents(responseEvents(response, {
1240
1308
  signal: watchdog.signal,
1241
1309
  maxResponseBytes,
1242
- }), wireMeta, { serverToolTypes, onServerToolUsage })), output, piEvents, piModel, { grammarToolInputProperties });
1310
+ }), wireMeta, { serverToolTypes, isServerToolItem, declaredToolNames, onServerToolUsage })), output, piEvents, piModel, { grammarToolInputProperties });
1243
1311
  if (typeof wireMeta.responseModel === "string" &&
1244
1312
  wireMeta.responseModel.length > 0)
1245
1313
  output.responseModel = wireMeta.responseModel;
package/lib/route.js CHANGED
@@ -3,7 +3,7 @@ import { credentialRef } from "@deepseek-ai/dsh-credentials";
3
3
  import { SessionId } from "@deepseek-ai/dsh-session";
4
4
  import "@deepseek-ai/dsh-settings";
5
5
  import { attributionHeaders, resolveRetryPolicy } from "@deepseek-ai/dsh-llm";
6
- import { getBuiltinModels, getBuiltinProviders, } from "@earendil-works/pi-ai/providers/all";
6
+ import { getBuiltinModels, getBuiltinProviders, } from "./pi-responses-runtime.js";
7
7
  /** @param {unknown} value */
8
8
  function isRecord(value) {
9
9
  return value !== null && typeof value === "object" && !Array.isArray(value);
@@ -102,6 +102,20 @@ function retryAttempts(policy, fallback = 3) {
102
102
  export function settingsValue(ctx, namespace) {
103
103
  return asLlmSettingsSection(ctx?.settings?.get(namespace));
104
104
  }
105
+ /** The DSH 0.1.5 contract exposes deferred provider diagnostics separately from saved settings. */
106
+ function providerDirectoryUsable(ctx, provider) {
107
+ const llm = ctx?.llm;
108
+ const list = llm?.listConfigurableProviders;
109
+ if (typeof list !== "function")
110
+ return true;
111
+ try {
112
+ const entry = list.call(llm).find((candidate) => candidate.provider === provider && candidate.settingsNs === "llm-pi-ai");
113
+ return typeof entry?.error !== "string" || entry.error.trim() === "";
114
+ }
115
+ catch {
116
+ return false;
117
+ }
118
+ }
105
119
  /** @type {Set<keyof ResponsesCompat>} */
106
120
  const RESPONSES_COMPAT_FIELDS = new Set(["supportsDeveloperRole", "sessionAffinityFormat", "supportsStrictMode", "supportsLongCacheRetention", "supportsOpenAIGrammarTools", "supportsAdditionalTools", "supportsToolSearch", "supportsExplicitPromptCacheMode", "supportsMaxOutputTokens"]);
107
121
  /** @param {ResponsesCompat} target @param {unknown} source */
@@ -196,7 +210,7 @@ function isLcxCapabilityRoute(provider, model) {
196
210
  export function resolveResponsesRouteConfig(ctx, options, policy) {
197
211
  const provider = String(options?.provider ?? "");
198
212
  const model = String(options?.model ?? "");
199
- if (!provider.trim() || !/^gpt-/iu.test(model))
213
+ if (!provider.trim() || !/^gpt-/iu.test(model) || !providerDirectoryUsable(ctx, provider))
200
214
  return undefined;
201
215
  const section = settingsValue(ctx, "llm-pi-ai");
202
216
  const profile = section?.providers?.[provider];
@@ -254,7 +268,7 @@ export function resolveResponsesRouteConfig(ctx, options, policy) {
254
268
  export function resolveGrokResponsesRouteConfig(ctx, options, policy) {
255
269
  const provider = String(options?.provider ?? "");
256
270
  const model = String(options?.model ?? "");
257
- if (!provider.trim() || !/^grok/iu.test(model))
271
+ if (!provider.trim() || !/^grok/iu.test(model) || !providerDirectoryUsable(ctx, provider))
258
272
  return undefined;
259
273
  const section = settingsValue(ctx, "llm-pi-ai");
260
274
  const configured = section?.providers?.[provider];
@@ -0,0 +1,86 @@
1
+ export const object = (v) => typeof v === 'object' && v !== null && !Array.isArray(v);
2
+ const count = (v) => typeof v === 'number' && Number.isSafeInteger(v) && v >= 0;
3
+ export const zeroBuckets = () => ({ uncachedInputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0 });
4
+ export const isSearchTool = (name) => name === 'web_search' || name === 'websearch_gpt_advanced';
5
+ export function auxiliaryUsageOf(event, toolName) {
6
+ if (!object(event) || event.type !== 'tool/result' || !object(event.data?.meta))
7
+ return [];
8
+ // Only our owned search results may contribute auxiliary billing.
9
+ // Official ToolMessageSource carries only callId; resolve its name from tool/call.
10
+ if (!isSearchTool(toolName))
11
+ return [];
12
+ const raw = event.data.meta.auxiliaryUsage;
13
+ if (!Array.isArray(raw))
14
+ return [];
15
+ const ids = new Set(), result = [];
16
+ for (const item of raw) {
17
+ if (!object(item) || typeof item.requestId !== 'string' || !item.requestId
18
+ || typeof item.provider !== 'string' || !item.provider || typeof item.model !== 'string' || !item.model
19
+ || !object(item.usage) || ids.has(item.requestId))
20
+ continue;
21
+ const u = item.usage;
22
+ if (![u.inputTokens, u.outputTokens, u.totalTokens, u.cacheReadTokens, u.cacheWriteTokens].every(count)
23
+ || u.inputTokens + u.outputTokens + u.cacheReadTokens + u.cacheWriteTokens !== u.totalTokens)
24
+ continue;
25
+ ids.add(item.requestId);
26
+ result.push(item);
27
+ }
28
+ return result;
29
+ }
30
+ export function addUsage(base, records) {
31
+ if (!records.length)
32
+ return base;
33
+ const next = { ...base };
34
+ for (const { usage: u } of records) {
35
+ next.uncachedInputTokens += u.inputTokens;
36
+ next.outputTokens += u.outputTokens;
37
+ next.cacheReadTokens += u.cacheReadTokens;
38
+ next.cacheWriteTokens += u.cacheWriteTokens;
39
+ }
40
+ if (!Object.values(next).every(count))
41
+ throw new Error('LCX search usage exceeds safe counters');
42
+ return next;
43
+ }
44
+ export function mergeBuckets(base, extra) {
45
+ if (!object(base) || Object.values(extra).every(n => n === 0))
46
+ return base;
47
+ const next = { ...base };
48
+ for (const key of Object.keys(extra)) {
49
+ // Keep unavailable host buckets unavailable rather than inventing zero.
50
+ if (count(base[key]))
51
+ next[key] = base[key] + extra[key];
52
+ }
53
+ return next;
54
+ }
55
+ export function addTurnUsage(base, records) {
56
+ if (!object(base) || !records.length || !count(base.totalTokens))
57
+ return base;
58
+ const next = mergeBuckets(base, addUsage(zeroBuckets(), records));
59
+ next.totalTokens = base.totalTokens + records.reduce((n, r) => n + r.usage.totalTokens, 0);
60
+ // Auxiliary responses do not always disclose a reasoning subset.
61
+ delete next.reasoningTokens;
62
+ if (Array.isArray(base.routes)) {
63
+ const routes = new Map();
64
+ for (const r of [...base.routes, ...records])
65
+ routes.set(`${r.provider}\0${r.model}`, { provider: r.provider, model: r.model });
66
+ next.routes = [...routes.values()];
67
+ }
68
+ return next;
69
+ }
70
+ /** The metadata is presentation/accounting state; it never changes request messages. */
71
+ export function aggregateContextOf(event) {
72
+ if (!object(event) || !['assistant/message', 'assistant/attempt'].includes(event.type) || !object(event.data))
73
+ return;
74
+ const chunks = Array.isArray(event.data.stream) ? event.data.stream.filter((e) => e.type === 'chunk').map((e) => e.chunk) : [];
75
+ const sample = event.data.usage ?? chunks.findLast((c) => c?.type === 'usage')?.usage;
76
+ if (!object(sample))
77
+ return;
78
+ const replay = chunks.findLast((c) => c?.type === 'finish')?.replayState;
79
+ const mark = replay?.response?.lcxUsage;
80
+ if (mark?.version === 1 && ['request', 'aggregate'].includes(mark.inputTokenScope))
81
+ return mark.inputTokenScope === 'aggregate';
82
+ // Read the already-installed local candidate without rewriting its logs.
83
+ if (sample.inputTokenScope === 'aggregate' || sample.inputTokenScope === 'request')
84
+ return sample.inputTokenScope === 'aggregate';
85
+ return replay?.grokNative?.kind === 'xai-responses-native-search' && replay.grokNative.version === 3;
86
+ }
@@ -0,0 +1,86 @@
1
+ import { z } from 'zod';
2
+ import { SessionSeq } from '@deepseek-ai/dsh-session';
3
+ import { addUsage, aggregateContextOf, auxiliaryUsageOf, isSearchTool, zeroBuckets } from './search-accounting.js';
4
+ const n = z.number().int().nonnegative();
5
+ const schema = z.object({ auxiliary: z.object({ uncachedInputTokens: n, outputTokens: n, cacheReadTokens: n, cacheWriteTokens: n }).strict(), aggregateContext: z.boolean() }).strict();
6
+ /** Separate, durable extension: the host's own usage projection remains primary-call billing. */
7
+ export const searchUsageProjection = {
8
+ key: 'lcxSearchUsage', stateVersion: 2, stateSchema: schema.extend({ pending: z.record(z.string(), z.string()) }),
9
+ init: () => ({ auxiliary: zeroBuckets(), aggregateContext: false, pending: {} }),
10
+ apply(state, event) {
11
+ if (event.type === 'tool/call' && isSearchTool(event.data.name))
12
+ return { ...state, pending: { ...state.pending, [event.data.callId]: event.data.name } };
13
+ const callId = event.type === 'tool/result' ? event.data.message.source.callId : undefined;
14
+ const auxiliary = addUsage(state.auxiliary, auxiliaryUsageOf(event, callId ? state.pending[callId] : undefined));
15
+ const aggregateContext = aggregateContextOf(event) ?? state.aggregateContext;
16
+ if (callId && Object.hasOwn(state.pending, callId)) {
17
+ const pending = { ...state.pending };
18
+ delete pending[callId];
19
+ return { auxiliary, aggregateContext, pending };
20
+ }
21
+ if (event.type === 'turn/end' && Object.keys(state.pending).length)
22
+ return { auxiliary, aggregateContext, pending: {} };
23
+ return auxiliary === state.auxiliary && aggregateContext === state.aggregateContext ? state : { ...state, auxiliary, aggregateContext };
24
+ },
25
+ wire: { viewSchema: schema, view: ({ auxiliary, aggregateContext }) => ({ auxiliary, aggregateContext }) },
26
+ };
27
+ const installed = new WeakMap();
28
+ /** Own a reversible adapter on the public measure method, never a DSH file or private fold. */
29
+ export function installSearchMeasurement(meter) {
30
+ const existing = installed.get(meter);
31
+ if (existing) {
32
+ existing.refs++;
33
+ let active = true;
34
+ return () => { if (active) {
35
+ active = false;
36
+ release(meter);
37
+ } };
38
+ }
39
+ const original = meter.measure, descriptor = Object.getOwnPropertyDescriptor(meter, 'measure');
40
+ const cursors = new WeakMap();
41
+ function measure(session, header) {
42
+ const value = original.call(this, session, header);
43
+ let state = cursors.get(session) ?? { seq: 0, aggregate: false };
44
+ while (state.seq < session.seq) {
45
+ const e = session.eventAt(SessionSeq(state.seq++));
46
+ if (e?.type === 'assistant/message')
47
+ state.aggregate = aggregateContextOf(e) ?? false;
48
+ }
49
+ cursors.set(session, state);
50
+ if (!state.aggregate || value.baseline.kind !== 'usage')
51
+ return value;
52
+ // The official measure already prices retained images/files and surface replacements.
53
+ // Only discard its unsuitable aggregate anchor; keep its current surface and node prices.
54
+ const tools = (header ?? session.requestHeader())?.tools;
55
+ const toolTokens = !tools?.length ? 0 : Math.ceil(JSON.stringify(tools).length / 4) + 4;
56
+ const tokens = value.surfaceTokens + toolTokens;
57
+ return Object.freeze({ ...value, baseline: Object.freeze({ kind: 'estimated', tokens }), surfaceDeltaTokens: 0, totalTokens: tokens });
58
+ }
59
+ Object.defineProperty(meter, 'measure', { configurable: true, writable: true, value: measure });
60
+ installed.set(meter, { refs: 1, release() {
61
+ if (Object.getOwnPropertyDescriptor(meter, 'measure')?.value !== measure)
62
+ return;
63
+ if (descriptor)
64
+ Object.defineProperty(meter, 'measure', descriptor);
65
+ else
66
+ delete meter.measure;
67
+ } });
68
+ let active = true;
69
+ return () => { if (active) {
70
+ active = false;
71
+ release(meter);
72
+ } };
73
+ }
74
+ function release(meter) {
75
+ const record = installed.get(meter);
76
+ if (record && --record.refs === 0) {
77
+ record.release();
78
+ installed.delete(meter);
79
+ }
80
+ }
81
+ export function installSearchUsage(ctx) {
82
+ ctx.inject(['sessionProjections'], c => { c.sessionProjections.register(searchUsageProjection); });
83
+ ctx.inject(['tokenMeter'], c => {
84
+ c.effect(() => installSearchMeasurement(c.tokenMeter), 'lcx search context measurement');
85
+ });
86
+ }