@oh-my-pi/pi-ai 18.4.0 → 18.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +12 -0
  3. package/dist/types/auth/oauth.d.ts +2 -2
  4. package/dist/types/auth/refresh.d.ts +3 -3
  5. package/dist/types/auth/store.d.ts +4 -2
  6. package/dist/types/auth/types.d.ts +20 -3
  7. package/dist/types/auth/usage-cache.d.ts +9 -2
  8. package/dist/types/auth/usage.d.ts +6 -5
  9. package/dist/types/auth-broker/client.d.ts +2 -2
  10. package/dist/types/auth-broker/remote-store.d.ts +2 -2
  11. package/dist/types/error/flags.d.ts +9 -2
  12. package/dist/types/error/rate-limit.d.ts +3 -2
  13. package/dist/types/providers/cursor/exec-modern.d.ts +1 -1
  14. package/dist/types/providers/cursor-pi-args.d.ts +13 -18
  15. package/dist/types/stream.d.ts +7 -0
  16. package/dist/types/usage/cursor.d.ts +3 -1
  17. package/dist/types/usage/zai.d.ts +2 -0
  18. package/dist/types/usage.d.ts +6 -0
  19. package/package.json +6 -6
  20. package/src/auth/cascade.ts +1 -0
  21. package/src/auth/oauth.ts +4 -3
  22. package/src/auth/refresh.ts +105 -16
  23. package/src/auth/resets.ts +6 -2
  24. package/src/auth/select.ts +11 -1
  25. package/src/auth/store.ts +4 -0
  26. package/src/auth/types.ts +22 -2
  27. package/src/auth/usage-cache.ts +12 -8
  28. package/src/auth/usage.ts +45 -19
  29. package/src/auth-broker/client.ts +8 -3
  30. package/src/auth-broker/remote-store.ts +4 -2
  31. package/src/auth-broker/server.ts +7 -1
  32. package/src/auth-gateway/dispatch.ts +1 -0
  33. package/src/auth-retry.ts +5 -1
  34. package/src/error/flags.ts +10 -3
  35. package/src/error/rate-limit.ts +23 -3
  36. package/src/providers/anthropic.ts +62 -53
  37. package/src/providers/cowork-fetch.ts +45 -3
  38. package/src/providers/cursor/exec-modern.ts +1 -1
  39. package/src/providers/cursor-pi-args.ts +13 -18
  40. package/src/providers/cursor.ts +179 -57
  41. package/src/providers/devin.ts +43 -13
  42. package/src/providers/openai-completions.ts +122 -30
  43. package/src/stream.ts +77 -16
  44. package/src/usage/claude.ts +5 -0
  45. package/src/usage/cursor.ts +43 -1
  46. package/src/usage/devin.ts +8 -7
  47. package/src/usage/registry.ts +2 -1
  48. package/src/usage/zai.ts +20 -0
  49. package/src/usage.ts +6 -0
@@ -59,7 +59,6 @@ import {
59
59
  GrepContentResultSchema,
60
60
  GrepCountResultSchema,
61
61
  GrepErrorSchema,
62
- type GrepFileCount,
63
62
  GrepFileCountSchema,
64
63
  GrepFileMatchSchema,
65
64
  GrepFilesResultSchema,
@@ -99,6 +98,7 @@ import {
99
98
  McpToolResultSchema,
100
99
  ModelDetailsSchema,
101
100
  ReadErrorSchema,
101
+ ReadFileNotFoundSchema,
102
102
  ReadMcpResourceErrorSchema,
103
103
  type ReadMcpResourceExecResult,
104
104
  ReadMcpResourceExecResultSchema,
@@ -233,7 +233,8 @@ import {
233
233
  buildPiWriteError,
234
234
  buildPiWriteRejected,
235
235
  buildPiWriteResult,
236
- cursorEditOwnedReadPath,
236
+ cursorExecReadPath,
237
+ cursorRawReadPath,
237
238
  omitUndefinedArgs,
238
239
  piEscapeRegexLiteral,
239
240
  piGrepSkip,
@@ -1654,39 +1655,93 @@ async function handleExecServerMessage(
1654
1655
  const args = execMsg.message.value;
1655
1656
  if (!args.toolCallId) args.toolCallId = crypto.randomUUID();
1656
1657
  const editOwned = isEditOwnedToolCallId(state, output, args.toolCallId);
1657
- // Native StrReplace materializes by reading then writing the same
1658
- // toolCallId. The server treats the read result as file bytes, so a
1659
- // hashline-formatted native read would be written back as markup.
1660
- // Force `:raw` and skip the extra transcript block — the edit card
1661
- // already owns this id.
1662
- const composed = editOwned ? cursorEditOwnedReadPath(args.path, args.offset, args.limit) : args.path;
1663
- const handlerArgs = editOwned
1664
- ? composed === null
1658
+ // Cursor numbers raw content itself; StrReplace also writes it back.
1659
+ // Neither frame may receive hashline gutters or summarized bodies.
1660
+ // Edit-owned reads omit the extra transcript block owned by the edit card.
1661
+ // The handler needs the original negative offset to resolve it against
1662
+ // the file length; positive windows can be composed before dispatch.
1663
+ const negativeOffset = args.offset !== undefined && args.offset < 0;
1664
+ const composed = negativeOffset
1665
+ ? cursorRawReadPath(args.path)
1666
+ : cursorExecReadPath(args.path, args.offset, args.limit);
1667
+ const handlerArgs =
1668
+ composed === null
1665
1669
  ? args
1666
- : { ...args, path: composed, offset: undefined, limit: undefined }
1667
- : args;
1670
+ : {
1671
+ ...args,
1672
+ path: composed,
1673
+ offset: negativeOffset ? args.offset : undefined,
1674
+ limit: negativeOffset ? args.limit : undefined,
1675
+ };
1668
1676
  if (!editOwned) {
1669
- // The same composed selector the bridge executes: showing a bare path
1670
- // for a ranged read makes the returned slice look like the whole
1671
- // file in every rebuilt transcript.
1672
- synthesizeCursorExecToolCall(output, stream, state, args.toolCallId, "read", {
1673
- path: piReadDisplayPath(args.path, args.offset, args.limit),
1674
- });
1677
+ synthesizeCursorExecToolCall(
1678
+ output,
1679
+ stream,
1680
+ state,
1681
+ args.toolCallId,
1682
+ "read",
1683
+ negativeOffset
1684
+ ? { path: composed, offset: args.offset, limit: args.limit }
1685
+ : { path: piReadDisplayPath(args.path, args.offset, args.limit) },
1686
+ );
1675
1687
  }
1676
- const { execResult } = await resolveExecHandler(
1688
+ const { execResult: readResult, toolResult } = await resolveExecHandler(
1677
1689
  handlerArgs,
1678
1690
  execHandlers?.read?.bind(execHandlers),
1679
1691
  editOwned ? undefined : onToolResult,
1680
- toolResult =>
1692
+ result =>
1681
1693
  buildReadResultFromToolResult(
1682
1694
  args.path,
1683
- toolResult,
1695
+ result,
1684
1696
  args.offset !== undefined || args.limit !== undefined || piReadPathHasRange(args.path),
1685
1697
  ),
1686
1698
  reason => buildReadRejectedResult(args.path, reason),
1687
1699
  error => buildReadErrorResult(args.path, error),
1688
1700
  editOwned ? null : { toolCallId: args.toolCallId, toolName: "read" },
1689
1701
  );
1702
+ let execResult = readResult;
1703
+ // StrReplace writes the returned bytes back. A truncated raw read would
1704
+ // make every replacement below the read limit invisible to the server.
1705
+ if (
1706
+ editOwned &&
1707
+ args.offset === undefined &&
1708
+ args.limit === undefined &&
1709
+ !piReadPathHasRange(args.path) &&
1710
+ toolResult &&
1711
+ !toolResult.isError &&
1712
+ toolResultWasTruncated(toolResult)
1713
+ ) {
1714
+ const details = toolResult.details;
1715
+ const source = details && typeof details === "object" && "meta" in details && details.meta;
1716
+ const path = source && typeof source === "object" && "source" in source && source.source;
1717
+ if (
1718
+ path &&
1719
+ typeof path === "object" &&
1720
+ "type" in path &&
1721
+ path.type === "path" &&
1722
+ "value" in path &&
1723
+ typeof path.value === "string"
1724
+ ) {
1725
+ try {
1726
+ const content = await readEditMaterialization(path.value);
1727
+ execResult =
1728
+ content === null
1729
+ ? buildReadErrorResult(
1730
+ args.path,
1731
+ `File exceeds ${EDIT_MATERIALIZATION_MAX_BYTES} bytes; StrReplace cannot load it whole. Use a line-range read and a targeted edit instead.`,
1732
+ )
1733
+ : buildReadResultFromToolResult(args.path, {
1734
+ ...toolResult,
1735
+ content: [{ type: "text", text: content }],
1736
+ details: { fileSize: Buffer.byteLength(content, "utf8") },
1737
+ });
1738
+ } catch (error) {
1739
+ execResult = buildReadErrorResult(args.path, error instanceof Error ? error.message : String(error));
1740
+ }
1741
+ } else {
1742
+ execResult = buildReadErrorResult(args.path, "Unable to read the complete file for StrReplace");
1743
+ }
1744
+ }
1690
1745
  sendExecClientMessage(h2Request, execMsg, "readResult", execResult);
1691
1746
  return;
1692
1747
  }
@@ -2842,9 +2897,50 @@ function readFileSizeFromDetails(toolResult: ToolResultMessage): number | undefi
2842
2897
  return typeof fileSize === "number" && Number.isSafeInteger(fileSize) && fileSize >= 0 ? fileSize : undefined;
2843
2898
  }
2844
2899
 
2900
+ /**
2901
+ * Largest file a native StrReplace materializes whole; matches the local
2902
+ * `read` tool's whole-file snapshot cap (`SNAPSHOT_MAX_BYTES`).
2903
+ */
2904
+ const EDIT_MATERIALIZATION_MAX_BYTES = 4 * 1024 * 1024;
2905
+
2906
+ /**
2907
+ * Read a file for native StrReplace materialization, or `null` when it exceeds
2908
+ * {@link EDIT_MATERIALIZATION_MAX_BYTES}. The buffer is sized from the handle's
2909
+ * stat and reading stops one byte past it, so a file growing during the read
2910
+ * cannot exhaust memory.
2911
+ *
2912
+ * Throws when the file grew during the read while still under the cap.
2913
+ */
2914
+ async function readEditMaterialization(filePath: string): Promise<string | null> {
2915
+ const handle = await fs.open(filePath, "r");
2916
+ try {
2917
+ const { size } = await handle.stat();
2918
+ if (size > EDIT_MATERIALIZATION_MAX_BYTES) return null;
2919
+ const buffer = Buffer.allocUnsafe(size + 1);
2920
+ let length = 0;
2921
+ while (length < buffer.length) {
2922
+ const { bytesRead } = await handle.read(buffer, length, buffer.length - length, length);
2923
+ if (bytesRead === 0) break;
2924
+ length += bytesRead;
2925
+ }
2926
+ if (length > size) {
2927
+ if (size === EDIT_MATERIALIZATION_MAX_BYTES) return null;
2928
+ throw new Error(`File changed while reading: ${filePath}`);
2929
+ }
2930
+ return buffer.toString("utf8", 0, length);
2931
+ } finally {
2932
+ await handle.close();
2933
+ }
2934
+ }
2935
+
2845
2936
  function buildReadResultFromToolResult(path: string, toolResult: ToolResultMessage, rangeApplied = false) {
2846
2937
  const text = toolResultToText(toolResult);
2847
2938
  if (toolResult.isError) {
2939
+ if (/^Path '.*' not found$/.test(text)) {
2940
+ return create(ReadResultSchema, {
2941
+ result: { case: "fileNotFound", value: create(ReadFileNotFoundSchema, { path }) },
2942
+ });
2943
+ }
2848
2944
  return buildReadErrorResult(path, text || "Read failed");
2849
2945
  }
2850
2946
  // Counting the payload is only the file's length when the payload is the
@@ -2942,7 +3038,7 @@ function buildDeleteResultFromToolResult(path: string, toolResult: ToolResultMes
2942
3038
  value: create(DeleteSuccessSchema, {
2943
3039
  path,
2944
3040
  deletedFile: path,
2945
- fileSize: BigInt(0),
3041
+ fileSize: BigInt(readFileSizeFromDetails(toolResult) ?? 0),
2946
3042
  prevContent: "",
2947
3043
  }),
2948
3044
  },
@@ -2973,7 +3069,14 @@ function buildShellResultFromToolResult(
2973
3069
  ) {
2974
3070
  const output = toolResultToText(toolResult);
2975
3071
  if (toolResult.isError) {
2976
- return buildShellFailureResult(args.command, args.workingDirectory, output || "Shell failed");
3072
+ const details = toolResult.details;
3073
+ const code = details && typeof details === "object" && "exitCode" in details ? details.exitCode : undefined;
3074
+ return buildShellFailureResult(
3075
+ args.command,
3076
+ args.workingDirectory,
3077
+ output || "Shell failed",
3078
+ typeof code === "number" && Number.isInteger(code) ? code : 1,
3079
+ );
2977
3080
  }
2978
3081
  return create(ShellResultSchema, {
2979
3082
  result: {
@@ -2991,14 +3094,14 @@ function buildShellResultFromToolResult(
2991
3094
  });
2992
3095
  }
2993
3096
 
2994
- function buildShellFailureResult(command: string, workingDirectory: string, error: string) {
3097
+ function buildShellFailureResult(command: string, workingDirectory: string, error: string, exitCode = 1) {
2995
3098
  return create(ShellResultSchema, {
2996
3099
  result: {
2997
3100
  case: "failure",
2998
3101
  value: create(ShellFailureSchema, {
2999
3102
  command,
3000
3103
  workingDirectory,
3001
- exitCode: 1,
3104
+ exitCode,
3002
3105
  signal: "",
3003
3106
  stdout: "",
3004
3107
  stderr: error,
@@ -3101,16 +3204,16 @@ function buildGrepResultFromToolResult(
3101
3204
 
3102
3205
  const outputMode = args.outputMode || "content";
3103
3206
  const clientTruncated = toolResultDetailBoolean(toolResult, "truncated");
3104
- const lines = text
3105
- .split("\n")
3106
- .map(line => line.trimEnd())
3107
- .filter(line => line.length > 0 && !line.startsWith("[") && !line.toLowerCase().startsWith("no matches"));
3207
+ const details = toolResult.details;
3208
+ const files =
3209
+ details && typeof details === "object" && "files" in details && Array.isArray(details.files)
3210
+ ? details.files.filter((file): file is string => typeof file === "string")
3211
+ : [];
3108
3212
 
3109
3213
  const workspaceKey = args.path || ".";
3110
3214
  let unionResult: GrepUnionResult;
3111
3215
 
3112
3216
  if (outputMode === "files_with_matches") {
3113
- const files = lines;
3114
3217
  unionResult = create(GrepUnionResultSchema, {
3115
3218
  result: {
3116
3219
  case: "files",
@@ -3128,20 +3231,21 @@ function buildGrepResultFromToolResult(
3128
3231
  },
3129
3232
  });
3130
3233
  } else if (outputMode === "count") {
3131
- const counts = lines
3132
- .map(line => {
3133
- const separatorIndex = line.lastIndexOf(":");
3134
- if (separatorIndex === -1) {
3135
- return null;
3136
- }
3137
- const file = line.slice(0, separatorIndex);
3138
- const count = Number.parseInt(line.slice(separatorIndex + 1), 10);
3139
- if (!file || Number.isNaN(count)) {
3140
- return null;
3141
- }
3142
- return create(GrepFileCountSchema, { file, count });
3143
- })
3144
- .filter((entry): entry is GrepFileCount => entry !== null);
3234
+ const fileMatches =
3235
+ details && typeof details === "object" && "fileMatches" in details && Array.isArray(details.fileMatches)
3236
+ ? details.fileMatches
3237
+ : [];
3238
+ const counts = fileMatches
3239
+ .filter(
3240
+ (entry): entry is { path: string; count: number } =>
3241
+ entry !== null &&
3242
+ typeof entry === "object" &&
3243
+ "path" in entry &&
3244
+ typeof entry.path === "string" &&
3245
+ "count" in entry &&
3246
+ typeof entry.count === "number",
3247
+ )
3248
+ .map(({ path, count }) => create(GrepFileCountSchema, { file: path, count }));
3145
3249
  const totalMatches = counts.reduce((sum, entry) => sum + entry.count, 0);
3146
3250
  unionResult = create(GrepUnionResultSchema, {
3147
3251
  result: {
@@ -3159,22 +3263,40 @@ function buildGrepResultFromToolResult(
3159
3263
  } else {
3160
3264
  const matchMap = new Map<string, Array<{ line: number; content: string; isContextLine: boolean }>>();
3161
3265
  let totalMatchedLines = 0;
3162
-
3163
- for (const line of lines) {
3164
- const matchLine = line.match(/^(.+?):(\d+):\s?(.*)$/);
3165
- const contextLine = line.match(/^(.+?)-(\d+)-\s?(.*)$/);
3166
- const match = matchLine ?? contextLine;
3167
- if (!match) {
3266
+ const directories = new Map<number, string>();
3267
+ let currentFile = files.length === 1 ? files[0] : undefined;
3268
+
3269
+ for (const line of text.split("\n")) {
3270
+ const header = /^(#+) (.+)$/.exec(line);
3271
+ if (header) {
3272
+ const depth = header[1]!.length;
3273
+ const name = header[2]!;
3274
+ for (const level of directories.keys()) {
3275
+ if (level >= depth) directories.delete(level);
3276
+ }
3277
+ const parent = directories.get(depth - 1);
3278
+ const candidate = parent ? `${parent}/${name}` : name;
3279
+ if (name.endsWith("/")) {
3280
+ directories.set(depth, candidate.slice(0, -1));
3281
+ currentFile = undefined;
3282
+ } else {
3283
+ currentFile = files.includes(candidate) ? candidate : candidate.replace(/#[0-9a-f]{4,}$/i, "");
3284
+ }
3168
3285
  continue;
3169
3286
  }
3170
- const [, file, lineNumber, content] = match;
3171
- const isContextLine = Boolean(contextLine);
3172
- const list = matchMap.get(file) ?? [];
3173
- list.push({ line: Number(lineNumber), content, isContextLine });
3174
- matchMap.set(file, list);
3175
- if (!isContextLine) {
3176
- totalMatchedLines += 1;
3287
+ const singleHeader = /^\[(.+)#[0-9a-f]{4,}\]$/i.exec(line);
3288
+ if (singleHeader) {
3289
+ currentFile = files[0] ?? singleHeader[1]!;
3290
+ continue;
3177
3291
  }
3292
+ const match = /^([* ])(\d+)[:|](.*)$/.exec(line);
3293
+ if (!match || !currentFile) continue;
3294
+ const [, marker, lineNumber, content] = match;
3295
+ const isContextLine = marker === " ";
3296
+ const list = matchMap.get(currentFile) ?? [];
3297
+ list.push({ line: Number(lineNumber), content, isContextLine });
3298
+ matchMap.set(currentFile, list);
3299
+ if (!isContextLine) totalMatchedLines++;
3178
3300
  }
3179
3301
 
3180
3302
  const matches = Array.from(matchMap.entries()).map(([file, matches]) =>
@@ -21,6 +21,7 @@ import {
21
21
  GetUserJwtResponseSchema,
22
22
  type ImageData,
23
23
  ImageDataSchema,
24
+ type Metadata,
24
25
  MetadataSchema,
25
26
  type ModelAssignment,
26
27
  PromptCacheOptionsSchema,
@@ -28,7 +29,7 @@ import {
28
29
  } from "@oh-my-pi/pi-catalog/discovery/devin-proto";
29
30
  import { create, fromBinary, toBinary } from "@oh-my-pi/pi-catalog/discovery/protobuf";
30
31
  import { calculateCost } from "@oh-my-pi/pi-catalog/models";
31
- import { DEVIN_DEFAULT_BASE_URL, devinCliMetadata } from "@oh-my-pi/pi-catalog/wire/devin";
32
+ import { DEVIN_DEFAULT_BASE_URL, devinCliMetadata, devinWireMetadata } from "@oh-my-pi/pi-catalog/wire/devin";
32
33
  import { decodeDevinUnaryMessage } from "@oh-my-pi/pi-catalog/wire/devin-proto";
33
34
  import { isRecord, logger, parseStreamingJson, parseStreamingJsonThrottled, sanitizeText } from "@oh-my-pi/pi-utils";
34
35
  import * as AIError from "../error";
@@ -221,7 +222,7 @@ export const streamDevin: StreamFunction<"devin-agent"> = (
221
222
  const auth = await fetchDevinAuthMetadata(options?.apiKey, baseUrl, fetchImpl, options?.signal);
222
223
  const chatBaseUrl = auth.baseUrl ?? baseUrl;
223
224
  const turn: DevinTurn = {
224
- apiKey: options?.apiKey,
225
+ apiKey: auth.apiKey,
225
226
  userJwt: auth.userJwt,
226
227
  cascadeId: options?.conversationId ?? options?.sessionId ?? crypto.randomUUID(),
227
228
  messages: transformMessages(context.messages, model),
@@ -526,7 +527,7 @@ export const streamDevin: StreamFunction<"devin-agent"> = (
526
527
 
527
528
  /** Per-turn wire state shared by `AssignModel` and `GetChatMessage`. */
528
529
  interface DevinTurn {
529
- apiKey: string | undefined;
530
+ apiKey: string;
530
531
  userJwt: string;
531
532
  /** Cascade thread id; assignment and chat must agree on it or the JWT is rejected. */
532
533
  cascadeId: string;
@@ -534,13 +535,18 @@ interface DevinTurn {
534
535
  messages: Message[];
535
536
  }
536
537
 
537
- async function fetchDevinAuthMetadata(
538
- apiKey: string | undefined,
538
+ interface DevinAuthAttempt {
539
+ response: Response;
540
+ payload: Uint8Array;
541
+ }
542
+
543
+ async function requestDevinAuth(
544
+ metadata: Metadata,
539
545
  baseUrl: string,
540
546
  fetchImpl: NonNullable<StreamOptions["fetch"]>,
541
547
  signal: AbortSignal | undefined,
542
- ): Promise<{ userJwt: string; baseUrl?: string }> {
543
- const request = create(GetUserJwtRequestSchema, { metadata: create(MetadataSchema, devinCliMetadata(apiKey)) });
548
+ ): Promise<DevinAuthAttempt> {
549
+ const request = create(GetUserJwtRequestSchema, { metadata });
544
550
  const response = await fetchImpl(`${baseUrl}${DEVIN_AUTH_PATH}`, {
545
551
  method: "POST",
546
552
  headers: {
@@ -551,9 +557,29 @@ async function fetchDevinAuthMetadata(
551
557
  body: toBinary(GetUserJwtRequestSchema, request),
552
558
  signal,
553
559
  });
554
- const payload = new Uint8Array(await response.arrayBuffer());
555
- if (!response.ok) throw createDevinHttpError("auth", response, payload);
556
- const decoded = decodeDevinUnaryMessage(GetUserJwtResponseSchema, payload);
560
+ return { response, payload: new Uint8Array(await response.arrayBuffer()) };
561
+ }
562
+
563
+ async function fetchDevinAuthMetadata(
564
+ apiKey: string | undefined,
565
+ baseUrl: string,
566
+ fetchImpl: NonNullable<StreamOptions["fetch"]>,
567
+ signal: AbortSignal | undefined,
568
+ ): Promise<{ userJwt: string; apiKey: string; baseUrl?: string }> {
569
+ const sessionMetadata = create(MetadataSchema, devinCliMetadata(apiKey));
570
+ let wireApiKey = sessionMetadata.apiKey;
571
+ let attempt = await requestDevinAuth(sessionMetadata, baseUrl, fetchImpl, signal);
572
+
573
+ if (attempt.response.status === 401) {
574
+ const apiKeyMetadata = create(MetadataSchema, devinWireMetadata(apiKey));
575
+ if (apiKeyMetadata.apiKey && apiKeyMetadata.apiKey !== sessionMetadata.apiKey) {
576
+ attempt = await requestDevinAuth(apiKeyMetadata, baseUrl, fetchImpl, signal);
577
+ wireApiKey = apiKeyMetadata.apiKey;
578
+ }
579
+ }
580
+
581
+ if (!attempt.response.ok) throw createDevinHttpError("auth", attempt.response, attempt.payload);
582
+ const decoded = decodeDevinUnaryMessage(GetUserJwtResponseSchema, attempt.payload);
557
583
  if (!decoded?.userJwt) {
558
584
  throw new AIError.ProviderResponseError("Devin auth error: GetUserJwt returned an empty user JWT", {
559
585
  provider: "devin",
@@ -561,7 +587,11 @@ async function fetchDevinAuthMetadata(
561
587
  });
562
588
  }
563
589
  const customBaseUrl = decoded.customApiServerUrl.trim();
564
- return { userJwt: decoded.userJwt, ...(customBaseUrl ? { baseUrl: customBaseUrl.replace(/\/+$/, "") } : undefined) };
590
+ return {
591
+ userJwt: decoded.userJwt,
592
+ apiKey: wireApiKey,
593
+ ...(customBaseUrl ? { baseUrl: customBaseUrl.replace(/\/+$/, "") } : undefined),
594
+ };
565
595
  }
566
596
 
567
597
  /**
@@ -578,7 +608,7 @@ async function assignDevinModel(
578
608
  signal: AbortSignal | undefined,
579
609
  ): Promise<ModelAssignment> {
580
610
  const request = create(AssignModelRequestSchema, {
581
- metadata: create(MetadataSchema, devinCliMetadata(turn.apiKey)),
611
+ metadata: create(MetadataSchema, devinWireMetadata(turn.apiKey)),
582
612
  modelRouterUid: model.requestModelId ?? model.id,
583
613
  cascadeId: turn.cascadeId,
584
614
  chatMessagePrompt: buildRouterPrompt(turn.messages),
@@ -660,7 +690,7 @@ function buildDevinChatRequest(
660
690
  });
661
691
  });
662
692
  return create(GetChatMessageRequestSchema, {
663
- metadata: create(MetadataSchema, devinCliMetadata(turn.apiKey, turn.userJwt)),
693
+ metadata: create(MetadataSchema, devinWireMetadata(turn.apiKey, turn.userJwt)),
664
694
  prompt: normalizeSystemPrompts(context.systemPrompt).join("\n\n"),
665
695
  chatMessagePrompts: buildChatMessagePrompts(turn.messages, turn.cascadeId, model),
666
696
  chatModelUid,
@@ -230,9 +230,60 @@ function mergeStoredGeminiSignature(existing: string | undefined, update: Stored
230
230
  return JSON.stringify(merged);
231
231
  }
232
232
 
233
+ /**
234
+ * LiteLLM's wire shape for Anthropic thinking: streamed as
235
+ * `delta.thinking_blocks` (mirrored under
236
+ * `delta.provider_specific_fields.thinking_blocks`) and required back verbatim,
237
+ * in order, on the assistant turn that carries `tool_calls` — otherwise the
238
+ * upstream rejects the turn or LiteLLM silently drops its thinking.
239
+ * https://docs.litellm.ai/docs/reasoning_content#tool-calling-with-thinking
240
+ */
241
+ type LiteLLMThinkingBlock =
242
+ | { type: "thinking"; thinking: string; signature: string }
243
+ | { type: "redacted_thinking"; data: string };
244
+
245
+ function getLiteLLMThinkingBlocksDelta(delta: object): unknown[] | undefined {
246
+ const direct = Reflect.get(delta, "thinking_blocks");
247
+ if (Array.isArray(direct) && direct.length > 0) return direct;
248
+ const providerFields = Reflect.get(delta, "provider_specific_fields");
249
+ if (typeof providerFields !== "object" || providerFields === null) return undefined;
250
+ const mirrored = Reflect.get(providerFields, "thinking_blocks");
251
+ return Array.isArray(mirrored) && mirrored.length > 0 ? mirrored : undefined;
252
+ }
253
+
254
+ /**
255
+ * Rebuilds LiteLLM `thinking_blocks` from a stored assistant turn. Thinking
256
+ * blocks parsed from `thinking_blocks` keep the raw Anthropic signature as
257
+ * `thinkingSignature`; blocks parsed from `reasoning_content`-style fields
258
+ * carry the field name instead and are skipped. `transformMessages` strips
259
+ * signatures and redacted blocks on cross-model replays, so only the issuing
260
+ * model ever receives them back.
261
+ */
262
+ function encodeLiteLLMThinkingBlocks(content: AssistantMessage["content"]): LiteLLMThinkingBlock[] {
263
+ const blocks: LiteLLMThinkingBlock[] = [];
264
+ for (const block of content) {
265
+ if (block.type === "thinking") {
266
+ const signature = block.thinkingSignature;
267
+ if (
268
+ !signature ||
269
+ signature === "reasoning_content" ||
270
+ signature === "reasoning" ||
271
+ signature === "reasoning_text"
272
+ ) {
273
+ continue;
274
+ }
275
+ blocks.push({ type: "thinking", thinking: block.thinking, signature });
276
+ } else if (block.type === "redactedThinking") {
277
+ blocks.push({ type: "redacted_thinking", data: block.data });
278
+ }
279
+ }
280
+ return blocks;
281
+ }
282
+
233
283
  type OpenAICompletionsAssistantMessageParam = ChatCompletionAssistantMessageParam &
234
284
  Partial<Record<OpenAICompletionsReasoningField | GeminiMessageThoughtSignatureField, string>> & {
235
285
  reasoning_details?: unknown[];
286
+ thinking_blocks?: LiteLLMThinkingBlock[];
236
287
  };
237
288
 
238
289
  type OpenAICompletionsToolMessageParam = ChatCompletionToolMessageParam & {
@@ -1059,37 +1110,68 @@ const streamOpenAICompletionsOnce = (
1059
1110
  partial: message,
1060
1111
  });
1061
1112
  };
1062
- const appendThinking = (
1063
- message: AssistantMessage,
1064
- eventStream: AssistantMessageEventStream,
1065
- thinking: string,
1066
- signature?: string,
1067
- ): void => {
1068
- if (
1069
- currentBlock?.type !== "thinking" ||
1070
- (signature !== undefined && currentBlock.thinkingSignature !== signature)
1071
- ) {
1072
- // Same as appendText: leave toolCall blocks pending so index-only
1073
- // continuation deltas can still find them.
1113
+ const openThinkingBlock = (signature?: string): ThinkingContent => {
1114
+ // Same as appendText: leave toolCall blocks pending so index-only
1115
+ // continuation deltas can still find them.
1116
+ if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1117
+ const block: ThinkingContent = { type: "thinking", thinking: "", thinkingSignature: signature };
1118
+ currentBlock = block;
1119
+ output.content.push(block);
1120
+ stream.push({ type: "thinking_start", contentIndex: blockIndex(block), partial: output });
1121
+ return block;
1122
+ };
1123
+ const appendThinking = (thinking: string, signature?: string): void => {
1124
+ const block =
1125
+ currentBlock?.type === "thinking" &&
1126
+ (signature === undefined || currentBlock.thinkingSignature === signature)
1127
+ ? currentBlock
1128
+ : openThinkingBlock(signature);
1129
+ if (signature !== undefined && !block.thinkingSignature) {
1130
+ block.thinkingSignature = signature;
1131
+ }
1132
+ block.thinking += thinking;
1133
+ stream.push({ type: "thinking_delta", contentIndex: blockIndex(block), delta: thinking, partial: output });
1134
+ };
1135
+ // LiteLLM `thinking_blocks` entries each carry a text fragment, a
1136
+ // signature fragment, or a whole redacted block. A signature seals its
1137
+ // block, so text arriving after one opens the next block; signature
1138
+ // fragments concatenate like Anthropic `signature_delta`. The raw
1139
+ // signature lands in `thinkingSignature` for `encodeLiteLLMThinkingBlocks`.
1140
+ let liteLLMThinkingBlock: ThinkingContent | undefined;
1141
+ const appendLiteLLMThinkingBlock = (entry: unknown): void => {
1142
+ if (typeof entry !== "object" || entry === null) return;
1143
+ const type = Reflect.get(entry, "type");
1144
+ if (type === "redacted_thinking") {
1145
+ const data = Reflect.get(entry, "data");
1146
+ if (typeof data !== "string" || data.length === 0) return;
1074
1147
  if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
1075
- currentBlock = { type: "thinking", thinking: "", thinkingSignature: signature };
1076
- message.content.push(currentBlock);
1077
- eventStream.push({
1078
- type: "thinking_start",
1079
- contentIndex: blockIndex(currentBlock),
1080
- partial: message,
1081
- });
1148
+ currentBlock = undefined;
1149
+ liteLLMThinkingBlock = undefined;
1150
+ output.content.push({ type: "redactedThinking", data });
1151
+ return;
1082
1152
  }
1083
- if (signature !== undefined && !currentBlock.thinkingSignature) {
1084
- currentBlock.thinkingSignature = signature;
1153
+ if (type !== "thinking") return;
1154
+ const rawThinking = Reflect.get(entry, "thinking");
1155
+ const rawSignature = Reflect.get(entry, "signature");
1156
+ const thinking = typeof rawThinking === "string" ? rawThinking : "";
1157
+ const signature = typeof rawSignature === "string" ? rawSignature : "";
1158
+ if (!thinking && !signature) return;
1159
+ if (!firstTokenTime) firstTokenTime = performance.now();
1160
+ let block = currentBlock === liteLLMThinkingBlock ? liteLLMThinkingBlock : undefined;
1161
+ if (!block || (thinking && block.thinkingSignature)) {
1162
+ block = openThinkingBlock();
1163
+ liteLLMThinkingBlock = block;
1085
1164
  }
1086
- currentBlock.thinking += thinking;
1087
- eventStream.push({
1088
- type: "thinking_delta",
1089
- contentIndex: blockIndex(currentBlock),
1090
- delta: thinking,
1091
- partial: message,
1092
- });
1165
+ if (thinking) {
1166
+ block.thinking += thinking;
1167
+ stream.push({
1168
+ type: "thinking_delta",
1169
+ contentIndex: blockIndex(block),
1170
+ delta: thinking,
1171
+ partial: output,
1172
+ });
1173
+ }
1174
+ if (signature) block.thinkingSignature = (block.thinkingSignature ?? "") + signature;
1093
1175
  };
1094
1176
 
1095
1177
  const appendTextDelta = (text: string): void => {
@@ -1122,7 +1204,7 @@ const streamOpenAICompletionsOnce = (
1122
1204
  if (!emittedThinking) return;
1123
1205
  }
1124
1206
  if (!firstTokenTime) firstTokenTime = performance.now();
1125
- appendThinking(output, stream, emittedThinking, signature);
1207
+ appendThinking(emittedThinking, signature);
1126
1208
  };
1127
1209
 
1128
1210
  let deepseekStripBuffer = "";
@@ -1308,7 +1390,14 @@ const streamOpenAICompletionsOnce = (
1308
1390
  }
1309
1391
  }
1310
1392
 
1311
- if (foundReasoningField) {
1393
+ // LiteLLM mirrors Anthropic thinking into both `reasoning_content`
1394
+ // and `thinking_blocks`; only the latter carries the signature, so
1395
+ // it wins and the text alias is skipped to avoid duplication.
1396
+ const liteLLMThinkingBlocks = getLiteLLMThinkingBlocksDelta(choice.delta);
1397
+ if (liteLLMThinkingBlocks) {
1398
+ for (const entry of liteLLMThinkingBlocks) appendLiteLLMThinkingBlock(entry);
1399
+ suppressHealedThinking = true;
1400
+ } else if (foundReasoningField) {
1312
1401
  appendThinkingDelta(
1313
1402
  foundReasoningDelta,
1314
1403
  foundReasoningField,
@@ -2366,6 +2455,9 @@ export function convertMessages(
2366
2455
  }
2367
2456
  }
2368
2457
 
2458
+ const liteLLMThinkingBlocks = encodeLiteLLMThinkingBlocks(msg.content);
2459
+ if (liteLLMThinkingBlocks.length > 0) assistantMsg.thinking_blocks = liteLLMThinkingBlocks;
2460
+
2369
2461
  const toolCalls = msg.content.filter(b => b.type === "toolCall") as ToolCall[];
2370
2462
  // Replay reasoning_content on assistant turns for backends that validate
2371
2463
  // thinking-mode history. DeepSeek V4 requires reasoning_content on EVERY