@tormentalabs/claude-code-wire-compat 0.1.0-rc.17 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/CHANGELOG.md +101 -2
  2. package/README.md +67 -2
  3. package/dist/betas.d.ts +37 -0
  4. package/dist/betas.d.ts.map +1 -1
  5. package/dist/betas.js +55 -23
  6. package/dist/betas.js.map +1 -1
  7. package/dist/build-request.d.ts +11 -0
  8. package/dist/build-request.d.ts.map +1 -1
  9. package/dist/build-request.js +133 -11
  10. package/dist/build-request.js.map +1 -1
  11. package/dist/contracts.d.ts +53 -0
  12. package/dist/contracts.d.ts.map +1 -1
  13. package/dist/contracts.js.map +1 -1
  14. package/dist/fingerprint.d.ts +29 -2
  15. package/dist/fingerprint.d.ts.map +1 -1
  16. package/dist/fingerprint.js +60 -7
  17. package/dist/fingerprint.js.map +1 -1
  18. package/dist/headers.d.ts.map +1 -1
  19. package/dist/headers.js +12 -4
  20. package/dist/headers.js.map +1 -1
  21. package/dist/index.d.ts +1 -0
  22. package/dist/index.d.ts.map +1 -1
  23. package/dist/index.js +1 -0
  24. package/dist/index.js.map +1 -1
  25. package/dist/model-capabilities.d.ts +31 -3
  26. package/dist/model-capabilities.d.ts.map +1 -1
  27. package/dist/model-capabilities.js +145 -12
  28. package/dist/model-capabilities.js.map +1 -1
  29. package/dist/models.d.ts.map +1 -1
  30. package/dist/models.js +8 -4
  31. package/dist/models.js.map +1 -1
  32. package/dist/profile-behaviors.d.ts +61 -0
  33. package/dist/profile-behaviors.d.ts.map +1 -0
  34. package/dist/profile-behaviors.js +53 -0
  35. package/dist/profile-behaviors.js.map +1 -0
  36. package/dist/profiles/beta-registry-2.1.233.d.ts +140 -0
  37. package/dist/profiles/beta-registry-2.1.233.d.ts.map +1 -0
  38. package/dist/profiles/beta-registry-2.1.233.js +183 -0
  39. package/dist/profiles/beta-registry-2.1.233.js.map +1 -0
  40. package/dist/profiles/claude-code-2.1.195.d.ts.map +1 -1
  41. package/dist/profiles/claude-code-2.1.195.js +14 -0
  42. package/dist/profiles/claude-code-2.1.195.js.map +1 -1
  43. package/dist/profiles/claude-code-2.1.233.d.ts +3 -0
  44. package/dist/profiles/claude-code-2.1.233.d.ts.map +1 -0
  45. package/dist/profiles/claude-code-2.1.233.js +235 -0
  46. package/dist/profiles/claude-code-2.1.233.js.map +1 -0
  47. package/dist/redaction.d.ts.map +1 -1
  48. package/dist/redaction.js +14 -1
  49. package/dist/redaction.js.map +1 -1
  50. package/dist/request-body.d.ts.map +1 -1
  51. package/dist/request-body.js +12 -10
  52. package/dist/request-body.js.map +1 -1
  53. package/dist/thinking.d.ts +33 -7
  54. package/dist/thinking.d.ts.map +1 -1
  55. package/dist/thinking.js +105 -36
  56. package/dist/thinking.js.map +1 -1
  57. package/package.json +10 -2
  58. package/src/anti-verbosity.ts +219 -0
  59. package/src/beta-registry.ts +140 -0
  60. package/src/betas.ts +302 -0
  61. package/src/build-request.ts +1799 -0
  62. package/src/contracts.ts +1286 -0
  63. package/src/count-tokens.ts +84 -0
  64. package/src/fingerprint.ts +155 -0
  65. package/src/headers.ts +448 -0
  66. package/src/index.ts +63 -0
  67. package/src/metadata.ts +331 -0
  68. package/src/model-capabilities.ts +453 -0
  69. package/src/model-identity.ts +45 -0
  70. package/src/models.ts +50 -0
  71. package/src/profile-behaviors.ts +114 -0
  72. package/src/profiles/beta-registry-2.1.233.ts +200 -0
  73. package/src/profiles/claude-code-2.1.195.ts +168 -0
  74. package/src/profiles/claude-code-2.1.233.ts +240 -0
  75. package/src/redaction.ts +536 -0
  76. package/src/request-body.ts +1931 -0
  77. package/src/sha256.ts +114 -0
  78. package/src/system-prompt.ts +222 -0
  79. package/src/thinking.ts +346 -0
  80. package/src/unicode.ts +24 -0
package/src/sha256.ts ADDED
@@ -0,0 +1,114 @@
1
+ // SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+ const SHA256_INITIAL = new Uint32Array([
4
+ 0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c,
5
+ 0x1f83d9ab, 0x5be0cd19,
6
+ ]);
7
+ const SHA256_ROUNDS = new Uint32Array([
8
+ 0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1,
9
+ 0x923f82a4, 0xab1c5ed5, 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3,
10
+ 0x72be5d74, 0x80deb1fe, 0x9bdc06a7, 0xc19bf174, 0xe49b69c1, 0xefbe4786,
11
+ 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f, 0x4a7484aa, 0x5cb0a9dc, 0x76f988da,
12
+ 0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7, 0xc6e00bf3, 0xd5a79147,
13
+ 0x06ca6351, 0x14292967, 0x27b70a85, 0x2e1b2138, 0x4d2c6dfc, 0x53380d13,
14
+ 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85, 0xa2bfe8a1, 0xa81a664b,
15
+ 0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070,
16
+ 0x19a4c116, 0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a,
17
+ 0x5b9cca4f, 0x682e6ff3, 0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208,
18
+ 0x90befffa, 0xa4506ceb, 0xbef9a3f7, 0xc67178f2,
19
+ ]);
20
+
21
+ function rotateRight(value: number, count: number): number {
22
+ return (value >>> count) | (value << (32 - count));
23
+ }
24
+
25
+ function wordAt(words: Uint32Array, index: number): number {
26
+ // The fallback is unreachable at runtime because every call site indexes
27
+ // within the allocated array. It exists solely to satisfy
28
+ // noUncheckedIndexedAccess and is therefore this file's sole uncovered
29
+ // branch, intentionally not covered by a test.
30
+ return words[index] ?? 0;
31
+ }
32
+
33
+ /**
34
+ * Computes a digest only so the synchronous parser can verify the integrity of
35
+ * evidence produced by this package. The asynchronous builder path uses the
36
+ * injected Web Crypto implementations in `src/fingerprint.ts` and
37
+ * `src/redaction.ts` and MUST continue to do so. This checks non-secret,
38
+ * self-produced data; it is not a MAC and must not inform security decisions.
39
+ */
40
+ export function sha256Hex(value: string): string {
41
+ const source = new TextEncoder().encode(value);
42
+ const paddedLength = Math.ceil((source.length + 9) / 64) * 64;
43
+ const padded = new Uint8Array(paddedLength);
44
+ padded.set(source);
45
+ padded[source.length] = 0x80;
46
+ const bitLength = source.length * 8;
47
+ const view = new DataView(padded.buffer);
48
+ view.setUint32(paddedLength - 8, Math.floor(bitLength / 0x1_0000_0000));
49
+ view.setUint32(paddedLength - 4, bitLength >>> 0);
50
+ const state = new Uint32Array(SHA256_INITIAL);
51
+ const words = new Uint32Array(64);
52
+
53
+ for (let offset = 0; offset < paddedLength; offset += 64) {
54
+ for (let index = 0; index < 16; index += 1) {
55
+ words[index] = view.getUint32(offset + index * 4);
56
+ }
57
+ for (let index = 16; index < 64; index += 1) {
58
+ const previous15 = wordAt(words, index - 15);
59
+ const previous2 = wordAt(words, index - 2);
60
+ const sigma0 =
61
+ rotateRight(previous15, 7) ^
62
+ rotateRight(previous15, 18) ^
63
+ (previous15 >>> 3);
64
+ const sigma1 =
65
+ rotateRight(previous2, 17) ^
66
+ rotateRight(previous2, 19) ^
67
+ (previous2 >>> 10);
68
+ words[index] =
69
+ wordAt(words, index - 16) + sigma0 + wordAt(words, index - 7) + sigma1;
70
+ }
71
+
72
+ let a = wordAt(state, 0);
73
+ let b = wordAt(state, 1);
74
+ let c = wordAt(state, 2);
75
+ let d = wordAt(state, 3);
76
+ let e = wordAt(state, 4);
77
+ let f = wordAt(state, 5);
78
+ let g = wordAt(state, 6);
79
+ let h = wordAt(state, 7);
80
+ for (let index = 0; index < 64; index += 1) {
81
+ const sum1 = rotateRight(e, 6) ^ rotateRight(e, 11) ^ rotateRight(e, 25);
82
+ const choice = (e & f) ^ (~e & g);
83
+ const temporary1 =
84
+ (h +
85
+ sum1 +
86
+ choice +
87
+ wordAt(SHA256_ROUNDS, index) +
88
+ wordAt(words, index)) >>>
89
+ 0;
90
+ const sum0 = rotateRight(a, 2) ^ rotateRight(a, 13) ^ rotateRight(a, 22);
91
+ const majority = (a & b) ^ (a & c) ^ (b & c);
92
+ const temporary2 = (sum0 + majority) >>> 0;
93
+ h = g;
94
+ g = f;
95
+ f = e;
96
+ e = (d + temporary1) >>> 0;
97
+ d = c;
98
+ c = b;
99
+ b = a;
100
+ a = (temporary1 + temporary2) >>> 0;
101
+ }
102
+ state[0] = wordAt(state, 0) + a;
103
+ state[1] = wordAt(state, 1) + b;
104
+ state[2] = wordAt(state, 2) + c;
105
+ state[3] = wordAt(state, 3) + d;
106
+ state[4] = wordAt(state, 4) + e;
107
+ state[5] = wordAt(state, 5) + f;
108
+ state[6] = wordAt(state, 6) + g;
109
+ state[7] = wordAt(state, 7) + h;
110
+ }
111
+ return Array.from(state, (word) => word.toString(16).padStart(8, "0")).join(
112
+ "",
113
+ );
114
+ }
@@ -0,0 +1,222 @@
1
+ // SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+ import type {
4
+ ClaudeCodeRuntimeIdentity,
5
+ SystemInput,
6
+ TextBlock,
7
+ } from "./contracts.js";
8
+ import { ClaudeCodeWireError } from "./contracts.js";
9
+ import { classifySurrogateAt } from "./unicode.js";
10
+
11
+ /**
12
+ * The pinned identity text, byte-exact.
13
+ *
14
+ * It is exported because it is the byte-exact probe the parser uses to CONFIRM
15
+ * the canonical system prefix. A caller block equal to it is dropped by
16
+ * `buildCanonicalSystem` — unconditionally, even when `suppressIdentityBlock`
17
+ * removed the canonical one — so it appears at most once in a built body.
18
+ *
19
+ * The parser no longer INFERS the prefix length from this text's position: the
20
+ * root seams `suppressBillingBlock` and `suppressIdentityBlock` are recorded in
21
+ * `evidence.billingBlockSuppressed` / `evidence.identityBlockSuppressed`, which
22
+ * state which canonical blocks were emitted. This text is what the parser then
23
+ * checks the identity slot against, so evidence is verified structurally rather
24
+ * than trusted.
25
+ */
26
+ export const IDENTITY_TEXT =
27
+ "You are Claude Code, Anthropic's official CLI for Claude.";
28
+ const MAX_INPUT_DEPTH = 64;
29
+ const MAX_INPUT_SIZE = 1_000_000;
30
+ type UnknownRecord = Readonly<Record<PropertyKey, unknown>>;
31
+
32
+ function fail(
33
+ code:
34
+ | "INVALID_INPUT"
35
+ | "INVALID_UNICODE"
36
+ | "INPUT_TOO_DEEP"
37
+ | "INPUT_TOO_LARGE"
38
+ | "CYCLIC_INPUT",
39
+ ): never {
40
+ throw new ClaudeCodeWireError(code);
41
+ }
42
+
43
+ function validateText(text: string): void {
44
+ for (let index = 0; index < text.length; index += 1) {
45
+ const codeUnit = text.charCodeAt(index);
46
+
47
+ if (
48
+ codeUnit === 0 ||
49
+ (codeUnit < 0x20 &&
50
+ codeUnit !== 0x09 &&
51
+ codeUnit !== 0x0a &&
52
+ codeUnit !== 0x0d) ||
53
+ (codeUnit >= 0x7f && codeUnit <= 0x9f)
54
+ ) {
55
+ fail("INVALID_UNICODE");
56
+ }
57
+
58
+ const classification = classifySurrogateAt(text, index);
59
+ if (classification === "loneSurrogate") fail("INVALID_UNICODE");
60
+ if (classification === "surrogatePair") index += 1;
61
+ }
62
+ }
63
+
64
+ function isUnknownRecord(value: unknown): value is UnknownRecord {
65
+ return value !== null && typeof value === "object";
66
+ }
67
+
68
+ function validateStructure(value: unknown): void {
69
+ const ancestors = new WeakSet();
70
+ let size = 0;
71
+
72
+ function visit(current: unknown, depth: number): void {
73
+ if (depth > MAX_INPUT_DEPTH) fail("INPUT_TOO_DEEP");
74
+
75
+ if (typeof current === "string") {
76
+ size += current.length;
77
+ if (size > MAX_INPUT_SIZE) fail("INPUT_TOO_LARGE");
78
+ validateText(current);
79
+ return;
80
+ }
81
+
82
+ if (!isUnknownRecord(current)) return;
83
+ if (ancestors.has(current)) fail("CYCLIC_INPUT");
84
+
85
+ ancestors.add(current);
86
+ for (const key of Reflect.ownKeys(current)) {
87
+ if (typeof key === "string") {
88
+ size += key.length;
89
+ if (size > MAX_INPUT_SIZE) fail("INPUT_TOO_LARGE");
90
+ validateText(key);
91
+ }
92
+ visit(current[key], depth + 1);
93
+ }
94
+ ancestors.delete(current);
95
+ }
96
+
97
+ visit(value, 0);
98
+ }
99
+
100
+ function cloneTextBlock(value: unknown): TextBlock {
101
+ if (!isUnknownRecord(value)) fail("INVALID_INPUT");
102
+
103
+ const type = value["type"];
104
+ const text = value["text"];
105
+ if (type !== "text" || typeof text !== "string") fail("INVALID_INPUT");
106
+
107
+ const cacheControl = value["cache_control"];
108
+ if (cacheControl === undefined) return Object.freeze({ type: "text", text });
109
+ if (!isUnknownRecord(cacheControl)) fail("INVALID_INPUT");
110
+
111
+ const cacheType = cacheControl["type"];
112
+ const ttl = cacheControl["ttl"];
113
+ const scope = cacheControl["scope"];
114
+ if (
115
+ cacheType !== "ephemeral" ||
116
+ (ttl !== undefined && ttl !== "5m" && ttl !== "1h") ||
117
+ (scope !== undefined && scope !== "global")
118
+ ) {
119
+ fail("INVALID_INPUT");
120
+ }
121
+
122
+ const clonedCacheControl: {
123
+ type: "ephemeral";
124
+ ttl?: "5m" | "1h";
125
+ scope?: "global";
126
+ } = { type: "ephemeral" };
127
+ if (ttl !== undefined) clonedCacheControl.ttl = ttl;
128
+ if (scope !== undefined) clonedCacheControl.scope = scope;
129
+
130
+ return Object.freeze({
131
+ type: "text",
132
+ text,
133
+ cache_control: Object.freeze(clonedCacheControl),
134
+ });
135
+ }
136
+
137
+ function equalCacheControl(
138
+ left: TextBlock["cache_control"],
139
+ right: TextBlock["cache_control"],
140
+ ): boolean {
141
+ if (left === undefined || right === undefined) return left === right;
142
+ if (left === null || right === null) return left === right;
143
+ return left.ttl === right.ttl && left.scope === right.scope;
144
+ }
145
+
146
+ function joinTextBlocks(left: TextBlock, right: TextBlock): TextBlock {
147
+ return Object.freeze({
148
+ type: "text",
149
+ text: `${left.text}\n${right.text}`,
150
+ ...(left.cache_control === undefined
151
+ ? {}
152
+ : { cache_control: left.cache_control }),
153
+ });
154
+ }
155
+
156
+ /** Builds the pinned Claude Code system block sequence without changing caller data. */
157
+ export function buildCanonicalSystem(
158
+ input: readonly SystemInput[] | undefined,
159
+ billingBlock: TextBlock,
160
+ identity: ClaudeCodeRuntimeIdentity,
161
+ suppressBillingBlock = false,
162
+ suppressIdentityBlock = false,
163
+ ): readonly TextBlock[] {
164
+ validateStructure(input);
165
+ if (input !== undefined && !Array.isArray(input)) fail("INVALID_INPUT");
166
+ if (billingBlock.cache_control !== undefined) fail("INVALID_INPUT");
167
+
168
+ const clonedBilling = cloneTextBlock(billingBlock);
169
+ const canonicalBilling = Object.isFrozen(billingBlock)
170
+ ? billingBlock
171
+ : clonedBilling;
172
+
173
+ // The runtime identity is accepted for parity with the request builder. The
174
+ // pinned identity system text itself intentionally contains no identifiers.
175
+ void identity;
176
+
177
+ // Package extension: `suppressBillingBlock` and `suppressIdentityBlock` are
178
+ // the only ways to omit a canonical block. Both default to `false`, which
179
+ // keeps the two-block canonical prefix the genuine client always emits. With
180
+ // both active the canonical prefix is empty and the emitted `system` array
181
+ // holds caller blocks only.
182
+ const blocks: TextBlock[] = suppressBillingBlock ? [] : [canonicalBilling];
183
+ if (!suppressIdentityBlock) {
184
+ blocks.push(
185
+ Object.freeze({
186
+ type: "text",
187
+ text: IDENTITY_TEXT,
188
+ cache_control: Object.freeze({ type: "ephemeral", ttl: "1h" }),
189
+ }),
190
+ );
191
+ }
192
+
193
+ if (input !== undefined) {
194
+ let run: TextBlock | undefined;
195
+ for (const entry of input) {
196
+ const block: TextBlock =
197
+ typeof entry === "string"
198
+ ? Object.freeze({ type: "text" as const, text: entry })
199
+ : cloneTextBlock(entry);
200
+
201
+ // Upstream recognizes only the byte-for-byte identity constant. Similar
202
+ // caller text remains ordinary prompt content.
203
+ //
204
+ // The drop stays UNCONDITIONAL under `suppressIdentityBlock`: the genuine
205
+ // client drops it too, and a caller block equal to the identity text
206
+ // landing at the front of a suppressed prefix would defeat the parser's
207
+ // structural check of the canonical prefix.
208
+ if (block.text === IDENTITY_TEXT) continue;
209
+ if (run === undefined) {
210
+ run = block;
211
+ } else if (equalCacheControl(run.cache_control, block.cache_control)) {
212
+ run = joinTextBlocks(run, block);
213
+ } else {
214
+ blocks.push(run);
215
+ run = block;
216
+ }
217
+ }
218
+ if (run !== undefined) blocks.push(run);
219
+ }
220
+
221
+ return Object.freeze(blocks);
222
+ }
@@ -0,0 +1,346 @@
1
+ // SPDX-License-Identifier: GPL-3.0-or-later
2
+
3
+ import type {
4
+ ClaudeCodeBetaPolicy,
5
+ ClaudeCodeCapabilities,
6
+ ClaudeCodeProtocolProfile,
7
+ } from "./contracts.js";
8
+ import { profileBehaviors } from "./profile-behaviors.js";
9
+ import { CLAUDE_CODE_2_1_195_PROFILE } from "./profiles/claude-code-2.1.195.js";
10
+
11
+ /**
12
+ * Extended-thinking resolution, ported from the genuine client's request
13
+ * builder at byte offset 238154330.
14
+ *
15
+ * The single most surprising thing in here, and the reason this module exists
16
+ * rather than a handful of inline branches: **the caller does not choose
17
+ * between adaptive and enabled thinking — the model does.**
18
+ *
19
+ * Upstream the choice is
20
+ *
21
+ * ```
22
+ * cn = aSr(s.model);
23
+ * if (cn !== void 0 ? cn === "adaptive" : Uot(u) && !zt) { adaptive } else { enabled }
24
+ * ```
25
+ *
26
+ * `aSr` is `Bt.thinkingTypeOverrides.get(e)`, a host-side override map that is
27
+ * empty on a default install, so `cn` is undefined and the ternary falls through
28
+ * to `Uot(u)` — the adaptive-thinking capability predicate. `zt` additionally
29
+ * requires an environment variable that is unset by default.
30
+ *
31
+ * So a caller asking for `type: "enabled"` against an adaptive-capable model
32
+ * gets `{type:"adaptive"}` on the wire and their `budgetTokens` is discarded,
33
+ * and a caller asking for `type: "adaptive"` against a model without the
34
+ * capability gets `{type:"enabled",budget_tokens:…}`. The caller's `type` is
35
+ * load-bearing in exactly one way: whether or not it is `"disabled"`.
36
+ *
37
+ * This package reproduces that. Rejecting the mismatch instead — which is what
38
+ * it used to do — would make its traffic distinguishable from the real client's,
39
+ * which is the one thing it exists to avoid.
40
+ *
41
+ * This module also owns `modelOutputTokenLimits` and `clampMaxTokens`, which
42
+ * bound `max_tokens` rather than anything thinking-specific. They live here
43
+ * because upstream derives both from one table (`Xxe`) and feeds one clamped
44
+ * value (`Fi`) into both the emitted `max_tokens` and the thinking budget, so
45
+ * splitting them would separate two things that must not drift apart. If a
46
+ * third consumer of the limit table ever appears, extract all three into their
47
+ * own module at that point.
48
+ */
49
+
50
+ /** Permitted values of the `display` property, from the schema at 241453966. */
51
+ export type ThinkingDisplay = "summarized" | "omitted";
52
+
53
+ export interface ThinkingRequest {
54
+ readonly type: "enabled" | "adaptive" | "disabled";
55
+ readonly budgetTokens?: number;
56
+ readonly display?: ThinkingDisplay;
57
+ }
58
+
59
+ export interface ModelOutputTokenLimits {
60
+ readonly default: number;
61
+ readonly upperLimit: number;
62
+ }
63
+
64
+ export interface ResolvedThinking {
65
+ /** The object to place at `body.thinking`, or undefined to omit the field. */
66
+ readonly emitted: Readonly<Record<string, unknown>> | undefined;
67
+ /**
68
+ * Whether the caller asked for thinking at all, regardless of whether any
69
+ * `thinking` object survived resolution. This — not `emitted` — is what
70
+ * suppresses `temperature`, matching upstream `nr`.
71
+ */
72
+ readonly requestActive: boolean;
73
+ /** Whether `tool_choice` of type `tool` must be demoted to `auto`. */
74
+ readonly extendedThinkingActive: boolean;
75
+ }
76
+
77
+ /**
78
+ * Per-model output token limits, ported from upstream `Xxe` at byte offset
79
+ * 227378240. Keyed on the NORMALISED model id.
80
+ *
81
+ * Resolution order, since Fase 1.2:
82
+ *
83
+ * 1. The pinned 2.1.195 catalogue, when the id has an entry carrying
84
+ * `maxOutputTokens`. That is the single source of truth for every model
85
+ * the profile knows, and `token-limits-equivalence.test.ts` pins the two
86
+ * sources cell by cell.
87
+ * 2. Otherwise the transcribed `Xxe` table below, preserved intact as the
88
+ * demarcated fallback. It is NOT dead code and must not be trimmed to
89
+ * "only the ids the catalogue lacks": `claude-3-opus`, `claude-3-sonnet`
90
+ * and `claude-3-haiku` are reachable through the normaliser with no
91
+ * catalogue entry, `claude-mythos-5` is absent from the catalogue by
92
+ * product decision D-1, and any id from a newer client lands on the
93
+ * final fallback row.
94
+ *
95
+ * Both fields are load-bearing. `upperLimit` seeds the thinking budget when the
96
+ * caller supplies none (upstream `wvi = Xxe(e).upperLimit - 1`); `default` caps
97
+ * the emitted `max_tokens` (upstream `qct`, see `clampMaxTokens`).
98
+ *
99
+ * Upstream additionally consults `Vkd` and `bvi`, neither of which is modelled:
100
+ *
101
+ * - `Vkd(e)` lowers `default` from the host config object `heather_vale`.
102
+ * That object is absent on a default install, so it returns null and makes
103
+ * no adjustment. Same class as `W9` in `model-capabilities.ts`: a host-side
104
+ * override this package cannot observe.
105
+ * - `bvi(e)` adjusts BOTH fields, but sits behind `_vi()`, which returns a
106
+ * hard `false`. Dead code upstream.
107
+ *
108
+ * From 2.1.222 onward upstream grew a THIRD adjustment, this one derived from
109
+ * the request rather than from host state, and therefore observable: see
110
+ * `requestedMaxTokens` below.
111
+ *
112
+ * @param requestedMaxTokens
113
+ * The caller's own `max_tokens`, when the call site has it. Modelled for
114
+ * profiles from 2.1.222 onward only; see the demarcated block below.
115
+ */
116
+ export function modelOutputTokenLimits(
117
+ normalizedId: string,
118
+ profile: ClaudeCodeProtocolProfile = CLAUDE_CODE_2_1_195_PROFILE,
119
+ requestedMaxTokens?: number,
120
+ ): ModelOutputTokenLimits {
121
+ const resolved = resolveDeclaredLimits(normalizedId, profile);
122
+
123
+ /*
124
+ * ---- Demarcated: request-derived upper bound, upstream 2.1.222+. ----
125
+ *
126
+ * Upstream raises `upperLimit` to the caller's own `max_tokens` and lowers
127
+ * `default` to fit under it:
128
+ *
129
+ * upperLimit = requestedMaxTokens;
130
+ * default = Math.min(default, upperLimit);
131
+ *
132
+ * Verified byte-identical between upstream 2.1.222 and 2.1.233.
133
+ *
134
+ * The gate is STRUCTURAL, not a capability flag: this behaviour exists in
135
+ * upstream 2.1.222+ and 2.1.195 does not have it, so the 195 profile must
136
+ * never see it. Which profiles are on which side is `profile-behaviors.ts`'s
137
+ * question, not this module's.
138
+ *
139
+ * `Number.isSafeInteger` is deliberately stricter than upstream's truthy
140
+ * check. The upstream runtime only ever produces integers in this field, so
141
+ * the two agree on every reachable input; here a NaN or Infinity would
142
+ * propagate straight into `budget_tokens`, which must stay an integer.
143
+ */
144
+ if (
145
+ profileBehaviors(profile).requestDerivedTokenCeiling &&
146
+ requestedMaxTokens !== undefined &&
147
+ Number.isSafeInteger(requestedMaxTokens) &&
148
+ requestedMaxTokens >= 4096
149
+ ) {
150
+ const upperLimit = requestedMaxTokens;
151
+ return { default: Math.min(resolved.default, upperLimit), upperLimit };
152
+ }
153
+
154
+ return resolved;
155
+ }
156
+
157
+ function resolveDeclaredLimits(
158
+ normalizedId: string,
159
+ profile: ClaudeCodeProtocolProfile,
160
+ ): ModelOutputTokenLimits {
161
+ const declared = profile.supportedModels[normalizedId]?.maxOutputTokens;
162
+ if (declared !== undefined) {
163
+ // `upper` is the catalogue's name for what this module calls `upperLimit`;
164
+ // the rename happens here and nowhere else.
165
+ return { default: declared.default, upperLimit: declared.upper };
166
+ }
167
+
168
+ /*
169
+ * ---- Demarcated fallback: the reachable remainder of `Xxe`. ----
170
+ *
171
+ * No catalogue id reaches this point. Every entry of the 2.1.195 catalogue
172
+ * declares `maxOutputTokens`, and `capability-equivalence.test.ts` fails if
173
+ * one stops doing so, which is what keeps the rows below to the ids the
174
+ * catalogue genuinely cannot answer for:
175
+ *
176
+ * - `claude-mythos-5` has no catalogue entry by product decision D-1.
177
+ * - `claude-3-opus`, `claude-3-sonnet` and `claude-3-haiku` are reachable
178
+ * through the normaliser and predate the catalogue.
179
+ *
180
+ * The rows for catalogued ids were deleted rather than kept "just in case":
181
+ * they were unreachable, so they could be neither covered nor
182
+ * mutation-killed, and a second copy of a limit that no longer serves any
183
+ * request is exactly the duplicated table
184
+ * `test/governance/single-source-of-truth.test.ts` exists to prevent.
185
+ *
186
+ * This mirrors upstream 2.1.222, where derivation is catalogue-first and
187
+ * the surviving legacy rows are the `claude-3-*` ones plus a generic tail.
188
+ *
189
+ * If a future profile omits `maxOutputTokens` for some id, that id lands on
190
+ * the generic tail below -- 32000/128000 -- rather than on a stale
191
+ * per-model row. That is deliberate: a wrong-but-loud generic limit is
192
+ * recoverable, a silently stale per-model limit is not. The equivalence
193
+ * guard fires first in any case.
194
+ */
195
+ if (normalizedId === "claude-mythos-5") {
196
+ return { default: 64000, upperLimit: 128000 };
197
+ }
198
+ if (normalizedId === "claude-3-opus") {
199
+ return { default: 4096, upperLimit: 4096 };
200
+ }
201
+ if (normalizedId === "claude-3-sonnet") {
202
+ return { default: 8192, upperLimit: 8192 };
203
+ }
204
+ if (normalizedId === "claude-3-haiku") {
205
+ return { default: 4096, upperLimit: 4096 };
206
+ }
207
+ return { default: 32000, upperLimit: 128000 };
208
+ }
209
+
210
+ /**
211
+ * Caps the caller's `max_tokens` at the model's default output limit, porting
212
+ * upstream `Fi = Math.min(En?.maxTokensOverride || s.maxOutputTokensOverride
213
+ * || la, la)` where `la = qct(u)`.
214
+ *
215
+ * `qct` is `Fue("CLAUDE_CODE_MAX_OUTPUT_TOKENS", <env>, t.default,
216
+ * t.upperLimit).effective` over `t = Xxe(model)`. `Fue` returns `t.default`
217
+ * untouched whenever the environment variable is unset, and only ever clamps
218
+ * the ENVIRONMENT value against `t.upperLimit` — never the default. This
219
+ * package reads no environment, so `qct` reduces to `Xxe(model).default` and
220
+ * `upperLimit` plays no part in this bound.
221
+ *
222
+ * The result is load-bearing twice over: it is the emitted `max_tokens`, and it
223
+ * is the `Fi` that `resolveThinking` clamps the thinking budget against via
224
+ * `Tr = Math.min(Fi - 1, Tr)`. Both call sites must receive the CLAMPED value.
225
+ *
226
+ * Upstream uses `||`, not `??`, so a zero override would fall back to the
227
+ * default. Unreachable here: `max_tokens` is validated as a positive integer
228
+ * before this runs.
229
+ *
230
+ * `requested` is forwarded as the request-derived bound so that this call site
231
+ * reads the same table upstream reads. It cannot change the result: the
232
+ * override only ever lowers `default` to `requested`, and
233
+ * `min(requested, min(default, requested)) === min(requested, default)`. It is
234
+ * passed for coherence of reading, not for effect.
235
+ */
236
+ export function clampMaxTokens(
237
+ requested: number,
238
+ normalizedId: string,
239
+ profile: ClaudeCodeProtocolProfile = CLAUDE_CODE_2_1_195_PROFILE,
240
+ ): number {
241
+ return Math.min(
242
+ requested,
243
+ modelOutputTokenLimits(normalizedId, profile, requested).default,
244
+ );
245
+ }
246
+
247
+ /**
248
+ * Upstream `Yn = nr && CM() && QOt(u) ? n.display : void 0`, combined with the
249
+ * `if (Xn && Yn)` guard on the beta splice.
250
+ *
251
+ * This answers the splice question on its own, without consulting whether a
252
+ * `thinking` object was actually emitted, because the two are equivalent: a true
253
+ * result requires `type !== "disabled"` and `capabilities.thinking`, and those
254
+ * two conditions are exactly what drives `resolveThinking` into one of its two
255
+ * emitting branches. Upstream's `Xn` is therefore always set whenever `Yn` is.
256
+ *
257
+ * Deliberately tolerant of unvalidated input so that request assembly can ask
258
+ * this question before the body validator has run. Anything malformed answers
259
+ * false here and is rejected later by `buildCanonicalBody`.
260
+ */
261
+ export function isThinkingDisplayActive(
262
+ request: unknown,
263
+ capabilities: ClaudeCodeCapabilities,
264
+ betaPolicy: ClaudeCodeBetaPolicy,
265
+ ): boolean {
266
+ if (request === null || typeof request !== "object") return false;
267
+ const record = request as Record<string, unknown>;
268
+ if (record["type"] === "disabled") return false;
269
+ const display = record["display"];
270
+ if (display !== "summarized" && display !== "omitted") return false;
271
+ return (
272
+ capabilities.thinking &&
273
+ capabilities.interleavedThinking &&
274
+ betaPolicy.experimentalBetasEnabled
275
+ );
276
+ }
277
+
278
+ /**
279
+ * Resolves the caller's thinking request into the object the genuine client
280
+ * would put on the wire.
281
+ *
282
+ * Key order is load-bearing. Upstream emits `{budget_tokens, type, display}`
283
+ * for the enabled branch — `budget_tokens` FIRST — and `{type, display}` for
284
+ * adaptive. Serialised bodies are compared byte for byte, so the insertion
285
+ * order below must not be rearranged.
286
+ */
287
+ export function resolveThinking(
288
+ request: ThinkingRequest | undefined,
289
+ normalizedId: string,
290
+ capabilities: ClaudeCodeCapabilities,
291
+ betaPolicy: ClaudeCodeBetaPolicy,
292
+ maxTokens: number,
293
+ profile: ClaudeCodeProtocolProfile = CLAUDE_CODE_2_1_195_PROFILE,
294
+ ): ResolvedThinking {
295
+ // Upstream `nr = n.type !== "disabled" && !CLAUDE_CODE_DISABLE_THINKING`.
296
+ const requestActive = request !== undefined && request.type !== "disabled";
297
+ const displayActive = isThinkingDisplayActive(
298
+ request,
299
+ capabilities,
300
+ betaPolicy,
301
+ );
302
+ const display = displayActive ? request?.display : undefined;
303
+
304
+ let emitted: Record<string, unknown> | undefined;
305
+
306
+ if (requestActive && capabilities.thinking) {
307
+ if (capabilities.adaptiveThinking) {
308
+ emitted = { type: "adaptive" };
309
+ if (display !== undefined) emitted["display"] = display;
310
+ } else {
311
+ // Upstream: `let Tr = wvi(u)` — the model's upper limit minus one —
312
+ // overridden by the caller's budget when supplied, then clamped by
313
+ // `Tr = Math.min(Fi - 1, Tr)` where `Fi` is the emitted `max_tokens`.
314
+ //
315
+ // This is the one wire-visible consumer of the request-derived bound: on
316
+ // a 2.1.222+ profile a caller asking for a `max_tokens` above the
317
+ // catalogue's upper limit seeds the default budget from THEIR number
318
+ // minus one, not from the catalogue's.
319
+ const requested =
320
+ request.budgetTokens ??
321
+ modelOutputTokenLimits(normalizedId, profile, maxTokens).upperLimit - 1;
322
+ emitted = { budget_tokens: Math.min(maxTokens - 1, requested) };
323
+ emitted["type"] = "enabled";
324
+ if (display !== undefined) emitted["display"] = display;
325
+ }
326
+ } else if (
327
+ request?.type === "disabled" &&
328
+ capabilities.thinking &&
329
+ !capabilities.rejectsDisabledThinking
330
+ ) {
331
+ emitted = { type: "disabled" };
332
+ }
333
+
334
+ // Upstream `Jr = Xn?.type === "enabled" || Xn?.type === "adaptive"
335
+ // || Xn === void 0 && U4e(u)`.
336
+ const extendedThinkingActive =
337
+ emitted?.["type"] === "enabled" ||
338
+ emitted?.["type"] === "adaptive" ||
339
+ (emitted === undefined && capabilities.rejectsDisabledThinking);
340
+
341
+ return Object.freeze({
342
+ emitted: emitted === undefined ? undefined : Object.freeze(emitted),
343
+ requestActive,
344
+ extendedThinkingActive,
345
+ });
346
+ }