@semanticist14/clco 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/token.ts ADDED
@@ -0,0 +1,298 @@
1
+ // Short-lived Copilot token management (exchange + cache + refresh) and
2
+ // Copilot model discovery.
3
+
4
+ import {
5
+ GITHUB_API_BASE_URL,
6
+ copilotBaseUrl,
7
+ copilotFetch,
8
+ copilotRequestHeaders,
9
+ githubRequestHeaders,
10
+ isMockMode,
11
+ setCopilotBase,
12
+ } from "./api"
13
+ import { ensureGithubToken } from "./auth"
14
+ import { advertisedId } from "./catalog"
15
+ import { isTlsTrustError } from "./tls"
16
+ import { setModelAliases } from "./translate"
17
+
18
+ export interface ModelMapping {
19
+ opus: string
20
+ sonnet: string
21
+ haiku: string
22
+ fable: string
23
+ }
24
+
25
+ interface CopilotTokenResponse {
26
+ token: string
27
+ expires_at: number // epoch seconds
28
+ refresh_in?: number
29
+ // Business/Enterprise accounts are served from a different host — official
30
+ // clients route to whatever this field names.
31
+ endpoints?: { api?: string }
32
+ }
33
+
34
+ let cached: { token: string; expiresAt: number } | null = null
35
+ let githubToken: string | null = null
36
+ let pending: Promise<string> | null = null
37
+
38
+ const EXPIRY_MARGIN_MS = 5 * 60 * 1000
39
+
40
+ export async function getCopilotToken(force = false): Promise<string> {
41
+ if (isMockMode()) return "mock"
42
+ if (!force && cached && cached.expiresAt - EXPIRY_MARGIN_MS > Date.now()) {
43
+ return cached.token
44
+ }
45
+ // Deduplicate concurrent refreshes — including forced ones after a 401.
46
+ // Any in-flight fetch returns a freshly minted token, so sharing is always
47
+ // correct and concurrent retries never race into parallel exchanges.
48
+ if (pending) return pending
49
+ pending = fetchCopilotToken().finally(() => {
50
+ pending = null
51
+ })
52
+ return pending
53
+ }
54
+
55
+ async function fetchCopilotToken(): Promise<string> {
56
+ githubToken ??= (await ensureGithubToken()).token
57
+ const res = await copilotFetch(`${GITHUB_API_BASE_URL}/copilot_internal/v2/token`, {
58
+ headers: githubRequestHeaders(githubToken),
59
+ // Short metadata call — never let a hung connection stall startup.
60
+ signal: AbortSignal.timeout(10_000),
61
+ })
62
+ if (res.status === 401 || res.status === 403) {
63
+ // The stored GitHub token is revoked/expired — drop it so the next
64
+ // ensureGithubToken can pick up a fresh one after re-auth.
65
+ githubToken = null
66
+ cached = null
67
+ throw new Error(
68
+ "GitHub token was rejected - re-authenticate with `clco auth`",
69
+ )
70
+ }
71
+ if (!res.ok) {
72
+ throw new Error(
73
+ `Copilot token exchange failed: HTTP ${res.status} (check that your Copilot subscription is active)`,
74
+ )
75
+ }
76
+ const data = (await res.json()) as CopilotTokenResponse
77
+ if (typeof data.expires_at !== "number" || !Number.isFinite(data.expires_at)) {
78
+ throw new Error("Copilot token response has no expires_at")
79
+ }
80
+ cached = { token: data.token, expiresAt: data.expires_at * 1000 }
81
+ setCopilotBase(data.endpoints?.api ?? null)
82
+ return cached.token
83
+ }
84
+
85
+ export function invalidateCopilotToken(): void {
86
+ cached = null
87
+ }
88
+
89
+ /**
90
+ * Non-secret facts Copilot encodes in its own token (plan and expiry). The
91
+ * plan is what actually explains a 402, and it needs no extra network call.
92
+ */
93
+ export async function copilotTokenFacts(): Promise<{
94
+ sku?: string
95
+ expiresAt?: number
96
+ }> {
97
+ if (isMockMode()) return {}
98
+ const token = await getCopilotToken()
99
+ const field = (key: string) =>
100
+ token
101
+ .split(";")
102
+ .find((part) => part.startsWith(`${key}=`))
103
+ ?.slice(key.length + 1)
104
+ const exp = Number(field("exp"))
105
+ return {
106
+ sku: field("sku"),
107
+ expiresAt: Number.isFinite(exp) && exp > 0 ? exp * 1000 : undefined,
108
+ }
109
+ }
110
+
111
+ const FALLBACK_MODELS: ModelMapping = {
112
+ opus: "claude-opus-4.1",
113
+ sonnet: "claude-sonnet-4.5",
114
+ haiku: "claude-sonnet-4.5",
115
+ fable: "claude-sonnet-4.5",
116
+ }
117
+
118
+ // Highest version wins ("claude-sonnet-4.5" > "claude-sonnet-4" > "...-3.7").
119
+ function pickModel(ids: string[], needle: string): string | undefined {
120
+ const matches = ids.filter((id) => id.includes(needle))
121
+ matches.sort((a, b) => b.localeCompare(a, undefined, { numeric: true }))
122
+ return matches[0]
123
+ }
124
+
125
+ /** What Copilot's /models tells us about one model. */
126
+ export interface UpstreamModel {
127
+ id: string
128
+ name: string
129
+ /** e.g. ["/v1/messages", "/chat/completions"] — drives dialect routing. */
130
+ endpoints: string[]
131
+ /** Declared reasoning_effort values, or null when the model has none. */
132
+ efforts: string[] | null
133
+ maxPromptTokens?: number
134
+ maxContextTokens?: number
135
+ policyState?: string
136
+ pickerEnabled?: boolean
137
+ /** capabilities.type — "chat" for anything you can hold a turn with. */
138
+ type?: string
139
+ /** capabilities.family — a model name, or a role for internal plumbing. */
140
+ family?: string
141
+ }
142
+
143
+ // Raw upstream model list, captured during discovery at startup. The server
144
+ // serves GET /v1/models from this so Claude Code's 3-second discovery
145
+ // timeout is never hit waiting on a live upstream fetch.
146
+ let cachedModelList: UpstreamModel[] | null = null
147
+
148
+ export function upstreamModels(): UpstreamModel[] {
149
+ return cachedModelList ?? []
150
+ }
151
+
152
+ export function modelInfo(id: string): UpstreamModel | undefined {
153
+ return cachedModelList?.find((m) => m.id === id)
154
+ }
155
+
156
+ /** Copilot serves Claude models through the native Anthropic endpoint. */
157
+ export function supportsNativeMessages(id: string): boolean {
158
+ return modelInfo(id)?.endpoints.includes("/v1/messages") ?? false
159
+ }
160
+
161
+ interface RawModel {
162
+ id?: string
163
+ name?: string
164
+ slug?: string
165
+ supported_endpoints?: string[]
166
+ model_picker_enabled?: boolean
167
+ policy?: { state?: string }
168
+ capabilities?: {
169
+ type?: string
170
+ family?: string
171
+ limits?: { max_prompt_tokens?: number; max_context_window_tokens?: number }
172
+ supports?: { reasoning_effort?: string[] }
173
+ }
174
+ }
175
+
176
+ function toUpstreamModel(m: RawModel): UpstreamModel | null {
177
+ const id = m.id ?? m.slug
178
+ if (!id) return null
179
+ const limits = m.capabilities?.limits
180
+ return {
181
+ id,
182
+ name: m.name ?? id,
183
+ endpoints: Array.isArray(m.supported_endpoints) ? m.supported_endpoints : [],
184
+ efforts: Array.isArray(m.capabilities?.supports?.reasoning_effort)
185
+ ? m.capabilities.supports.reasoning_effort
186
+ : null,
187
+ maxPromptTokens: numberOrUndefined(limits?.max_prompt_tokens),
188
+ maxContextTokens: numberOrUndefined(limits?.max_context_window_tokens),
189
+ policyState: m.policy?.state,
190
+ pickerEnabled: m.model_picker_enabled,
191
+ type: m.capabilities?.type,
192
+ family: m.capabilities?.family,
193
+ }
194
+ }
195
+
196
+ function numberOrUndefined(value: unknown): number | undefined {
197
+ return typeof value === "number" && Number.isFinite(value) && value > 0
198
+ ? value
199
+ : undefined
200
+ }
201
+
202
+ // Why discovery fell back, if it did. Held rather than printed so the caller
203
+ // can surface it after its spinner stops.
204
+ let softFailure: string | null = null
205
+
206
+ export function takeDiscoverySoftFailure(): string | null {
207
+ const out = softFailure
208
+ softFailure = null
209
+ return out
210
+ }
211
+
212
+ // Resolve Copilot model slugs for Claude Code's opus/sonnet/haiku slots.
213
+ // Priority: env overrides > Copilot /models discovery > hardcoded fallback.
214
+ export async function discoverModels(): Promise<ModelMapping> {
215
+ softFailure = null
216
+ const env = (name: string) => process.env[name]?.trim() || undefined
217
+ const overrides = {
218
+ opus: env("CLCO_OPUS"),
219
+ sonnet: env("CLCO_SONNET"),
220
+ haiku: env("CLCO_HAIKU"),
221
+ fable: env("CLCO_FABLE"),
222
+ }
223
+ let reason = ""
224
+ let noClaudeModels = false
225
+ try {
226
+ const token = await getCopilotToken()
227
+ const res = await copilotFetch(`${copilotBaseUrl()}/models`, {
228
+ headers: copilotRequestHeaders(token),
229
+ signal: AbortSignal.timeout(10_000),
230
+ })
231
+ if (res.ok) {
232
+ const body = (await res.json()) as {
233
+ data?: RawModel[]
234
+ models?: RawModel[]
235
+ }
236
+ const raw = body.models ?? body.data ?? []
237
+ cachedModelList = raw
238
+ .map(toUpstreamModel)
239
+ .filter((m): m is UpstreamModel => m !== null)
240
+ // Teach the translator every id the picker is about to advertise, so a
241
+ // model chosen by its catalog-form id still reaches the right slug.
242
+ setModelAliases(
243
+ new Map(
244
+ cachedModelList.flatMap((m) => {
245
+ const advertised = advertisedId(m.id)
246
+ return advertised && advertised !== m.id
247
+ ? ([[advertised, m.id]] as [string, string][])
248
+ : []
249
+ }),
250
+ ),
251
+ )
252
+ const ids = cachedModelList.map((m) => m.id)
253
+ const opus = overrides.opus ?? pickModel(ids, "claude-opus")
254
+ const sonnet = overrides.sonnet ?? pickModel(ids, "claude-sonnet")
255
+ const haiku =
256
+ overrides.haiku ??
257
+ pickModel(ids, "claude-haiku") ??
258
+ pickModel(ids, "claude-sonnet")
259
+ const fable =
260
+ overrides.fable ??
261
+ pickModel(ids, "claude-fable") ??
262
+ sonnet ??
263
+ FALLBACK_MODELS.fable
264
+ if (sonnet) {
265
+ return {
266
+ opus: opus ?? FALLBACK_MODELS.opus,
267
+ sonnet,
268
+ haiku: haiku ?? FALLBACK_MODELS.haiku,
269
+ fable,
270
+ }
271
+ }
272
+ // The list arrived and simply has no Claude models in it - a fact about
273
+ // this plan, not about the network. Falling through to the generic
274
+ // message below announced "discovery failed" about a request that
275
+ // returned 200 and whose result is already driving the picker.
276
+ noClaudeModels = true
277
+ }
278
+ } catch (err) {
279
+ reason = err instanceof Error ? err.message : String(err)
280
+ // A broken trust chain is not a "carry on with fallback slugs" situation
281
+ // — it is the user's actual blocker, and every later request will fail
282
+ // the same way. Let it reach main().catch so the TLS hint gets printed.
283
+ if (isTlsTrustError(err)) throw err
284
+ }
285
+ // Reported by the caller after any progress spinner has stopped; printing
286
+ // here would be painted over by the spinner that wraps this call.
287
+ softFailure = noClaudeModels
288
+ ? `[clco] Copilot serves no Claude model on this plan - the opus/sonnet/haiku` +
289
+ ` slots are guesses (override with CLCO_OPUS/SONNET/HAIKU)`
290
+ : `[clco] Copilot /models discovery failed (${copilotBaseUrl()}${reason ? `: ${reason}` : ""})` +
291
+ ` - falling back to default slugs (override with CLCO_OPUS/SONNET/HAIKU)`
292
+ return {
293
+ opus: overrides.opus ?? FALLBACK_MODELS.opus,
294
+ sonnet: overrides.sonnet ?? FALLBACK_MODELS.sonnet,
295
+ haiku: overrides.haiku ?? FALLBACK_MODELS.haiku,
296
+ fable: overrides.fable ?? FALLBACK_MODELS.fable,
297
+ }
298
+ }
package/src/tokens.ts ADDED
@@ -0,0 +1,55 @@
1
+ import { classifyContent } from "./blocks"
2
+ import type { AnthropicRequest } from "./wire"
3
+
4
+ export const ONE_MILLION_TOKENS = 1_000_000
5
+ // An unknown model must not inherit a larger budget than a common 128k model.
6
+ // Discovery can lower this fallback further; it is not a claimed model limit.
7
+ export const UNKNOWN_MODEL_WINDOW = 128_000
8
+ export const TOKEN_WARNING_RATIO = 0.98
9
+ const CHARS_PER_TOKEN = 3.5
10
+ const IMAGE_TOKENS = 1_600
11
+ const MESSAGE_TOKENS = 4
12
+
13
+ export function fallbackInputWindow(windows: Array<number | undefined>): number {
14
+ return Math.min(UNKNOWN_MODEL_WINDOW, ...windows.filter(
15
+ (n): n is number => n !== undefined && Number.isFinite(n) && n > 0,
16
+ ))
17
+ }
18
+
19
+ /** Rough text estimate, not a tokenizer or proof a request exceeds a limit. */
20
+ export function estimateTokens(payload: AnthropicRequest): number {
21
+ let chars = 0
22
+ let images = 0
23
+ const count = (content: unknown): void => {
24
+ for (const view of classifyContent(content)) {
25
+ switch (view.kind) {
26
+ case "text":
27
+ case "unsupported":
28
+ chars += view.text.length
29
+ break
30
+ case "image":
31
+ images++
32
+ break
33
+ case "tool_use":
34
+ // Arguments are model-visible JSON; ids and wire envelopes are not.
35
+ chars += view.block.name.length + JSON.stringify(view.block.input ?? {}).length
36
+ break
37
+ case "tool_result":
38
+ count(view.block.content)
39
+ break
40
+ default: {
41
+ const exhaustive: never = view
42
+ throw new Error(`unhandled content: ${exhaustive}`)
43
+ }
44
+ }
45
+ }
46
+ }
47
+ for (const message of payload.messages ?? []) count(message.content)
48
+ count(payload.system)
49
+ for (const tool of payload.tools ?? []) {
50
+ chars += tool.name.length + (tool.description?.length ?? 0)
51
+ chars += JSON.stringify(tool.input_schema).length
52
+ }
53
+ return Math.ceil(chars / CHARS_PER_TOKEN) + images * IMAGE_TOKENS +
54
+ (payload.messages?.length ?? 0) * MESSAGE_TOKENS
55
+ }