@semanticist14/clco 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +279 -0
- package/bin/clco +27 -0
- package/bun.lock +39 -0
- package/install.sh +161 -0
- package/package.json +35 -0
- package/scripts/mock-upstream.ts +129 -0
- package/scripts/write-launcher.sh +47 -0
- package/src/api.ts +90 -0
- package/src/auth.ts +130 -0
- package/src/blocks.ts +198 -0
- package/src/browsermcp.ts +430 -0
- package/src/catalog.ts +147 -0
- package/src/claudehome.ts +269 -0
- package/src/cli.ts +889 -0
- package/src/config.ts +164 -0
- package/src/responses.ts +435 -0
- package/src/route.ts +52 -0
- package/src/server.ts +641 -0
- package/src/setup.ts +218 -0
- package/src/spawn.ts +453 -0
- package/src/stream.ts +235 -0
- package/src/tls.ts +234 -0
- package/src/token.ts +298 -0
- package/src/tokens.ts +55 -0
- package/src/translate.ts +384 -0
- package/src/wire.ts +149 -0
- package/uninstall.sh +42 -0
package/src/server.ts
ADDED
|
@@ -0,0 +1,641 @@
|
|
|
1
|
+
// Local Anthropic-compatible adapter server. Claude Code talks to this; it
|
|
2
|
+
// translates and forwards to GitHub Copilot.
|
|
3
|
+
|
|
4
|
+
import { StreamTranslator } from "./stream"
|
|
5
|
+
import { estimateTokens, ONE_MILLION_TOKENS, TOKEN_WARNING_RATIO } from "./tokens"
|
|
6
|
+
import { DialectRouter, type Dialect } from "./route"
|
|
7
|
+
import type { AnthropicRequest, OpenAIRequest, OpenAIResponse, StreamEventData } from "./wire"
|
|
8
|
+
import {
|
|
9
|
+
copilotBaseUrl,
|
|
10
|
+
copilotFetch,
|
|
11
|
+
copilotRequestHeaders,
|
|
12
|
+
isMockMode,
|
|
13
|
+
} from "./api"
|
|
14
|
+
import { isTlsTrustError, tlsHint } from "./tls"
|
|
15
|
+
import {
|
|
16
|
+
getCopilotToken,
|
|
17
|
+
invalidateCopilotToken,
|
|
18
|
+
modelInfo,
|
|
19
|
+
upstreamModels,
|
|
20
|
+
} from "./token"
|
|
21
|
+
import {
|
|
22
|
+
normalizeModel,
|
|
23
|
+
translateRequest,
|
|
24
|
+
translateResponse,
|
|
25
|
+
} from "./translate"
|
|
26
|
+
import {
|
|
27
|
+
ResponsesEventAdapter,
|
|
28
|
+
responsesToOpenAIResponse,
|
|
29
|
+
toResponsesRequest,
|
|
30
|
+
} from "./responses"
|
|
31
|
+
|
|
32
|
+
export interface ServerHandle {
|
|
33
|
+
url: string
|
|
34
|
+
port: number
|
|
35
|
+
stop(): void
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export interface ServerOptions {
|
|
39
|
+
port?: number
|
|
40
|
+
/** Override the Copilot base URL (tests inject a mock directly).
|
|
41
|
+
* Implies mock-token mode: no GitHub exchange is attempted. */
|
|
42
|
+
upstream?: string
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
// Token injected into claude via ANTHROPIC_AUTH_TOKEN; requests without it
|
|
46
|
+
// are rejected so stray local processes (or web pages doing no-cors POSTs)
|
|
47
|
+
// cannot burn Copilot quota through the adapter.
|
|
48
|
+
const LOCAL_TOKEN = "clco-local"
|
|
49
|
+
|
|
50
|
+
const MAX_BODY_BYTES = 64 * 1024 * 1024
|
|
51
|
+
|
|
52
|
+
const QUOTA_GUIDANCE =
|
|
53
|
+
"Copilot monthly premium quota exhausted, or no subscription - check github.com/settings/copilot"
|
|
54
|
+
|
|
55
|
+
// clco appends [1m] to picker rows whose real window exceeds the default
|
|
56
|
+
// ceiling, which makes Claude Code request the 1M-context beta. Forward that
|
|
57
|
+
// only to a model that genuinely has the window — Copilot rejects the header
|
|
58
|
+
// otherwise, and the [1m] suffix is the only per-row window channel the
|
|
59
|
+
// picker schema offers, so we cannot simply stop using it.
|
|
60
|
+
export function sanitizeBeta(
|
|
61
|
+
beta: string | undefined,
|
|
62
|
+
model: string,
|
|
63
|
+
): string | undefined {
|
|
64
|
+
if (!beta) return beta
|
|
65
|
+
const window =
|
|
66
|
+
modelInfo(model)?.maxPromptTokens ?? modelInfo(model)?.maxContextTokens
|
|
67
|
+
if ((window ?? 0) >= ONE_MILLION_TOKENS) return beta
|
|
68
|
+
const kept = beta
|
|
69
|
+
.split(",")
|
|
70
|
+
.map((v) => v.trim())
|
|
71
|
+
.filter((v) => v && !v.startsWith("context-1m"))
|
|
72
|
+
return kept.length > 0 ? kept.join(",") : undefined
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function timestamp(): string {
|
|
76
|
+
return new Date().toTimeString().slice(0, 8)
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
// Request logs go to the terminal in serve mode; in run mode claude's TUI
|
|
80
|
+
// owns stdout/stderr, so the sink is silenced (or redirected to a file
|
|
81
|
+
// under CLCO_DEBUG) to keep the chat view clean.
|
|
82
|
+
let logSink: ((line: string) => void) | null = (line) => console.log(line)
|
|
83
|
+
|
|
84
|
+
export function setAdapterLogSink(
|
|
85
|
+
sink: ((line: string) => void) | null,
|
|
86
|
+
): void {
|
|
87
|
+
logSink = sink
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function logLine(line: string): void {
|
|
91
|
+
logSink?.(line)
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
function debug(...args: unknown[]): void {
|
|
95
|
+
if (process.env.CLCO_DEBUG) console.error("[clco:debug]", ...args)
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function anthropicError(status: number, message: string): Response {
|
|
99
|
+
const type =
|
|
100
|
+
status === 401 || status === 403
|
|
101
|
+
? "authentication_error"
|
|
102
|
+
: status === 429
|
|
103
|
+
? "rate_limit_error"
|
|
104
|
+
: status === 404
|
|
105
|
+
? "not_found_error"
|
|
106
|
+
: status === 400 || status === 402 || status === 413 || status === 422
|
|
107
|
+
? "invalid_request_error" // terminal — Claude Code must not retry
|
|
108
|
+
: "api_error"
|
|
109
|
+
return Response.json(
|
|
110
|
+
{ type: "error", error: { type, message } },
|
|
111
|
+
{ status },
|
|
112
|
+
)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
// Turn a raw upstream error body into an actionable message: known error
|
|
116
|
+
// codes get guidance; the raw body is debug-gated to avoid leaking upstream
|
|
117
|
+
// internals into terminals and pasted bug reports.
|
|
118
|
+
function friendlyUpstreamError(bodyText: string): string {
|
|
119
|
+
try {
|
|
120
|
+
const parsed = JSON.parse(bodyText) as {
|
|
121
|
+
error?: { message?: string; code?: string | number }
|
|
122
|
+
}
|
|
123
|
+
const message = parsed?.error?.message
|
|
124
|
+
if (
|
|
125
|
+
parsed?.error?.code === "quota_exceeded" ||
|
|
126
|
+
(message && /quota/i.test(message))
|
|
127
|
+
) {
|
|
128
|
+
return QUOTA_GUIDANCE
|
|
129
|
+
}
|
|
130
|
+
if (message) return message
|
|
131
|
+
} catch {
|
|
132
|
+
// not JSON — fall through
|
|
133
|
+
}
|
|
134
|
+
if (process.env.CLCO_DEBUG) {
|
|
135
|
+
return `upstream error: ${bodyText.slice(0, 500)}`
|
|
136
|
+
}
|
|
137
|
+
return "upstream error (set CLCO_DEBUG=1 for the raw body)"
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
function detectVision(payload: OpenAIRequest): boolean {
|
|
141
|
+
return payload.messages.some((m) =>
|
|
142
|
+
typeof m.content !== "string" && Array.isArray(m.content)
|
|
143
|
+
? m.content.some((p) => p.type === "image_url")
|
|
144
|
+
: false,
|
|
145
|
+
)
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
const routes = new DialectRouter()
|
|
149
|
+
|
|
150
|
+
type ChatResult =
|
|
151
|
+
| { ok: true; res: Response; dialect: Dialect }
|
|
152
|
+
| { ok: false; status: number; message: string }
|
|
153
|
+
|
|
154
|
+
/** The request as Claude Code sent it, for the native passthrough path. */
|
|
155
|
+
interface NativeRequest {
|
|
156
|
+
body: string
|
|
157
|
+
anthropicVersion?: string
|
|
158
|
+
anthropicBeta?: string
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
async function copilotChat(
|
|
162
|
+
payload: OpenAIRequest,
|
|
163
|
+
anthropic: AnthropicRequest,
|
|
164
|
+
upstreamBase: string,
|
|
165
|
+
mockToken: boolean,
|
|
166
|
+
native?: NativeRequest,
|
|
167
|
+
allowedEfforts?: string[] | null,
|
|
168
|
+
): Promise<ChatResult> {
|
|
169
|
+
const chatBody = JSON.stringify(payload)
|
|
170
|
+
const agentInitiated = payload.messages.some(
|
|
171
|
+
(m) => m.role === "assistant" || m.role === "tool",
|
|
172
|
+
)
|
|
173
|
+
const vision = detectVision(payload)
|
|
174
|
+
|
|
175
|
+
const info = modelInfo(payload.model)
|
|
176
|
+
let dialect = routes.select(payload.model, info, native !== undefined && !process.env.CLCO_NO_PASSTHROUGH)
|
|
177
|
+
let attempt = 0
|
|
178
|
+
while (attempt < 2) {
|
|
179
|
+
const token = mockToken ? "mock" : await getCopilotToken(attempt > 0)
|
|
180
|
+
const endpoint = dialect === "native"
|
|
181
|
+
? "/v1/messages"
|
|
182
|
+
: dialect === "responses"
|
|
183
|
+
? "/responses"
|
|
184
|
+
: "/chat/completions"
|
|
185
|
+
const headers = copilotRequestHeaders(token, {
|
|
186
|
+
agentInitiated,
|
|
187
|
+
vision,
|
|
188
|
+
accept: payload.stream ? "text/event-stream" : "application/json",
|
|
189
|
+
})
|
|
190
|
+
if (dialect === "native" && native) {
|
|
191
|
+
// Forward the protocol headers verbatim; the upstream needs them to
|
|
192
|
+
// honour the same betas Claude Code asked for.
|
|
193
|
+
if (native.anthropicVersion) headers["anthropic-version"] = native.anthropicVersion
|
|
194
|
+
const beta = sanitizeBeta(native.anthropicBeta, payload.model)
|
|
195
|
+
if (beta) headers["anthropic-beta"] = beta
|
|
196
|
+
}
|
|
197
|
+
const res = await copilotFetch(`${upstreamBase}${endpoint}`, {
|
|
198
|
+
method: "POST",
|
|
199
|
+
headers,
|
|
200
|
+
body: dialect === "native"
|
|
201
|
+
? native!.body
|
|
202
|
+
: dialect === "responses"
|
|
203
|
+
? JSON.stringify(toResponsesRequest(anthropic, allowedEfforts))
|
|
204
|
+
: chatBody,
|
|
205
|
+
})
|
|
206
|
+
|
|
207
|
+
if ((res.status === 401 || res.status === 403) && attempt === 0) {
|
|
208
|
+
// Free the parked socket before retrying with a fresh token.
|
|
209
|
+
await res.body?.cancel().catch(() => {})
|
|
210
|
+
invalidateCopilotToken()
|
|
211
|
+
attempt++
|
|
212
|
+
continue
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
if (res.ok) {
|
|
216
|
+
return {
|
|
217
|
+
ok: true,
|
|
218
|
+
res,
|
|
219
|
+
dialect,
|
|
220
|
+
}
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
const text = await res.text()
|
|
224
|
+
// A shape rejection falls back for this request. Only model/endpoint
|
|
225
|
+
// evidence changes future routing. Quota, auth and rate-limit failures
|
|
226
|
+
// are not shape problems, so they surface
|
|
227
|
+
// as-is rather than burning a second upstream call.
|
|
228
|
+
if (
|
|
229
|
+
dialect === "native" &&
|
|
230
|
+
[400, 404, 415, 422].includes(res.status)
|
|
231
|
+
) {
|
|
232
|
+
debug("native /v1/messages rejected:", res.status, text.slice(0, 300))
|
|
233
|
+
const remembered = routes.rejectNative(payload.model, res.status, text)
|
|
234
|
+
dialect = routes.translated(payload.model, info)
|
|
235
|
+
logLine(`[${timestamp()}] -> ${payload.model}: native rejected (${res.status}); ${dialect} fallback ${remembered ? "remembered for this process" : "for this request only"}`)
|
|
236
|
+
continue
|
|
237
|
+
}
|
|
238
|
+
// Copilot serves some models only via the Responses API; switch and
|
|
239
|
+
// retry within the same attempt budget.
|
|
240
|
+
if (
|
|
241
|
+
res.status === 400 &&
|
|
242
|
+
dialect === "chat" &&
|
|
243
|
+
/chat\/completions endpoint/i.test(text)
|
|
244
|
+
) {
|
|
245
|
+
routes.requireResponses(payload.model)
|
|
246
|
+
dialect = "responses"
|
|
247
|
+
continue
|
|
248
|
+
}
|
|
249
|
+
return { ok: false, status: res.status, message: friendlyUpstreamError(text) }
|
|
250
|
+
}
|
|
251
|
+
return { ok: false, status: 502, message: "upstream retry exhausted" }
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// Parse the upstream SSE stream into Anthropic SSE events. The "responses"
|
|
255
|
+
// dialect first maps Responses events onto OpenAI-style chunks so the same
|
|
256
|
+
// StreamTranslator renders both.
|
|
257
|
+
function sseResponse(
|
|
258
|
+
upstream: Response,
|
|
259
|
+
model: string,
|
|
260
|
+
dialect: "chat" | "responses" = "chat",
|
|
261
|
+
): Response {
|
|
262
|
+
const translator = new StreamTranslator(model)
|
|
263
|
+
const eventAdapter =
|
|
264
|
+
dialect === "responses" ? new ResponsesEventAdapter() : null
|
|
265
|
+
// Held in the closure so client disconnects can cancel the upstream read
|
|
266
|
+
// (upstream.body.cancel() would fail: the stream is locked by the reader).
|
|
267
|
+
let reader: ReadableStreamDefaultReader<Uint8Array> | null = null
|
|
268
|
+
|
|
269
|
+
const stream = new ReadableStream<Uint8Array>({
|
|
270
|
+
async start(controller) {
|
|
271
|
+
const enc = new TextEncoder()
|
|
272
|
+
let closed = false
|
|
273
|
+
const send = (event: StreamEventData) => {
|
|
274
|
+
if (closed) return
|
|
275
|
+
try {
|
|
276
|
+
controller.enqueue(
|
|
277
|
+
enc.encode(
|
|
278
|
+
`event: ${event.event}\ndata: ${JSON.stringify(event.data)}\n\n`,
|
|
279
|
+
),
|
|
280
|
+
)
|
|
281
|
+
} catch {
|
|
282
|
+
closed = true
|
|
283
|
+
}
|
|
284
|
+
}
|
|
285
|
+
const stop = () => {
|
|
286
|
+
closed = true
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
// Flush headers immediately and keep the connection warm while the
|
|
290
|
+
// upstream thinks; Claude Code aborts a silent stream after 300s.
|
|
291
|
+
send({ event: "ping", data: { type: "ping" } })
|
|
292
|
+
const pinger = setInterval(
|
|
293
|
+
() => send({ event: "ping", data: { type: "ping" } }),
|
|
294
|
+
15000,
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
try {
|
|
298
|
+
reader = upstream.body!.getReader()
|
|
299
|
+
const decoder = new TextDecoder()
|
|
300
|
+
let buffer = ""
|
|
301
|
+
let errored = false
|
|
302
|
+
readLoop: while (true) {
|
|
303
|
+
const { done, value } = await reader.read()
|
|
304
|
+
if (closed) break
|
|
305
|
+
buffer += done ? decoder.decode() : decoder.decode(value, { stream: true })
|
|
306
|
+
if (done && buffer) buffer += "\n"
|
|
307
|
+
let idx: number
|
|
308
|
+
while ((idx = buffer.indexOf("\n")) !== -1) {
|
|
309
|
+
const line = buffer.slice(0, idx).replace(/\r$/, "")
|
|
310
|
+
buffer = buffer.slice(idx + 1)
|
|
311
|
+
if (!line.startsWith("data:")) continue
|
|
312
|
+
const data = line.slice(5).trim()
|
|
313
|
+
if (!data || data === "[DONE]") continue
|
|
314
|
+
|
|
315
|
+
let parsed: unknown
|
|
316
|
+
try {
|
|
317
|
+
parsed = JSON.parse(data)
|
|
318
|
+
} catch {
|
|
319
|
+
debug("unparsable upstream SSE line:", data.slice(0, 200))
|
|
320
|
+
continue
|
|
321
|
+
}
|
|
322
|
+
let obj = parsed as OpenAIResponse
|
|
323
|
+
if (eventAdapter) {
|
|
324
|
+
const chunk = eventAdapter.pushEvent(parsed as Record<string, unknown>)
|
|
325
|
+
if (!chunk) continue
|
|
326
|
+
obj = chunk
|
|
327
|
+
}
|
|
328
|
+
if (obj && typeof obj === "object" && obj.error) {
|
|
329
|
+
// In-band upstream error (quota, entitlement, moderation).
|
|
330
|
+
// The error event is terminal — no fake message termination
|
|
331
|
+
// after it, so the client cannot mistake this for success.
|
|
332
|
+
debug("upstream error chunk:", data.slice(0, 500))
|
|
333
|
+
errored = true
|
|
334
|
+
const message = (obj.error.message as string | undefined) ?? ""
|
|
335
|
+
const isQuota =
|
|
336
|
+
(obj.error.code as string | undefined) === "quota_exceeded" ||
|
|
337
|
+
/quota/i.test(message)
|
|
338
|
+
send({
|
|
339
|
+
event: "error",
|
|
340
|
+
data: {
|
|
341
|
+
type: "error",
|
|
342
|
+
error: {
|
|
343
|
+
type: isQuota ? "invalid_request_error" : "api_error",
|
|
344
|
+
message: isQuota
|
|
345
|
+
? QUOTA_GUIDANCE
|
|
346
|
+
: message || `upstream error chunk (see CLCO_DEBUG)`,
|
|
347
|
+
},
|
|
348
|
+
},
|
|
349
|
+
})
|
|
350
|
+
break readLoop
|
|
351
|
+
}
|
|
352
|
+
if (!Array.isArray(obj?.choices)) {
|
|
353
|
+
debug("non-conforming upstream chunk:", data.slice(0, 200))
|
|
354
|
+
continue
|
|
355
|
+
}
|
|
356
|
+
for (const ev of translator.pushChunk(obj)) send(ev)
|
|
357
|
+
if (closed) break readLoop
|
|
358
|
+
}
|
|
359
|
+
if (done) break
|
|
360
|
+
}
|
|
361
|
+
if (!errored) {
|
|
362
|
+
for (const ev of translator.finish()) send(ev)
|
|
363
|
+
}
|
|
364
|
+
} catch (err) {
|
|
365
|
+
send({
|
|
366
|
+
event: "error",
|
|
367
|
+
data: {
|
|
368
|
+
type: "error",
|
|
369
|
+
error: { type: "api_error", message: String(err) },
|
|
370
|
+
},
|
|
371
|
+
})
|
|
372
|
+
} finally {
|
|
373
|
+
clearInterval(pinger)
|
|
374
|
+
stop()
|
|
375
|
+
try {
|
|
376
|
+
controller.close()
|
|
377
|
+
} catch {
|
|
378
|
+
// already closed by client cancellation
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
},
|
|
382
|
+
cancel() {
|
|
383
|
+
reader?.cancel().catch(() => {})
|
|
384
|
+
},
|
|
385
|
+
})
|
|
386
|
+
return new Response(stream, {
|
|
387
|
+
headers: {
|
|
388
|
+
"content-type": "text/event-stream",
|
|
389
|
+
"cache-control": "no-cache",
|
|
390
|
+
"x-accel-buffering": "no",
|
|
391
|
+
},
|
|
392
|
+
})
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
async function handleMessages(
|
|
396
|
+
req: Request,
|
|
397
|
+
upstreamBase: string,
|
|
398
|
+
mockToken: boolean,
|
|
399
|
+
): Promise<Response> {
|
|
400
|
+
const started = Date.now()
|
|
401
|
+
// Read the body as text so the native path can forward it essentially
|
|
402
|
+
// untouched; the parse is only for inspection and the translation paths.
|
|
403
|
+
let rawBody: string
|
|
404
|
+
let payload: AnthropicRequest
|
|
405
|
+
try {
|
|
406
|
+
rawBody = await req.text()
|
|
407
|
+
payload = JSON.parse(rawBody) as AnthropicRequest
|
|
408
|
+
} catch {
|
|
409
|
+
return anthropicError(400, "invalid JSON body")
|
|
410
|
+
}
|
|
411
|
+
const stream = payload.stream === true
|
|
412
|
+
logLine(
|
|
413
|
+
`[${timestamp()}] POST /v1/messages model=${payload.model} stream=${stream}`,
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
const info = modelInfo(normalizeModel(payload.model))
|
|
417
|
+
const allowedEfforts = info?.efforts ?? null
|
|
418
|
+
// The local estimate cannot establish overflow across different tokenizers.
|
|
419
|
+
// Keep requests flowing; the selected upstream is authoritative about fit.
|
|
420
|
+
const limit = info?.maxPromptTokens ?? info?.maxContextTokens
|
|
421
|
+
if (limit) {
|
|
422
|
+
const estimate = estimateTokens(payload)
|
|
423
|
+
if (estimate > limit * TOKEN_WARNING_RATIO) {
|
|
424
|
+
logLine(`[${timestamp()}] -> warning: estimated input ~${estimate}, ${payload.model} limit ${limit}; forwarding to upstream`)
|
|
425
|
+
}
|
|
426
|
+
}
|
|
427
|
+
const upstreamPayload = translateRequest(payload, allowedEfforts)
|
|
428
|
+
// Only the model name is rewritten for the native path; every other field
|
|
429
|
+
// (thinking, cache_control, output_config, tools) rides through as sent.
|
|
430
|
+
const nativeBody =
|
|
431
|
+
payload.model === upstreamPayload.model
|
|
432
|
+
? rawBody
|
|
433
|
+
: JSON.stringify({ ...payload, model: upstreamPayload.model })
|
|
434
|
+
let result: Awaited<ReturnType<typeof copilotChat>>
|
|
435
|
+
try {
|
|
436
|
+
result = await copilotChat(
|
|
437
|
+
upstreamPayload,
|
|
438
|
+
payload,
|
|
439
|
+
upstreamBase,
|
|
440
|
+
mockToken,
|
|
441
|
+
{
|
|
442
|
+
body: nativeBody,
|
|
443
|
+
anthropicVersion: req.headers.get("anthropic-version") ?? undefined,
|
|
444
|
+
anthropicBeta: req.headers.get("anthropic-beta") ?? undefined,
|
|
445
|
+
},
|
|
446
|
+
allowedEfforts,
|
|
447
|
+
)
|
|
448
|
+
} catch (err) {
|
|
449
|
+
// This is the only place a TLS failure can reach the user: it renders as
|
|
450
|
+
// an API error inside claude's UI, so the remedy has to travel with it.
|
|
451
|
+
const detail = String(err)
|
|
452
|
+
logLine(`[${timestamp()}] -> upstream request failed: ${detail}`)
|
|
453
|
+
return anthropicError(
|
|
454
|
+
502,
|
|
455
|
+
`upstream request failed: ${detail}` +
|
|
456
|
+
(isTlsTrustError(err) ? tlsHint() : ""),
|
|
457
|
+
)
|
|
458
|
+
}
|
|
459
|
+
if (!result.ok) {
|
|
460
|
+
logLine(`[${timestamp()}] -> HTTP ${result.status} (${Date.now() - started}ms)`)
|
|
461
|
+
return anthropicError(result.status, result.message)
|
|
462
|
+
}
|
|
463
|
+
const res = result.res
|
|
464
|
+
|
|
465
|
+
if (result.dialect === "native") {
|
|
466
|
+
// Already Anthropic-shaped: relay verbatim, streaming included.
|
|
467
|
+
logLine(
|
|
468
|
+
`[${timestamp()}] -> ${res.status} native${stream ? " streaming" : ""} (${Date.now() - started}ms)`,
|
|
469
|
+
)
|
|
470
|
+
const headers: Record<string, string> = {
|
|
471
|
+
"content-type": res.headers.get("content-type") ?? "application/json",
|
|
472
|
+
}
|
|
473
|
+
if (stream) {
|
|
474
|
+
headers["cache-control"] = "no-cache"
|
|
475
|
+
headers["x-accel-buffering"] = "no"
|
|
476
|
+
}
|
|
477
|
+
return new Response(res.body, { status: res.status, headers })
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
if (!stream) {
|
|
481
|
+
let data: OpenAIResponse
|
|
482
|
+
try {
|
|
483
|
+
const raw = (await res.json()) as unknown
|
|
484
|
+
if (result.dialect === "responses") {
|
|
485
|
+
const body = raw as {
|
|
486
|
+
error?: unknown
|
|
487
|
+
status?: string
|
|
488
|
+
}
|
|
489
|
+
// An in-band failure body must not convert into an empty
|
|
490
|
+
// "successful" assistant message.
|
|
491
|
+
if (body?.error || body?.status === "failed") {
|
|
492
|
+
logLine(
|
|
493
|
+
`[${timestamp()}] -> responses error body (${Date.now() - started}ms)`,
|
|
494
|
+
)
|
|
495
|
+
return anthropicError(
|
|
496
|
+
502,
|
|
497
|
+
`upstream error: ${JSON.stringify(body.error ?? body.status).slice(0, 300)}`,
|
|
498
|
+
)
|
|
499
|
+
}
|
|
500
|
+
data = responsesToOpenAIResponse(raw as Record<string, unknown>)
|
|
501
|
+
} else {
|
|
502
|
+
data = raw as OpenAIResponse
|
|
503
|
+
}
|
|
504
|
+
} catch {
|
|
505
|
+
// A 200 with a non-JSON body (proxy page, HTML error) is an upstream
|
|
506
|
+
// failure, not a client error — must be terminal 502, not a retryable
|
|
507
|
+
// 500.
|
|
508
|
+
return anthropicError(502, "upstream returned a non-JSON body")
|
|
509
|
+
}
|
|
510
|
+
if (data && typeof data === "object" && data.error) {
|
|
511
|
+
logLine(`[${timestamp()}] -> upstream error body (${Date.now() - started}ms)`)
|
|
512
|
+
return anthropicError(
|
|
513
|
+
502,
|
|
514
|
+
`upstream error: ${JSON.stringify(data.error).slice(0, 500)}`,
|
|
515
|
+
)
|
|
516
|
+
}
|
|
517
|
+
if (!Array.isArray(data?.choices)) {
|
|
518
|
+
return anthropicError(502, "upstream returned a non-conforming response")
|
|
519
|
+
}
|
|
520
|
+
logLine(`[${timestamp()}] -> 200 (${Date.now() - started}ms)`)
|
|
521
|
+
return Response.json(translateResponse(data))
|
|
522
|
+
}
|
|
523
|
+
|
|
524
|
+
// A 200 that is not actually an SSE stream is an error body, not a message.
|
|
525
|
+
const contentType = res.headers.get("content-type") ?? ""
|
|
526
|
+
if (!contentType.includes("text/event-stream")) {
|
|
527
|
+
const detail = (await res.text()).slice(0, 500)
|
|
528
|
+
logLine(
|
|
529
|
+
`[${timestamp()}] -> 200 non-SSE body: ${contentType} (${Date.now() - started}ms)`,
|
|
530
|
+
)
|
|
531
|
+
return anthropicError(502, `upstream returned non-streaming body: ${detail}`)
|
|
532
|
+
}
|
|
533
|
+
|
|
534
|
+
logLine(`[${timestamp()}] -> 200 streaming (${result.dialect})`)
|
|
535
|
+
return sseResponse(res, upstreamPayload.model, result.dialect)
|
|
536
|
+
}
|
|
537
|
+
|
|
538
|
+
// Bun derives req.url from the client-supplied Host header, so comparing
|
|
539
|
+
// the Host header against url.host is a tautology. Compare against the
|
|
540
|
+
// origin we actually bound instead.
|
|
541
|
+
function authorize(req: Request, selfHost: string): Response | null {
|
|
542
|
+
if (req.headers.get("host") !== selfHost) {
|
|
543
|
+
return anthropicError(403, "host header mismatch")
|
|
544
|
+
}
|
|
545
|
+
if (req.headers.get("authorization") !== `Bearer ${LOCAL_TOKEN}`) {
|
|
546
|
+
return anthropicError(401, "missing or invalid adapter token")
|
|
547
|
+
}
|
|
548
|
+
return null
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
async function handle(
|
|
552
|
+
req: Request,
|
|
553
|
+
upstreamBase: string,
|
|
554
|
+
mockToken: boolean,
|
|
555
|
+
selfHost: string,
|
|
556
|
+
): Promise<Response> {
|
|
557
|
+
const url = new URL(req.url)
|
|
558
|
+
const path = url.pathname
|
|
559
|
+
|
|
560
|
+
if (path === "/api/hello") {
|
|
561
|
+
// Claude Code's connection-warming probe; auth keeps it from being a
|
|
562
|
+
// port-scan oracle. Rejecting it is harmless by design.
|
|
563
|
+
return authorize(req, selfHost) ?? new Response(null, { status: 200 })
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
if (path !== "/v1/messages" && path !== "/v1/messages/count_tokens") {
|
|
567
|
+
if (req.method === "GET" && path === "/v1/models") {
|
|
568
|
+
const denied = authorize(req, selfHost)
|
|
569
|
+
if (denied) return denied
|
|
570
|
+
// Gateway model discovery (CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY):
|
|
571
|
+
// serve the list captured at startup — Claude Code aborts discovery
|
|
572
|
+
// after 3 seconds, so a live upstream fetch would silently fail.
|
|
573
|
+
logLine(`[${timestamp()}] GET /v1/models`)
|
|
574
|
+
const entries = upstreamModels()
|
|
575
|
+
if (entries.length === 0) {
|
|
576
|
+
return anthropicError(502, "model list unavailable (discovery failed at startup)")
|
|
577
|
+
}
|
|
578
|
+
return Response.json({
|
|
579
|
+
data: entries.map((e) => ({
|
|
580
|
+
type: "model",
|
|
581
|
+
id: e.id,
|
|
582
|
+
display_name: e.name,
|
|
583
|
+
})),
|
|
584
|
+
has_more: false,
|
|
585
|
+
})
|
|
586
|
+
}
|
|
587
|
+
return anthropicError(404, `not found: ${req.method} ${path}`)
|
|
588
|
+
}
|
|
589
|
+
if (req.method !== "POST") {
|
|
590
|
+
return anthropicError(404, `not found: ${req.method} ${path}`)
|
|
591
|
+
}
|
|
592
|
+
const denied = authorize(req, selfHost)
|
|
593
|
+
if (denied) return denied
|
|
594
|
+
|
|
595
|
+
const contentType = req.headers.get("content-type") ?? ""
|
|
596
|
+
if (!contentType.includes("application/json")) {
|
|
597
|
+
return anthropicError(400, "content-type must be application/json")
|
|
598
|
+
}
|
|
599
|
+
const contentLength = Number(req.headers.get("content-length") ?? "0")
|
|
600
|
+
if (contentLength > MAX_BODY_BYTES) {
|
|
601
|
+
return anthropicError(413, "request body too large")
|
|
602
|
+
}
|
|
603
|
+
|
|
604
|
+
if (path === "/v1/messages/count_tokens") {
|
|
605
|
+
try {
|
|
606
|
+
const payload = (await req.json()) as AnthropicRequest
|
|
607
|
+
logLine(`[${timestamp()}] POST /v1/messages/count_tokens`)
|
|
608
|
+
return Response.json({ input_tokens: estimateTokens(payload) })
|
|
609
|
+
} catch {
|
|
610
|
+
return anthropicError(400, "invalid JSON body")
|
|
611
|
+
}
|
|
612
|
+
}
|
|
613
|
+
|
|
614
|
+
return handleMessages(req, upstreamBase, mockToken)
|
|
615
|
+
}
|
|
616
|
+
|
|
617
|
+
export async function startServer(
|
|
618
|
+
opts: ServerOptions = {},
|
|
619
|
+
): Promise<ServerHandle> {
|
|
620
|
+
const upstreamBase = (opts.upstream ?? copilotBaseUrl()).replace(/\/$/, "")
|
|
621
|
+
const mockToken = opts.upstream !== undefined || isMockMode()
|
|
622
|
+
// Filled in once the port is bound; the fetch closure only runs afterwards.
|
|
623
|
+
let selfHost = ""
|
|
624
|
+
const server = Bun.serve({
|
|
625
|
+
hostname: "127.0.0.1",
|
|
626
|
+
port: opts.port ?? 0,
|
|
627
|
+
fetch: (req) =>
|
|
628
|
+
handle(req, upstreamBase, mockToken, selfHost).catch((err) =>
|
|
629
|
+
anthropicError(500, String(err)),
|
|
630
|
+
),
|
|
631
|
+
})
|
|
632
|
+
selfHost = `127.0.0.1:${server.port ?? 0}`
|
|
633
|
+
return {
|
|
634
|
+
url: server.url.toString().replace(/\/$/, ""),
|
|
635
|
+
port: server.port ?? 0,
|
|
636
|
+
stop: () => server.stop(true),
|
|
637
|
+
}
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
// Imported lazily by name to keep the module graph simple.
|
|
641
|
+
|