@semanticist14/clco 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/server.ts ADDED
@@ -0,0 +1,641 @@
1
+ // Local Anthropic-compatible adapter server. Claude Code talks to this; it
2
+ // translates and forwards to GitHub Copilot.
3
+
4
+ import { StreamTranslator } from "./stream"
5
+ import { estimateTokens, ONE_MILLION_TOKENS, TOKEN_WARNING_RATIO } from "./tokens"
6
+ import { DialectRouter, type Dialect } from "./route"
7
+ import type { AnthropicRequest, OpenAIRequest, OpenAIResponse, StreamEventData } from "./wire"
8
+ import {
9
+ copilotBaseUrl,
10
+ copilotFetch,
11
+ copilotRequestHeaders,
12
+ isMockMode,
13
+ } from "./api"
14
+ import { isTlsTrustError, tlsHint } from "./tls"
15
+ import {
16
+ getCopilotToken,
17
+ invalidateCopilotToken,
18
+ modelInfo,
19
+ upstreamModels,
20
+ } from "./token"
21
+ import {
22
+ normalizeModel,
23
+ translateRequest,
24
+ translateResponse,
25
+ } from "./translate"
26
+ import {
27
+ ResponsesEventAdapter,
28
+ responsesToOpenAIResponse,
29
+ toResponsesRequest,
30
+ } from "./responses"
31
+
32
+ export interface ServerHandle {
33
+ url: string
34
+ port: number
35
+ stop(): void
36
+ }
37
+
38
+ export interface ServerOptions {
39
+ port?: number
40
+ /** Override the Copilot base URL (tests inject a mock directly).
41
+ * Implies mock-token mode: no GitHub exchange is attempted. */
42
+ upstream?: string
43
+ }
44
+
45
+ // Token injected into claude via ANTHROPIC_AUTH_TOKEN; requests without it
46
+ // are rejected so stray local processes (or web pages doing no-cors POSTs)
47
+ // cannot burn Copilot quota through the adapter.
48
+ const LOCAL_TOKEN = "clco-local"
49
+
50
+ const MAX_BODY_BYTES = 64 * 1024 * 1024
51
+
52
+ const QUOTA_GUIDANCE =
53
+ "Copilot monthly premium quota exhausted, or no subscription - check github.com/settings/copilot"
54
+
55
+ // clco appends [1m] to picker rows whose real window exceeds the default
56
+ // ceiling, which makes Claude Code request the 1M-context beta. Forward that
57
+ // only to a model that genuinely has the window — Copilot rejects the header
58
+ // otherwise, and the [1m] suffix is the only per-row window channel the
59
+ // picker schema offers, so we cannot simply stop using it.
60
+ export function sanitizeBeta(
61
+ beta: string | undefined,
62
+ model: string,
63
+ ): string | undefined {
64
+ if (!beta) return beta
65
+ const window =
66
+ modelInfo(model)?.maxPromptTokens ?? modelInfo(model)?.maxContextTokens
67
+ if ((window ?? 0) >= ONE_MILLION_TOKENS) return beta
68
+ const kept = beta
69
+ .split(",")
70
+ .map((v) => v.trim())
71
+ .filter((v) => v && !v.startsWith("context-1m"))
72
+ return kept.length > 0 ? kept.join(",") : undefined
73
+ }
74
+
75
+ function timestamp(): string {
76
+ return new Date().toTimeString().slice(0, 8)
77
+ }
78
+
79
+ // Request logs go to the terminal in serve mode; in run mode claude's TUI
80
+ // owns stdout/stderr, so the sink is silenced (or redirected to a file
81
+ // under CLCO_DEBUG) to keep the chat view clean.
82
+ let logSink: ((line: string) => void) | null = (line) => console.log(line)
83
+
84
+ export function setAdapterLogSink(
85
+ sink: ((line: string) => void) | null,
86
+ ): void {
87
+ logSink = sink
88
+ }
89
+
90
+ function logLine(line: string): void {
91
+ logSink?.(line)
92
+ }
93
+
94
+ function debug(...args: unknown[]): void {
95
+ if (process.env.CLCO_DEBUG) console.error("[clco:debug]", ...args)
96
+ }
97
+
98
+ function anthropicError(status: number, message: string): Response {
99
+ const type =
100
+ status === 401 || status === 403
101
+ ? "authentication_error"
102
+ : status === 429
103
+ ? "rate_limit_error"
104
+ : status === 404
105
+ ? "not_found_error"
106
+ : status === 400 || status === 402 || status === 413 || status === 422
107
+ ? "invalid_request_error" // terminal — Claude Code must not retry
108
+ : "api_error"
109
+ return Response.json(
110
+ { type: "error", error: { type, message } },
111
+ { status },
112
+ )
113
+ }
114
+
115
+ // Turn a raw upstream error body into an actionable message: known error
116
+ // codes get guidance; the raw body is debug-gated to avoid leaking upstream
117
+ // internals into terminals and pasted bug reports.
118
+ function friendlyUpstreamError(bodyText: string): string {
119
+ try {
120
+ const parsed = JSON.parse(bodyText) as {
121
+ error?: { message?: string; code?: string | number }
122
+ }
123
+ const message = parsed?.error?.message
124
+ if (
125
+ parsed?.error?.code === "quota_exceeded" ||
126
+ (message && /quota/i.test(message))
127
+ ) {
128
+ return QUOTA_GUIDANCE
129
+ }
130
+ if (message) return message
131
+ } catch {
132
+ // not JSON — fall through
133
+ }
134
+ if (process.env.CLCO_DEBUG) {
135
+ return `upstream error: ${bodyText.slice(0, 500)}`
136
+ }
137
+ return "upstream error (set CLCO_DEBUG=1 for the raw body)"
138
+ }
139
+
140
+ function detectVision(payload: OpenAIRequest): boolean {
141
+ return payload.messages.some((m) =>
142
+ typeof m.content !== "string" && Array.isArray(m.content)
143
+ ? m.content.some((p) => p.type === "image_url")
144
+ : false,
145
+ )
146
+ }
147
+
148
+ const routes = new DialectRouter()
149
+
150
+ type ChatResult =
151
+ | { ok: true; res: Response; dialect: Dialect }
152
+ | { ok: false; status: number; message: string }
153
+
154
+ /** The request as Claude Code sent it, for the native passthrough path. */
155
+ interface NativeRequest {
156
+ body: string
157
+ anthropicVersion?: string
158
+ anthropicBeta?: string
159
+ }
160
+
161
+ async function copilotChat(
162
+ payload: OpenAIRequest,
163
+ anthropic: AnthropicRequest,
164
+ upstreamBase: string,
165
+ mockToken: boolean,
166
+ native?: NativeRequest,
167
+ allowedEfforts?: string[] | null,
168
+ ): Promise<ChatResult> {
169
+ const chatBody = JSON.stringify(payload)
170
+ const agentInitiated = payload.messages.some(
171
+ (m) => m.role === "assistant" || m.role === "tool",
172
+ )
173
+ const vision = detectVision(payload)
174
+
175
+ const info = modelInfo(payload.model)
176
+ let dialect = routes.select(payload.model, info, native !== undefined && !process.env.CLCO_NO_PASSTHROUGH)
177
+ let attempt = 0
178
+ while (attempt < 2) {
179
+ const token = mockToken ? "mock" : await getCopilotToken(attempt > 0)
180
+ const endpoint = dialect === "native"
181
+ ? "/v1/messages"
182
+ : dialect === "responses"
183
+ ? "/responses"
184
+ : "/chat/completions"
185
+ const headers = copilotRequestHeaders(token, {
186
+ agentInitiated,
187
+ vision,
188
+ accept: payload.stream ? "text/event-stream" : "application/json",
189
+ })
190
+ if (dialect === "native" && native) {
191
+ // Forward the protocol headers verbatim; the upstream needs them to
192
+ // honour the same betas Claude Code asked for.
193
+ if (native.anthropicVersion) headers["anthropic-version"] = native.anthropicVersion
194
+ const beta = sanitizeBeta(native.anthropicBeta, payload.model)
195
+ if (beta) headers["anthropic-beta"] = beta
196
+ }
197
+ const res = await copilotFetch(`${upstreamBase}${endpoint}`, {
198
+ method: "POST",
199
+ headers,
200
+ body: dialect === "native"
201
+ ? native!.body
202
+ : dialect === "responses"
203
+ ? JSON.stringify(toResponsesRequest(anthropic, allowedEfforts))
204
+ : chatBody,
205
+ })
206
+
207
+ if ((res.status === 401 || res.status === 403) && attempt === 0) {
208
+ // Free the parked socket before retrying with a fresh token.
209
+ await res.body?.cancel().catch(() => {})
210
+ invalidateCopilotToken()
211
+ attempt++
212
+ continue
213
+ }
214
+
215
+ if (res.ok) {
216
+ return {
217
+ ok: true,
218
+ res,
219
+ dialect,
220
+ }
221
+ }
222
+
223
+ const text = await res.text()
224
+ // A shape rejection falls back for this request. Only model/endpoint
225
+ // evidence changes future routing. Quota, auth and rate-limit failures
226
+ // are not shape problems, so they surface
227
+ // as-is rather than burning a second upstream call.
228
+ if (
229
+ dialect === "native" &&
230
+ [400, 404, 415, 422].includes(res.status)
231
+ ) {
232
+ debug("native /v1/messages rejected:", res.status, text.slice(0, 300))
233
+ const remembered = routes.rejectNative(payload.model, res.status, text)
234
+ dialect = routes.translated(payload.model, info)
235
+ logLine(`[${timestamp()}] -> ${payload.model}: native rejected (${res.status}); ${dialect} fallback ${remembered ? "remembered for this process" : "for this request only"}`)
236
+ continue
237
+ }
238
+ // Copilot serves some models only via the Responses API; switch and
239
+ // retry within the same attempt budget.
240
+ if (
241
+ res.status === 400 &&
242
+ dialect === "chat" &&
243
+ /chat\/completions endpoint/i.test(text)
244
+ ) {
245
+ routes.requireResponses(payload.model)
246
+ dialect = "responses"
247
+ continue
248
+ }
249
+ return { ok: false, status: res.status, message: friendlyUpstreamError(text) }
250
+ }
251
+ return { ok: false, status: 502, message: "upstream retry exhausted" }
252
+ }
253
+
254
+ // Parse the upstream SSE stream into Anthropic SSE events. The "responses"
255
+ // dialect first maps Responses events onto OpenAI-style chunks so the same
256
+ // StreamTranslator renders both.
257
+ function sseResponse(
258
+ upstream: Response,
259
+ model: string,
260
+ dialect: "chat" | "responses" = "chat",
261
+ ): Response {
262
+ const translator = new StreamTranslator(model)
263
+ const eventAdapter =
264
+ dialect === "responses" ? new ResponsesEventAdapter() : null
265
+ // Held in the closure so client disconnects can cancel the upstream read
266
+ // (upstream.body.cancel() would fail: the stream is locked by the reader).
267
+ let reader: ReadableStreamDefaultReader<Uint8Array> | null = null
268
+
269
+ const stream = new ReadableStream<Uint8Array>({
270
+ async start(controller) {
271
+ const enc = new TextEncoder()
272
+ let closed = false
273
+ const send = (event: StreamEventData) => {
274
+ if (closed) return
275
+ try {
276
+ controller.enqueue(
277
+ enc.encode(
278
+ `event: ${event.event}\ndata: ${JSON.stringify(event.data)}\n\n`,
279
+ ),
280
+ )
281
+ } catch {
282
+ closed = true
283
+ }
284
+ }
285
+ const stop = () => {
286
+ closed = true
287
+ }
288
+
289
+ // Flush headers immediately and keep the connection warm while the
290
+ // upstream thinks; Claude Code aborts a silent stream after 300s.
291
+ send({ event: "ping", data: { type: "ping" } })
292
+ const pinger = setInterval(
293
+ () => send({ event: "ping", data: { type: "ping" } }),
294
+ 15000,
295
+ )
296
+
297
+ try {
298
+ reader = upstream.body!.getReader()
299
+ const decoder = new TextDecoder()
300
+ let buffer = ""
301
+ let errored = false
302
+ readLoop: while (true) {
303
+ const { done, value } = await reader.read()
304
+ if (closed) break
305
+ buffer += done ? decoder.decode() : decoder.decode(value, { stream: true })
306
+ if (done && buffer) buffer += "\n"
307
+ let idx: number
308
+ while ((idx = buffer.indexOf("\n")) !== -1) {
309
+ const line = buffer.slice(0, idx).replace(/\r$/, "")
310
+ buffer = buffer.slice(idx + 1)
311
+ if (!line.startsWith("data:")) continue
312
+ const data = line.slice(5).trim()
313
+ if (!data || data === "[DONE]") continue
314
+
315
+ let parsed: unknown
316
+ try {
317
+ parsed = JSON.parse(data)
318
+ } catch {
319
+ debug("unparsable upstream SSE line:", data.slice(0, 200))
320
+ continue
321
+ }
322
+ let obj = parsed as OpenAIResponse
323
+ if (eventAdapter) {
324
+ const chunk = eventAdapter.pushEvent(parsed as Record<string, unknown>)
325
+ if (!chunk) continue
326
+ obj = chunk
327
+ }
328
+ if (obj && typeof obj === "object" && obj.error) {
329
+ // In-band upstream error (quota, entitlement, moderation).
330
+ // The error event is terminal — no fake message termination
331
+ // after it, so the client cannot mistake this for success.
332
+ debug("upstream error chunk:", data.slice(0, 500))
333
+ errored = true
334
+ const message = (obj.error.message as string | undefined) ?? ""
335
+ const isQuota =
336
+ (obj.error.code as string | undefined) === "quota_exceeded" ||
337
+ /quota/i.test(message)
338
+ send({
339
+ event: "error",
340
+ data: {
341
+ type: "error",
342
+ error: {
343
+ type: isQuota ? "invalid_request_error" : "api_error",
344
+ message: isQuota
345
+ ? QUOTA_GUIDANCE
346
+ : message || `upstream error chunk (see CLCO_DEBUG)`,
347
+ },
348
+ },
349
+ })
350
+ break readLoop
351
+ }
352
+ if (!Array.isArray(obj?.choices)) {
353
+ debug("non-conforming upstream chunk:", data.slice(0, 200))
354
+ continue
355
+ }
356
+ for (const ev of translator.pushChunk(obj)) send(ev)
357
+ if (closed) break readLoop
358
+ }
359
+ if (done) break
360
+ }
361
+ if (!errored) {
362
+ for (const ev of translator.finish()) send(ev)
363
+ }
364
+ } catch (err) {
365
+ send({
366
+ event: "error",
367
+ data: {
368
+ type: "error",
369
+ error: { type: "api_error", message: String(err) },
370
+ },
371
+ })
372
+ } finally {
373
+ clearInterval(pinger)
374
+ stop()
375
+ try {
376
+ controller.close()
377
+ } catch {
378
+ // already closed by client cancellation
379
+ }
380
+ }
381
+ },
382
+ cancel() {
383
+ reader?.cancel().catch(() => {})
384
+ },
385
+ })
386
+ return new Response(stream, {
387
+ headers: {
388
+ "content-type": "text/event-stream",
389
+ "cache-control": "no-cache",
390
+ "x-accel-buffering": "no",
391
+ },
392
+ })
393
+ }
394
+
395
+ async function handleMessages(
396
+ req: Request,
397
+ upstreamBase: string,
398
+ mockToken: boolean,
399
+ ): Promise<Response> {
400
+ const started = Date.now()
401
+ // Read the body as text so the native path can forward it essentially
402
+ // untouched; the parse is only for inspection and the translation paths.
403
+ let rawBody: string
404
+ let payload: AnthropicRequest
405
+ try {
406
+ rawBody = await req.text()
407
+ payload = JSON.parse(rawBody) as AnthropicRequest
408
+ } catch {
409
+ return anthropicError(400, "invalid JSON body")
410
+ }
411
+ const stream = payload.stream === true
412
+ logLine(
413
+ `[${timestamp()}] POST /v1/messages model=${payload.model} stream=${stream}`,
414
+ )
415
+
416
+ const info = modelInfo(normalizeModel(payload.model))
417
+ const allowedEfforts = info?.efforts ?? null
418
+ // The local estimate cannot establish overflow across different tokenizers.
419
+ // Keep requests flowing; the selected upstream is authoritative about fit.
420
+ const limit = info?.maxPromptTokens ?? info?.maxContextTokens
421
+ if (limit) {
422
+ const estimate = estimateTokens(payload)
423
+ if (estimate > limit * TOKEN_WARNING_RATIO) {
424
+ logLine(`[${timestamp()}] -> warning: estimated input ~${estimate}, ${payload.model} limit ${limit}; forwarding to upstream`)
425
+ }
426
+ }
427
+ const upstreamPayload = translateRequest(payload, allowedEfforts)
428
+ // Only the model name is rewritten for the native path; every other field
429
+ // (thinking, cache_control, output_config, tools) rides through as sent.
430
+ const nativeBody =
431
+ payload.model === upstreamPayload.model
432
+ ? rawBody
433
+ : JSON.stringify({ ...payload, model: upstreamPayload.model })
434
+ let result: Awaited<ReturnType<typeof copilotChat>>
435
+ try {
436
+ result = await copilotChat(
437
+ upstreamPayload,
438
+ payload,
439
+ upstreamBase,
440
+ mockToken,
441
+ {
442
+ body: nativeBody,
443
+ anthropicVersion: req.headers.get("anthropic-version") ?? undefined,
444
+ anthropicBeta: req.headers.get("anthropic-beta") ?? undefined,
445
+ },
446
+ allowedEfforts,
447
+ )
448
+ } catch (err) {
449
+ // This is the only place a TLS failure can reach the user: it renders as
450
+ // an API error inside claude's UI, so the remedy has to travel with it.
451
+ const detail = String(err)
452
+ logLine(`[${timestamp()}] -> upstream request failed: ${detail}`)
453
+ return anthropicError(
454
+ 502,
455
+ `upstream request failed: ${detail}` +
456
+ (isTlsTrustError(err) ? tlsHint() : ""),
457
+ )
458
+ }
459
+ if (!result.ok) {
460
+ logLine(`[${timestamp()}] -> HTTP ${result.status} (${Date.now() - started}ms)`)
461
+ return anthropicError(result.status, result.message)
462
+ }
463
+ const res = result.res
464
+
465
+ if (result.dialect === "native") {
466
+ // Already Anthropic-shaped: relay verbatim, streaming included.
467
+ logLine(
468
+ `[${timestamp()}] -> ${res.status} native${stream ? " streaming" : ""} (${Date.now() - started}ms)`,
469
+ )
470
+ const headers: Record<string, string> = {
471
+ "content-type": res.headers.get("content-type") ?? "application/json",
472
+ }
473
+ if (stream) {
474
+ headers["cache-control"] = "no-cache"
475
+ headers["x-accel-buffering"] = "no"
476
+ }
477
+ return new Response(res.body, { status: res.status, headers })
478
+ }
479
+
480
+ if (!stream) {
481
+ let data: OpenAIResponse
482
+ try {
483
+ const raw = (await res.json()) as unknown
484
+ if (result.dialect === "responses") {
485
+ const body = raw as {
486
+ error?: unknown
487
+ status?: string
488
+ }
489
+ // An in-band failure body must not convert into an empty
490
+ // "successful" assistant message.
491
+ if (body?.error || body?.status === "failed") {
492
+ logLine(
493
+ `[${timestamp()}] -> responses error body (${Date.now() - started}ms)`,
494
+ )
495
+ return anthropicError(
496
+ 502,
497
+ `upstream error: ${JSON.stringify(body.error ?? body.status).slice(0, 300)}`,
498
+ )
499
+ }
500
+ data = responsesToOpenAIResponse(raw as Record<string, unknown>)
501
+ } else {
502
+ data = raw as OpenAIResponse
503
+ }
504
+ } catch {
505
+ // A 200 with a non-JSON body (proxy page, HTML error) is an upstream
506
+ // failure, not a client error — must be terminal 502, not a retryable
507
+ // 500.
508
+ return anthropicError(502, "upstream returned a non-JSON body")
509
+ }
510
+ if (data && typeof data === "object" && data.error) {
511
+ logLine(`[${timestamp()}] -> upstream error body (${Date.now() - started}ms)`)
512
+ return anthropicError(
513
+ 502,
514
+ `upstream error: ${JSON.stringify(data.error).slice(0, 500)}`,
515
+ )
516
+ }
517
+ if (!Array.isArray(data?.choices)) {
518
+ return anthropicError(502, "upstream returned a non-conforming response")
519
+ }
520
+ logLine(`[${timestamp()}] -> 200 (${Date.now() - started}ms)`)
521
+ return Response.json(translateResponse(data))
522
+ }
523
+
524
+ // A 200 that is not actually an SSE stream is an error body, not a message.
525
+ const contentType = res.headers.get("content-type") ?? ""
526
+ if (!contentType.includes("text/event-stream")) {
527
+ const detail = (await res.text()).slice(0, 500)
528
+ logLine(
529
+ `[${timestamp()}] -> 200 non-SSE body: ${contentType} (${Date.now() - started}ms)`,
530
+ )
531
+ return anthropicError(502, `upstream returned non-streaming body: ${detail}`)
532
+ }
533
+
534
+ logLine(`[${timestamp()}] -> 200 streaming (${result.dialect})`)
535
+ return sseResponse(res, upstreamPayload.model, result.dialect)
536
+ }
537
+
538
+ // Bun derives req.url from the client-supplied Host header, so comparing
539
+ // the Host header against url.host is a tautology. Compare against the
540
+ // origin we actually bound instead.
541
+ function authorize(req: Request, selfHost: string): Response | null {
542
+ if (req.headers.get("host") !== selfHost) {
543
+ return anthropicError(403, "host header mismatch")
544
+ }
545
+ if (req.headers.get("authorization") !== `Bearer ${LOCAL_TOKEN}`) {
546
+ return anthropicError(401, "missing or invalid adapter token")
547
+ }
548
+ return null
549
+ }
550
+
551
+ async function handle(
552
+ req: Request,
553
+ upstreamBase: string,
554
+ mockToken: boolean,
555
+ selfHost: string,
556
+ ): Promise<Response> {
557
+ const url = new URL(req.url)
558
+ const path = url.pathname
559
+
560
+ if (path === "/api/hello") {
561
+ // Claude Code's connection-warming probe; auth keeps it from being a
562
+ // port-scan oracle. Rejecting it is harmless by design.
563
+ return authorize(req, selfHost) ?? new Response(null, { status: 200 })
564
+ }
565
+
566
+ if (path !== "/v1/messages" && path !== "/v1/messages/count_tokens") {
567
+ if (req.method === "GET" && path === "/v1/models") {
568
+ const denied = authorize(req, selfHost)
569
+ if (denied) return denied
570
+ // Gateway model discovery (CLAUDE_CODE_ENABLE_GATEWAY_MODEL_DISCOVERY):
571
+ // serve the list captured at startup — Claude Code aborts discovery
572
+ // after 3 seconds, so a live upstream fetch would silently fail.
573
+ logLine(`[${timestamp()}] GET /v1/models`)
574
+ const entries = upstreamModels()
575
+ if (entries.length === 0) {
576
+ return anthropicError(502, "model list unavailable (discovery failed at startup)")
577
+ }
578
+ return Response.json({
579
+ data: entries.map((e) => ({
580
+ type: "model",
581
+ id: e.id,
582
+ display_name: e.name,
583
+ })),
584
+ has_more: false,
585
+ })
586
+ }
587
+ return anthropicError(404, `not found: ${req.method} ${path}`)
588
+ }
589
+ if (req.method !== "POST") {
590
+ return anthropicError(404, `not found: ${req.method} ${path}`)
591
+ }
592
+ const denied = authorize(req, selfHost)
593
+ if (denied) return denied
594
+
595
+ const contentType = req.headers.get("content-type") ?? ""
596
+ if (!contentType.includes("application/json")) {
597
+ return anthropicError(400, "content-type must be application/json")
598
+ }
599
+ const contentLength = Number(req.headers.get("content-length") ?? "0")
600
+ if (contentLength > MAX_BODY_BYTES) {
601
+ return anthropicError(413, "request body too large")
602
+ }
603
+
604
+ if (path === "/v1/messages/count_tokens") {
605
+ try {
606
+ const payload = (await req.json()) as AnthropicRequest
607
+ logLine(`[${timestamp()}] POST /v1/messages/count_tokens`)
608
+ return Response.json({ input_tokens: estimateTokens(payload) })
609
+ } catch {
610
+ return anthropicError(400, "invalid JSON body")
611
+ }
612
+ }
613
+
614
+ return handleMessages(req, upstreamBase, mockToken)
615
+ }
616
+
617
+ export async function startServer(
618
+ opts: ServerOptions = {},
619
+ ): Promise<ServerHandle> {
620
+ const upstreamBase = (opts.upstream ?? copilotBaseUrl()).replace(/\/$/, "")
621
+ const mockToken = opts.upstream !== undefined || isMockMode()
622
+ // Filled in once the port is bound; the fetch closure only runs afterwards.
623
+ let selfHost = ""
624
+ const server = Bun.serve({
625
+ hostname: "127.0.0.1",
626
+ port: opts.port ?? 0,
627
+ fetch: (req) =>
628
+ handle(req, upstreamBase, mockToken, selfHost).catch((err) =>
629
+ anthropicError(500, String(err)),
630
+ ),
631
+ })
632
+ selfHost = `127.0.0.1:${server.port ?? 0}`
633
+ return {
634
+ url: server.url.toString().replace(/\/$/, ""),
635
+ port: server.port ?? 0,
636
+ stop: () => server.stop(true),
637
+ }
638
+ }
639
+
640
+ // Imported lazily by name to keep the module graph simple.
641
+