myapikey 0.48.0 → 0.48.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/packages/core/src/server/proxy.ts +83 -35
- package/packages/core/src/server/tokens.ts +66 -0
- package/packages/core/src/shared/types.ts +2 -2
- package/packages/web/dist/assets/{index-Bdt_PL4a.js → index-CIGEPkrP.js} +4 -4
- package/packages/web/dist/assets/index-CWGc5RaE.css +1 -0
- package/packages/web/dist/index.html +2 -2
- package/packages/web/dist/assets/index-COQLT2h4.css +0 -1
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "myapikey",
|
|
3
|
-
"version": "0.48.
|
|
3
|
+
"version": "0.48.2",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Personal LLM API gateway & proxy — one address + one API key for all your models. Forwards OpenAI & Anthropic calls to your backends with failover and a circuit breaker. Pure passthrough, no format translation. Self-hosted (CLI + web UI).",
|
|
6
6
|
"keywords": [
|
|
@@ -2,7 +2,7 @@ import { Hono, type Context, type MiddlewareHandler } from "hono";
|
|
|
2
2
|
import { trimBase } from "../shared/config";
|
|
3
3
|
import type { DebugCapture, Format, Provider, RouteKey, Usage } from "../shared/types";
|
|
4
4
|
import { CAPTURE_BODY_MAX, type Store } from "./store";
|
|
5
|
-
import { UsageCollector } from "./tokens";
|
|
5
|
+
import { UsageCollector, applyWireUsage, outboundUsage } from "./tokens";
|
|
6
6
|
|
|
7
7
|
/** HTTP statuses that should trigger failover to the next provider. 401/403
|
|
8
8
|
* included: a banned/invalid credential (e.g. "User has been banned") is dead
|
|
@@ -312,7 +312,7 @@ export function anthropicAuthHeaders(apiKey: string, version: string): Record<st
|
|
|
312
312
|
}
|
|
313
313
|
|
|
314
314
|
/** Exported for the admin source-test (direct upstream ping, no routing). */
|
|
315
|
-
export function upstreamHeaders(provider: Provider, format:
|
|
315
|
+
export function upstreamHeaders(provider: Provider, format: RouteKey, clientVersion?: string): Record<string, string> {
|
|
316
316
|
const h: Record<string, string> = { "content-type": "application/json" };
|
|
317
317
|
if (format === "openai") h.authorization = `Bearer ${provider.apiKey}`;
|
|
318
318
|
else Object.assign(h, anthropicAuthHeaders(provider.apiKey, clientVersion || "2023-06-01"));
|
|
@@ -375,6 +375,27 @@ interface SettleInfo {
|
|
|
375
375
|
usage?: Usage;
|
|
376
376
|
}
|
|
377
377
|
|
|
378
|
+
/** A final SSE frame carrying the gateway's normalized usage. OpenAI chat
|
|
379
|
+
* clients get a standard final chunk; Responses and Anthropic get a tagged
|
|
380
|
+
* gateway event so we never fabricate a protocol-specific terminal event. */
|
|
381
|
+
function usageFrame(key: RouteKey, usage: Usage): string {
|
|
382
|
+
const data = { type: "myapikey.usage", usage: outboundUsage(usage, key) };
|
|
383
|
+
if (key === "openai") return `data: ${JSON.stringify({ choices: [], usage: data.usage })}\n\n`;
|
|
384
|
+
return `event: myapikey.usage\ndata: ${JSON.stringify(data)}\n\n`;
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
/** OpenAI chat streams only report usage when explicitly requested. Force it on
|
|
388
|
+
* so the gateway's outbound usage is exact instead of a tokenizer estimate; the
|
|
389
|
+
* other two wires already carry usage on their normal terminal events. */
|
|
390
|
+
function requireUpstreamUsage(body: Record<string, unknown>, key: RouteKey, stream: boolean): void {
|
|
391
|
+
if (key !== "openai" || !stream) return;
|
|
392
|
+
const options = body.stream_options;
|
|
393
|
+
body.stream_options = {
|
|
394
|
+
...(options && typeof options === "object" && !Array.isArray(options) ? options : {}),
|
|
395
|
+
include_usage: true,
|
|
396
|
+
};
|
|
397
|
+
}
|
|
398
|
+
|
|
378
399
|
/** Wrap an upstream body so every byte is forwarded to the client VERBATIM while
|
|
379
400
|
* we watch — out of band — for whether the stream completed cleanly. The 200
|
|
380
401
|
* status is already committed before the body flows, so on a bad end we can't
|
|
@@ -382,6 +403,8 @@ interface SettleInfo {
|
|
|
382
403
|
* so the client learns the stream died rather than seeing a silent EOF, and
|
|
383
404
|
* (b) settle {ok:false} so dispatch logs a 502 and trips the circuit (the NEXT
|
|
384
405
|
* call then fails over — this call can't be salvaged once streaming started).
|
|
406
|
+
* A successful body gets one normalized usage frame before its terminal marker
|
|
407
|
+
* (streaming) or one final JSON body with usage filled in (non-streaming).
|
|
385
408
|
*
|
|
386
409
|
* Detection keys on the stream's terminal marker (anthropic message_stop /
|
|
387
410
|
* openai [DONE] / responses response.completed), buffered across chunk
|
|
@@ -444,6 +467,7 @@ function observedBody(
|
|
|
444
467
|
let tail = ""; // rolling window so a marker split across chunks is still caught
|
|
445
468
|
let terminal = false;
|
|
446
469
|
let settled = false;
|
|
470
|
+
const heldTerminal: Uint8Array[] = [];
|
|
447
471
|
const usage = new UsageCollector();
|
|
448
472
|
|
|
449
473
|
const settle = (info: SettleInfo) => {
|
|
@@ -462,38 +486,61 @@ function observedBody(
|
|
|
462
486
|
controller.close();
|
|
463
487
|
return;
|
|
464
488
|
}
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
if (
|
|
469
|
-
const
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
489
|
+
while (true) {
|
|
490
|
+
try {
|
|
491
|
+
const { done, value } = await reader.read();
|
|
492
|
+
if (done) {
|
|
493
|
+
const finalUsage = usage.finalize({ stream: opts.stream, key: opts.key, requestMessages: opts.requestMessages });
|
|
494
|
+
if (opts.stream && !terminal) {
|
|
495
|
+
const reason = "upstream stream truncated (no terminal marker)";
|
|
496
|
+
injectError(controller, reason);
|
|
497
|
+
settle({ ok: false, status: 502, error: reason });
|
|
498
|
+
} else {
|
|
499
|
+
settle({ ok: true, status: 200, usage: finalUsage });
|
|
500
|
+
if (opts.stream) {
|
|
501
|
+
controller.enqueue(enc.encode(usageFrame(opts.key, finalUsage ?? { input: 0, output: 0 })));
|
|
502
|
+
for (const value of heldTerminal) controller.enqueue(value);
|
|
503
|
+
} else {
|
|
504
|
+
let body = usage.bodyText();
|
|
505
|
+
try {
|
|
506
|
+
const parsed = JSON.parse(body) as Record<string, unknown>;
|
|
507
|
+
applyWireUsage(parsed, opts.key, finalUsage ?? { input: 0, output: 0 });
|
|
508
|
+
body = JSON.stringify(parsed);
|
|
509
|
+
} catch {
|
|
510
|
+
// Leave malformed upstream bodies untouched; usage is still logged.
|
|
511
|
+
}
|
|
512
|
+
controller.enqueue(enc.encode(body));
|
|
513
|
+
}
|
|
514
|
+
}
|
|
515
|
+
controller.close();
|
|
516
|
+
return;
|
|
517
|
+
}
|
|
518
|
+
const txt = dec.decode(value, { stream: true });
|
|
519
|
+
usage.feed(txt, { stream: opts.stream, key: opts.key });
|
|
520
|
+
opts.onText?.(txt);
|
|
521
|
+
if (!terminal && markers.length) {
|
|
522
|
+
const win = tail + txt;
|
|
523
|
+
if (markers.some((m) => win.includes(m))) terminal = true;
|
|
524
|
+
tail = win.slice(-128);
|
|
525
|
+
}
|
|
526
|
+
if (terminal) {
|
|
527
|
+
heldTerminal.push(value);
|
|
528
|
+
continue;
|
|
529
|
+
}
|
|
530
|
+
if (opts.stream) {
|
|
531
|
+
controller.enqueue(value);
|
|
474
532
|
}
|
|
533
|
+
return;
|
|
534
|
+
} catch (e) {
|
|
535
|
+
const reason = `upstream stream error: ${e instanceof Error ? e.message : String(e)}`;
|
|
536
|
+
injectError(controller, reason);
|
|
537
|
+
settle({ ok: false, status: 502, error: reason });
|
|
475
538
|
controller.close();
|
|
476
539
|
return;
|
|
477
540
|
}
|
|
478
|
-
const txt = dec.decode(value, { stream: true });
|
|
479
|
-
usage.feed(txt, { stream: opts.stream, key: opts.key });
|
|
480
|
-
opts.onText?.(txt);
|
|
481
|
-
if (!terminal && markers.length) {
|
|
482
|
-
const win = tail + txt;
|
|
483
|
-
if (markers.some((m) => win.includes(m))) terminal = true;
|
|
484
|
-
tail = win.slice(-128);
|
|
485
|
-
}
|
|
486
|
-
controller.enqueue(value);
|
|
487
|
-
} catch (e) {
|
|
488
|
-
const reason = `upstream stream error: ${e instanceof Error ? e.message : String(e)}`;
|
|
489
|
-
injectError(controller, reason);
|
|
490
|
-
settle({ ok: false, status: 502, error: reason });
|
|
491
|
-
controller.close();
|
|
492
541
|
}
|
|
493
542
|
},
|
|
494
543
|
cancel() {
|
|
495
|
-
// Client abort (Esc / disconnect) — not a provider failure. Suppress the
|
|
496
|
-
// settle so we neither log nor cool down a source the client simply left.
|
|
497
544
|
settled = true;
|
|
498
545
|
reader?.cancel().catch(() => {});
|
|
499
546
|
},
|
|
@@ -559,6 +606,7 @@ export function proxyApi(
|
|
|
559
606
|
const model: string = body.model;
|
|
560
607
|
const wire: Format = key === "anthropic" ? "anthropic" : "openai";
|
|
561
608
|
const stream = body.stream === true;
|
|
609
|
+
requireUpstreamUsage(body, key, stream);
|
|
562
610
|
// The request's own thinking switches, snapshotted BEFORE the failover
|
|
563
611
|
// loop — each attempt mutates the shared body, and a passthrough slot
|
|
564
612
|
// (no default) must restore these originals.
|
|
@@ -641,7 +689,7 @@ export function proxyApi(
|
|
|
641
689
|
const retryAfter = Math.max(1, Math.ceil(RPM_SECONDS / paceRpm));
|
|
642
690
|
const message = `rate limited: even-pacing queue for '${model}' is full (max wait ${Math.round(PACE_MAX_WAIT_S)}s; retry in ~${retryAfter}s)`;
|
|
643
691
|
if (!isProbe) rt.warn(`proxy model=${model}: paced out (429)`);
|
|
644
|
-
store.pushLog({ ts: Date.now(), model, provider: "", format:
|
|
692
|
+
store.pushLog({ ts: Date.now(), model, provider: "", format: key, status: 429, ms: Date.now() - start, stream, error: "even-pacing queue full" });
|
|
645
693
|
const headers = { "content-type": "application/json", "retry-after": String(retryAfter) };
|
|
646
694
|
if (wire === "anthropic") {
|
|
647
695
|
return c.json({ type: "error", error: { type: "rate_limit_error", message } }, 429, headers);
|
|
@@ -737,7 +785,7 @@ export function proxyApi(
|
|
|
737
785
|
model,
|
|
738
786
|
provider: provider.name,
|
|
739
787
|
providerId: provider.id,
|
|
740
|
-
format:
|
|
788
|
+
format: key,
|
|
741
789
|
status,
|
|
742
790
|
ms: Date.now() - attemptStart,
|
|
743
791
|
stream,
|
|
@@ -767,7 +815,7 @@ export function proxyApi(
|
|
|
767
815
|
sayFailover(provider, "network error");
|
|
768
816
|
const r = store.recordCircuitFailure(provider.id, lastStatus, lastErr);
|
|
769
817
|
if (r.entered) {
|
|
770
|
-
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format:
|
|
818
|
+
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format: key, status: lastStatus, ms: Date.now() - start, stream, thinking: think, sampling, kind: "cooldown", cooldownMs: r.cooldownMs, fails: r.fails, error: lastErr });
|
|
771
819
|
sayCooldown(provider, r);
|
|
772
820
|
}
|
|
773
821
|
continue;
|
|
@@ -805,14 +853,14 @@ export function proxyApi(
|
|
|
805
853
|
onSettle: (info) => {
|
|
806
854
|
if (info.ok) {
|
|
807
855
|
store.recordCircuitSuccess(provider.id);
|
|
808
|
-
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format:
|
|
856
|
+
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format: key, status: 200, ms: ttfb, stream, thinking: think, sampling, usage: info.usage });
|
|
809
857
|
capture(200, capAcc.text, capAcc.truncated);
|
|
810
858
|
} else {
|
|
811
859
|
// A pinned per-source probe takes no circuit side-effects (a manual
|
|
812
860
|
// test must not trip the breaker) — mirrors the retryable branch.
|
|
813
861
|
if (pinIndex == null) store.recordCircuitFailure(provider.id, info.status, info.error || "stream failed");
|
|
814
862
|
if (!isProbe) rt.warn(`proxy stream failed: provider '${provider.name}' status=${info.status} (${info.error || "stream failed"})`);
|
|
815
|
-
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format:
|
|
863
|
+
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format: key, status: info.status, ms: ttfb, stream, thinking: think, sampling, error: info.error });
|
|
816
864
|
capture(info.status, capAcc.text, capAcc.truncated, info.error);
|
|
817
865
|
}
|
|
818
866
|
},
|
|
@@ -842,13 +890,13 @@ export function proxyApi(
|
|
|
842
890
|
const resetInMs = retryAfterMs ? undefined : parseResetFromBody(txt);
|
|
843
891
|
const r = store.recordCircuitFailure(provider.id, lastStatus, lastErr, retryAfterMs ?? resetInMs, !!resetInMs);
|
|
844
892
|
if (r.entered) {
|
|
845
|
-
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format:
|
|
893
|
+
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format: key, status: lastStatus, ms: Date.now() - start, stream, thinking: think, sampling, kind: "cooldown", cooldownMs: r.cooldownMs, fails: r.fails, error: lastErr });
|
|
846
894
|
sayCooldown(provider, r);
|
|
847
895
|
}
|
|
848
896
|
continue;
|
|
849
897
|
}
|
|
850
898
|
// Non-retryable client error: return it to the caller as-is.
|
|
851
|
-
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format:
|
|
899
|
+
store.pushLog({ ts: Date.now(), model, upstreamModel, provider: provider.name, providerId: provider.id, format: key, status: upstream.status, ms: Date.now() - start, stream, thinking: think, sampling, error: lastErr });
|
|
852
900
|
return new Response(txt, { status: upstream.status, headers: downHeaders(upstream, isProbe ? provider.name : undefined) });
|
|
853
901
|
}
|
|
854
902
|
|
|
@@ -861,7 +909,7 @@ export function proxyApi(
|
|
|
861
909
|
return new Response(null, { status: 499 });
|
|
862
910
|
}
|
|
863
911
|
if (!isProbe) rt.error(`proxy all providers failed model=${model} (last status ${lastStatus})`);
|
|
864
|
-
store.pushLog({ ts: Date.now(), model, upstreamModel: lastUpstreamModel, provider: last.provider.name, providerId: last.provider.id, format:
|
|
912
|
+
store.pushLog({ ts: Date.now(), model, upstreamModel: lastUpstreamModel, provider: last.provider.name, providerId: last.provider.id, format: key, status: lastStatus, ms: Date.now() - start, stream, thinking: lastThink, sampling: lastSampling, error: lastErr || `all providers failed (last status ${lastStatus})` });
|
|
865
913
|
// A pinned (per-source) probe failed: surface the REAL upstream status the
|
|
866
914
|
// one slot returned (429/500/…), not a collapsed 502, and tag it with
|
|
867
915
|
// x-myapikey-provider so the source-row badge names the tested source.
|
|
@@ -55,6 +55,66 @@ export function estimatePromptTokens(messages: unknown): number {
|
|
|
55
55
|
return n;
|
|
56
56
|
}
|
|
57
57
|
|
|
58
|
+
/** Turn the gateway's wire-neutral usage back into the usage shape expected on
|
|
59
|
+
* THIS outbound wire. `input` is the uncached prompt count, so OpenAI-family
|
|
60
|
+
* prompt tokens add cacheRead back; Anthropic keeps cache read/creation as its
|
|
61
|
+
* native sibling counters. Every emitted usage object carries the cache fields
|
|
62
|
+
* explicitly so clients never have to distinguish "no cache" from "not told". */
|
|
63
|
+
export function outboundUsage(usage: Usage, key: RouteKey): Record<string, unknown> {
|
|
64
|
+
if (key === "anthropic") {
|
|
65
|
+
return {
|
|
66
|
+
input_tokens: usage.input,
|
|
67
|
+
output_tokens: usage.output,
|
|
68
|
+
cache_read_input_tokens: usage.cacheRead ?? 0,
|
|
69
|
+
cache_creation_input_tokens: usage.cacheCreation ?? 0,
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
if (key === "responses") {
|
|
73
|
+
return {
|
|
74
|
+
input_tokens: usage.input + (usage.cacheRead ?? 0),
|
|
75
|
+
output_tokens: usage.output,
|
|
76
|
+
total_tokens: usage.input + (usage.cacheRead ?? 0) + usage.output,
|
|
77
|
+
input_tokens_details: { cached_tokens: usage.cacheRead ?? 0 },
|
|
78
|
+
};
|
|
79
|
+
}
|
|
80
|
+
const promptTokens = usage.input + (usage.cacheRead ?? 0);
|
|
81
|
+
return {
|
|
82
|
+
prompt_tokens: promptTokens,
|
|
83
|
+
completion_tokens: usage.output,
|
|
84
|
+
total_tokens: promptTokens + usage.output,
|
|
85
|
+
prompt_tokens_details: { cached_tokens: usage.cacheRead ?? 0 },
|
|
86
|
+
};
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/** Attach `outboundUsage` to a parsed non-streaming body without discarding any
|
|
90
|
+
* provider-specific fields already present. A `prompt_tokens_details: null`
|
|
91
|
+
* (or absent details object) is normalized to `{cached_tokens:…}`; upstream
|
|
92
|
+
* extensions such as audio/reasoning token details survive. */
|
|
93
|
+
export function applyWireUsage(obj: Record<string, unknown>, key: RouteKey, usage: Usage): void {
|
|
94
|
+
const native = outboundUsage(usage, key);
|
|
95
|
+
if (key === "responses") {
|
|
96
|
+
const response = obj.response as Record<string, unknown> | null | undefined;
|
|
97
|
+
const parent = typeof response === "object" && response ? response : obj;
|
|
98
|
+
const old = parent.usage as Record<string, unknown> | null | undefined;
|
|
99
|
+
parent.usage = { ...(typeof old === "object" && old ? old : {}), ...native };
|
|
100
|
+
return;
|
|
101
|
+
}
|
|
102
|
+
const old = obj.usage as Record<string, unknown> | null | undefined;
|
|
103
|
+
if (key === "anthropic") {
|
|
104
|
+
obj.usage = { ...(typeof old === "object" && old ? old : {}), ...native };
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
const details = old?.prompt_tokens_details as Record<string, unknown> | null | undefined;
|
|
108
|
+
obj.usage = {
|
|
109
|
+
...(typeof old === "object" && old ? old : {}),
|
|
110
|
+
...native,
|
|
111
|
+
prompt_tokens_details: {
|
|
112
|
+
...(typeof details === "object" && details ? details : {}),
|
|
113
|
+
...(typeof native.prompt_tokens_details === "object" ? native.prompt_tokens_details : {}),
|
|
114
|
+
},
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
|
|
58
118
|
/** Coerce a JSON value to a finite non-negative integer, or undefined. */
|
|
59
119
|
function num(v: unknown): number | undefined {
|
|
60
120
|
return typeof v === "number" && Number.isFinite(v) && v >= 0 ? Math.trunc(v) : undefined;
|
|
@@ -151,6 +211,12 @@ export class UsageCollector {
|
|
|
151
211
|
* local completion-token estimate when the upstream omits usage. */
|
|
152
212
|
private completionText = "";
|
|
153
213
|
|
|
214
|
+
/** The complete non-streaming body. Available after `feed()`; lets the proxy
|
|
215
|
+
* augment the exact JSON it already buffered without keeping a second copy. */
|
|
216
|
+
bodyText(): string {
|
|
217
|
+
return this.buf;
|
|
218
|
+
}
|
|
219
|
+
|
|
154
220
|
feed(text: string, opts: { stream: boolean; key: RouteKey }): void {
|
|
155
221
|
if (!opts.stream) {
|
|
156
222
|
this.buf += text;
|
|
@@ -172,7 +172,7 @@ export interface LogEntry {
|
|
|
172
172
|
* provider doesn't split its history; the display name is resolved from the
|
|
173
173
|
* live config at read time. Absent on legacy lines → fall back to `provider`. */
|
|
174
174
|
providerId?: string;
|
|
175
|
-
format:
|
|
175
|
+
format: RouteKey;
|
|
176
176
|
status: number;
|
|
177
177
|
/** End-to-end latency (ms): from the dispatch start to the returned response. */
|
|
178
178
|
ms: number;
|
|
@@ -219,7 +219,7 @@ export interface DebugCapture {
|
|
|
219
219
|
model: string;
|
|
220
220
|
provider: string;
|
|
221
221
|
providerId: string;
|
|
222
|
-
format:
|
|
222
|
+
format: RouteKey;
|
|
223
223
|
/** The upstream model name actually sent this attempt (post per-slot
|
|
224
224
|
* rewrite). Absent when the public name went through verbatim. */
|
|
225
225
|
upstreamModel?: string;
|