ghc-proxy 0.9.0 → 0.9.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +568 -529
- package/dist/main.mjs +2090 -733
- package/package.json +1 -1
package/dist/main.mjs
CHANGED
|
@@ -8,7 +8,7 @@ import fs, { mkdir, readdir, unlink, writeFile } from "node:fs/promises";
|
|
|
8
8
|
import os from "node:os";
|
|
9
9
|
import { randomUUID } from "node:crypto";
|
|
10
10
|
import { execFile, execFileSync } from "node:child_process";
|
|
11
|
-
import nodeHTTP from "node:http";
|
|
11
|
+
import nodeHTTP, { createServer } from "node:http";
|
|
12
12
|
import { PassThrough, Readable } from "node:stream";
|
|
13
13
|
import nodeHTTP2 from "node:http2";
|
|
14
14
|
import nodeHTTPS from "node:https";
|
|
@@ -5769,7 +5769,9 @@ const configFileSchema = object({
|
|
|
5769
5769
|
to: string()
|
|
5770
5770
|
})).optional(),
|
|
5771
5771
|
upstreamQueueConcurrency: number().int().positive().optional(),
|
|
5772
|
-
upstreamQueueMaxRetries: number().int().
|
|
5772
|
+
upstreamQueueMaxRetries: number().int().min(0).max(2).optional(),
|
|
5773
|
+
upstreamRecoveryBudgetSeconds: number().int().min(1).max(120).optional(),
|
|
5774
|
+
overloadFallbacks: record(string(), string()).optional(),
|
|
5773
5775
|
upstreamQueueBaseDelaySeconds: number().int().nonnegative().optional(),
|
|
5774
5776
|
upstreamQueueMaxDelaySeconds: number().int().positive().optional(),
|
|
5775
5777
|
gheDomain: string().optional()
|
|
@@ -5798,12 +5800,12 @@ async function readConfig() {
|
|
|
5798
5800
|
const fieldResult = schema.safeParse(rawObj[key]);
|
|
5799
5801
|
if (fieldResult.success) partial[key] = fieldResult.data;
|
|
5800
5802
|
}
|
|
5801
|
-
cachedConfig = partial;
|
|
5803
|
+
cachedConfig = sanitizeConfig(partial);
|
|
5802
5804
|
return cachedConfig;
|
|
5803
5805
|
}
|
|
5804
5806
|
const unknownKeys = Object.keys(raw).filter((key) => !KNOWN_CONFIG_KEYS.has(key));
|
|
5805
5807
|
if (unknownKeys.length > 0) consola.warn(`config.json contains unknown fields: ${unknownKeys.join(", ")}`);
|
|
5806
|
-
cachedConfig = result.data;
|
|
5808
|
+
cachedConfig = sanitizeConfig(result.data);
|
|
5807
5809
|
return cachedConfig;
|
|
5808
5810
|
} catch (error) {
|
|
5809
5811
|
if (error.code === "ENOENT") {
|
|
@@ -5815,6 +5817,33 @@ async function readConfig() {
|
|
|
5815
5817
|
return {};
|
|
5816
5818
|
}
|
|
5817
5819
|
}
|
|
5820
|
+
function sanitizeConfig(config) {
|
|
5821
|
+
if (!config.overloadFallbacks) return config;
|
|
5822
|
+
const normalizedFallbacks = {};
|
|
5823
|
+
const invalidSources = [];
|
|
5824
|
+
for (const [source, target] of Object.entries(config.overloadFallbacks)) {
|
|
5825
|
+
const normalizedSource = source.trim();
|
|
5826
|
+
const normalizedTarget = target.trim();
|
|
5827
|
+
if (!normalizedSource || !normalizedTarget || normalizedSource === normalizedTarget) {
|
|
5828
|
+
invalidSources.push(source || "<blank>");
|
|
5829
|
+
continue;
|
|
5830
|
+
}
|
|
5831
|
+
normalizedFallbacks[normalizedSource] = normalizedTarget;
|
|
5832
|
+
}
|
|
5833
|
+
const overloadFallbacks = {};
|
|
5834
|
+
for (const [source, target] of Object.entries(normalizedFallbacks)) {
|
|
5835
|
+
if (normalizedFallbacks[target] === source) {
|
|
5836
|
+
invalidSources.push(source);
|
|
5837
|
+
continue;
|
|
5838
|
+
}
|
|
5839
|
+
overloadFallbacks[source] = target;
|
|
5840
|
+
}
|
|
5841
|
+
if (invalidSources.length > 0) consola.warn(`config.json contains invalid overloadFallbacks entries: ${invalidSources.join(", ")}. Ignoring those entries.`);
|
|
5842
|
+
return {
|
|
5843
|
+
...config,
|
|
5844
|
+
overloadFallbacks
|
|
5845
|
+
};
|
|
5846
|
+
}
|
|
5818
5847
|
function getCachedConfig() {
|
|
5819
5848
|
return cachedConfig;
|
|
5820
5849
|
}
|
|
@@ -5898,6 +5927,12 @@ var ConfigStore = class {
|
|
|
5898
5927
|
getModelFallback() {
|
|
5899
5928
|
return getCachedConfig().modelFallback;
|
|
5900
5929
|
}
|
|
5930
|
+
getOverloadFallback(sourceModel) {
|
|
5931
|
+
return getCachedConfig().overloadFallbacks?.[sourceModel];
|
|
5932
|
+
}
|
|
5933
|
+
hasOverloadFallbacks() {
|
|
5934
|
+
return Object.keys(getCachedConfig().overloadFallbacks ?? {}).length > 0;
|
|
5935
|
+
}
|
|
5901
5936
|
};
|
|
5902
5937
|
const configStore = new ConfigStore();
|
|
5903
5938
|
//#endregion
|
|
@@ -5996,18 +6031,80 @@ var TranslationFailure = class extends Error {
|
|
|
5996
6031
|
}
|
|
5997
6032
|
};
|
|
5998
6033
|
//#endregion
|
|
6034
|
+
//#region src/lib/timeout-error.ts
|
|
6035
|
+
/**
|
|
6036
|
+
* Whether an error represents a request that timed out or was aborted.
|
|
6037
|
+
*
|
|
6038
|
+
* The shape differs by runtime, so the check is structural rather than a
|
|
6039
|
+
* single `name` comparison:
|
|
6040
|
+
* - Bun rejects with a flat `DOMException` named `TimeoutError` (its ~300s
|
|
6041
|
+
* `fetch` ceiling, `AbortSignal.timeout`) or `AbortError`.
|
|
6042
|
+
* - Node rejects with `TypeError('fetch failed' | 'terminated')` and puts the
|
|
6043
|
+
* real undici error on `.cause` (`HeadersTimeoutError`, `BodyTimeoutError`,
|
|
6044
|
+
* `ConnectTimeoutError`), so the top-level error carries no signal at all —
|
|
6045
|
+
* `TypeError('fetch failed')` is also what `ECONNREFUSED` and DNS failures
|
|
6046
|
+
* look like. The discriminator is the cause's `name`/`code`.
|
|
6047
|
+
*
|
|
6048
|
+
* Both runtimes enforce a ~300s upstream ceiling by default (Node's is
|
|
6049
|
+
* undici's `headersTimeout`/`bodyTimeout` default of `300e3`). It is an
|
|
6050
|
+
* **idle** timer on both, not a total-duration cap: it resets on every byte, so
|
|
6051
|
+
* it fires on a stalled stream well before the configured `--upstream-timeout`
|
|
6052
|
+
* of 1800s, and never on one that keeps streaming. See
|
|
6053
|
+
* `docs/design/streaming.md`.
|
|
6054
|
+
*
|
|
6055
|
+
* Kept in one place because the rule is checked on both sides of the stream
|
|
6056
|
+
* boundary: `src/server.ts` maps it to a 504 before the first byte, and the
|
|
6057
|
+
* Anthropic stream transducer maps it to an SSE error frame after. Two
|
|
6058
|
+
* implementations of "what counts as a timeout" is how one of them ends up
|
|
6059
|
+
* recognizing only half the errors.
|
|
6060
|
+
*/
|
|
6061
|
+
const TIMEOUT_ERROR_NAMES = new Set([
|
|
6062
|
+
"AbortError",
|
|
6063
|
+
"TimeoutError",
|
|
6064
|
+
"ConnectTimeoutError",
|
|
6065
|
+
"HeadersTimeoutError",
|
|
6066
|
+
"BodyTimeoutError"
|
|
6067
|
+
]);
|
|
6068
|
+
const TIMEOUT_ERROR_CODES = new Set([
|
|
6069
|
+
"UND_ERR_CONNECT_TIMEOUT",
|
|
6070
|
+
"UND_ERR_HEADERS_TIMEOUT",
|
|
6071
|
+
"UND_ERR_BODY_TIMEOUT",
|
|
6072
|
+
"ETIMEDOUT"
|
|
6073
|
+
]);
|
|
6074
|
+
const MAX_CAUSE_DEPTH = 5;
|
|
6075
|
+
function errorCauseChainSome(value, predicate, depth = 0) {
|
|
6076
|
+
if (typeof value !== "object" || value === null) return false;
|
|
6077
|
+
const candidate = value;
|
|
6078
|
+
if (predicate(candidate)) return true;
|
|
6079
|
+
if (depth >= MAX_CAUSE_DEPTH) return false;
|
|
6080
|
+
if (errorCauseChainSome(candidate.cause, predicate, depth + 1)) return true;
|
|
6081
|
+
return Array.isArray(candidate.errors) && candidate.errors.some((inner) => errorCauseChainSome(inner, predicate, depth + 1));
|
|
6082
|
+
}
|
|
6083
|
+
function isTimeoutLikeError(error) {
|
|
6084
|
+
try {
|
|
6085
|
+
return errorCauseChainSome(error, (candidate) => typeof candidate.name === "string" && TIMEOUT_ERROR_NAMES.has(candidate.name) || typeof candidate.code === "string" && TIMEOUT_ERROR_CODES.has(candidate.code));
|
|
6086
|
+
} catch {
|
|
6087
|
+
return false;
|
|
6088
|
+
}
|
|
6089
|
+
}
|
|
6090
|
+
//#endregion
|
|
5999
6091
|
//#region src/lib/error.ts
|
|
6000
6092
|
var HTTPError = class extends Error {
|
|
6001
6093
|
status;
|
|
6002
6094
|
body;
|
|
6003
|
-
|
|
6095
|
+
headers;
|
|
6096
|
+
constructor(status, body, options = {}) {
|
|
6004
6097
|
super(body.error.message);
|
|
6005
6098
|
this.name = "HTTPError";
|
|
6006
6099
|
this.status = status;
|
|
6007
6100
|
this.body = body;
|
|
6101
|
+
this.headers = new Headers(options.headers);
|
|
6008
6102
|
}
|
|
6009
6103
|
toResponse() {
|
|
6010
|
-
return Response.json(this.body, {
|
|
6104
|
+
return Response.json(this.body, {
|
|
6105
|
+
status: this.status,
|
|
6106
|
+
headers: this.headers
|
|
6107
|
+
});
|
|
6011
6108
|
}
|
|
6012
6109
|
};
|
|
6013
6110
|
const TRANSIENT_UPSTREAM_STATUSES = new Set([
|
|
@@ -6027,13 +6124,29 @@ const TRANSIENT_UPSTREAM_STATUSES = new Set([
|
|
|
6027
6124
|
function isTransientUpstreamStatus(status) {
|
|
6028
6125
|
return TRANSIENT_UPSTREAM_STATUSES.has(status);
|
|
6029
6126
|
}
|
|
6030
|
-
|
|
6031
|
-
|
|
6032
|
-
|
|
6033
|
-
|
|
6034
|
-
|
|
6035
|
-
|
|
6036
|
-
|
|
6127
|
+
function resolveCapacityCooldownScope(status, effectiveModel) {
|
|
6128
|
+
if (status === 429) return "account";
|
|
6129
|
+
if (status === 529) return effectiveModel ? "model" : "request";
|
|
6130
|
+
}
|
|
6131
|
+
const CONNECTION_ESTABLISHMENT_CODES = {
|
|
6132
|
+
ENOTFOUND: "dns",
|
|
6133
|
+
EAI_AGAIN: "dns",
|
|
6134
|
+
ECONNREFUSED: "connection-refused",
|
|
6135
|
+
ConnectionRefused: "connection-refused"
|
|
6136
|
+
};
|
|
6137
|
+
function isRetryableConnectionEstablishmentError(error) {
|
|
6138
|
+
if (isTimeoutLikeError(error)) return void 0;
|
|
6139
|
+
let connectionClass;
|
|
6140
|
+
try {
|
|
6141
|
+
errorCauseChainSome(error, (candidate) => {
|
|
6142
|
+
if (typeof candidate.code !== "string") return false;
|
|
6143
|
+
connectionClass = CONNECTION_ESTABLISHMENT_CODES[candidate.code];
|
|
6144
|
+
return connectionClass !== void 0;
|
|
6145
|
+
});
|
|
6146
|
+
} catch {
|
|
6147
|
+
return;
|
|
6148
|
+
}
|
|
6149
|
+
return connectionClass;
|
|
6037
6150
|
}
|
|
6038
6151
|
/**
|
|
6039
6152
|
* Reject a request locally with a 400.
|
|
@@ -6133,7 +6246,8 @@ async function throwUpstreamError(message, response) {
|
|
|
6133
6246
|
rawBody: rawText ? previewBody(rawText) : "<empty>",
|
|
6134
6247
|
headers: getDiagnosticHeaders(response)
|
|
6135
6248
|
});
|
|
6136
|
-
|
|
6249
|
+
const retryAfter = response.headers.get("retry-after");
|
|
6250
|
+
throw new HTTPError(response.status, body, retryAfter ? { headers: { "retry-after": retryAfter } } : void 0);
|
|
6137
6251
|
}
|
|
6138
6252
|
//#endregion
|
|
6139
6253
|
//#region src/util/sleep.ts
|
|
@@ -6579,17 +6693,896 @@ const GITHUB_BASE_URL = "https://github.com";
|
|
|
6579
6693
|
const GITHUB_CLIENT_ID = "Iv1.b507a08c87ecfe98";
|
|
6580
6694
|
const GITHUB_APP_SCOPES = ["read:user"].join(" ");
|
|
6581
6695
|
//#endregion
|
|
6696
|
+
//#region src/util/duration.ts
|
|
6697
|
+
/**
|
|
6698
|
+
* Formats a millisecond duration as a compact human-readable string:
|
|
6699
|
+
* `<n>ms` under one second, otherwise `<n>s` rounded to whole seconds.
|
|
6700
|
+
*/
|
|
6701
|
+
function formatDurationMs(ms) {
|
|
6702
|
+
return ms < 1e3 ? `${ms}ms` : `${Math.round(ms / 1e3)}s`;
|
|
6703
|
+
}
|
|
6704
|
+
//#endregion
|
|
6705
|
+
//#region src/lib/request-logger.ts
|
|
6706
|
+
/**
|
|
6707
|
+
* Per-request model mapping store.
|
|
6708
|
+
* Route handlers write to this; the after-response hook reads from it.
|
|
6709
|
+
* Uses WeakMap so entries are GC'd when the Request is collected.
|
|
6710
|
+
*/
|
|
6711
|
+
const requestModelMapping = /* @__PURE__ */ new WeakMap();
|
|
6712
|
+
const requestCorrelation = /* @__PURE__ */ new WeakMap();
|
|
6713
|
+
/**
|
|
6714
|
+
* Per-request start timestamp for the access log.
|
|
6715
|
+
*
|
|
6716
|
+
* Keyed on the `Request` the same way `requestCorrelation` is, so entries are
|
|
6717
|
+
* GC'd with it. This lives here rather than in `derive()` because `derive` does
|
|
6718
|
+
* not run on a route Elysia never matched — `onRequest` does, and it preserves
|
|
6719
|
+
* `Request` identity through to `onAfterResponse`. Reading a missing `derive`
|
|
6720
|
+
* value is what rendered every unmatched route's duration as the literal string
|
|
6721
|
+
* `NaNs`.
|
|
6722
|
+
*
|
|
6723
|
+
* That lifecycle behavior was measured once by hand on Bun 1.3.14 and on Node
|
|
6724
|
+
* 24.18 via `@elysiajs/node`. **The automated suite runs only under Bun**, so
|
|
6725
|
+
* the Node half is a point-in-time observation rather than standing coverage —
|
|
6726
|
+
* see `docs/solutions/testing/green-suite-is-evidence-about-one-runtime.md`. If
|
|
6727
|
+
* the Node adapter ever stops firing `onRequest` on an unmatched path, or
|
|
6728
|
+
* re-wraps the `Request` between hooks, the lookup misses and the duration
|
|
6729
|
+
* silently degrades to `-` on that runtime with a green suite.
|
|
6730
|
+
*/
|
|
6731
|
+
const requestStartTimes = /* @__PURE__ */ new WeakMap();
|
|
6732
|
+
/**
|
|
6733
|
+
* Record when a request arrived. Called from `onRequest`, which fires on every
|
|
6734
|
+
* path including ones no route matches.
|
|
6735
|
+
*
|
|
6736
|
+
* The `void` return type is load-bearing, not decoration. `WeakMap.set` returns
|
|
6737
|
+
* the WeakMap, and Elysia turns any non-undefined `onRequest` return into the
|
|
6738
|
+
* response body — a setter written as a concise arrow over `.set()` makes every
|
|
6739
|
+
* response in the proxy the string `[object WeakMap]`. `tsc` does not catch it.
|
|
6740
|
+
*/
|
|
6741
|
+
function markRequestStart(request) {
|
|
6742
|
+
requestStartTimes.set(request, Date.now());
|
|
6743
|
+
}
|
|
6744
|
+
function getRequestStart(request) {
|
|
6745
|
+
return requestStartTimes.get(request);
|
|
6746
|
+
}
|
|
6747
|
+
function getOrCreateRequestCorrelation(request) {
|
|
6748
|
+
const existing = requestCorrelation.get(request);
|
|
6749
|
+
if (existing) return existing;
|
|
6750
|
+
const requestId = crypto.randomUUID();
|
|
6751
|
+
const callerRequestId = request.headers.get("x-request-id") ?? void 0;
|
|
6752
|
+
const correlation = {
|
|
6753
|
+
requestId,
|
|
6754
|
+
...callerRequestId ? { callerRequestId } : {},
|
|
6755
|
+
responseRequestId: callerRequestId ?? requestId
|
|
6756
|
+
};
|
|
6757
|
+
requestCorrelation.set(request, correlation);
|
|
6758
|
+
return correlation;
|
|
6759
|
+
}
|
|
6760
|
+
const MAX_LOGGED_CALLER_REQUEST_ID_LENGTH = 128;
|
|
6761
|
+
const UNSAFE_CALLER_REQUEST_ID_CHARACTERS = /[^\w.:@/-]/g;
|
|
6762
|
+
function sanitizeCallerRequestId(value) {
|
|
6763
|
+
return value ? value.replace(UNSAFE_CALLER_REQUEST_ID_CHARACTERS, "_").slice(0, MAX_LOGGED_CALLER_REQUEST_ID_LENGTH) : void 0;
|
|
6764
|
+
}
|
|
6765
|
+
const RECOVERY_EVENT_OPTIONAL_FIELDS = [
|
|
6766
|
+
"retryCount",
|
|
6767
|
+
"status",
|
|
6768
|
+
"connectionClass",
|
|
6769
|
+
"effectiveModel",
|
|
6770
|
+
"scope",
|
|
6771
|
+
"activeSlots",
|
|
6772
|
+
"maxSlots",
|
|
6773
|
+
"pendingDepth",
|
|
6774
|
+
"maxPendingDepth",
|
|
6775
|
+
"queueWaitMs",
|
|
6776
|
+
"delaySource",
|
|
6777
|
+
"delayMs",
|
|
6778
|
+
"elapsedMs",
|
|
6779
|
+
"remainingBudgetMs",
|
|
6780
|
+
"nextRetryAt",
|
|
6781
|
+
"decision"
|
|
6782
|
+
];
|
|
6783
|
+
function logRecoveryEvent(input, logger = consola) {
|
|
6784
|
+
const callerRequestId = sanitizeCallerRequestId(input.callerRequestId);
|
|
6785
|
+
const fields = {
|
|
6786
|
+
requestId: input.requestId,
|
|
6787
|
+
...callerRequestId ? { callerRequestId } : {},
|
|
6788
|
+
event: input.event
|
|
6789
|
+
};
|
|
6790
|
+
for (const key of RECOVERY_EVENT_OPTIONAL_FIELDS) if (input[key] !== void 0) Object.assign(fields, { [key]: input[key] });
|
|
6791
|
+
logger.info("Upstream recovery", fields);
|
|
6792
|
+
}
|
|
6793
|
+
function setRequestModelMapping(request, info) {
|
|
6794
|
+
requestModelMapping.set(request, info);
|
|
6795
|
+
}
|
|
6796
|
+
function getRequestModelMapping(request) {
|
|
6797
|
+
return requestModelMapping.get(request);
|
|
6798
|
+
}
|
|
6799
|
+
/**
|
|
6800
|
+
* Format how long a request took, given the timestamp recorded at arrival.
|
|
6801
|
+
*
|
|
6802
|
+
* Renders `-` rather than a number when the start is missing or the arithmetic
|
|
6803
|
+
* is not finite. The caller is expected to supply a real start (see
|
|
6804
|
+
* {@link markRequestStart}); this is the shared-formatter backstop, so a future
|
|
6805
|
+
* code path that skips the `onRequest` hook degrades to an honest `-` instead
|
|
6806
|
+
* of printing `NaNs`.
|
|
6807
|
+
*/
|
|
6808
|
+
function formatElapsed(start) {
|
|
6809
|
+
if (start === void 0) return "-";
|
|
6810
|
+
const elapsed = Date.now() - start;
|
|
6811
|
+
return Number.isFinite(elapsed) ? formatDurationMs(elapsed) : "-";
|
|
6812
|
+
}
|
|
6813
|
+
function formatPath(rawUrl) {
|
|
6814
|
+
try {
|
|
6815
|
+
const url = new URL(rawUrl);
|
|
6816
|
+
return `${url.pathname}${url.search}`;
|
|
6817
|
+
} catch {
|
|
6818
|
+
return rawUrl;
|
|
6819
|
+
}
|
|
6820
|
+
}
|
|
6821
|
+
function colorizeStatus(status) {
|
|
6822
|
+
if (status >= 500) return colorize("red", status);
|
|
6823
|
+
if (status >= 400) return colorize("yellow", status);
|
|
6824
|
+
if (status >= 300) return colorize("cyan", status);
|
|
6825
|
+
return colorize("green", status);
|
|
6826
|
+
}
|
|
6827
|
+
const methodColors = {
|
|
6828
|
+
GET: "cyan",
|
|
6829
|
+
POST: "magenta",
|
|
6830
|
+
PUT: "yellow",
|
|
6831
|
+
PATCH: "yellow",
|
|
6832
|
+
DELETE: "red"
|
|
6833
|
+
};
|
|
6834
|
+
function colorizeMethod(method) {
|
|
6835
|
+
return colorize(methodColors[method] ?? "white", method);
|
|
6836
|
+
}
|
|
6837
|
+
function getEffectiveModel(info) {
|
|
6838
|
+
return info.steps.length > 0 ? info.steps.at(-1).to : info.originalModel ?? "-";
|
|
6839
|
+
}
|
|
6840
|
+
/**
|
|
6841
|
+
* Mutate `modelMapping` in place by appending a transform step.
|
|
6842
|
+
* Strategy contexts hold a reference to the same `modelMapping`,
|
|
6843
|
+
* so steps are pushed directly rather than returning a new object.
|
|
6844
|
+
*/
|
|
6845
|
+
function appendModelStepInPlace(info, tag, newModel) {
|
|
6846
|
+
const current = getEffectiveModel(info);
|
|
6847
|
+
if (newModel !== current) info.steps.push({
|
|
6848
|
+
tag,
|
|
6849
|
+
from: current,
|
|
6850
|
+
to: newModel
|
|
6851
|
+
});
|
|
6852
|
+
}
|
|
6853
|
+
function formatModelMapping(info) {
|
|
6854
|
+
if (!info) return "";
|
|
6855
|
+
const { originalModel, steps } = info;
|
|
6856
|
+
if (!originalModel && steps.length === 0) return "";
|
|
6857
|
+
const parts = [colorize("blueBright", originalModel ?? "-")];
|
|
6858
|
+
for (let i = 0; i < steps.length; i++) {
|
|
6859
|
+
const step = steps[i];
|
|
6860
|
+
const isLast = i === steps.length - 1;
|
|
6861
|
+
parts.push(colorize("dim", `-[${step.tag}]->`));
|
|
6862
|
+
parts.push(colorize(isLast ? "greenBright" : "cyanBright", step.to));
|
|
6863
|
+
}
|
|
6864
|
+
return ` ${colorize("dim", "model=")}${parts.join(" ")}`;
|
|
6865
|
+
}
|
|
6866
|
+
/**
|
|
6867
|
+
* Request logging function.
|
|
6868
|
+
* Logs a formatted request line with method, path, status, elapsed time,
|
|
6869
|
+
* and optional model mapping info.
|
|
6870
|
+
*/
|
|
6871
|
+
function logRequest(method, url, status, elapsed, modelInfo, requestId, callerRequestId) {
|
|
6872
|
+
const path = formatPath(url);
|
|
6873
|
+
const line = [
|
|
6874
|
+
colorize("dim", "<-"),
|
|
6875
|
+
colorizeMethod(method),
|
|
6876
|
+
colorize("white", path),
|
|
6877
|
+
colorizeStatus(status),
|
|
6878
|
+
colorize("dim", elapsed)
|
|
6879
|
+
].join(" ");
|
|
6880
|
+
const rid = requestId ? ` ${colorize("dim", `rid=${requestId.slice(0, 8)}`)}` : "";
|
|
6881
|
+
const safeCallerRequestId = sanitizeCallerRequestId(callerRequestId);
|
|
6882
|
+
const callerRid = safeCallerRequestId ? ` ${colorize("dim", `callerRid=${safeCallerRequestId}`)}` : "";
|
|
6883
|
+
console.log(`${line}${formatModelMapping(modelInfo)}${rid}${callerRid}`);
|
|
6884
|
+
}
|
|
6885
|
+
//#endregion
|
|
6886
|
+
//#region src/clients/upstream-queue.ts
|
|
6887
|
+
const DEFAULT_UPSTREAM_QUEUE_OPTIONS = {
|
|
6888
|
+
concurrency: 10,
|
|
6889
|
+
maxRetries: 1,
|
|
6890
|
+
baseDelayMs: 2e3,
|
|
6891
|
+
maxDelayMs: 6e4,
|
|
6892
|
+
maxQueueDepth: 1e3,
|
|
6893
|
+
recoveryBudgetMs: 60 * 1e3
|
|
6894
|
+
};
|
|
6895
|
+
const MAX_TIMER_DELAY_MS = 2147483647;
|
|
6896
|
+
const RETRY_AFTER_SECONDS_RE = /^\d+(?:\.\d+)?$/;
|
|
6897
|
+
const RETRY_AFTER_HTTP_DATE_RE = /^(?:Mon|Tue|Wed|Thu|Fri|Sat|Sun), \d{2} (?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec) \d{4} \d{2}:\d{2}:\d{2} GMT$/;
|
|
6898
|
+
var TerminalUpstreamRecoveryError = class extends HTTPError {
|
|
6899
|
+
recovery;
|
|
6900
|
+
fallbackClaimed = false;
|
|
6901
|
+
constructor(source, recovery) {
|
|
6902
|
+
super(source.status, source.body, { headers: source.headers });
|
|
6903
|
+
this.name = "TerminalUpstreamRecoveryError";
|
|
6904
|
+
this.recovery = recovery;
|
|
6905
|
+
}
|
|
6906
|
+
claimFallback() {
|
|
6907
|
+
if (this.fallbackClaimed) return false;
|
|
6908
|
+
this.fallbackClaimed = true;
|
|
6909
|
+
return true;
|
|
6910
|
+
}
|
|
6911
|
+
};
|
|
6912
|
+
var LocalModelCooldownError = class extends TerminalUpstreamRecoveryError {
|
|
6913
|
+
constructor(recovery, retryAfter) {
|
|
6914
|
+
super(new HTTPError(529, { error: {
|
|
6915
|
+
message: "The selected upstream model is temporarily overloaded.",
|
|
6916
|
+
type: "overloaded_error"
|
|
6917
|
+
} }, { headers: { "retry-after": retryAfter } }), recovery);
|
|
6918
|
+
this.name = "LocalModelCooldownError";
|
|
6919
|
+
}
|
|
6920
|
+
};
|
|
6921
|
+
var FallbackCooldownError = class extends Error {
|
|
6922
|
+
scope;
|
|
6923
|
+
effectiveModel;
|
|
6924
|
+
constructor(cooldown) {
|
|
6925
|
+
super("Fallback target is locally cooled");
|
|
6926
|
+
this.name = "FallbackCooldownError";
|
|
6927
|
+
this.scope = cooldown.scope;
|
|
6928
|
+
this.effectiveModel = cooldown.effectiveModel;
|
|
6929
|
+
}
|
|
6930
|
+
};
|
|
6931
|
+
var UpstreamRequestQueue = class {
|
|
6932
|
+
sleep;
|
|
6933
|
+
now;
|
|
6934
|
+
wallNow;
|
|
6935
|
+
random;
|
|
6936
|
+
logger;
|
|
6937
|
+
setTimer;
|
|
6938
|
+
clearTimer;
|
|
6939
|
+
options;
|
|
6940
|
+
active = 0;
|
|
6941
|
+
accountNotBefore = 0;
|
|
6942
|
+
modelNotBefore = /* @__PURE__ */ new Map();
|
|
6943
|
+
drainTimer;
|
|
6944
|
+
drainTimerAt;
|
|
6945
|
+
waiters = [];
|
|
6946
|
+
terminalRecoveries = /* @__PURE__ */ new WeakSet();
|
|
6947
|
+
constructor(options = {}, deps = {}) {
|
|
6948
|
+
this.options = normalizeOptions(options);
|
|
6949
|
+
this.sleep = deps.sleep;
|
|
6950
|
+
this.now = deps.now ?? (() => performance.now());
|
|
6951
|
+
this.wallNow = deps.wallNow ?? Date.now;
|
|
6952
|
+
this.random = deps.random ?? Math.random;
|
|
6953
|
+
this.logger = deps.logger ?? consola;
|
|
6954
|
+
this.setTimer = deps.setTimeout ?? globalThis.setTimeout;
|
|
6955
|
+
this.clearTimer = deps.clearTimeout ?? globalThis.clearTimeout;
|
|
6956
|
+
}
|
|
6957
|
+
updateOptions(options) {
|
|
6958
|
+
this.options = normalizeOptions(mergeDefinedOptions(this.options, options));
|
|
6959
|
+
this.drain();
|
|
6960
|
+
}
|
|
6961
|
+
async dispatch(fetcher, inputContext, signal) {
|
|
6962
|
+
const recovery = inputContext.recovery ?? {
|
|
6963
|
+
requestId: crypto.randomUUID(),
|
|
6964
|
+
...signal ? { callerSignal: signal } : {},
|
|
6965
|
+
retryCount: 0
|
|
6966
|
+
};
|
|
6967
|
+
const context = {
|
|
6968
|
+
...inputContext,
|
|
6969
|
+
recovery
|
|
6970
|
+
};
|
|
6971
|
+
recovery.sourceModel ??= context.effectiveModel;
|
|
6972
|
+
signal?.throwIfAborted();
|
|
6973
|
+
this.throwIfFallbackCooled(context);
|
|
6974
|
+
const localCooldown = this.getActiveCooldown(context.effectiveModel);
|
|
6975
|
+
if ((typeof context.offerLocalModelCooldown === "function" ? Boolean(context.effectiveModel && context.offerLocalModelCooldown(context.effectiveModel)) : context.offerLocalModelCooldown) && localCooldown?.scope === "model" && localCooldown.notBeforeMonotonicMs > this.now()) {
|
|
6976
|
+
this.startRecovery(recovery);
|
|
6977
|
+
this.setRecoveryCooldown(recovery, localCooldown);
|
|
6978
|
+
const retryAfter = formatRetryAfter(localCooldown.notBeforeMonotonicMs - this.now());
|
|
6979
|
+
recovery.publicError = {
|
|
6980
|
+
status: 529,
|
|
6981
|
+
retryAfter
|
|
6982
|
+
};
|
|
6983
|
+
this.emitTerminal("retry", context, {
|
|
6984
|
+
retryCount: recovery.retryCount,
|
|
6985
|
+
status: 529,
|
|
6986
|
+
scope: "model",
|
|
6987
|
+
decision: "local-cooldown"
|
|
6988
|
+
});
|
|
6989
|
+
throw new LocalModelCooldownError(recovery, retryAfter);
|
|
6990
|
+
}
|
|
6991
|
+
try {
|
|
6992
|
+
return await this.runDispatch(fetcher, context, signal);
|
|
6993
|
+
} catch (error) {
|
|
6994
|
+
const connectionClass = isRetryableConnectionEstablishmentError(error);
|
|
6995
|
+
if (recovery.callerSignal?.aborted) this.emitTerminal(recovery.startedAtMonotonicMs === void 0 ? "admission" : "retry", context, {
|
|
6996
|
+
retryCount: recovery.retryCount,
|
|
6997
|
+
decision: "cancelled"
|
|
6998
|
+
});
|
|
6999
|
+
else if (signal?.aborted) this.emitTerminal("budget", context, {
|
|
7000
|
+
retryCount: recovery.retryCount,
|
|
7001
|
+
status: 504,
|
|
7002
|
+
decision: "deadline-exceeded"
|
|
7003
|
+
});
|
|
7004
|
+
else if (error instanceof HTTPError && error.status === 504 || this.remainingBudget(recovery) === 0) this.emitTerminal("budget", context, {
|
|
7005
|
+
retryCount: recovery.retryCount,
|
|
7006
|
+
status: error instanceof HTTPError ? error.status : void 0,
|
|
7007
|
+
decision: "deadline-exceeded"
|
|
7008
|
+
});
|
|
7009
|
+
else if (connectionClass && !this.canRetry(recovery)) this.emitTerminal("retry", context, {
|
|
7010
|
+
retryCount: recovery.retryCount,
|
|
7011
|
+
connectionClass,
|
|
7012
|
+
decision: "retry-exhausted"
|
|
7013
|
+
});
|
|
7014
|
+
else if (recovery.startedAtMonotonicMs !== void 0) this.emitTerminal("retry", context, {
|
|
7015
|
+
retryCount: recovery.retryCount,
|
|
7016
|
+
status: error instanceof HTTPError ? error.status : void 0,
|
|
7017
|
+
connectionClass,
|
|
7018
|
+
decision: "failed"
|
|
7019
|
+
});
|
|
7020
|
+
throw error;
|
|
7021
|
+
}
|
|
7022
|
+
}
|
|
7023
|
+
async runDispatch(fetcher, context, signal) {
|
|
7024
|
+
const { recovery } = context;
|
|
7025
|
+
let lastConnectionError;
|
|
7026
|
+
for (;;) {
|
|
7027
|
+
signal?.throwIfAborted();
|
|
7028
|
+
this.throwIfRecoveryExpired(recovery, lastConnectionError);
|
|
7029
|
+
const lease = await this.acquire(context, signal, lastConnectionError);
|
|
7030
|
+
let response;
|
|
7031
|
+
try {
|
|
7032
|
+
this.throwIfRecoveryExpired(recovery, lastConnectionError);
|
|
7033
|
+
if (context.fallbackAttempt) recovery.fallbackFetchStarted = true;
|
|
7034
|
+
response = await this.fetchBeforeDeadline(fetcher, signal, recovery, lastConnectionError);
|
|
7035
|
+
lastConnectionError = void 0;
|
|
7036
|
+
} catch (error) {
|
|
7037
|
+
lease.release();
|
|
7038
|
+
if (signal?.aborted) throw signal.reason;
|
|
7039
|
+
if (error instanceof RecoveryBudgetError) throw error.cause ?? error;
|
|
7040
|
+
const connectionClass = isRetryableConnectionEstablishmentError(error);
|
|
7041
|
+
if (!connectionClass || !context.retryable) throw error;
|
|
7042
|
+
this.startRecovery(recovery);
|
|
7043
|
+
if (!this.canRetry(recovery)) throw error;
|
|
7044
|
+
lastConnectionError = error;
|
|
7045
|
+
const delay = this.getBackoffDelay(recovery.retryCount, this.remainingBudget(recovery));
|
|
7046
|
+
recovery.retryCount++;
|
|
7047
|
+
this.emit("retry", context, {
|
|
7048
|
+
retryCount: recovery.retryCount,
|
|
7049
|
+
connectionClass,
|
|
7050
|
+
delaySource: "backoff",
|
|
7051
|
+
delayMs: delay,
|
|
7052
|
+
decision: "retry"
|
|
7053
|
+
});
|
|
7054
|
+
await this.waitForRecovery(delay, signal, recovery, error);
|
|
7055
|
+
continue;
|
|
7056
|
+
}
|
|
7057
|
+
try {
|
|
7058
|
+
const status = response.status;
|
|
7059
|
+
const scope = resolveCapacityCooldownScope(status, context.effectiveModel);
|
|
7060
|
+
const capacity = scope !== void 0;
|
|
7061
|
+
const mayReplay = context.retryable === "capacity" ? capacity : context.retryable === true && isTransientUpstreamStatus(status);
|
|
7062
|
+
let retryDelay;
|
|
7063
|
+
if (capacity) {
|
|
7064
|
+
this.startRecovery(recovery);
|
|
7065
|
+
retryDelay = this.getRetryDelay(response, recovery.retryCount, recovery);
|
|
7066
|
+
this.installCooldown(scope, context.effectiveModel, retryDelay.delayMs, context);
|
|
7067
|
+
recovery.publicError = {
|
|
7068
|
+
status,
|
|
7069
|
+
retryAfter: retryDelay.retryAfter ?? formatRetryAfter(retryDelay.delayMs)
|
|
7070
|
+
};
|
|
7071
|
+
}
|
|
7072
|
+
if (!mayReplay) return this.committed(response, lease, context, capacity ? retryDelay : void 0, capacity ? "capacity-terminal" : "upstream-terminal");
|
|
7073
|
+
this.startRecovery(recovery);
|
|
7074
|
+
if (!capacity) recovery.publicError = { status };
|
|
7075
|
+
retryDelay ??= this.getRetryDelay(response, recovery.retryCount, recovery);
|
|
7076
|
+
const remaining = this.remainingBudget(recovery);
|
|
7077
|
+
const serverMinimumDoesNotFit = retryDelay.source === "retry-after" && retryDelay.delayMs >= remaining;
|
|
7078
|
+
if (!this.canRetry(recovery) || serverMinimumDoesNotFit) {
|
|
7079
|
+
const decision = serverMinimumDoesNotFit ? "server-delay-exceeds-budget" : "retry-limit";
|
|
7080
|
+
this.emitTerminal("budget", context, {
|
|
7081
|
+
retryCount: recovery.retryCount,
|
|
7082
|
+
status,
|
|
7083
|
+
scope,
|
|
7084
|
+
delaySource: retryDelay.source,
|
|
7085
|
+
delayMs: retryDelay.delayMs,
|
|
7086
|
+
remainingBudgetMs: remaining,
|
|
7087
|
+
decision
|
|
7088
|
+
});
|
|
7089
|
+
return this.committed(response, lease, context, capacity ? retryDelay : void 0, decision);
|
|
7090
|
+
}
|
|
7091
|
+
discardResponse(response);
|
|
7092
|
+
lease.release();
|
|
7093
|
+
recovery.retryCount++;
|
|
7094
|
+
this.logger.warn([
|
|
7095
|
+
`Upstream ${status};`,
|
|
7096
|
+
`retrying ${formatRequestContext(context)}`,
|
|
7097
|
+
`in ${formatDurationMs(retryDelay.delayMs)}`,
|
|
7098
|
+
`(attempt ${recovery.retryCount}/${recovery.retryLimit})`
|
|
7099
|
+
].join(" "));
|
|
7100
|
+
this.emit("retry", context, {
|
|
7101
|
+
retryCount: recovery.retryCount,
|
|
7102
|
+
status,
|
|
7103
|
+
scope,
|
|
7104
|
+
delaySource: retryDelay.source,
|
|
7105
|
+
delayMs: retryDelay.delayMs,
|
|
7106
|
+
decision: "retry"
|
|
7107
|
+
});
|
|
7108
|
+
await this.waitForRecovery(retryDelay.delayMs, signal, recovery);
|
|
7109
|
+
} catch (error) {
|
|
7110
|
+
discardResponse(response);
|
|
7111
|
+
lease.release();
|
|
7112
|
+
throw error;
|
|
7113
|
+
}
|
|
7114
|
+
}
|
|
7115
|
+
}
|
|
7116
|
+
async acquire(context, signal, causalError) {
|
|
7117
|
+
signal?.throwIfAborted();
|
|
7118
|
+
this.throwIfFallbackCooled(context);
|
|
7119
|
+
this.prepareCooldownWait(context);
|
|
7120
|
+
this.throwIfRecoveryExpired(context.recovery, causalError);
|
|
7121
|
+
const eligible = this.isEligible(context);
|
|
7122
|
+
if (this.active < this.options.concurrency && (eligible || this.drainTimerAt !== void 0 && this.drainTimerAt <= this.now())) {
|
|
7123
|
+
this.drain();
|
|
7124
|
+
if (eligible && this.active < this.options.concurrency) return this.grant(context, 0);
|
|
7125
|
+
}
|
|
7126
|
+
if (this.waiters.length >= this.options.maxQueueDepth) this.drain();
|
|
7127
|
+
if (this.waiters.length >= this.options.maxQueueDepth) {
|
|
7128
|
+
this.emit("admission", context, { decision: "queue-full" });
|
|
7129
|
+
throw new HTTPError(503, { error: {
|
|
7130
|
+
message: "Upstream queue full",
|
|
7131
|
+
type: "overloaded_error"
|
|
7132
|
+
} });
|
|
7133
|
+
}
|
|
7134
|
+
return new Promise((resolve, reject) => {
|
|
7135
|
+
const waiter = {
|
|
7136
|
+
context,
|
|
7137
|
+
causalError,
|
|
7138
|
+
enqueuedAt: this.now(),
|
|
7139
|
+
resolve,
|
|
7140
|
+
reject,
|
|
7141
|
+
signal
|
|
7142
|
+
};
|
|
7143
|
+
if (signal) {
|
|
7144
|
+
waiter.onAbort = () => {
|
|
7145
|
+
const index = this.waiters.indexOf(waiter);
|
|
7146
|
+
if (index === -1) return;
|
|
7147
|
+
this.waiters.splice(index, 1);
|
|
7148
|
+
reject(signal.reason);
|
|
7149
|
+
if (waiter.wakeAt === this.drainTimerAt && !this.waiters.some((candidate) => candidate.wakeAt === waiter.wakeAt)) this.scheduleNextWake();
|
|
7150
|
+
};
|
|
7151
|
+
signal.addEventListener("abort", waiter.onAbort, { once: true });
|
|
7152
|
+
}
|
|
7153
|
+
this.waiters.push(waiter);
|
|
7154
|
+
this.emit("admission", context, { decision: "queued" });
|
|
7155
|
+
this.scheduleNextWake(waiter);
|
|
7156
|
+
});
|
|
7157
|
+
}
|
|
7158
|
+
prepareCooldownWait(context) {
|
|
7159
|
+
const cooldown = this.getActiveCooldown(context.effectiveModel);
|
|
7160
|
+
if (!cooldown) return;
|
|
7161
|
+
this.startRecovery(context.recovery);
|
|
7162
|
+
this.setRecoveryCooldown(context.recovery, cooldown);
|
|
7163
|
+
const status = cooldown.scope === "account" ? 429 : 529;
|
|
7164
|
+
context.recovery.publicError = {
|
|
7165
|
+
status,
|
|
7166
|
+
retryAfter: formatRetryAfter(cooldown.notBeforeMonotonicMs - this.now())
|
|
7167
|
+
};
|
|
7168
|
+
if (cooldown.notBeforeMonotonicMs > context.recovery.deadlineMonotonicMs) throw createLocalCapacityError(context.recovery);
|
|
7169
|
+
}
|
|
7170
|
+
drain() {
|
|
7171
|
+
this.clearExpiredModels();
|
|
7172
|
+
for (let index = this.waiters.length - 1; index >= 0; index--) {
|
|
7173
|
+
const waiter = this.waiters[index];
|
|
7174
|
+
const fallbackCooldown = waiter.context.fallbackAttempt ? this.getActiveCooldown(waiter.context.effectiveModel) : void 0;
|
|
7175
|
+
if (fallbackCooldown) {
|
|
7176
|
+
this.waiters.splice(index, 1);
|
|
7177
|
+
this.cleanupWaiter(waiter);
|
|
7178
|
+
this.emit("admission", waiter.context, {
|
|
7179
|
+
scope: fallbackCooldown.scope,
|
|
7180
|
+
decision: "fallback-cooldown"
|
|
7181
|
+
});
|
|
7182
|
+
waiter.reject(new FallbackCooldownError(fallbackCooldown));
|
|
7183
|
+
continue;
|
|
7184
|
+
}
|
|
7185
|
+
const deadline = waiter.context.recovery.deadlineMonotonicMs;
|
|
7186
|
+
if (deadline !== void 0 && this.now() >= deadline) {
|
|
7187
|
+
this.waiters.splice(index, 1);
|
|
7188
|
+
this.cleanupWaiter(waiter);
|
|
7189
|
+
waiter.reject(waiter.causalError ?? createLocalCapacityError(waiter.context.recovery));
|
|
7190
|
+
}
|
|
7191
|
+
}
|
|
7192
|
+
while (this.active < this.options.concurrency) {
|
|
7193
|
+
const index = this.waiters.findIndex((waiter) => this.isEligible(waiter.context));
|
|
7194
|
+
if (index === -1) break;
|
|
7195
|
+
const waiter = this.waiters.splice(index, 1)[0];
|
|
7196
|
+
this.cleanupWaiter(waiter);
|
|
7197
|
+
waiter.resolve(this.grant(waiter.context, this.now() - waiter.enqueuedAt));
|
|
7198
|
+
}
|
|
7199
|
+
this.scheduleNextWake();
|
|
7200
|
+
}
|
|
7201
|
+
scheduleNextWake(addedWaiter) {
|
|
7202
|
+
const now = this.now();
|
|
7203
|
+
if (addedWaiter) {
|
|
7204
|
+
const wakeAt = this.getWaiterWakeAt(addedWaiter, now);
|
|
7205
|
+
addedWaiter.wakeAt = wakeAt;
|
|
7206
|
+
if (wakeAt === void 0 || this.drainTimerAt !== void 0 && wakeAt >= this.drainTimerAt) return;
|
|
7207
|
+
this.replaceDrainTimer(wakeAt, now);
|
|
7208
|
+
return;
|
|
7209
|
+
}
|
|
7210
|
+
let wakeAt;
|
|
7211
|
+
for (const waiter of this.waiters) {
|
|
7212
|
+
waiter.wakeAt = this.getWaiterWakeAt(waiter, now);
|
|
7213
|
+
if (waiter.wakeAt !== void 0) wakeAt = Math.min(wakeAt ?? Number.POSITIVE_INFINITY, waiter.wakeAt);
|
|
7214
|
+
}
|
|
7215
|
+
this.replaceDrainTimer(wakeAt, now);
|
|
7216
|
+
}
|
|
7217
|
+
getWaiterWakeAt(waiter, now) {
|
|
7218
|
+
let wakeAt;
|
|
7219
|
+
const cooldown = this.getActiveCooldown(waiter.context.effectiveModel);
|
|
7220
|
+
if (cooldown && cooldown.notBeforeMonotonicMs > now) wakeAt = cooldown.notBeforeMonotonicMs;
|
|
7221
|
+
const deadline = waiter.context.recovery.deadlineMonotonicMs;
|
|
7222
|
+
if (deadline !== void 0 && deadline > now) wakeAt = Math.min(wakeAt ?? Number.POSITIVE_INFINITY, deadline);
|
|
7223
|
+
return wakeAt;
|
|
7224
|
+
}
|
|
7225
|
+
replaceDrainTimer(wakeAt, now) {
|
|
7226
|
+
if (wakeAt === this.drainTimerAt) return;
|
|
7227
|
+
if (this.drainTimer) {
|
|
7228
|
+
this.clearTimer(this.drainTimer);
|
|
7229
|
+
this.drainTimer = void 0;
|
|
7230
|
+
this.drainTimerAt = void 0;
|
|
7231
|
+
}
|
|
7232
|
+
if (wakeAt === void 0) return;
|
|
7233
|
+
this.drainTimerAt = wakeAt;
|
|
7234
|
+
this.drainTimer = this.setTimer(() => {
|
|
7235
|
+
this.drainTimer = void 0;
|
|
7236
|
+
this.drainTimerAt = void 0;
|
|
7237
|
+
this.drain();
|
|
7238
|
+
}, Math.min(MAX_TIMER_DELAY_MS, Math.max(0, wakeAt - now)));
|
|
7239
|
+
}
|
|
7240
|
+
grant(context, queueWaitMs) {
|
|
7241
|
+
let released = false;
|
|
7242
|
+
this.active++;
|
|
7243
|
+
this.emit("grant", context, {
|
|
7244
|
+
queueWaitMs,
|
|
7245
|
+
decision: "granted"
|
|
7246
|
+
});
|
|
7247
|
+
return { release: () => {
|
|
7248
|
+
if (released) return;
|
|
7249
|
+
released = true;
|
|
7250
|
+
this.active--;
|
|
7251
|
+
this.drain();
|
|
7252
|
+
} };
|
|
7253
|
+
}
|
|
7254
|
+
isEligible(context) {
|
|
7255
|
+
return this.getActiveCooldown(context.effectiveModel) === void 0;
|
|
7256
|
+
}
|
|
7257
|
+
throwIfFallbackCooled(context) {
|
|
7258
|
+
if (!context.fallbackAttempt) return;
|
|
7259
|
+
const cooldown = this.getActiveCooldown(context.effectiveModel);
|
|
7260
|
+
if (cooldown) throw new FallbackCooldownError(cooldown);
|
|
7261
|
+
}
|
|
7262
|
+
getActiveCooldown(effectiveModel) {
|
|
7263
|
+
const now = this.now();
|
|
7264
|
+
if (this.accountNotBefore > now) return {
|
|
7265
|
+
scope: "account",
|
|
7266
|
+
notBeforeMonotonicMs: this.accountNotBefore
|
|
7267
|
+
};
|
|
7268
|
+
if (!effectiveModel) return void 0;
|
|
7269
|
+
const modelDeadline = this.modelNotBefore.get(effectiveModel);
|
|
7270
|
+
if (modelDeadline === void 0) return void 0;
|
|
7271
|
+
if (modelDeadline <= now) {
|
|
7272
|
+
this.modelNotBefore.delete(effectiveModel);
|
|
7273
|
+
return;
|
|
7274
|
+
}
|
|
7275
|
+
return {
|
|
7276
|
+
scope: "model",
|
|
7277
|
+
notBeforeMonotonicMs: modelDeadline,
|
|
7278
|
+
effectiveModel
|
|
7279
|
+
};
|
|
7280
|
+
}
|
|
7281
|
+
clearExpiredModels() {
|
|
7282
|
+
const now = this.now();
|
|
7283
|
+
for (const [model, deadline] of this.modelNotBefore) if (deadline <= now) this.modelNotBefore.delete(model);
|
|
7284
|
+
}
|
|
7285
|
+
installCooldown(scope, effectiveModel, delayMs, context) {
|
|
7286
|
+
const deadline = this.now() + delayMs;
|
|
7287
|
+
let stored = deadline;
|
|
7288
|
+
if (scope === "account") {
|
|
7289
|
+
this.accountNotBefore = Math.max(this.accountNotBefore, deadline);
|
|
7290
|
+
stored = this.accountNotBefore;
|
|
7291
|
+
} else if (scope === "model" && effectiveModel) {
|
|
7292
|
+
stored = Math.max(this.modelNotBefore.get(effectiveModel) ?? 0, deadline);
|
|
7293
|
+
this.modelNotBefore.set(effectiveModel, stored);
|
|
7294
|
+
}
|
|
7295
|
+
this.setRecoveryCooldown(context.recovery, {
|
|
7296
|
+
scope,
|
|
7297
|
+
notBeforeMonotonicMs: stored,
|
|
7298
|
+
...effectiveModel && scope === "model" ? { effectiveModel } : {}
|
|
7299
|
+
});
|
|
7300
|
+
this.emit("cooldown", context, {
|
|
7301
|
+
scope,
|
|
7302
|
+
delayMs,
|
|
7303
|
+
nextRetryAt: formatNextRetryAt(this.wallNow() + delayMs),
|
|
7304
|
+
decision: scope === "request" ? "request-local" : "installed"
|
|
7305
|
+
});
|
|
7306
|
+
this.drain();
|
|
7307
|
+
}
|
|
7308
|
+
startRecovery(recovery) {
|
|
7309
|
+
if (recovery.deadlineMonotonicMs !== void 0) return;
|
|
7310
|
+
const startedAt = this.now();
|
|
7311
|
+
recovery.startedAtMonotonicMs = startedAt;
|
|
7312
|
+
recovery.deadlineMonotonicMs = startedAt + this.options.recoveryBudgetMs;
|
|
7313
|
+
recovery.retryLimit ??= this.options.maxRetries;
|
|
7314
|
+
}
|
|
7315
|
+
setRecoveryCooldown(recovery, cooldown) {
|
|
7316
|
+
if (!recovery.cooldown || cooldown.notBeforeMonotonicMs >= recovery.cooldown.notBeforeMonotonicMs) recovery.cooldown = cooldown;
|
|
7317
|
+
}
|
|
7318
|
+
canRetry(recovery) {
|
|
7319
|
+
return recovery.retryCount < (recovery.retryLimit ?? this.options.maxRetries);
|
|
7320
|
+
}
|
|
7321
|
+
remainingBudget(recovery) {
|
|
7322
|
+
return recovery.deadlineMonotonicMs === void 0 ? this.options.recoveryBudgetMs : Math.max(0, recovery.deadlineMonotonicMs - this.now());
|
|
7323
|
+
}
|
|
7324
|
+
throwIfRecoveryExpired(recovery, lastConnectionError) {
|
|
7325
|
+
if (recovery.deadlineMonotonicMs !== void 0 && this.now() >= recovery.deadlineMonotonicMs) throw lastConnectionError ?? createLocalCapacityError(recovery);
|
|
7326
|
+
}
|
|
7327
|
+
getRetryDelay(response, attempt, recovery) {
|
|
7328
|
+
const retryAfterMs = parseRetryAfterMs(response.headers, this.wallNow());
|
|
7329
|
+
const retryAfter = response.headers.get("retry-after") ?? void 0;
|
|
7330
|
+
if (retryAfterMs !== void 0) return {
|
|
7331
|
+
delayMs: retryAfterMs,
|
|
7332
|
+
source: "retry-after",
|
|
7333
|
+
retryAfter
|
|
7334
|
+
};
|
|
7335
|
+
return {
|
|
7336
|
+
delayMs: this.getBackoffDelay(attempt, this.remainingBudget(recovery)),
|
|
7337
|
+
source: "backoff"
|
|
7338
|
+
};
|
|
7339
|
+
}
|
|
7340
|
+
getBackoffDelay(attempt, remainingBudgetMs) {
|
|
7341
|
+
const cap = Math.min(this.options.baseDelayMs * 2 ** attempt, this.options.maxDelayMs, Math.max(0, remainingBudgetMs));
|
|
7342
|
+
const random = Math.min(1, Math.max(0, this.random()));
|
|
7343
|
+
return Math.floor(cap * random);
|
|
7344
|
+
}
|
|
7345
|
+
async waitForRecovery(delayMs, signal, recovery, lastConnectionError) {
|
|
7346
|
+
const remaining = this.remainingBudget(recovery);
|
|
7347
|
+
if (delayMs > remaining) throw lastConnectionError ?? createLocalCapacityError(recovery);
|
|
7348
|
+
const deadline = createDeadlineSignal(signal, remaining, this.setTimer, this.clearTimer);
|
|
7349
|
+
try {
|
|
7350
|
+
await abortableSleep(this.sleep, delayMs, deadline.signal, this.setTimer, this.clearTimer);
|
|
7351
|
+
} catch (error) {
|
|
7352
|
+
if (signal?.aborted) throw signal.reason;
|
|
7353
|
+
if (deadline.timedOut()) throw lastConnectionError ?? createLocalCapacityError(recovery);
|
|
7354
|
+
throw error;
|
|
7355
|
+
} finally {
|
|
7356
|
+
deadline.cleanup();
|
|
7357
|
+
}
|
|
7358
|
+
this.throwIfRecoveryExpired(recovery, lastConnectionError);
|
|
7359
|
+
}
|
|
7360
|
+
async fetchBeforeDeadline(fetcher, signal, recovery, lastConnectionError) {
|
|
7361
|
+
if (recovery.deadlineMonotonicMs === void 0) return fetcher(signal);
|
|
7362
|
+
const deadline = createDeadlineSignal(signal, this.remainingBudget(recovery), this.setTimer, this.clearTimer);
|
|
7363
|
+
try {
|
|
7364
|
+
const response = await fetcher(deadline.signal);
|
|
7365
|
+
if (this.now() >= recovery.deadlineMonotonicMs) {
|
|
7366
|
+
discardResponse(response);
|
|
7367
|
+
throw new RecoveryBudgetError(recovery.fallbackFetchStarted ? createRecoveryTimeoutError() : lastConnectionError ?? createLocalCapacityError(recovery));
|
|
7368
|
+
}
|
|
7369
|
+
return response;
|
|
7370
|
+
} catch (error) {
|
|
7371
|
+
if (signal?.aborted) throw signal.reason;
|
|
7372
|
+
if (deadline.timedOut()) throw new RecoveryBudgetError(recovery.fallbackFetchStarted ? createRecoveryTimeoutError() : lastConnectionError ?? createLocalCapacityError(recovery));
|
|
7373
|
+
throw error;
|
|
7374
|
+
} finally {
|
|
7375
|
+
deadline.cleanup();
|
|
7376
|
+
}
|
|
7377
|
+
}
|
|
7378
|
+
committed(response, lease, context, retryDelay, terminalDecision = "upstream-terminal") {
|
|
7379
|
+
const { recovery } = context;
|
|
7380
|
+
if (recovery.startedAtMonotonicMs !== void 0) this.emitTerminal("retry", context, {
|
|
7381
|
+
retryCount: recovery.retryCount,
|
|
7382
|
+
status: response.status,
|
|
7383
|
+
scope: resolveCapacityCooldownScope(response.status, context.effectiveModel),
|
|
7384
|
+
decision: response.ok ? "recovered" : terminalDecision
|
|
7385
|
+
});
|
|
7386
|
+
return {
|
|
7387
|
+
response: retryDelay ? ensureRetryAfter(response, retryDelay.retryAfter ?? formatRetryAfter(retryDelay.delayMs)) : response,
|
|
7388
|
+
release: lease.release,
|
|
7389
|
+
recovery
|
|
7390
|
+
};
|
|
7391
|
+
}
|
|
7392
|
+
emit(event, context, fields) {
|
|
7393
|
+
const recovery = context.recovery;
|
|
7394
|
+
recovery.queueMetrics = {
|
|
7395
|
+
activeSlots: this.active,
|
|
7396
|
+
maxSlots: this.options.concurrency,
|
|
7397
|
+
pendingDepth: this.waiters.length,
|
|
7398
|
+
maxPendingDepth: this.options.maxQueueDepth
|
|
7399
|
+
};
|
|
7400
|
+
if (!this.logger.info) return;
|
|
7401
|
+
logRecoveryEvent({
|
|
7402
|
+
requestId: recovery.requestId,
|
|
7403
|
+
callerRequestId: recovery.callerRequestId,
|
|
7404
|
+
event,
|
|
7405
|
+
effectiveModel: context.effectiveModel,
|
|
7406
|
+
...recovery.queueMetrics,
|
|
7407
|
+
...recovery.startedAtMonotonicMs !== void 0 ? {
|
|
7408
|
+
elapsedMs: Math.max(0, this.now() - recovery.startedAtMonotonicMs),
|
|
7409
|
+
remainingBudgetMs: this.remainingBudget(recovery)
|
|
7410
|
+
} : {},
|
|
7411
|
+
...fields
|
|
7412
|
+
}, { info: this.logger.info.bind(this.logger) });
|
|
7413
|
+
}
|
|
7414
|
+
emitTerminal(event, context, fields) {
|
|
7415
|
+
if (this.terminalRecoveries.has(context.recovery)) return;
|
|
7416
|
+
this.terminalRecoveries.add(context.recovery);
|
|
7417
|
+
this.emit(event, context, fields);
|
|
7418
|
+
}
|
|
7419
|
+
cleanupWaiter(waiter) {
|
|
7420
|
+
if (waiter.signal && waiter.onAbort) waiter.signal.removeEventListener("abort", waiter.onAbort);
|
|
7421
|
+
}
|
|
7422
|
+
};
|
|
7423
|
+
function createDefaultUpstreamRequestQueue() {
|
|
7424
|
+
return new UpstreamRequestQueue(DEFAULT_UPSTREAM_QUEUE_OPTIONS);
|
|
7425
|
+
}
|
|
7426
|
+
function parseRetryAfterMs(headers, now = Date.now()) {
|
|
7427
|
+
const value = headers.get("retry-after");
|
|
7428
|
+
if (!value) return void 0;
|
|
7429
|
+
if (RETRY_AFTER_SECONDS_RE.test(value)) {
|
|
7430
|
+
const milliseconds = Number(value) * 1e3;
|
|
7431
|
+
return Number.isFinite(milliseconds) ? Math.ceil(milliseconds) : void 0;
|
|
7432
|
+
}
|
|
7433
|
+
if (!RETRY_AFTER_HTTP_DATE_RE.test(value)) return void 0;
|
|
7434
|
+
const retryAt = Date.parse(value);
|
|
7435
|
+
if (Number.isNaN(retryAt)) return void 0;
|
|
7436
|
+
const serverDate = headers.get("date");
|
|
7437
|
+
const parsedServerDate = serverDate ? Date.parse(serverDate) : NaN;
|
|
7438
|
+
return Math.max(0, retryAt - (Number.isNaN(parsedServerDate) ? now : parsedServerDate));
|
|
7439
|
+
}
|
|
7440
|
+
function finiteOr(value, fallback) {
|
|
7441
|
+
return value !== void 0 && Number.isFinite(value) ? value : fallback;
|
|
7442
|
+
}
|
|
7443
|
+
function normalizeOptions(options) {
|
|
7444
|
+
return {
|
|
7445
|
+
concurrency: Math.max(1, Math.floor(finiteOr(options.concurrency, DEFAULT_UPSTREAM_QUEUE_OPTIONS.concurrency))),
|
|
7446
|
+
maxRetries: Math.min(2, Math.max(0, Math.floor(finiteOr(options.maxRetries, DEFAULT_UPSTREAM_QUEUE_OPTIONS.maxRetries)))),
|
|
7447
|
+
baseDelayMs: Math.max(0, Math.floor(finiteOr(options.baseDelayMs, DEFAULT_UPSTREAM_QUEUE_OPTIONS.baseDelayMs))),
|
|
7448
|
+
maxDelayMs: Math.max(1, Math.floor(finiteOr(options.maxDelayMs, DEFAULT_UPSTREAM_QUEUE_OPTIONS.maxDelayMs))),
|
|
7449
|
+
maxQueueDepth: Math.max(1, Math.floor(finiteOr(options.maxQueueDepth, DEFAULT_UPSTREAM_QUEUE_OPTIONS.maxQueueDepth))),
|
|
7450
|
+
recoveryBudgetMs: Math.min(120 * 1e3, Math.max(1 * 1e3, Math.floor(finiteOr(options.recoveryBudgetMs, DEFAULT_UPSTREAM_QUEUE_OPTIONS.recoveryBudgetMs))))
|
|
7451
|
+
};
|
|
7452
|
+
}
|
|
7453
|
+
function mergeDefinedOptions(current, next) {
|
|
7454
|
+
return {
|
|
7455
|
+
concurrency: next.concurrency ?? current.concurrency,
|
|
7456
|
+
maxRetries: next.maxRetries ?? current.maxRetries,
|
|
7457
|
+
baseDelayMs: next.baseDelayMs ?? current.baseDelayMs,
|
|
7458
|
+
maxDelayMs: next.maxDelayMs ?? current.maxDelayMs,
|
|
7459
|
+
maxQueueDepth: next.maxQueueDepth ?? current.maxQueueDepth,
|
|
7460
|
+
recoveryBudgetMs: next.recoveryBudgetMs ?? current.recoveryBudgetMs
|
|
7461
|
+
};
|
|
7462
|
+
}
|
|
7463
|
+
function discardResponse(response) {
|
|
7464
|
+
try {
|
|
7465
|
+
response.body?.cancel().catch(() => {});
|
|
7466
|
+
} catch {}
|
|
7467
|
+
}
|
|
7468
|
+
function ensureRetryAfter(response, retryAfter) {
|
|
7469
|
+
if (response.headers.get("retry-after") === retryAfter) return response;
|
|
7470
|
+
const headers = new Headers(response.headers);
|
|
7471
|
+
headers.set("retry-after", retryAfter);
|
|
7472
|
+
return new Response(response.body, {
|
|
7473
|
+
status: response.status,
|
|
7474
|
+
statusText: response.statusText,
|
|
7475
|
+
headers
|
|
7476
|
+
});
|
|
7477
|
+
}
|
|
7478
|
+
function formatRetryAfter(delayMs) {
|
|
7479
|
+
return String(Math.max(0, Math.ceil(delayMs / 1e3)));
|
|
7480
|
+
}
|
|
7481
|
+
function formatNextRetryAt(timestampMs) {
|
|
7482
|
+
const retryAt = new Date(timestampMs);
|
|
7483
|
+
return Number.isNaN(retryAt.getTime()) ? void 0 : retryAt.toISOString();
|
|
7484
|
+
}
|
|
7485
|
+
function formatRequestContext(context) {
|
|
7486
|
+
try {
|
|
7487
|
+
const url = new URL(context.url);
|
|
7488
|
+
return `${context.method ?? "GET"} ${url.pathname}`;
|
|
7489
|
+
} catch {
|
|
7490
|
+
return `${context.method ?? "GET"} ${context.url}`;
|
|
7491
|
+
}
|
|
7492
|
+
}
|
|
7493
|
+
function abortableSleep(sleep, ms, signal, setTimer = globalThis.setTimeout, clearTimer = globalThis.clearTimeout) {
|
|
7494
|
+
if (sleep && !signal) return sleep(ms);
|
|
7495
|
+
signal?.throwIfAborted();
|
|
7496
|
+
return new Promise((resolve, reject) => {
|
|
7497
|
+
let timer;
|
|
7498
|
+
function cleanup() {
|
|
7499
|
+
if (timer !== void 0) clearTimer(timer);
|
|
7500
|
+
signal?.removeEventListener("abort", onAbort);
|
|
7501
|
+
}
|
|
7502
|
+
function onAbort() {
|
|
7503
|
+
cleanup();
|
|
7504
|
+
reject(signal?.reason);
|
|
7505
|
+
}
|
|
7506
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
7507
|
+
if (!sleep) {
|
|
7508
|
+
timer = setTimer(() => {
|
|
7509
|
+
cleanup();
|
|
7510
|
+
resolve();
|
|
7511
|
+
}, ms);
|
|
7512
|
+
return;
|
|
7513
|
+
}
|
|
7514
|
+
sleep(ms).then(() => {
|
|
7515
|
+
cleanup();
|
|
7516
|
+
resolve();
|
|
7517
|
+
}, (error) => {
|
|
7518
|
+
cleanup();
|
|
7519
|
+
reject(error);
|
|
7520
|
+
});
|
|
7521
|
+
});
|
|
7522
|
+
}
|
|
7523
|
+
function createDeadlineSignal(parent, delayMs, setTimer, clearTimer) {
|
|
7524
|
+
const controller = new AbortController();
|
|
7525
|
+
let didTimeOut = false;
|
|
7526
|
+
const timer = setTimer(() => {
|
|
7527
|
+
didTimeOut = true;
|
|
7528
|
+
controller.abort(new DOMException("Recovery deadline exceeded", "TimeoutError"));
|
|
7529
|
+
}, Math.max(0, delayMs));
|
|
7530
|
+
return {
|
|
7531
|
+
signal: parent ? AbortSignal.any([parent, controller.signal]) : controller.signal,
|
|
7532
|
+
timedOut: () => didTimeOut,
|
|
7533
|
+
cleanup: () => clearTimer(timer)
|
|
7534
|
+
};
|
|
7535
|
+
}
|
|
7536
|
+
var RecoveryBudgetError = class extends Error {
|
|
7537
|
+
cause;
|
|
7538
|
+
constructor(cause) {
|
|
7539
|
+
super("Upstream recovery deadline exceeded");
|
|
7540
|
+
this.name = "RecoveryBudgetError";
|
|
7541
|
+
this.cause = cause;
|
|
7542
|
+
}
|
|
7543
|
+
};
|
|
7544
|
+
function createRecoveryTimeoutError() {
|
|
7545
|
+
return new HTTPError(504, { error: {
|
|
7546
|
+
message: localCapacityErrorMessage(504),
|
|
7547
|
+
type: "timeout_error"
|
|
7548
|
+
} });
|
|
7549
|
+
}
|
|
7550
|
+
function createLocalCapacityError(recovery) {
|
|
7551
|
+
const status = recovery.publicError?.status ?? 504;
|
|
7552
|
+
const retryAfter = recovery.publicError?.retryAfter ?? (recovery.cooldown ? formatRetryAfter(recovery.cooldown.notBeforeMonotonicMs - (recovery.deadlineMonotonicMs ?? 0)) : void 0);
|
|
7553
|
+
const errorType = status === 504 ? "timeout_error" : upstreamErrorType(status);
|
|
7554
|
+
const error = new HTTPError(status, { error: {
|
|
7555
|
+
message: localCapacityErrorMessage(status),
|
|
7556
|
+
type: errorType
|
|
7557
|
+
} }, retryAfter ? { headers: { "retry-after": retryAfter } } : void 0);
|
|
7558
|
+
return status === 529 ? new TerminalUpstreamRecoveryError(error, recovery) : error;
|
|
7559
|
+
}
|
|
7560
|
+
function localCapacityErrorMessage(status) {
|
|
7561
|
+
switch (status) {
|
|
7562
|
+
case 429: return "The upstream account is temporarily rate limited.";
|
|
7563
|
+
case 529: return "The selected upstream model is temporarily overloaded.";
|
|
7564
|
+
case 504: return "The upstream recovery budget was exhausted.";
|
|
7565
|
+
default: return `The last upstream attempt failed with status ${status}.`;
|
|
7566
|
+
}
|
|
7567
|
+
}
|
|
7568
|
+
//#endregion
|
|
6582
7569
|
//#region src/clients/copilot-client.ts
|
|
6583
7570
|
var CopilotClient = class {
|
|
6584
7571
|
auth;
|
|
6585
7572
|
config;
|
|
6586
7573
|
fetchImpl;
|
|
6587
7574
|
requestQueue;
|
|
7575
|
+
recovery;
|
|
7576
|
+
offerLocalModelCooldown;
|
|
7577
|
+
fallbackAttempt;
|
|
6588
7578
|
constructor(auth, config, deps) {
|
|
6589
7579
|
this.auth = auth;
|
|
6590
7580
|
this.config = config;
|
|
6591
7581
|
this.fetchImpl = deps?.fetch ?? fetch;
|
|
6592
7582
|
this.requestQueue = deps?.requestQueue;
|
|
7583
|
+
this.recovery = deps?.recovery;
|
|
7584
|
+
this.offerLocalModelCooldown = deps?.offerLocalModelCooldown ?? false;
|
|
7585
|
+
this.fallbackAttempt = deps?.fallbackAttempt ?? false;
|
|
6593
7586
|
}
|
|
6594
7587
|
requireToken() {
|
|
6595
7588
|
if (!this.auth.copilotToken) throw new Error("Copilot token not found");
|
|
@@ -6610,10 +7603,13 @@ var CopilotClient = class {
|
|
|
6610
7603
|
signal: options.signal
|
|
6611
7604
|
}
|
|
6612
7605
|
};
|
|
6613
|
-
const queuedResponse = await this.fetchWithQueue(request, options.retryable);
|
|
7606
|
+
const queuedResponse = await this.fetchWithQueue(request, options.retryable, options.effectiveModel);
|
|
6614
7607
|
const { response } = queuedResponse;
|
|
6615
7608
|
if (!response.ok) try {
|
|
6616
7609
|
await throwUpstreamError(errorMessage, response);
|
|
7610
|
+
} catch (error) {
|
|
7611
|
+
if (error instanceof HTTPError && response.status === 529 && queuedResponse.recovery) throw new TerminalUpstreamRecoveryError(error, queuedResponse.recovery);
|
|
7612
|
+
throw error;
|
|
6617
7613
|
} finally {
|
|
6618
7614
|
queuedResponse.release();
|
|
6619
7615
|
}
|
|
@@ -6634,6 +7630,7 @@ var CopilotClient = class {
|
|
|
6634
7630
|
method: "POST",
|
|
6635
7631
|
body: JSON.stringify(payload),
|
|
6636
7632
|
retryable: "capacity",
|
|
7633
|
+
effectiveModel: typeof payload.model === "string" ? payload.model : void 0,
|
|
6637
7634
|
...options
|
|
6638
7635
|
});
|
|
6639
7636
|
if (payload.stream) return withRelease(events(response), release);
|
|
@@ -6643,16 +7640,28 @@ var CopilotClient = class {
|
|
|
6643
7640
|
release();
|
|
6644
7641
|
}
|
|
6645
7642
|
}
|
|
6646
|
-
async fetchWithQueue(request, retryable) {
|
|
6647
|
-
const fetcher = () => this.fetchImpl(request.url,
|
|
7643
|
+
async fetchWithQueue(request, retryable, effectiveModel) {
|
|
7644
|
+
const fetcher = (signal) => this.fetchImpl(request.url, {
|
|
7645
|
+
...request.init,
|
|
7646
|
+
signal: signal ?? request.init.signal
|
|
7647
|
+
});
|
|
6648
7648
|
if (this.requestQueue) return this.requestQueue.dispatch(fetcher, {
|
|
6649
7649
|
method: request.init.method,
|
|
6650
7650
|
url: request.url,
|
|
6651
|
-
retryable
|
|
7651
|
+
retryable,
|
|
7652
|
+
effectiveModel,
|
|
7653
|
+
recovery: this.recovery,
|
|
7654
|
+
offerLocalModelCooldown: this.offerLocalModelCooldown,
|
|
7655
|
+
fallbackAttempt: this.fallbackAttempt
|
|
6652
7656
|
}, request.init.signal ?? void 0);
|
|
7657
|
+
const recovery = this.recovery ?? {
|
|
7658
|
+
requestId: crypto.randomUUID(),
|
|
7659
|
+
retryCount: 0
|
|
7660
|
+
};
|
|
6653
7661
|
return {
|
|
6654
|
-
response: await fetcher(),
|
|
6655
|
-
release: () => {}
|
|
7662
|
+
response: await fetcher(request.init.signal ?? void 0),
|
|
7663
|
+
release: () => {},
|
|
7664
|
+
recovery
|
|
6656
7665
|
};
|
|
6657
7666
|
}
|
|
6658
7667
|
async createChatCompletions(payload, options) {
|
|
@@ -6672,6 +7681,7 @@ var CopilotClient = class {
|
|
|
6672
7681
|
method: "POST",
|
|
6673
7682
|
body: JSON.stringify(payload),
|
|
6674
7683
|
signal: options?.signal,
|
|
7684
|
+
effectiveModel: payload.model,
|
|
6675
7685
|
retryable: true
|
|
6676
7686
|
});
|
|
6677
7687
|
}
|
|
@@ -6968,202 +7978,6 @@ function buildGitHubUrls(gheDomain) {
|
|
|
6968
7978
|
};
|
|
6969
7979
|
}
|
|
6970
7980
|
//#endregion
|
|
6971
|
-
//#region src/util/duration.ts
|
|
6972
|
-
/**
|
|
6973
|
-
* Formats a millisecond duration as a compact human-readable string:
|
|
6974
|
-
* `<n>ms` under one second, otherwise `<n>s` rounded to whole seconds.
|
|
6975
|
-
*/
|
|
6976
|
-
function formatDurationMs(ms) {
|
|
6977
|
-
return ms < 1e3 ? `${ms}ms` : `${Math.round(ms / 1e3)}s`;
|
|
6978
|
-
}
|
|
6979
|
-
//#endregion
|
|
6980
|
-
//#region src/clients/upstream-queue.ts
|
|
6981
|
-
const DEFAULT_UPSTREAM_QUEUE_OPTIONS = {
|
|
6982
|
-
concurrency: 10,
|
|
6983
|
-
maxRetries: 5,
|
|
6984
|
-
baseDelayMs: 2e3,
|
|
6985
|
-
maxDelayMs: 6e4,
|
|
6986
|
-
maxQueueDepth: 1e3
|
|
6987
|
-
};
|
|
6988
|
-
var UpstreamRequestQueue = class {
|
|
6989
|
-
sleep;
|
|
6990
|
-
now;
|
|
6991
|
-
logger;
|
|
6992
|
-
setTimer;
|
|
6993
|
-
clearTimer;
|
|
6994
|
-
options;
|
|
6995
|
-
active = 0;
|
|
6996
|
-
cooldownUntil = 0;
|
|
6997
|
-
drainTimer;
|
|
6998
|
-
waiters = [];
|
|
6999
|
-
constructor(options = {}, deps = {}) {
|
|
7000
|
-
this.options = normalizeOptions(options);
|
|
7001
|
-
this.sleep = deps.sleep ?? sleep;
|
|
7002
|
-
this.now = deps.now ?? Date.now;
|
|
7003
|
-
this.logger = deps.logger ?? consola;
|
|
7004
|
-
this.setTimer = deps.setTimeout ?? globalThis.setTimeout;
|
|
7005
|
-
this.clearTimer = deps.clearTimeout ?? globalThis.clearTimeout;
|
|
7006
|
-
}
|
|
7007
|
-
updateOptions(options) {
|
|
7008
|
-
this.options = normalizeOptions(mergeDefinedOptions(this.options, options));
|
|
7009
|
-
this.drain();
|
|
7010
|
-
}
|
|
7011
|
-
async dispatch(fetcher, context, signal) {
|
|
7012
|
-
let attempt = 0;
|
|
7013
|
-
for (;;) {
|
|
7014
|
-
signal?.throwIfAborted();
|
|
7015
|
-
const lease = await this.acquire(signal);
|
|
7016
|
-
let response;
|
|
7017
|
-
try {
|
|
7018
|
-
response = await fetcher();
|
|
7019
|
-
} catch (error) {
|
|
7020
|
-
lease.release();
|
|
7021
|
-
throw error;
|
|
7022
|
-
}
|
|
7023
|
-
const { status } = response;
|
|
7024
|
-
const isCapacityLimit = isCapacityLimitStatus(status);
|
|
7025
|
-
if (!((context.retryable === "capacity" ? isCapacityLimit : context.retryable === true && isTransientUpstreamStatus(status)) && attempt < this.options.maxRetries)) {
|
|
7026
|
-
if (isCapacityLimit) this.applyCooldown(this.getRetryDelayMs(response, 0));
|
|
7027
|
-
return {
|
|
7028
|
-
response,
|
|
7029
|
-
release: lease.release
|
|
7030
|
-
};
|
|
7031
|
-
}
|
|
7032
|
-
const delayMs = this.getRetryDelayMs(response, attempt);
|
|
7033
|
-
await discardResponse(response);
|
|
7034
|
-
if (isCapacityLimit) this.applyCooldown(delayMs);
|
|
7035
|
-
lease.release();
|
|
7036
|
-
this.logger.warn([
|
|
7037
|
-
`Upstream ${status};`,
|
|
7038
|
-
`retrying ${formatRequestContext(context)}`,
|
|
7039
|
-
`in ${formatDurationMs(delayMs)}`,
|
|
7040
|
-
`(attempt ${attempt + 1}/${this.options.maxRetries})`
|
|
7041
|
-
].join(" "));
|
|
7042
|
-
await abortableSleep(this.sleep, delayMs, signal);
|
|
7043
|
-
attempt++;
|
|
7044
|
-
}
|
|
7045
|
-
}
|
|
7046
|
-
acquire(signal) {
|
|
7047
|
-
signal?.throwIfAborted();
|
|
7048
|
-
if (this.waiters.length >= this.options.maxQueueDepth) return Promise.reject(new HTTPError(503, { error: {
|
|
7049
|
-
message: "Upstream queue full",
|
|
7050
|
-
type: "overloaded_error"
|
|
7051
|
-
} }));
|
|
7052
|
-
return new Promise((resolve, reject) => {
|
|
7053
|
-
let resolved = false;
|
|
7054
|
-
const waiter = (lease) => {
|
|
7055
|
-
resolved = true;
|
|
7056
|
-
resolve(lease);
|
|
7057
|
-
};
|
|
7058
|
-
this.waiters.push(waiter);
|
|
7059
|
-
if (signal) signal.addEventListener("abort", () => {
|
|
7060
|
-
if (resolved) return;
|
|
7061
|
-
const idx = this.waiters.indexOf(waiter);
|
|
7062
|
-
if (idx !== -1) this.waiters.splice(idx, 1);
|
|
7063
|
-
reject(signal.reason);
|
|
7064
|
-
}, { once: true });
|
|
7065
|
-
this.drain();
|
|
7066
|
-
});
|
|
7067
|
-
}
|
|
7068
|
-
drain() {
|
|
7069
|
-
if (this.drainTimer) {
|
|
7070
|
-
this.clearTimer(this.drainTimer);
|
|
7071
|
-
this.drainTimer = void 0;
|
|
7072
|
-
}
|
|
7073
|
-
const cooldownMs = this.cooldownUntil - this.now();
|
|
7074
|
-
if (cooldownMs > 0) {
|
|
7075
|
-
this.drainTimer = this.setTimer(() => this.drain(), cooldownMs);
|
|
7076
|
-
return;
|
|
7077
|
-
}
|
|
7078
|
-
while (this.active < this.options.concurrency && this.waiters.length > 0) {
|
|
7079
|
-
const resolve = this.waiters.shift();
|
|
7080
|
-
let released = false;
|
|
7081
|
-
this.active++;
|
|
7082
|
-
resolve({ release: () => {
|
|
7083
|
-
if (released) return;
|
|
7084
|
-
released = true;
|
|
7085
|
-
this.active--;
|
|
7086
|
-
this.drain();
|
|
7087
|
-
} });
|
|
7088
|
-
}
|
|
7089
|
-
}
|
|
7090
|
-
applyCooldown(delayMs) {
|
|
7091
|
-
this.cooldownUntil = Math.max(this.cooldownUntil, this.now() + delayMs);
|
|
7092
|
-
this.drain();
|
|
7093
|
-
}
|
|
7094
|
-
getRetryDelayMs(response, attempt) {
|
|
7095
|
-
const retryAfterMs = parseRetryAfterMs(response.headers, this.now());
|
|
7096
|
-
if (retryAfterMs !== void 0) return clampDelay(retryAfterMs, this.options.maxDelayMs);
|
|
7097
|
-
return clampDelay(this.options.baseDelayMs * 2 ** attempt, this.options.maxDelayMs);
|
|
7098
|
-
}
|
|
7099
|
-
};
|
|
7100
|
-
function createDefaultUpstreamRequestQueue() {
|
|
7101
|
-
return new UpstreamRequestQueue(DEFAULT_UPSTREAM_QUEUE_OPTIONS);
|
|
7102
|
-
}
|
|
7103
|
-
function parseRetryAfterMs(headers, now = Date.now()) {
|
|
7104
|
-
const retryAfter = headers.get("retry-after");
|
|
7105
|
-
if (!retryAfter) return;
|
|
7106
|
-
const retryAfterSeconds = Number.parseFloat(retryAfter);
|
|
7107
|
-
if (Number.isFinite(retryAfterSeconds)) return Math.max(0, retryAfterSeconds * 1e3);
|
|
7108
|
-
const retryAt = Date.parse(retryAfter);
|
|
7109
|
-
if (Number.isNaN(retryAt)) return;
|
|
7110
|
-
return Math.max(0, retryAt - now);
|
|
7111
|
-
}
|
|
7112
|
-
function finiteOr(value, fallback) {
|
|
7113
|
-
return value !== void 0 && Number.isFinite(value) ? value : fallback;
|
|
7114
|
-
}
|
|
7115
|
-
function normalizeOptions(options) {
|
|
7116
|
-
return {
|
|
7117
|
-
concurrency: Math.max(1, Math.floor(finiteOr(options.concurrency, DEFAULT_UPSTREAM_QUEUE_OPTIONS.concurrency))),
|
|
7118
|
-
maxRetries: Math.max(0, Math.floor(finiteOr(options.maxRetries, DEFAULT_UPSTREAM_QUEUE_OPTIONS.maxRetries))),
|
|
7119
|
-
baseDelayMs: Math.max(0, Math.floor(finiteOr(options.baseDelayMs, DEFAULT_UPSTREAM_QUEUE_OPTIONS.baseDelayMs))),
|
|
7120
|
-
maxDelayMs: Math.max(1, Math.floor(finiteOr(options.maxDelayMs, DEFAULT_UPSTREAM_QUEUE_OPTIONS.maxDelayMs))),
|
|
7121
|
-
maxQueueDepth: Math.max(1, Math.floor(finiteOr(options.maxQueueDepth, DEFAULT_UPSTREAM_QUEUE_OPTIONS.maxQueueDepth)))
|
|
7122
|
-
};
|
|
7123
|
-
}
|
|
7124
|
-
function mergeDefinedOptions(current, next) {
|
|
7125
|
-
return {
|
|
7126
|
-
concurrency: next.concurrency ?? current.concurrency,
|
|
7127
|
-
maxRetries: next.maxRetries ?? current.maxRetries,
|
|
7128
|
-
baseDelayMs: next.baseDelayMs ?? current.baseDelayMs,
|
|
7129
|
-
maxDelayMs: next.maxDelayMs ?? current.maxDelayMs,
|
|
7130
|
-
maxQueueDepth: next.maxQueueDepth ?? current.maxQueueDepth
|
|
7131
|
-
};
|
|
7132
|
-
}
|
|
7133
|
-
function clampDelay(delayMs, maxDelayMs) {
|
|
7134
|
-
return Math.min(Math.max(0, Math.ceil(delayMs)), maxDelayMs);
|
|
7135
|
-
}
|
|
7136
|
-
async function discardResponse(response) {
|
|
7137
|
-
try {
|
|
7138
|
-
await response.body?.cancel();
|
|
7139
|
-
} catch {}
|
|
7140
|
-
}
|
|
7141
|
-
function formatRequestContext(context) {
|
|
7142
|
-
try {
|
|
7143
|
-
const url = new URL(context.url);
|
|
7144
|
-
return `${context.method ?? "GET"} ${url.pathname}`;
|
|
7145
|
-
} catch {
|
|
7146
|
-
return `${context.method ?? "GET"} ${context.url}`;
|
|
7147
|
-
}
|
|
7148
|
-
}
|
|
7149
|
-
function abortableSleep(sleep, ms, signal) {
|
|
7150
|
-
if (!signal) return sleep(ms);
|
|
7151
|
-
signal.throwIfAborted();
|
|
7152
|
-
return new Promise((resolve, reject) => {
|
|
7153
|
-
let done = false;
|
|
7154
|
-
signal.addEventListener("abort", () => {
|
|
7155
|
-
if (done) return;
|
|
7156
|
-
done = true;
|
|
7157
|
-
reject(signal.reason);
|
|
7158
|
-
}, { once: true });
|
|
7159
|
-
sleep(ms).then(() => {
|
|
7160
|
-
if (done) return;
|
|
7161
|
-
done = true;
|
|
7162
|
-
resolve();
|
|
7163
|
-
});
|
|
7164
|
-
});
|
|
7165
|
-
}
|
|
7166
|
-
//#endregion
|
|
7167
7981
|
//#region src/clients/factory.ts
|
|
7168
7982
|
const upstreamRequestQueue = createDefaultUpstreamRequestQueue();
|
|
7169
7983
|
function configureUpstreamRequestQueue(options) {
|
|
@@ -7179,8 +7993,12 @@ function getClientConfig() {
|
|
|
7179
7993
|
githubApiBaseUrl: apiBaseUrl
|
|
7180
7994
|
};
|
|
7181
7995
|
}
|
|
7182
|
-
function createCopilotClient() {
|
|
7183
|
-
return new CopilotClient(authStore, getClientConfig(), {
|
|
7996
|
+
function createCopilotClient(recovery, options = {}) {
|
|
7997
|
+
return new CopilotClient(authStore, getClientConfig(), {
|
|
7998
|
+
requestQueue: upstreamRequestQueue,
|
|
7999
|
+
recovery,
|
|
8000
|
+
...options
|
|
8001
|
+
});
|
|
7184
8002
|
}
|
|
7185
8003
|
async function cacheModels(client) {
|
|
7186
8004
|
const models = await (client ?? createCopilotClient()).getModels();
|
|
@@ -7419,7 +8237,7 @@ const checkUsage = defineCommand({
|
|
|
7419
8237
|
});
|
|
7420
8238
|
//#endregion
|
|
7421
8239
|
//#region src/util/version.ts
|
|
7422
|
-
const VERSION = "0.9.
|
|
8240
|
+
const VERSION = "0.9.2";
|
|
7423
8241
|
//#endregion
|
|
7424
8242
|
//#region src/debug.ts
|
|
7425
8243
|
function getRuntimeInfo() {
|
|
@@ -7495,336 +8313,6 @@ const debug = defineCommand({
|
|
|
7495
8313
|
}
|
|
7496
8314
|
});
|
|
7497
8315
|
//#endregion
|
|
7498
|
-
//#region src/lib/tokenizer.ts
|
|
7499
|
-
const ENCODING_MAP = {
|
|
7500
|
-
o200k_base: () => import("./o200k_base-DXNwToXP.mjs"),
|
|
7501
|
-
cl100k_base: () => import("./cl100k_base-ChJqEXhP.mjs"),
|
|
7502
|
-
p50k_base: () => import("./p50k_base-Cab7w92R.mjs"),
|
|
7503
|
-
p50k_edit: () => import("./p50k_edit-DkrRw_em.mjs"),
|
|
7504
|
-
r50k_base: () => import("./r50k_base-1vVxWqTY.mjs")
|
|
7505
|
-
};
|
|
7506
|
-
const encodingCache = /* @__PURE__ */ new Map();
|
|
7507
|
-
const TOKENS_PER_MESSAGE = 3;
|
|
7508
|
-
const TOKENS_PER_NAME = 1;
|
|
7509
|
-
const REPLY_PRIMING_TOKENS = 3;
|
|
7510
|
-
const BASE_CONSTANTS = {
|
|
7511
|
-
propertyInitOverhead: 3,
|
|
7512
|
-
propertyKeyOverhead: 3,
|
|
7513
|
-
enumOverhead: -3,
|
|
7514
|
-
enumItemCost: 3,
|
|
7515
|
-
functionEndOverhead: 12
|
|
7516
|
-
};
|
|
7517
|
-
/**
|
|
7518
|
-
* Calculate tokens for tool calls
|
|
7519
|
-
*/
|
|
7520
|
-
function calculateToolCallsTokens(toolCalls, encoder, constants) {
|
|
7521
|
-
let tokens = 0;
|
|
7522
|
-
for (const toolCall of toolCalls) {
|
|
7523
|
-
tokens += constants.functionInitOverhead;
|
|
7524
|
-
tokens += encoder.encode(JSON.stringify(toolCall)).length;
|
|
7525
|
-
}
|
|
7526
|
-
tokens += constants.functionEndOverhead;
|
|
7527
|
-
return tokens;
|
|
7528
|
-
}
|
|
7529
|
-
/**
|
|
7530
|
-
* Calculate tokens for content parts
|
|
7531
|
-
*/
|
|
7532
|
-
function calculateContentPartsTokens(contentParts, encoder) {
|
|
7533
|
-
let tokens = 0;
|
|
7534
|
-
for (const part of contentParts) if (part.type === "image_url") tokens += encoder.encode(part.image_url.url).length + 85;
|
|
7535
|
-
else if (part.text) tokens += encoder.encode(part.text).length;
|
|
7536
|
-
return tokens;
|
|
7537
|
-
}
|
|
7538
|
-
/**
|
|
7539
|
-
* Calculate tokens for a single message
|
|
7540
|
-
*/
|
|
7541
|
-
function calculateMessageTokens(message, encoder, constants) {
|
|
7542
|
-
let tokens = TOKENS_PER_MESSAGE;
|
|
7543
|
-
for (const [key, value] of Object.entries(message)) {
|
|
7544
|
-
if (typeof value === "string") tokens += encoder.encode(value).length;
|
|
7545
|
-
if (key === "name") tokens += TOKENS_PER_NAME;
|
|
7546
|
-
if (key === "tool_calls") tokens += calculateToolCallsTokens(value, encoder, constants);
|
|
7547
|
-
if (key === "content" && Array.isArray(value)) tokens += calculateContentPartsTokens(value, encoder);
|
|
7548
|
-
}
|
|
7549
|
-
return tokens;
|
|
7550
|
-
}
|
|
7551
|
-
/**
|
|
7552
|
-
* Calculate tokens using custom algorithm
|
|
7553
|
-
*/
|
|
7554
|
-
function calculateTokens(messages, encoder, constants) {
|
|
7555
|
-
if (messages.length === 0) return 0;
|
|
7556
|
-
let numTokens = 0;
|
|
7557
|
-
for (const message of messages) numTokens += calculateMessageTokens(message, encoder, constants);
|
|
7558
|
-
numTokens += REPLY_PRIMING_TOKENS;
|
|
7559
|
-
return numTokens;
|
|
7560
|
-
}
|
|
7561
|
-
/**
|
|
7562
|
-
* Get the corresponding encoder module based on encoding type
|
|
7563
|
-
*/
|
|
7564
|
-
async function getEncoder(encoding) {
|
|
7565
|
-
const cached = encodingCache.get(encoding);
|
|
7566
|
-
if (cached) return cached;
|
|
7567
|
-
const supportedEncoding = encoding;
|
|
7568
|
-
if (!(supportedEncoding in ENCODING_MAP)) {
|
|
7569
|
-
const fallbackModule = await ENCODING_MAP.o200k_base();
|
|
7570
|
-
encodingCache.set(encoding, fallbackModule);
|
|
7571
|
-
return fallbackModule;
|
|
7572
|
-
}
|
|
7573
|
-
const encodingModule = await ENCODING_MAP[supportedEncoding]();
|
|
7574
|
-
encodingCache.set(encoding, encodingModule);
|
|
7575
|
-
return encodingModule;
|
|
7576
|
-
}
|
|
7577
|
-
/**
|
|
7578
|
-
* Get tokenizer type from model information
|
|
7579
|
-
*/
|
|
7580
|
-
function getTokenizerFromModel(model) {
|
|
7581
|
-
return model.capabilities.tokenizer || "o200k_base";
|
|
7582
|
-
}
|
|
7583
|
-
/**
|
|
7584
|
-
* Get model-specific constants for token calculation
|
|
7585
|
-
*/
|
|
7586
|
-
function getModelConstants(model) {
|
|
7587
|
-
const isLegacy = model.id === "gpt-3.5-turbo" || model.id === "gpt-4";
|
|
7588
|
-
return {
|
|
7589
|
-
...BASE_CONSTANTS,
|
|
7590
|
-
functionInitOverhead: isLegacy ? 10 : 7
|
|
7591
|
-
};
|
|
7592
|
-
}
|
|
7593
|
-
/**
|
|
7594
|
-
* Calculate tokens for a single parameter
|
|
7595
|
-
*/
|
|
7596
|
-
function calculateParameterTokens(key, prop, context) {
|
|
7597
|
-
const { encoder, constants } = context;
|
|
7598
|
-
let tokens = constants.propertyKeyOverhead;
|
|
7599
|
-
if (typeof prop !== "object" || prop === null) return tokens;
|
|
7600
|
-
const param = prop;
|
|
7601
|
-
const paramName = key;
|
|
7602
|
-
const paramType = param.type || "string";
|
|
7603
|
-
let paramDesc = param.description || "";
|
|
7604
|
-
if (param.enum && Array.isArray(param.enum)) {
|
|
7605
|
-
tokens += constants.enumOverhead;
|
|
7606
|
-
for (const item of param.enum) {
|
|
7607
|
-
tokens += constants.enumItemCost;
|
|
7608
|
-
tokens += encoder.encode(String(item)).length;
|
|
7609
|
-
}
|
|
7610
|
-
}
|
|
7611
|
-
if (paramDesc.endsWith(".")) paramDesc = paramDesc.slice(0, -1);
|
|
7612
|
-
const line = `${paramName}:${paramType}:${paramDesc}`;
|
|
7613
|
-
tokens += encoder.encode(line).length;
|
|
7614
|
-
const excludedKeys = new Set([
|
|
7615
|
-
"type",
|
|
7616
|
-
"description",
|
|
7617
|
-
"enum"
|
|
7618
|
-
]);
|
|
7619
|
-
for (const propertyName of Object.keys(param)) if (!excludedKeys.has(propertyName)) {
|
|
7620
|
-
const propertyValue = param[propertyName];
|
|
7621
|
-
const propertyText = typeof propertyValue === "string" ? propertyValue : JSON.stringify(propertyValue);
|
|
7622
|
-
tokens += encoder.encode(`${propertyName}:${propertyText}`).length;
|
|
7623
|
-
}
|
|
7624
|
-
return tokens;
|
|
7625
|
-
}
|
|
7626
|
-
/**
|
|
7627
|
-
* Calculate tokens for function parameters
|
|
7628
|
-
*/
|
|
7629
|
-
function calculateParametersTokens(parameters, encoder, constants) {
|
|
7630
|
-
if (!parameters || typeof parameters !== "object") return 0;
|
|
7631
|
-
const params = parameters;
|
|
7632
|
-
let tokens = 0;
|
|
7633
|
-
for (const [key, value] of Object.entries(params)) if (key === "properties") {
|
|
7634
|
-
const properties = value;
|
|
7635
|
-
if (Object.keys(properties).length > 0) {
|
|
7636
|
-
tokens += constants.propertyInitOverhead;
|
|
7637
|
-
for (const propKey of Object.keys(properties)) tokens += calculateParameterTokens(propKey, properties[propKey], {
|
|
7638
|
-
encoder,
|
|
7639
|
-
constants
|
|
7640
|
-
});
|
|
7641
|
-
}
|
|
7642
|
-
} else {
|
|
7643
|
-
const paramText = typeof value === "string" ? value : JSON.stringify(value);
|
|
7644
|
-
tokens += encoder.encode(`${key}:${paramText}`).length;
|
|
7645
|
-
}
|
|
7646
|
-
return tokens;
|
|
7647
|
-
}
|
|
7648
|
-
/**
|
|
7649
|
-
* Calculate tokens for a single tool
|
|
7650
|
-
*/
|
|
7651
|
-
function calculateToolTokens(tool, encoder, constants) {
|
|
7652
|
-
let tokens = constants.functionInitOverhead;
|
|
7653
|
-
const func = tool.function;
|
|
7654
|
-
const functionName = func.name;
|
|
7655
|
-
let functionDescription = func.description || "";
|
|
7656
|
-
if (functionDescription.endsWith(".")) functionDescription = functionDescription.slice(0, -1);
|
|
7657
|
-
const line = `${functionName}:${functionDescription}`;
|
|
7658
|
-
tokens += encoder.encode(line).length;
|
|
7659
|
-
if (typeof func.parameters === "object" && func.parameters !== null) tokens += calculateParametersTokens(func.parameters, encoder, constants);
|
|
7660
|
-
return tokens;
|
|
7661
|
-
}
|
|
7662
|
-
/**
|
|
7663
|
-
* Calculate token count for tools based on model
|
|
7664
|
-
*/
|
|
7665
|
-
function numTokensForTools(tools, encoder, constants) {
|
|
7666
|
-
let toolTokenCount = 0;
|
|
7667
|
-
for (const tool of tools) toolTokenCount += calculateToolTokens(tool, encoder, constants);
|
|
7668
|
-
toolTokenCount += constants.functionEndOverhead;
|
|
7669
|
-
return toolTokenCount;
|
|
7670
|
-
}
|
|
7671
|
-
/**
|
|
7672
|
-
* Calculate the token count of messages, supporting multiple GPT encoders
|
|
7673
|
-
*/
|
|
7674
|
-
async function getTokenCount(payload, model) {
|
|
7675
|
-
const encoder = await getEncoder(getTokenizerFromModel(model));
|
|
7676
|
-
const inputMessages = payload.messages.filter((msg) => msg.role !== "assistant");
|
|
7677
|
-
const outputMessages = payload.messages.filter((msg) => msg.role === "assistant");
|
|
7678
|
-
const constants = getModelConstants(model);
|
|
7679
|
-
let inputTokens = calculateTokens(inputMessages, encoder, constants);
|
|
7680
|
-
if (payload.tools && payload.tools.length > 0) inputTokens += numTokensForTools(payload.tools, encoder, constants);
|
|
7681
|
-
const outputTokens = calculateTokens(outputMessages, encoder, constants);
|
|
7682
|
-
return {
|
|
7683
|
-
input: inputTokens,
|
|
7684
|
-
output: outputTokens
|
|
7685
|
-
};
|
|
7686
|
-
}
|
|
7687
|
-
async function estimateResponsesInputTokens(inputItems, model) {
|
|
7688
|
-
return (await getEncoder(getTokenizerFromModel(model))).encode(JSON.stringify(inputItems)).length;
|
|
7689
|
-
}
|
|
7690
|
-
//#endregion
|
|
7691
|
-
//#region src/selfcheck.ts
|
|
7692
|
-
const PROBE_ENCODINGS = [
|
|
7693
|
-
"o200k_base",
|
|
7694
|
-
"cl100k_base",
|
|
7695
|
-
"p50k_base",
|
|
7696
|
-
"p50k_edit",
|
|
7697
|
-
"r50k_base"
|
|
7698
|
-
];
|
|
7699
|
-
const PROBE_MESSAGE = "ghc-proxy selfcheck: probe text for tokenizer chunk load";
|
|
7700
|
-
async function probeEncoding(encoding) {
|
|
7701
|
-
try {
|
|
7702
|
-
const count = await getTokenCount({ messages: [{
|
|
7703
|
-
role: "user",
|
|
7704
|
-
content: PROBE_MESSAGE
|
|
7705
|
-
}] }, {
|
|
7706
|
-
id: `selfcheck-${encoding}`,
|
|
7707
|
-
capabilities: { tokenizer: encoding }
|
|
7708
|
-
});
|
|
7709
|
-
if (count.input <= 0) throw new Error(`encoder for ${encoding} returned 0 tokens for non-empty input`);
|
|
7710
|
-
return {
|
|
7711
|
-
encoding,
|
|
7712
|
-
ok: true,
|
|
7713
|
-
tokenCount: count.input
|
|
7714
|
-
};
|
|
7715
|
-
} catch (error) {
|
|
7716
|
-
return {
|
|
7717
|
-
encoding,
|
|
7718
|
-
ok: false,
|
|
7719
|
-
error: error instanceof Error ? error.message : String(error)
|
|
7720
|
-
};
|
|
7721
|
-
}
|
|
7722
|
-
}
|
|
7723
|
-
async function runSelfCheck(options) {
|
|
7724
|
-
const probes = await Promise.all(PROBE_ENCODINGS.map(probeEncoding));
|
|
7725
|
-
const failed = probes.filter((p) => !p.ok);
|
|
7726
|
-
const result = {
|
|
7727
|
-
ok: failed.length === 0,
|
|
7728
|
-
probes,
|
|
7729
|
-
failedCount: failed.length
|
|
7730
|
-
};
|
|
7731
|
-
if (options.json) process$1.stdout.write(`${JSON.stringify(result, null, 2)}\n`);
|
|
7732
|
-
else {
|
|
7733
|
-
process$1.stdout.write(`ghc-proxy selfcheck — tokenizer dynamic-chunk load\n\n`);
|
|
7734
|
-
for (const probe of probes) {
|
|
7735
|
-
const mark = probe.ok ? "ok " : "FAIL";
|
|
7736
|
-
const detail = probe.ok ? `tokens=${probe.tokenCount}` : `error=${probe.error}`;
|
|
7737
|
-
process$1.stdout.write(` [${mark}] ${probe.encoding.padEnd(12)} ${detail}\n`);
|
|
7738
|
-
}
|
|
7739
|
-
process$1.stdout.write(`\n${result.ok ? "PASS" : "FAIL"} — ${probes.length - failed.length}/${probes.length} encodings loaded\n`);
|
|
7740
|
-
}
|
|
7741
|
-
if (!result.ok) process$1.exitCode = 1;
|
|
7742
|
-
}
|
|
7743
|
-
const selfcheck = defineCommand({
|
|
7744
|
-
meta: {
|
|
7745
|
-
name: "selfcheck",
|
|
7746
|
-
description: "Probe the packaged bundle for runtime regressions (loads every gpt-tokenizer dynamic chunk and encodes a probe string)."
|
|
7747
|
-
},
|
|
7748
|
-
args: { json: {
|
|
7749
|
-
type: "boolean",
|
|
7750
|
-
default: false,
|
|
7751
|
-
description: "Output probe results as JSON"
|
|
7752
|
-
} },
|
|
7753
|
-
run({ args }) {
|
|
7754
|
-
return runSelfCheck({ json: args.json });
|
|
7755
|
-
}
|
|
7756
|
-
});
|
|
7757
|
-
//#endregion
|
|
7758
|
-
//#region node_modules/proxy-from-env/index.js
|
|
7759
|
-
var DEFAULT_PORTS = {
|
|
7760
|
-
ftp: 21,
|
|
7761
|
-
gopher: 70,
|
|
7762
|
-
http: 80,
|
|
7763
|
-
https: 443,
|
|
7764
|
-
ws: 80,
|
|
7765
|
-
wss: 443
|
|
7766
|
-
};
|
|
7767
|
-
function parseUrl(urlString) {
|
|
7768
|
-
try {
|
|
7769
|
-
return new URL(urlString);
|
|
7770
|
-
} catch {
|
|
7771
|
-
return null;
|
|
7772
|
-
}
|
|
7773
|
-
}
|
|
7774
|
-
/**
|
|
7775
|
-
* @param {string|object|URL} url - The URL as a string or URL instance, or a
|
|
7776
|
-
* compatible object (such as the result from legacy url.parse).
|
|
7777
|
-
* @return {string} The URL of the proxy that should handle the request to the
|
|
7778
|
-
* given URL. If no proxy is set, this will be an empty string.
|
|
7779
|
-
*/
|
|
7780
|
-
function getProxyForUrl(url) {
|
|
7781
|
-
var parsedUrl = (typeof url === "string" ? parseUrl(url) : url) || {};
|
|
7782
|
-
var proto = parsedUrl.protocol;
|
|
7783
|
-
var hostname = parsedUrl.host;
|
|
7784
|
-
var port = parsedUrl.port;
|
|
7785
|
-
if (typeof hostname !== "string" || !hostname || typeof proto !== "string") return "";
|
|
7786
|
-
proto = proto.split(":", 1)[0];
|
|
7787
|
-
hostname = hostname.replace(/:\d*$/, "");
|
|
7788
|
-
port = parseInt(port) || DEFAULT_PORTS[proto] || 0;
|
|
7789
|
-
if (!shouldProxy(hostname, port)) return "";
|
|
7790
|
-
var proxy = getEnv(proto + "_proxy") || getEnv("all_proxy");
|
|
7791
|
-
if (proxy && proxy.indexOf("://") === -1) proxy = proto + "://" + proxy;
|
|
7792
|
-
return proxy;
|
|
7793
|
-
}
|
|
7794
|
-
/**
|
|
7795
|
-
* Determines whether a given URL should be proxied.
|
|
7796
|
-
*
|
|
7797
|
-
* @param {string} hostname - The host name of the URL.
|
|
7798
|
-
* @param {number} port - The effective port of the URL.
|
|
7799
|
-
* @returns {boolean} Whether the given URL should be proxied.
|
|
7800
|
-
* @private
|
|
7801
|
-
*/
|
|
7802
|
-
function shouldProxy(hostname, port) {
|
|
7803
|
-
var NO_PROXY = getEnv("no_proxy").toLowerCase();
|
|
7804
|
-
if (!NO_PROXY) return true;
|
|
7805
|
-
if (NO_PROXY === "*") return false;
|
|
7806
|
-
return NO_PROXY.split(/[,\s]/).every(function(proxy) {
|
|
7807
|
-
if (!proxy) return true;
|
|
7808
|
-
var parsedProxy = proxy.match(/^(.+):(\d+)$/);
|
|
7809
|
-
var parsedProxyHostname = parsedProxy ? parsedProxy[1] : proxy;
|
|
7810
|
-
var parsedProxyPort = parsedProxy ? parseInt(parsedProxy[2]) : 0;
|
|
7811
|
-
if (parsedProxyPort && parsedProxyPort !== port) return true;
|
|
7812
|
-
if (!/^[.*]/.test(parsedProxyHostname)) return hostname !== parsedProxyHostname;
|
|
7813
|
-
if (parsedProxyHostname.charAt(0) === "*") parsedProxyHostname = parsedProxyHostname.slice(1);
|
|
7814
|
-
return !hostname.endsWith(parsedProxyHostname);
|
|
7815
|
-
});
|
|
7816
|
-
}
|
|
7817
|
-
/**
|
|
7818
|
-
* Get the value for an environment variable.
|
|
7819
|
-
*
|
|
7820
|
-
* @param {string} key - The name of the environment variable.
|
|
7821
|
-
* @return {string} The value of the environment variable.
|
|
7822
|
-
* @private
|
|
7823
|
-
*/
|
|
7824
|
-
function getEnv(key) {
|
|
7825
|
-
return process.env[key.toLowerCase()] || process.env[key.toUpperCase()] || "";
|
|
7826
|
-
}
|
|
7827
|
-
//#endregion
|
|
7828
8316
|
//#region node_modules/undici/lib/core/symbols.js
|
|
7829
8317
|
var require_symbols = /* @__PURE__ */ __commonJSMin$1(((exports, module) => {
|
|
7830
8318
|
module.exports = {
|
|
@@ -29186,7 +29674,7 @@ var require_eventsource = /* @__PURE__ */ __commonJSMin$1(((exports, module) =>
|
|
|
29186
29674
|
};
|
|
29187
29675
|
}));
|
|
29188
29676
|
//#endregion
|
|
29189
|
-
//#region src/
|
|
29677
|
+
//#region src/lib/tokenizer.ts
|
|
29190
29678
|
var import_undici = (/* @__PURE__ */ __commonJSMin$1(((exports, module) => {
|
|
29191
29679
|
const Client = require_client();
|
|
29192
29680
|
const Dispatcher = require_dispatcher();
|
|
@@ -29353,6 +29841,524 @@ var import_undici = (/* @__PURE__ */ __commonJSMin$1(((exports, module) => {
|
|
|
29353
29841
|
}
|
|
29354
29842
|
module.exports.install = install;
|
|
29355
29843
|
})))();
|
|
29844
|
+
const ENCODING_MAP = {
|
|
29845
|
+
o200k_base: () => import("./o200k_base-DXNwToXP.mjs"),
|
|
29846
|
+
cl100k_base: () => import("./cl100k_base-ChJqEXhP.mjs"),
|
|
29847
|
+
p50k_base: () => import("./p50k_base-Cab7w92R.mjs"),
|
|
29848
|
+
p50k_edit: () => import("./p50k_edit-DkrRw_em.mjs"),
|
|
29849
|
+
r50k_base: () => import("./r50k_base-1vVxWqTY.mjs")
|
|
29850
|
+
};
|
|
29851
|
+
const encodingCache = /* @__PURE__ */ new Map();
|
|
29852
|
+
const TOKENS_PER_MESSAGE = 3;
|
|
29853
|
+
const TOKENS_PER_NAME = 1;
|
|
29854
|
+
const REPLY_PRIMING_TOKENS = 3;
|
|
29855
|
+
const BASE_CONSTANTS = {
|
|
29856
|
+
propertyInitOverhead: 3,
|
|
29857
|
+
propertyKeyOverhead: 3,
|
|
29858
|
+
enumOverhead: -3,
|
|
29859
|
+
enumItemCost: 3,
|
|
29860
|
+
functionEndOverhead: 12
|
|
29861
|
+
};
|
|
29862
|
+
/**
|
|
29863
|
+
* Calculate tokens for tool calls
|
|
29864
|
+
*/
|
|
29865
|
+
function calculateToolCallsTokens(toolCalls, encoder, constants) {
|
|
29866
|
+
let tokens = 0;
|
|
29867
|
+
for (const toolCall of toolCalls) {
|
|
29868
|
+
tokens += constants.functionInitOverhead;
|
|
29869
|
+
tokens += encoder.encode(JSON.stringify(toolCall)).length;
|
|
29870
|
+
}
|
|
29871
|
+
tokens += constants.functionEndOverhead;
|
|
29872
|
+
return tokens;
|
|
29873
|
+
}
|
|
29874
|
+
/**
|
|
29875
|
+
* Calculate tokens for content parts
|
|
29876
|
+
*/
|
|
29877
|
+
function calculateContentPartsTokens(contentParts, encoder) {
|
|
29878
|
+
let tokens = 0;
|
|
29879
|
+
for (const part of contentParts) if (part.type === "image_url") tokens += encoder.encode(part.image_url.url).length + 85;
|
|
29880
|
+
else if (part.text) tokens += encoder.encode(part.text).length;
|
|
29881
|
+
return tokens;
|
|
29882
|
+
}
|
|
29883
|
+
/**
|
|
29884
|
+
* Calculate tokens for a single message
|
|
29885
|
+
*/
|
|
29886
|
+
function calculateMessageTokens(message, encoder, constants) {
|
|
29887
|
+
let tokens = TOKENS_PER_MESSAGE;
|
|
29888
|
+
for (const [key, value] of Object.entries(message)) {
|
|
29889
|
+
if (typeof value === "string") tokens += encoder.encode(value).length;
|
|
29890
|
+
if (key === "name") tokens += TOKENS_PER_NAME;
|
|
29891
|
+
if (key === "tool_calls") tokens += calculateToolCallsTokens(value, encoder, constants);
|
|
29892
|
+
if (key === "content" && Array.isArray(value)) tokens += calculateContentPartsTokens(value, encoder);
|
|
29893
|
+
}
|
|
29894
|
+
return tokens;
|
|
29895
|
+
}
|
|
29896
|
+
/**
|
|
29897
|
+
* Calculate tokens using custom algorithm
|
|
29898
|
+
*/
|
|
29899
|
+
function calculateTokens(messages, encoder, constants) {
|
|
29900
|
+
if (messages.length === 0) return 0;
|
|
29901
|
+
let numTokens = 0;
|
|
29902
|
+
for (const message of messages) numTokens += calculateMessageTokens(message, encoder, constants);
|
|
29903
|
+
numTokens += REPLY_PRIMING_TOKENS;
|
|
29904
|
+
return numTokens;
|
|
29905
|
+
}
|
|
29906
|
+
/**
|
|
29907
|
+
* Get the corresponding encoder module based on encoding type
|
|
29908
|
+
*/
|
|
29909
|
+
async function getEncoder(encoding) {
|
|
29910
|
+
const cached = encodingCache.get(encoding);
|
|
29911
|
+
if (cached) return cached;
|
|
29912
|
+
const supportedEncoding = encoding;
|
|
29913
|
+
if (!(supportedEncoding in ENCODING_MAP)) {
|
|
29914
|
+
const fallbackModule = await ENCODING_MAP.o200k_base();
|
|
29915
|
+
encodingCache.set(encoding, fallbackModule);
|
|
29916
|
+
return fallbackModule;
|
|
29917
|
+
}
|
|
29918
|
+
const encodingModule = await ENCODING_MAP[supportedEncoding]();
|
|
29919
|
+
encodingCache.set(encoding, encodingModule);
|
|
29920
|
+
return encodingModule;
|
|
29921
|
+
}
|
|
29922
|
+
/**
|
|
29923
|
+
* Get tokenizer type from model information
|
|
29924
|
+
*/
|
|
29925
|
+
function getTokenizerFromModel(model) {
|
|
29926
|
+
return model.capabilities.tokenizer || "o200k_base";
|
|
29927
|
+
}
|
|
29928
|
+
/**
|
|
29929
|
+
* Get model-specific constants for token calculation
|
|
29930
|
+
*/
|
|
29931
|
+
function getModelConstants(model) {
|
|
29932
|
+
const isLegacy = model.id === "gpt-3.5-turbo" || model.id === "gpt-4";
|
|
29933
|
+
return {
|
|
29934
|
+
...BASE_CONSTANTS,
|
|
29935
|
+
functionInitOverhead: isLegacy ? 10 : 7
|
|
29936
|
+
};
|
|
29937
|
+
}
|
|
29938
|
+
/**
|
|
29939
|
+
* Calculate tokens for a single parameter
|
|
29940
|
+
*/
|
|
29941
|
+
function calculateParameterTokens(key, prop, context) {
|
|
29942
|
+
const { encoder, constants } = context;
|
|
29943
|
+
let tokens = constants.propertyKeyOverhead;
|
|
29944
|
+
if (typeof prop !== "object" || prop === null) return tokens;
|
|
29945
|
+
const param = prop;
|
|
29946
|
+
const paramName = key;
|
|
29947
|
+
const paramType = param.type || "string";
|
|
29948
|
+
let paramDesc = param.description || "";
|
|
29949
|
+
if (param.enum && Array.isArray(param.enum)) {
|
|
29950
|
+
tokens += constants.enumOverhead;
|
|
29951
|
+
for (const item of param.enum) {
|
|
29952
|
+
tokens += constants.enumItemCost;
|
|
29953
|
+
tokens += encoder.encode(String(item)).length;
|
|
29954
|
+
}
|
|
29955
|
+
}
|
|
29956
|
+
if (paramDesc.endsWith(".")) paramDesc = paramDesc.slice(0, -1);
|
|
29957
|
+
const line = `${paramName}:${paramType}:${paramDesc}`;
|
|
29958
|
+
tokens += encoder.encode(line).length;
|
|
29959
|
+
const excludedKeys = new Set([
|
|
29960
|
+
"type",
|
|
29961
|
+
"description",
|
|
29962
|
+
"enum"
|
|
29963
|
+
]);
|
|
29964
|
+
for (const propertyName of Object.keys(param)) if (!excludedKeys.has(propertyName)) {
|
|
29965
|
+
const propertyValue = param[propertyName];
|
|
29966
|
+
const propertyText = typeof propertyValue === "string" ? propertyValue : JSON.stringify(propertyValue);
|
|
29967
|
+
tokens += encoder.encode(`${propertyName}:${propertyText}`).length;
|
|
29968
|
+
}
|
|
29969
|
+
return tokens;
|
|
29970
|
+
}
|
|
29971
|
+
/**
|
|
29972
|
+
* Calculate tokens for function parameters
|
|
29973
|
+
*/
|
|
29974
|
+
function calculateParametersTokens(parameters, encoder, constants) {
|
|
29975
|
+
if (!parameters || typeof parameters !== "object") return 0;
|
|
29976
|
+
const params = parameters;
|
|
29977
|
+
let tokens = 0;
|
|
29978
|
+
for (const [key, value] of Object.entries(params)) if (key === "properties") {
|
|
29979
|
+
const properties = value;
|
|
29980
|
+
if (Object.keys(properties).length > 0) {
|
|
29981
|
+
tokens += constants.propertyInitOverhead;
|
|
29982
|
+
for (const propKey of Object.keys(properties)) tokens += calculateParameterTokens(propKey, properties[propKey], {
|
|
29983
|
+
encoder,
|
|
29984
|
+
constants
|
|
29985
|
+
});
|
|
29986
|
+
}
|
|
29987
|
+
} else {
|
|
29988
|
+
const paramText = typeof value === "string" ? value : JSON.stringify(value);
|
|
29989
|
+
tokens += encoder.encode(`${key}:${paramText}`).length;
|
|
29990
|
+
}
|
|
29991
|
+
return tokens;
|
|
29992
|
+
}
|
|
29993
|
+
/**
|
|
29994
|
+
* Calculate tokens for a single tool
|
|
29995
|
+
*/
|
|
29996
|
+
function calculateToolTokens(tool, encoder, constants) {
|
|
29997
|
+
let tokens = constants.functionInitOverhead;
|
|
29998
|
+
const func = tool.function;
|
|
29999
|
+
const functionName = func.name;
|
|
30000
|
+
let functionDescription = func.description || "";
|
|
30001
|
+
if (functionDescription.endsWith(".")) functionDescription = functionDescription.slice(0, -1);
|
|
30002
|
+
const line = `${functionName}:${functionDescription}`;
|
|
30003
|
+
tokens += encoder.encode(line).length;
|
|
30004
|
+
if (typeof func.parameters === "object" && func.parameters !== null) tokens += calculateParametersTokens(func.parameters, encoder, constants);
|
|
30005
|
+
return tokens;
|
|
30006
|
+
}
|
|
30007
|
+
/**
|
|
30008
|
+
* Calculate token count for tools based on model
|
|
30009
|
+
*/
|
|
30010
|
+
function numTokensForTools(tools, encoder, constants) {
|
|
30011
|
+
let toolTokenCount = 0;
|
|
30012
|
+
for (const tool of tools) toolTokenCount += calculateToolTokens(tool, encoder, constants);
|
|
30013
|
+
toolTokenCount += constants.functionEndOverhead;
|
|
30014
|
+
return toolTokenCount;
|
|
30015
|
+
}
|
|
30016
|
+
/**
|
|
30017
|
+
* Calculate the token count of messages, supporting multiple GPT encoders
|
|
30018
|
+
*/
|
|
30019
|
+
async function getTokenCount(payload, model) {
|
|
30020
|
+
const encoder = await getEncoder(getTokenizerFromModel(model));
|
|
30021
|
+
const inputMessages = payload.messages.filter((msg) => msg.role !== "assistant");
|
|
30022
|
+
const outputMessages = payload.messages.filter((msg) => msg.role === "assistant");
|
|
30023
|
+
const constants = getModelConstants(model);
|
|
30024
|
+
let inputTokens = calculateTokens(inputMessages, encoder, constants);
|
|
30025
|
+
if (payload.tools && payload.tools.length > 0) inputTokens += numTokensForTools(payload.tools, encoder, constants);
|
|
30026
|
+
const outputTokens = calculateTokens(outputMessages, encoder, constants);
|
|
30027
|
+
return {
|
|
30028
|
+
input: inputTokens,
|
|
30029
|
+
output: outputTokens
|
|
30030
|
+
};
|
|
30031
|
+
}
|
|
30032
|
+
async function estimateResponsesInputTokens(inputItems, model) {
|
|
30033
|
+
return (await getEncoder(getTokenizerFromModel(model))).encode(JSON.stringify(inputItems)).length;
|
|
30034
|
+
}
|
|
30035
|
+
//#endregion
|
|
30036
|
+
//#region src/selfcheck.ts
|
|
30037
|
+
const PROBE_ENCODINGS = [
|
|
30038
|
+
"o200k_base",
|
|
30039
|
+
"cl100k_base",
|
|
30040
|
+
"p50k_base",
|
|
30041
|
+
"p50k_edit",
|
|
30042
|
+
"r50k_base"
|
|
30043
|
+
];
|
|
30044
|
+
const PROBE_MESSAGE = "ghc-proxy selfcheck: probe text for tokenizer chunk load";
|
|
30045
|
+
const RUNTIME_PROBES = [
|
|
30046
|
+
["http-error-response-contract", probeHttpErrorResponseContract],
|
|
30047
|
+
["connection-error-classification", probeConnectionErrorClassification],
|
|
30048
|
+
["response-body-cancellation", probeResponseBodyCancellation],
|
|
30049
|
+
["response-commit-boundary", probeResponseCommitBoundary],
|
|
30050
|
+
["caller-cancellation", probeCallerCancellation],
|
|
30051
|
+
["protocol-payload-contract", probeProtocolPayloadContract]
|
|
30052
|
+
];
|
|
30053
|
+
async function probeEncoding(encoding) {
|
|
30054
|
+
try {
|
|
30055
|
+
const count = await getTokenCount({ messages: [{
|
|
30056
|
+
role: "user",
|
|
30057
|
+
content: PROBE_MESSAGE
|
|
30058
|
+
}] }, {
|
|
30059
|
+
id: `selfcheck-${encoding}`,
|
|
30060
|
+
capabilities: { tokenizer: encoding }
|
|
30061
|
+
});
|
|
30062
|
+
if (count.input <= 0) throw new Error(`encoder for ${encoding} returned 0 tokens for non-empty input`);
|
|
30063
|
+
return {
|
|
30064
|
+
encoding,
|
|
30065
|
+
ok: true,
|
|
30066
|
+
tokenCount: count.input
|
|
30067
|
+
};
|
|
30068
|
+
} catch (error) {
|
|
30069
|
+
return {
|
|
30070
|
+
encoding,
|
|
30071
|
+
ok: false,
|
|
30072
|
+
error: error instanceof Error ? error.message : String(error)
|
|
30073
|
+
};
|
|
30074
|
+
}
|
|
30075
|
+
}
|
|
30076
|
+
async function runRuntimeProbe(name, probe) {
|
|
30077
|
+
try {
|
|
30078
|
+
await probe();
|
|
30079
|
+
return {
|
|
30080
|
+
name,
|
|
30081
|
+
ok: true
|
|
30082
|
+
};
|
|
30083
|
+
} catch (error) {
|
|
30084
|
+
return {
|
|
30085
|
+
name,
|
|
30086
|
+
ok: false,
|
|
30087
|
+
error: error instanceof Error ? error.message : String(error)
|
|
30088
|
+
};
|
|
30089
|
+
}
|
|
30090
|
+
}
|
|
30091
|
+
async function probeHttpErrorResponseContract() {
|
|
30092
|
+
const response = new HTTPError(529, { error: {
|
|
30093
|
+
message: "upstream overloaded",
|
|
30094
|
+
type: "overloaded_error"
|
|
30095
|
+
} }, { headers: { "retry-after": "17" } }).toResponse();
|
|
30096
|
+
assertProbe(response.status === 529, `expected status 529, received ${response.status}`);
|
|
30097
|
+
assertProbe(response.headers.get("retry-after") === "17", "Retry-After was not preserved");
|
|
30098
|
+
assertProbe((await response.json()).error?.type === "overloaded_error", "error payload changed during toResponse()");
|
|
30099
|
+
}
|
|
30100
|
+
function probeConnectionErrorClassification() {
|
|
30101
|
+
assertProbe(isRetryableConnectionEstablishmentError({ code: "ConnectionRefused" }) === "connection-refused", "Bun ConnectionRefused was not classified");
|
|
30102
|
+
assertProbe(isRetryableConnectionEstablishmentError(new TypeError("fetch failed", { cause: { code: "ECONNREFUSED" } })) === "connection-refused", "Node ECONNREFUSED was not classified");
|
|
30103
|
+
assertProbe(isRetryableConnectionEstablishmentError(new TypeError("fetch failed", { cause: { code: "ENOTFOUND" } })) === "dns", "Node ENOTFOUND was not classified");
|
|
30104
|
+
assertProbe(isRetryableConnectionEstablishmentError({
|
|
30105
|
+
name: "TimeoutError",
|
|
30106
|
+
code: "ECONNREFUSED"
|
|
30107
|
+
}) === void 0, "timeout-shaped error was classified as a connection-establishment failure");
|
|
30108
|
+
}
|
|
30109
|
+
async function probeResponseBodyCancellation() {
|
|
30110
|
+
const queue = createRuntimeProbeQueue({ maxRetries: 1 });
|
|
30111
|
+
const dispatcher = process$1.versions.bun ? void 0 : new import_undici.Agent({ connections: 1 });
|
|
30112
|
+
const sockets = /* @__PURE__ */ new Set();
|
|
30113
|
+
let requests = 0;
|
|
30114
|
+
let firstResponseClosed = false;
|
|
30115
|
+
const server = createServer((_request, response) => {
|
|
30116
|
+
requests++;
|
|
30117
|
+
if (requests === 1) {
|
|
30118
|
+
response.once("close", () => {
|
|
30119
|
+
firstResponseClosed = true;
|
|
30120
|
+
});
|
|
30121
|
+
response.writeHead(529, { "retry-after": "0" });
|
|
30122
|
+
response.write("retryable response remains open");
|
|
30123
|
+
return;
|
|
30124
|
+
}
|
|
30125
|
+
response.end("ok");
|
|
30126
|
+
});
|
|
30127
|
+
server.on("connection", (socket) => {
|
|
30128
|
+
sockets.add(socket);
|
|
30129
|
+
socket.once("close", () => sockets.delete(socket));
|
|
30130
|
+
});
|
|
30131
|
+
await new Promise((resolve, reject) => {
|
|
30132
|
+
server.once("error", reject);
|
|
30133
|
+
server.listen(0, "127.0.0.1", () => {
|
|
30134
|
+
server.off("error", reject);
|
|
30135
|
+
resolve();
|
|
30136
|
+
});
|
|
30137
|
+
});
|
|
30138
|
+
try {
|
|
30139
|
+
const address = server.address();
|
|
30140
|
+
assertProbe(address !== null && typeof address === "object", "loopback server has no address");
|
|
30141
|
+
const url = `http://127.0.0.1:${address.port}/retry`;
|
|
30142
|
+
const result = await queue.dispatch((signal) => fetch(url, {
|
|
30143
|
+
signal,
|
|
30144
|
+
...dispatcher ? { dispatcher } : {}
|
|
30145
|
+
}), {
|
|
30146
|
+
url,
|
|
30147
|
+
retryable: "capacity"
|
|
30148
|
+
});
|
|
30149
|
+
try {
|
|
30150
|
+
assertProbe(result.response.status === 200, `expected retry status 200, received ${result.response.status}`);
|
|
30151
|
+
assertProbe(await result.response.text() === "ok", "retry response body changed");
|
|
30152
|
+
} finally {
|
|
30153
|
+
result.release();
|
|
30154
|
+
}
|
|
30155
|
+
assertProbe(requests === 2, `expected one retry, observed ${requests - 1}`);
|
|
30156
|
+
if (!process$1.versions.bun) assertProbe(firstResponseClosed, "retryable response did not release its transport");
|
|
30157
|
+
} finally {
|
|
30158
|
+
for (const socket of sockets) socket.destroy();
|
|
30159
|
+
await new Promise((resolve) => server.close(() => resolve()));
|
|
30160
|
+
await dispatcher?.close();
|
|
30161
|
+
}
|
|
30162
|
+
}
|
|
30163
|
+
async function probeResponseCommitBoundary() {
|
|
30164
|
+
const queue = createRuntimeProbeQueue({ maxRetries: 1 });
|
|
30165
|
+
let attempts = 0;
|
|
30166
|
+
const result = await queue.dispatch(async () => {
|
|
30167
|
+
attempts++;
|
|
30168
|
+
return new Response(new ReadableStream({ start(controller) {
|
|
30169
|
+
controller.error(/* @__PURE__ */ new Error("probe stream failure"));
|
|
30170
|
+
} }));
|
|
30171
|
+
}, {
|
|
30172
|
+
url: "https://example.invalid/v1/messages",
|
|
30173
|
+
retryable: "capacity"
|
|
30174
|
+
});
|
|
30175
|
+
let bodyFailed = false;
|
|
30176
|
+
try {
|
|
30177
|
+
await result.response.text();
|
|
30178
|
+
} catch {
|
|
30179
|
+
bodyFailed = true;
|
|
30180
|
+
} finally {
|
|
30181
|
+
result.release();
|
|
30182
|
+
}
|
|
30183
|
+
assertProbe(bodyFailed, "probe stream did not fail during body consumption");
|
|
30184
|
+
assertProbe(attempts === 1, `committed response was replayed ${attempts - 1} time(s)`);
|
|
30185
|
+
}
|
|
30186
|
+
async function probeCallerCancellation() {
|
|
30187
|
+
const queue = createRuntimeProbeQueue();
|
|
30188
|
+
const controller = new AbortController();
|
|
30189
|
+
const reason = /* @__PURE__ */ new Error("selfcheck caller cancellation");
|
|
30190
|
+
let observedSignal;
|
|
30191
|
+
const pending = queue.dispatch(async (signal) => {
|
|
30192
|
+
observedSignal = signal;
|
|
30193
|
+
return new Promise((_resolve, reject) => {
|
|
30194
|
+
if (!signal) {
|
|
30195
|
+
reject(/* @__PURE__ */ new Error("queue did not pass the caller signal to fetch"));
|
|
30196
|
+
return;
|
|
30197
|
+
}
|
|
30198
|
+
signal.addEventListener("abort", () => reject(signal.reason), { once: true });
|
|
30199
|
+
});
|
|
30200
|
+
}, {
|
|
30201
|
+
url: "https://example.invalid/v1/messages",
|
|
30202
|
+
retryable: "capacity"
|
|
30203
|
+
}, controller.signal);
|
|
30204
|
+
await Promise.resolve();
|
|
30205
|
+
await Promise.resolve();
|
|
30206
|
+
controller.abort(reason);
|
|
30207
|
+
let rejection;
|
|
30208
|
+
try {
|
|
30209
|
+
await pending;
|
|
30210
|
+
} catch (error) {
|
|
30211
|
+
rejection = error;
|
|
30212
|
+
}
|
|
30213
|
+
assertProbe(observedSignal === controller.signal, "fetch did not receive the caller signal");
|
|
30214
|
+
assertProbe(rejection === reason, "caller abort reason was not preserved");
|
|
30215
|
+
}
|
|
30216
|
+
async function probeProtocolPayloadContract() {
|
|
30217
|
+
const response = new HTTPError(429, { error: {
|
|
30218
|
+
message: "rate limited",
|
|
30219
|
+
type: "rate_limit_error"
|
|
30220
|
+
} }, { headers: { "retry-after": "5" } }).toResponse();
|
|
30221
|
+
const payload = await response.json();
|
|
30222
|
+
const error = payload.error;
|
|
30223
|
+
assertProbe(Object.keys(payload).join(",") === "error", "public error payload gained a top-level extension");
|
|
30224
|
+
assertProbe(error !== void 0 && Object.keys(error).sort().join(",") === "message,type", "public error object gained a recovery extension");
|
|
30225
|
+
assertProbe([...response.headers.keys()].every((name) => !name.startsWith("x-ghc-")), "public response gained a non-standard recovery header");
|
|
30226
|
+
}
|
|
30227
|
+
function createRuntimeProbeQueue(options = {}) {
|
|
30228
|
+
return new UpstreamRequestQueue({
|
|
30229
|
+
concurrency: 1,
|
|
30230
|
+
maxRetries: options.maxRetries ?? 0,
|
|
30231
|
+
baseDelayMs: 0,
|
|
30232
|
+
maxDelayMs: 1,
|
|
30233
|
+
maxQueueDepth: 1,
|
|
30234
|
+
recoveryBudgetMs: 1e3
|
|
30235
|
+
}, {
|
|
30236
|
+
sleep: async () => {},
|
|
30237
|
+
random: () => 0,
|
|
30238
|
+
logger: {
|
|
30239
|
+
warn() {},
|
|
30240
|
+
info() {}
|
|
30241
|
+
}
|
|
30242
|
+
});
|
|
30243
|
+
}
|
|
30244
|
+
function assertProbe(condition, message) {
|
|
30245
|
+
if (!condition) throw new Error(message);
|
|
30246
|
+
}
|
|
30247
|
+
async function runSelfCheck(options) {
|
|
30248
|
+
const probes = await Promise.all(PROBE_ENCODINGS.map(probeEncoding));
|
|
30249
|
+
const runtimeProbes = await Promise.all(RUNTIME_PROBES.map(([name, probe]) => runRuntimeProbe(name, probe)));
|
|
30250
|
+
const failed = [...probes, ...runtimeProbes].filter((p) => !p.ok);
|
|
30251
|
+
const result = {
|
|
30252
|
+
ok: failed.length === 0,
|
|
30253
|
+
probes,
|
|
30254
|
+
runtimeProbes,
|
|
30255
|
+
failedCount: failed.length
|
|
30256
|
+
};
|
|
30257
|
+
if (options.json) process$1.stdout.write(`${JSON.stringify(result, null, 2)}\n`);
|
|
30258
|
+
else {
|
|
30259
|
+
process$1.stdout.write(`ghc-proxy selfcheck — tokenizer dynamic-chunk load\n\n`);
|
|
30260
|
+
for (const probe of probes) {
|
|
30261
|
+
const mark = probe.ok ? "ok " : "FAIL";
|
|
30262
|
+
const detail = probe.ok ? `tokens=${probe.tokenCount}` : `error=${probe.error}`;
|
|
30263
|
+
process$1.stdout.write(` [${mark}] ${probe.encoding.padEnd(12)} ${detail}\n`);
|
|
30264
|
+
}
|
|
30265
|
+
process$1.stdout.write(`\nghc-proxy runtime probes\n\n`);
|
|
30266
|
+
for (const probe of runtimeProbes) {
|
|
30267
|
+
const mark = probe.ok ? "ok " : "FAIL";
|
|
30268
|
+
const detail = probe.ok ? "" : ` error=${probe.error}`;
|
|
30269
|
+
process$1.stdout.write(` [${mark}] ${probe.name}${detail}\n`);
|
|
30270
|
+
}
|
|
30271
|
+
const passed = probes.length + runtimeProbes.length - failed.length;
|
|
30272
|
+
process$1.stdout.write(`\n${result.ok ? "PASS" : "FAIL"} — ${passed}/${probes.length + runtimeProbes.length} probes passed\n`);
|
|
30273
|
+
}
|
|
30274
|
+
if (!result.ok) process$1.exitCode = 1;
|
|
30275
|
+
}
|
|
30276
|
+
const selfcheck = defineCommand({
|
|
30277
|
+
meta: {
|
|
30278
|
+
name: "selfcheck",
|
|
30279
|
+
description: "Probe the packaged bundle for tokenizer and cross-runtime regressions."
|
|
30280
|
+
},
|
|
30281
|
+
args: { json: {
|
|
30282
|
+
type: "boolean",
|
|
30283
|
+
default: false,
|
|
30284
|
+
description: "Output probe results as JSON"
|
|
30285
|
+
} },
|
|
30286
|
+
run({ args }) {
|
|
30287
|
+
return runSelfCheck({ json: args.json });
|
|
30288
|
+
}
|
|
30289
|
+
});
|
|
30290
|
+
//#endregion
|
|
30291
|
+
//#region node_modules/proxy-from-env/index.js
|
|
30292
|
+
var DEFAULT_PORTS = {
|
|
30293
|
+
ftp: 21,
|
|
30294
|
+
gopher: 70,
|
|
30295
|
+
http: 80,
|
|
30296
|
+
https: 443,
|
|
30297
|
+
ws: 80,
|
|
30298
|
+
wss: 443
|
|
30299
|
+
};
|
|
30300
|
+
function parseUrl(urlString) {
|
|
30301
|
+
try {
|
|
30302
|
+
return new URL(urlString);
|
|
30303
|
+
} catch {
|
|
30304
|
+
return null;
|
|
30305
|
+
}
|
|
30306
|
+
}
|
|
30307
|
+
/**
|
|
30308
|
+
* @param {string|object|URL} url - The URL as a string or URL instance, or a
|
|
30309
|
+
* compatible object (such as the result from legacy url.parse).
|
|
30310
|
+
* @return {string} The URL of the proxy that should handle the request to the
|
|
30311
|
+
* given URL. If no proxy is set, this will be an empty string.
|
|
30312
|
+
*/
|
|
30313
|
+
function getProxyForUrl(url) {
|
|
30314
|
+
var parsedUrl = (typeof url === "string" ? parseUrl(url) : url) || {};
|
|
30315
|
+
var proto = parsedUrl.protocol;
|
|
30316
|
+
var hostname = parsedUrl.host;
|
|
30317
|
+
var port = parsedUrl.port;
|
|
30318
|
+
if (typeof hostname !== "string" || !hostname || typeof proto !== "string") return "";
|
|
30319
|
+
proto = proto.split(":", 1)[0];
|
|
30320
|
+
hostname = hostname.replace(/:\d*$/, "");
|
|
30321
|
+
port = parseInt(port) || DEFAULT_PORTS[proto] || 0;
|
|
30322
|
+
if (!shouldProxy(hostname, port)) return "";
|
|
30323
|
+
var proxy = getEnv(proto + "_proxy") || getEnv("all_proxy");
|
|
30324
|
+
if (proxy && proxy.indexOf("://") === -1) proxy = proto + "://" + proxy;
|
|
30325
|
+
return proxy;
|
|
30326
|
+
}
|
|
30327
|
+
/**
|
|
30328
|
+
* Determines whether a given URL should be proxied.
|
|
30329
|
+
*
|
|
30330
|
+
* @param {string} hostname - The host name of the URL.
|
|
30331
|
+
* @param {number} port - The effective port of the URL.
|
|
30332
|
+
* @returns {boolean} Whether the given URL should be proxied.
|
|
30333
|
+
* @private
|
|
30334
|
+
*/
|
|
30335
|
+
function shouldProxy(hostname, port) {
|
|
30336
|
+
var NO_PROXY = getEnv("no_proxy").toLowerCase();
|
|
30337
|
+
if (!NO_PROXY) return true;
|
|
30338
|
+
if (NO_PROXY === "*") return false;
|
|
30339
|
+
return NO_PROXY.split(/[,\s]/).every(function(proxy) {
|
|
30340
|
+
if (!proxy) return true;
|
|
30341
|
+
var parsedProxy = proxy.match(/^(.+):(\d+)$/);
|
|
30342
|
+
var parsedProxyHostname = parsedProxy ? parsedProxy[1] : proxy;
|
|
30343
|
+
var parsedProxyPort = parsedProxy ? parseInt(parsedProxy[2]) : 0;
|
|
30344
|
+
if (parsedProxyPort && parsedProxyPort !== port) return true;
|
|
30345
|
+
if (!/^[.*]/.test(parsedProxyHostname)) return hostname !== parsedProxyHostname;
|
|
30346
|
+
if (parsedProxyHostname.charAt(0) === "*") parsedProxyHostname = parsedProxyHostname.slice(1);
|
|
30347
|
+
return !hostname.endsWith(parsedProxyHostname);
|
|
30348
|
+
});
|
|
30349
|
+
}
|
|
30350
|
+
/**
|
|
30351
|
+
* Get the value for an environment variable.
|
|
30352
|
+
*
|
|
30353
|
+
* @param {string} key - The name of the environment variable.
|
|
30354
|
+
* @return {string} The value of the environment variable.
|
|
30355
|
+
* @private
|
|
30356
|
+
*/
|
|
30357
|
+
function getEnv(key) {
|
|
30358
|
+
return process.env[key.toLowerCase()] || process.env[key.toUpperCase()] || "";
|
|
30359
|
+
}
|
|
30360
|
+
//#endregion
|
|
30361
|
+
//#region src/cli/proxy.ts
|
|
29356
30362
|
function initProxyFromEnv() {
|
|
29357
30363
|
if (typeof Bun !== "undefined") return;
|
|
29358
30364
|
try {
|
|
@@ -47781,93 +48787,6 @@ var node = () => {
|
|
|
47781
48787
|
};
|
|
47782
48788
|
};
|
|
47783
48789
|
//#endregion
|
|
47784
|
-
//#region src/lib/request-logger.ts
|
|
47785
|
-
/**
|
|
47786
|
-
* Per-request model mapping store.
|
|
47787
|
-
* Route handlers write to this; the after-response hook reads from it.
|
|
47788
|
-
* Uses WeakMap so entries are GC'd when the Request is collected.
|
|
47789
|
-
*/
|
|
47790
|
-
const requestModelMapping = /* @__PURE__ */ new WeakMap();
|
|
47791
|
-
function setRequestModelMapping(request, info) {
|
|
47792
|
-
requestModelMapping.set(request, info);
|
|
47793
|
-
}
|
|
47794
|
-
function getRequestModelMapping(request) {
|
|
47795
|
-
return requestModelMapping.get(request);
|
|
47796
|
-
}
|
|
47797
|
-
function formatElapsed(start) {
|
|
47798
|
-
return formatDurationMs(Date.now() - start);
|
|
47799
|
-
}
|
|
47800
|
-
function formatPath(rawUrl) {
|
|
47801
|
-
try {
|
|
47802
|
-
const url = new URL(rawUrl);
|
|
47803
|
-
return `${url.pathname}${url.search}`;
|
|
47804
|
-
} catch {
|
|
47805
|
-
return rawUrl;
|
|
47806
|
-
}
|
|
47807
|
-
}
|
|
47808
|
-
function colorizeStatus(status) {
|
|
47809
|
-
if (status >= 500) return colorize("red", status);
|
|
47810
|
-
if (status >= 400) return colorize("yellow", status);
|
|
47811
|
-
if (status >= 300) return colorize("cyan", status);
|
|
47812
|
-
return colorize("green", status);
|
|
47813
|
-
}
|
|
47814
|
-
const methodColors = {
|
|
47815
|
-
GET: "cyan",
|
|
47816
|
-
POST: "magenta",
|
|
47817
|
-
PUT: "yellow",
|
|
47818
|
-
PATCH: "yellow",
|
|
47819
|
-
DELETE: "red"
|
|
47820
|
-
};
|
|
47821
|
-
function colorizeMethod(method) {
|
|
47822
|
-
return colorize(methodColors[method] ?? "white", method);
|
|
47823
|
-
}
|
|
47824
|
-
function getEffectiveModel(info) {
|
|
47825
|
-
return info.steps.length > 0 ? info.steps.at(-1).to : info.originalModel ?? "-";
|
|
47826
|
-
}
|
|
47827
|
-
/**
|
|
47828
|
-
* Mutate `modelMapping` in place by appending a transform step.
|
|
47829
|
-
* Strategy contexts hold a reference to the same `modelMapping`,
|
|
47830
|
-
* so steps are pushed directly rather than returning a new object.
|
|
47831
|
-
*/
|
|
47832
|
-
function appendModelStepInPlace(info, tag, newModel) {
|
|
47833
|
-
const current = getEffectiveModel(info);
|
|
47834
|
-
if (newModel !== current) info.steps.push({
|
|
47835
|
-
tag,
|
|
47836
|
-
from: current,
|
|
47837
|
-
to: newModel
|
|
47838
|
-
});
|
|
47839
|
-
}
|
|
47840
|
-
function formatModelMapping(info) {
|
|
47841
|
-
if (!info) return "";
|
|
47842
|
-
const { originalModel, steps } = info;
|
|
47843
|
-
if (!originalModel && steps.length === 0) return "";
|
|
47844
|
-
const parts = [colorize("blueBright", originalModel ?? "-")];
|
|
47845
|
-
for (let i = 0; i < steps.length; i++) {
|
|
47846
|
-
const step = steps[i];
|
|
47847
|
-
const isLast = i === steps.length - 1;
|
|
47848
|
-
parts.push(colorize("dim", `-[${step.tag}]->`));
|
|
47849
|
-
parts.push(colorize(isLast ? "greenBright" : "cyanBright", step.to));
|
|
47850
|
-
}
|
|
47851
|
-
return ` ${colorize("dim", "model=")}${parts.join(" ")}`;
|
|
47852
|
-
}
|
|
47853
|
-
/**
|
|
47854
|
-
* Request logging function.
|
|
47855
|
-
* Logs a formatted request line with method, path, status, elapsed time,
|
|
47856
|
-
* and optional model mapping info.
|
|
47857
|
-
*/
|
|
47858
|
-
function logRequest(method, url, status, elapsed, modelInfo, requestId) {
|
|
47859
|
-
const path = formatPath(url);
|
|
47860
|
-
const line = [
|
|
47861
|
-
colorize("dim", "<-"),
|
|
47862
|
-
colorizeMethod(method),
|
|
47863
|
-
colorize("white", path),
|
|
47864
|
-
colorizeStatus(status),
|
|
47865
|
-
colorize("dim", elapsed)
|
|
47866
|
-
].join(" ");
|
|
47867
|
-
const rid = requestId ? ` ${colorize("dim", `rid=${requestId.slice(0, 8)}`)}` : "";
|
|
47868
|
-
console.log(`${line}${formatModelMapping(modelInfo)}${rid}`);
|
|
47869
|
-
}
|
|
47870
|
-
//#endregion
|
|
47871
48790
|
//#region src/lib/sse-adapter.ts
|
|
47872
48791
|
/**
|
|
47873
48792
|
* Serializes Anthropic stream events into SSE output items
|
|
@@ -49143,7 +50062,7 @@ const responsesToolSchema = union([object({
|
|
|
49143
50062
|
type: literal("function"),
|
|
49144
50063
|
name: string().min(1),
|
|
49145
50064
|
parameters: jsonObjectSchema.nullable().optional(),
|
|
49146
|
-
strict: boolean().optional(),
|
|
50065
|
+
strict: boolean().nullable().optional(),
|
|
49147
50066
|
description: string().nullable().optional()
|
|
49148
50067
|
}).loose(), object({ type: string().min(1) }).catchall(unknown()).superRefine((tool, ctx) => {
|
|
49149
50068
|
if (tool.type === "function") ctx.addIssue({
|
|
@@ -49162,6 +50081,8 @@ const responsesToolChoiceSchema = union([
|
|
|
49162
50081
|
}).loose(),
|
|
49163
50082
|
object({ type: _enum([
|
|
49164
50083
|
"file_search",
|
|
50084
|
+
"web_search",
|
|
50085
|
+
"web_search_2025_08_26",
|
|
49165
50086
|
"web_search_preview",
|
|
49166
50087
|
"web_search_preview_2025_03_11",
|
|
49167
50088
|
"computer_use_preview",
|
|
@@ -49402,6 +50323,10 @@ protocolRegistry.register("embeddings", embeddingsProtocol);
|
|
|
49402
50323
|
//#endregion
|
|
49403
50324
|
//#region src/lib/upstream-signal.ts
|
|
49404
50325
|
const DEFAULT_TIMEOUT_MS = 18e5;
|
|
50326
|
+
function createUpstreamDeadlineFromConfig(now = performance.now()) {
|
|
50327
|
+
const timeoutMs = authStore.upstreamTimeoutSeconds !== void 0 ? authStore.upstreamTimeoutSeconds * 1e3 : DEFAULT_TIMEOUT_MS;
|
|
50328
|
+
return timeoutMs > 0 ? now + timeoutMs : null;
|
|
50329
|
+
}
|
|
49405
50330
|
function createUpstreamSignal(clientSignal, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
49406
50331
|
const controller = new AbortController();
|
|
49407
50332
|
const timeout = timeoutMs > 0 ? setTimeout(() => controller.abort(), timeoutMs) : void 0;
|
|
@@ -49418,9 +50343,30 @@ function createUpstreamSignal(clientSignal, timeoutMs = DEFAULT_TIMEOUT_MS) {
|
|
|
49418
50343
|
}
|
|
49419
50344
|
/**
|
|
49420
50345
|
* Convenience wrapper that reads the upstream timeout from runtime config.
|
|
50346
|
+
*
|
|
50347
|
+
* This signal is a *total-duration* limit. Both runtimes separately apply an
|
|
50348
|
+
* ~300s **idle** timeout to `fetch` — Bun's is built in, Node's is undici's
|
|
50349
|
+
* `headersTimeout` / `bodyTimeout` default of `300e3` — which resets on every
|
|
50350
|
+
* byte received. A response that keeps streaming therefore runs past 300s and
|
|
50351
|
+
* is bounded only by this signal; a stalled one is rejected at ~300s by the
|
|
50352
|
+
* runtime instead. `isTimeoutLikeError` recognizes both runtimes' shapes so
|
|
50353
|
+
* every path maps to a 504.
|
|
49421
50354
|
*/
|
|
49422
|
-
function createUpstreamSignalFromConfig(clientSignal) {
|
|
49423
|
-
|
|
50355
|
+
function createUpstreamSignalFromConfig(clientSignal, deadlineMonotonicMs = createUpstreamDeadlineFromConfig()) {
|
|
50356
|
+
const remainingMs = deadlineMonotonicMs === null ? void 0 : deadlineMonotonicMs - performance.now();
|
|
50357
|
+
return {
|
|
50358
|
+
...remainingMs !== void 0 && remainingMs <= 0 ? createExpiredUpstreamSignal(clientSignal) : createUpstreamSignal(clientSignal, remainingMs ?? 0),
|
|
50359
|
+
deadlineMonotonicMs
|
|
50360
|
+
};
|
|
50361
|
+
}
|
|
50362
|
+
function createExpiredUpstreamSignal(clientSignal) {
|
|
50363
|
+
const controller = new AbortController();
|
|
50364
|
+
controller.abort();
|
|
50365
|
+
return {
|
|
50366
|
+
signal: controller.signal,
|
|
50367
|
+
clientSignal,
|
|
50368
|
+
cleanup: () => {}
|
|
50369
|
+
};
|
|
49424
50370
|
}
|
|
49425
50371
|
//#endregion
|
|
49426
50372
|
//#region src/transform/constants.ts
|
|
@@ -49576,6 +50522,8 @@ function resolveRequestModel({ payload, betaHeaders, applyPolicy }) {
|
|
|
49576
50522
|
//#endregion
|
|
49577
50523
|
//#region src/pipeline/runner.ts
|
|
49578
50524
|
async function runPipeline(params, config) {
|
|
50525
|
+
const upstreamDeadlineMonotonicMs = createUpstreamDeadlineFromConfig();
|
|
50526
|
+
const recovery = createRecoveryRecord(params);
|
|
49579
50527
|
const ingested = protocolRegistry.ingest(config.protocol, params.body, params.headers);
|
|
49580
50528
|
const meta = ingested.meta;
|
|
49581
50529
|
const payload = config.afterIngest ? config.afterIngest({
|
|
@@ -49583,31 +50531,321 @@ async function runPipeline(params, config) {
|
|
|
49583
50531
|
meta,
|
|
49584
50532
|
headers: params.headers
|
|
49585
50533
|
}) : ingested.payload;
|
|
49586
|
-
const
|
|
50534
|
+
const baseSourceModel = resolveBaseModel(payload, meta, config);
|
|
50535
|
+
const fallbackPossible = shouldPreservePristinePayload(config.protocol, baseSourceModel);
|
|
50536
|
+
const pristinePayload = fallbackPossible ? structuredClone(payload) : payload;
|
|
50537
|
+
const sourceAttempt = await prepareAttempt(payload, meta, params, config, recovery, upstreamDeadlineMonotonicMs, { offerLocalModelCooldown: (sourceModel) => fallbackPossible && validateFallback(config.protocol, pristinePayload, sourceModel).ok });
|
|
50538
|
+
try {
|
|
50539
|
+
return {
|
|
50540
|
+
result: await sourceAttempt.execute(),
|
|
50541
|
+
modelMapping: sourceAttempt.modelMapping
|
|
50542
|
+
};
|
|
50543
|
+
} catch (error) {
|
|
50544
|
+
if (!(error instanceof TerminalUpstreamRecoveryError) || error.status !== 529) throw error;
|
|
50545
|
+
const sourceModel = error.recovery.sourceModel;
|
|
50546
|
+
if (!sourceModel) {
|
|
50547
|
+
emitFallbackEvent(error.recovery, void 0, "missing-source-model", 529);
|
|
50548
|
+
throw error;
|
|
50549
|
+
}
|
|
50550
|
+
const candidate = fallbackPossible ? validateFallback(config.protocol, pristinePayload, sourceModel) : {
|
|
50551
|
+
ok: false,
|
|
50552
|
+
reason: "not-configured"
|
|
50553
|
+
};
|
|
50554
|
+
if (!candidate.ok) {
|
|
50555
|
+
emitFallbackEvent(error.recovery, sourceModel, candidate.reason, 529);
|
|
50556
|
+
throw error;
|
|
50557
|
+
}
|
|
50558
|
+
if (sourceAttempt.baseModel !== resolveBaseModel(pristinePayload, meta, config) || getEffectiveModel(sourceAttempt.modelMapping) !== sourceModel) {
|
|
50559
|
+
emitFallbackEvent(error.recovery, sourceModel, "source-resolution-changed", 529);
|
|
50560
|
+
throw error;
|
|
50561
|
+
}
|
|
50562
|
+
params.signal.throwIfAborted();
|
|
50563
|
+
if (!error.claimFallback()) throw error;
|
|
50564
|
+
error.recovery.fallbackFetchStarted = false;
|
|
50565
|
+
const fallbackMapping = {
|
|
50566
|
+
originalModel: sourceAttempt.modelMapping.originalModel,
|
|
50567
|
+
steps: [...sourceAttempt.modelMapping.steps]
|
|
50568
|
+
};
|
|
50569
|
+
appendModelStepInPlace(fallbackMapping, "OVERLOAD_FALLBACK", candidate.target.id);
|
|
50570
|
+
let fallbackAttempt;
|
|
50571
|
+
try {
|
|
50572
|
+
fallbackAttempt = await prepareAttempt(structuredClone(pristinePayload), meta, params, config, error.recovery, upstreamDeadlineMonotonicMs, {
|
|
50573
|
+
target: candidate.target,
|
|
50574
|
+
modelMapping: fallbackMapping,
|
|
50575
|
+
fallbackAttempt: true
|
|
50576
|
+
});
|
|
50577
|
+
} catch {
|
|
50578
|
+
if (params.signal.aborted) throw params.signal.reason;
|
|
50579
|
+
emitFallbackEvent(error.recovery, candidate.target.id, "preflight-rejected", 529);
|
|
50580
|
+
throw error;
|
|
50581
|
+
}
|
|
50582
|
+
error.recovery.retryLimit = error.recovery.retryCount;
|
|
50583
|
+
emitFallbackEvent(error.recovery, candidate.target.id, "selected", 529);
|
|
50584
|
+
try {
|
|
50585
|
+
const result = await fallbackAttempt.execute();
|
|
50586
|
+
emitFallbackEvent(error.recovery, candidate.target.id, "succeeded");
|
|
50587
|
+
return {
|
|
50588
|
+
result: discloseActualModel(result, candidate.target.id),
|
|
50589
|
+
modelMapping: fallbackMapping
|
|
50590
|
+
};
|
|
50591
|
+
} catch (fallbackError) {
|
|
50592
|
+
if (params.signal.aborted) throw params.signal.reason;
|
|
50593
|
+
if (fallbackError instanceof FallbackCooldownError) {
|
|
50594
|
+
emitFallbackEvent(error.recovery, candidate.target.id, "target-cooldown", 529);
|
|
50595
|
+
throw error;
|
|
50596
|
+
}
|
|
50597
|
+
if (!error.recovery.fallbackFetchStarted) {
|
|
50598
|
+
emitFallbackEvent(error.recovery, candidate.target.id, "pre-fetch-failed", 529);
|
|
50599
|
+
throw error;
|
|
50600
|
+
}
|
|
50601
|
+
emitFallbackEvent(error.recovery, candidate.target.id, "target-failed", fallbackError instanceof HTTPError ? fallbackError.status : void 0, isRetryableConnectionEstablishmentError(fallbackError));
|
|
50602
|
+
throw fallbackError;
|
|
50603
|
+
}
|
|
50604
|
+
}
|
|
50605
|
+
}
|
|
50606
|
+
async function prepareAttempt(payload, meta, params, config, recovery, upstreamDeadlineMonotonicMs, options = {}) {
|
|
50607
|
+
const resolved = resolveRequestModel({
|
|
49587
50608
|
payload,
|
|
49588
50609
|
betaHeaders: meta.betaHeaders,
|
|
49589
50610
|
applyPolicy: config.applyModelPolicy
|
|
49590
50611
|
});
|
|
50612
|
+
const baseModel = resolved.model;
|
|
50613
|
+
const selectedModel = options.target ?? resolved.resolvedModel;
|
|
50614
|
+
const modelMapping = options.modelMapping ?? resolved.modelMapping;
|
|
50615
|
+
if (options.target) payload.model = options.target.id;
|
|
49591
50616
|
if (config.afterTransform) await config.afterTransform({
|
|
49592
50617
|
payload,
|
|
49593
50618
|
meta,
|
|
49594
50619
|
headers: params.headers,
|
|
49595
50620
|
selectedModel
|
|
49596
50621
|
});
|
|
49597
|
-
|
|
49598
|
-
const
|
|
49599
|
-
const
|
|
49600
|
-
|
|
49601
|
-
|
|
49602
|
-
headers: params.headers,
|
|
49603
|
-
selectedModel,
|
|
49604
|
-
copilotClient,
|
|
49605
|
-
upstreamSignal,
|
|
49606
|
-
modelMapping
|
|
50622
|
+
params.signal.throwIfAborted();
|
|
50623
|
+
const upstreamSignal = createUpstreamSignalFromConfig(params.signal, upstreamDeadlineMonotonicMs);
|
|
50624
|
+
const copilotClient = createCopilotClient(recovery, {
|
|
50625
|
+
offerLocalModelCooldown: options.offerLocalModelCooldown,
|
|
50626
|
+
fallbackAttempt: options.fallbackAttempt
|
|
49607
50627
|
});
|
|
50628
|
+
try {
|
|
50629
|
+
const ctx = config.buildStrategyContext({
|
|
50630
|
+
payload,
|
|
50631
|
+
meta,
|
|
50632
|
+
headers: params.headers,
|
|
50633
|
+
selectedModel,
|
|
50634
|
+
copilotClient,
|
|
50635
|
+
upstreamSignal,
|
|
50636
|
+
modelMapping,
|
|
50637
|
+
recovery
|
|
50638
|
+
});
|
|
50639
|
+
const entry = config.strategyRegistry.select(selectedModel, ctx);
|
|
50640
|
+
return {
|
|
50641
|
+
baseModel,
|
|
50642
|
+
modelMapping,
|
|
50643
|
+
execute: async () => {
|
|
50644
|
+
try {
|
|
50645
|
+
params.signal.throwIfAborted();
|
|
50646
|
+
return await entry.execute(ctx);
|
|
50647
|
+
} catch (error) {
|
|
50648
|
+
upstreamSignal.cleanup();
|
|
50649
|
+
throw error;
|
|
50650
|
+
}
|
|
50651
|
+
}
|
|
50652
|
+
};
|
|
50653
|
+
} catch (error) {
|
|
50654
|
+
upstreamSignal.cleanup();
|
|
50655
|
+
throw error;
|
|
50656
|
+
}
|
|
50657
|
+
}
|
|
50658
|
+
function resolveBaseModel(pristinePayload, meta, config) {
|
|
50659
|
+
return resolveRequestModel({
|
|
50660
|
+
payload: { ...pristinePayload },
|
|
50661
|
+
betaHeaders: meta.betaHeaders,
|
|
50662
|
+
applyPolicy: config.applyModelPolicy
|
|
50663
|
+
}).model;
|
|
50664
|
+
}
|
|
50665
|
+
function shouldPreservePristinePayload(protocol, baseSourceModel) {
|
|
50666
|
+
if (configStore.getOverloadFallback(baseSourceModel)?.trim()) return true;
|
|
50667
|
+
if (protocol !== "anthropic-messages" || !configStore.hasOverloadFallbacks()) return false;
|
|
50668
|
+
const model = modelCache.findById(baseSourceModel);
|
|
50669
|
+
return !model || !modelCache.supportsEndpoint(model, "/v1/messages") && !modelCache.supportsEndpoint(model, "/responses");
|
|
50670
|
+
}
|
|
50671
|
+
function validateFallback(protocol, payload, sourceModel) {
|
|
50672
|
+
const targetId = configStore.getOverloadFallback(sourceModel)?.trim();
|
|
50673
|
+
if (!targetId) return {
|
|
50674
|
+
ok: false,
|
|
50675
|
+
reason: "not-configured"
|
|
50676
|
+
};
|
|
50677
|
+
if (targetId === sourceModel) return {
|
|
50678
|
+
ok: false,
|
|
50679
|
+
reason: "same-model"
|
|
50680
|
+
};
|
|
50681
|
+
const target = modelCache.findById(targetId);
|
|
50682
|
+
if (!target) return {
|
|
50683
|
+
ok: false,
|
|
50684
|
+
reason: "unknown-target"
|
|
50685
|
+
};
|
|
50686
|
+
if (protocol === "responses" && !modelCache.supportsEndpoint(target, "/responses")) return {
|
|
50687
|
+
ok: false,
|
|
50688
|
+
reason: "unsupported-endpoint"
|
|
50689
|
+
};
|
|
50690
|
+
if (requestsTools(payload) && !modelCache.supportsToolCalls(target)) return {
|
|
50691
|
+
ok: false,
|
|
50692
|
+
reason: "unsupported-tools"
|
|
50693
|
+
};
|
|
50694
|
+
if (requestsParallelToolCalls(protocol, payload) && target.capabilities.supports.parallel_tool_calls !== true) return {
|
|
50695
|
+
ok: false,
|
|
50696
|
+
reason: "unsupported-parallel-tools"
|
|
50697
|
+
};
|
|
50698
|
+
if (requestsStreaming(payload) && target.capabilities.supports.streaming === false) return {
|
|
50699
|
+
ok: false,
|
|
50700
|
+
reason: "unsupported-streaming"
|
|
50701
|
+
};
|
|
50702
|
+
if (requestsVision(payload) && !modelCache.supportsVision(target)) return {
|
|
50703
|
+
ok: false,
|
|
50704
|
+
reason: "unsupported-vision"
|
|
50705
|
+
};
|
|
50706
|
+
if (requestsReasoningEffort(protocol, payload) && !modelCache.supportsReasoningEffort(target)) return {
|
|
50707
|
+
ok: false,
|
|
50708
|
+
reason: "unsupported-reasoning"
|
|
50709
|
+
};
|
|
50710
|
+
if (requestsThinking(protocol, payload) && !modelCache.supportsAdaptiveThinking(target) && !modelCache.supportsReasoningEffort(target)) return {
|
|
50711
|
+
ok: false,
|
|
50712
|
+
reason: "unsupported-thinking"
|
|
50713
|
+
};
|
|
50714
|
+
if (requestsStructuredOutput(protocol, payload) && !supportsStructuredOutput(protocol, target)) return {
|
|
50715
|
+
ok: false,
|
|
50716
|
+
reason: "unsupported-structured-output"
|
|
50717
|
+
};
|
|
50718
|
+
return {
|
|
50719
|
+
ok: true,
|
|
50720
|
+
target
|
|
50721
|
+
};
|
|
50722
|
+
}
|
|
50723
|
+
function requestsTools(payload) {
|
|
50724
|
+
const tools = asRecord(payload)?.tools;
|
|
50725
|
+
return Array.isArray(tools) && tools.length > 0;
|
|
50726
|
+
}
|
|
50727
|
+
function requestsParallelToolCalls(protocol, payload) {
|
|
50728
|
+
if (protocol === "anthropic-messages") return requestsTools(payload);
|
|
50729
|
+
return asRecord(payload)?.parallel_tool_calls === true;
|
|
50730
|
+
}
|
|
50731
|
+
function requestsStreaming(payload) {
|
|
50732
|
+
return asRecord(payload)?.stream === true;
|
|
50733
|
+
}
|
|
50734
|
+
function requestsVision(payload) {
|
|
50735
|
+
return containsVisionPart(asRecord(payload)?.messages) || containsVisionPart(asRecord(payload)?.input);
|
|
50736
|
+
}
|
|
50737
|
+
function containsVisionPart(value) {
|
|
50738
|
+
if (Array.isArray(value)) return value.some(containsVisionPart);
|
|
50739
|
+
const record = asRecord(value);
|
|
50740
|
+
if (!record) return false;
|
|
50741
|
+
if (record.type === "image" || record.type === "image_url" || record.type === "input_image") return true;
|
|
50742
|
+
return containsVisionPart(record.content);
|
|
50743
|
+
}
|
|
50744
|
+
function requestsReasoningEffort(protocol, payload) {
|
|
50745
|
+
const record = asRecord(payload);
|
|
50746
|
+
if (!record) return false;
|
|
50747
|
+
const effort = protocol === "anthropic-messages" ? asRecord(record.output_config)?.effort : protocol === "responses" ? asRecord(record.reasoning)?.effort : record.reasoning_effort;
|
|
50748
|
+
return effort !== void 0 && effort !== "none";
|
|
50749
|
+
}
|
|
50750
|
+
function requestsThinking(protocol, payload) {
|
|
50751
|
+
const record = asRecord(payload);
|
|
50752
|
+
if (!record) return false;
|
|
50753
|
+
if (protocol === "anthropic-messages") {
|
|
50754
|
+
const type = asRecord(record.thinking)?.type;
|
|
50755
|
+
return type === "enabled" || type === "adaptive";
|
|
50756
|
+
}
|
|
50757
|
+
if (protocol === "openai-chat") return typeof record.thinking_budget === "number" && record.thinking_budget > 0;
|
|
50758
|
+
return false;
|
|
50759
|
+
}
|
|
50760
|
+
function requestsStructuredOutput(protocol, payload) {
|
|
50761
|
+
const record = asRecord(payload);
|
|
50762
|
+
if (!record) return false;
|
|
50763
|
+
if (protocol === "anthropic-messages") return asRecord(record.output_config)?.format !== void 0;
|
|
50764
|
+
if (protocol === "responses") {
|
|
50765
|
+
const type = asRecord(asRecord(record.text)?.format)?.type;
|
|
50766
|
+
return type !== void 0 && type !== "text";
|
|
50767
|
+
}
|
|
50768
|
+
const responseFormat = asRecord(record.response_format);
|
|
50769
|
+
return responseFormat?.type !== void 0 && responseFormat.type !== "text";
|
|
50770
|
+
}
|
|
50771
|
+
function supportsStructuredOutput(protocol, model) {
|
|
50772
|
+
if (protocol === "anthropic-messages") return modelCache.supportsStructuredOutputs(model) || modelCache.supportsEndpoint(model, "/responses");
|
|
50773
|
+
return model.capabilities.supports.structured_outputs ?? false;
|
|
50774
|
+
}
|
|
50775
|
+
function asRecord(value) {
|
|
50776
|
+
return value !== null && typeof value === "object" && !Array.isArray(value) ? value : void 0;
|
|
50777
|
+
}
|
|
50778
|
+
function discloseActualModel(result, model) {
|
|
50779
|
+
if (result.kind === "json") return {
|
|
50780
|
+
kind: "json",
|
|
50781
|
+
data: replaceKnownModelIdentity(result.data, model)
|
|
50782
|
+
};
|
|
50783
|
+
return {
|
|
50784
|
+
kind: "stream",
|
|
50785
|
+
generator: discloseStreamModel(result.generator, model)
|
|
50786
|
+
};
|
|
50787
|
+
}
|
|
50788
|
+
async function* discloseStreamModel(generator, model) {
|
|
50789
|
+
for await (const chunk of generator) {
|
|
50790
|
+
if (!chunk.data || chunk.data === "[DONE]") {
|
|
50791
|
+
yield chunk;
|
|
50792
|
+
continue;
|
|
50793
|
+
}
|
|
50794
|
+
try {
|
|
50795
|
+
const parsed = JSON.parse(chunk.data);
|
|
50796
|
+
const replaced = replaceKnownModelIdentity(parsed, model);
|
|
50797
|
+
yield replaced === parsed ? chunk : {
|
|
50798
|
+
...chunk,
|
|
50799
|
+
data: JSON.stringify(replaced)
|
|
50800
|
+
};
|
|
50801
|
+
} catch {
|
|
50802
|
+
yield chunk;
|
|
50803
|
+
}
|
|
50804
|
+
}
|
|
50805
|
+
}
|
|
50806
|
+
function replaceKnownModelIdentity(value, model) {
|
|
50807
|
+
const record = asRecord(value);
|
|
50808
|
+
if (!record) return value;
|
|
50809
|
+
let changed = false;
|
|
50810
|
+
const next = { ...record };
|
|
50811
|
+
if (typeof record.model === "string") {
|
|
50812
|
+
next.model = model;
|
|
50813
|
+
changed = true;
|
|
50814
|
+
}
|
|
50815
|
+
for (const key of ["message", "response"]) {
|
|
50816
|
+
const nested = asRecord(record[key]);
|
|
50817
|
+
if (nested && typeof nested.model === "string") {
|
|
50818
|
+
next[key] = {
|
|
50819
|
+
...nested,
|
|
50820
|
+
model
|
|
50821
|
+
};
|
|
50822
|
+
changed = true;
|
|
50823
|
+
}
|
|
50824
|
+
}
|
|
50825
|
+
return changed ? next : value;
|
|
50826
|
+
}
|
|
50827
|
+
function emitFallbackEvent(recovery, effectiveModel, decision, status, connectionClass) {
|
|
50828
|
+
const now = performance.now();
|
|
50829
|
+
logRecoveryEvent({
|
|
50830
|
+
requestId: recovery.requestId,
|
|
50831
|
+
callerRequestId: recovery.callerRequestId,
|
|
50832
|
+
event: "fallback",
|
|
50833
|
+
retryCount: recovery.retryCount,
|
|
50834
|
+
effectiveModel,
|
|
50835
|
+
status,
|
|
50836
|
+
connectionClass,
|
|
50837
|
+
...recovery.queueMetrics,
|
|
50838
|
+
...recovery.startedAtMonotonicMs !== void 0 ? { elapsedMs: Math.max(0, now - recovery.startedAtMonotonicMs) } : {},
|
|
50839
|
+
...recovery.deadlineMonotonicMs !== void 0 ? { remainingBudgetMs: Math.max(0, recovery.deadlineMonotonicMs - now) } : {},
|
|
50840
|
+
decision
|
|
50841
|
+
});
|
|
50842
|
+
}
|
|
50843
|
+
function createRecoveryRecord(request) {
|
|
49608
50844
|
return {
|
|
49609
|
-
|
|
49610
|
-
|
|
50845
|
+
requestId: request.requestId,
|
|
50846
|
+
...request.callerRequestId ? { callerRequestId: request.callerRequestId } : {},
|
|
50847
|
+
callerSignal: request.signal,
|
|
50848
|
+
retryCount: 0
|
|
49611
50849
|
};
|
|
49612
50850
|
}
|
|
49613
50851
|
//#endregion
|
|
@@ -50465,14 +51703,9 @@ var AnthropicStreamTranslator = class {
|
|
|
50465
51703
|
this.state.messageStartSent = true;
|
|
50466
51704
|
}
|
|
50467
51705
|
getErrorMessage(error) {
|
|
50468
|
-
if (
|
|
51706
|
+
if (isTimeoutLikeError(error)) return "Upstream streaming request timed out. Please retry.";
|
|
50469
51707
|
return "An unexpected error occurred during streaming.";
|
|
50470
51708
|
}
|
|
50471
|
-
isTimeoutError(error) {
|
|
50472
|
-
if (error instanceof DOMException) return error.name === "TimeoutError";
|
|
50473
|
-
if (error instanceof Error) return error.name === "TimeoutError";
|
|
50474
|
-
return false;
|
|
50475
|
-
}
|
|
50476
51709
|
toConversationDeltas(chunk) {
|
|
50477
51710
|
if (chunk.choices.length === 0) return [];
|
|
50478
51711
|
const choice = chunk.choices.toSorted((left, right) => left.index - right.index)[0];
|
|
@@ -51092,11 +52325,13 @@ const chatCompletionsStrategyRegistry = new StrategyRegistry();
|
|
|
51092
52325
|
chatCompletionsStrategyRegistry.register(chatCompletionsEntry$1);
|
|
51093
52326
|
//#endregion
|
|
51094
52327
|
//#region src/routes/chat-completions/handler.ts
|
|
51095
|
-
async function handleCompletionCore({ body, signal, headers }) {
|
|
52328
|
+
async function handleCompletionCore({ body, signal, headers, requestId, callerRequestId }) {
|
|
51096
52329
|
return runPipeline({
|
|
51097
52330
|
body,
|
|
51098
52331
|
signal,
|
|
51099
|
-
headers
|
|
52332
|
+
headers,
|
|
52333
|
+
requestId,
|
|
52334
|
+
callerRequestId
|
|
51100
52335
|
}, {
|
|
51101
52336
|
protocol: "openai-chat",
|
|
51102
52337
|
strategyRegistry: chatCompletionsStrategyRegistry,
|
|
@@ -51138,7 +52373,8 @@ function createCompletionRoutes() {
|
|
|
51138
52373
|
const { result, modelMapping } = await handleCompletionCore({
|
|
51139
52374
|
body,
|
|
51140
52375
|
signal: request.signal,
|
|
51141
|
-
headers: request.headers
|
|
52376
|
+
headers: request.headers,
|
|
52377
|
+
...getOrCreateRequestCorrelation(request)
|
|
51142
52378
|
});
|
|
51143
52379
|
const delivery = deliverResult(request, result, modelMapping);
|
|
51144
52380
|
if (!delivery.streaming) return delivery.data;
|
|
@@ -51363,12 +52599,19 @@ function normalizeSchemaNode(node) {
|
|
|
51363
52599
|
}
|
|
51364
52600
|
normalized[key] = normalizeSchemaNode(value);
|
|
51365
52601
|
}
|
|
51366
|
-
if (node.type === "object" || isRecord$2(normalized.properties)) {
|
|
51367
|
-
normalized.required = isRecord$2(normalized.properties) ? Object.keys(normalized.properties) : [];
|
|
51368
|
-
normalized.additionalProperties = false;
|
|
51369
|
-
}
|
|
51370
52602
|
return normalized;
|
|
51371
52603
|
}
|
|
52604
|
+
/**
|
|
52605
|
+
* Strip JSON Schema / OpenAPI annotations Copilot's function-schema validator
|
|
52606
|
+
* rejects, leaving the structural schema — including the caller's own
|
|
52607
|
+
* `required` array and `additionalProperties` — untouched.
|
|
52608
|
+
*
|
|
52609
|
+
* The annotation stripping is currently inert: probed 2026-08-06
|
|
52610
|
+
* (`scripts/probes/tool-strict.ts`), upstream accepts every annotation in the
|
|
52611
|
+
* list on every `/responses` model with `strict` omitted. It stays anyway — the
|
|
52612
|
+
* list was written against the upstream of 2026-04, and a probe result is a
|
|
52613
|
+
* dated snapshot rather than a permanent fact.
|
|
52614
|
+
*/
|
|
51372
52615
|
function normalizeFunctionParametersSchemaForCopilot(schema) {
|
|
51373
52616
|
if (!schema) return schema;
|
|
51374
52617
|
return normalizeSchemaNode(schema);
|
|
@@ -51684,7 +52927,6 @@ function convertAnthropicTools(tools) {
|
|
|
51684
52927
|
type: "function",
|
|
51685
52928
|
name: tool.name,
|
|
51686
52929
|
parameters: normalizeFunctionParametersSchemaForCopilot(tool.input_schema),
|
|
51687
|
-
strict: false,
|
|
51688
52930
|
...tool.description ? { description: tool.description } : {}
|
|
51689
52931
|
}));
|
|
51690
52932
|
}
|
|
@@ -52559,34 +53801,45 @@ defaultStrategyRegistry.register(responsesApiEntry);
|
|
|
52559
53801
|
defaultStrategyRegistry.register(chatCompletionsEntry);
|
|
52560
53802
|
//#endregion
|
|
52561
53803
|
//#region src/routes/messages/handler.ts
|
|
52562
|
-
async function handleMessagesCore({ body, signal, headers }) {
|
|
53804
|
+
async function handleMessagesCore({ body, signal, headers, requestId, callerRequestId }) {
|
|
52563
53805
|
let anthropicBetaHeader;
|
|
52564
|
-
|
|
52565
|
-
|
|
52566
|
-
|
|
52567
|
-
|
|
52568
|
-
|
|
52569
|
-
|
|
52570
|
-
|
|
52571
|
-
|
|
52572
|
-
|
|
52573
|
-
|
|
52574
|
-
|
|
52575
|
-
|
|
52576
|
-
|
|
52577
|
-
|
|
52578
|
-
|
|
52579
|
-
|
|
52580
|
-
|
|
52581
|
-
|
|
52582
|
-
|
|
52583
|
-
|
|
52584
|
-
|
|
52585
|
-
|
|
52586
|
-
|
|
52587
|
-
|
|
52588
|
-
|
|
52589
|
-
|
|
53806
|
+
try {
|
|
53807
|
+
return await runPipeline({
|
|
53808
|
+
body,
|
|
53809
|
+
signal,
|
|
53810
|
+
headers,
|
|
53811
|
+
requestId,
|
|
53812
|
+
callerRequestId
|
|
53813
|
+
}, {
|
|
53814
|
+
protocol: "anthropic-messages",
|
|
53815
|
+
applyModelPolicy: true,
|
|
53816
|
+
strategyRegistry: defaultStrategyRegistry,
|
|
53817
|
+
afterIngest({ payload, headers: reqHeaders }) {
|
|
53818
|
+
if (consola.level >= 4) consola.debug("Anthropic request payload:", JSON.stringify(payload));
|
|
53819
|
+
anthropicBetaHeader = processAnthropicBetaHeader(reqHeaders.get("anthropic-beta"));
|
|
53820
|
+
return payload;
|
|
53821
|
+
},
|
|
53822
|
+
buildStrategyContext({ payload, meta, headers: reqHeaders, selectedModel, copilotClient, upstreamSignal, modelMapping }) {
|
|
53823
|
+
return {
|
|
53824
|
+
copilotClient,
|
|
53825
|
+
anthropicPayload: payload,
|
|
53826
|
+
anthropicBetaHeader,
|
|
53827
|
+
selectedModel,
|
|
53828
|
+
upstreamSignal,
|
|
53829
|
+
headers: reqHeaders,
|
|
53830
|
+
requestContext: meta.requestContext ?? {},
|
|
53831
|
+
modelMapping
|
|
53832
|
+
};
|
|
53833
|
+
}
|
|
53834
|
+
});
|
|
53835
|
+
} catch (error) {
|
|
53836
|
+
if (!(error instanceof HTTPError)) throw error;
|
|
53837
|
+
const body = "type" in error.body ? error.body : {
|
|
53838
|
+
type: "error",
|
|
53839
|
+
...error.body
|
|
53840
|
+
};
|
|
53841
|
+
throw new HTTPError(error.status, body, { headers: error.headers });
|
|
53842
|
+
}
|
|
52590
53843
|
}
|
|
52591
53844
|
//#endregion
|
|
52592
53845
|
//#region src/routes/messages/route.ts
|
|
@@ -52596,7 +53849,8 @@ function createMessageRoutes() {
|
|
|
52596
53849
|
const { result, modelMapping } = await handleMessagesCore({
|
|
52597
53850
|
body,
|
|
52598
53851
|
signal: request.signal,
|
|
52599
|
-
headers: request.headers
|
|
53852
|
+
headers: request.headers,
|
|
53853
|
+
...getOrCreateRequestCorrelation(request)
|
|
52600
53854
|
});
|
|
52601
53855
|
const delivery = deliverResult(request, result, modelMapping);
|
|
52602
53856
|
if (!delivery.streaming) return delivery.data;
|
|
@@ -53008,14 +54262,16 @@ const HTTP_URL_RE = /^https?:\/\//i;
|
|
|
53008
54262
|
* emulator request prep, tool/input policies, and context management applied
|
|
53009
54263
|
* through the afterIngest / afterTransform lifecycle hooks.
|
|
53010
54264
|
*/
|
|
53011
|
-
async function handleResponsesCore({ body, signal, headers }) {
|
|
54265
|
+
async function handleResponsesCore({ body, signal, headers, requestId, callerRequestId }) {
|
|
53012
54266
|
const emulatorMode = configStore.isEmulatorEnabled();
|
|
53013
54267
|
let originalPayload;
|
|
53014
54268
|
let emulatorPrepared;
|
|
53015
54269
|
return await runPipeline({
|
|
53016
54270
|
body,
|
|
53017
54271
|
signal,
|
|
53018
|
-
headers
|
|
54272
|
+
headers,
|
|
54273
|
+
requestId,
|
|
54274
|
+
callerRequestId
|
|
53019
54275
|
}, {
|
|
53020
54276
|
protocol: "responses",
|
|
53021
54277
|
strategyRegistry: responsesStrategyRegistry,
|
|
@@ -53035,7 +54291,7 @@ async function handleResponsesCore({ body, signal, headers }) {
|
|
|
53035
54291
|
clampResponsesOutputTokens(payload);
|
|
53036
54292
|
clampResponsesReasoningEffort(payload, selectedModel);
|
|
53037
54293
|
},
|
|
53038
|
-
buildStrategyContext({ payload, meta, copilotClient, upstreamSignal }) {
|
|
54294
|
+
buildStrategyContext({ payload, meta, selectedModel, copilotClient, upstreamSignal }) {
|
|
53039
54295
|
const { vision, initiator } = getResponsesRequestOptions(payload);
|
|
53040
54296
|
const prepared = emulatorPrepared;
|
|
53041
54297
|
const requestPayload = originalPayload ?? payload;
|
|
@@ -53046,7 +54302,10 @@ async function handleResponsesCore({ body, signal, headers }) {
|
|
|
53046
54302
|
requestContext: meta.requestContext ?? {},
|
|
53047
54303
|
vision,
|
|
53048
54304
|
initiator,
|
|
53049
|
-
decorateResponse: prepared ? (response) => decorateStoredResponse(
|
|
54305
|
+
decorateResponse: prepared ? (response) => decorateStoredResponse({
|
|
54306
|
+
...response,
|
|
54307
|
+
model: selectedModel?.id ?? response.model
|
|
54308
|
+
}, requestPayload, prepared) : void 0,
|
|
53050
54309
|
onTerminalResponse: prepared ? (terminalResponse) => {
|
|
53051
54310
|
if (!prepared.shouldStore) return;
|
|
53052
54311
|
persistEmulatorResponse(terminalResponse, prepared.effectiveInputItems);
|
|
@@ -53058,16 +54317,16 @@ async function handleResponsesCore({ body, signal, headers }) {
|
|
|
53058
54317
|
function applyResponsesToolTransforms(payload) {
|
|
53059
54318
|
applyFunctionApplyPatch(payload);
|
|
53060
54319
|
applyFunctionToolCompatibilityDefaults(payload);
|
|
53061
|
-
rejectUnsupportedBuiltinTools(payload);
|
|
53062
54320
|
}
|
|
53063
54321
|
function applyFunctionToolCompatibilityDefaults(payload) {
|
|
53064
54322
|
if (!Array.isArray(payload.tools)) return;
|
|
53065
54323
|
payload.tools = payload.tools.map((tool) => {
|
|
53066
54324
|
if (!isResponseFunctionTool(tool)) return tool;
|
|
54325
|
+
const { strict, ...rest } = tool;
|
|
53067
54326
|
return {
|
|
53068
|
-
...
|
|
54327
|
+
...rest,
|
|
53069
54328
|
parameters: normalizeFunctionParametersSchemaForCopilot(tool.parameters),
|
|
53070
|
-
strict
|
|
54329
|
+
...strict != null ? { strict } : {}
|
|
53071
54330
|
};
|
|
53072
54331
|
});
|
|
53073
54332
|
}
|
|
@@ -53094,11 +54353,6 @@ function applyFunctionApplyPatch(payload) {
|
|
|
53094
54353
|
return tool;
|
|
53095
54354
|
});
|
|
53096
54355
|
}
|
|
53097
|
-
function rejectUnsupportedBuiltinTools(payload) {
|
|
53098
|
-
if (payload.tool_choice && typeof payload.tool_choice === "object" && "type" in payload.tool_choice && (payload.tool_choice.type === "web_search_preview" || payload.tool_choice.type === "web_search_preview_2025_03_11")) throwInvalidRequestError("The selected Copilot endpoint does not support the Responses web_search tool.", "tool_choice", "unsupported_tool_web_search");
|
|
53099
|
-
if (!Array.isArray(payload.tools)) return;
|
|
53100
|
-
for (const tool of payload.tools) if (tool.type === "web_search") throwInvalidRequestError("The selected Copilot endpoint does not support the Responses web_search tool.", "tools", "unsupported_tool_web_search");
|
|
53101
|
-
}
|
|
53102
54356
|
function applyResponsesInputPolicies(payload) {
|
|
53103
54357
|
payload.store = false;
|
|
53104
54358
|
stripUnresolvableInputItems(payload);
|
|
@@ -53280,7 +54534,8 @@ function createResponsesRoutes() {
|
|
|
53280
54534
|
const { result, modelMapping } = await handleResponsesCore({
|
|
53281
54535
|
body,
|
|
53282
54536
|
signal: request.signal,
|
|
53283
|
-
headers: request.headers
|
|
54537
|
+
headers: request.headers,
|
|
54538
|
+
...getOrCreateRequestCorrelation(request)
|
|
53284
54539
|
});
|
|
53285
54540
|
const delivery = deliverResult(request, result, modelMapping);
|
|
53286
54541
|
if (!delivery.streaming) return delivery.data;
|
|
@@ -53344,37 +54599,122 @@ function createUsageRoute() {
|
|
|
53344
54599
|
//#endregion
|
|
53345
54600
|
//#region src/server.ts
|
|
53346
54601
|
const isBun = typeof globalThis.Bun !== "undefined";
|
|
53347
|
-
|
|
54602
|
+
/**
|
|
54603
|
+
* Smallest and largest status an error may claim for itself.
|
|
54604
|
+
*
|
|
54605
|
+
* An error that reports a 2xx/3xx — or a nonsense number — is not describing a
|
|
54606
|
+
* failure the client can act on, so it falls through to 500 rather than turning
|
|
54607
|
+
* a thrown exception into an apparent success.
|
|
54608
|
+
*/
|
|
54609
|
+
const MIN_ERROR_STATUS = 400;
|
|
54610
|
+
const MAX_ERROR_STATUS = 599;
|
|
54611
|
+
/**
|
|
54612
|
+
* The status a thrown value claims for itself, when it claims a plausible one.
|
|
54613
|
+
*
|
|
54614
|
+
* Elysia's built-in error classes (`NotFoundError`, `ParseError`,
|
|
54615
|
+
* `ValidationError`, `InternalServerError`) each declare `status: number` as
|
|
54616
|
+
* part of their public class contract, and this repo's own `TranslationFailure`
|
|
54617
|
+
* declares `status: 400 | 502`. Reading the property instead of mapping
|
|
54618
|
+
* `code` covers all of them, and covers whatever Elysia adds next.
|
|
54619
|
+
*
|
|
54620
|
+
* The read is wrapped because this runs inside the error handler: `status` may
|
|
54621
|
+
* be a getter and `error` may be a Proxy, either of which can throw. This
|
|
54622
|
+
* hardens this function only — Elysia itself does `set.status = error.status`
|
|
54623
|
+
* after `onError` returns (`elysia/dist/compose.mjs`), so a hostile getter
|
|
54624
|
+
* still escapes `app.handle()`. Measured on the pre-fix tree as well, so that
|
|
54625
|
+
* escape is Elysia's, not something reading the property here introduced.
|
|
54626
|
+
* `isTimeoutLikeError` guards its own traversal the same way.
|
|
54627
|
+
*/
|
|
54628
|
+
function claimedErrorStatus(error) {
|
|
54629
|
+
if (typeof error !== "object" || error === null) return void 0;
|
|
54630
|
+
let status;
|
|
54631
|
+
try {
|
|
54632
|
+
if (!("status" in error)) return void 0;
|
|
54633
|
+
status = error.status;
|
|
54634
|
+
} catch {
|
|
54635
|
+
return;
|
|
54636
|
+
}
|
|
54637
|
+
return typeof status === "number" && Number.isInteger(status) && status >= MIN_ERROR_STATUS && status <= MAX_ERROR_STATUS ? status : void 0;
|
|
54638
|
+
}
|
|
54639
|
+
/**
|
|
54640
|
+
* Client-facing error `type` for a locally generated failure.
|
|
54641
|
+
*
|
|
54642
|
+
* The proxy's other error paths already classify by meaning
|
|
54643
|
+
* (`invalid_request_error`, `upstream_error`, `rate_limit_error`,
|
|
54644
|
+
* `timeout_error`), and `upstreamErrorType` in `src/lib/error.ts` states the
|
|
54645
|
+
* rule: a non-standard `type` at the proxy boundary breaks client error
|
|
54646
|
+
* handling. Honoring the thrown error's status made this branch reachable at
|
|
54647
|
+
* 404/400/422 rather than only 500, so it classifies too instead of labelling
|
|
54648
|
+
* every one of them `error`.
|
|
54649
|
+
*/
|
|
54650
|
+
function localErrorType(status) {
|
|
54651
|
+
if (status === 404) return "not_found_error";
|
|
54652
|
+
if (status >= 400 && status < 500) return "invalid_request_error";
|
|
54653
|
+
return "error";
|
|
54654
|
+
}
|
|
54655
|
+
/**
|
|
54656
|
+
* Elysia's `NotFoundError` carries the bare string `NOT_FOUND` as its message.
|
|
54657
|
+
* That is an internal token, not something a client should be shown, so an
|
|
54658
|
+
* unmatched route gets a sentence instead.
|
|
54659
|
+
*/
|
|
54660
|
+
const NOT_FOUND_MESSAGE = "Unknown endpoint. Check the request path.";
|
|
54661
|
+
/**
|
|
54662
|
+
* Maps a thrown error to a client response.
|
|
54663
|
+
*
|
|
54664
|
+
* `set.status` is written on every branch because `onError` returns a fresh
|
|
54665
|
+
* `Response` instead of falling through Elysia's normal path — `set.status`
|
|
54666
|
+
* would otherwise still hold whatever it was before the throw, and the access
|
|
54667
|
+
* log in `onAfterResponse` reads it. Without the write-back, a 504 is logged
|
|
54668
|
+
* as a 500.
|
|
54669
|
+
*
|
|
54670
|
+
* Errors that carry their own plausible status keep it. Flattening every
|
|
54671
|
+
* non-HTTPError to 500 turned an unmatched route into `500 NaNs` on an
|
|
54672
|
+
* OpenAI-compatible surface where an unknown path owes the client a 404, and
|
|
54673
|
+
* hid `TranslationFailure`'s 502 behind a generic 500.
|
|
54674
|
+
*
|
|
54675
|
+
* Exported so tests exercise this mapping rather than a copy of it.
|
|
54676
|
+
*/
|
|
54677
|
+
function handleRouteError({ code, error, set }) {
|
|
54678
|
+
if (code === "HTTP") return;
|
|
54679
|
+
if (isTimeoutLikeError(error)) {
|
|
54680
|
+
set.status = 504;
|
|
54681
|
+
return Response.json({ error: {
|
|
54682
|
+
message: "Upstream request timed out before a response was received.",
|
|
54683
|
+
type: "timeout_error"
|
|
54684
|
+
} }, { status: 504 });
|
|
54685
|
+
}
|
|
54686
|
+
const status = claimedErrorStatus(error) ?? 500;
|
|
54687
|
+
const rawMessage = error instanceof Error ? error.message : String(error);
|
|
54688
|
+
const message = code === "NOT_FOUND" ? NOT_FOUND_MESSAGE : rawMessage;
|
|
54689
|
+
set.status = status;
|
|
54690
|
+
return Response.json({ error: {
|
|
54691
|
+
message,
|
|
54692
|
+
type: localErrorType(status)
|
|
54693
|
+
} }, { status });
|
|
54694
|
+
}
|
|
54695
|
+
function createServer$1(options) {
|
|
53348
54696
|
return new Elysia({
|
|
53349
54697
|
adapter: isBun ? void 0 : node(),
|
|
53350
54698
|
serve: options?.idleTimeout !== void 0 ? { idleTimeout: options.idleTimeout } : void 0
|
|
53351
|
-
}).use(cors()).error({ HTTP: HTTPError }).
|
|
53352
|
-
|
|
53353
|
-
|
|
53354
|
-
|
|
54699
|
+
}).use(cors()).error({ HTTP: HTTPError }).onRequest(({ request }) => {
|
|
54700
|
+
markRequestStart(request);
|
|
54701
|
+
}).derive(({ request }) => ({ ...getOrCreateRequestCorrelation(request) })).onBeforeHandle(({ body, request, responseRequestId, set }) => {
|
|
54702
|
+
set.headers["x-request-id"] = responseRequestId;
|
|
53355
54703
|
if (request.method !== "POST") return;
|
|
53356
54704
|
const model = body && typeof body === "object" && "model" in body ? body.model : void 0;
|
|
53357
54705
|
if (typeof model === "string") setRequestModelMapping(request, {
|
|
53358
54706
|
originalModel: model,
|
|
53359
54707
|
steps: []
|
|
53360
54708
|
});
|
|
53361
|
-
}).onAfterResponse(({
|
|
53362
|
-
|
|
53363
|
-
const elapsed = formatElapsed(requestStart);
|
|
54709
|
+
}).onAfterResponse(({ callerRequestId, request, requestId, set }) => {
|
|
54710
|
+
const elapsed = formatElapsed(getRequestStart(request));
|
|
53364
54711
|
const status = typeof set.status === "number" ? set.status : 200;
|
|
53365
|
-
logRequest(request.method, request.url, status, elapsed, getRequestModelMapping(request), requestId);
|
|
53366
|
-
}).onError(({ code, error }) => {
|
|
53367
|
-
|
|
53368
|
-
|
|
53369
|
-
|
|
53370
|
-
|
|
53371
|
-
} }, { status: 504 });
|
|
53372
|
-
const message = error instanceof Error ? error.message : String(error);
|
|
53373
|
-
return Response.json({ error: {
|
|
53374
|
-
message,
|
|
53375
|
-
type: "error"
|
|
53376
|
-
} }, { status: 500 });
|
|
53377
|
-
}).get("/", () => "Server running").get("/health", () => ({
|
|
54712
|
+
logRequest(request.method, request.url, status, elapsed, getRequestModelMapping(request), requestId, callerRequestId);
|
|
54713
|
+
}).onError(({ code, error, set }) => handleRouteError({
|
|
54714
|
+
code,
|
|
54715
|
+
error,
|
|
54716
|
+
set
|
|
54717
|
+
})).get("/", () => "Server running").get("/health", () => ({
|
|
53378
54718
|
status: "ok",
|
|
53379
54719
|
copilotToken: !!authStore.copilotToken,
|
|
53380
54720
|
modelsLoaded: !!modelCache.getModels(),
|
|
@@ -53432,11 +54772,13 @@ async function runServer(options) {
|
|
|
53432
54772
|
const cachedConfig = getCachedConfig();
|
|
53433
54773
|
const upstreamQueueConcurrency = options.upstreamQueueConcurrency ?? cachedConfig.upstreamQueueConcurrency;
|
|
53434
54774
|
const upstreamQueueMaxRetries = options.upstreamQueueMaxRetries ?? cachedConfig.upstreamQueueMaxRetries;
|
|
54775
|
+
const upstreamRecoveryBudgetSeconds = options.upstreamRecoveryBudgetSeconds ?? cachedConfig.upstreamRecoveryBudgetSeconds;
|
|
53435
54776
|
const upstreamQueueBaseDelaySeconds = options.upstreamQueueBaseDelaySeconds ?? cachedConfig.upstreamQueueBaseDelaySeconds;
|
|
53436
54777
|
const upstreamQueueMaxDelaySeconds = options.upstreamQueueMaxDelaySeconds ?? cachedConfig.upstreamQueueMaxDelaySeconds;
|
|
53437
54778
|
configureUpstreamRequestQueue({
|
|
53438
54779
|
concurrency: upstreamQueueConcurrency,
|
|
53439
54780
|
maxRetries: upstreamQueueMaxRetries,
|
|
54781
|
+
recoveryBudgetMs: secondsToMs(upstreamRecoveryBudgetSeconds),
|
|
53440
54782
|
baseDelayMs: secondsToMs(upstreamQueueBaseDelaySeconds),
|
|
53441
54783
|
maxDelayMs: secondsToMs(upstreamQueueMaxDelaySeconds)
|
|
53442
54784
|
});
|
|
@@ -53451,7 +54793,7 @@ async function runServer(options) {
|
|
|
53451
54793
|
await maybeCopyClaudeCodeCommand(serverUrl);
|
|
53452
54794
|
}
|
|
53453
54795
|
printStartupBanner(serverUrl);
|
|
53454
|
-
const app = createServer({ idleTimeout: options.idleTimeoutSeconds });
|
|
54796
|
+
const app = createServer$1({ idleTimeout: options.idleTimeoutSeconds });
|
|
53455
54797
|
app.listen(options.port);
|
|
53456
54798
|
const shutdown = async () => {
|
|
53457
54799
|
consola.info("Shutting down gracefully...");
|
|
@@ -53471,6 +54813,15 @@ function parseIntArg(raw, name, fallbackMsg) {
|
|
|
53471
54813
|
}
|
|
53472
54814
|
return n;
|
|
53473
54815
|
}
|
|
54816
|
+
function parseBoundedIntArg(raw, name, fallbackMsg, min, max) {
|
|
54817
|
+
if (raw === void 0) return void 0;
|
|
54818
|
+
const value = Number(raw);
|
|
54819
|
+
if (!raw.trim() || !Number.isInteger(value) || value < min || value > max) {
|
|
54820
|
+
consola.warn(`Invalid --${name} value "${raw}". ${fallbackMsg}`);
|
|
54821
|
+
return;
|
|
54822
|
+
}
|
|
54823
|
+
return value;
|
|
54824
|
+
}
|
|
53474
54825
|
function secondsToMs(seconds) {
|
|
53475
54826
|
return seconds === void 0 ? void 0 : seconds * 1e3;
|
|
53476
54827
|
}
|
|
@@ -53547,7 +54898,7 @@ const start = defineCommand({
|
|
|
53547
54898
|
"upstream-timeout": {
|
|
53548
54899
|
type: "string",
|
|
53549
54900
|
default: "1800",
|
|
53550
|
-
description: "Upstream request timeout in seconds (0 to disable)"
|
|
54901
|
+
description: "Upstream request timeout in seconds (0 to disable). Enforced as a total-duration limit; both runtimes additionally apply their own ~300s idle timeout to fetch, which a steadily streaming response does not trip."
|
|
53551
54902
|
},
|
|
53552
54903
|
"upstream-queue-concurrency": {
|
|
53553
54904
|
type: "string",
|
|
@@ -53555,7 +54906,11 @@ const start = defineCommand({
|
|
|
53555
54906
|
},
|
|
53556
54907
|
"upstream-queue-retries": {
|
|
53557
54908
|
type: "string",
|
|
53558
|
-
description: "Maximum retries for transient upstream responses (default:
|
|
54909
|
+
description: "Maximum retries for transient upstream responses (0-2, default: 1)"
|
|
54910
|
+
},
|
|
54911
|
+
"upstream-recovery-budget": {
|
|
54912
|
+
type: "string",
|
|
54913
|
+
description: "Recovery budget in seconds after the first retryable outcome (1-120, default: 60)"
|
|
53559
54914
|
},
|
|
53560
54915
|
"upstream-queue-base-delay": {
|
|
53561
54916
|
type: "string",
|
|
@@ -53582,7 +54937,8 @@ const start = defineCommand({
|
|
|
53582
54937
|
const idleTimeoutSeconds = parseIntArg(args["idle-timeout"], "idle-timeout", "Falling back to Bun default.");
|
|
53583
54938
|
const upstreamTimeoutSeconds = parseIntArg(args["upstream-timeout"], "upstream-timeout", "Falling back to default (300s).");
|
|
53584
54939
|
const upstreamQueueConcurrency = parseIntArg(args["upstream-queue-concurrency"], "upstream-queue-concurrency", "Using default upstream queue concurrency.");
|
|
53585
|
-
const upstreamQueueMaxRetries =
|
|
54940
|
+
const upstreamQueueMaxRetries = parseBoundedIntArg(args["upstream-queue-retries"], "upstream-queue-retries", "Using default upstream queue retry count.", 0, 2);
|
|
54941
|
+
const upstreamRecoveryBudgetSeconds = parseBoundedIntArg(args["upstream-recovery-budget"], "upstream-recovery-budget", "Using default upstream recovery budget.", 1, 120);
|
|
53586
54942
|
const upstreamQueueBaseDelaySeconds = parseIntArg(args["upstream-queue-base-delay"], "upstream-queue-base-delay", "Using default upstream queue base delay.");
|
|
53587
54943
|
const upstreamQueueMaxDelaySeconds = parseIntArg(args["upstream-queue-max-delay"], "upstream-queue-max-delay", "Using default upstream queue max delay.");
|
|
53588
54944
|
return runServer({
|
|
@@ -53600,6 +54956,7 @@ const start = defineCommand({
|
|
|
53600
54956
|
upstreamTimeoutSeconds,
|
|
53601
54957
|
upstreamQueueConcurrency,
|
|
53602
54958
|
upstreamQueueMaxRetries,
|
|
54959
|
+
upstreamRecoveryBudgetSeconds,
|
|
53603
54960
|
upstreamQueueBaseDelaySeconds,
|
|
53604
54961
|
upstreamQueueMaxDelaySeconds,
|
|
53605
54962
|
gheDomain: args["ghe-domain"],
|