@kolisachint/hoocode-ai 0.5.47 → 0.5.49
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -0
- package/dist/index.js.map +1 -1
- package/dist/providers/anthropic.d.ts.map +1 -1
- package/dist/providers/anthropic.js +7 -3
- package/dist/providers/anthropic.js.map +1 -1
- package/dist/providers/azure-openai-responses.d.ts.map +1 -1
- package/dist/providers/azure-openai-responses.js +3 -1
- package/dist/providers/azure-openai-responses.js.map +1 -1
- package/dist/providers/openai-completions.d.ts.map +1 -1
- package/dist/providers/openai-completions.js +52 -11
- package/dist/providers/openai-completions.js.map +1 -1
- package/dist/providers/openai-responses.d.ts.map +1 -1
- package/dist/providers/openai-responses.js +5 -3
- package/dist/providers/openai-responses.js.map +1 -1
- package/dist/providers/param-fallback.d.ts +44 -0
- package/dist/providers/param-fallback.d.ts.map +1 -0
- package/dist/providers/param-fallback.js +90 -0
- package/dist/providers/param-fallback.js.map +1 -0
- package/dist/utils/retry-delay.d.ts +85 -0
- package/dist/utils/retry-delay.d.ts.map +1 -0
- package/dist/utils/retry-delay.js +163 -0
- package/dist/utils/retry-delay.js.map +1 -0
- package/package.json +1 -1
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recovery for endpoints that refuse the optional params we send.
|
|
3
|
+
*
|
|
4
|
+
* Every OpenAI-compatible request carries a few params that are not part of
|
|
5
|
+
* what a chat-completions endpoint has to accept: `prompt_cache_retention` and
|
|
6
|
+
* `prompt_cache_key` buy the long prompt cache, `store` opts out of
|
|
7
|
+
* server-side retention, `stream_options` asks for usage in the stream.
|
|
8
|
+
* OpenAI ignores the ones a model does not use. A gateway that validates its
|
|
9
|
+
* request schema strictly does not — a LiteLLM-style proxy in front of, say,
|
|
10
|
+
* Qwen answers `422 prompt_cache_retention: Extra inputs are not permitted`
|
|
11
|
+
* and the request never reaches the model at all.
|
|
12
|
+
*
|
|
13
|
+
* None of these params change the answer, so a rejection is recoverable: drop
|
|
14
|
+
* the ones the endpoint named and send the request again. Which params an
|
|
15
|
+
* endpoint refused is remembered per base URL, so the cost is one wasted round
|
|
16
|
+
* trip per process rather than one per request, and every later request is
|
|
17
|
+
* built without them from the start — including the summarization behind
|
|
18
|
+
* auto-compaction, which calls the provider directly and so cannot be patched
|
|
19
|
+
* by whatever sanitising the main loop's `onPayload` hook does.
|
|
20
|
+
*/
|
|
21
|
+
/**
|
|
22
|
+
* Params safe to drop when an endpoint rejects them. Deliberately a fixed
|
|
23
|
+
* list: a 4xx naming `messages` or `tools` is a bug in the request we built,
|
|
24
|
+
* not a param to quietly throw away.
|
|
25
|
+
*/
|
|
26
|
+
export declare const DROPPABLE_PARAMS: readonly ["prompt_cache_retention", "prompt_cache_key", "store", "stream_options"];
|
|
27
|
+
export type DroppableParam = (typeof DROPPABLE_PARAMS)[number];
|
|
28
|
+
/** The params `baseUrl` has already refused. Empty for an endpoint that never has. */
|
|
29
|
+
export declare function rejectedParamsFor(baseUrl: string): ReadonlySet<DroppableParam>;
|
|
30
|
+
/** Remember that `baseUrl` rejected these, so the next request omits them. */
|
|
31
|
+
export declare function noteRejectedParams(baseUrl: string, params: Iterable<DroppableParam>): void;
|
|
32
|
+
/** Forget everything learned so far. Tests only — the map is process-lifetime state. */
|
|
33
|
+
export declare function resetRejectedParams(): void;
|
|
34
|
+
/**
|
|
35
|
+
* The droppable params an error blames, if it is the kind of error that blames
|
|
36
|
+
* params at all.
|
|
37
|
+
*
|
|
38
|
+
* Only a 4xx is about the request we sent: a 5xx or a dropped connection says
|
|
39
|
+
* nothing about which params the endpoint accepts, and dropping caching on the
|
|
40
|
+
* strength of one would be a silent, permanent downgrade. Matching is on whole
|
|
41
|
+
* words so `store` does not answer for `datastore`.
|
|
42
|
+
*/
|
|
43
|
+
export declare function droppableParamsNamedBy(error: unknown): DroppableParam[];
|
|
44
|
+
//# sourceMappingURL=param-fallback.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"param-fallback.d.ts","sourceRoot":"","sources":["../../src/providers/param-fallback.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH;;;;GAIG;AACH,eAAO,MAAM,gBAAgB,oFAAqF,CAAC;AAEnH,MAAM,MAAM,cAAc,GAAG,CAAC,OAAO,gBAAgB,CAAC,CAAC,MAAM,CAAC,CAAC;AAO/D,sFAAsF;AACtF,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,MAAM,GAAG,WAAW,CAAC,cAAc,CAAC,CAE9E;AAED,8EAA8E;AAC9E,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,MAAM,EAAE,MAAM,EAAE,QAAQ,CAAC,cAAc,CAAC,GAAG,IAAI,CAS1F;AAED,0FAAwF;AACxF,wBAAgB,mBAAmB,IAAI,IAAI,CAE1C;AAuBD;;;;;;;;GAQG;AACH,wBAAgB,sBAAsB,CAAC,KAAK,EAAE,OAAO,GAAG,cAAc,EAAE,CAMvE","sourcesContent":["/**\n * Recovery for endpoints that refuse the optional params we send.\n *\n * Every OpenAI-compatible request carries a few params that are not part of\n * what a chat-completions endpoint has to accept: `prompt_cache_retention` and\n * `prompt_cache_key` buy the long prompt cache, `store` opts out of\n * server-side retention, `stream_options` asks for usage in the stream.\n * OpenAI ignores the ones a model does not use. A gateway that validates its\n * request schema strictly does not — a LiteLLM-style proxy in front of, say,\n * Qwen answers `422 prompt_cache_retention: Extra inputs are not permitted`\n * and the request never reaches the model at all.\n *\n * None of these params change the answer, so a rejection is recoverable: drop\n * the ones the endpoint named and send the request again. Which params an\n * endpoint refused is remembered per base URL, so the cost is one wasted round\n * trip per process rather than one per request, and every later request is\n * built without them from the start — including the summarization behind\n * auto-compaction, which calls the provider directly and so cannot be patched\n * by whatever sanitising the main loop's `onPayload` hook does.\n */\n\n/**\n * Params safe to drop when an endpoint rejects them. Deliberately a fixed\n * list: a 4xx naming `messages` or `tools` is a bug in the request we built,\n * not a param to quietly throw away.\n */\nexport const DROPPABLE_PARAMS = [\"prompt_cache_retention\", \"prompt_cache_key\", \"store\", \"stream_options\"] as const;\n\nexport type DroppableParam = (typeof DROPPABLE_PARAMS)[number];\n\n/** Params each base URL has been observed rejecting, for this process. */\nconst rejectedByBaseUrl = new Map<string, Set<DroppableParam>>();\n\nconst NONE_REJECTED: ReadonlySet<DroppableParam> = new Set();\n\n/** The params `baseUrl` has already refused. Empty for an endpoint that never has. */\nexport function rejectedParamsFor(baseUrl: string): ReadonlySet<DroppableParam> {\n\treturn rejectedByBaseUrl.get(baseUrl) ?? NONE_REJECTED;\n}\n\n/** Remember that `baseUrl` rejected these, so the next request omits them. */\nexport function noteRejectedParams(baseUrl: string, params: Iterable<DroppableParam>): void {\n\tlet rejected = rejectedByBaseUrl.get(baseUrl);\n\tif (!rejected) {\n\t\trejected = new Set();\n\t\trejectedByBaseUrl.set(baseUrl, rejected);\n\t}\n\tfor (const param of params) {\n\t\trejected.add(param);\n\t}\n}\n\n/** Forget everything learned so far. Tests only — the map is process-lifetime state. */\nexport function resetRejectedParams(): void {\n\trejectedByBaseUrl.clear();\n}\n\n/**\n * The text an error carries about the request, both the SDK's own message\n * (`422 prompt_cache_retention: Extra inputs are not permitted`) and the raw\n * body it parsed, since proxies differ in which one names the field.\n */\nfunction errorText(error: unknown): string {\n\tif (!error || typeof error !== \"object\") return \"\";\n\tconst parts: string[] = [];\n\tconst message = (error as { message?: unknown }).message;\n\tif (typeof message === \"string\") parts.push(message);\n\tconst body = (error as { error?: unknown }).error;\n\tif (body !== undefined) {\n\t\ttry {\n\t\t\tparts.push(JSON.stringify(body));\n\t\t} catch {\n\t\t\t// A body that will not serialize tells us nothing; the message still might.\n\t\t}\n\t}\n\treturn parts.join(\" \");\n}\n\n/**\n * The droppable params an error blames, if it is the kind of error that blames\n * params at all.\n *\n * Only a 4xx is about the request we sent: a 5xx or a dropped connection says\n * nothing about which params the endpoint accepts, and dropping caching on the\n * strength of one would be a silent, permanent downgrade. Matching is on whole\n * words so `store` does not answer for `datastore`.\n */\nexport function droppableParamsNamedBy(error: unknown): DroppableParam[] {\n\tconst status = (error as { status?: unknown } | null | undefined)?.status;\n\tif (typeof status !== \"number\" || status < 400 || status >= 500) return [];\n\tconst text = errorText(error);\n\tif (!text) return [];\n\treturn DROPPABLE_PARAMS.filter((param) => new RegExp(String.raw`\\b${param}\\b`).test(text));\n}\n"]}
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recovery for endpoints that refuse the optional params we send.
|
|
3
|
+
*
|
|
4
|
+
* Every OpenAI-compatible request carries a few params that are not part of
|
|
5
|
+
* what a chat-completions endpoint has to accept: `prompt_cache_retention` and
|
|
6
|
+
* `prompt_cache_key` buy the long prompt cache, `store` opts out of
|
|
7
|
+
* server-side retention, `stream_options` asks for usage in the stream.
|
|
8
|
+
* OpenAI ignores the ones a model does not use. A gateway that validates its
|
|
9
|
+
* request schema strictly does not — a LiteLLM-style proxy in front of, say,
|
|
10
|
+
* Qwen answers `422 prompt_cache_retention: Extra inputs are not permitted`
|
|
11
|
+
* and the request never reaches the model at all.
|
|
12
|
+
*
|
|
13
|
+
* None of these params change the answer, so a rejection is recoverable: drop
|
|
14
|
+
* the ones the endpoint named and send the request again. Which params an
|
|
15
|
+
* endpoint refused is remembered per base URL, so the cost is one wasted round
|
|
16
|
+
* trip per process rather than one per request, and every later request is
|
|
17
|
+
* built without them from the start — including the summarization behind
|
|
18
|
+
* auto-compaction, which calls the provider directly and so cannot be patched
|
|
19
|
+
* by whatever sanitising the main loop's `onPayload` hook does.
|
|
20
|
+
*/
|
|
21
|
+
/**
|
|
22
|
+
* Params safe to drop when an endpoint rejects them. Deliberately a fixed
|
|
23
|
+
* list: a 4xx naming `messages` or `tools` is a bug in the request we built,
|
|
24
|
+
* not a param to quietly throw away.
|
|
25
|
+
*/
|
|
26
|
+
export const DROPPABLE_PARAMS = ["prompt_cache_retention", "prompt_cache_key", "store", "stream_options"];
|
|
27
|
+
/** Params each base URL has been observed rejecting, for this process. */
|
|
28
|
+
const rejectedByBaseUrl = new Map();
|
|
29
|
+
const NONE_REJECTED = new Set();
|
|
30
|
+
/** The params `baseUrl` has already refused. Empty for an endpoint that never has. */
|
|
31
|
+
export function rejectedParamsFor(baseUrl) {
|
|
32
|
+
return rejectedByBaseUrl.get(baseUrl) ?? NONE_REJECTED;
|
|
33
|
+
}
|
|
34
|
+
/** Remember that `baseUrl` rejected these, so the next request omits them. */
|
|
35
|
+
export function noteRejectedParams(baseUrl, params) {
|
|
36
|
+
let rejected = rejectedByBaseUrl.get(baseUrl);
|
|
37
|
+
if (!rejected) {
|
|
38
|
+
rejected = new Set();
|
|
39
|
+
rejectedByBaseUrl.set(baseUrl, rejected);
|
|
40
|
+
}
|
|
41
|
+
for (const param of params) {
|
|
42
|
+
rejected.add(param);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
/** Forget everything learned so far. Tests only — the map is process-lifetime state. */
|
|
46
|
+
export function resetRejectedParams() {
|
|
47
|
+
rejectedByBaseUrl.clear();
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* The text an error carries about the request, both the SDK's own message
|
|
51
|
+
* (`422 prompt_cache_retention: Extra inputs are not permitted`) and the raw
|
|
52
|
+
* body it parsed, since proxies differ in which one names the field.
|
|
53
|
+
*/
|
|
54
|
+
function errorText(error) {
|
|
55
|
+
if (!error || typeof error !== "object")
|
|
56
|
+
return "";
|
|
57
|
+
const parts = [];
|
|
58
|
+
const message = error.message;
|
|
59
|
+
if (typeof message === "string")
|
|
60
|
+
parts.push(message);
|
|
61
|
+
const body = error.error;
|
|
62
|
+
if (body !== undefined) {
|
|
63
|
+
try {
|
|
64
|
+
parts.push(JSON.stringify(body));
|
|
65
|
+
}
|
|
66
|
+
catch {
|
|
67
|
+
// A body that will not serialize tells us nothing; the message still might.
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
return parts.join(" ");
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* The droppable params an error blames, if it is the kind of error that blames
|
|
74
|
+
* params at all.
|
|
75
|
+
*
|
|
76
|
+
* Only a 4xx is about the request we sent: a 5xx or a dropped connection says
|
|
77
|
+
* nothing about which params the endpoint accepts, and dropping caching on the
|
|
78
|
+
* strength of one would be a silent, permanent downgrade. Matching is on whole
|
|
79
|
+
* words so `store` does not answer for `datastore`.
|
|
80
|
+
*/
|
|
81
|
+
export function droppableParamsNamedBy(error) {
|
|
82
|
+
const status = error?.status;
|
|
83
|
+
if (typeof status !== "number" || status < 400 || status >= 500)
|
|
84
|
+
return [];
|
|
85
|
+
const text = errorText(error);
|
|
86
|
+
if (!text)
|
|
87
|
+
return [];
|
|
88
|
+
return DROPPABLE_PARAMS.filter((param) => new RegExp(String.raw `\b${param}\b`).test(text));
|
|
89
|
+
}
|
|
90
|
+
//# sourceMappingURL=param-fallback.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"param-fallback.js","sourceRoot":"","sources":["../../src/providers/param-fallback.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;;GAmBG;AAEH;;;;GAIG;AACH,MAAM,CAAC,MAAM,gBAAgB,GAAG,CAAC,wBAAwB,EAAE,kBAAkB,EAAE,OAAO,EAAE,gBAAgB,CAAU,CAAC;AAInH,0EAA0E;AAC1E,MAAM,iBAAiB,GAAG,IAAI,GAAG,EAA+B,CAAC;AAEjE,MAAM,aAAa,GAAgC,IAAI,GAAG,EAAE,CAAC;AAE7D,sFAAsF;AACtF,MAAM,UAAU,iBAAiB,CAAC,OAAe,EAA+B;IAC/E,OAAO,iBAAiB,CAAC,GAAG,CAAC,OAAO,CAAC,IAAI,aAAa,CAAC;AAAA,CACvD;AAED,8EAA8E;AAC9E,MAAM,UAAU,kBAAkB,CAAC,OAAe,EAAE,MAAgC,EAAQ;IAC3F,IAAI,QAAQ,GAAG,iBAAiB,CAAC,GAAG,CAAC,OAAO,CAAC,CAAC;IAC9C,IAAI,CAAC,QAAQ,EAAE,CAAC;QACf,QAAQ,GAAG,IAAI,GAAG,EAAE,CAAC;QACrB,iBAAiB,CAAC,GAAG,CAAC,OAAO,EAAE,QAAQ,CAAC,CAAC;IAC1C,CAAC;IACD,KAAK,MAAM,KAAK,IAAI,MAAM,EAAE,CAAC;QAC5B,QAAQ,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC;IACrB,CAAC;AAAA,CACD;AAED,0FAAwF;AACxF,MAAM,UAAU,mBAAmB,GAAS;IAC3C,iBAAiB,CAAC,KAAK,EAAE,CAAC;AAAA,CAC1B;AAED;;;;GAIG;AACH,SAAS,SAAS,CAAC,KAAc,EAAU;IAC1C,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,KAAK,QAAQ;QAAE,OAAO,EAAE,CAAC;IACnD,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,MAAM,OAAO,GAAI,KAA+B,CAAC,OAAO,CAAC;IACzD,IAAI,OAAO,OAAO,KAAK,QAAQ;QAAE,KAAK,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;IACrD,MAAM,IAAI,GAAI,KAA6B,CAAC,KAAK,CAAC;IAClD,IAAI,IAAI,KAAK,SAAS,EAAE,CAAC;QACxB,IAAI,CAAC;YACJ,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,SAAS,CAAC,IAAI,CAAC,CAAC,CAAC;QAClC,CAAC;QAAC,MAAM,CAAC;YACR,4EAA4E;QAC7E,CAAC;IACF,CAAC;IACD,OAAO,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AAAA,CACvB;AAED;;;;;;;;GAQG;AACH,MAAM,UAAU,sBAAsB,CAAC,KAAc,EAAoB;IACxE,MAAM,MAAM,GAAI,KAAiD,EAAE,MAAM,CAAC;IAC1E,IAAI,OAAO,MAAM,KAAK,QAAQ,IAAI,MAAM,GAAG,GAAG,IAAI,MAAM,IAAI,GAAG;QAAE,OAAO,EAAE,CAAC;IAC3E,MAAM,IAAI,GAAG,SAAS,CAAC,KAAK,CAAC,CAAC;IAC9B,IAAI,CAAC,IAAI;QAAE,OAAO,EAAE,CAAC;IACrB,OAAO,gBAAgB,CAAC,MAAM,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,IAAI,MAAM,CAAC,MAAM,CAAC,GAAG,CAAA,KAAK,KAAK,IAAI,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,CAAC;AAAA,CAC3F","sourcesContent":["/**\n * Recovery for endpoints that refuse the optional params we send.\n *\n * Every OpenAI-compatible request carries a few params that are not part of\n * what a chat-completions endpoint has to accept: `prompt_cache_retention` and\n * `prompt_cache_key` buy the long prompt cache, `store` opts out of\n * server-side retention, `stream_options` asks for usage in the stream.\n * OpenAI ignores the ones a model does not use. A gateway that validates its\n * request schema strictly does not — a LiteLLM-style proxy in front of, say,\n * Qwen answers `422 prompt_cache_retention: Extra inputs are not permitted`\n * and the request never reaches the model at all.\n *\n * None of these params change the answer, so a rejection is recoverable: drop\n * the ones the endpoint named and send the request again. Which params an\n * endpoint refused is remembered per base URL, so the cost is one wasted round\n * trip per process rather than one per request, and every later request is\n * built without them from the start — including the summarization behind\n * auto-compaction, which calls the provider directly and so cannot be patched\n * by whatever sanitising the main loop's `onPayload` hook does.\n */\n\n/**\n * Params safe to drop when an endpoint rejects them. Deliberately a fixed\n * list: a 4xx naming `messages` or `tools` is a bug in the request we built,\n * not a param to quietly throw away.\n */\nexport const DROPPABLE_PARAMS = [\"prompt_cache_retention\", \"prompt_cache_key\", \"store\", \"stream_options\"] as const;\n\nexport type DroppableParam = (typeof DROPPABLE_PARAMS)[number];\n\n/** Params each base URL has been observed rejecting, for this process. */\nconst rejectedByBaseUrl = new Map<string, Set<DroppableParam>>();\n\nconst NONE_REJECTED: ReadonlySet<DroppableParam> = new Set();\n\n/** The params `baseUrl` has already refused. Empty for an endpoint that never has. */\nexport function rejectedParamsFor(baseUrl: string): ReadonlySet<DroppableParam> {\n\treturn rejectedByBaseUrl.get(baseUrl) ?? NONE_REJECTED;\n}\n\n/** Remember that `baseUrl` rejected these, so the next request omits them. */\nexport function noteRejectedParams(baseUrl: string, params: Iterable<DroppableParam>): void {\n\tlet rejected = rejectedByBaseUrl.get(baseUrl);\n\tif (!rejected) {\n\t\trejected = new Set();\n\t\trejectedByBaseUrl.set(baseUrl, rejected);\n\t}\n\tfor (const param of params) {\n\t\trejected.add(param);\n\t}\n}\n\n/** Forget everything learned so far. Tests only — the map is process-lifetime state. */\nexport function resetRejectedParams(): void {\n\trejectedByBaseUrl.clear();\n}\n\n/**\n * The text an error carries about the request, both the SDK's own message\n * (`422 prompt_cache_retention: Extra inputs are not permitted`) and the raw\n * body it parsed, since proxies differ in which one names the field.\n */\nfunction errorText(error: unknown): string {\n\tif (!error || typeof error !== \"object\") return \"\";\n\tconst parts: string[] = [];\n\tconst message = (error as { message?: unknown }).message;\n\tif (typeof message === \"string\") parts.push(message);\n\tconst body = (error as { error?: unknown }).error;\n\tif (body !== undefined) {\n\t\ttry {\n\t\t\tparts.push(JSON.stringify(body));\n\t\t} catch {\n\t\t\t// A body that will not serialize tells us nothing; the message still might.\n\t\t}\n\t}\n\treturn parts.join(\" \");\n}\n\n/**\n * The droppable params an error blames, if it is the kind of error that blames\n * params at all.\n *\n * Only a 4xx is about the request we sent: a 5xx or a dropped connection says\n * nothing about which params the endpoint accepts, and dropping caching on the\n * strength of one would be a silent, permanent downgrade. Matching is on whole\n * words so `store` does not answer for `datastore`.\n */\nexport function droppableParamsNamedBy(error: unknown): DroppableParam[] {\n\tconst status = (error as { status?: unknown } | null | undefined)?.status;\n\tif (typeof status !== \"number\" || status < 400 || status >= 500) return [];\n\tconst text = errorText(error);\n\tif (!text) return [];\n\treturn DROPPABLE_PARAMS.filter((param) => new RegExp(String.raw`\\b${param}\\b`).test(text));\n}\n"]}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Server-requested retry delays, and the cap that keeps them sane.
|
|
3
|
+
*
|
|
4
|
+
* A provider that has run out of quota answers `429` with a `Retry-After`
|
|
5
|
+
* naming when the quota resets — which for a monthly plan is weeks away, not
|
|
6
|
+
* seconds. The vendor SDKs take that header literally and sleep for it, and a
|
|
7
|
+
* delay that large does not survive the trip: `setTimeout` holds a signed
|
|
8
|
+
* 32-bit millisecond count, so anything past ~24.9 days is silently clamped to
|
|
9
|
+
* 1ms. The "wait 28 days" becomes "retry immediately", the retry draws another
|
|
10
|
+
* 429, and the SDK burns its whole retry budget in a few milliseconds against a
|
|
11
|
+
* quota that will not move for a month.
|
|
12
|
+
*
|
|
13
|
+
* So a requested delay past the cap is not something to wait out — it is a
|
|
14
|
+
* different kind of failure, and the caller needs to see it rather than sit in
|
|
15
|
+
* a hot loop behind it.
|
|
16
|
+
*/
|
|
17
|
+
/**
|
|
18
|
+
* The largest delay a timer can actually hold.
|
|
19
|
+
*
|
|
20
|
+
* `setTimeout` stores its delay in a signed 32-bit integer. Larger values do
|
|
21
|
+
* not throw: Node warns and substitutes 1ms, turning the longest wait into the
|
|
22
|
+
* shortest one.
|
|
23
|
+
*/
|
|
24
|
+
export declare const MAX_TIMER_DELAY_MS: number;
|
|
25
|
+
/** Longest server-requested delay worth waiting out, when the caller names none. */
|
|
26
|
+
export declare const DEFAULT_MAX_RETRY_DELAY_MS = 60000;
|
|
27
|
+
/** Anything with a `get` — a `Headers`, or an SDK error's header bag. */
|
|
28
|
+
interface HeaderBag {
|
|
29
|
+
get(name: string): string | null | undefined;
|
|
30
|
+
}
|
|
31
|
+
/**
|
|
32
|
+
* The delay a response asks for, in milliseconds, or undefined if it asks for
|
|
33
|
+
* none.
|
|
34
|
+
*
|
|
35
|
+
* Mirrors what the OpenAI and Anthropic SDKs do with these headers, because the
|
|
36
|
+
* point is to predict the sleep they are about to take: `retry-after-ms` wins,
|
|
37
|
+
* then `retry-after` as seconds, then `retry-after` as an HTTP date.
|
|
38
|
+
*/
|
|
39
|
+
export declare function parseRetryAfterMs(headers: HeaderBag | undefined): number | undefined;
|
|
40
|
+
/**
|
|
41
|
+
* A duration as a person would say it: the two largest units that carry
|
|
42
|
+
* meaning, because "28d 15h" tells you to go do something else and
|
|
43
|
+
* "2472352s" does not.
|
|
44
|
+
*/
|
|
45
|
+
export declare function formatDelay(ms: number): string;
|
|
46
|
+
/**
|
|
47
|
+
* Wrap `fetch` so a response asking to wait longer than `maxRetryDelayMs` is
|
|
48
|
+
* marked as one the SDK must not retry.
|
|
49
|
+
*
|
|
50
|
+
* `x-should-retry: false` is the escape hatch both the OpenAI and Anthropic
|
|
51
|
+
* clients check before anything else, so stamping it is enough to stop the
|
|
52
|
+
* retry loop before it computes an unholdable sleep. The response is otherwise
|
|
53
|
+
* passed through untouched — same status, same body — so the error the caller
|
|
54
|
+
* finally sees is still the provider's own.
|
|
55
|
+
*
|
|
56
|
+
* A cap of zero (or a nonsensical one) disables the wrapper entirely, which is
|
|
57
|
+
* what the documented "set to 0 to disable" means.
|
|
58
|
+
*/
|
|
59
|
+
export declare function createRetryDelayCapFetch(maxRetryDelayMs: number, baseFetch?: typeof fetch): typeof fetch;
|
|
60
|
+
/**
|
|
61
|
+
* The `fetch` override to spread into an SDK client's options, or nothing when
|
|
62
|
+
* no cap applies.
|
|
63
|
+
*
|
|
64
|
+
* Spread rather than assigned so a disabled cap leaves the key absent
|
|
65
|
+
* altogether: passing `fetch: undefined` explicitly would override the client's
|
|
66
|
+
* own default instead of leaving it alone.
|
|
67
|
+
*/
|
|
68
|
+
export declare function retryDelayCapFetch(maxRetryDelayMs: number | undefined): {
|
|
69
|
+
fetch?: typeof fetch;
|
|
70
|
+
};
|
|
71
|
+
/**
|
|
72
|
+
* The provider's error message, plus how long it wants us gone.
|
|
73
|
+
*
|
|
74
|
+
* A bare "429 quota exceeded" is the same sentence whether the quota returns in
|
|
75
|
+
* thirty seconds or four weeks, and those call for opposite responses from the
|
|
76
|
+
* person reading it. The wait is already on the error; it just was never shown.
|
|
77
|
+
*/
|
|
78
|
+
export declare function describeProviderError(error: unknown, maxRetryDelayMs?: number): string;
|
|
79
|
+
/**
|
|
80
|
+
* Whether an error message describes a wait too long to sit through — the
|
|
81
|
+
* signal that retrying is pointless rather than merely slow.
|
|
82
|
+
*/
|
|
83
|
+
export declare function isLongRetryDelayError(message: string | undefined): boolean;
|
|
84
|
+
export {};
|
|
85
|
+
//# sourceMappingURL=retry-delay.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"retry-delay.d.ts","sourceRoot":"","sources":["../../src/utils/retry-delay.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAEH;;;;;;GAMG;AACH,eAAO,MAAM,kBAAkB,QAAc,CAAC;AAE9C,oFAAoF;AACpF,eAAO,MAAM,0BAA0B,QAAS,CAAC;AAKjD,2EAAyE;AACzE,UAAU,SAAS;IAClB,GAAG,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,GAAG,IAAI,GAAG,SAAS,CAAC;CAC7C;AAED;;;;;;;GAOG;AACH,wBAAgB,iBAAiB,CAAC,OAAO,EAAE,SAAS,GAAG,SAAS,GAAG,MAAM,GAAG,SAAS,CAkBpF;AAED;;;;GAIG;AACH,wBAAgB,WAAW,CAAC,EAAE,EAAE,MAAM,GAAG,MAAM,CAsB9C;AAED;;;;;;;;;;;;GAYG;AACH,wBAAgB,wBAAwB,CAAC,eAAe,EAAE,MAAM,EAAE,SAAS,GAAE,OAAO,KAAa,GAAG,OAAO,KAAK,CAqB/G;AAED;;;;;;;GAOG;AACH,wBAAgB,kBAAkB,CAAC,eAAe,EAAE,MAAM,GAAG,SAAS,GAAG;IAAE,KAAK,CAAC,EAAE,OAAO,KAAK,CAAA;CAAE,CAIhG;AAUD;;;;;;GAMG;AACH,wBAAgB,qBAAqB,CAAC,KAAK,EAAE,OAAO,EAAE,eAAe,SAA6B,GAAG,MAAM,CAO1G;AAED;;;GAGG;AACH,wBAAgB,qBAAqB,CAAC,OAAO,EAAE,MAAM,GAAG,SAAS,GAAG,OAAO,CAE1E","sourcesContent":["/**\n * Server-requested retry delays, and the cap that keeps them sane.\n *\n * A provider that has run out of quota answers `429` with a `Retry-After`\n * naming when the quota resets — which for a monthly plan is weeks away, not\n * seconds. The vendor SDKs take that header literally and sleep for it, and a\n * delay that large does not survive the trip: `setTimeout` holds a signed\n * 32-bit millisecond count, so anything past ~24.9 days is silently clamped to\n * 1ms. The \"wait 28 days\" becomes \"retry immediately\", the retry draws another\n * 429, and the SDK burns its whole retry budget in a few milliseconds against a\n * quota that will not move for a month.\n *\n * So a requested delay past the cap is not something to wait out — it is a\n * different kind of failure, and the caller needs to see it rather than sit in\n * a hot loop behind it.\n */\n\n/**\n * The largest delay a timer can actually hold.\n *\n * `setTimeout` stores its delay in a signed 32-bit integer. Larger values do\n * not throw: Node warns and substitutes 1ms, turning the longest wait into the\n * shortest one.\n */\nexport const MAX_TIMER_DELAY_MS = 2 ** 31 - 1;\n\n/** Longest server-requested delay worth waiting out, when the caller names none. */\nexport const DEFAULT_MAX_RETRY_DELAY_MS = 60_000;\n\n/** Phrase every capped-delay message carries, so callers can recognise one. */\nconst LONG_DELAY_MARKER = \"asked to wait\";\n\n/** Anything with a `get` — a `Headers`, or an SDK error's header bag. */\ninterface HeaderBag {\n\tget(name: string): string | null | undefined;\n}\n\n/**\n * The delay a response asks for, in milliseconds, or undefined if it asks for\n * none.\n *\n * Mirrors what the OpenAI and Anthropic SDKs do with these headers, because the\n * point is to predict the sleep they are about to take: `retry-after-ms` wins,\n * then `retry-after` as seconds, then `retry-after` as an HTTP date.\n */\nexport function parseRetryAfterMs(headers: HeaderBag | undefined): number | undefined {\n\tif (!headers) return undefined;\n\n\tconst afterMs = headers.get(\"retry-after-ms\");\n\tif (afterMs) {\n\t\tconst ms = Number.parseFloat(afterMs);\n\t\tif (!Number.isNaN(ms)) return ms;\n\t}\n\n\tconst after = headers.get(\"retry-after\");\n\tif (!after) return undefined;\n\n\tconst seconds = Number.parseFloat(after);\n\tif (!Number.isNaN(seconds)) return seconds * 1000;\n\n\t// The header's other legal form is an HTTP date.\n\tconst at = Date.parse(after);\n\treturn Number.isNaN(at) ? undefined : at - Date.now();\n}\n\n/**\n * A duration as a person would say it: the two largest units that carry\n * meaning, because \"28d 15h\" tells you to go do something else and\n * \"2472352s\" does not.\n */\nexport function formatDelay(ms: number): string {\n\tif (!Number.isFinite(ms) || ms < 0) return \"an unknown time\";\n\n\tconst totalSeconds = Math.round(ms / 1000);\n\tif (totalSeconds < 60) return `${totalSeconds}s`;\n\n\tconst units: Array<[number, string]> = [\n\t\t[86400, \"d\"],\n\t\t[3600, \"h\"],\n\t\t[60, \"m\"],\n\t\t[1, \"s\"],\n\t];\n\n\tconst parts: string[] = [];\n\tlet remaining = totalSeconds;\n\tfor (const [size, suffix] of units) {\n\t\tconst value = Math.floor(remaining / size);\n\t\tremaining -= value * size;\n\t\tif (value > 0) parts.push(`${value}${suffix}`);\n\t\tif (parts.length === 2) break;\n\t}\n\treturn parts.join(\" \");\n}\n\n/**\n * Wrap `fetch` so a response asking to wait longer than `maxRetryDelayMs` is\n * marked as one the SDK must not retry.\n *\n * `x-should-retry: false` is the escape hatch both the OpenAI and Anthropic\n * clients check before anything else, so stamping it is enough to stop the\n * retry loop before it computes an unholdable sleep. The response is otherwise\n * passed through untouched — same status, same body — so the error the caller\n * finally sees is still the provider's own.\n *\n * A cap of zero (or a nonsensical one) disables the wrapper entirely, which is\n * what the documented \"set to 0 to disable\" means.\n */\nexport function createRetryDelayCapFetch(maxRetryDelayMs: number, baseFetch: typeof fetch = fetch): typeof fetch {\n\tif (!Number.isFinite(maxRetryDelayMs) || maxRetryDelayMs <= 0) return baseFetch;\n\n\treturn async (input: Parameters<typeof fetch>[0], init?: Parameters<typeof fetch>[1]): Promise<Response> => {\n\t\tconst response = await baseFetch(input, init);\n\n\t\t// Only a failure is ever retried, and a server that already stated its\n\t\t// own preference outranks the cap.\n\t\tif (response.ok || response.headers.get(\"x-should-retry\") !== null) return response;\n\n\t\tconst delayMs = parseRetryAfterMs(response.headers);\n\t\tif (delayMs === undefined || delayMs <= maxRetryDelayMs) return response;\n\n\t\tconst headers = new Headers(response.headers);\n\t\theaders.set(\"x-should-retry\", \"false\");\n\t\treturn new Response(response.body, {\n\t\t\tstatus: response.status,\n\t\t\tstatusText: response.statusText,\n\t\t\theaders,\n\t\t});\n\t};\n}\n\n/**\n * The `fetch` override to spread into an SDK client's options, or nothing when\n * no cap applies.\n *\n * Spread rather than assigned so a disabled cap leaves the key absent\n * altogether: passing `fetch: undefined` explicitly would override the client's\n * own default instead of leaving it alone.\n */\nexport function retryDelayCapFetch(maxRetryDelayMs: number | undefined): { fetch?: typeof fetch } {\n\tif (maxRetryDelayMs === undefined) return {};\n\tconst capped = createRetryDelayCapFetch(maxRetryDelayMs);\n\treturn capped === fetch ? {} : { fetch: capped };\n}\n\n/** The header bag an SDK error carries, if it is that kind of error. */\nfunction errorHeaders(error: unknown): HeaderBag | undefined {\n\tif (!error || typeof error !== \"object\") return undefined;\n\tconst headers = (error as { headers?: unknown }).headers;\n\tif (!headers || typeof headers !== \"object\") return undefined;\n\treturn typeof (headers as HeaderBag).get === \"function\" ? (headers as HeaderBag) : undefined;\n}\n\n/**\n * The provider's error message, plus how long it wants us gone.\n *\n * A bare \"429 quota exceeded\" is the same sentence whether the quota returns in\n * thirty seconds or four weeks, and those call for opposite responses from the\n * person reading it. The wait is already on the error; it just was never shown.\n */\nexport function describeProviderError(error: unknown, maxRetryDelayMs = DEFAULT_MAX_RETRY_DELAY_MS): string {\n\tconst message = error instanceof Error ? error.message : JSON.stringify(error);\n\n\tconst delayMs = parseRetryAfterMs(errorHeaders(error));\n\tif (delayMs === undefined || delayMs <= Math.max(0, maxRetryDelayMs)) return message;\n\n\treturn `${message} (the provider ${LONG_DELAY_MARKER} ${formatDelay(delayMs)} before retrying, so no retry was attempted)`;\n}\n\n/**\n * Whether an error message describes a wait too long to sit through — the\n * signal that retrying is pointless rather than merely slow.\n */\nexport function isLongRetryDelayError(message: string | undefined): boolean {\n\treturn message?.includes(`provider ${LONG_DELAY_MARKER}`) ?? false;\n}\n"]}
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Server-requested retry delays, and the cap that keeps them sane.
|
|
3
|
+
*
|
|
4
|
+
* A provider that has run out of quota answers `429` with a `Retry-After`
|
|
5
|
+
* naming when the quota resets — which for a monthly plan is weeks away, not
|
|
6
|
+
* seconds. The vendor SDKs take that header literally and sleep for it, and a
|
|
7
|
+
* delay that large does not survive the trip: `setTimeout` holds a signed
|
|
8
|
+
* 32-bit millisecond count, so anything past ~24.9 days is silently clamped to
|
|
9
|
+
* 1ms. The "wait 28 days" becomes "retry immediately", the retry draws another
|
|
10
|
+
* 429, and the SDK burns its whole retry budget in a few milliseconds against a
|
|
11
|
+
* quota that will not move for a month.
|
|
12
|
+
*
|
|
13
|
+
* So a requested delay past the cap is not something to wait out — it is a
|
|
14
|
+
* different kind of failure, and the caller needs to see it rather than sit in
|
|
15
|
+
* a hot loop behind it.
|
|
16
|
+
*/
|
|
17
|
+
/**
|
|
18
|
+
* The largest delay a timer can actually hold.
|
|
19
|
+
*
|
|
20
|
+
* `setTimeout` stores its delay in a signed 32-bit integer. Larger values do
|
|
21
|
+
* not throw: Node warns and substitutes 1ms, turning the longest wait into the
|
|
22
|
+
* shortest one.
|
|
23
|
+
*/
|
|
24
|
+
export const MAX_TIMER_DELAY_MS = 2 ** 31 - 1;
|
|
25
|
+
/** Longest server-requested delay worth waiting out, when the caller names none. */
|
|
26
|
+
export const DEFAULT_MAX_RETRY_DELAY_MS = 60_000;
|
|
27
|
+
/** Phrase every capped-delay message carries, so callers can recognise one. */
|
|
28
|
+
const LONG_DELAY_MARKER = "asked to wait";
|
|
29
|
+
/**
|
|
30
|
+
* The delay a response asks for, in milliseconds, or undefined if it asks for
|
|
31
|
+
* none.
|
|
32
|
+
*
|
|
33
|
+
* Mirrors what the OpenAI and Anthropic SDKs do with these headers, because the
|
|
34
|
+
* point is to predict the sleep they are about to take: `retry-after-ms` wins,
|
|
35
|
+
* then `retry-after` as seconds, then `retry-after` as an HTTP date.
|
|
36
|
+
*/
|
|
37
|
+
export function parseRetryAfterMs(headers) {
|
|
38
|
+
if (!headers)
|
|
39
|
+
return undefined;
|
|
40
|
+
const afterMs = headers.get("retry-after-ms");
|
|
41
|
+
if (afterMs) {
|
|
42
|
+
const ms = Number.parseFloat(afterMs);
|
|
43
|
+
if (!Number.isNaN(ms))
|
|
44
|
+
return ms;
|
|
45
|
+
}
|
|
46
|
+
const after = headers.get("retry-after");
|
|
47
|
+
if (!after)
|
|
48
|
+
return undefined;
|
|
49
|
+
const seconds = Number.parseFloat(after);
|
|
50
|
+
if (!Number.isNaN(seconds))
|
|
51
|
+
return seconds * 1000;
|
|
52
|
+
// The header's other legal form is an HTTP date.
|
|
53
|
+
const at = Date.parse(after);
|
|
54
|
+
return Number.isNaN(at) ? undefined : at - Date.now();
|
|
55
|
+
}
|
|
56
|
+
/**
|
|
57
|
+
* A duration as a person would say it: the two largest units that carry
|
|
58
|
+
* meaning, because "28d 15h" tells you to go do something else and
|
|
59
|
+
* "2472352s" does not.
|
|
60
|
+
*/
|
|
61
|
+
export function formatDelay(ms) {
|
|
62
|
+
if (!Number.isFinite(ms) || ms < 0)
|
|
63
|
+
return "an unknown time";
|
|
64
|
+
const totalSeconds = Math.round(ms / 1000);
|
|
65
|
+
if (totalSeconds < 60)
|
|
66
|
+
return `${totalSeconds}s`;
|
|
67
|
+
const units = [
|
|
68
|
+
[86400, "d"],
|
|
69
|
+
[3600, "h"],
|
|
70
|
+
[60, "m"],
|
|
71
|
+
[1, "s"],
|
|
72
|
+
];
|
|
73
|
+
const parts = [];
|
|
74
|
+
let remaining = totalSeconds;
|
|
75
|
+
for (const [size, suffix] of units) {
|
|
76
|
+
const value = Math.floor(remaining / size);
|
|
77
|
+
remaining -= value * size;
|
|
78
|
+
if (value > 0)
|
|
79
|
+
parts.push(`${value}${suffix}`);
|
|
80
|
+
if (parts.length === 2)
|
|
81
|
+
break;
|
|
82
|
+
}
|
|
83
|
+
return parts.join(" ");
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Wrap `fetch` so a response asking to wait longer than `maxRetryDelayMs` is
|
|
87
|
+
* marked as one the SDK must not retry.
|
|
88
|
+
*
|
|
89
|
+
* `x-should-retry: false` is the escape hatch both the OpenAI and Anthropic
|
|
90
|
+
* clients check before anything else, so stamping it is enough to stop the
|
|
91
|
+
* retry loop before it computes an unholdable sleep. The response is otherwise
|
|
92
|
+
* passed through untouched — same status, same body — so the error the caller
|
|
93
|
+
* finally sees is still the provider's own.
|
|
94
|
+
*
|
|
95
|
+
* A cap of zero (or a nonsensical one) disables the wrapper entirely, which is
|
|
96
|
+
* what the documented "set to 0 to disable" means.
|
|
97
|
+
*/
|
|
98
|
+
export function createRetryDelayCapFetch(maxRetryDelayMs, baseFetch = fetch) {
|
|
99
|
+
if (!Number.isFinite(maxRetryDelayMs) || maxRetryDelayMs <= 0)
|
|
100
|
+
return baseFetch;
|
|
101
|
+
return async (input, init) => {
|
|
102
|
+
const response = await baseFetch(input, init);
|
|
103
|
+
// Only a failure is ever retried, and a server that already stated its
|
|
104
|
+
// own preference outranks the cap.
|
|
105
|
+
if (response.ok || response.headers.get("x-should-retry") !== null)
|
|
106
|
+
return response;
|
|
107
|
+
const delayMs = parseRetryAfterMs(response.headers);
|
|
108
|
+
if (delayMs === undefined || delayMs <= maxRetryDelayMs)
|
|
109
|
+
return response;
|
|
110
|
+
const headers = new Headers(response.headers);
|
|
111
|
+
headers.set("x-should-retry", "false");
|
|
112
|
+
return new Response(response.body, {
|
|
113
|
+
status: response.status,
|
|
114
|
+
statusText: response.statusText,
|
|
115
|
+
headers,
|
|
116
|
+
});
|
|
117
|
+
};
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* The `fetch` override to spread into an SDK client's options, or nothing when
|
|
121
|
+
* no cap applies.
|
|
122
|
+
*
|
|
123
|
+
* Spread rather than assigned so a disabled cap leaves the key absent
|
|
124
|
+
* altogether: passing `fetch: undefined` explicitly would override the client's
|
|
125
|
+
* own default instead of leaving it alone.
|
|
126
|
+
*/
|
|
127
|
+
export function retryDelayCapFetch(maxRetryDelayMs) {
|
|
128
|
+
if (maxRetryDelayMs === undefined)
|
|
129
|
+
return {};
|
|
130
|
+
const capped = createRetryDelayCapFetch(maxRetryDelayMs);
|
|
131
|
+
return capped === fetch ? {} : { fetch: capped };
|
|
132
|
+
}
|
|
133
|
+
/** The header bag an SDK error carries, if it is that kind of error. */
|
|
134
|
+
function errorHeaders(error) {
|
|
135
|
+
if (!error || typeof error !== "object")
|
|
136
|
+
return undefined;
|
|
137
|
+
const headers = error.headers;
|
|
138
|
+
if (!headers || typeof headers !== "object")
|
|
139
|
+
return undefined;
|
|
140
|
+
return typeof headers.get === "function" ? headers : undefined;
|
|
141
|
+
}
|
|
142
|
+
/**
|
|
143
|
+
* The provider's error message, plus how long it wants us gone.
|
|
144
|
+
*
|
|
145
|
+
* A bare "429 quota exceeded" is the same sentence whether the quota returns in
|
|
146
|
+
* thirty seconds or four weeks, and those call for opposite responses from the
|
|
147
|
+
* person reading it. The wait is already on the error; it just was never shown.
|
|
148
|
+
*/
|
|
149
|
+
export function describeProviderError(error, maxRetryDelayMs = DEFAULT_MAX_RETRY_DELAY_MS) {
|
|
150
|
+
const message = error instanceof Error ? error.message : JSON.stringify(error);
|
|
151
|
+
const delayMs = parseRetryAfterMs(errorHeaders(error));
|
|
152
|
+
if (delayMs === undefined || delayMs <= Math.max(0, maxRetryDelayMs))
|
|
153
|
+
return message;
|
|
154
|
+
return `${message} (the provider ${LONG_DELAY_MARKER} ${formatDelay(delayMs)} before retrying, so no retry was attempted)`;
|
|
155
|
+
}
|
|
156
|
+
/**
|
|
157
|
+
* Whether an error message describes a wait too long to sit through — the
|
|
158
|
+
* signal that retrying is pointless rather than merely slow.
|
|
159
|
+
*/
|
|
160
|
+
export function isLongRetryDelayError(message) {
|
|
161
|
+
return message?.includes(`provider ${LONG_DELAY_MARKER}`) ?? false;
|
|
162
|
+
}
|
|
163
|
+
//# sourceMappingURL=retry-delay.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"retry-delay.js","sourceRoot":"","sources":["../../src/utils/retry-delay.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAEH;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,CAAC,IAAI,EAAE,GAAG,CAAC,CAAC;AAE9C,oFAAoF;AACpF,MAAM,CAAC,MAAM,0BAA0B,GAAG,MAAM,CAAC;AAEjD,+EAA+E;AAC/E,MAAM,iBAAiB,GAAG,eAAe,CAAC;AAO1C;;;;;;;GAOG;AACH,MAAM,UAAU,iBAAiB,CAAC,OAA8B,EAAsB;IACrF,IAAI,CAAC,OAAO;QAAE,OAAO,SAAS,CAAC;IAE/B,MAAM,OAAO,GAAG,OAAO,CAAC,GAAG,CAAC,gBAAgB,CAAC,CAAC;IAC9C,IAAI,OAAO,EAAE,CAAC;QACb,MAAM,EAAE,GAAG,MAAM,CAAC,UAAU,CAAC,OAAO,CAAC,CAAC;QACtC,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC;YAAE,OAAO,EAAE,CAAC;IAClC,CAAC;IAED,MAAM,KAAK,GAAG,OAAO,CAAC,GAAG,CAAC,aAAa,CAAC,CAAC;IACzC,IAAI,CAAC,KAAK;QAAE,OAAO,SAAS,CAAC;IAE7B,MAAM,OAAO,GAAG,MAAM,CAAC,UAAU,CAAC,KAAK,CAAC,CAAC;IACzC,IAAI,CAAC,MAAM,CAAC,KAAK,CAAC,OAAO,CAAC;QAAE,OAAO,OAAO,GAAG,IAAI,CAAC;IAElD,iDAAiD;IACjD,MAAM,EAAE,GAAG,IAAI,CAAC,KAAK,CAAC,KAAK,CAAC,CAAC;IAC7B,OAAO,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,EAAE,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC;AAAA,CACtD;AAED;;;;GAIG;AACH,MAAM,UAAU,WAAW,CAAC,EAAU,EAAU;IAC/C,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,EAAE,CAAC,IAAI,EAAE,GAAG,CAAC;QAAE,OAAO,iBAAiB,CAAC;IAE7D,MAAM,YAAY,GAAG,IAAI,CAAC,KAAK,CAAC,EAAE,GAAG,IAAI,CAAC,CAAC;IAC3C,IAAI,YAAY,GAAG,EAAE;QAAE,OAAO,GAAG,YAAY,GAAG,CAAC;IAEjD,MAAM,KAAK,GAA4B;QACtC,CAAC,KAAK,EAAE,GAAG,CAAC;QACZ,CAAC,IAAI,EAAE,GAAG,CAAC;QACX,CAAC,EAAE,EAAE,GAAG,CAAC;QACT,CAAC,CAAC,EAAE,GAAG,CAAC;KACR,CAAC;IAEF,MAAM,KAAK,GAAa,EAAE,CAAC;IAC3B,IAAI,SAAS,GAAG,YAAY,CAAC;IAC7B,KAAK,MAAM,CAAC,IAAI,EAAE,MAAM,CAAC,IAAI,KAAK,EAAE,CAAC;QACpC,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,SAAS,GAAG,IAAI,CAAC,CAAC;QAC3C,SAAS,IAAI,KAAK,GAAG,IAAI,CAAC;QAC1B,IAAI,KAAK,GAAG,CAAC;YAAE,KAAK,CAAC,IAAI,CAAC,GAAG,KAAK,GAAG,MAAM,EAAE,CAAC,CAAC;QAC/C,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC;YAAE,MAAM;IAC/B,CAAC;IACD,OAAO,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,CAAC;AAAA,CACvB;AAED;;;;;;;;;;;;GAYG;AACH,MAAM,UAAU,wBAAwB,CAAC,eAAuB,EAAE,SAAS,GAAiB,KAAK,EAAgB;IAChH,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,eAAe,CAAC,IAAI,eAAe,IAAI,CAAC;QAAE,OAAO,SAAS,CAAC;IAEhF,OAAO,KAAK,EAAE,KAAkC,EAAE,IAAkC,EAAqB,EAAE,CAAC;QAC3G,MAAM,QAAQ,GAAG,MAAM,SAAS,CAAC,KAAK,EAAE,IAAI,CAAC,CAAC;QAE9C,uEAAuE;QACvE,mCAAmC;QACnC,IAAI,QAAQ,CAAC,EAAE,IAAI,QAAQ,CAAC,OAAO,CAAC,GAAG,CAAC,gBAAgB,CAAC,KAAK,IAAI;YAAE,OAAO,QAAQ,CAAC;QAEpF,MAAM,OAAO,GAAG,iBAAiB,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC;QACpD,IAAI,OAAO,KAAK,SAAS,IAAI,OAAO,IAAI,eAAe;YAAE,OAAO,QAAQ,CAAC;QAEzE,MAAM,OAAO,GAAG,IAAI,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAC,CAAC;QAC9C,OAAO,CAAC,GAAG,CAAC,gBAAgB,EAAE,OAAO,CAAC,CAAC;QACvC,OAAO,IAAI,QAAQ,CAAC,QAAQ,CAAC,IAAI,EAAE;YAClC,MAAM,EAAE,QAAQ,CAAC,MAAM;YACvB,UAAU,EAAE,QAAQ,CAAC,UAAU;YAC/B,OAAO;SACP,CAAC,CAAC;IAAA,CACH,CAAC;AAAA,CACF;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,kBAAkB,CAAC,eAAmC,EAA4B;IACjG,IAAI,eAAe,KAAK,SAAS;QAAE,OAAO,EAAE,CAAC;IAC7C,MAAM,MAAM,GAAG,wBAAwB,CAAC,eAAe,CAAC,CAAC;IACzD,OAAO,MAAM,KAAK,KAAK,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,EAAE,KAAK,EAAE,MAAM,EAAE,CAAC;AAAA,CACjD;AAED,wEAAwE;AACxE,SAAS,YAAY,CAAC,KAAc,EAAyB;IAC5D,IAAI,CAAC,KAAK,IAAI,OAAO,KAAK,KAAK,QAAQ;QAAE,OAAO,SAAS,CAAC;IAC1D,MAAM,OAAO,GAAI,KAA+B,CAAC,OAAO,CAAC;IACzD,IAAI,CAAC,OAAO,IAAI,OAAO,OAAO,KAAK,QAAQ;QAAE,OAAO,SAAS,CAAC;IAC9D,OAAO,OAAQ,OAAqB,CAAC,GAAG,KAAK,UAAU,CAAC,CAAC,CAAE,OAAqB,CAAC,CAAC,CAAC,SAAS,CAAC;AAAA,CAC7F;AAED;;;;;;GAMG;AACH,MAAM,UAAU,qBAAqB,CAAC,KAAc,EAAE,eAAe,GAAG,0BAA0B,EAAU;IAC3G,MAAM,OAAO,GAAG,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,IAAI,CAAC,SAAS,CAAC,KAAK,CAAC,CAAC;IAE/E,MAAM,OAAO,GAAG,iBAAiB,CAAC,YAAY,CAAC,KAAK,CAAC,CAAC,CAAC;IACvD,IAAI,OAAO,KAAK,SAAS,IAAI,OAAO,IAAI,IAAI,CAAC,GAAG,CAAC,CAAC,EAAE,eAAe,CAAC;QAAE,OAAO,OAAO,CAAC;IAErF,OAAO,GAAG,OAAO,kBAAkB,iBAAiB,IAAI,WAAW,CAAC,OAAO,CAAC,8CAA8C,CAAC;AAAA,CAC3H;AAED;;;GAGG;AACH,MAAM,UAAU,qBAAqB,CAAC,OAA2B,EAAW;IAC3E,OAAO,OAAO,EAAE,QAAQ,CAAC,YAAY,iBAAiB,EAAE,CAAC,IAAI,KAAK,CAAC;AAAA,CACnE","sourcesContent":["/**\n * Server-requested retry delays, and the cap that keeps them sane.\n *\n * A provider that has run out of quota answers `429` with a `Retry-After`\n * naming when the quota resets — which for a monthly plan is weeks away, not\n * seconds. The vendor SDKs take that header literally and sleep for it, and a\n * delay that large does not survive the trip: `setTimeout` holds a signed\n * 32-bit millisecond count, so anything past ~24.9 days is silently clamped to\n * 1ms. The \"wait 28 days\" becomes \"retry immediately\", the retry draws another\n * 429, and the SDK burns its whole retry budget in a few milliseconds against a\n * quota that will not move for a month.\n *\n * So a requested delay past the cap is not something to wait out — it is a\n * different kind of failure, and the caller needs to see it rather than sit in\n * a hot loop behind it.\n */\n\n/**\n * The largest delay a timer can actually hold.\n *\n * `setTimeout` stores its delay in a signed 32-bit integer. Larger values do\n * not throw: Node warns and substitutes 1ms, turning the longest wait into the\n * shortest one.\n */\nexport const MAX_TIMER_DELAY_MS = 2 ** 31 - 1;\n\n/** Longest server-requested delay worth waiting out, when the caller names none. */\nexport const DEFAULT_MAX_RETRY_DELAY_MS = 60_000;\n\n/** Phrase every capped-delay message carries, so callers can recognise one. */\nconst LONG_DELAY_MARKER = \"asked to wait\";\n\n/** Anything with a `get` — a `Headers`, or an SDK error's header bag. */\ninterface HeaderBag {\n\tget(name: string): string | null | undefined;\n}\n\n/**\n * The delay a response asks for, in milliseconds, or undefined if it asks for\n * none.\n *\n * Mirrors what the OpenAI and Anthropic SDKs do with these headers, because the\n * point is to predict the sleep they are about to take: `retry-after-ms` wins,\n * then `retry-after` as seconds, then `retry-after` as an HTTP date.\n */\nexport function parseRetryAfterMs(headers: HeaderBag | undefined): number | undefined {\n\tif (!headers) return undefined;\n\n\tconst afterMs = headers.get(\"retry-after-ms\");\n\tif (afterMs) {\n\t\tconst ms = Number.parseFloat(afterMs);\n\t\tif (!Number.isNaN(ms)) return ms;\n\t}\n\n\tconst after = headers.get(\"retry-after\");\n\tif (!after) return undefined;\n\n\tconst seconds = Number.parseFloat(after);\n\tif (!Number.isNaN(seconds)) return seconds * 1000;\n\n\t// The header's other legal form is an HTTP date.\n\tconst at = Date.parse(after);\n\treturn Number.isNaN(at) ? undefined : at - Date.now();\n}\n\n/**\n * A duration as a person would say it: the two largest units that carry\n * meaning, because \"28d 15h\" tells you to go do something else and\n * \"2472352s\" does not.\n */\nexport function formatDelay(ms: number): string {\n\tif (!Number.isFinite(ms) || ms < 0) return \"an unknown time\";\n\n\tconst totalSeconds = Math.round(ms / 1000);\n\tif (totalSeconds < 60) return `${totalSeconds}s`;\n\n\tconst units: Array<[number, string]> = [\n\t\t[86400, \"d\"],\n\t\t[3600, \"h\"],\n\t\t[60, \"m\"],\n\t\t[1, \"s\"],\n\t];\n\n\tconst parts: string[] = [];\n\tlet remaining = totalSeconds;\n\tfor (const [size, suffix] of units) {\n\t\tconst value = Math.floor(remaining / size);\n\t\tremaining -= value * size;\n\t\tif (value > 0) parts.push(`${value}${suffix}`);\n\t\tif (parts.length === 2) break;\n\t}\n\treturn parts.join(\" \");\n}\n\n/**\n * Wrap `fetch` so a response asking to wait longer than `maxRetryDelayMs` is\n * marked as one the SDK must not retry.\n *\n * `x-should-retry: false` is the escape hatch both the OpenAI and Anthropic\n * clients check before anything else, so stamping it is enough to stop the\n * retry loop before it computes an unholdable sleep. The response is otherwise\n * passed through untouched — same status, same body — so the error the caller\n * finally sees is still the provider's own.\n *\n * A cap of zero (or a nonsensical one) disables the wrapper entirely, which is\n * what the documented \"set to 0 to disable\" means.\n */\nexport function createRetryDelayCapFetch(maxRetryDelayMs: number, baseFetch: typeof fetch = fetch): typeof fetch {\n\tif (!Number.isFinite(maxRetryDelayMs) || maxRetryDelayMs <= 0) return baseFetch;\n\n\treturn async (input: Parameters<typeof fetch>[0], init?: Parameters<typeof fetch>[1]): Promise<Response> => {\n\t\tconst response = await baseFetch(input, init);\n\n\t\t// Only a failure is ever retried, and a server that already stated its\n\t\t// own preference outranks the cap.\n\t\tif (response.ok || response.headers.get(\"x-should-retry\") !== null) return response;\n\n\t\tconst delayMs = parseRetryAfterMs(response.headers);\n\t\tif (delayMs === undefined || delayMs <= maxRetryDelayMs) return response;\n\n\t\tconst headers = new Headers(response.headers);\n\t\theaders.set(\"x-should-retry\", \"false\");\n\t\treturn new Response(response.body, {\n\t\t\tstatus: response.status,\n\t\t\tstatusText: response.statusText,\n\t\t\theaders,\n\t\t});\n\t};\n}\n\n/**\n * The `fetch` override to spread into an SDK client's options, or nothing when\n * no cap applies.\n *\n * Spread rather than assigned so a disabled cap leaves the key absent\n * altogether: passing `fetch: undefined` explicitly would override the client's\n * own default instead of leaving it alone.\n */\nexport function retryDelayCapFetch(maxRetryDelayMs: number | undefined): { fetch?: typeof fetch } {\n\tif (maxRetryDelayMs === undefined) return {};\n\tconst capped = createRetryDelayCapFetch(maxRetryDelayMs);\n\treturn capped === fetch ? {} : { fetch: capped };\n}\n\n/** The header bag an SDK error carries, if it is that kind of error. */\nfunction errorHeaders(error: unknown): HeaderBag | undefined {\n\tif (!error || typeof error !== \"object\") return undefined;\n\tconst headers = (error as { headers?: unknown }).headers;\n\tif (!headers || typeof headers !== \"object\") return undefined;\n\treturn typeof (headers as HeaderBag).get === \"function\" ? (headers as HeaderBag) : undefined;\n}\n\n/**\n * The provider's error message, plus how long it wants us gone.\n *\n * A bare \"429 quota exceeded\" is the same sentence whether the quota returns in\n * thirty seconds or four weeks, and those call for opposite responses from the\n * person reading it. The wait is already on the error; it just was never shown.\n */\nexport function describeProviderError(error: unknown, maxRetryDelayMs = DEFAULT_MAX_RETRY_DELAY_MS): string {\n\tconst message = error instanceof Error ? error.message : JSON.stringify(error);\n\n\tconst delayMs = parseRetryAfterMs(errorHeaders(error));\n\tif (delayMs === undefined || delayMs <= Math.max(0, maxRetryDelayMs)) return message;\n\n\treturn `${message} (the provider ${LONG_DELAY_MARKER} ${formatDelay(delayMs)} before retrying, so no retry was attempted)`;\n}\n\n/**\n * Whether an error message describes a wait too long to sit through — the\n * signal that retrying is pointless rather than merely slow.\n */\nexport function isLongRetryDelayError(message: string | undefined): boolean {\n\treturn message?.includes(`provider ${LONG_DELAY_MARKER}`) ?? false;\n}\n"]}
|