@xlaunch/llm 0.2.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +22 -0
- package/README.md +174 -0
- package/lib/index.js +2300 -0
- package/lib/invariant.js +84 -0
- package/lib/typert.host.d.ts +3 -0
- package/lib/typert.host.js +547 -0
- package/lib/typert.remote-client.d.ts +26 -0
- package/lib/typert.remote-client.js +102 -0
- package/lib/types/adapter-failure.d.ts +14 -0
- package/lib/types/adapter-failure.js +105 -0
- package/lib/types/api-key.d.ts +28 -0
- package/lib/types/api-key.js +34 -0
- package/lib/types/assembler.d.ts +75 -0
- package/lib/types/assembler.js +191 -0
- package/lib/types/assistant-stream.d.ts +166 -0
- package/lib/types/assistant-stream.js +458 -0
- package/lib/types/attribution.d.ts +47 -0
- package/lib/types/attribution.js +46 -0
- package/lib/types/brand.d.ts +56 -0
- package/lib/types/brand.js +53 -0
- package/lib/types/call-config.d.ts +53 -0
- package/lib/types/call-config.js +46 -0
- package/lib/types/content.d.ts +130 -0
- package/lib/types/content.js +284 -0
- package/lib/types/error.d.ts +73 -0
- package/lib/types/error.js +145 -0
- package/lib/types/index.d.ts +408 -0
- package/lib/types/index.js +920 -0
- package/lib/types/invariant.d.ts +13 -0
- package/lib/types/invariant.js +100 -0
- package/lib/types/message.d.ts +197 -0
- package/lib/types/message.js +82 -0
- package/lib/types/retry-policy.d.ts +66 -0
- package/lib/types/retry-policy.js +127 -0
- package/lib/types/types.d.ts +430 -0
- package/lib/types/types.js +7 -0
- package/package.json +81 -0
package/lib/index.js
ADDED
|
@@ -0,0 +1,2300 @@
|
|
|
1
|
+
import { createRequire } from "node:module";
|
|
2
|
+
import { Remote, RemoteError, TypertRemoteService } from "@xlaunch/typert-protocol";
|
|
3
|
+
import { assertNever, deepFreeze, snapshotJsonValue } from "@xlaunch/util-values";
|
|
4
|
+
import { randomUUID } from "@xlaunch/util-crypto";
|
|
5
|
+
import { brandString } from "@xlaunch/brand";
|
|
6
|
+
import z from "@xlaunch/schemastery";
|
|
7
|
+
import { MAX_TIMER_DELAY_MS } from "@xlaunch/timeout";
|
|
8
|
+
//#region lib/types/message.js
|
|
9
|
+
/** Message value types, identity, and immutable construction helpers. */
|
|
10
|
+
/**
|
|
11
|
+
* Bound for a `notice` summary. The account rides a collapsed transcript row
|
|
12
|
+
* and is committed to the durable log, while its inputs — task labels, goal
|
|
13
|
+
* objectives, tool arguments — are caller text with no length of their own.
|
|
14
|
+
*/
|
|
15
|
+
const CONTEXT_SUMMARY_MAX_CHARS = 120;
|
|
16
|
+
/**
|
|
17
|
+
* Bound one `notice` summary to {@link CONTEXT_SUMMARY_MAX_CHARS}.
|
|
18
|
+
* @param summary - the producer's one-line account, of any length.
|
|
19
|
+
* @returns the account, ellipsized when it exceeds the bound.
|
|
20
|
+
*/
|
|
21
|
+
function boundContextSummary(summary) {
|
|
22
|
+
return summary.length <= 120 ? summary : `${summary.slice(0, 119)}…`;
|
|
23
|
+
}
|
|
24
|
+
/**
|
|
25
|
+
* Detach and deep-freeze a message whose identity already exists.
|
|
26
|
+
* @param message - complete message, including its stable identity.
|
|
27
|
+
* @returns an immutable snapshot that preserves the identity.
|
|
28
|
+
*/
|
|
29
|
+
function freezeMessage(message) {
|
|
30
|
+
return deepFreeze(structuredClone(message));
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* Create one identified message and freeze it before publication.
|
|
34
|
+
* @param input - complete role, content, and source for a new message.
|
|
35
|
+
* @returns an immutable message with a fresh stable identity.
|
|
36
|
+
*/
|
|
37
|
+
function createMessage(input) {
|
|
38
|
+
return freezeMessage({
|
|
39
|
+
...input,
|
|
40
|
+
id: brandString(randomUUID())
|
|
41
|
+
});
|
|
42
|
+
}
|
|
43
|
+
/**
|
|
44
|
+
* Create one identified user-role message and freeze it before publication.
|
|
45
|
+
* @param input - complete content and source for a new user message.
|
|
46
|
+
* @returns an immutable user message with a fresh stable identity.
|
|
47
|
+
*/
|
|
48
|
+
function createUserMessage(input) {
|
|
49
|
+
return createMessage({
|
|
50
|
+
...input,
|
|
51
|
+
role: "user"
|
|
52
|
+
});
|
|
53
|
+
}
|
|
54
|
+
/**
|
|
55
|
+
* Create one identified model-produced assistant message and freeze it before publication.
|
|
56
|
+
* @param input - complete content plus the provider, model, and optional replay state for a new assistant message.
|
|
57
|
+
* @returns an immutable assistant message with fixed role/source tags and a fresh stable identity.
|
|
58
|
+
*/
|
|
59
|
+
function createAssistantMessage(input) {
|
|
60
|
+
return createMessage({
|
|
61
|
+
role: "assistant",
|
|
62
|
+
content: input.content,
|
|
63
|
+
source: {
|
|
64
|
+
kind: "model",
|
|
65
|
+
...input.source
|
|
66
|
+
}
|
|
67
|
+
});
|
|
68
|
+
}
|
|
69
|
+
/**
|
|
70
|
+
* Create and freeze one identified tool-result message.
|
|
71
|
+
* @param input - call identity, raw result blocks, and outcome.
|
|
72
|
+
* @returns an immutable user-role tool-result message.
|
|
73
|
+
*/
|
|
74
|
+
function createToolResultMessage(input) {
|
|
75
|
+
return createUserMessage({
|
|
76
|
+
source: {
|
|
77
|
+
kind: "tool",
|
|
78
|
+
callId: input.callId
|
|
79
|
+
},
|
|
80
|
+
content: [{
|
|
81
|
+
type: "tool-result",
|
|
82
|
+
toolCallId: input.callId,
|
|
83
|
+
content: input.content,
|
|
84
|
+
isError: input.isError
|
|
85
|
+
}]
|
|
86
|
+
});
|
|
87
|
+
}
|
|
88
|
+
//#endregion
|
|
89
|
+
//#region lib/types/error.js
|
|
90
|
+
/**
|
|
91
|
+
* Harness error base with a stable machine-routable code and chained cause.
|
|
92
|
+
* Package errors extend it so tool results and replay can retain failure class.
|
|
93
|
+
* @module @xlaunch/llm/error
|
|
94
|
+
*/
|
|
95
|
+
/**
|
|
96
|
+
* Base class for all harness errors. Carries a `code` (stable, programmatic —
|
|
97
|
+
* e.g. `NO_ADAPTER`, `INVALID_ARGS`, `INVARIANT`) distinct from the
|
|
98
|
+
* human-readable `message`, and supports `cause` chaining via the standard
|
|
99
|
+
* `ErrorOptions`. `name` defaults to the subclass constructor name.
|
|
100
|
+
*/
|
|
101
|
+
var HarnessError = class extends Error {
|
|
102
|
+
/** Stable machine-routable failure class (e.g. `RATE_LIMIT`); route on this, never by parsing `message`. */
|
|
103
|
+
code;
|
|
104
|
+
constructor(message, code, options) {
|
|
105
|
+
super(message, options);
|
|
106
|
+
this.code = code;
|
|
107
|
+
this.name = new.target.name;
|
|
108
|
+
}
|
|
109
|
+
};
|
|
110
|
+
/** Canonical provider-neutral code for a model request rejected because its context window was exceeded. */
|
|
111
|
+
const CONTEXT_WINDOW_EXCEEDED_CODE = "CONTEXT_WINDOW_EXCEEDED";
|
|
112
|
+
/** Canonical provider-neutral code for an exhausted account quota or balance. */
|
|
113
|
+
const QUOTA_EXCEEDED_CODE = "QUOTA";
|
|
114
|
+
/**
|
|
115
|
+
* Canonical provider-neutral code for a response that completed normally but
|
|
116
|
+
* carried no content blocks at all. Providers occasionally emit a degenerate
|
|
117
|
+
* completion (a terminal stop with zero output); adapters classify it as this
|
|
118
|
+
* failure instead of yielding an empty assistant message, because an empty
|
|
119
|
+
* message silently ends the turn with nothing for the user or the loop to act
|
|
120
|
+
* on. The attempt produced nothing durable, so retry policy treats it as safe
|
|
121
|
+
* to repeat.
|
|
122
|
+
*/
|
|
123
|
+
const EMPTY_RESPONSE_CODE = "EMPTY_RESPONSE";
|
|
124
|
+
/**
|
|
125
|
+
* Canonical provider-neutral code for a credential that was supplied but
|
|
126
|
+
* cannot be used — malformed rather than absent. Distinct from
|
|
127
|
+
* `MISSING_CREDENTIAL` because the fix differs: correct the stored value
|
|
128
|
+
* rather than supply one. Deliberately outside the default retryable set —
|
|
129
|
+
* a malformed credential fails identically on every attempt.
|
|
130
|
+
*/
|
|
131
|
+
const INVALID_CREDENTIAL_CODE = "INVALID_CREDENTIAL";
|
|
132
|
+
/** Structured codes and plain phrases that explicitly name a context bound being exceeded. */
|
|
133
|
+
const STRUCTURED_CONTEXT_OVERFLOW = new RegExp(String.raw`(?:^|[^a-z0-9])context[\s_-](?:length|window)[\s_-]` + String.raw`(?:exceed(?:ed|s)?|overflow(?:ed)?|limit[\s_-]exceeded)(?:$|[^a-z0-9])`, "i");
|
|
134
|
+
/** Request-size wording that ties "too large" directly to model context capacity. */
|
|
135
|
+
const TOO_LARGE_FOR_CONTEXT = new RegExp(String.raw`\b(?:request|prompt|input|messages?)\s+(?:is\s+|are\s+)?` + String.raw`too\s+(?:large|long)\s+for\s+(?:(?:this|the)\s+)?` + String.raw`(?:model(?:'s)?\s+)?context(?:\s+window)?\b`, "i");
|
|
136
|
+
/** "Exceeds" wording is safe only when its object is explicitly the model context. */
|
|
137
|
+
const EXCEEDS_MODEL_CONTEXT = new RegExp(String.raw`\b(?:input|prompt|request|messages?)\b.{0,40}` + String.raw`\b(?:exceed(?:s|ed)?|overflows?|is\s+larger\s+than)\b.{0,40}` + String.raw`\b(?:the\s+)?(?:model(?:'s)?\s+)?context(?:\s+(?:length|window))?\b`, "i");
|
|
138
|
+
/**
|
|
139
|
+
* Recognize the context-overflow wording used by OpenAI-compatible providers
|
|
140
|
+
* and library adapters. Adapters pass all available provider code, type, and
|
|
141
|
+
* message text so both thrown and in-band delivery styles share one classifier.
|
|
142
|
+
* @param detail - provider error code/type/message text joined into one string.
|
|
143
|
+
* @returns true when the detail identifies a request exceeding the model context window.
|
|
144
|
+
*/
|
|
145
|
+
function isContextWindowExceededError(detail) {
|
|
146
|
+
return STRUCTURED_CONTEXT_OVERFLOW.test(detail) || /\b(?:maximum|max)(?:\s+(?:allowed|supported))?\s+context\s+(?:length|window)\b/i.test(detail) || TOO_LARGE_FOR_CONTEXT.test(detail) || /\b(?:input|prompt|request)\s+(?:is\s+)?too\s+(?:long|large)\s+for\s+(?:this|the)\s+model\b/i.test(detail) || EXCEEDS_MODEL_CONTEXT.test(detail);
|
|
147
|
+
}
|
|
148
|
+
/**
|
|
149
|
+
* Recognize provider wording that identifies an exhausted account quota rather
|
|
150
|
+
* than a transient request-rate limit.
|
|
151
|
+
* @param detail - provider error code/type/message text joined into one string.
|
|
152
|
+
* @returns true only for terminal quota, balance, credit, budget, or usage-limit wording.
|
|
153
|
+
*/
|
|
154
|
+
function isQuotaExceededError(detail) {
|
|
155
|
+
return /\binsufficient[\s_-]+(?:quota|balance|credits?)\b/i.test(detail) || /\b(?:quota|usage[\s_-]+limit)[\s_-]+(?:exceeded|exhausted|reached)\b/i.test(detail) || /\bexceed(?:ed|s)?[\s_-]+(?:(?:your|the)[\s_-]+)?(?:current[\s_-]+)?quota\b/i.test(detail) || /\b(?:balance|credits?)[\s_-]+(?:exhausted|depleted)\b/i.test(detail) || /\bout[\s_-]+of[\s_-]+(?:credits?|budget)\b/i.test(detail);
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Render a thrown value with its full `cause` chain and AggregateError
|
|
159
|
+
* members, so transport wrappers like undici's `TypeError: fetch failed`
|
|
160
|
+
* surface the underlying failure instead of masking it. Plain structured
|
|
161
|
+
* failures render their own data-backed `message`. Diagnostic-surface
|
|
162
|
+
* rendering only (messages, notices, logs) — never parse the result; route on
|
|
163
|
+
* {@link HarnessError.code}.
|
|
164
|
+
* @param value - the caught value (`unknown` in catch clauses).
|
|
165
|
+
* @returns the outermost message first, each cause appended with `: ` (skipped
|
|
166
|
+
* when it repeats the wrapper message verbatim), and AggregateError members
|
|
167
|
+
* bracketed and `; `-joined.
|
|
168
|
+
*/
|
|
169
|
+
function errorChain(value) {
|
|
170
|
+
const path = /* @__PURE__ */ new Set();
|
|
171
|
+
const render = (current) => {
|
|
172
|
+
if (path.has(current)) return "<circular cause>";
|
|
173
|
+
path.add(current);
|
|
174
|
+
try {
|
|
175
|
+
if (!(current instanceof Error)) {
|
|
176
|
+
if (typeof current === "object" && current !== null) {
|
|
177
|
+
const descriptor = Object.getOwnPropertyDescriptor(current, "message");
|
|
178
|
+
if (descriptor !== void 0 && "value" in descriptor && typeof descriptor.value === "string") return descriptor.value;
|
|
179
|
+
}
|
|
180
|
+
return String(current);
|
|
181
|
+
}
|
|
182
|
+
const message = current.message === "" ? current.name : current.message;
|
|
183
|
+
const members = current instanceof AggregateError && current.errors.length > 0 ? ` [${current.errors.map(render).join("; ")}]` : "";
|
|
184
|
+
const causeText = current.cause === void 0 || current.cause === null ? "" : render(current.cause);
|
|
185
|
+
return `${message}${members}${causeText === "" || causeText === message ? "" : `: ${causeText}`}`;
|
|
186
|
+
} catch {
|
|
187
|
+
return "<unrenderable value>";
|
|
188
|
+
} finally {
|
|
189
|
+
path.delete(current);
|
|
190
|
+
}
|
|
191
|
+
};
|
|
192
|
+
return render(value);
|
|
193
|
+
}
|
|
194
|
+
/**
|
|
195
|
+
* Narrow an arbitrary thrown value to a HarnessError (for `instanceof` at runtime boundaries).
|
|
196
|
+
* @param value - the caught value (`unknown` in catch clauses).
|
|
197
|
+
* @returns true only for real instances; duck-typed or cross-realm errors do not narrow.
|
|
198
|
+
*/
|
|
199
|
+
function isHarnessError(value) {
|
|
200
|
+
return value instanceof HarnessError;
|
|
201
|
+
}
|
|
202
|
+
//#endregion
|
|
203
|
+
//#region lib/types/retry-policy.js
|
|
204
|
+
/**
|
|
205
|
+
* Provider-owned request-retry policy configuration and resolution.
|
|
206
|
+
*
|
|
207
|
+
* Adapters expose one resolved policy per registered provider route; the
|
|
208
|
+
* optional xlaunch-llm-retry plugin executes it on the agent's failed-step extension point.
|
|
209
|
+
*
|
|
210
|
+
* @module @xlaunch/llm/retry-policy
|
|
211
|
+
*/
|
|
212
|
+
const DEFAULT_MAX_RETRIES = 5;
|
|
213
|
+
const DEFAULT_INITIAL_DELAY_MS = 500;
|
|
214
|
+
const DEFAULT_MAX_DELAY_MS = 1e4;
|
|
215
|
+
const DEFAULT_JITTER_RATIO = .1;
|
|
216
|
+
const DEFAULT_RETRYABLE_CODES = Object.freeze([
|
|
217
|
+
EMPTY_RESPONSE_CODE,
|
|
218
|
+
"RATE_LIMIT",
|
|
219
|
+
"SERVER",
|
|
220
|
+
"TIMEOUT",
|
|
221
|
+
"TRANSPORT"
|
|
222
|
+
]);
|
|
223
|
+
const backoffSchema = z.object({
|
|
224
|
+
initialDelayMs: z.number().max(MAX_TIMER_DELAY_MS).default(DEFAULT_INITIAL_DELAY_MS),
|
|
225
|
+
maxDelayMs: z.number().max(MAX_TIMER_DELAY_MS).default(DEFAULT_MAX_DELAY_MS),
|
|
226
|
+
jitterRatio: z.number().min(0).max(1).default(DEFAULT_JITTER_RATIO)
|
|
227
|
+
});
|
|
228
|
+
const normalPolicySchema = z.object({
|
|
229
|
+
mode: z.const("normal").required(),
|
|
230
|
+
maxRetries: z.number().step(1).min(0).max(Number.MAX_SAFE_INTEGER).default(DEFAULT_MAX_RETRIES),
|
|
231
|
+
retryableCodes: z.array(z.string()).default([...DEFAULT_RETRYABLE_CODES]),
|
|
232
|
+
backoff: backoffSchema
|
|
233
|
+
});
|
|
234
|
+
const alwaysPolicySchema = z.object({
|
|
235
|
+
mode: z.const("always").required(),
|
|
236
|
+
backoff: backoffSchema
|
|
237
|
+
});
|
|
238
|
+
/** Cordis schema embedded by each concrete provider configuration. */
|
|
239
|
+
const RetryPolicySchema = z.union([normalPolicySchema, alwaysPolicySchema]);
|
|
240
|
+
const NORMAL_POLICY_KEYS = new Set([
|
|
241
|
+
"mode",
|
|
242
|
+
"maxRetries",
|
|
243
|
+
"retryableCodes",
|
|
244
|
+
"backoff"
|
|
245
|
+
]);
|
|
246
|
+
const ALWAYS_POLICY_KEYS = new Set([
|
|
247
|
+
"mode",
|
|
248
|
+
"maxRetries",
|
|
249
|
+
"retryableCodes",
|
|
250
|
+
"backoff"
|
|
251
|
+
]);
|
|
252
|
+
const BACKOFF_KEYS = new Set([
|
|
253
|
+
"initialDelayMs",
|
|
254
|
+
"maxDelayMs",
|
|
255
|
+
"jitterRatio"
|
|
256
|
+
]);
|
|
257
|
+
function validateKeys(value, allowed, path) {
|
|
258
|
+
for (const key of Object.keys(value)) if (!allowed.has(key)) throw new Error(`${path}: unknown key "${key}"`);
|
|
259
|
+
}
|
|
260
|
+
function resolveBackoff(config, path) {
|
|
261
|
+
if (config !== void 0) validateKeys(config, BACKOFF_KEYS, path);
|
|
262
|
+
const initialDelayMs = config?.initialDelayMs ?? DEFAULT_INITIAL_DELAY_MS;
|
|
263
|
+
const maxDelayMs = config?.maxDelayMs ?? DEFAULT_MAX_DELAY_MS;
|
|
264
|
+
const jitterRatio = config?.jitterRatio ?? DEFAULT_JITTER_RATIO;
|
|
265
|
+
if (!Number.isFinite(initialDelayMs) || initialDelayMs <= 0 || initialDelayMs > MAX_TIMER_DELAY_MS) throw new Error(`${path}.initialDelayMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`);
|
|
266
|
+
if (!Number.isFinite(maxDelayMs) || maxDelayMs <= 0 || maxDelayMs > MAX_TIMER_DELAY_MS) throw new Error(`${path}.maxDelayMs must be a positive finite number no greater than ${MAX_TIMER_DELAY_MS}`);
|
|
267
|
+
if (initialDelayMs > maxDelayMs) throw new Error(`${path}.initialDelayMs must be less than or equal to maxDelayMs`);
|
|
268
|
+
if (!Number.isFinite(jitterRatio) || jitterRatio < 0 || jitterRatio > 1) throw new Error(`${path}.jitterRatio must be between 0 and 1`);
|
|
269
|
+
return Object.freeze({
|
|
270
|
+
initialDelayMs,
|
|
271
|
+
maxDelayMs,
|
|
272
|
+
jitterRatio
|
|
273
|
+
});
|
|
274
|
+
}
|
|
275
|
+
/**
|
|
276
|
+
* Validate, default, and detach one provider-owned retry policy.
|
|
277
|
+
* @param config - optional provider configuration; omission selects normal defaults.
|
|
278
|
+
* @param path - diagnostic path naming the provider config that owns the value.
|
|
279
|
+
* @returns an immutable policy safe to capture in provider registration state.
|
|
280
|
+
*/
|
|
281
|
+
function resolveRetryPolicy(config, path) {
|
|
282
|
+
if (config === void 0) return Object.freeze({
|
|
283
|
+
mode: "normal",
|
|
284
|
+
maxRetries: DEFAULT_MAX_RETRIES,
|
|
285
|
+
retryableCodes: DEFAULT_RETRYABLE_CODES,
|
|
286
|
+
...resolveBackoff(void 0, `${path}.backoff`)
|
|
287
|
+
});
|
|
288
|
+
switch (config.mode) {
|
|
289
|
+
case "normal": {
|
|
290
|
+
validateKeys(config, NORMAL_POLICY_KEYS, path);
|
|
291
|
+
const maxRetries = config.maxRetries ?? DEFAULT_MAX_RETRIES;
|
|
292
|
+
const retryableCodes = config.retryableCodes ?? [...DEFAULT_RETRYABLE_CODES];
|
|
293
|
+
if (!Number.isSafeInteger(maxRetries) || maxRetries < 0) throw new Error(`${path}.maxRetries must be a non-negative safe integer`);
|
|
294
|
+
if (retryableCodes.length === 0) throw new Error(`${path}.retryableCodes must not be empty`);
|
|
295
|
+
if (retryableCodes.some((code) => typeof code !== "string" || code.length === 0)) throw new Error(`${path}.retryableCodes must contain only non-empty strings`);
|
|
296
|
+
if (new Set(retryableCodes).size !== retryableCodes.length) throw new Error(`${path}.retryableCodes must not contain duplicates`);
|
|
297
|
+
return Object.freeze({
|
|
298
|
+
mode: "normal",
|
|
299
|
+
maxRetries,
|
|
300
|
+
retryableCodes: Object.freeze([...retryableCodes]),
|
|
301
|
+
...resolveBackoff(config.backoff, `${path}.backoff`)
|
|
302
|
+
});
|
|
303
|
+
}
|
|
304
|
+
case "always":
|
|
305
|
+
validateKeys(config, ALWAYS_POLICY_KEYS, path);
|
|
306
|
+
return Object.freeze({
|
|
307
|
+
mode: "always",
|
|
308
|
+
...resolveBackoff(config.backoff, `${path}.backoff`)
|
|
309
|
+
});
|
|
310
|
+
default: throw new Error(`${path}.mode must be "normal" or "always"`);
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
//#endregion
|
|
314
|
+
//#region lib/types/call-config.js
|
|
315
|
+
/**
|
|
316
|
+
* Conversation call configuration and freeze utilities. Provider routing,
|
|
317
|
+
* model, reasoning effort, and sampling values are request-header state that
|
|
318
|
+
* can affect cache reuse; request waterfalls replace them and the loop logs
|
|
319
|
+
* changed snapshots instead of allowing silent per-call drift.
|
|
320
|
+
* @module xlaunch-llm/call-config
|
|
321
|
+
*/
|
|
322
|
+
/** Process-local identities of request objects assembled by xlaunch-agent-loop. */
|
|
323
|
+
const AGENT_LOOP_REQUESTS = /* @__PURE__ */ new WeakSet();
|
|
324
|
+
/**
|
|
325
|
+
* Field-wise equality over {@link LlmCallConfig} — the comparison a caller
|
|
326
|
+
* runs to decide whether a proposed configuration is a real change (worth a
|
|
327
|
+
* logged header snapshot) or the held one restated.
|
|
328
|
+
* @param a - one configuration.
|
|
329
|
+
* @param b - the other.
|
|
330
|
+
* @returns whether every field (including the `stop` list, element-wise) matches.
|
|
331
|
+
*/
|
|
332
|
+
function callConfigEquals(a, b) {
|
|
333
|
+
if (a.provider !== b.provider || a.model !== b.model || a.reasoningEffort !== b.reasoningEffort || a.temperature !== b.temperature || a.maxTokens !== b.maxTokens) return false;
|
|
334
|
+
if (a.stop === void 0 || b.stop === void 0) return a.stop === b.stop;
|
|
335
|
+
return a.stop.length === b.stop.length && a.stop.every((s, i) => s === b.stop?.[i]);
|
|
336
|
+
}
|
|
337
|
+
/**
|
|
338
|
+
* Mark one exact request object as assembled by xlaunch-agent-loop.
|
|
339
|
+
* @param request - loop-owned request envelope before LLM dispatch.
|
|
340
|
+
* @returns the same request object marked as created by the process-local agent loop.
|
|
341
|
+
*/
|
|
342
|
+
function markAgentLoopRequest(request) {
|
|
343
|
+
AGENT_LOOP_REQUESTS.add(request);
|
|
344
|
+
return request;
|
|
345
|
+
}
|
|
346
|
+
/**
|
|
347
|
+
* Test whether the exact request object was assembled by xlaunch-agent-loop.
|
|
348
|
+
* @param request - request envelope observed at the LLM waterfall.
|
|
349
|
+
* @returns whether {@link markAgentLoopRequest} recorded this object.
|
|
350
|
+
*/
|
|
351
|
+
function isAgentLoopRequest(request) {
|
|
352
|
+
return AGENT_LOOP_REQUESTS.has(request);
|
|
353
|
+
}
|
|
354
|
+
//#endregion
|
|
355
|
+
//#region lib/types/adapter-failure.js
|
|
356
|
+
/**
|
|
357
|
+
* Normalization for values thrown by a final LLM adapter boundary.
|
|
358
|
+
*
|
|
359
|
+
* @module @xlaunch/llm/adapter-failure
|
|
360
|
+
*/
|
|
361
|
+
/**
|
|
362
|
+
* Detach serializable provider facts from a value thrown by an adapter.
|
|
363
|
+
* @param value - arbitrary value thrown during adapter dispatch or iteration.
|
|
364
|
+
* @returns immutable provider-neutral facts suitable for a terminal finish chunk.
|
|
365
|
+
* @internal
|
|
366
|
+
*/
|
|
367
|
+
function normalizeLlmFailure(value) {
|
|
368
|
+
const error = value instanceof Error ? value : new HarnessError(thrownMessage(value), "UNKNOWN", { cause: value });
|
|
369
|
+
const carried = ownFailureSnapshot(error);
|
|
370
|
+
if (carried !== void 0 && carried.code === ownErrorCode(error)) return carried;
|
|
371
|
+
return Object.freeze({
|
|
372
|
+
message: errorMessage(error),
|
|
373
|
+
code: harnessErrorCode(error)
|
|
374
|
+
});
|
|
375
|
+
}
|
|
376
|
+
/** Render a non-Error throw without letting hostile coercion escape normalization. */
|
|
377
|
+
function thrownMessage(value) {
|
|
378
|
+
try {
|
|
379
|
+
const message = String(value);
|
|
380
|
+
return message.length > 0 ? message : "LLM adapter failed";
|
|
381
|
+
} catch (_hostileThrownValue) {
|
|
382
|
+
return "LLM adapter failed";
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
/** Read a foreign error's own data-backed `code` without invoking accessors. */
|
|
386
|
+
function ownErrorCode(error) {
|
|
387
|
+
try {
|
|
388
|
+
const descriptor = Object.getOwnPropertyDescriptor(error, "code");
|
|
389
|
+
return descriptor !== void 0 && "value" in descriptor ? descriptor.value : void 0;
|
|
390
|
+
} catch (_sdkPropertyTrap) {
|
|
391
|
+
return;
|
|
392
|
+
}
|
|
393
|
+
}
|
|
394
|
+
/** Snapshot an own data property without invoking an SDK-defined accessor. */
|
|
395
|
+
function ownFailureSnapshot(error) {
|
|
396
|
+
try {
|
|
397
|
+
const descriptor = Object.getOwnPropertyDescriptor(error, "failure");
|
|
398
|
+
return descriptor !== void 0 && "value" in descriptor ? failureSnapshot(descriptor.value) : void 0;
|
|
399
|
+
} catch (_sdkPropertyTrap) {
|
|
400
|
+
return;
|
|
401
|
+
}
|
|
402
|
+
}
|
|
403
|
+
/** Validate and detach an arbitrary serializable failure payload. */
|
|
404
|
+
function failureSnapshot(value) {
|
|
405
|
+
if (typeof value !== "object" || value === null) return void 0;
|
|
406
|
+
try {
|
|
407
|
+
const candidate = value;
|
|
408
|
+
const message = candidate.message;
|
|
409
|
+
const code = candidate.code;
|
|
410
|
+
const status = candidate.status;
|
|
411
|
+
const providerRetryAfterMs = candidate.providerRetryAfterMs;
|
|
412
|
+
const requestId = candidate.requestId;
|
|
413
|
+
if (typeof message !== "string" || message.length === 0 || typeof code !== "string" || code.length === 0 || status !== void 0 && (!Number.isInteger(status) || status < 100 || status > 599) || providerRetryAfterMs !== void 0 && (!Number.isFinite(providerRetryAfterMs) || providerRetryAfterMs <= 0) || requestId !== void 0 && (typeof requestId !== "string" || requestId.length === 0)) return void 0;
|
|
414
|
+
return Object.freeze({
|
|
415
|
+
message,
|
|
416
|
+
code,
|
|
417
|
+
...status === void 0 ? {} : { status },
|
|
418
|
+
...providerRetryAfterMs === void 0 ? {} : { providerRetryAfterMs },
|
|
419
|
+
...requestId === void 0 ? {} : { requestId }
|
|
420
|
+
});
|
|
421
|
+
} catch (_sdkFailureGetter) {
|
|
422
|
+
return;
|
|
423
|
+
}
|
|
424
|
+
}
|
|
425
|
+
/** Read an SDK error message without letting an accessor replace the primary failure. */
|
|
426
|
+
function errorMessage(error) {
|
|
427
|
+
try {
|
|
428
|
+
const message = error.message;
|
|
429
|
+
if (typeof message === "string" && message.length > 0) return message;
|
|
430
|
+
} catch (_sdkMessageGetter) {}
|
|
431
|
+
return "LLM adapter failed";
|
|
432
|
+
}
|
|
433
|
+
/** Trust only Harness-owned codes; third-party SDK codes are not our taxonomy. */
|
|
434
|
+
function harnessErrorCode(error) {
|
|
435
|
+
return error instanceof HarnessError ? error.code : "UNKNOWN";
|
|
436
|
+
}
|
|
437
|
+
//#endregion
|
|
438
|
+
//#region lib/types/api-key.js
|
|
439
|
+
/**
|
|
440
|
+
* The one definition of a well-formed provider API key, shared by every
|
|
441
|
+
* adapter that puts one in an HTTP header.
|
|
442
|
+
* @module @xlaunch/llm/api-key
|
|
443
|
+
*/
|
|
444
|
+
/**
|
|
445
|
+
* Characters an HTTP header value carries verbatim and every known provider
|
|
446
|
+
* key uses: printable ASCII, space excluded. A key outside this set cannot
|
|
447
|
+
* reach any provider — `fetch` refuses to build the header — so this is a
|
|
448
|
+
* transport invariant rather than one provider's policy. Latin-1 is excluded
|
|
449
|
+
* deliberately: a header could carry it, but no provider issues it, and
|
|
450
|
+
* admitting it trades a local explained refusal for an opaque 401.
|
|
451
|
+
*/
|
|
452
|
+
const LEGAL_API_KEY = /^[\x21-\x7E]+$/;
|
|
453
|
+
/**
|
|
454
|
+
* Judge one *supplied* API key, trimming surrounding whitespace first.
|
|
455
|
+
*
|
|
456
|
+
* Trimming is silent because a padded key has one unambiguous reading; every
|
|
457
|
+
* other defect is reported. Absence is a configuration state this function
|
|
458
|
+
* never sees — a profile naming no credential authenticates through the
|
|
459
|
+
* provider's own ambient discovery or OAuth — so callers decide whether a
|
|
460
|
+
* value was supplied before asking.
|
|
461
|
+
* @param raw - the key exactly as configured, stored, or typed.
|
|
462
|
+
* @returns the trimmed key, or why it cannot be used.
|
|
463
|
+
*/
|
|
464
|
+
function normalizeApiKey(raw) {
|
|
465
|
+
const value = raw.trim();
|
|
466
|
+
if (value.length === 0) return {
|
|
467
|
+
ok: false,
|
|
468
|
+
reason: "empty"
|
|
469
|
+
};
|
|
470
|
+
if (!LEGAL_API_KEY.test(value)) return {
|
|
471
|
+
ok: false,
|
|
472
|
+
reason: "illegalCharacters"
|
|
473
|
+
};
|
|
474
|
+
return {
|
|
475
|
+
ok: true,
|
|
476
|
+
value
|
|
477
|
+
};
|
|
478
|
+
}
|
|
479
|
+
//#endregion
|
|
480
|
+
//#region lib/types/content.js
|
|
481
|
+
/** Content-block structure helpers. @module @xlaunch/llm/content */
|
|
482
|
+
/**
|
|
483
|
+
* Bridge one attachment provider's host object location into the mounted
|
|
484
|
+
* tool execution world. The consumer supplies the current filesystem
|
|
485
|
+
* provider's mapping without making attachment or LLM definitions depend on it.
|
|
486
|
+
* @param attachments - provider that owns the normalized attachment object.
|
|
487
|
+
* @param mapHostPath - map one absolute host path into the current tool execution world.
|
|
488
|
+
* @param ref - durable normalized attachment reference.
|
|
489
|
+
* @returns a read-only execution-world path, or undefined when either provider exposes no mapping.
|
|
490
|
+
* @throws an attachment error when the durable reference is invalid.
|
|
491
|
+
*/
|
|
492
|
+
function resolveImageAttachmentAccess(attachments, mapHostPath, ref) {
|
|
493
|
+
const hostPath = attachments.imageHostPath(ref);
|
|
494
|
+
if (hostPath === void 0) return void 0;
|
|
495
|
+
const readonlyPath = mapHostPath(hostPath);
|
|
496
|
+
return readonlyPath === void 0 ? void 0 : { readonlyPath };
|
|
497
|
+
}
|
|
498
|
+
function quoted(value) {
|
|
499
|
+
return JSON.stringify(value);
|
|
500
|
+
}
|
|
501
|
+
function imageIdentity(ref) {
|
|
502
|
+
return ref.name === void 0 ? String(ref.attachmentId) : `${quoted(ref.name)} (${ref.attachmentId})`;
|
|
503
|
+
}
|
|
504
|
+
function extension(mediaType) {
|
|
505
|
+
switch (mediaType) {
|
|
506
|
+
case "image/png": return ".png";
|
|
507
|
+
case "image/jpeg": return ".jpg";
|
|
508
|
+
case "image/webp": return ".webp";
|
|
509
|
+
case "image/gif": return ".gif";
|
|
510
|
+
default: return assertNever(mediaType, "image extension");
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
function normalizedAccessText(ref, access) {
|
|
514
|
+
return ` Normalized copy (read-only; may be resized or re-encoded): ${quoted(access.readonlyPath)} (${ref.width}x${ref.height}px, ${ref.mediaType}). Source dimensions, format, and byte size may differ. Copy to a writable path ending in ${extension(ref.mediaType)} before editing.`;
|
|
515
|
+
}
|
|
516
|
+
/**
|
|
517
|
+
* Stable text shown to a model that cannot accept one durable image reference.
|
|
518
|
+
* @param ref - durable normalized attachment omitted from the request.
|
|
519
|
+
* @returns deterministic text-only placeholder.
|
|
520
|
+
*/
|
|
521
|
+
function textOnlyImageText(ref) {
|
|
522
|
+
return `[image omitted because this model accepts text only; attachment sha256:${String(ref.attachmentId).slice(7, 15)}]`;
|
|
523
|
+
}
|
|
524
|
+
/**
|
|
525
|
+
* Stable model-facing handle for one exact request image. Identity comes from
|
|
526
|
+
* the occurrence's own durable reference: request versions are prepared per
|
|
527
|
+
* attachment id, so one shared version may serve occurrences whose display
|
|
528
|
+
* names differ.
|
|
529
|
+
* @param ref - the occurrence's durable normalized attachment.
|
|
530
|
+
* @param version - exact request-image dimensions shown beside the text.
|
|
531
|
+
* @param access - optional path resolved for the current tool execution world.
|
|
532
|
+
* @returns attachment handle and request-image dimensions.
|
|
533
|
+
*/
|
|
534
|
+
function requestImageHandleText(ref, version, access) {
|
|
535
|
+
const preview = `Image ${imageIdentity(ref)}; request preview ${version.width}x${version.height}px.`;
|
|
536
|
+
return access === void 0 ? `${preview} It may be resized or re-encoded; source dimensions, format, and byte size may differ.` : preview + normalizedAccessText(ref, access);
|
|
537
|
+
}
|
|
538
|
+
/**
|
|
539
|
+
* Stable per-image placeholder for a request-limit omission.
|
|
540
|
+
* @param ref - durable normalized attachment omitted from this request.
|
|
541
|
+
* @param access - optional provider-resolved path for model tools.
|
|
542
|
+
* @returns identity, normalized metadata, and the available recovery path.
|
|
543
|
+
*/
|
|
544
|
+
function offloadedImageText(ref, access) {
|
|
545
|
+
const identity = `image omitted to fit request image limits; ${imageIdentity(ref)}.`;
|
|
546
|
+
if (access === void 0) return `[${identity} No local normalized image path is available; ask the user to attach it again if needed.]`;
|
|
547
|
+
return `[${identity}${normalizedAccessText(ref, access)}]`;
|
|
548
|
+
}
|
|
549
|
+
/**
|
|
550
|
+
* True when typed model content contains an image block, walking nested
|
|
551
|
+
* tool-result content. This is the one recursive image walk shared by every
|
|
552
|
+
* image policy (capability gating, text-only serialization, compaction
|
|
553
|
+
* survey), so a consumer cannot silently diverge on nesting depth.
|
|
554
|
+
* @param content - typed model content blocks.
|
|
555
|
+
* @returns whether any nested block is an image.
|
|
556
|
+
*/
|
|
557
|
+
function contentHasImage(content) {
|
|
558
|
+
return content.some((block) => block.type === "image" || block.type === "tool-result" && contentHasImage(block.content));
|
|
559
|
+
}
|
|
560
|
+
/**
|
|
561
|
+
* True when typed model content contains a file block, walking nested
|
|
562
|
+
* tool-result content on the same recursion every file policy shares.
|
|
563
|
+
* @param content - typed model content blocks.
|
|
564
|
+
* @returns whether any nested block is a file.
|
|
565
|
+
*/
|
|
566
|
+
function contentHasFile(content) {
|
|
567
|
+
return content.some((block) => block.type === "file" || block.type === "tool-result" && contentHasFile(block.content));
|
|
568
|
+
}
|
|
569
|
+
/**
|
|
570
|
+
* Stable model-facing handle for one durable file reference: the address of
|
|
571
|
+
* the verbatim stored copy and the instruction to read it on demand. This is
|
|
572
|
+
* the only representation a provider ever receives for a file.
|
|
573
|
+
* @param ref - durable verbatim file reference.
|
|
574
|
+
* @param readonlyPath - execution-world path of the stored copy, when resolvable.
|
|
575
|
+
* @returns deterministic handle text naming the file, its size, and its address.
|
|
576
|
+
*/
|
|
577
|
+
function fileHandleText(ref, readonlyPath) {
|
|
578
|
+
const digest = String(ref.attachmentId).slice(7, 15);
|
|
579
|
+
const identity = `File ${quoted(ref.name)} (${ref.bytes} bytes, sha256:${digest})`;
|
|
580
|
+
if (readonlyPath === void 0) return `[${identity} was uploaded, but the current execution environment cannot access a readable path. Report that limitation if its contents are needed; do not claim to have read it.]`;
|
|
581
|
+
return `[${identity}: verbatim read-only copy saved at ${quoted(readonlyPath)}. Read that path with your file tools when its contents are needed; copy it to a writable location before modifying it. When delegating file work, include this saved path in the delegation prompt; only subagents sharing this execution environment can read it.]`;
|
|
582
|
+
}
|
|
583
|
+
/** Replace every file occurrence, including nested tool results, with handle text. */
|
|
584
|
+
function replaceFilesWithHandles(blocks, resolvePath) {
|
|
585
|
+
let next;
|
|
586
|
+
for (const [index, block] of blocks.entries()) {
|
|
587
|
+
if (block.type === "file") {
|
|
588
|
+
next ??= blocks.slice(0, index);
|
|
589
|
+
next.push({
|
|
590
|
+
type: "text",
|
|
591
|
+
text: fileHandleText(block.attachment, resolvePath(block.attachment))
|
|
592
|
+
});
|
|
593
|
+
continue;
|
|
594
|
+
}
|
|
595
|
+
if (block.type === "tool-result") {
|
|
596
|
+
const content = replaceFilesWithHandles(block.content, resolvePath);
|
|
597
|
+
if (content !== block.content) {
|
|
598
|
+
next ??= blocks.slice(0, index);
|
|
599
|
+
next.push({
|
|
600
|
+
...block,
|
|
601
|
+
content
|
|
602
|
+
});
|
|
603
|
+
continue;
|
|
604
|
+
}
|
|
605
|
+
}
|
|
606
|
+
next?.push(block);
|
|
607
|
+
}
|
|
608
|
+
return next ?? blocks;
|
|
609
|
+
}
|
|
610
|
+
/**
|
|
611
|
+
* Project durable file history into deterministic handle text for every model
|
|
612
|
+
* route. Unlike images, no provider receives file blocks natively, so this
|
|
613
|
+
* projection is unconditional in request assembly.
|
|
614
|
+
* @param messages - complete request history.
|
|
615
|
+
* @param resolvePath - resolve one reference's current execution-world read path.
|
|
616
|
+
* @returns the original list without files, otherwise shallow message copies with handle text.
|
|
617
|
+
*/
|
|
618
|
+
function projectFilesToText(messages, resolvePath) {
|
|
619
|
+
if (!messages.some((message) => contentHasFile(message.content))) return messages;
|
|
620
|
+
return messages.map((message) => {
|
|
621
|
+
const content = replaceFilesWithHandles(message.content, resolvePath);
|
|
622
|
+
return content === message.content ? message : {
|
|
623
|
+
...message,
|
|
624
|
+
content
|
|
625
|
+
};
|
|
626
|
+
});
|
|
627
|
+
}
|
|
628
|
+
/** Base64 length of raw image bytes, including padding. */
|
|
629
|
+
function base64Length(bytes) {
|
|
630
|
+
return Math.ceil(bytes / 3) * 4;
|
|
631
|
+
}
|
|
632
|
+
/** Collect represented image lengths in request and nested-block order. */
|
|
633
|
+
function collectImageLengths(blocks, lengths, policy) {
|
|
634
|
+
for (const block of blocks) if (block.type === "image") {
|
|
635
|
+
const bytes = policy.byteLength === void 0 ? block.attachment.bytes : policy.byteLength(block.attachment);
|
|
636
|
+
lengths.push(policy.representation === "base64" ? base64Length(bytes) : bytes);
|
|
637
|
+
} else if (block.type === "tool-result") collectImageLengths(block.content, lengths, policy);
|
|
638
|
+
}
|
|
639
|
+
/** Replace the first `remaining.count` image occurrences without mutating durable messages. */
|
|
640
|
+
function replaceOldestImages(blocks, remaining, placeholder) {
|
|
641
|
+
let next;
|
|
642
|
+
for (const [index, block] of blocks.entries()) {
|
|
643
|
+
if (block.type === "image" && remaining.count > 0) {
|
|
644
|
+
remaining.count -= 1;
|
|
645
|
+
next ??= blocks.slice(0, index);
|
|
646
|
+
next.push({
|
|
647
|
+
type: "text",
|
|
648
|
+
text: placeholder(block.attachment)
|
|
649
|
+
});
|
|
650
|
+
continue;
|
|
651
|
+
}
|
|
652
|
+
if (block.type === "tool-result") {
|
|
653
|
+
const content = replaceOldestImages(block.content, remaining, placeholder);
|
|
654
|
+
if (content !== block.content) {
|
|
655
|
+
next ??= blocks.slice(0, index);
|
|
656
|
+
next.push({
|
|
657
|
+
...block,
|
|
658
|
+
content
|
|
659
|
+
});
|
|
660
|
+
continue;
|
|
661
|
+
}
|
|
662
|
+
}
|
|
663
|
+
next?.push(block);
|
|
664
|
+
}
|
|
665
|
+
return next ?? blocks;
|
|
666
|
+
}
|
|
667
|
+
/** Replace every image occurrence, including nested tool results, for a text-only model. */
|
|
668
|
+
function replaceImagesForTextModel(blocks) {
|
|
669
|
+
let next;
|
|
670
|
+
for (const [index, block] of blocks.entries()) {
|
|
671
|
+
if (block.type === "image") {
|
|
672
|
+
next ??= blocks.slice(0, index);
|
|
673
|
+
next.push({
|
|
674
|
+
type: "text",
|
|
675
|
+
text: textOnlyImageText(block.attachment)
|
|
676
|
+
});
|
|
677
|
+
continue;
|
|
678
|
+
}
|
|
679
|
+
if (block.type === "tool-result") {
|
|
680
|
+
const content = replaceImagesForTextModel(block.content);
|
|
681
|
+
if (content !== block.content) {
|
|
682
|
+
next ??= blocks.slice(0, index);
|
|
683
|
+
next.push({
|
|
684
|
+
...block,
|
|
685
|
+
content
|
|
686
|
+
});
|
|
687
|
+
continue;
|
|
688
|
+
}
|
|
689
|
+
}
|
|
690
|
+
next?.push(block);
|
|
691
|
+
}
|
|
692
|
+
return next ?? blocks;
|
|
693
|
+
}
|
|
694
|
+
/**
|
|
695
|
+
* Project durable image history into deterministic text for an exact text-only model.
|
|
696
|
+
* @param messages - complete request history.
|
|
697
|
+
* @returns the original list without images, otherwise shallow message copies with stable placeholders.
|
|
698
|
+
*/
|
|
699
|
+
function projectImagesForTextModel(messages) {
|
|
700
|
+
if (!messages.some((message) => contentHasImage(message.content))) return messages;
|
|
701
|
+
return messages.map((message) => {
|
|
702
|
+
const content = replaceImagesForTextModel(message.content);
|
|
703
|
+
return content === message.content ? message : {
|
|
704
|
+
...message,
|
|
705
|
+
content
|
|
706
|
+
};
|
|
707
|
+
});
|
|
708
|
+
}
|
|
709
|
+
/**
|
|
710
|
+
* Number of oldest image occurrences one request projection removes, in whole
|
|
711
|
+
* count and byte quanta, once a route budget is exceeded. The result depends
|
|
712
|
+
* only on the represented lengths, so provider request pricing reproduces the
|
|
713
|
+
* exact serialization decision without building the projected messages.
|
|
714
|
+
* @param lengths - represented byte length of every occurrence, in request order.
|
|
715
|
+
* @param policy - count/byte budgets and removal quanta; unbounded when absent.
|
|
716
|
+
* @returns how many leading occurrences the projection replaces with placeholders.
|
|
717
|
+
*/
|
|
718
|
+
function offloadedImagePrefixCount(lengths, policy) {
|
|
719
|
+
const total = lengths.reduce((sum, bytes) => sum + bytes, 0);
|
|
720
|
+
const excessCount = policy.maxImages === void 0 ? 0 : Math.max(0, lengths.length - policy.maxImages);
|
|
721
|
+
const excessBytes = policy.maxBytes === void 0 ? 0 : Math.max(0, total - policy.maxBytes);
|
|
722
|
+
if (excessCount === 0 && excessBytes === 0) return 0;
|
|
723
|
+
const countQuantum = policy.countQuantum ?? 1;
|
|
724
|
+
const byteQuantum = policy.byteQuantum ?? 1;
|
|
725
|
+
const removeCount = excessCount === 0 ? 0 : Math.ceil(excessCount / countQuantum) * countQuantum;
|
|
726
|
+
const removeBytes = excessBytes === 0 ? 0 : Math.ceil(excessBytes / byteQuantum) * byteQuantum;
|
|
727
|
+
let count = 0;
|
|
728
|
+
let removedBytes = 0;
|
|
729
|
+
for (const imageBytes of lengths) {
|
|
730
|
+
if (count >= removeCount && (removeBytes === 0 || (byteQuantum === 1 ? removedBytes >= removeBytes : removedBytes > removeBytes))) break;
|
|
731
|
+
removedBytes += imageBytes;
|
|
732
|
+
count += 1;
|
|
733
|
+
}
|
|
734
|
+
return count;
|
|
735
|
+
}
|
|
736
|
+
/**
|
|
737
|
+
* Return a deterministic transient projection whose oldest images are replaced
|
|
738
|
+
* in whole count and byte quanta after a route budget is exceeded. The target
|
|
739
|
+
* depends only on complete durable history: at 129 one-megabyte images under
|
|
740
|
+
* a 128 MiB bound with a 64 MiB quantum, the oldest 65 images are removed so
|
|
741
|
+
* 64 MiB remain; that removed prefix stays fixed until total history exceeds
|
|
742
|
+
* 192 MiB.
|
|
743
|
+
* @param messages - complete request history, oldest first.
|
|
744
|
+
* @param policy - route representation, budgets, and removal quanta.
|
|
745
|
+
* @returns original messages below both bounds, otherwise shallow copies with deterministic placeholders.
|
|
746
|
+
*/
|
|
747
|
+
function offloadRequestImagesWithPolicy(messages, policy) {
|
|
748
|
+
const lengths = [];
|
|
749
|
+
for (const message of messages) collectImageLengths(message.content, lengths, policy);
|
|
750
|
+
const count = offloadedImagePrefixCount(lengths, policy);
|
|
751
|
+
if (count === 0) return messages;
|
|
752
|
+
const remaining = { count };
|
|
753
|
+
return messages.map((message) => {
|
|
754
|
+
const content = replaceOldestImages(message.content, remaining, policy.placeholder);
|
|
755
|
+
return content === message.content ? message : {
|
|
756
|
+
...message,
|
|
757
|
+
content
|
|
758
|
+
};
|
|
759
|
+
});
|
|
760
|
+
}
|
|
761
|
+
//#endregion
|
|
762
|
+
//#region lib/types/attribution.js
|
|
763
|
+
/**
|
|
764
|
+
* Centralize the non-secret product identity every provider request sends as `User-Agent`, keeping
|
|
765
|
+
* adapters from drifting. See
|
|
766
|
+
* `.agents/notes/implemented/architecture/2026-06-21-mandatory-app-attribution-headers.md`.
|
|
767
|
+
*
|
|
768
|
+
* App-attribution vocabulary for provider requests.
|
|
769
|
+
* @module @xlaunch/llm/attribution
|
|
770
|
+
*/
|
|
771
|
+
const { version } = createRequire(import.meta.url)("../package.json");
|
|
772
|
+
/**
|
|
773
|
+
* The harness's own identity: the default every adapter sends. Deployments
|
|
774
|
+
* that need a white-label identity pass their own {@link AppIdentity} to
|
|
775
|
+
* {@link attributionHeaders} — omission falls back to this default; nothing
|
|
776
|
+
* can suppress attribution entirely.
|
|
777
|
+
*/
|
|
778
|
+
const APP_IDENTITY = {
|
|
779
|
+
product: "xlaunch",
|
|
780
|
+
version,
|
|
781
|
+
url: "https://github.com/Northlatch-Labs-LLC/xlaunch-agent"
|
|
782
|
+
};
|
|
783
|
+
/**
|
|
784
|
+
* The standard `User-Agent` value: `product/version (+url)`. The
|
|
785
|
+
* parenthesized `+url` comment is the conventional self-identification form
|
|
786
|
+
* (RFC 9110 §10.1.5 product + comment syntax).
|
|
787
|
+
* @param identity - the identity to render; defaults to {@link APP_IDENTITY}.
|
|
788
|
+
* @returns the ready-to-send header value.
|
|
789
|
+
*/
|
|
790
|
+
function userAgent(identity = APP_IDENTITY) {
|
|
791
|
+
return `${identity.product}/${identity.version} (+${identity.url})`;
|
|
792
|
+
}
|
|
793
|
+
/**
|
|
794
|
+
* Build the attribution headers an adapter must send on every provider
|
|
795
|
+
* request. Header names are lowercase (HTTP field names are case-insensitive
|
|
796
|
+
* on the wire).
|
|
797
|
+
* @param identity - the identity to send; defaults to {@link APP_IDENTITY} — omission cannot suppress attribution.
|
|
798
|
+
* @returns headers to merge into the provider request (currently just `user-agent`).
|
|
799
|
+
*/
|
|
800
|
+
function attributionHeaders(identity = APP_IDENTITY) {
|
|
801
|
+
return { "user-agent": userAgent(identity) };
|
|
802
|
+
}
|
|
803
|
+
//#endregion
|
|
804
|
+
//#region lib/types/brand.js
|
|
805
|
+
/**
|
|
806
|
+
* xlaunch-llm's owned branded ids: tool-call correlation and provider request
|
|
807
|
+
* diagnostics.
|
|
808
|
+
*
|
|
809
|
+
* The `Branded<B>` primitive and stateless constructor live in
|
|
810
|
+
* `@xlaunch/brand` so every owner of a cross-boundary id can brand it
|
|
811
|
+
* without depending on xlaunch-llm; see that package's README for the
|
|
812
|
+
* nominal-typing policy.
|
|
813
|
+
*
|
|
814
|
+
* @module @xlaunch/llm/brand
|
|
815
|
+
*/
|
|
816
|
+
/**
|
|
817
|
+
* Brand a message identifier.
|
|
818
|
+
* @param id - the opaque message identifier.
|
|
819
|
+
* @returns the same string with the message-id brand.
|
|
820
|
+
*/
|
|
821
|
+
function MessageId(id) {
|
|
822
|
+
return brandString(id);
|
|
823
|
+
}
|
|
824
|
+
/**
|
|
825
|
+
* Brand a string as a {@link ToolCallId}.
|
|
826
|
+
* @param id - the provider-issued or synthesized call id.
|
|
827
|
+
* @returns the same string with the tool-call-id brand.
|
|
828
|
+
*/
|
|
829
|
+
function ToolCallId(id) {
|
|
830
|
+
return brandString(id);
|
|
831
|
+
}
|
|
832
|
+
/**
|
|
833
|
+
* Brand a provider-issued request identifier.
|
|
834
|
+
* @param id - the opaque provider-issued string.
|
|
835
|
+
* @returns the same string, branded; no validation is performed.
|
|
836
|
+
*/
|
|
837
|
+
function ProviderRequestId(id) {
|
|
838
|
+
return brandString(id);
|
|
839
|
+
}
|
|
840
|
+
/**
|
|
841
|
+
* Brand one loop-owned streaming attempt identifier.
|
|
842
|
+
* @param id - the opaque Agent-lifecycle-local identifier.
|
|
843
|
+
* @returns the same string with the attempt-id brand.
|
|
844
|
+
*/
|
|
845
|
+
function LlmAttemptId(id) {
|
|
846
|
+
return brandString(id);
|
|
847
|
+
}
|
|
848
|
+
/**
|
|
849
|
+
* Brand an adapter-owned reasoning-effort identifier.
|
|
850
|
+
* @param id - the opaque identifier exposed by one model capability.
|
|
851
|
+
* @returns the same string, branded; no validation is performed.
|
|
852
|
+
*/
|
|
853
|
+
function ReasoningEffortId(id) {
|
|
854
|
+
return brandString(id);
|
|
855
|
+
}
|
|
856
|
+
//#endregion
|
|
857
|
+
//#region lib/types/assembler.js
|
|
858
|
+
/**
|
|
859
|
+
* Incremental chunk-to-message assembler. This is the single canonical assembly
|
|
860
|
+
* algorithm used by the agent loop to build an assistant message from a chunk
|
|
861
|
+
* stream while logging the raw chunks for replay fidelity.
|
|
862
|
+
*
|
|
863
|
+
* @module @xlaunch/llm/assembler
|
|
864
|
+
*/
|
|
865
|
+
/**
|
|
866
|
+
* Incrementally assembles raw {@link StreamChunk}s into complete
|
|
867
|
+
* {@link ContentBlock}s and a final assistant {@link Message}.
|
|
868
|
+
*
|
|
869
|
+
* The agent loop feeds it while logging raw chunks for replay fidelity, then
|
|
870
|
+
* reads `blocks()` / `message()` / `usage` / `finish` once the stream ends,
|
|
871
|
+
* or `interruptedBlocks()` when cancellation cut the stream short.
|
|
872
|
+
*
|
|
873
|
+
* Tolerant of delta-only protocols (no block-start/end); deltas arriving for
|
|
874
|
+
* an index already closed by `block-end` are ignored (malformed stream) so a
|
|
875
|
+
* misbehaving adapter cannot grow memory or corrupt a completed block.
|
|
876
|
+
*/
|
|
877
|
+
var BlockAssembler = class {
|
|
878
|
+
partials = /* @__PURE__ */ new Map();
|
|
879
|
+
order = [];
|
|
880
|
+
_usage;
|
|
881
|
+
_finish;
|
|
882
|
+
_replayState;
|
|
883
|
+
/**
|
|
884
|
+
* Feed one chunk into the assembly state.
|
|
885
|
+
* @param chunk - the next raw chunk, in stream order.
|
|
886
|
+
*/
|
|
887
|
+
push(chunk) {
|
|
888
|
+
switch (chunk.type) {
|
|
889
|
+
case "block-start":
|
|
890
|
+
if (!this.partials.has(chunk.index)) {
|
|
891
|
+
this.order.push(chunk.index);
|
|
892
|
+
this.partials.set(chunk.index, {
|
|
893
|
+
blockType: chunk.blockType,
|
|
894
|
+
text: "",
|
|
895
|
+
toolCallArguments: ""
|
|
896
|
+
});
|
|
897
|
+
}
|
|
898
|
+
return;
|
|
899
|
+
case "text-delta":
|
|
900
|
+
case "reasoning-delta": {
|
|
901
|
+
const partial = this.ensure(chunk.index, chunk.type === "text-delta" ? "text" : "reasoning");
|
|
902
|
+
if (partial.block) return;
|
|
903
|
+
partial.text += chunk.text;
|
|
904
|
+
return;
|
|
905
|
+
}
|
|
906
|
+
case "tool-call-delta": {
|
|
907
|
+
const partial = this.ensure(chunk.index, "tool-call");
|
|
908
|
+
if (partial.block) return;
|
|
909
|
+
partial.toolCallId = chunk.id;
|
|
910
|
+
if (chunk.name) partial.toolCallName = chunk.name;
|
|
911
|
+
partial.toolCallArguments += chunk.argumentsDelta;
|
|
912
|
+
return;
|
|
913
|
+
}
|
|
914
|
+
case "block-end": {
|
|
915
|
+
const partial = this.ensure(chunk.index, chunk.block.type);
|
|
916
|
+
if (partial.block) return;
|
|
917
|
+
partial.block = chunk.block;
|
|
918
|
+
return;
|
|
919
|
+
}
|
|
920
|
+
case "usage":
|
|
921
|
+
this._usage = chunk.usage;
|
|
922
|
+
return;
|
|
923
|
+
case "finish":
|
|
924
|
+
this._finish = chunk.reason;
|
|
925
|
+
this._replayState = chunk.replayState;
|
|
926
|
+
return;
|
|
927
|
+
default: return assertNever(chunk, "BlockAssembler.push");
|
|
928
|
+
}
|
|
929
|
+
}
|
|
930
|
+
ensure(index, blockType) {
|
|
931
|
+
let partial = this.partials.get(index);
|
|
932
|
+
if (!partial) {
|
|
933
|
+
partial = {
|
|
934
|
+
blockType,
|
|
935
|
+
text: "",
|
|
936
|
+
toolCallArguments: ""
|
|
937
|
+
};
|
|
938
|
+
this.partials.set(index, partial);
|
|
939
|
+
this.order.push(index);
|
|
940
|
+
}
|
|
941
|
+
return partial;
|
|
942
|
+
}
|
|
943
|
+
assemble(partial, index) {
|
|
944
|
+
if (partial.block) return partial.block;
|
|
945
|
+
switch (partial.blockType) {
|
|
946
|
+
case "text": return {
|
|
947
|
+
type: "text",
|
|
948
|
+
text: partial.text
|
|
949
|
+
};
|
|
950
|
+
case "reasoning": return {
|
|
951
|
+
type: "reasoning",
|
|
952
|
+
text: partial.text
|
|
953
|
+
};
|
|
954
|
+
case "tool-call": return {
|
|
955
|
+
type: "tool-call",
|
|
956
|
+
id: partial.toolCallId ?? brandString(`call-${index}`),
|
|
957
|
+
name: partial.toolCallName ?? "",
|
|
958
|
+
arguments: partial.toolCallArguments
|
|
959
|
+
};
|
|
960
|
+
default: throw new Error(`cannot assemble incomplete block of type "${partial.blockType}"`);
|
|
961
|
+
}
|
|
962
|
+
}
|
|
963
|
+
/** Invariant accessor: every index in `order` has a partial. */
|
|
964
|
+
mustGet(index) {
|
|
965
|
+
const partial = this.partials.get(index);
|
|
966
|
+
if (!partial) throw new Error(`BlockAssembler invariant violated: no partial for index ${index}`);
|
|
967
|
+
return partial;
|
|
968
|
+
}
|
|
969
|
+
/**
|
|
970
|
+
* The one shared keep/drop decision over all seen blocks: max-token
|
|
971
|
+
* truncation drops tool calls that cannot be executed safely. Emitted blocks
|
|
972
|
+
* and replay metadata both derive from this result, so they cannot disagree.
|
|
973
|
+
*/
|
|
974
|
+
assembled() {
|
|
975
|
+
const all = this.order.map((index) => this.assemble(this.mustGet(index), index));
|
|
976
|
+
const kept = this.finish.kind === "max-tokens" ? all.map((block) => block.type !== "tool-call") : void 0;
|
|
977
|
+
const blocks = kept === void 0 ? all : all.filter((_, position) => kept[position]);
|
|
978
|
+
const envelope = this._replayState;
|
|
979
|
+
if (envelope?.blocks === void 0) return {
|
|
980
|
+
blocks,
|
|
981
|
+
replay: envelope
|
|
982
|
+
};
|
|
983
|
+
if (envelope.blocks.length !== all.length) return {
|
|
984
|
+
blocks,
|
|
985
|
+
replay: void 0
|
|
986
|
+
};
|
|
987
|
+
return {
|
|
988
|
+
blocks,
|
|
989
|
+
replay: kept === void 0 || blocks.length === all.length ? envelope : {
|
|
990
|
+
response: envelope.response,
|
|
991
|
+
blocks: envelope.blocks.filter((_, position) => kept[position])
|
|
992
|
+
}
|
|
993
|
+
};
|
|
994
|
+
}
|
|
995
|
+
/**
|
|
996
|
+
* Assemble all blocks seen so far, in stream order.
|
|
997
|
+
* @returns one block per seen index, except that max-token truncation drops
|
|
998
|
+
* tool calls that cannot be executed safely; an open block assembles from
|
|
999
|
+
* its accumulated deltas (an unknown block type never closed by `block-end` throws).
|
|
1000
|
+
*/
|
|
1001
|
+
blocks() {
|
|
1002
|
+
return this.assembled().blocks;
|
|
1003
|
+
}
|
|
1004
|
+
/**
|
|
1005
|
+
* Assemble the prefix an interrupted stream can safely finalize: closed and
|
|
1006
|
+
* open text/reasoning blocks with non-whitespace content, in stream order.
|
|
1007
|
+
* Tool calls are omitted because interruption precedes dispatch; retaining
|
|
1008
|
+
* one would require a fabricated result. Open unknown blocks are also omitted.
|
|
1009
|
+
* @returns the kept blocks; empty when nothing streamed before the interruption.
|
|
1010
|
+
*/
|
|
1011
|
+
interruptedBlocks() {
|
|
1012
|
+
return this.order.map((index) => {
|
|
1013
|
+
const partial = this.mustGet(index);
|
|
1014
|
+
const type = partial.block?.type ?? partial.blockType;
|
|
1015
|
+
if (type !== "text" && type !== "reasoning") return void 0;
|
|
1016
|
+
return this.assemble(partial, index);
|
|
1017
|
+
}).filter((block) => (block?.type === "text" || block?.type === "reasoning") && block.text.trim() !== "");
|
|
1018
|
+
}
|
|
1019
|
+
/** Usage from the `usage` chunk; undefined until one arrives. */
|
|
1020
|
+
get usage() {
|
|
1021
|
+
return this._usage;
|
|
1022
|
+
}
|
|
1023
|
+
/** Finish reason from the `finish` chunk; `{kind: 'stop'}` when the stream ended without one. */
|
|
1024
|
+
get finish() {
|
|
1025
|
+
return this._finish ?? { kind: "stop" };
|
|
1026
|
+
}
|
|
1027
|
+
/**
|
|
1028
|
+
* Replay metadata from the terminal finish chunk, if any, with per-block
|
|
1029
|
+
* entries pruned in step with {@link blocks}. Undefined when the envelope's
|
|
1030
|
+
* entries do not align with the emitted blocks.
|
|
1031
|
+
*/
|
|
1032
|
+
get replayState() {
|
|
1033
|
+
return this.assembled().replay;
|
|
1034
|
+
}
|
|
1035
|
+
/**
|
|
1036
|
+
* The assembled assistant message.
|
|
1037
|
+
* @param source - producer attribution for the assembled message.
|
|
1038
|
+
* @returns a frozen assistant-role message over `blocks()` (same open-block assembly rules).
|
|
1039
|
+
*/
|
|
1040
|
+
message(source = {
|
|
1041
|
+
kind: "plugin",
|
|
1042
|
+
plugin: "xlaunch-llm/assembler"
|
|
1043
|
+
}) {
|
|
1044
|
+
return createMessage({
|
|
1045
|
+
role: "assistant",
|
|
1046
|
+
content: this.blocks(),
|
|
1047
|
+
source
|
|
1048
|
+
});
|
|
1049
|
+
}
|
|
1050
|
+
};
|
|
1051
|
+
//#endregion
|
|
1052
|
+
//#region lib/types/assistant-stream.js
|
|
1053
|
+
/**
|
|
1054
|
+
* Lossless compact representation of one model-stream attempt, plus record-level
|
|
1055
|
+
* readers that answer common consumer questions without materializing members.
|
|
1056
|
+
* Readers trust the static record type; expandAssistantStream is the validating
|
|
1057
|
+
* path for records read at a durable boundary.
|
|
1058
|
+
*/
|
|
1059
|
+
function safeTime(value) {
|
|
1060
|
+
if (!Number.isSafeInteger(value)) throw new TypeError(`Assistant stream time must be a safe integer, got ${String(value)}`);
|
|
1061
|
+
return value;
|
|
1062
|
+
}
|
|
1063
|
+
function safeIndex(value, label) {
|
|
1064
|
+
if (!Number.isSafeInteger(value) || value < 0 || Object.is(value, -0)) throw new TypeError(`${label} index must be a non-negative safe integer`);
|
|
1065
|
+
return value;
|
|
1066
|
+
}
|
|
1067
|
+
function snapshotChunk(chunk) {
|
|
1068
|
+
const snapshot = snapshotJsonValue(chunk);
|
|
1069
|
+
if (snapshot === void 0) throw new TypeError("Assistant stream chunk must be losslessly JSON-serializable");
|
|
1070
|
+
return snapshot;
|
|
1071
|
+
}
|
|
1072
|
+
function safeGap(previous, next) {
|
|
1073
|
+
const gap = next - previous;
|
|
1074
|
+
return Number.isSafeInteger(gap) && previous + gap === next ? gap : void 0;
|
|
1075
|
+
}
|
|
1076
|
+
/** Incrementally compacts one attempt without retaining a second raw-chunk list. */
|
|
1077
|
+
var AssistantStreamAccumulator = class {
|
|
1078
|
+
records = [];
|
|
1079
|
+
/**
|
|
1080
|
+
* Add one timed chunk to the compact attempt stream.
|
|
1081
|
+
* @param value - model chunk and its original Session timestamp.
|
|
1082
|
+
* @returns a detached immutable copy for assembly and live publication.
|
|
1083
|
+
*/
|
|
1084
|
+
push(value) {
|
|
1085
|
+
const time = safeTime(value.time);
|
|
1086
|
+
const chunk = snapshotChunk(value.chunk);
|
|
1087
|
+
const timed = deepFreeze({
|
|
1088
|
+
time,
|
|
1089
|
+
chunk
|
|
1090
|
+
});
|
|
1091
|
+
const previous = this.records.at(-1);
|
|
1092
|
+
switch (chunk.type) {
|
|
1093
|
+
case "text-delta":
|
|
1094
|
+
case "reasoning-delta": {
|
|
1095
|
+
safeIndex(chunk.index, chunk.type);
|
|
1096
|
+
if (typeof chunk.text !== "string") throw new TypeError(`${chunk.type} text must be a string`);
|
|
1097
|
+
const type = chunk.type === "text-delta" ? "text-chunks" : "reasoning-chunks";
|
|
1098
|
+
const gap = previous !== void 0 && previous.type === type ? safeGap(previous.lastTime, time) : void 0;
|
|
1099
|
+
if (previous !== void 0 && previous.type === type && previous.index === chunk.index && gap !== void 0) {
|
|
1100
|
+
previous.dt.push(gap);
|
|
1101
|
+
previous.texts.push(chunk.text);
|
|
1102
|
+
previous.lastTime = time;
|
|
1103
|
+
} else this.records.push({
|
|
1104
|
+
type,
|
|
1105
|
+
time0: time,
|
|
1106
|
+
index: chunk.index,
|
|
1107
|
+
dt: [],
|
|
1108
|
+
texts: [chunk.text],
|
|
1109
|
+
lastTime: time
|
|
1110
|
+
});
|
|
1111
|
+
return timed;
|
|
1112
|
+
}
|
|
1113
|
+
case "tool-call-delta": {
|
|
1114
|
+
safeIndex(chunk.index, chunk.type);
|
|
1115
|
+
if (typeof chunk.id !== "string") throw new TypeError("tool-call-delta id must be a string");
|
|
1116
|
+
if (Object.hasOwn(chunk, "name") && typeof chunk.name !== "string") throw new TypeError("tool-call-delta name must be a string");
|
|
1117
|
+
if (typeof chunk.argumentsDelta !== "string") throw new TypeError("tool-call-delta argumentsDelta must be a string");
|
|
1118
|
+
if (chunk.id.length === 0 || chunk.name === "") {
|
|
1119
|
+
this.records.push({
|
|
1120
|
+
type: "chunk",
|
|
1121
|
+
time,
|
|
1122
|
+
chunk
|
|
1123
|
+
});
|
|
1124
|
+
return timed;
|
|
1125
|
+
}
|
|
1126
|
+
const gap = previous?.type === "tool-call-chunks" ? safeGap(previous.lastTime, time) : void 0;
|
|
1127
|
+
const sameName = previous?.type === "tool-call-chunks" && Object.hasOwn(previous, "name") === Object.hasOwn(chunk, "name") && previous.name === chunk.name;
|
|
1128
|
+
if (previous?.type === "tool-call-chunks" && previous.index === chunk.index && previous.id === chunk.id && sameName && gap !== void 0) {
|
|
1129
|
+
previous.dt.push(gap);
|
|
1130
|
+
previous.args.push(chunk.argumentsDelta);
|
|
1131
|
+
previous.lastTime = time;
|
|
1132
|
+
} else this.records.push({
|
|
1133
|
+
type: "tool-call-chunks",
|
|
1134
|
+
time0: time,
|
|
1135
|
+
index: chunk.index,
|
|
1136
|
+
dt: [],
|
|
1137
|
+
id: chunk.id,
|
|
1138
|
+
...Object.hasOwn(chunk, "name") ? { name: chunk.name } : {},
|
|
1139
|
+
args: [chunk.argumentsDelta],
|
|
1140
|
+
lastTime: time
|
|
1141
|
+
});
|
|
1142
|
+
return timed;
|
|
1143
|
+
}
|
|
1144
|
+
case "block-start":
|
|
1145
|
+
case "block-end":
|
|
1146
|
+
case "usage":
|
|
1147
|
+
case "finish":
|
|
1148
|
+
this.records.push({
|
|
1149
|
+
type: "chunk",
|
|
1150
|
+
time,
|
|
1151
|
+
chunk
|
|
1152
|
+
});
|
|
1153
|
+
return timed;
|
|
1154
|
+
default: return assertNever(chunk, "AssistantStreamAccumulator.push");
|
|
1155
|
+
}
|
|
1156
|
+
}
|
|
1157
|
+
/**
|
|
1158
|
+
* Return the current compact attempt stream.
|
|
1159
|
+
* @returns a detached immutable record list suitable for a durable event.
|
|
1160
|
+
*/
|
|
1161
|
+
snapshot() {
|
|
1162
|
+
return deepFreeze(this.records.map((record) => {
|
|
1163
|
+
if (record.type === "chunk") return { ...record };
|
|
1164
|
+
const { lastTime: _lastTime, ...durable } = record;
|
|
1165
|
+
if (durable.type === "tool-call-chunks") return {
|
|
1166
|
+
...durable,
|
|
1167
|
+
dt: [...durable.dt],
|
|
1168
|
+
args: [...durable.args]
|
|
1169
|
+
};
|
|
1170
|
+
return {
|
|
1171
|
+
...durable,
|
|
1172
|
+
dt: [...durable.dt],
|
|
1173
|
+
texts: [...durable.texts]
|
|
1174
|
+
};
|
|
1175
|
+
}));
|
|
1176
|
+
}
|
|
1177
|
+
};
|
|
1178
|
+
/**
|
|
1179
|
+
* Expand compact records into the exact timed chunk sequence.
|
|
1180
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1181
|
+
* @returns detached timed chunks with every original delta boundary preserved.
|
|
1182
|
+
* @throws {TypeError} when a record or reconstructed timestamp is invalid.
|
|
1183
|
+
*/
|
|
1184
|
+
function expandAssistantStream(stream) {
|
|
1185
|
+
const chunks = [];
|
|
1186
|
+
for (const candidate of stream) {
|
|
1187
|
+
const record = validateRecord(candidate);
|
|
1188
|
+
if (record.type === "chunk") {
|
|
1189
|
+
chunks.push({
|
|
1190
|
+
time: record.time,
|
|
1191
|
+
chunk: record.chunk
|
|
1192
|
+
});
|
|
1193
|
+
continue;
|
|
1194
|
+
}
|
|
1195
|
+
const members = record.type === "tool-call-chunks" ? record.args : record.texts;
|
|
1196
|
+
let time = record.time0;
|
|
1197
|
+
for (let index = 0; index < members.length; index += 1) {
|
|
1198
|
+
if (index > 0) time += record.dt[index - 1];
|
|
1199
|
+
let chunk;
|
|
1200
|
+
if (record.type === "text-chunks") chunk = {
|
|
1201
|
+
type: "text-delta",
|
|
1202
|
+
index: record.index,
|
|
1203
|
+
text: members[index]
|
|
1204
|
+
};
|
|
1205
|
+
else if (record.type === "reasoning-chunks") chunk = {
|
|
1206
|
+
type: "reasoning-delta",
|
|
1207
|
+
index: record.index,
|
|
1208
|
+
text: members[index]
|
|
1209
|
+
};
|
|
1210
|
+
else chunk = {
|
|
1211
|
+
type: "tool-call-delta",
|
|
1212
|
+
index: record.index,
|
|
1213
|
+
id: record.id,
|
|
1214
|
+
...Object.hasOwn(record, "name") ? { name: record.name } : {},
|
|
1215
|
+
argumentsDelta: members[index]
|
|
1216
|
+
};
|
|
1217
|
+
chunks.push({
|
|
1218
|
+
time,
|
|
1219
|
+
chunk
|
|
1220
|
+
});
|
|
1221
|
+
}
|
|
1222
|
+
}
|
|
1223
|
+
return chunks;
|
|
1224
|
+
}
|
|
1225
|
+
function hasNonWhitespace(text) {
|
|
1226
|
+
return /\S/.test(text);
|
|
1227
|
+
}
|
|
1228
|
+
function blockIsVisible(block) {
|
|
1229
|
+
if (block.type === "tool-call") return false;
|
|
1230
|
+
if (block.type === "text" || block.type === "reasoning") return hasNonWhitespace(block.text);
|
|
1231
|
+
return true;
|
|
1232
|
+
}
|
|
1233
|
+
/**
|
|
1234
|
+
* Whether one chunk carries the model's first output token for latency measurement.
|
|
1235
|
+
* @param chunk - any stream chunk.
|
|
1236
|
+
* @returns true for a non-empty text, reasoning, or Tool-call arguments fragment and for
|
|
1237
|
+
* every name-bearing Tool-call delta; false for block, usage, and finish chunks.
|
|
1238
|
+
*/
|
|
1239
|
+
function isTokenDelta(chunk) {
|
|
1240
|
+
switch (chunk.type) {
|
|
1241
|
+
case "text-delta":
|
|
1242
|
+
case "reasoning-delta": return chunk.text !== "";
|
|
1243
|
+
case "tool-call-delta": return chunk.argumentsDelta !== "" || chunk.name !== void 0;
|
|
1244
|
+
default: return false;
|
|
1245
|
+
}
|
|
1246
|
+
}
|
|
1247
|
+
/**
|
|
1248
|
+
* Whether one chunk by itself contributes reader-visible transcript content.
|
|
1249
|
+
* Text and reasoning count only with non-whitespace content, streamed as a delta or
|
|
1250
|
+
* completed as a block; a block of any other kind counts at its start and its end,
|
|
1251
|
+
* except a Tool call, which is protocol rather than content. Usage and finish never count.
|
|
1252
|
+
* @param chunk - any stream chunk.
|
|
1253
|
+
* @returns whether a transcript reader would see this chunk.
|
|
1254
|
+
*/
|
|
1255
|
+
function isVisibleChunk(chunk) {
|
|
1256
|
+
switch (chunk.type) {
|
|
1257
|
+
case "text-delta":
|
|
1258
|
+
case "reasoning-delta": return hasNonWhitespace(chunk.text);
|
|
1259
|
+
case "block-start": return chunk.blockType !== "text" && chunk.blockType !== "reasoning" && chunk.blockType !== "tool-call";
|
|
1260
|
+
case "block-end": return blockIsVisible(chunk.block);
|
|
1261
|
+
default: return false;
|
|
1262
|
+
}
|
|
1263
|
+
}
|
|
1264
|
+
/**
|
|
1265
|
+
* Whether one chunk carries non-whitespace text, as a text delta or a completed text block.
|
|
1266
|
+
* Reasoning, Tool calls, and other block kinds never count.
|
|
1267
|
+
* @param chunk - any stream chunk.
|
|
1268
|
+
* @returns whether the chunk contributes visible text.
|
|
1269
|
+
*/
|
|
1270
|
+
function chunkHasVisibleText(chunk) {
|
|
1271
|
+
if (chunk.type === "text-delta") return hasNonWhitespace(chunk.text);
|
|
1272
|
+
return chunk.type === "block-end" && chunk.block.type === "text" && hasNonWhitespace(chunk.block.text);
|
|
1273
|
+
}
|
|
1274
|
+
function firstRunMemberTime(run, predicate) {
|
|
1275
|
+
const fragments = run.type === "tool-call-chunks" ? run.args : run.texts;
|
|
1276
|
+
let time = run.time0;
|
|
1277
|
+
for (let index = 0; index < fragments.length; index += 1) {
|
|
1278
|
+
if (index > 0) time += run.dt[index - 1];
|
|
1279
|
+
if (predicate(fragments[index])) return time;
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
/**
|
|
1283
|
+
* Time of the first member of one packed run that {@link isTokenDelta} accepts: a
|
|
1284
|
+
* name-bearing Tool-call run starts at its first member, otherwise the first non-empty fragment.
|
|
1285
|
+
* Stops scanning at that member.
|
|
1286
|
+
* @param run - one packed delta run.
|
|
1287
|
+
* @returns the member's reconstructed time, or undefined when no member qualifies.
|
|
1288
|
+
*/
|
|
1289
|
+
function runFirstTokenTime(run) {
|
|
1290
|
+
if (run.type === "tool-call-chunks" && run.name !== void 0) return run.time0;
|
|
1291
|
+
return firstRunMemberTime(run, (fragment) => fragment !== "");
|
|
1292
|
+
}
|
|
1293
|
+
/**
|
|
1294
|
+
* Time of the first member of one packed run that {@link isVisibleChunk} accepts: the first
|
|
1295
|
+
* non-whitespace text or reasoning fragment. A Tool-call run has none. Stops scanning at that member.
|
|
1296
|
+
* @param run - one packed delta run.
|
|
1297
|
+
* @returns the member's reconstructed time, or undefined when no member qualifies.
|
|
1298
|
+
*/
|
|
1299
|
+
function runFirstVisibleTime(run) {
|
|
1300
|
+
return run.type === "tool-call-chunks" ? void 0 : firstRunMemberTime(run, hasNonWhitespace);
|
|
1301
|
+
}
|
|
1302
|
+
/**
|
|
1303
|
+
* Time of the first token in one compact stream per {@link isTokenDelta}, read from the
|
|
1304
|
+
* records themselves and stopping at the first qualifying member.
|
|
1305
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1306
|
+
* @returns the first token's time, or undefined when the stream carries no token.
|
|
1307
|
+
*/
|
|
1308
|
+
function assistantStreamFirstTokenTime(stream) {
|
|
1309
|
+
for (const record of stream) {
|
|
1310
|
+
const time = record.type === "chunk" ? isTokenDelta(record.chunk) ? record.time : void 0 : runFirstTokenTime(record);
|
|
1311
|
+
if (time !== void 0) return time;
|
|
1312
|
+
}
|
|
1313
|
+
}
|
|
1314
|
+
/**
|
|
1315
|
+
* Whether one compact stream carries any reader-visible content per {@link isVisibleChunk},
|
|
1316
|
+
* stopping at the first qualifying member.
|
|
1317
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1318
|
+
* @returns whether a transcript reader would see anything from this stream.
|
|
1319
|
+
*/
|
|
1320
|
+
function assistantStreamHasVisibleContent(stream) {
|
|
1321
|
+
return stream.some((record) => record.type === "chunk" ? isVisibleChunk(record.chunk) : runFirstVisibleTime(record) !== void 0);
|
|
1322
|
+
}
|
|
1323
|
+
/**
|
|
1324
|
+
* Whether one compact stream carries non-whitespace text per {@link chunkHasVisibleText},
|
|
1325
|
+
* stopping at the first qualifying member.
|
|
1326
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1327
|
+
* @returns whether the stream contributes visible text.
|
|
1328
|
+
*/
|
|
1329
|
+
function assistantStreamHasVisibleText(stream) {
|
|
1330
|
+
return stream.some((record) => record.type === "text-chunks" ? record.texts.some(hasNonWhitespace) : record.type === "chunk" && chunkHasVisibleText(record.chunk));
|
|
1331
|
+
}
|
|
1332
|
+
/**
|
|
1333
|
+
* The last raw chunk of one never-packed type, scanning backwards and stopping at the first hit.
|
|
1334
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1335
|
+
* @param type - chunk type that only appears as a raw record.
|
|
1336
|
+
* @returns the stream's final chunk of that type, or undefined when it has none.
|
|
1337
|
+
*/
|
|
1338
|
+
function lastAssistantStreamChunk(stream, type) {
|
|
1339
|
+
for (let index = stream.length - 1; index >= 0; index -= 1) {
|
|
1340
|
+
const record = stream[index];
|
|
1341
|
+
if (record.type === "chunk" && record.chunk.type === type) return record.chunk;
|
|
1342
|
+
}
|
|
1343
|
+
}
|
|
1344
|
+
/**
|
|
1345
|
+
* Every raw chunk of one never-packed type, in stream order.
|
|
1346
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1347
|
+
* @param type - chunk type that only appears as a raw record.
|
|
1348
|
+
* @returns the matching chunks; empty when the stream has none.
|
|
1349
|
+
*/
|
|
1350
|
+
function assistantStreamChunks(stream, type) {
|
|
1351
|
+
const chunks = [];
|
|
1352
|
+
for (const record of stream) if (record.type === "chunk" && record.chunk.type === type) chunks.push(record.chunk);
|
|
1353
|
+
return chunks;
|
|
1354
|
+
}
|
|
1355
|
+
/**
|
|
1356
|
+
* Every streamed text-delta fragment joined in stream order; reasoning and Tool-call fragments are excluded.
|
|
1357
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1358
|
+
* @returns the joined text, empty when the stream carries no text delta.
|
|
1359
|
+
*/
|
|
1360
|
+
function joinAssistantStreamText(stream) {
|
|
1361
|
+
const parts = [];
|
|
1362
|
+
for (const record of stream) if (record.type === "text-chunks") parts.push(record.texts.join(""));
|
|
1363
|
+
else if (record.type === "chunk" && record.chunk.type === "text-delta") parts.push(record.chunk.text);
|
|
1364
|
+
return parts.join("");
|
|
1365
|
+
}
|
|
1366
|
+
/**
|
|
1367
|
+
* Feed one compact stream into a {@link BlockAssembler} without materializing members.
|
|
1368
|
+
* Each run contributes one delta carrying its joined fragments, which assembles the same
|
|
1369
|
+
* blocks as the original per-member deltas because assembly only concatenates them;
|
|
1370
|
+
* raw chunks are pushed as recorded. The records are trusted, not validated: validate a
|
|
1371
|
+
* stream read at a durable boundary with {@link expandAssistantStream} first.
|
|
1372
|
+
* @param stream - compact records from one durable Assistant settlement.
|
|
1373
|
+
* @param assembler - assembler to feed; a fresh one by default.
|
|
1374
|
+
* @returns the same assembler after every record was pushed.
|
|
1375
|
+
*/
|
|
1376
|
+
function assembleAssistantStream(stream, assembler = new BlockAssembler()) {
|
|
1377
|
+
for (const record of stream) switch (record.type) {
|
|
1378
|
+
case "chunk":
|
|
1379
|
+
assembler.push(record.chunk);
|
|
1380
|
+
break;
|
|
1381
|
+
case "text-chunks":
|
|
1382
|
+
assembler.push({
|
|
1383
|
+
type: "text-delta",
|
|
1384
|
+
index: record.index,
|
|
1385
|
+
text: record.texts.join("")
|
|
1386
|
+
});
|
|
1387
|
+
break;
|
|
1388
|
+
case "reasoning-chunks":
|
|
1389
|
+
assembler.push({
|
|
1390
|
+
type: "reasoning-delta",
|
|
1391
|
+
index: record.index,
|
|
1392
|
+
text: record.texts.join("")
|
|
1393
|
+
});
|
|
1394
|
+
break;
|
|
1395
|
+
case "tool-call-chunks":
|
|
1396
|
+
assembler.push({
|
|
1397
|
+
type: "tool-call-delta",
|
|
1398
|
+
index: record.index,
|
|
1399
|
+
id: record.id,
|
|
1400
|
+
...record.name === void 0 ? {} : { name: record.name },
|
|
1401
|
+
argumentsDelta: record.args.join("")
|
|
1402
|
+
});
|
|
1403
|
+
break;
|
|
1404
|
+
default: assertNever(record, "assembleAssistantStream");
|
|
1405
|
+
}
|
|
1406
|
+
return assembler;
|
|
1407
|
+
}
|
|
1408
|
+
function validateRecord(value) {
|
|
1409
|
+
if (typeof value !== "object" || value === null || Array.isArray(value)) throw new TypeError("Assistant stream record must be an object");
|
|
1410
|
+
const record = value;
|
|
1411
|
+
switch (record.type) {
|
|
1412
|
+
case "text-chunks":
|
|
1413
|
+
case "reasoning-chunks": {
|
|
1414
|
+
exactKeys(record, [
|
|
1415
|
+
"type",
|
|
1416
|
+
"time0",
|
|
1417
|
+
"index",
|
|
1418
|
+
"dt",
|
|
1419
|
+
"texts"
|
|
1420
|
+
], record.type);
|
|
1421
|
+
const texts = stringArray(record.texts, `${record.type} texts`);
|
|
1422
|
+
if (texts.length === 0) throw new TypeError(`${record.type} texts must be non-empty`);
|
|
1423
|
+
validateRun(record, texts.length, record.type);
|
|
1424
|
+
return record;
|
|
1425
|
+
}
|
|
1426
|
+
case "tool-call-chunks": {
|
|
1427
|
+
exactKeys(record, Object.hasOwn(record, "name") ? [
|
|
1428
|
+
"type",
|
|
1429
|
+
"time0",
|
|
1430
|
+
"index",
|
|
1431
|
+
"dt",
|
|
1432
|
+
"id",
|
|
1433
|
+
"name",
|
|
1434
|
+
"args"
|
|
1435
|
+
] : [
|
|
1436
|
+
"type",
|
|
1437
|
+
"time0",
|
|
1438
|
+
"index",
|
|
1439
|
+
"dt",
|
|
1440
|
+
"id",
|
|
1441
|
+
"args"
|
|
1442
|
+
], record.type);
|
|
1443
|
+
const args = stringArray(record.args, "tool-call-chunks args");
|
|
1444
|
+
if (args.length === 0) throw new TypeError("tool-call-chunks args must be non-empty");
|
|
1445
|
+
if (typeof record.id !== "string" || record.id.length === 0) throw new TypeError("tool-call-chunks id must be a non-empty string");
|
|
1446
|
+
if (record.name !== void 0 && (typeof record.name !== "string" || record.name.length === 0)) throw new TypeError("tool-call-chunks name must be a non-empty string");
|
|
1447
|
+
validateRun(record, args.length, record.type);
|
|
1448
|
+
return record;
|
|
1449
|
+
}
|
|
1450
|
+
case "chunk": {
|
|
1451
|
+
exactKeys(record, [
|
|
1452
|
+
"type",
|
|
1453
|
+
"time",
|
|
1454
|
+
"chunk"
|
|
1455
|
+
], "chunk");
|
|
1456
|
+
const time = safeTime(record.time);
|
|
1457
|
+
if (typeof record.chunk !== "object" || record.chunk === null || Array.isArray(record.chunk)) throw new TypeError("Assistant stream raw chunk must be a lossless JSON object");
|
|
1458
|
+
let chunk;
|
|
1459
|
+
try {
|
|
1460
|
+
chunk = snapshotChunk(record.chunk);
|
|
1461
|
+
} catch (error) {
|
|
1462
|
+
throw new TypeError("Assistant stream raw chunk must be a lossless JSON object", { cause: error });
|
|
1463
|
+
}
|
|
1464
|
+
return deepFreeze({
|
|
1465
|
+
type: "chunk",
|
|
1466
|
+
time,
|
|
1467
|
+
chunk
|
|
1468
|
+
});
|
|
1469
|
+
}
|
|
1470
|
+
default: throw new TypeError(`Unsupported Assistant stream record ${JSON.stringify(record.type)}`);
|
|
1471
|
+
}
|
|
1472
|
+
}
|
|
1473
|
+
function validateRun(record, members, label) {
|
|
1474
|
+
safeTime(record.time0);
|
|
1475
|
+
safeIndex(record.index, label);
|
|
1476
|
+
if (!Array.isArray(record.dt) || record.dt.some((value) => !Number.isSafeInteger(value))) throw new TypeError(`${label} dt must contain safe integers`);
|
|
1477
|
+
if (record.dt.length !== members - 1) throw new TypeError(`${label} dt length must be one less than its members`);
|
|
1478
|
+
let time = record.time0;
|
|
1479
|
+
for (const gap of record.dt) {
|
|
1480
|
+
time += gap;
|
|
1481
|
+
if (!Number.isSafeInteger(time)) throw new TypeError(`${label} member times must stay safe integers`);
|
|
1482
|
+
}
|
|
1483
|
+
}
|
|
1484
|
+
function stringArray(value, label) {
|
|
1485
|
+
if (!Array.isArray(value) || value.some((member) => typeof member !== "string")) throw new TypeError(`${label} must be a string array`);
|
|
1486
|
+
return value;
|
|
1487
|
+
}
|
|
1488
|
+
function exactKeys(record, keys, label) {
|
|
1489
|
+
if (Object.keys(record).length !== keys.length || !keys.every((key) => Object.hasOwn(record, key))) throw new TypeError(`${label} Assistant stream record must contain exactly ${keys.join(", ")}`);
|
|
1490
|
+
}
|
|
1491
|
+
//#endregion
|
|
1492
|
+
//#region lib/types/index.js
|
|
1493
|
+
/**
|
|
1494
|
+
* LLM service: adapter registry with a waterfall-interceptable streaming call
|
|
1495
|
+
* API. Exports the `LlmRuntime` default, the abstract `LlmAdapter` for
|
|
1496
|
+
* provider backends, and `BlockAssembler` for chunk assembly.
|
|
1497
|
+
*
|
|
1498
|
+
* @module @xlaunch/llm
|
|
1499
|
+
*/
|
|
1500
|
+
var __runInitializers = function(thisArg, initializers, value) {
|
|
1501
|
+
var useValue = arguments.length > 2;
|
|
1502
|
+
for (var i = 0; i < initializers.length; i++) value = useValue ? initializers[i].call(thisArg, value) : initializers[i].call(thisArg);
|
|
1503
|
+
return useValue ? value : void 0;
|
|
1504
|
+
};
|
|
1505
|
+
var __esDecorate = function(ctor, descriptorIn, decorators, contextIn, initializers, extraInitializers) {
|
|
1506
|
+
function accept(f) {
|
|
1507
|
+
if (f !== void 0 && typeof f !== "function") throw new TypeError("Function expected");
|
|
1508
|
+
return f;
|
|
1509
|
+
}
|
|
1510
|
+
var kind = contextIn.kind, key = kind === "getter" ? "get" : kind === "setter" ? "set" : "value";
|
|
1511
|
+
var target = !descriptorIn && ctor ? contextIn["static"] ? ctor : ctor.prototype : null;
|
|
1512
|
+
var descriptor = descriptorIn || (target ? Object.getOwnPropertyDescriptor(target, contextIn.name) : {});
|
|
1513
|
+
var _, done = false;
|
|
1514
|
+
for (var i = decorators.length - 1; i >= 0; i--) {
|
|
1515
|
+
var context = {};
|
|
1516
|
+
for (var p in contextIn) context[p] = p === "access" ? {} : contextIn[p];
|
|
1517
|
+
for (var p in contextIn.access) context.access[p] = contextIn.access[p];
|
|
1518
|
+
context.addInitializer = function(f) {
|
|
1519
|
+
if (done) throw new TypeError("Cannot add initializers after decoration has completed");
|
|
1520
|
+
extraInitializers.push(accept(f || null));
|
|
1521
|
+
};
|
|
1522
|
+
var result = (0, decorators[i])(kind === "accessor" ? {
|
|
1523
|
+
get: descriptor.get,
|
|
1524
|
+
set: descriptor.set
|
|
1525
|
+
} : descriptor[key], context);
|
|
1526
|
+
if (kind === "accessor") {
|
|
1527
|
+
if (result === void 0) continue;
|
|
1528
|
+
if (result === null || typeof result !== "object") throw new TypeError("Object expected");
|
|
1529
|
+
if (_ = accept(result.get)) descriptor.get = _;
|
|
1530
|
+
if (_ = accept(result.set)) descriptor.set = _;
|
|
1531
|
+
if (_ = accept(result.init)) initializers.unshift(_);
|
|
1532
|
+
} else if (_ = accept(result)) if (kind === "field") initializers.unshift(_);
|
|
1533
|
+
else descriptor[key] = _;
|
|
1534
|
+
}
|
|
1535
|
+
if (target) Object.defineProperty(target, contextIn.name, descriptor);
|
|
1536
|
+
done = true;
|
|
1537
|
+
};
|
|
1538
|
+
/**
|
|
1539
|
+
* Typed error for LLM-related failures. Extends {@link HarnessError}, so the
|
|
1540
|
+
* `code` string (e.g. `AUTH`, `RATE_LIMIT`, `NO_ADAPTER`) is shared taxonomy.
|
|
1541
|
+
*/
|
|
1542
|
+
var LlmError = class extends HarnessError {
|
|
1543
|
+
/** Serializable facts retained beside this live Error. */
|
|
1544
|
+
failure;
|
|
1545
|
+
/**
|
|
1546
|
+
* @param message - non-empty human-readable failure summary.
|
|
1547
|
+
* @param code - non-empty stable provider-neutral machine code.
|
|
1548
|
+
* @param options - optional cause and validated serializable provider facts.
|
|
1549
|
+
*/
|
|
1550
|
+
constructor(message, code, options) {
|
|
1551
|
+
if (typeof message !== "string" || message.length === 0) throw new Error("LlmError message must be a non-empty string");
|
|
1552
|
+
if (typeof code !== "string" || code.length === 0) throw new Error("LlmError code must be a non-empty string");
|
|
1553
|
+
if (options?.status !== void 0 && (!Number.isInteger(options.status) || options.status < 100 || options.status > 599)) throw new Error("LlmError status must be an integer from 100 through 599");
|
|
1554
|
+
if (options?.providerRetryAfterMs !== void 0 && (!Number.isFinite(options.providerRetryAfterMs) || options.providerRetryAfterMs <= 0)) throw new Error("LlmError providerRetryAfterMs must be a positive finite number");
|
|
1555
|
+
if (options?.requestId !== void 0 && (typeof options.requestId !== "string" || options.requestId.length === 0)) throw new Error("LlmError requestId must be a non-empty string");
|
|
1556
|
+
super(message, code, options);
|
|
1557
|
+
this.name = "LlmError";
|
|
1558
|
+
this.failure = Object.freeze({
|
|
1559
|
+
message,
|
|
1560
|
+
code,
|
|
1561
|
+
...options?.status === void 0 ? {} : { status: options.status },
|
|
1562
|
+
...options?.providerRetryAfterMs === void 0 ? {} : { providerRetryAfterMs: options.providerRetryAfterMs },
|
|
1563
|
+
...options?.requestId === void 0 ? {} : { requestId: options.requestId }
|
|
1564
|
+
});
|
|
1565
|
+
}
|
|
1566
|
+
};
|
|
1567
|
+
/**
|
|
1568
|
+
* Accept one supplied credential, or refuse it as unusable.
|
|
1569
|
+
*
|
|
1570
|
+
* A stored key arrives from the credentials seam, a `.env` line, or a shell
|
|
1571
|
+
* export, all of which pick up surrounding whitespace, so trimming is silent.
|
|
1572
|
+
* Anything else fails here rather than inside `fetch`, whose ByteString
|
|
1573
|
+
* refusal names a UTF-16 code point instead of the setting to change. The key
|
|
1574
|
+
* never enters the message: `ref` names where to fix it, and echoing any part
|
|
1575
|
+
* of a secret into a log or a UI is the failure this diagnosis avoids.
|
|
1576
|
+
*
|
|
1577
|
+
* Lives beside {@link LlmError} rather than in `./api-key.ts` so the predicate
|
|
1578
|
+
* module stays dependency-free; every adapter shares this one diagnosis instead
|
|
1579
|
+
* of keeping near-identical local copies.
|
|
1580
|
+
* @param raw - the credential exactly as supplied.
|
|
1581
|
+
* @param pkg - the refusing package name, prefixed to the diagnostic.
|
|
1582
|
+
* @param ref - the credential reference the value resolved through.
|
|
1583
|
+
* @returns the trimmed, usable key.
|
|
1584
|
+
*/
|
|
1585
|
+
function assertUsableApiKey(raw, pkg, ref) {
|
|
1586
|
+
const checked = normalizeApiKey(raw);
|
|
1587
|
+
if (checked.ok) return checked.value;
|
|
1588
|
+
throw new LlmError(checked.reason === "empty" ? `${pkg}: the API key resolved from ${ref} is blank; set ${ref} to the raw key (the web Models page writes it) or export it in the launching environment` : `${pkg}: the API key resolved from ${ref} contains characters no HTTP header can carry; set ${ref} to the raw key alone (the web Models page writes it)`, INVALID_CREDENTIAL_CODE);
|
|
1589
|
+
}
|
|
1590
|
+
/**
|
|
1591
|
+
* Provider-wire adapter for the harness message and stream vocabulary. Register implementations
|
|
1592
|
+
* with `ctx.llm.registerAdapter(providers, adapter)`. Every provider HTTP request must include
|
|
1593
|
+
* `attributionHeaders()`; prove the headers are added in the wire request or library header hook. The
|
|
1594
|
+
* library-backed gateway adapter meets this contract through gateway's header hook.
|
|
1595
|
+
*/
|
|
1596
|
+
var LlmAdapter = class {
|
|
1597
|
+
/**
|
|
1598
|
+
* Describe one provider route owned by this adapter.
|
|
1599
|
+
* @param provider - a route passed to `registerAdapter()` for this instance.
|
|
1600
|
+
* @returns detached display metadata whose id must equal `provider`.
|
|
1601
|
+
*/
|
|
1602
|
+
providerInfo(provider) {
|
|
1603
|
+
return {
|
|
1604
|
+
id: provider,
|
|
1605
|
+
name: provider
|
|
1606
|
+
};
|
|
1607
|
+
}
|
|
1608
|
+
/**
|
|
1609
|
+
* Return the provider-owned retry policy captured with this route.
|
|
1610
|
+
* @param _provider - a route passed to `registerAdapter()` for this instance.
|
|
1611
|
+
* @returns a resolved policy, or `undefined` to use the normal defaults.
|
|
1612
|
+
*/
|
|
1613
|
+
providerRetryPolicy(_provider) {}
|
|
1614
|
+
/**
|
|
1615
|
+
* Resolve provider-side request-image pricing for one exact model route.
|
|
1616
|
+
* The default declares none, so consumers fall back to their own neutral
|
|
1617
|
+
* estimate. Implementations must answer synchronously without I/O; the
|
|
1618
|
+
* token meter resolves this per measurement.
|
|
1619
|
+
* @param _provider - a route passed to `registerAdapter()` for this instance.
|
|
1620
|
+
* @param _model - exact model id passed to {@link GenerateOptions.model}.
|
|
1621
|
+
* @returns route-owned image pricing, or `undefined` when the route declares none.
|
|
1622
|
+
*/
|
|
1623
|
+
imageRequestPricing(_provider, _model) {}
|
|
1624
|
+
/**
|
|
1625
|
+
* List models this adapter can currently advertise for one owned provider.
|
|
1626
|
+
* The result is advisory: an adapter may accept unlisted model ids, and
|
|
1627
|
+
* consumers must not turn absence into request rejection.
|
|
1628
|
+
* @param _provider - one provider route owned by this adapter.
|
|
1629
|
+
* @returns discoverable models in adapter-preferred order.
|
|
1630
|
+
*/
|
|
1631
|
+
listModels(_provider) {
|
|
1632
|
+
return Promise.resolve([]);
|
|
1633
|
+
}
|
|
1634
|
+
/**
|
|
1635
|
+
* Resolve all metadata available for one exact model. This query is
|
|
1636
|
+
* independent of the advisory catalog and does not validate request routing.
|
|
1637
|
+
* @param provider - one provider route owned by this adapter.
|
|
1638
|
+
* @param model - exact model id passed to {@link GenerateOptions.model}.
|
|
1639
|
+
* @param _signal - cancellation for this exact-model lookup; asynchronous
|
|
1640
|
+
* implementations must settle promptly after it aborts.
|
|
1641
|
+
* @returns provider/model identity plus any context, call-default, and reasoning metadata.
|
|
1642
|
+
*/
|
|
1643
|
+
resolveModel(provider, model, _signal) {
|
|
1644
|
+
return Promise.resolve({
|
|
1645
|
+
provider,
|
|
1646
|
+
id: model,
|
|
1647
|
+
name: model
|
|
1648
|
+
});
|
|
1649
|
+
}
|
|
1650
|
+
/**
|
|
1651
|
+
* Bind exact model metadata and the eventual request dispatch to one adapter generation.
|
|
1652
|
+
* Dynamic adapters override this so settings changes between preparation and
|
|
1653
|
+
* dispatch cannot combine one generation's capabilities with another's endpoint.
|
|
1654
|
+
* @param provider - registered provider route.
|
|
1655
|
+
* @param model - exact model id.
|
|
1656
|
+
* @param signal - cancellation for model resolution.
|
|
1657
|
+
* @returns model metadata and a one-generation stream entry point.
|
|
1658
|
+
*/
|
|
1659
|
+
async prepareCall(provider, model, signal) {
|
|
1660
|
+
return {
|
|
1661
|
+
model: await this.resolveModel(provider, model, signal),
|
|
1662
|
+
stream: (options) => this.stream(options)
|
|
1663
|
+
};
|
|
1664
|
+
}
|
|
1665
|
+
};
|
|
1666
|
+
/**
|
|
1667
|
+
* The abstract `llm` service: an adapter registry plus a streaming model-call
|
|
1668
|
+
* API, interceptable via the `llm/stream` waterfall.
|
|
1669
|
+
*/
|
|
1670
|
+
let LlmRuntime = (() => {
|
|
1671
|
+
let _classSuper = TypertRemoteService;
|
|
1672
|
+
let _instanceExtraInitializers = [];
|
|
1673
|
+
let _listProviders_decorators;
|
|
1674
|
+
let _listConfigurableProviders_decorators;
|
|
1675
|
+
let _remoteDiscoverModels_decorators;
|
|
1676
|
+
return class LlmRuntime extends _classSuper {
|
|
1677
|
+
static {
|
|
1678
|
+
const _metadata = typeof Symbol === "function" && Symbol.metadata ? Object.create(_classSuper[Symbol.metadata] ?? null) : void 0;
|
|
1679
|
+
_listProviders_decorators = [Remote];
|
|
1680
|
+
_listConfigurableProviders_decorators = [Remote];
|
|
1681
|
+
_remoteDiscoverModels_decorators = [Remote("discoverModels")];
|
|
1682
|
+
__esDecorate(this, null, _listProviders_decorators, {
|
|
1683
|
+
kind: "method",
|
|
1684
|
+
name: "listProviders",
|
|
1685
|
+
static: false,
|
|
1686
|
+
private: false,
|
|
1687
|
+
access: {
|
|
1688
|
+
has: (obj) => "listProviders" in obj,
|
|
1689
|
+
get: (obj) => obj.listProviders
|
|
1690
|
+
},
|
|
1691
|
+
metadata: _metadata
|
|
1692
|
+
}, null, _instanceExtraInitializers);
|
|
1693
|
+
__esDecorate(this, null, _listConfigurableProviders_decorators, {
|
|
1694
|
+
kind: "method",
|
|
1695
|
+
name: "listConfigurableProviders",
|
|
1696
|
+
static: false,
|
|
1697
|
+
private: false,
|
|
1698
|
+
access: {
|
|
1699
|
+
has: (obj) => "listConfigurableProviders" in obj,
|
|
1700
|
+
get: (obj) => obj.listConfigurableProviders
|
|
1701
|
+
},
|
|
1702
|
+
metadata: _metadata
|
|
1703
|
+
}, null, _instanceExtraInitializers);
|
|
1704
|
+
__esDecorate(this, null, _remoteDiscoverModels_decorators, {
|
|
1705
|
+
kind: "method",
|
|
1706
|
+
name: "remoteDiscoverModels",
|
|
1707
|
+
static: false,
|
|
1708
|
+
private: false,
|
|
1709
|
+
access: {
|
|
1710
|
+
has: (obj) => "remoteDiscoverModels" in obj,
|
|
1711
|
+
get: (obj) => obj.remoteDiscoverModels
|
|
1712
|
+
},
|
|
1713
|
+
metadata: _metadata
|
|
1714
|
+
}, null, _instanceExtraInitializers);
|
|
1715
|
+
if (_metadata) Object.defineProperty(this, Symbol.metadata, {
|
|
1716
|
+
enumerable: true,
|
|
1717
|
+
configurable: true,
|
|
1718
|
+
writable: true,
|
|
1719
|
+
value: _metadata
|
|
1720
|
+
});
|
|
1721
|
+
}
|
|
1722
|
+
adapters = (__runInitializers(this, _instanceExtraInitializers), /* @__PURE__ */ new Map());
|
|
1723
|
+
directory = /* @__PURE__ */ new Map();
|
|
1724
|
+
discoveries = /* @__PURE__ */ new Map();
|
|
1725
|
+
constructor(ctx) {
|
|
1726
|
+
super(ctx, "llm");
|
|
1727
|
+
}
|
|
1728
|
+
/** Notify topology observers without letting one broken listener veto the commit. */
|
|
1729
|
+
emitAdaptersUpdated() {
|
|
1730
|
+
let invariantFailure;
|
|
1731
|
+
for (const listener of this.ctx.events.dispatch("emit", ["llm/adapters-updated"])) try {
|
|
1732
|
+
const returned = listener();
|
|
1733
|
+
if (returned != null && typeof returned.then === "function") Promise.resolve(returned).then(void 0, (error) => {
|
|
1734
|
+
this.warnAdaptersListenerFailure(error);
|
|
1735
|
+
});
|
|
1736
|
+
} catch (error) {
|
|
1737
|
+
if (error?.code === "INVARIANT") {
|
|
1738
|
+
invariantFailure ??= error;
|
|
1739
|
+
continue;
|
|
1740
|
+
}
|
|
1741
|
+
this.warnAdaptersListenerFailure(error);
|
|
1742
|
+
}
|
|
1743
|
+
if (invariantFailure !== void 0) throw invariantFailure;
|
|
1744
|
+
}
|
|
1745
|
+
/** Contained-listener diagnostic shared by the sync and async failure paths. */
|
|
1746
|
+
warnAdaptersListenerFailure(error) {
|
|
1747
|
+
this.ctx.logger.warn("llm: an llm/adapters-updated listener failed");
|
|
1748
|
+
this.ctx.logger.warn(error);
|
|
1749
|
+
}
|
|
1750
|
+
/**
|
|
1751
|
+
* Register an adapter for the given provider routes. Throws `LlmError` with code
|
|
1752
|
+
* `DUPLICATE_ADAPTER` if any provider already has an adapter (all-or-nothing).
|
|
1753
|
+
* Disposed with the fiber.
|
|
1754
|
+
* @param providers - every provider route this adapter should serve.
|
|
1755
|
+
* @param adapter - the adapter that streams calls for those providers.
|
|
1756
|
+
* @returns the disposer, carrying {@link AdapterRegistrationHandle.replace}.
|
|
1757
|
+
*/
|
|
1758
|
+
registerAdapter(providers, adapter) {
|
|
1759
|
+
const owned = /* @__PURE__ */ new Set();
|
|
1760
|
+
let released = false;
|
|
1761
|
+
const dispose = this.ctx.effect(function* () {
|
|
1762
|
+
if (providers.length === 0) throw new LlmError("an adapter must register at least one provider", "INVALID_ADAPTER");
|
|
1763
|
+
this.commitRoutes(owned, this.prepareRoutes(providers, adapter, owned));
|
|
1764
|
+
yield () => {
|
|
1765
|
+
released = true;
|
|
1766
|
+
for (const provider of owned) this.adapters.delete(provider);
|
|
1767
|
+
owned.clear();
|
|
1768
|
+
this.emitAdaptersUpdated();
|
|
1769
|
+
};
|
|
1770
|
+
}.bind(this), "llm.registerAdapter()");
|
|
1771
|
+
const handle = (() => void dispose());
|
|
1772
|
+
handle.replace = (next) => {
|
|
1773
|
+
if (released) throw new LlmError("a disposed adapter registration cannot replace its routes", "REGISTRATION_DISPOSED");
|
|
1774
|
+
this.commitRoutes(owned, this.prepareRoutes(next, adapter, owned));
|
|
1775
|
+
};
|
|
1776
|
+
return handle;
|
|
1777
|
+
}
|
|
1778
|
+
/**
|
|
1779
|
+
* Validate one candidate route set for `adapter`, treating routes this
|
|
1780
|
+
* registration already holds as available. Nothing is mutated: a rejected
|
|
1781
|
+
* candidate leaves the registry exactly as it was.
|
|
1782
|
+
*/
|
|
1783
|
+
prepareRoutes(providers, adapter, owned) {
|
|
1784
|
+
const unique = /* @__PURE__ */ new Set();
|
|
1785
|
+
const registrations = [];
|
|
1786
|
+
for (const provider of providers) {
|
|
1787
|
+
if (provider.length === 0) throw new LlmError("adapter provider names must be non-empty", "INVALID_ADAPTER");
|
|
1788
|
+
if (unique.has(provider) || this.adapters.has(provider) && !owned.has(provider)) throw new LlmError(`an adapter for provider "${provider}" is already registered`, "DUPLICATE_ADAPTER");
|
|
1789
|
+
const info = adapter.providerInfo(provider);
|
|
1790
|
+
if (typeof info.id !== "string" || info.id !== provider || typeof info.name !== "string" || info.name.length === 0) throw new LlmError(`adapter metadata for provider "${provider}" must preserve its id and have a non-empty name`, "INVALID_ADAPTER");
|
|
1791
|
+
unique.add(provider);
|
|
1792
|
+
const retryPolicy = adapter.providerRetryPolicy(provider) ?? resolveRetryPolicy(void 0, `llm: provider "${provider}" retryPolicy`);
|
|
1793
|
+
registrations.push({
|
|
1794
|
+
adapter,
|
|
1795
|
+
provider: {
|
|
1796
|
+
id: info.id,
|
|
1797
|
+
name: info.name
|
|
1798
|
+
},
|
|
1799
|
+
retryPolicy
|
|
1800
|
+
});
|
|
1801
|
+
}
|
|
1802
|
+
return registrations;
|
|
1803
|
+
}
|
|
1804
|
+
/**
|
|
1805
|
+
* Swap this registration's routes for the prepared ones in one synchronous
|
|
1806
|
+
* section, so no observer can see the registry between the release and the
|
|
1807
|
+
* re-registration. The route set's one mutation point is also where
|
|
1808
|
+
* `llm/adapters-updated` is published, so a `replace` announces itself
|
|
1809
|
+
* exactly like a first registration.
|
|
1810
|
+
*/
|
|
1811
|
+
commitRoutes(owned, registrations) {
|
|
1812
|
+
for (const provider of owned) this.adapters.delete(provider);
|
|
1813
|
+
owned.clear();
|
|
1814
|
+
for (const registration of registrations) {
|
|
1815
|
+
this.adapters.set(registration.provider.id, registration);
|
|
1816
|
+
owned.add(registration.provider.id);
|
|
1817
|
+
}
|
|
1818
|
+
this.emitAdaptersUpdated();
|
|
1819
|
+
}
|
|
1820
|
+
/**
|
|
1821
|
+
* Describe provider routes with a registered adapter.
|
|
1822
|
+
* @returns detached provider metadata in registration order.
|
|
1823
|
+
*/
|
|
1824
|
+
listProviders() {
|
|
1825
|
+
return [...this.adapters.values()].map(({ provider }) => ({ ...provider }));
|
|
1826
|
+
}
|
|
1827
|
+
/**
|
|
1828
|
+
* Declare provider routes an adapter plugin can activate through
|
|
1829
|
+
* configuration. Registration is all-or-nothing: an empty list, invalid
|
|
1830
|
+
* entry, or a provider already declared by any registration throws
|
|
1831
|
+
* `LlmError` without registering the rest. Disposed with the fiber.
|
|
1832
|
+
* @param entries - every configurable provider this plugin owns.
|
|
1833
|
+
* @returns a handle that withdraws all of them, and can atomically replace them.
|
|
1834
|
+
*/
|
|
1835
|
+
registerConfigurableProviders(entries) {
|
|
1836
|
+
let held = [];
|
|
1837
|
+
let disposed = false;
|
|
1838
|
+
/**
|
|
1839
|
+
* Validate a candidate set in full against everything this registration
|
|
1840
|
+
* does not already hold, then publish it. Nothing is written until the
|
|
1841
|
+
* whole set passes, so a refused candidate leaves the current entries in
|
|
1842
|
+
* place — the property that makes `replace` a swap rather than a
|
|
1843
|
+
* delete-then-add that can strand the directory empty.
|
|
1844
|
+
*/
|
|
1845
|
+
const commit = (candidates) => {
|
|
1846
|
+
const detached = [];
|
|
1847
|
+
const own = new Set(held.map((entry) => entry.provider));
|
|
1848
|
+
for (const entry of candidates) {
|
|
1849
|
+
if (entry.provider.length === 0 || entry.displayName.length === 0 || entry.settingsNs.length === 0) throw new LlmError("configurable providers need a non-empty provider, displayName, and settingsNs", "INVALID_DIRECTORY");
|
|
1850
|
+
if (entry.settingsPath.some((segment) => segment.length === 0)) throw new LlmError(`configurable provider "${entry.provider}" has an empty settingsPath segment`, "INVALID_DIRECTORY");
|
|
1851
|
+
if (this.directory.has(entry.provider) && !own.has(entry.provider) || detached.some((seen) => seen.provider === entry.provider)) throw new LlmError(`configurable provider "${entry.provider}" is already declared`, "DUPLICATE_DIRECTORY");
|
|
1852
|
+
detached.push({
|
|
1853
|
+
...entry,
|
|
1854
|
+
settingsPath: [...entry.settingsPath]
|
|
1855
|
+
});
|
|
1856
|
+
}
|
|
1857
|
+
for (const entry of held) this.directory.delete(entry.provider);
|
|
1858
|
+
for (const entry of detached) this.directory.set(entry.provider, entry);
|
|
1859
|
+
held = detached;
|
|
1860
|
+
this.emitAdaptersUpdated();
|
|
1861
|
+
};
|
|
1862
|
+
const dispose = this.ctx.effect(function* () {
|
|
1863
|
+
if (entries.length === 0) throw new LlmError("a configurable-provider registration must declare at least one provider", "INVALID_DIRECTORY");
|
|
1864
|
+
commit(entries);
|
|
1865
|
+
yield () => {
|
|
1866
|
+
disposed = true;
|
|
1867
|
+
for (const entry of held) this.directory.delete(entry.provider);
|
|
1868
|
+
held = [];
|
|
1869
|
+
this.emitAdaptersUpdated();
|
|
1870
|
+
};
|
|
1871
|
+
}.bind(this), "llm.registerConfigurableProviders()");
|
|
1872
|
+
const handle = (() => void dispose());
|
|
1873
|
+
handle.replace = (next) => {
|
|
1874
|
+
if (disposed) throw new LlmError("this configurable-provider registration was disposed", "REGISTRATION_DISPOSED");
|
|
1875
|
+
commit(next);
|
|
1876
|
+
};
|
|
1877
|
+
return handle;
|
|
1878
|
+
}
|
|
1879
|
+
/**
|
|
1880
|
+
* List every declared configurable provider, registered or dormant.
|
|
1881
|
+
* @returns detached directory entries in declaration order.
|
|
1882
|
+
*/
|
|
1883
|
+
listConfigurableProviders() {
|
|
1884
|
+
return [...this.directory.values()].map((entry) => ({
|
|
1885
|
+
...entry,
|
|
1886
|
+
settingsPath: [...entry.settingsPath]
|
|
1887
|
+
}));
|
|
1888
|
+
}
|
|
1889
|
+
/**
|
|
1890
|
+
* Offer to interrogate provider endpoints on behalf of the settings
|
|
1891
|
+
* namespace this plugin owns. The namespace is the key because that is what
|
|
1892
|
+
* a configuration surface already holds from the configurable-provider
|
|
1893
|
+
* directory, and because a provider being *added* has no route to name yet.
|
|
1894
|
+
* Disposed with the fiber.
|
|
1895
|
+
* @param settingsNs - the namespace whose profiles this discovery serves.
|
|
1896
|
+
* @param discover - interrogates one endpoint and must honor the supplied signal.
|
|
1897
|
+
* @returns the disposer that withdraws the offer.
|
|
1898
|
+
*/
|
|
1899
|
+
registerModelDiscovery(settingsNs, discover) {
|
|
1900
|
+
const dispose = this.ctx.effect(function* () {
|
|
1901
|
+
if (settingsNs.length === 0) throw new LlmError("model discovery needs a non-empty settings namespace", "INVALID_DISCOVERY");
|
|
1902
|
+
if (this.discoveries.has(settingsNs)) throw new LlmError(`model discovery for "${settingsNs}" is already registered`, "DUPLICATE_DISCOVERY");
|
|
1903
|
+
this.discoveries.set(settingsNs, discover);
|
|
1904
|
+
yield () => {
|
|
1905
|
+
this.discoveries.delete(settingsNs);
|
|
1906
|
+
};
|
|
1907
|
+
}.bind(this), "llm.registerModelDiscovery()");
|
|
1908
|
+
return () => void dispose();
|
|
1909
|
+
}
|
|
1910
|
+
/**
|
|
1911
|
+
* Interrogate one provider endpoint for the models it advertises. The
|
|
1912
|
+
* request describes a draft, not a stored route, so nothing here reads or
|
|
1913
|
+
* writes settings or credentials — the caller owns both, and the reply is
|
|
1914
|
+
* candidate metadata a surface may offer for adoption.
|
|
1915
|
+
* @param settingsNs - namespace whose registered discovery serves this draft.
|
|
1916
|
+
* @param request - the endpoint, protocol, and one-shot credential to use.
|
|
1917
|
+
* @param signal - caller cancellation.
|
|
1918
|
+
* @returns the advertised models, deduplicated in endpoint order.
|
|
1919
|
+
*/
|
|
1920
|
+
async discoverModels(settingsNs, request, signal) {
|
|
1921
|
+
const discover = this.discoveries.get(settingsNs);
|
|
1922
|
+
if (discover === void 0) throw new LlmError(`no model discovery is registered for "${settingsNs}"`, "NO_DISCOVERY");
|
|
1923
|
+
if ((request.provider ?? "").length === 0 && (request.baseURL ?? "").length === 0) throw new LlmError("model discovery needs a provider route or a baseURL", "INVALID_DISCOVERY");
|
|
1924
|
+
const discovered = signal === void 0 ? await discover(request) : await discover(request, signal);
|
|
1925
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1926
|
+
const models = [];
|
|
1927
|
+
for (const model of discovered) {
|
|
1928
|
+
if (typeof model.id !== "string" || model.id.length === 0 || seen.has(model.id)) continue;
|
|
1929
|
+
seen.add(model.id);
|
|
1930
|
+
models.push({
|
|
1931
|
+
id: model.id,
|
|
1932
|
+
...model.name === void 0 ? {} : { name: model.name },
|
|
1933
|
+
...model.contextWindow === void 0 ? {} : { contextWindow: model.contextWindow },
|
|
1934
|
+
...model.maxTokens === void 0 ? {} : { maxTokens: model.maxTokens }
|
|
1935
|
+
});
|
|
1936
|
+
}
|
|
1937
|
+
return models;
|
|
1938
|
+
}
|
|
1939
|
+
/**
|
|
1940
|
+
* Remote adapter for one draft provider interrogation.
|
|
1941
|
+
* @param settingsNs - namespace whose registered discovery serves this draft.
|
|
1942
|
+
* @param request - endpoint, protocol, and one-shot credential to use.
|
|
1943
|
+
* @param signal - caller cancellation supplied by the Remote carrier.
|
|
1944
|
+
* @returns advertised models in endpoint order.
|
|
1945
|
+
* @throws RemoteError with `llm/model-discovery-rejected` when discovery refuses or fails.
|
|
1946
|
+
*/
|
|
1947
|
+
async remoteDiscoverModels(settingsNs, request, signal) {
|
|
1948
|
+
try {
|
|
1949
|
+
return await this.discoverModels(settingsNs, request, signal);
|
|
1950
|
+
} catch (error) {
|
|
1951
|
+
throw new RemoteError("llm/model-discovery-rejected", error instanceof Error ? error.message : String(error), {
|
|
1952
|
+
settingsNs,
|
|
1953
|
+
...request.baseURL === void 0 ? {} : { baseURL: request.baseURL }
|
|
1954
|
+
}, { cause: error });
|
|
1955
|
+
}
|
|
1956
|
+
}
|
|
1957
|
+
/**
|
|
1958
|
+
* Resolve the retry policy captured when one provider route was registered.
|
|
1959
|
+
* @param provider - registered provider route to inspect.
|
|
1960
|
+
* @returns the provider-owned policy, with normal defaults already resolved.
|
|
1961
|
+
*/
|
|
1962
|
+
providerRetryPolicy(provider) {
|
|
1963
|
+
return this.registration(provider).retryPolicy;
|
|
1964
|
+
}
|
|
1965
|
+
/**
|
|
1966
|
+
* Resolve provider-side request-image pricing for one exact route, or
|
|
1967
|
+
* `undefined` when the provider is unregistered or declares none. Unknown
|
|
1968
|
+
* providers degrade to `undefined` rather than throwing because callers
|
|
1969
|
+
* price durable history whose route may no longer be mounted.
|
|
1970
|
+
* @param provider - provider route named by a request header.
|
|
1971
|
+
* @param model - exact model id named by the same header.
|
|
1972
|
+
* @returns the owning adapter's image pricing for the route, when declared.
|
|
1973
|
+
*/
|
|
1974
|
+
imageRequestPricing(provider, model) {
|
|
1975
|
+
return this.adapters.get(provider)?.adapter.imageRequestPricing(provider, model);
|
|
1976
|
+
}
|
|
1977
|
+
/**
|
|
1978
|
+
* Resolve the exact text one durable file occurrence contributes to every
|
|
1979
|
+
* provider request in the current execution environment.
|
|
1980
|
+
* @param ref - durable verbatim file reference from model history.
|
|
1981
|
+
* @returns the same deterministic handle text used at adapter dispatch.
|
|
1982
|
+
*/
|
|
1983
|
+
fileRequestText(ref) {
|
|
1984
|
+
return fileHandleText(ref, this.fileReadPath(ref));
|
|
1985
|
+
}
|
|
1986
|
+
/** Detach typed adapter-owned modality metadata. */
|
|
1987
|
+
detachedModalities(modalities) {
|
|
1988
|
+
return modalities === void 0 ? void 0 : [...modalities];
|
|
1989
|
+
}
|
|
1990
|
+
/**
|
|
1991
|
+
* Discover models advertised by one registered provider. Catalog membership
|
|
1992
|
+
* is advisory and never changes routing or request validation.
|
|
1993
|
+
* @param provider - registered provider route to inspect.
|
|
1994
|
+
* @returns detached model metadata in adapter-preferred order.
|
|
1995
|
+
*/
|
|
1996
|
+
async listModels(provider) {
|
|
1997
|
+
const models = await this.registration(provider).adapter.listModels(provider);
|
|
1998
|
+
const seen = /* @__PURE__ */ new Set();
|
|
1999
|
+
return models.map((model) => {
|
|
2000
|
+
if (typeof model.provider !== "string" || model.provider !== provider || typeof model.id !== "string" || model.id.length === 0 || typeof model.name !== "string" || model.name.length === 0 || model.description !== void 0 && typeof model.description !== "string" || seen.has(model.id)) throw new LlmError(`adapter returned invalid or duplicate model metadata for provider "${provider}"`, "INVALID_CATALOG");
|
|
2001
|
+
seen.add(model.id);
|
|
2002
|
+
const inputModalities = this.detachedModalities(model.inputModalities);
|
|
2003
|
+
return {
|
|
2004
|
+
provider: model.provider,
|
|
2005
|
+
id: model.id,
|
|
2006
|
+
name: model.name,
|
|
2007
|
+
...model.description === void 0 ? {} : { description: model.description },
|
|
2008
|
+
...inputModalities === void 0 ? {} : { inputModalities }
|
|
2009
|
+
};
|
|
2010
|
+
});
|
|
2011
|
+
}
|
|
2012
|
+
/**
|
|
2013
|
+
* Resolve and validate all metadata from the adapter that owns one exact
|
|
2014
|
+
* route. The result is detached from adapter-owned objects; catalog
|
|
2015
|
+
* membership remains advisory and does not control request routing.
|
|
2016
|
+
* @param provider - registered provider route to inspect.
|
|
2017
|
+
* @param model - exact model id passed to the adapter.
|
|
2018
|
+
* @param signal - optional cancellation for adapter-owned asynchronous lookup.
|
|
2019
|
+
* @returns exact model identity plus available context and reasoning metadata.
|
|
2020
|
+
*/
|
|
2021
|
+
async resolveModelInfo(provider, model, signal) {
|
|
2022
|
+
return this.resolveModelInfoFor(this.registration(provider), model, signal);
|
|
2023
|
+
}
|
|
2024
|
+
async resolveModelInfoFor(registration, model, signal) {
|
|
2025
|
+
const resolved = await registration.adapter.resolveModel(registration.provider.id, model, signal);
|
|
2026
|
+
return this.normalizeModelInfo(registration, model, resolved);
|
|
2027
|
+
}
|
|
2028
|
+
/** Validate and detach one adapter-returned exact model result. */
|
|
2029
|
+
normalizeModelInfo(registration, model, resolved) {
|
|
2030
|
+
const provider = registration.provider.id;
|
|
2031
|
+
if (typeof resolved.provider !== "string" || resolved.provider !== provider || typeof resolved.id !== "string" || resolved.id !== model || typeof resolved.name !== "string" || resolved.name.length === 0 || resolved.description !== void 0 && typeof resolved.description !== "string") throw new LlmError(`adapter returned invalid exact model metadata for provider "${provider}" model "${model}"`, "INVALID_MODEL_INFO");
|
|
2032
|
+
const context = resolved.context;
|
|
2033
|
+
if (context !== void 0 && (!Number.isInteger(context.contextWindow) || context.contextWindow <= 0)) throw new LlmError(`adapter returned invalid context metadata for provider "${provider}" model "${model}"`, "INVALID_MODEL_CONTEXT");
|
|
2034
|
+
const inputModalities = this.detachedModalities(resolved.inputModalities);
|
|
2035
|
+
const defaultMaxTokens = resolved.defaultMaxTokens;
|
|
2036
|
+
if (defaultMaxTokens !== void 0 && (!Number.isSafeInteger(defaultMaxTokens) || defaultMaxTokens <= 0)) throw new LlmError(`adapter returned invalid default maxTokens for provider "${provider}" model "${model}"`, "INVALID_MODEL_MAX_TOKENS");
|
|
2037
|
+
const info = {
|
|
2038
|
+
provider,
|
|
2039
|
+
id: model,
|
|
2040
|
+
name: resolved.name,
|
|
2041
|
+
...resolved.description === void 0 ? {} : { description: resolved.description },
|
|
2042
|
+
...inputModalities === void 0 ? {} : { inputModalities },
|
|
2043
|
+
...context === void 0 ? {} : { context: { contextWindow: context.contextWindow } },
|
|
2044
|
+
...defaultMaxTokens === void 0 ? {} : { defaultMaxTokens }
|
|
2045
|
+
};
|
|
2046
|
+
const reasoning = resolved.reasoning;
|
|
2047
|
+
if (reasoning === void 0) return info;
|
|
2048
|
+
if (reasoning.efforts.length === 0) throw new LlmError(`adapter returned invalid reasoning metadata for provider "${provider}" model "${model}"`, "INVALID_MODEL_REASONING");
|
|
2049
|
+
const seen = /* @__PURE__ */ new Set();
|
|
2050
|
+
const efforts = reasoning.efforts.map((effort) => {
|
|
2051
|
+
if (typeof effort.id !== "string" || effort.id.length === 0 || typeof effort.name !== "string" || effort.name.length === 0 || effort.description !== void 0 && typeof effort.description !== "string" || seen.has(effort.id)) throw new LlmError(`adapter returned invalid or duplicate reasoning effort metadata for provider "${provider}" model "${model}"`, "INVALID_MODEL_REASONING");
|
|
2052
|
+
seen.add(effort.id);
|
|
2053
|
+
return {
|
|
2054
|
+
id: effort.id,
|
|
2055
|
+
name: effort.name,
|
|
2056
|
+
...effort.description === void 0 ? {} : { description: effort.description }
|
|
2057
|
+
};
|
|
2058
|
+
});
|
|
2059
|
+
if (reasoning.defaultEffort !== void 0 && !seen.has(reasoning.defaultEffort)) throw new LlmError(`adapter returned an unknown default reasoning effort for provider "${provider}" model "${model}"`, "INVALID_MODEL_REASONING");
|
|
2060
|
+
return {
|
|
2061
|
+
...info,
|
|
2062
|
+
reasoning: {
|
|
2063
|
+
efforts,
|
|
2064
|
+
...reasoning.defaultEffort === void 0 ? {} : { defaultEffort: reasoning.defaultEffort }
|
|
2065
|
+
}
|
|
2066
|
+
};
|
|
2067
|
+
}
|
|
2068
|
+
/**
|
|
2069
|
+
* Validate a conversation call config against its exact model capability and
|
|
2070
|
+
* materialize adapter-configured defaults. Unsupported explicit efforts
|
|
2071
|
+
* reject before provider I/O; no clamping or aliasing is performed. This
|
|
2072
|
+
* standalone query does not bind a later dispatch; use {@link prepareCall}
|
|
2073
|
+
* when logging and streaming must share one adapter registration.
|
|
2074
|
+
* @param config - provider/model route and optional request controls.
|
|
2075
|
+
* @param signal - optional cancellation for adapter-owned capability lookup.
|
|
2076
|
+
* @returns a detached config only when a default must be materialized.
|
|
2077
|
+
*/
|
|
2078
|
+
async resolveCallConfig(config, signal) {
|
|
2079
|
+
return (await this.resolveCallFor(this.registration(config.provider), config, signal)).config;
|
|
2080
|
+
}
|
|
2081
|
+
async resolveCallFor(registration, config, signal) {
|
|
2082
|
+
const info = await this.resolveModelInfoFor(registration, config.model, signal);
|
|
2083
|
+
return this.resolveCallWithInfo(config, info);
|
|
2084
|
+
}
|
|
2085
|
+
/** Validate request controls against one already-bound exact model result. */
|
|
2086
|
+
resolveCallWithInfo(config, info) {
|
|
2087
|
+
const defaulted = config.maxTokens === void 0 && info.defaultMaxTokens !== void 0 ? {
|
|
2088
|
+
...config,
|
|
2089
|
+
maxTokens: info.defaultMaxTokens
|
|
2090
|
+
} : config;
|
|
2091
|
+
const reasoning = info.reasoning;
|
|
2092
|
+
const requested = defaulted.reasoningEffort;
|
|
2093
|
+
let resolvedConfig = defaulted;
|
|
2094
|
+
if (reasoning === void 0) {
|
|
2095
|
+
if (requested !== void 0) throw new LlmError(`provider "${config.provider}" model "${config.model}" does not support reasoning effort "${requested}"`, "UNSUPPORTED_REASONING_EFFORT");
|
|
2096
|
+
} else {
|
|
2097
|
+
const effective = requested ?? reasoning.defaultEffort;
|
|
2098
|
+
if (effective !== void 0) {
|
|
2099
|
+
if (!reasoning.efforts.some((effort) => effort.id === effective)) throw new LlmError(`provider "${config.provider}" model "${config.model}" does not support reasoning effort "${effective}"`, "UNSUPPORTED_REASONING_EFFORT");
|
|
2100
|
+
if (requested !== effective) resolvedConfig = {
|
|
2101
|
+
...defaulted,
|
|
2102
|
+
reasoningEffort: effective
|
|
2103
|
+
};
|
|
2104
|
+
}
|
|
2105
|
+
}
|
|
2106
|
+
return {
|
|
2107
|
+
config: resolvedConfig,
|
|
2108
|
+
...info.context === void 0 ? {} : { context: info.context },
|
|
2109
|
+
modelInfo: info
|
|
2110
|
+
};
|
|
2111
|
+
}
|
|
2112
|
+
/**
|
|
2113
|
+
* Resolve one call under its current adapter registration. The returned
|
|
2114
|
+
* one-shot handle keeps that registration across header logging and dispatch,
|
|
2115
|
+
* so HMR cannot combine one adapter's capability result with another adapter.
|
|
2116
|
+
* @param config - provider/model route and optional request controls.
|
|
2117
|
+
* @param signal - optional cancellation for adapter-owned capability lookup.
|
|
2118
|
+
* @returns a prepared config and its registration-bound stream entry point.
|
|
2119
|
+
*/
|
|
2120
|
+
async prepareCall(config, signal) {
|
|
2121
|
+
const registration = this.registration(config.provider);
|
|
2122
|
+
const adapterCall = await registration.adapter.prepareCall(config.provider, config.model, signal);
|
|
2123
|
+
const modelInfo = this.normalizeModelInfo(registration, config.model, adapterCall.model);
|
|
2124
|
+
const resolved = this.resolveCallWithInfo(config, modelInfo);
|
|
2125
|
+
const resolvedConfig = deepFreeze(structuredClone(resolved.config));
|
|
2126
|
+
const context = resolved.context === void 0 ? void 0 : deepFreeze(structuredClone(resolved.context));
|
|
2127
|
+
const adapterDefaults = deepFreeze({
|
|
2128
|
+
...config.reasoningEffort === void 0 && resolvedConfig.reasoningEffort !== void 0 ? { reasoningEffort: true } : {},
|
|
2129
|
+
...config.maxTokens === void 0 && resolvedConfig.maxTokens !== void 0 ? { maxTokens: true } : {}
|
|
2130
|
+
});
|
|
2131
|
+
let dispatched = false;
|
|
2132
|
+
return Object.freeze({
|
|
2133
|
+
config: resolvedConfig,
|
|
2134
|
+
retryPolicy: registration.retryPolicy,
|
|
2135
|
+
adapterDefaults,
|
|
2136
|
+
...context === void 0 ? {} : { context },
|
|
2137
|
+
...modelInfo.inputModalities === void 0 ? {} : { inputModalities: Object.freeze([...modelInfo.inputModalities]) },
|
|
2138
|
+
stream: (options) => {
|
|
2139
|
+
if (dispatched) throw new LlmError("a prepared LLM call can only be dispatched once", "INVALID_PREPARED_CALL");
|
|
2140
|
+
if (!callConfigEquals(options, resolvedConfig)) throw new LlmError("prepared LLM call config changed before adapter dispatch", "INVALID_PREPARED_CALL");
|
|
2141
|
+
dispatched = true;
|
|
2142
|
+
return this.streamWithRegistration(options, {
|
|
2143
|
+
registration,
|
|
2144
|
+
config: resolvedConfig,
|
|
2145
|
+
modelInfo,
|
|
2146
|
+
dispatch: (options) => adapterCall.stream(options)
|
|
2147
|
+
});
|
|
2148
|
+
}
|
|
2149
|
+
});
|
|
2150
|
+
}
|
|
2151
|
+
registration(provider) {
|
|
2152
|
+
const registration = this.adapters.get(provider);
|
|
2153
|
+
if (!registration) throw new LlmError(`no adapter registered for provider "${provider}"`, "NO_ADAPTER");
|
|
2154
|
+
return registration;
|
|
2155
|
+
}
|
|
2156
|
+
/** Remove replay state whose historical route is owned by another adapter. */
|
|
2157
|
+
forAdapter(options, adapter) {
|
|
2158
|
+
const messages = options.messages.map((message) => {
|
|
2159
|
+
const source = message.source;
|
|
2160
|
+
if (message.role !== "assistant" || source.kind !== "model" || source.replayState === void 0) return message;
|
|
2161
|
+
if (this.adapters.get(source.provider)?.adapter === adapter) return message;
|
|
2162
|
+
return freezeMessage({
|
|
2163
|
+
...message,
|
|
2164
|
+
source: {
|
|
2165
|
+
kind: "model",
|
|
2166
|
+
provider: source.provider,
|
|
2167
|
+
model: source.model
|
|
2168
|
+
}
|
|
2169
|
+
});
|
|
2170
|
+
});
|
|
2171
|
+
if (messages.every((message, index) => message === options.messages[index])) return options;
|
|
2172
|
+
const filtered = {
|
|
2173
|
+
...options,
|
|
2174
|
+
messages
|
|
2175
|
+
};
|
|
2176
|
+
return Object.isFrozen(options) ? deepFreeze(filtered) : filtered;
|
|
2177
|
+
}
|
|
2178
|
+
/**
|
|
2179
|
+
* Resolve the current execution-world read path of one durable file
|
|
2180
|
+
* reference through the mounted attachment and filesystem providers.
|
|
2181
|
+
*/
|
|
2182
|
+
fileReadPath(ref) {
|
|
2183
|
+
let hostPath;
|
|
2184
|
+
try {
|
|
2185
|
+
hostPath = this.ctx.get("attachments")?.fileHostPath(ref);
|
|
2186
|
+
} catch {
|
|
2187
|
+
return;
|
|
2188
|
+
}
|
|
2189
|
+
if (hostPath === void 0) return void 0;
|
|
2190
|
+
return this.ctx.get("fs")?.processPathFromHostPath(hostPath);
|
|
2191
|
+
}
|
|
2192
|
+
/**
|
|
2193
|
+
* Final adapter boundary. Adapter selection, dispatch, iterator construction,
|
|
2194
|
+
* and iteration failures become one terminal failure chunk. Middleware and
|
|
2195
|
+
* downstream consumer failures remain thrown plugin or consumer errors.
|
|
2196
|
+
*/
|
|
2197
|
+
async *adapterStream(options, prepared) {
|
|
2198
|
+
let iterator;
|
|
2199
|
+
try {
|
|
2200
|
+
const registration = prepared?.registration ?? this.registration(options.provider);
|
|
2201
|
+
const adapter = registration.adapter;
|
|
2202
|
+
let modelInfo;
|
|
2203
|
+
let resolvedConfig;
|
|
2204
|
+
let dispatch;
|
|
2205
|
+
if (prepared === void 0) {
|
|
2206
|
+
const adapterCall = await adapter.prepareCall(options.provider, options.model, options.signal);
|
|
2207
|
+
modelInfo = this.normalizeModelInfo(registration, options.model, adapterCall.model);
|
|
2208
|
+
resolvedConfig = this.resolveCallWithInfo(options, modelInfo).config;
|
|
2209
|
+
dispatch = (options) => adapterCall.stream(options);
|
|
2210
|
+
} else {
|
|
2211
|
+
modelInfo = prepared.modelInfo;
|
|
2212
|
+
resolvedConfig = prepared.config;
|
|
2213
|
+
dispatch = prepared.dispatch;
|
|
2214
|
+
}
|
|
2215
|
+
if (prepared !== void 0 && !callConfigEquals(options, resolvedConfig)) throw new LlmError("prepared LLM call config changed before adapter dispatch", "INVALID_PREPARED_CALL");
|
|
2216
|
+
const resolvedOptions = callConfigEquals(options, resolvedConfig) ? options : Object.isFrozen(options) ? deepFreeze({
|
|
2217
|
+
...options,
|
|
2218
|
+
...resolvedConfig
|
|
2219
|
+
}) : {
|
|
2220
|
+
...options,
|
|
2221
|
+
...resolvedConfig
|
|
2222
|
+
};
|
|
2223
|
+
let projectedMessages = resolvedOptions.messages;
|
|
2224
|
+
if (projectedMessages.some((message) => contentHasFile(message.content))) projectedMessages = projectFilesToText(projectedMessages, (ref) => this.fileReadPath(ref));
|
|
2225
|
+
if (modelInfo.inputModalities !== void 0 && !modelInfo.inputModalities.includes("image") && projectedMessages.some((message) => contentHasImage(message.content))) projectedMessages = projectImagesForTextModel(projectedMessages);
|
|
2226
|
+
const projectedOptions = projectedMessages === resolvedOptions.messages ? resolvedOptions : Object.isFrozen(resolvedOptions) ? deepFreeze({
|
|
2227
|
+
...resolvedOptions,
|
|
2228
|
+
messages: projectedMessages
|
|
2229
|
+
}) : {
|
|
2230
|
+
...resolvedOptions,
|
|
2231
|
+
messages: projectedMessages
|
|
2232
|
+
};
|
|
2233
|
+
iterator = dispatch(this.forAdapter(projectedOptions, adapter))[Symbol.asyncIterator]();
|
|
2234
|
+
} catch (error) {
|
|
2235
|
+
yield adapterFailureChunk(error, options.signal);
|
|
2236
|
+
return;
|
|
2237
|
+
}
|
|
2238
|
+
let completed = false;
|
|
2239
|
+
try {
|
|
2240
|
+
while (true) {
|
|
2241
|
+
let item;
|
|
2242
|
+
try {
|
|
2243
|
+
const next = await iterator.next();
|
|
2244
|
+
item = next.done ? { done: true } : {
|
|
2245
|
+
done: false,
|
|
2246
|
+
value: next.value
|
|
2247
|
+
};
|
|
2248
|
+
} catch (error) {
|
|
2249
|
+
completed = true;
|
|
2250
|
+
yield adapterFailureChunk(error, options.signal);
|
|
2251
|
+
return;
|
|
2252
|
+
}
|
|
2253
|
+
if (item.done) {
|
|
2254
|
+
completed = true;
|
|
2255
|
+
return;
|
|
2256
|
+
}
|
|
2257
|
+
yield item.value;
|
|
2258
|
+
}
|
|
2259
|
+
} finally {
|
|
2260
|
+
if (!completed) {
|
|
2261
|
+
const close = iterator.return?.bind(iterator);
|
|
2262
|
+
if (close) await close();
|
|
2263
|
+
}
|
|
2264
|
+
}
|
|
2265
|
+
}
|
|
2266
|
+
/**
|
|
2267
|
+
* Stream one model call as raw chunks (token-level deltas). Replay state is
|
|
2268
|
+
* retained only when the same adapter instance owns its historical provider
|
|
2269
|
+
* and the target provider. Final adapter selection remains fixed through
|
|
2270
|
+
* asynchronous exact-model resolution and dispatch. Adapter selection,
|
|
2271
|
+
* dispatch, and iteration failures become terminal `error` or `aborted`
|
|
2272
|
+
* finish chunks; middleware, nested-call, cleanup, and consumer failures
|
|
2273
|
+
* remain thrown.
|
|
2274
|
+
* @param options - the full request; `options.provider` selects the adapter.
|
|
2275
|
+
* @returns the chunk stream, possibly wrapped by `llm/stream` listeners.
|
|
2276
|
+
*/
|
|
2277
|
+
stream(options) {
|
|
2278
|
+
return this.streamWithRegistration(options);
|
|
2279
|
+
}
|
|
2280
|
+
streamWithRegistration(options, prepared) {
|
|
2281
|
+
return this.ctx.waterfall(this, "llm/stream", options, () => this.adapterStream(options, prepared));
|
|
2282
|
+
}
|
|
2283
|
+
};
|
|
2284
|
+
})();
|
|
2285
|
+
/** Convert one adapter throw into the stream protocol's terminal outcome. */
|
|
2286
|
+
function adapterFailureChunk(error, signal) {
|
|
2287
|
+
const failure = normalizeLlmFailure(error);
|
|
2288
|
+
return {
|
|
2289
|
+
type: "finish",
|
|
2290
|
+
reason: signal?.aborted || failure.code === "ABORTED" ? {
|
|
2291
|
+
kind: "aborted",
|
|
2292
|
+
failure
|
|
2293
|
+
} : {
|
|
2294
|
+
kind: "error",
|
|
2295
|
+
failure
|
|
2296
|
+
}
|
|
2297
|
+
};
|
|
2298
|
+
}
|
|
2299
|
+
//#endregion
|
|
2300
|
+
export { APP_IDENTITY, AssistantStreamAccumulator, BlockAssembler, CONTEXT_SUMMARY_MAX_CHARS, CONTEXT_WINDOW_EXCEEDED_CODE, EMPTY_RESPONSE_CODE, HarnessError, INVALID_CREDENTIAL_CODE, LlmAdapter, LlmAttemptId, LlmError, LlmRuntime, LlmRuntime as default, MessageId, ProviderRequestId, QUOTA_EXCEEDED_CODE, ReasoningEffortId, RetryPolicySchema, ToolCallId, assembleAssistantStream, assertUsableApiKey, assistantStreamChunks, assistantStreamFirstTokenTime, assistantStreamHasVisibleContent, assistantStreamHasVisibleText, attributionHeaders, boundContextSummary, callConfigEquals, chunkHasVisibleText, contentHasFile, contentHasImage, createAssistantMessage, createMessage, createToolResultMessage, createUserMessage, errorChain, expandAssistantStream, fileHandleText, freezeMessage, isAgentLoopRequest, isContextWindowExceededError, isHarnessError, isQuotaExceededError, isTokenDelta, isVisibleChunk, joinAssistantStreamText, lastAssistantStreamChunk, markAgentLoopRequest, normalizeApiKey, offloadRequestImagesWithPolicy, offloadedImagePrefixCount, offloadedImageText, projectFilesToText, projectImagesForTextModel, requestImageHandleText, resolveImageAttachmentAccess, resolveRetryPolicy, runFirstTokenTime, runFirstVisibleTime, textOnlyImageText, userAgent };
|