@zq-silk/yui 2.1.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +4 -0
- package/dist/cli/commandCatalog.js +24 -3
- package/dist/cli/commandDiscovery.js +4 -1
- package/dist/cli.js +23 -3078
- package/dist/commands/taskCommands.js +79 -21
- package/dist/commands/taskIntegrationCommands.js +3 -1
- package/dist/commands/taskUpstreamCommands.js +3 -1
- package/dist/context/runContextPack.js +132 -5
- package/dist/context/sourceRunContext.js +4 -2
- package/dist/context/taskContext.js +6 -1
- package/dist/controlPlaneCli.js +3092 -0
- package/dist/controller/fileSchedulerStoreAdapter.js +19 -6
- package/dist/controller/jobControl.js +21 -0
- package/dist/executor/agentExecutor.js +1 -1
- package/dist/executor/effectiveLaunch.js +18 -5
- package/dist/executor/fileRoleLaunchPlanner.js +15 -19
- package/dist/integration/gitIntegrationService.js +50 -15
- package/dist/message/messageContinuation.js +7 -2
- package/dist/nativeAgent/agent.js +176 -74
- package/dist/nativeAgent/cliDemo.js +37 -0
- package/dist/nativeAgent/codingTools.js +11 -0
- package/dist/nativeAgent/commandTool.js +215 -0
- package/dist/nativeAgent/compactionDemo.js +158 -0
- package/dist/nativeAgent/composition.js +71 -0
- package/dist/nativeAgent/context/budget.js +44 -0
- package/dist/nativeAgent/context/index.js +339 -0
- package/dist/nativeAgent/context/providerCompressor.js +103 -0
- package/dist/nativeAgent/demo.js +12 -1
- package/dist/nativeAgent/evaluation/cases.js +38 -0
- package/dist/nativeAgent/evaluation/checks.js +91 -0
- package/dist/nativeAgent/evaluation/demo.js +19 -0
- package/dist/nativeAgent/evaluation/files.js +54 -0
- package/dist/nativeAgent/evaluation/fixture.js +36 -0
- package/dist/nativeAgent/evaluation/index.js +239 -0
- package/dist/nativeAgent/executionOwner.js +209 -0
- package/dist/nativeAgent/filePatterns.js +170 -0
- package/dist/nativeAgent/fileToolsSupport.js +202 -0
- package/dist/nativeAgent/index.js +13 -0
- package/dist/nativeAgent/interaction/cli.js +358 -0
- package/dist/nativeAgent/interaction/contracts.js +1 -0
- package/dist/nativeAgent/interaction/index.js +3 -0
- package/dist/nativeAgent/interaction/memoryDemo.js +115 -0
- package/dist/nativeAgent/interaction/renderer.js +34 -0
- package/dist/nativeAgent/localSafety.js +258 -0
- package/dist/nativeAgent/model/anthropicMessages.js +204 -0
- package/dist/nativeAgent/model/chatCompletions.js +210 -0
- package/dist/nativeAgent/model/errors.js +47 -0
- package/dist/nativeAgent/model/gateway.js +423 -0
- package/dist/nativeAgent/model/index.js +7 -0
- package/dist/nativeAgent/model/observationAdapter.js +21 -0
- package/dist/nativeAgent/model/protocols.js +19 -0
- package/dist/nativeAgent/model/responses.js +263 -0
- package/dist/nativeAgent/model/types.js +1 -0
- package/dist/nativeAgent/model/wire.js +73 -0
- package/dist/nativeAgent/observability/index.js +220 -0
- package/dist/nativeAgent/product/catalog.js +82 -0
- package/dist/nativeAgent/product/config.js +295 -0
- package/dist/nativeAgent/product/facts.js +30 -0
- package/dist/nativeAgent/product/index.js +62 -0
- package/dist/nativeAgent/product/location.js +44 -0
- package/dist/nativeAgent/product/runtime.js +276 -0
- package/dist/nativeAgent/product/storage.js +49 -0
- package/dist/nativeAgent/product/tools.js +47 -0
- package/dist/nativeAgent/product/transport.js +54 -0
- package/dist/nativeAgent/projectGuidance/index.js +425 -0
- package/dist/nativeAgent/searchTools.js +305 -0
- package/dist/nativeAgent/session/backends.js +293 -0
- package/dist/nativeAgent/session/catalog.js +97 -0
- package/dist/nativeAgent/session/catalogDemo.js +87 -0
- package/dist/nativeAgent/session/contracts.js +1 -0
- package/dist/nativeAgent/session/format.js +269 -0
- package/dist/nativeAgent/session/index.js +5 -0
- package/dist/nativeAgent/session/location.js +36 -0
- package/dist/nativeAgent/session/sqliteFormat.js +134 -0
- package/dist/nativeAgent/session/store.js +248 -0
- package/dist/nativeAgent/textTools.js +270 -133
- package/dist/nativeAgent/toolManager/executor.js +290 -0
- package/dist/nativeAgent/toolManager/index.js +4 -0
- package/dist/nativeAgent/validation.js +2 -2
- package/dist/task/taskAuthority.js +56 -0
- package/dist/web/assets/client/app.js +64 -3
- package/dist/web/assets/client/detail.js +6 -6
- package/dist/web/assets/client/i18n.js +4 -0
- package/dist/web/assets/client/overview.js +12 -9
- package/dist/web/assets/client/sidebar.js +31 -2
- package/dist/web/assets/shell.js +2 -1
- package/dist/web/assets/styles/components.js +2 -0
- package/dist/web/assets/styles/layout.js +11 -4
- package/dist/web/assets/styles/responsive.js +12 -4
- package/dist/web/assets/styles/views.js +16 -13
- package/docs/agent-result-consumption.md +16 -0
- package/docs/agent-result-consumption.zh-CN.md +13 -0
- package/docs/examples/agent-offline.mjs +194 -0
- package/docs/native-agent.md +283 -0
- package/docs/release-workflow.md +47 -9
- package/docs/release-workflow.zh-CN.md +36 -6
- package/docs/roles-and-configuration.md +32 -0
- package/docs/roles-and-configuration.zh-CN.md +26 -0
- package/package.json +1 -1
- package/skills/yui-leader/SKILL.md +9 -0
- package/skills/yui-reviewer/SKILL.md +5 -0
- package/skills/yui-runtime/SKILL.md +7 -0
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
const messages = {
|
|
2
|
+
configuration: 'Invalid explicit model configuration',
|
|
3
|
+
request: 'Invalid or oversized model request',
|
|
4
|
+
protocol: 'Invalid or oversized model response',
|
|
5
|
+
incomplete: 'Model response did not finish completely',
|
|
6
|
+
authentication: 'Model request authentication or access rejected',
|
|
7
|
+
quota: 'Model account quota rejected',
|
|
8
|
+
rate_limit: 'Temporary model rate limit rejection',
|
|
9
|
+
http: 'Model HTTP request failed; not eligible for automatic replay',
|
|
10
|
+
transport: 'Model transport failed; remote effects unknown',
|
|
11
|
+
cancelled: 'Model request cancelled; remote cancellation is not rollback',
|
|
12
|
+
deadline: 'Model request deadline exceeded; remote effects may be unknown',
|
|
13
|
+
};
|
|
14
|
+
/** No raw cause, URL, request/response bodies, headers, credentials or provider message. */
|
|
15
|
+
export class ModelGatewayError extends Error {
|
|
16
|
+
code;
|
|
17
|
+
effect;
|
|
18
|
+
stopReason;
|
|
19
|
+
requestId;
|
|
20
|
+
source = 'live';
|
|
21
|
+
attempts;
|
|
22
|
+
constructor(code, effect = 'unknown', attempts = [], stopReason, requestId) {
|
|
23
|
+
super(messages[code]);
|
|
24
|
+
this.code = code;
|
|
25
|
+
this.effect = effect;
|
|
26
|
+
this.stopReason = stopReason;
|
|
27
|
+
this.requestId = requestId;
|
|
28
|
+
this.name = 'ModelGatewayError';
|
|
29
|
+
this.attempts = Object.freeze(attempts.map(a => Object.freeze({
|
|
30
|
+
...a, ...(a.usage ? { usage: Object.freeze({ ...a.usage }) } : {}),
|
|
31
|
+
})));
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
export function protocol() { throw new ModelGatewayError('protocol'); }
|
|
35
|
+
export function object(value) {
|
|
36
|
+
if (!value || typeof value !== 'object' || Array.isArray(value))
|
|
37
|
+
return protocol();
|
|
38
|
+
return value;
|
|
39
|
+
}
|
|
40
|
+
export function parse(text) {
|
|
41
|
+
try {
|
|
42
|
+
return JSON.parse(text);
|
|
43
|
+
}
|
|
44
|
+
catch {
|
|
45
|
+
return protocol();
|
|
46
|
+
}
|
|
47
|
+
}
|
|
@@ -0,0 +1,423 @@
|
|
|
1
|
+
import { performance } from 'node:perf_hooks';
|
|
2
|
+
import { randomUUID } from 'node:crypto';
|
|
3
|
+
import { setTimeout as delay } from 'node:timers/promises';
|
|
4
|
+
import { ModelGatewayError } from './errors.js';
|
|
5
|
+
import { createProtocolAdapter, getProtocolCapabilities } from './protocols.js';
|
|
6
|
+
export const fetchTransport = (endpoint, init) => fetch(endpoint, init);
|
|
7
|
+
const bytes = (v) => Buffer.byteLength(JSON.stringify(v));
|
|
8
|
+
function frozen(v) {
|
|
9
|
+
const copy = structuredClone(v);
|
|
10
|
+
const freeze = (x) => {
|
|
11
|
+
if (x && typeof x === 'object') {
|
|
12
|
+
Object.values(x).forEach(freeze);
|
|
13
|
+
Object.freeze(x);
|
|
14
|
+
}
|
|
15
|
+
};
|
|
16
|
+
freeze(copy);
|
|
17
|
+
return copy;
|
|
18
|
+
}
|
|
19
|
+
function plain(v) {
|
|
20
|
+
return v !== null && typeof v === 'object' && !Array.isArray(v) && Object.getPrototypeOf(v) === Object.prototype;
|
|
21
|
+
}
|
|
22
|
+
function json(v, depth = 0) {
|
|
23
|
+
if (depth > 32)
|
|
24
|
+
return false;
|
|
25
|
+
if (v === null || typeof v === 'boolean' || typeof v === 'string')
|
|
26
|
+
return true;
|
|
27
|
+
if (typeof v === 'number')
|
|
28
|
+
return Number.isFinite(v);
|
|
29
|
+
return (Array.isArray(v) || plain(v)) && Object.values(v).every(child => json(child, depth + 1));
|
|
30
|
+
}
|
|
31
|
+
const nonempty = (v) => typeof v === 'string' && v.trim().length > 0;
|
|
32
|
+
const keys = (v, allowed) => Object.keys(v).every(k => allowed.includes(k));
|
|
33
|
+
function validCalls(v) {
|
|
34
|
+
return Array.isArray(v) && v.length <= 8 && v.every(c => plain(c) && nonempty(c.id) && nonempty(c.name)
|
|
35
|
+
&& keys(c, ['id', 'name', 'arguments'])
|
|
36
|
+
&& plain(c.arguments) && json(c.arguments) && bytes(c.arguments) <= 64 * 1024)
|
|
37
|
+
&& new Set(v.map(c => c.id)).size === v.length;
|
|
38
|
+
}
|
|
39
|
+
function requestSnapshot(input) {
|
|
40
|
+
try {
|
|
41
|
+
if (!plain(input) || !keys(input, ['sessionId', 'turnId', 'step', 'messages', 'tools'])
|
|
42
|
+
|| !json(input) || bytes(input) > 1024 * 1024 || !nonempty(input.sessionId) || !nonempty(input.turnId)
|
|
43
|
+
|| !Number.isSafeInteger(input.step) || input.step < 1 || !Array.isArray(input.messages)
|
|
44
|
+
|| !Array.isArray(input.tools))
|
|
45
|
+
throw 0;
|
|
46
|
+
const used = new Set(), pending = new Map();
|
|
47
|
+
for (const m of input.messages) {
|
|
48
|
+
if (!plain(m) || bytes(m) > 512 * 1024)
|
|
49
|
+
throw 0;
|
|
50
|
+
if (m.role === 'tool') {
|
|
51
|
+
if (!keys(m, ['role', 'toolCallId', 'name', 'outcome']))
|
|
52
|
+
throw 0;
|
|
53
|
+
if (!nonempty(m.toolCallId) || !pending.has(m.toolCallId) || pending.get(m.toolCallId) !== m.name || !plain(m.outcome))
|
|
54
|
+
throw 0;
|
|
55
|
+
if (m.outcome.ok === true) {
|
|
56
|
+
if (!keys(m.outcome, ['ok', 'content']) || typeof m.outcome.content !== 'string')
|
|
57
|
+
throw 0;
|
|
58
|
+
}
|
|
59
|
+
else if (m.outcome.ok !== false || !keys(m.outcome, ['ok', 'error']) || !plain(m.outcome.error)
|
|
60
|
+
|| !keys(m.outcome.error, ['code', 'message', 'effect']) || !nonempty(m.outcome.error.code)
|
|
61
|
+
|| typeof m.outcome.error.message !== 'string'
|
|
62
|
+
|| (m.outcome.error.effect !== 'none' && m.outcome.error.effect !== 'unknown'))
|
|
63
|
+
throw 0;
|
|
64
|
+
pending.delete(m.toolCallId);
|
|
65
|
+
}
|
|
66
|
+
else {
|
|
67
|
+
if (pending.size || typeof m.content !== 'string')
|
|
68
|
+
throw 0;
|
|
69
|
+
if (m.role === 'assistant') {
|
|
70
|
+
if (!keys(m, ['role', 'content', 'toolCalls']))
|
|
71
|
+
throw 0;
|
|
72
|
+
if (!validCalls(m.toolCalls))
|
|
73
|
+
throw 0;
|
|
74
|
+
for (const c of m.toolCalls) {
|
|
75
|
+
if (used.has(c.id))
|
|
76
|
+
throw 0;
|
|
77
|
+
used.add(c.id);
|
|
78
|
+
pending.set(c.id, c.name);
|
|
79
|
+
}
|
|
80
|
+
}
|
|
81
|
+
else if ((m.role !== 'system' && m.role !== 'user') || !keys(m, ['role', 'content']))
|
|
82
|
+
throw 0;
|
|
83
|
+
}
|
|
84
|
+
}
|
|
85
|
+
if (pending.size || !input.tools.every(t => plain(t) && nonempty(t.name)
|
|
86
|
+
&& keys(t, ['name', 'description', 'inputSchema']) && typeof t.description === 'string' && plain(t.inputSchema))
|
|
87
|
+
|| new Set(input.tools.map(t => t.name)).size !== input.tools.length)
|
|
88
|
+
throw 0;
|
|
89
|
+
return frozen(input);
|
|
90
|
+
}
|
|
91
|
+
catch {
|
|
92
|
+
throw new ModelGatewayError('request', 'none');
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
function validateResponse(response, request) {
|
|
96
|
+
const used = new Set(request.messages.flatMap(m => m.role === 'assistant' ? m.toolCalls.map(c => c.id) : []));
|
|
97
|
+
const names = new Set(request.tools.map(t => t.name));
|
|
98
|
+
if (!plain(response) || !json(response) || bytes(response) > 512 * 1024 || typeof response.content !== 'string'
|
|
99
|
+
|| !(response.kind === 'final' || (response.kind === 'tool_calls' && validCalls(response.calls)
|
|
100
|
+
&& response.calls.length > 0 && response.calls.every(c => names.has(c.name) && !used.has(c.id))))) {
|
|
101
|
+
throw new ModelGatewayError('protocol');
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
/** Owns the reader on every exit, including early DONE, parse failure, cancellation and size limit. */
|
|
105
|
+
async function* readBody(response, signal, limit) {
|
|
106
|
+
if (!response.body)
|
|
107
|
+
throw new ModelGatewayError('protocol');
|
|
108
|
+
const reader = response.body.getReader();
|
|
109
|
+
const abort = () => { void reader.cancel().catch(() => { }); };
|
|
110
|
+
signal.addEventListener('abort', abort, { once: true });
|
|
111
|
+
const decoder = new TextDecoder('utf-8', { fatal: true });
|
|
112
|
+
let count = 0;
|
|
113
|
+
try {
|
|
114
|
+
while (true) {
|
|
115
|
+
signal.throwIfAborted();
|
|
116
|
+
const part = await reader.read();
|
|
117
|
+
signal.throwIfAborted();
|
|
118
|
+
if (part.done)
|
|
119
|
+
break;
|
|
120
|
+
count += part.value.byteLength;
|
|
121
|
+
if (count > limit)
|
|
122
|
+
throw new ModelGatewayError('protocol');
|
|
123
|
+
let decoded;
|
|
124
|
+
try {
|
|
125
|
+
decoded = decoder.decode(part.value, { stream: true });
|
|
126
|
+
}
|
|
127
|
+
catch {
|
|
128
|
+
throw new ModelGatewayError('protocol');
|
|
129
|
+
}
|
|
130
|
+
yield decoded;
|
|
131
|
+
}
|
|
132
|
+
try {
|
|
133
|
+
yield decoder.decode();
|
|
134
|
+
}
|
|
135
|
+
catch {
|
|
136
|
+
throw new ModelGatewayError('protocol');
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
finally {
|
|
140
|
+
signal.removeEventListener('abort', abort);
|
|
141
|
+
try {
|
|
142
|
+
await reader.cancel();
|
|
143
|
+
}
|
|
144
|
+
finally {
|
|
145
|
+
reader.releaseLock();
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
function retryAfter(value) {
|
|
150
|
+
if (value === null)
|
|
151
|
+
return 0;
|
|
152
|
+
if (/^\d+(\.\d+)?$/.test(value.trim()))
|
|
153
|
+
return Number(value) * 1000;
|
|
154
|
+
const date = Date.parse(value);
|
|
155
|
+
return Number.isFinite(date) ? Math.max(0, date - Date.now()) : 0;
|
|
156
|
+
}
|
|
157
|
+
function reportedUsage(value) {
|
|
158
|
+
if (!plain(value))
|
|
159
|
+
throw new ModelGatewayError('protocol');
|
|
160
|
+
const usage = {};
|
|
161
|
+
for (const key of ['inputTokens', 'outputTokens', 'totalTokens', 'cachedInputTokens', 'cacheWriteInputTokens']) {
|
|
162
|
+
const n = value[key];
|
|
163
|
+
if (n === undefined)
|
|
164
|
+
continue;
|
|
165
|
+
if (typeof n !== 'number' || !Number.isSafeInteger(n) || n < 0)
|
|
166
|
+
throw new ModelGatewayError('protocol');
|
|
167
|
+
usage[key] = n;
|
|
168
|
+
}
|
|
169
|
+
if (!Object.keys(usage).length || !keys(value, Object.keys(usage)))
|
|
170
|
+
throw new ModelGatewayError('protocol');
|
|
171
|
+
return usage;
|
|
172
|
+
}
|
|
173
|
+
export function createModelGateway(options) {
|
|
174
|
+
let endpoint, headers;
|
|
175
|
+
const adapter = options.adapter ?? createProtocolAdapter(options.protocol ?? 'chat-completions');
|
|
176
|
+
const protocol = adapter.protocol ?? 'custom';
|
|
177
|
+
const capabilities = { ...(protocol === 'custom' ? { text: true, functionTools: true, streaming: true }
|
|
178
|
+
: getProtocolCapabilities(protocol)), ...options.modelCapabilities };
|
|
179
|
+
let generation;
|
|
180
|
+
const retry = { maxAttempts: 3, maxElapsedMs: 30_000, baseDelayMs: 250, ...options.retry };
|
|
181
|
+
try {
|
|
182
|
+
if (!plain(options) || !keys(options, ['endpoint', 'model', 'account', 'stream', 'adapter', 'transport',
|
|
183
|
+
'onObservation', 'retry', 'clock', 'protocol', 'generation', 'capacity', 'modelCapabilities'])
|
|
184
|
+
|| (options.protocol !== undefined && options.protocol !== protocol)
|
|
185
|
+
|| (options.generation !== undefined && (!plain(options.generation) || !keys(options.generation, ['maxOutputTokens'])))
|
|
186
|
+
|| (options.capacity !== undefined && (!plain(options.capacity)
|
|
187
|
+
|| !keys(options.capacity, ['contextWindowTokens', 'maxOutputTokens'])))
|
|
188
|
+
|| (options.modelCapabilities !== undefined && (!plain(options.modelCapabilities)
|
|
189
|
+
|| !keys(options.modelCapabilities, ['text', 'functionTools', 'streaming'])))
|
|
190
|
+
|| !Object.values(capabilities).every(v => typeof v === 'boolean')
|
|
191
|
+
|| !capabilities.text || (options.stream && !capabilities.streaming)
|
|
192
|
+
|| !plain(options.account) || !keys(options.account, options.account.kind === 'none' ? ['kind'] : ['kind', 'token']))
|
|
193
|
+
throw 0;
|
|
194
|
+
const bounds = [...Object.values(options.generation ?? {}), ...Object.values(options.capacity ?? {})];
|
|
195
|
+
if (!bounds.every(n => n === undefined || (typeof n === 'number' && Number.isSafeInteger(n) && n > 0)))
|
|
196
|
+
throw 0;
|
|
197
|
+
generation = frozen(options.generation ?? {});
|
|
198
|
+
if ((protocol === 'custom' && Object.keys(generation).length)
|
|
199
|
+
|| (protocol === 'anthropic-messages' && generation.maxOutputTokens === undefined)
|
|
200
|
+
|| (generation.maxOutputTokens !== undefined && options.capacity?.maxOutputTokens !== undefined
|
|
201
|
+
&& generation.maxOutputTokens > options.capacity.maxOutputTokens))
|
|
202
|
+
throw 0;
|
|
203
|
+
const url = new URL(options.endpoint);
|
|
204
|
+
if (url.username || url.password || url.hash || url.search
|
|
205
|
+
|| !(url.protocol === 'https:' || (url.protocol === 'http:' && ['localhost', '127.0.0.1', '[::1]'].includes(url.hostname)))
|
|
206
|
+
|| !nonempty(options.model) || options.model.length > 256
|
|
207
|
+
|| (options.stream !== undefined && typeof options.stream !== 'boolean')
|
|
208
|
+
|| !Number.isInteger(retry.maxAttempts) || retry.maxAttempts < 1 || retry.maxAttempts > 10
|
|
209
|
+
|| !Number.isInteger(retry.maxElapsedMs) || retry.maxElapsedMs < 1 || retry.maxElapsedMs > 300_000
|
|
210
|
+
|| !Number.isInteger(retry.baseDelayMs) || retry.baseDelayMs < 1 || retry.baseDelayMs > 30_000)
|
|
211
|
+
throw 0;
|
|
212
|
+
headers = { 'content-type': 'application/json', accept: options.stream ? 'text/event-stream' : 'application/json' };
|
|
213
|
+
if (options.account.kind === 'bearer') {
|
|
214
|
+
if (!nonempty(options.account.token) || /[\r\n]/.test(options.account.token))
|
|
215
|
+
throw 0;
|
|
216
|
+
if (protocol === 'anthropic-messages')
|
|
217
|
+
throw 0;
|
|
218
|
+
headers.authorization = `Bearer ${options.account.token}`;
|
|
219
|
+
}
|
|
220
|
+
else if (options.account.kind === 'api-key') {
|
|
221
|
+
if (protocol !== 'anthropic-messages' || !nonempty(options.account.token) || /[\r\n]/.test(options.account.token))
|
|
222
|
+
throw 0;
|
|
223
|
+
headers['x-api-key'] = options.account.token;
|
|
224
|
+
}
|
|
225
|
+
else if (options.account.kind !== 'none')
|
|
226
|
+
throw 0;
|
|
227
|
+
if (protocol === 'anthropic-messages')
|
|
228
|
+
headers['anthropic-version'] = '2023-06-01';
|
|
229
|
+
endpoint = url.href;
|
|
230
|
+
}
|
|
231
|
+
catch {
|
|
232
|
+
throw new ModelGatewayError('configuration', 'none');
|
|
233
|
+
}
|
|
234
|
+
const model = options.model, streaming = options.stream ?? false;
|
|
235
|
+
const profile = frozen({ protocol, model, capabilities, ...(options.capacity ? { capacity: options.capacity } : {}) });
|
|
236
|
+
const transport = options.transport ?? fetchTransport;
|
|
237
|
+
const observe = options.onObservation;
|
|
238
|
+
const credential = options.account.kind !== 'none' ? options.account.token : undefined;
|
|
239
|
+
const clock = options.clock ?? { now: () => performance.now(), sleep: async (ms, signal) => {
|
|
240
|
+
await delay(ms, undefined, { signal });
|
|
241
|
+
} };
|
|
242
|
+
async function generate(input, external) {
|
|
243
|
+
const requestId = randomUUID(), attempts = [];
|
|
244
|
+
let request;
|
|
245
|
+
try {
|
|
246
|
+
request = requestSnapshot(input);
|
|
247
|
+
if (!capabilities.functionTools && (request.tools.length
|
|
248
|
+
|| request.messages.some(m => m.role === 'tool' || (m.role === 'assistant' && m.toolCalls.length))))
|
|
249
|
+
throw 0;
|
|
250
|
+
}
|
|
251
|
+
catch {
|
|
252
|
+
throw new ModelGatewayError('request', 'none', [], undefined, requestId);
|
|
253
|
+
}
|
|
254
|
+
const controller = new AbortController();
|
|
255
|
+
const cancel = () => controller.abort();
|
|
256
|
+
external.addEventListener('abort', cancel, { once: true });
|
|
257
|
+
if (external.aborted)
|
|
258
|
+
cancel();
|
|
259
|
+
const start = clock.now();
|
|
260
|
+
const timer = setTimeout(cancel, retry.maxElapsedMs);
|
|
261
|
+
let sent = false;
|
|
262
|
+
const elapsed = () => Math.max(0, clock.now() - start);
|
|
263
|
+
const check = () => {
|
|
264
|
+
if (external.aborted)
|
|
265
|
+
throw new ModelGatewayError('cancelled', sent ? 'unknown' : 'none');
|
|
266
|
+
if (controller.signal.aborted || elapsed() >= retry.maxElapsedMs) {
|
|
267
|
+
controller.abort();
|
|
268
|
+
throw new ModelGatewayError('deadline', sent ? 'unknown' : 'none');
|
|
269
|
+
}
|
|
270
|
+
};
|
|
271
|
+
const emit = (attempt, data) => {
|
|
272
|
+
if (!observe)
|
|
273
|
+
return;
|
|
274
|
+
try {
|
|
275
|
+
void Promise.resolve(observe(frozen({ sessionId: request.sessionId, turnId: request.turnId,
|
|
276
|
+
step: request.step, requestId, source: 'live', attempt, data }))).catch(() => { });
|
|
277
|
+
}
|
|
278
|
+
catch { /* Optional display is not a required storage boundary. */ }
|
|
279
|
+
};
|
|
280
|
+
try {
|
|
281
|
+
check();
|
|
282
|
+
let body;
|
|
283
|
+
try {
|
|
284
|
+
const encoded = adapter.encode(request, model, streaming, generation);
|
|
285
|
+
if (!json(encoded) || bytes(encoded) > 2 * 1024 * 1024)
|
|
286
|
+
throw 0;
|
|
287
|
+
body = JSON.stringify(encoded);
|
|
288
|
+
}
|
|
289
|
+
catch {
|
|
290
|
+
throw new ModelGatewayError('request', 'none');
|
|
291
|
+
}
|
|
292
|
+
for (let attempt = 1; attempt <= retry.maxAttempts; attempt++) {
|
|
293
|
+
check();
|
|
294
|
+
let status;
|
|
295
|
+
let wait = 0;
|
|
296
|
+
let rejectedForRateLimit = false;
|
|
297
|
+
let classifiedRejection = false;
|
|
298
|
+
let acquiredResponse;
|
|
299
|
+
const clientRequestId = `${requestId}-${attempt}`;
|
|
300
|
+
let providerRequestId, usage;
|
|
301
|
+
try {
|
|
302
|
+
sent = true;
|
|
303
|
+
const response = await transport(endpoint, { method: 'POST',
|
|
304
|
+
headers: { ...headers, 'x-client-request-id': clientRequestId }, body,
|
|
305
|
+
signal: controller.signal, redirect: 'error' });
|
|
306
|
+
acquiredResponse = response;
|
|
307
|
+
status = response.status;
|
|
308
|
+
const remoteId = response.headers.get(protocol === 'anthropic-messages' ? 'request-id' : 'x-request-id');
|
|
309
|
+
if (remoteId && /^[A-Za-z0-9_-]{1,128}$/.test(remoteId)
|
|
310
|
+
&& !(credential && remoteId.includes(credential)))
|
|
311
|
+
providerRequestId = remoteId;
|
|
312
|
+
// A transport returning after cancellation still transfers body ownership here.
|
|
313
|
+
if (controller.signal.aborted || external.aborted || elapsed() >= retry.maxElapsedMs) {
|
|
314
|
+
await response.body?.cancel();
|
|
315
|
+
check();
|
|
316
|
+
}
|
|
317
|
+
if (!response.ok) {
|
|
318
|
+
let text = '';
|
|
319
|
+
for await (const part of readBody(response, controller.signal, 64 * 1024))
|
|
320
|
+
text += part;
|
|
321
|
+
check();
|
|
322
|
+
let errorBody;
|
|
323
|
+
try {
|
|
324
|
+
errorBody = JSON.parse(text);
|
|
325
|
+
}
|
|
326
|
+
catch {
|
|
327
|
+
errorBody = undefined;
|
|
328
|
+
}
|
|
329
|
+
const code = adapter.classify(status, errorBody);
|
|
330
|
+
// Even a custom classifier cannot turn accepted output/server uncertainty into safe replay.
|
|
331
|
+
const safe = status >= 400 && status < 500;
|
|
332
|
+
const classified = code === 'rate_limit' && status !== 429 ? 'http' : code;
|
|
333
|
+
classifiedRejection = safe;
|
|
334
|
+
rejectedForRateLimit = classified === 'rate_limit' && status === 429;
|
|
335
|
+
wait = Math.max(retry.baseDelayMs * 2 ** (attempt - 1), retryAfter(response.headers.get('retry-after')));
|
|
336
|
+
throw new ModelGatewayError(classified, safe ? 'none' : 'unknown');
|
|
337
|
+
}
|
|
338
|
+
const bodyStream = readBody(response, controller.signal, 4 * 1024 * 1024);
|
|
339
|
+
let decoded;
|
|
340
|
+
try {
|
|
341
|
+
decoded = await adapter.decode(bodyStream, streaming, data => {
|
|
342
|
+
if (data.type === 'usage')
|
|
343
|
+
usage = reportedUsage(data.usage);
|
|
344
|
+
// Attempts and retries are gateway facts, not adapter-created observations.
|
|
345
|
+
if (data.type !== 'attempt_finished' && data.type !== 'retry')
|
|
346
|
+
emit(attempt, data);
|
|
347
|
+
});
|
|
348
|
+
}
|
|
349
|
+
finally {
|
|
350
|
+
await bodyStream.return(undefined);
|
|
351
|
+
}
|
|
352
|
+
check();
|
|
353
|
+
validateResponse(decoded.response, request);
|
|
354
|
+
if (decoded.usage !== undefined)
|
|
355
|
+
usage = reportedUsage(decoded.usage);
|
|
356
|
+
const record = { attempt, clientRequestId, elapsedMs: elapsed(), status,
|
|
357
|
+
...(providerRequestId ? { providerRequestId } : {}), ...(usage ? { usage } : {}),
|
|
358
|
+
outcome: 'success', effect: 'completed' };
|
|
359
|
+
attempts.push(record);
|
|
360
|
+
emit(attempt, { type: 'attempt_finished', record });
|
|
361
|
+
const cleanResponse = decoded.response.kind === 'final'
|
|
362
|
+
? { kind: 'final', content: decoded.response.content }
|
|
363
|
+
: { kind: 'tool_calls', content: decoded.response.content, calls: decoded.response.calls.map(c => ({
|
|
364
|
+
id: c.id, name: c.name, arguments: c.arguments,
|
|
365
|
+
})) };
|
|
366
|
+
return frozen({ requestId, source: 'live', response: cleanResponse,
|
|
367
|
+
...(usage ? { usage } : {}), attempts });
|
|
368
|
+
}
|
|
369
|
+
catch (error) {
|
|
370
|
+
let fault = error instanceof ModelGatewayError ? error : new ModelGatewayError('transport');
|
|
371
|
+
if (!classifiedRejection && fault.effect === 'none') {
|
|
372
|
+
fault = new ModelGatewayError(fault.code, 'unknown');
|
|
373
|
+
}
|
|
374
|
+
try {
|
|
375
|
+
check();
|
|
376
|
+
}
|
|
377
|
+
catch (cancelled) {
|
|
378
|
+
fault = cancelled;
|
|
379
|
+
}
|
|
380
|
+
const record = { attempt, clientRequestId, elapsedMs: elapsed(),
|
|
381
|
+
...(status !== undefined ? { status } : {}), ...(providerRequestId ? { providerRequestId } : {}),
|
|
382
|
+
...(usage ? { usage } : {}), outcome: fault.code, effect: fault.effect };
|
|
383
|
+
attempts.push(record);
|
|
384
|
+
emit(attempt, { type: 'attempt_finished', record });
|
|
385
|
+
if (fault.code !== 'rate_limit' || fault.effect !== 'none' || !rejectedForRateLimit) {
|
|
386
|
+
throw new ModelGatewayError(fault.code, fault.effect, attempts);
|
|
387
|
+
}
|
|
388
|
+
const stop = attempt >= retry.maxAttempts ? 'attempt_limit'
|
|
389
|
+
: elapsed() + wait >= retry.maxElapsedMs ? 'time_limit' : undefined;
|
|
390
|
+
if (stop)
|
|
391
|
+
throw new ModelGatewayError(fault.code, fault.effect, attempts, stop);
|
|
392
|
+
}
|
|
393
|
+
finally {
|
|
394
|
+
// The gateway owns acquired bodies even when a replacement adapter does not read them.
|
|
395
|
+
if (acquiredResponse?.body && !acquiredResponse.body.locked)
|
|
396
|
+
await acquiredResponse.body.cancel();
|
|
397
|
+
}
|
|
398
|
+
emit(attempt, { type: 'retry', delayMs: wait });
|
|
399
|
+
check();
|
|
400
|
+
await clock.sleep(wait, controller.signal);
|
|
401
|
+
}
|
|
402
|
+
throw new ModelGatewayError('deadline', 'unknown', attempts);
|
|
403
|
+
}
|
|
404
|
+
catch (error) {
|
|
405
|
+
if (error instanceof ModelGatewayError) {
|
|
406
|
+
throw new ModelGatewayError(error.code, error.effect, error.attempts.length ? error.attempts : attempts, error.stopReason, requestId);
|
|
407
|
+
}
|
|
408
|
+
try {
|
|
409
|
+
check();
|
|
410
|
+
}
|
|
411
|
+
catch (cancelled) {
|
|
412
|
+
const fault = cancelled;
|
|
413
|
+
throw new ModelGatewayError(fault.code, fault.effect, attempts, undefined, requestId);
|
|
414
|
+
}
|
|
415
|
+
throw new ModelGatewayError('transport', sent ? 'unknown' : 'none', attempts, undefined, requestId);
|
|
416
|
+
}
|
|
417
|
+
finally {
|
|
418
|
+
clearTimeout(timer);
|
|
419
|
+
external.removeEventListener('abort', cancel);
|
|
420
|
+
}
|
|
421
|
+
}
|
|
422
|
+
return { profile, generate, complete: async (request, signal) => (await generate(request, signal)).response };
|
|
423
|
+
}
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
export { ModelGatewayError } from './errors.js';
|
|
2
|
+
export { createModelGateway, fetchTransport } from './gateway.js';
|
|
3
|
+
export { createChatCompletionsAdapter } from './chatCompletions.js';
|
|
4
|
+
export { createResponsesAdapter } from './responses.js';
|
|
5
|
+
export { createAnthropicMessagesAdapter } from './anthropicMessages.js';
|
|
6
|
+
export { createProtocolAdapter, getProtocolCapabilities } from './protocols.js';
|
|
7
|
+
export { createModelObservationAdapter } from './observationAdapter.js';
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/** Independent, optional UI/diagnostic wiring. No import or required dependency on observability. */
|
|
2
|
+
export function createModelObservationAdapter(sinks) {
|
|
3
|
+
const { display, diagnostics } = sinks;
|
|
4
|
+
return event => {
|
|
5
|
+
const invoke = (sink) => {
|
|
6
|
+
try {
|
|
7
|
+
if (sink)
|
|
8
|
+
void Promise.resolve(sink()).catch(() => { });
|
|
9
|
+
}
|
|
10
|
+
catch { /* Optional consumer. */ }
|
|
11
|
+
};
|
|
12
|
+
if (event.data.type === 'text_delta' || event.data.type === 'tool_delta') {
|
|
13
|
+
invoke(display && (() => display(event)));
|
|
14
|
+
}
|
|
15
|
+
else if (event.data.type === 'attempt_finished' || event.data.type === 'retry') {
|
|
16
|
+
const projected = { ...event, data: event.data };
|
|
17
|
+
invoke(diagnostics && (() => diagnostics(projected)));
|
|
18
|
+
}
|
|
19
|
+
// Provisional usage frames are not separately counted; each attempt carries its reported usage.
|
|
20
|
+
};
|
|
21
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { ModelGatewayError } from './errors.js';
|
|
2
|
+
import { createChatCompletionsAdapter } from './chatCompletions.js';
|
|
3
|
+
import { createResponsesAdapter } from './responses.js';
|
|
4
|
+
import { createAnthropicMessagesAdapter } from './anthropicMessages.js';
|
|
5
|
+
/** Adapter subset, not a claim that every model/account supports these operations. */
|
|
6
|
+
export function getProtocolCapabilities(protocol) {
|
|
7
|
+
if (!['chat-completions', 'responses', 'anthropic-messages'].includes(protocol)) {
|
|
8
|
+
throw new ModelGatewayError('configuration', 'none');
|
|
9
|
+
}
|
|
10
|
+
return Object.freeze({ text: true, functionTools: true, streaming: true });
|
|
11
|
+
}
|
|
12
|
+
export function createProtocolAdapter(protocol) {
|
|
13
|
+
switch (protocol) {
|
|
14
|
+
case 'chat-completions': return createChatCompletionsAdapter();
|
|
15
|
+
case 'responses': return createResponsesAdapter();
|
|
16
|
+
case 'anthropic-messages': return createAnthropicMessagesAdapter();
|
|
17
|
+
default: throw new ModelGatewayError('configuration', 'none');
|
|
18
|
+
}
|
|
19
|
+
}
|