newmark-agent 0.5.14 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/dist/conversation-utility-host.bundle.cjs +4565 -2998
  2. package/dist/conversation-utility-host.js +1 -1
  3. package/dist/core/agent.d.ts +108 -16
  4. package/dist/core/agent.js +698 -195
  5. package/dist/core/agentKernel/agent.d.ts +1 -0
  6. package/dist/core/agentKernel/agent.js +9 -0
  7. package/dist/core/agentKernelDiagnostics.d.ts +4 -6
  8. package/dist/core/agentKernelDiagnostics.js +12 -9
  9. package/dist/core/agentKernelRunner.d.ts +5 -0
  10. package/dist/core/agentKernelRunner.js +250 -107
  11. package/dist/core/autoRouter.js +17 -27
  12. package/dist/core/config.d.ts +2 -0
  13. package/dist/core/config.js +7 -1
  14. package/dist/core/conversationCommandState.d.ts +17 -0
  15. package/dist/core/conversationCommandState.js +140 -0
  16. package/dist/core/conversationKernel.d.ts +25 -4
  17. package/dist/core/conversationKernel.js +360 -89
  18. package/dist/core/conversationListEvent.d.ts +6 -0
  19. package/dist/core/conversationListEvent.js +14 -0
  20. package/dist/core/electronUtilityAgentClient.d.ts +4 -1
  21. package/dist/core/electronUtilityAgentClient.js +4 -4
  22. package/dist/core/electronUtilityRuntimePool.d.ts +8 -2
  23. package/dist/core/electronUtilityRuntimePool.js +11 -5
  24. package/dist/core/installUpdate.js +41 -29
  25. package/dist/core/modelResponseHealth.d.ts +17 -0
  26. package/dist/core/modelResponseHealth.js +193 -0
  27. package/dist/core/providerUsageAccounting.d.ts +47 -0
  28. package/dist/core/providerUsageAccounting.js +71 -0
  29. package/dist/core/requestContextEstimate.d.ts +18 -0
  30. package/dist/core/requestContextEstimate.js +54 -0
  31. package/dist/core/subagent.d.ts +89 -5
  32. package/dist/core/subagent.js +264 -41
  33. package/dist/core/subagentCommunication.d.ts +43 -0
  34. package/dist/core/subagentCommunication.js +167 -0
  35. package/dist/core/types.d.ts +34 -1
  36. package/dist/core/utilityAgentProtocol.d.ts +4 -0
  37. package/dist/core/workEventCoalescer.js +4 -1
  38. package/dist/core/wslAgentClient.d.ts +18 -5
  39. package/dist/core/wslAgentClient.js +79 -27
  40. package/dist/core/wslAgentProtocol.d.ts +7 -0
  41. package/dist/core/wslAgentRuntimePool.d.ts +8 -2
  42. package/dist/core/wslAgentRuntimePool.js +19 -6
  43. package/dist/core/wslRuntimeProcessTree.d.ts +10 -0
  44. package/dist/core/wslRuntimeProcessTree.js +104 -0
  45. package/dist/llm/provider.d.ts +5 -12
  46. package/dist/llm/provider.js +150 -269
  47. package/dist/main.js +582 -356
  48. package/dist/preload.js +3 -3
  49. package/dist/providers/chat-completions.adapter.d.ts +2 -0
  50. package/dist/providers/chat-completions.adapter.js +84 -36
  51. package/dist/providers/provider-adapter.d.ts +2 -0
  52. package/dist/providers/provider-events.d.ts +23 -14
  53. package/dist/providers/provider-events.js +148 -43
  54. package/dist/providers/provider-headers.d.ts +7 -3
  55. package/dist/providers/provider-headers.js +109 -39
  56. package/dist/providers/provider-request-compat.d.ts +6 -0
  57. package/dist/providers/provider-request-compat.js +37 -0
  58. package/dist/providers/responses.adapter.d.ts +2 -0
  59. package/dist/providers/responses.adapter.js +192 -126
  60. package/dist/server.d.ts +9 -2
  61. package/dist/server.js +100 -24
  62. package/dist/tools/index.js +17 -1
  63. package/dist/ui/index.html +2515 -693
  64. package/dist/ui/lucide-sprite.svg +7 -0
  65. package/dist/ui/startup.html +6 -6
  66. package/dist/wsl-agent-host.bundle.cjs +4575 -3006
  67. package/dist/wsl-agent-host.js +9 -4
  68. package/package.json +23 -4
package/dist/preload.js CHANGED
@@ -10,7 +10,7 @@ contextBridge.exposeInMainWorld('api', {
10
10
  onStartupStatus: (callback) => {
11
11
  ipcRenderer.on('startup:status', (_event, payload) => callback(payload));
12
12
  },
13
- sendMessage: (message, target) => ipcRenderer.invoke('agent:send', message, target),
13
+ sendMessage: (message, target, options) => ipcRenderer.invoke('agent:send', message, target, options),
14
14
  enqueueGuide: (envelope) => ipcRenderer.invoke('agent:enqueueGuide', envelope),
15
15
  queueAction: (action, input, target) => ipcRenderer.invoke('agent:queueAction', action, input, target),
16
16
  checkpointConversation: (request) => ipcRenderer.invoke('agent:checkpointConversation', request),
@@ -19,7 +19,7 @@ contextBridge.exposeInMainWorld('api', {
19
19
  stopConversation: (request) => ipcRenderer.invoke('agent:stopConversation', request),
20
20
  setWorkRunExpanded: (request) => ipcRenderer.invoke('agent:setWorkRunExpanded', request),
21
21
  sendPrompt: (message, _model) => ipcRenderer.invoke('agent:sendPrompt', message),
22
- setMode: (mode) => ipcRenderer.invoke('agent:setMode', mode),
22
+ setMode: (mode, target) => ipcRenderer.invoke('agent:setMode', mode, target),
23
23
  setModel: (model) => ipcRenderer.invoke('agent:setModel', model),
24
24
  setIntelligence: (tier) => ipcRenderer.invoke('agent:setIntelligence', tier),
25
25
  setInputMode: (mode, target) => ipcRenderer.invoke('agent:setInputMode', mode, target),
@@ -48,7 +48,7 @@ contextBridge.exposeInMainWorld('api', {
48
48
  browserControl: (request) => ipcRenderer.invoke('browser:control', request),
49
49
  computerUseState: (target) => ipcRenderer.invoke('agent:computerUseState', target),
50
50
  setComputerUseEnabled: (target, enabled) => ipcRenderer.invoke('agent:setComputerUseEnabled', target, enabled),
51
- runFlow: (name, input, start) => ipcRenderer.invoke('flow:run', name, input, start),
51
+ runFlow: (name, input, start, target) => ipcRenderer.invoke('flow:run', name, input, start, target),
52
52
  resumeFlow: (response, target) => ipcRenderer.invoke('flow:resume', response, target),
53
53
  guideFlow: (message, target) => ipcRenderer.invoke('flow:guide', message, target),
54
54
  stopFlow: (target) => ipcRenderer.invoke('flow:stop', target),
@@ -16,6 +16,8 @@ export declare class ChatCompletionsAdapter implements ModelProviderAdapter {
16
16
  execute(request: SerializedProviderRequest, signal: AbortSignal, transport?: typeof defaultProviderTransport): AsyncIterable<NormalizedProviderEvent>;
17
17
  private emitNonStreaming;
18
18
  normalizeUsage(value: unknown): ActualApiUsage | null;
19
+ private validToolArguments;
20
+ private responseError;
19
21
  normalizeHeaders(headers: Headers | Record<string, string>): ProviderResponseMetadata;
20
22
  private extractText;
21
23
  }
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.ChatCompletionsAdapter = void 0;
4
4
  const provider_events_1 = require("./provider-events");
5
5
  const provider_headers_1 = require("./provider-headers");
6
+ const provider_request_compat_1 = require("./provider-request-compat");
6
7
  const chat_messages_1 = require("./chat-messages");
7
8
  const CHAT_SYSTEM_ROLE = 'system';
8
9
  /**
@@ -56,13 +57,9 @@ class ChatCompletionsAdapter {
56
57
  // body,否则省略,避免严格 API 拒绝未知字段。
57
58
  if (request.sessionId)
58
59
  body.session_id = request.sessionId;
59
- const base = request.baseUrl.replace(/\/+$/, '');
60
60
  return {
61
- url: `${base}/chat/completions`,
62
- headers: {
63
- 'Content-Type': 'application/json',
64
- 'Authorization': `Bearer ${request.apiKey}`,
65
- },
61
+ url: (0, provider_request_compat_1.providerEndpoint)(request.baseUrl, '/chat/completions'),
62
+ headers: (0, provider_request_compat_1.providerRequestHeaders)('openai', request.apiKey),
66
63
  body,
67
64
  };
68
65
  }
@@ -88,7 +85,7 @@ class ChatCompletionsAdapter {
88
85
  const contentType = response.headers.get('content-type') || '';
89
86
  if (!/text\/event-stream/i.test(contentType)) {
90
87
  const json = await response.json();
91
- yield* this.emitNonStreaming(json);
88
+ yield* this.emitNonStreaming(json, response.headers);
92
89
  return;
93
90
  }
94
91
  const reader = response.body?.getReader();
@@ -97,9 +94,10 @@ class ChatCompletionsAdapter {
97
94
  return;
98
95
  }
99
96
  const decoder = new TextDecoder();
100
- let buffer = '';
97
+ const sse = new provider_events_1.ProviderSseDecoder();
101
98
  const toolCalls = new Map();
102
99
  const toolCallOrder = [];
100
+ const toolIndexById = new Map();
103
101
  let syntheticToolIndex = 0;
104
102
  let lastToolIndex = 0;
105
103
  let contentPolicyBlocked = false;
@@ -107,22 +105,17 @@ class ChatCompletionsAdapter {
107
105
  let emittedTool = false;
108
106
  let emittedReasoning = false;
109
107
  let explicitCompletion = false;
108
+ let finishReason = '';
109
+ let streamError = '';
110
110
  try {
111
- while (true) {
111
+ stream: while (true) {
112
112
  const { done, value } = await (0, provider_events_1.readProviderStreamChunk)(reader, signal);
113
113
  if (done)
114
114
  break;
115
- buffer += decoder.decode(value, { stream: true });
116
- const lines = buffer.split('\n');
117
- buffer = lines.pop() || '';
118
- for (const line of lines) {
119
- const trimmed = line.trim();
120
- if (!trimmed.startsWith('data: '))
121
- continue;
122
- const data = trimmed.slice(6);
115
+ for (const { data } of sse.push(decoder.decode(value, { stream: true }))) {
123
116
  if (data === '[DONE]') {
124
117
  explicitCompletion = true;
125
- continue;
118
+ break stream;
126
119
  }
127
120
  let json;
128
121
  try {
@@ -136,12 +129,18 @@ class ChatCompletionsAdapter {
136
129
  if (usage)
137
130
  yield { type: 'usage.updated', usage };
138
131
  }
132
+ if (json.error) {
133
+ streamError = this.responseError(json, response.headers);
134
+ break stream;
135
+ }
139
136
  if ((0, provider_events_1.isContentPolicyBlocked)(json))
140
137
  contentPolicyBlocked = true;
141
138
  const choices = Array.isArray(json.choices) ? json.choices : [];
142
139
  const choice = choices[0];
143
- if (choice?.finish_reason !== undefined && choice.finish_reason !== null)
140
+ if (typeof choice?.finish_reason === 'string' && choice.finish_reason.trim()) {
144
141
  explicitCompletion = true;
142
+ finishReason = choice.finish_reason.trim();
143
+ }
145
144
  const delta = choice?.delta;
146
145
  if (!delta)
147
146
  continue;
@@ -162,14 +161,16 @@ class ChatCompletionsAdapter {
162
161
  const tc = raw;
163
162
  const fn = tc.function && typeof tc.function === 'object' ? tc.function : {};
164
163
  const rawIndex = Number(tc.index);
165
- const index = Number.isInteger(rawIndex) && rawIndex >= 0
166
- ? rawIndex
167
- : (tc.id ? syntheticToolIndex++ : lastToolIndex);
164
+ const id = String(tc.id || '');
165
+ while (toolCalls.has(syntheticToolIndex))
166
+ syntheticToolIndex++;
167
+ const index = toolIndexById.get(id) ?? (tc.index !== undefined && tc.index !== null && Number.isInteger(rawIndex) && rawIndex >= 0
168
+ ? rawIndex : (id ? syntheticToolIndex++ : lastToolIndex));
168
169
  lastToolIndex = index;
169
170
  let currentToolCall = toolCalls.get(index);
170
- if (!currentToolCall && tc.id) {
171
+ if (!currentToolCall) {
171
172
  currentToolCall = {
172
- id: String(tc.id || ''),
173
+ id,
173
174
  name: (0, chat_messages_1.openAIToolName)(String(fn.name || '')),
174
175
  argumentParts: [],
175
176
  };
@@ -177,6 +178,14 @@ class ChatCompletionsAdapter {
177
178
  toolCallOrder.push(index);
178
179
  yield { type: 'tool_call.started', id: currentToolCall.id, name: currentToolCall.name };
179
180
  }
181
+ if (id) {
182
+ if (currentToolCall.id && currentToolCall.id !== id) {
183
+ streamError = '[LLM Error] Provider changed a streamed tool call identity.';
184
+ break stream;
185
+ }
186
+ currentToolCall.id = id;
187
+ toolIndexById.set(id, index);
188
+ }
180
189
  if (currentToolCall && fn.name && !currentToolCall.name)
181
190
  currentToolCall.name = (0, chat_messages_1.openAIToolName)(String(fn.name));
182
191
  if (currentToolCall && fn.arguments !== undefined && fn.arguments !== null) {
@@ -189,17 +198,26 @@ class ChatCompletionsAdapter {
189
198
  }
190
199
  }
191
200
  }
201
+ if (streamError || !explicitCompletion || finishReason === 'length' || finishReason === 'content_filter') {
202
+ yield { type: 'response.failed', error: streamError || (finishReason === 'length'
203
+ ? '[LLM Error] Chat response was truncated by the output token limit.'
204
+ : finishReason === 'content_filter' ? '[Error] Content policy refusal (content_filter).'
205
+ : '[LLM Error] Chat stream ended before an explicit completion.') };
206
+ return;
207
+ }
192
208
  if (toolCallOrder.length) {
193
- for (const index of toolCallOrder) {
194
- const currentToolCall = toolCalls.get(index);
195
- if (!currentToolCall)
196
- continue;
209
+ const completeCalls = toolCallOrder.map(index => toolCalls.get(index)).map(call => ({ ...call, arguments: (0, provider_events_1.assembleCompatibleToolArguments)(call.argumentParts, true) }));
210
+ if (completeCalls.some(call => !call.id || !call.name || !this.validToolArguments(call.arguments))) {
211
+ yield { type: 'response.failed', error: '[LLM Error] Provider returned incomplete or invalid tool-call arguments.' };
212
+ return;
213
+ }
214
+ for (const currentToolCall of completeCalls) {
197
215
  emittedTool = true;
198
216
  yield {
199
217
  type: 'tool_call.completed',
200
218
  id: currentToolCall.id,
201
219
  name: currentToolCall.name,
202
- arguments: (0, provider_events_1.assembleCompatibleToolArguments)(currentToolCall.argumentParts),
220
+ arguments: currentToolCall.arguments,
203
221
  };
204
222
  }
205
223
  }
@@ -207,20 +225,24 @@ class ChatCompletionsAdapter {
207
225
  yield { type: 'response.failed', error: '[Error] Content policy refusal (content_filter).' };
208
226
  return;
209
227
  }
210
- if (!explicitCompletion && !emittedContent && !emittedTool && !emittedReasoning) {
211
- yield { type: 'response.failed', error: '[LLM Error] Chat stream ended before an explicit completion.' };
212
- return;
213
- }
214
228
  yield { type: 'response.completed' };
215
229
  }
216
230
  finally {
231
+ try {
232
+ await reader.cancel();
233
+ }
234
+ catch { }
217
235
  reader.releaseLock();
218
236
  }
219
237
  }
220
- async *emitNonStreaming(json) {
238
+ async *emitNonStreaming(json, headers) {
221
239
  const usage = this.normalizeUsage(json.usage);
222
240
  if (usage)
223
241
  yield { type: 'usage.updated', usage };
242
+ if (json.error) {
243
+ yield { type: 'response.failed', error: this.responseError(json, headers) };
244
+ return;
245
+ }
224
246
  const choices = Array.isArray(json.choices) ? json.choices : [];
225
247
  const choice = choices[0];
226
248
  const message = choice?.message && typeof choice.message === 'object' ? choice.message : {};
@@ -230,14 +252,27 @@ class ChatCompletionsAdapter {
230
252
  const text = this.extractText(message.content) || this.extractText(choice?.text);
231
253
  if (text)
232
254
  yield { type: 'text.delta', delta: text };
255
+ if (choice?.finish_reason === 'length' || choice?.finish_reason === 'content_filter') {
256
+ yield { type: 'response.failed', error: choice.finish_reason === 'length'
257
+ ? '[LLM Error] Chat response was truncated by the output token limit.'
258
+ : '[Error] Content policy refusal (content_filter).' };
259
+ return;
260
+ }
233
261
  const toolCalls = Array.isArray(message.tool_calls) ? message.tool_calls : [];
234
- for (const raw of toolCalls) {
262
+ const completeCalls = toolCalls.map(raw => {
235
263
  const tc = raw;
236
264
  const fn = tc.function && typeof tc.function === 'object' ? tc.function : {};
237
265
  const id = String(tc.id || '');
238
266
  const name = (0, chat_messages_1.openAIToolName)(String(fn.name || ''));
267
+ const args = fn.arguments === undefined || fn.arguments === null ? '' : String(fn.arguments);
268
+ return { id, name, args };
269
+ });
270
+ if (completeCalls.some(call => !call.id || !call.name || !this.validToolArguments(call.args))) {
271
+ yield { type: 'response.failed', error: '[LLM Error] Provider returned incomplete or invalid tool-call arguments.' };
272
+ return;
273
+ }
274
+ for (const { id, name, args } of completeCalls) {
239
275
  yield { type: 'tool_call.started', id, name };
240
- const args = String(fn.arguments || '{}');
241
276
  if (args && args !== '{}')
242
277
  yield { type: 'tool_call.arguments.delta', id, delta: args };
243
278
  yield { type: 'tool_call.completed', id, name, arguments: args };
@@ -254,6 +289,19 @@ class ChatCompletionsAdapter {
254
289
  return null;
255
290
  return normalized;
256
291
  }
292
+ validToolArguments(value) {
293
+ try {
294
+ const parsed = JSON.parse(value);
295
+ return !!parsed && typeof parsed === 'object' && !Array.isArray(parsed);
296
+ }
297
+ catch {
298
+ return false;
299
+ }
300
+ }
301
+ responseError(json, headers) {
302
+ const error = json.error && typeof json.error === 'object' ? json.error : {};
303
+ return (0, provider_events_1.providerSemanticErrorText)(error, this.extractText(error.message || error.type || json.error) || 'Provider returned an error response.', headers);
304
+ }
257
305
  normalizeHeaders(headers) {
258
306
  return (0, provider_headers_1.normalizeProviderHeaders)(headers);
259
307
  }
@@ -133,6 +133,8 @@ export interface ActualApiUsage {
133
133
  cacheReadTokens: number;
134
134
  cacheWriteTokens: number;
135
135
  totalTokens: number;
136
+ /** Legacy injected providers may omit this; normal protocol parsing fills it. */
137
+ reported?: import('../core/types').ProviderUsageReported;
136
138
  }
137
139
  export interface ProviderResponseMetadata {
138
140
  requestId?: string;
@@ -1,4 +1,6 @@
1
- import { NormalizedAgentRequest, SerializedProviderRequest, TransportResponse } from './provider-adapter';
1
+ import { ActualApiUsage, NormalizedAgentRequest, SerializedProviderRequest, TransportResponse } from './provider-adapter';
2
+ import { Dispatcher } from 'undici';
3
+ export declare function providerStreamingDispatcher(dispatcher?: unknown): Dispatcher;
2
4
  /** dev-0.5.14: let non-LLMProvider adapter fallbacks also honor the configured proxy. */
3
5
  export declare function setDefaultProviderDispatcher(dispatcher: unknown): void;
4
6
  /**
@@ -28,13 +30,24 @@ export declare function parseProviderSse(raw: string): Array<{
28
30
  event?: string;
29
31
  data: string;
30
32
  }>;
33
+ /** Incremental SSE framing: CRLF can straddle any transport chunk boundary. */
34
+ export declare class ProviderSseDecoder {
35
+ private line;
36
+ private skipLf;
37
+ private event;
38
+ private data;
39
+ push(text: string): Array<{
40
+ event?: string;
41
+ data: string;
42
+ }>;
43
+ }
31
44
  /**
32
45
  * Compatible gateways may stream function arguments as JSON deltas or repeat
33
46
  * a cumulative snapshot on every SSE frame. Keep the normal incremental form
34
47
  * when it parses, then fall back to snapshot folding. Returning malformed
35
48
  * concatenated snapshots would erase the model's correction at tool parsing.
36
49
  */
37
- export declare function assembleCompatibleToolArguments(parts: string[]): string;
50
+ export declare function assembleCompatibleToolArguments(parts: string[], strictFinal?: boolean): string;
38
51
  /**
39
52
  * Detect content-policy refusals across the provider failure shapes.
40
53
  * Mirrors `LLMProvider.contentPolicyBlocked` semantics exactly so the
@@ -46,18 +59,16 @@ export declare function assembleCompatibleToolArguments(parts: string[]): string
46
59
  * - `content_filter_results` / `prompt_filter_results` with `"filtered": true`
47
60
  */
48
61
  export declare function isContentPolicyBlocked(json: Record<string, unknown>): boolean;
62
+ export interface UsageNormalizationOptions {
63
+ /** Only the actual native protocol can change the meaning of input_tokens. */
64
+ protocol?: 'openai' | 'anthropic' | 'github_models';
65
+ }
49
66
  /**
50
67
  * Normalize the many provider usage shapes into ActualApiUsage.
51
68
  * Handles Chat Completions `usage`, Responses `response.usage`, and
52
69
  * Anthropic-style `usage` objects.
53
70
  */
54
- export declare function normalizeProviderUsage(value: unknown): {
55
- inputTokens: number;
56
- outputTokens: number;
57
- cacheReadTokens: number;
58
- cacheWriteTokens: number;
59
- totalTokens: number;
60
- } | null;
71
+ export declare function normalizeProviderUsage(value: unknown, options?: UsageNormalizationOptions): ActualApiUsage | null;
61
72
  export declare function estimateRequestTokens(request: NormalizedAgentRequest): {
62
73
  inputTokens: number;
63
74
  outputReservedTokens: number;
@@ -65,11 +76,9 @@ export declare function estimateRequestTokens(request: NormalizedAgentRequest):
65
76
  totalTokens: number;
66
77
  tokenizerSource: 'estimated' | 'compatible' | 'provider_tokenizer';
67
78
  };
68
- /**
69
- * Format a non-OK provider response into the exact error string the LLM
70
- * provider produces (`[LLM Error: <status>][ Retry-After: <n>s] <body>`).
71
- * Mirrors `LLMProvider.llmErrorText`.
72
- */
79
+ /** Keep semantic provider errors classifiable even when the HTTP status is 200. */
80
+ export declare function providerSemanticErrorText(value: unknown, fallback: string, headers?: Pick<Headers, 'get'>): string;
81
+ /** Format a non-OK response, retaining the server's retry delay. */
73
82
  export declare function providerErrorText(response: {
74
83
  status: number;
75
84
  headers?: {
@@ -1,5 +1,7 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.ProviderSseDecoder = void 0;
4
+ exports.providerStreamingDispatcher = providerStreamingDispatcher;
3
5
  exports.setDefaultProviderDispatcher = setDefaultProviderDispatcher;
4
6
  exports.defaultProviderTransport = defaultProviderTransport;
5
7
  exports.providerAbortError = providerAbortError;
@@ -10,12 +12,40 @@ exports.assembleCompatibleToolArguments = assembleCompatibleToolArguments;
10
12
  exports.isContentPolicyBlocked = isContentPolicyBlocked;
11
13
  exports.normalizeProviderUsage = normalizeProviderUsage;
12
14
  exports.estimateRequestTokens = estimateRequestTokens;
15
+ exports.providerSemanticErrorText = providerSemanticErrorText;
13
16
  exports.providerErrorText = providerErrorText;
14
17
  exports.normalizeResponsesPayload = normalizeResponsesPayload;
18
+ const undici_1 = require("undici");
19
+ const provider_headers_1 = require("./provider-headers");
15
20
  /**
16
21
  * Shared SSE parsing and usage normalization used by both adapters.
17
22
  */
18
23
  let defaultProviderDispatcher = null;
24
+ const directProviderStreamDispatcher = new undici_1.Agent();
25
+ const providerStreamDispatchers = new WeakMap();
26
+ class StreamingProviderDispatcher extends undici_1.Dispatcher {
27
+ delegate;
28
+ constructor(delegate) {
29
+ super();
30
+ this.delegate = delegate;
31
+ }
32
+ dispatch(options, handler) {
33
+ // The application already owns cancellation. Undici otherwise adds its
34
+ // own 300s headers/body deadline, including during valid long reasoning
35
+ // and non-streaming JSON model responses (for example generated titles).
36
+ // Keep connection establishment and all non-LLM requests unchanged.
37
+ return this.delegate.dispatch({ ...options, headersTimeout: 0, bodyTimeout: 0 }, handler);
38
+ }
39
+ }
40
+ function providerStreamingDispatcher(dispatcher) {
41
+ const delegate = (dispatcher || directProviderStreamDispatcher);
42
+ let streaming = providerStreamDispatchers.get(delegate);
43
+ if (!streaming) {
44
+ streaming = new StreamingProviderDispatcher(delegate);
45
+ providerStreamDispatchers.set(delegate, streaming);
46
+ }
47
+ return streaming;
48
+ }
19
49
  /** dev-0.5.14: let non-LLMProvider adapter fallbacks also honor the configured proxy. */
20
50
  function setDefaultProviderDispatcher(dispatcher) {
21
51
  defaultProviderDispatcher = dispatcher || null;
@@ -31,8 +61,11 @@ function defaultProviderTransport(request, signal) {
31
61
  body: JSON.stringify(request.body),
32
62
  signal,
33
63
  };
34
- if (defaultProviderDispatcher) {
35
- init.dispatcher = defaultProviderDispatcher;
64
+ // This transport is exclusively for model POSTs, regardless of whether the
65
+ // provider returns SSE or a complete JSON body. It never serves model GETs.
66
+ const dispatcher = providerStreamingDispatcher(defaultProviderDispatcher);
67
+ if (dispatcher) {
68
+ init.dispatcher = dispatcher;
36
69
  }
37
70
  return fetch(request.url, init);
38
71
  }
@@ -116,16 +149,64 @@ function parseProviderSse(raw) {
116
149
  }
117
150
  return events;
118
151
  }
152
+ /** Incremental SSE framing: CRLF can straddle any transport chunk boundary. */
153
+ class ProviderSseDecoder {
154
+ line = '';
155
+ skipLf = false;
156
+ event;
157
+ data = [];
158
+ push(text) {
159
+ const events = [];
160
+ let start = 0;
161
+ for (let i = 0; i < text.length; i += 1) {
162
+ const char = text[i];
163
+ if (this.skipLf) {
164
+ this.skipLf = false;
165
+ if (char === '\n') {
166
+ start = i + 1;
167
+ continue;
168
+ }
169
+ }
170
+ if (char !== '\r' && char !== '\n')
171
+ continue;
172
+ const line = this.line + text.slice(start, i);
173
+ this.line = '';
174
+ start = i + 1;
175
+ this.skipLf = char === '\r';
176
+ if (!line) {
177
+ if (this.data.length)
178
+ events.push({ event: this.event, data: this.data.join('\n') });
179
+ this.event = undefined;
180
+ this.data = [];
181
+ }
182
+ else if (!line.startsWith(':')) {
183
+ const colon = line.indexOf(':');
184
+ const field = colon < 0 ? line : line.slice(0, colon);
185
+ let value = colon < 0 ? '' : line.slice(colon + 1);
186
+ // SSE removes only the optional single ASCII space after the colon.
187
+ if (value.startsWith(' '))
188
+ value = value.slice(1);
189
+ if (field === 'event')
190
+ this.event = value;
191
+ else if (field === 'data')
192
+ this.data.push(value);
193
+ }
194
+ }
195
+ this.line += text.slice(start);
196
+ return events;
197
+ }
198
+ }
199
+ exports.ProviderSseDecoder = ProviderSseDecoder;
119
200
  /**
120
201
  * Compatible gateways may stream function arguments as JSON deltas or repeat
121
202
  * a cumulative snapshot on every SSE frame. Keep the normal incremental form
122
203
  * when it parses, then fall back to snapshot folding. Returning malformed
123
204
  * concatenated snapshots would erase the model's correction at tool parsing.
124
205
  */
125
- function assembleCompatibleToolArguments(parts) {
126
- const nonEmpty = (parts || []).map(String).filter(part => part && part !== 'null');
206
+ function assembleCompatibleToolArguments(parts, strictFinal = false) {
207
+ const nonEmpty = (parts || []).map(String).filter(part => part && (strictFinal || part !== 'null'));
127
208
  if (!nonEmpty.length)
128
- return '{}';
209
+ return strictFinal ? '' : '{}';
129
210
  const isJsonObject = (value) => {
130
211
  try {
131
212
  const parsed = JSON.parse(value);
@@ -146,13 +227,21 @@ function assembleCompatibleToolArguments(parts) {
146
227
  continue;
147
228
  else if (incoming.startsWith(compatible))
148
229
  compatible = incoming;
149
- else if (compatible.startsWith(incoming))
150
- continue;
230
+ else if (compatible.startsWith(incoming)) {
231
+ if (strictFinal && isJsonObject(compatible) && !isJsonObject(incoming))
232
+ compatible = incoming;
233
+ else
234
+ continue;
235
+ }
151
236
  else
152
237
  compatible += incoming;
153
238
  }
154
239
  if (isJsonObject(compatible))
155
240
  return compatible;
241
+ // At execution time an earlier valid snapshot cannot repair a newer,
242
+ // truncated correction. Only the final snapshot may replace bad assembly.
243
+ if (strictFinal)
244
+ return isJsonObject(nonEmpty[nonEmpty.length - 1]) ? nonEmpty[nonEmpty.length - 1] : compatible;
156
245
  return [...nonEmpty].reverse().find(isJsonObject) || compatible;
157
246
  }
158
247
  /**
@@ -181,49 +270,62 @@ function isContentPolicyBlocked(json) {
181
270
  const filterEvidence = JSON.stringify(choice.content_filter_results || json.prompt_filter_results || {});
182
271
  return /"filtered"\s*:\s*true/i.test(filterEvidence);
183
272
  }
184
- function extractUsageValue(value) {
185
- return Math.max(0, Number(value) || 0);
273
+ function reportedUsageValue(...values) {
274
+ for (const value of values) {
275
+ if (typeof value !== 'number' && (typeof value !== 'string' || !value.trim()))
276
+ continue;
277
+ const number = Number(value);
278
+ if (Number.isFinite(number) && number >= 0)
279
+ return { value: Math.floor(number), reported: true };
280
+ }
281
+ return { value: 0, reported: false };
186
282
  }
187
283
  /**
188
284
  * Normalize the many provider usage shapes into ActualApiUsage.
189
285
  * Handles Chat Completions `usage`, Responses `response.usage`, and
190
286
  * Anthropic-style `usage` objects.
191
287
  */
192
- function normalizeProviderUsage(value) {
288
+ function normalizeProviderUsage(value, options = {}) {
193
289
  if (!value || typeof value !== 'object' || Array.isArray(value))
194
290
  return null;
195
291
  const raw = value;
196
292
  // Nested response usage (Responses API: usage inside response).
197
293
  if (raw.response && typeof raw.response === 'object' && !Array.isArray(raw.response)) {
198
- return normalizeProviderUsage(raw.response);
294
+ return normalizeProviderUsage(raw.response, options);
199
295
  }
200
296
  // Top-level `usage` wrapper (Responses `response.completed` items carry
201
297
  // `{ response: { usage } }`, which strips down to `{ usage }`).
202
298
  if (raw.usage && typeof raw.usage === 'object' && !Array.isArray(raw.usage)) {
203
- return normalizeProviderUsage(raw.usage);
299
+ return normalizeProviderUsage(raw.usage, options);
204
300
  }
205
- const inputTokens = extractUsageValue(raw.input_tokens) ||
206
- extractUsageValue(raw.prompt_tokens) ||
207
- extractUsageValue(raw.input) ||
208
- 0;
209
- const outputTokens = extractUsageValue(raw.output_tokens) ||
210
- extractUsageValue(raw.completion_tokens) ||
211
- extractUsageValue(raw.output) ||
212
- 0;
213
- // token_details (Responses) / prompt_tokens_details (Chat).
214
- const details = raw.token_details ?? raw.prompt_tokens_details ?? {};
215
- const detailsObj = details && typeof details === 'object' && !Array.isArray(details) ? details : {};
216
- const cachedTokens = raw.cached_tokens ?? detailsObj.cached_tokens ?? 0;
217
- const cacheReadTokens = extractUsageValue(cachedTokens);
218
- const cacheCreationTokens = raw.cache_creation_tokens ?? raw.cache_read_input_tokens ?? detailsObj.cache_creation ?? 0;
219
- const cacheWriteTokens = extractUsageValue(cacheCreationTokens);
220
- const totalTokens = extractUsageValue(raw.total_tokens) || inputTokens + outputTokens;
301
+ const input = reportedUsageValue(raw.input_tokens, raw.prompt_tokens, raw.input);
302
+ const output = reportedUsageValue(raw.output_tokens, raw.completion_tokens, raw.output);
303
+ // Responses uses input_tokens_details; Chat uses prompt_tokens_details.
304
+ // Keep compatible top-level and legacy aliases without confusing reads
305
+ // with writes. An explicit zero is authoritative, not a missing value.
306
+ const details = [raw.input_tokens_details, raw.prompt_tokens_details, raw.token_details]
307
+ .filter((value) => !!value && typeof value === 'object' && !Array.isArray(value));
308
+ const detailValues = (...keys) => details.flatMap(value => keys.map(key => value[key]));
309
+ const cacheRead = reportedUsageValue(raw.cached_tokens, raw.cache_read_input_tokens, raw.cached_input_tokens, ...detailValues('cached_tokens', 'cache_read_tokens'), raw.prompt_cache_hit_tokens);
310
+ const cacheWrite = reportedUsageValue(raw.cache_creation_input_tokens, raw.cache_write_input_tokens, raw.cache_creation_tokens, ...detailValues('cache_write_tokens', 'cache_creation'));
311
+ const total = reportedUsageValue(raw.total_tokens);
312
+ const reported = { input: input.reported, output: output.reported, cacheRead: cacheRead.reported, cacheWrite: cacheWrite.reported };
313
+ if (!Object.values(reported).some(Boolean) && !total.reported)
314
+ return null;
315
+ // Native Anthropic input_tokens excludes both cache reads and writes. OpenAI
316
+ // compatible aliases alone never select this protocol-specific convention.
317
+ const inputTokens = input.value + (options.protocol === 'anthropic' ? cacheRead.value + cacheWrite.value : 0);
318
+ const outputTokens = output.value;
319
+ const cacheReadTokens = cacheRead.value;
320
+ const cacheWriteTokens = cacheWrite.value;
321
+ const totalTokens = total.reported ? total.value : inputTokens + outputTokens;
221
322
  return {
222
323
  inputTokens,
223
324
  outputTokens,
224
325
  cacheReadTokens,
225
326
  cacheWriteTokens,
226
327
  totalTokens,
328
+ reported,
227
329
  };
228
330
  }
229
331
  const CHARACTERS_PER_TOKEN = 4;
@@ -257,22 +359,25 @@ function estimateRequestTokens(request) {
257
359
  tokenizerSource: 'estimated',
258
360
  };
259
361
  }
260
- /**
261
- * Format a non-OK provider response into the exact error string the LLM
262
- * provider produces (`[LLM Error: <status>][ Retry-After: <n>s] <body>`).
263
- * Mirrors `LLMProvider.llmErrorText`.
264
- */
265
- function providerErrorText(response, body) {
266
- const rawRetryAfter = response.headers?.get('retry-after')?.trim() || '';
267
- let retryAfter = '';
268
- if (/^\d+(?:\.\d+)?$/.test(rawRetryAfter)) {
269
- retryAfter = ` Retry-After: ${rawRetryAfter}s`;
270
- }
271
- else if (rawRetryAfter) {
272
- const retryAt = Date.parse(rawRetryAfter);
273
- if (Number.isFinite(retryAt))
274
- retryAfter = ` Retry-After: ${Math.max(0, Math.ceil((retryAt - Date.now()) / 1000))}s`;
362
+ /** Keep semantic provider errors classifiable even when the HTTP status is 200. */
363
+ function providerSemanticErrorText(value, fallback, headers) {
364
+ const seconds = headers ? (0, provider_headers_1.normalizeProviderHeaders)(headers).retryAfterSeconds : undefined;
365
+ const prefix = `[LLM Error]${seconds === undefined ? '' : ` Retry-After: ${seconds}s`} `;
366
+ if (!value || typeof value !== 'object' || Array.isArray(value))
367
+ return prefix + fallback;
368
+ const error = value;
369
+ const detail = { message: typeof error.message === 'string' ? error.message : fallback };
370
+ for (const key of ['code', 'type']) {
371
+ const entry = error[key];
372
+ if (typeof entry === 'string' && /^[a-z0-9_.-]{1,80}$/i.test(entry))
373
+ detail[key] = entry;
275
374
  }
375
+ return prefix + (detail.code || detail.type ? JSON.stringify({ error: detail }) : detail.message);
376
+ }
377
+ /** Format a non-OK response, retaining the server's retry delay. */
378
+ function providerErrorText(response, body) {
379
+ const seconds = response.headers ? (0, provider_headers_1.normalizeProviderHeaders)(response.headers).retryAfterSeconds : undefined;
380
+ const retryAfter = seconds === undefined ? '' : ` Retry-After: ${seconds}s`;
276
381
  return `[LLM Error: ${response.status}]${retryAfter} ${body}`;
277
382
  }
278
383
  /**