@hifullmoon/aicommit 2.3.0 → 2.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,403 @@
1
+ import { stream as streamPi } from '@earendil-works/pi-ai/api/openai-completions';
2
+ import { getProviderAdapter, normalizeUsage } from './providers.js';
3
+ import { ERROR_CATEGORIES, fail } from './errors.js';
4
+ import { completionEvent, normalizeEventStream } from './provider-response.js';
5
+ import { estimateTokens } from './analysis-budget.js';
6
+
7
+ const DEFAULT_TIMEOUT_MS = 120_000;
8
+
9
+ const DEFAULT_RETRY_POLICY = Object.freeze({
10
+ maxAttempts: 3,
11
+ baseDelayMs: 500,
12
+ maxDelayMs: 5000,
13
+ });
14
+ const RETRYABLE_STATUS = new Set([429, 500, 502, 503, 504]);
15
+ const RETRYABLE_NETWORK_CODES = new Set([
16
+ 'ECONNRESET',
17
+ 'ECONNREFUSED',
18
+ 'EHOSTUNREACH',
19
+ 'ENETUNREACH',
20
+ 'EPIPE',
21
+ 'UND_ERR_CONNECT_TIMEOUT',
22
+ 'UND_ERR_SOCKET',
23
+ ]);
24
+
25
+ function secureEndpoint(apiUrl) {
26
+ const endpoint = new URL(apiUrl);
27
+ const loopback =
28
+ endpoint.hostname === 'localhost' ||
29
+ endpoint.hostname === '127.0.0.1' ||
30
+ endpoint.hostname.startsWith('127.') ||
31
+ endpoint.hostname === '[::1]';
32
+ if (endpoint.protocol !== 'https:' && !(endpoint.protocol === 'http:' && loopback)) {
33
+ throw new Error(
34
+ 'Refusing insecure API endpoint: use HTTPS, or HTTP only for localhost/loopback.',
35
+ );
36
+ }
37
+ return endpoint;
38
+ }
39
+
40
+ function retryPolicy(value = {}) {
41
+ return {
42
+ maxAttempts: value?.maxAttempts ?? DEFAULT_RETRY_POLICY.maxAttempts,
43
+ baseDelayMs: value?.baseDelayMs ?? DEFAULT_RETRY_POLICY.baseDelayMs,
44
+ maxDelayMs: value?.maxDelayMs ?? DEFAULT_RETRY_POLICY.maxDelayMs,
45
+ sleep:
46
+ value?.sleep ??
47
+ ((delayMs) =>
48
+ new Promise((resolve) => {
49
+ globalThis.setTimeout(resolve, delayMs);
50
+ })),
51
+ now: value?.now ?? (() => Date.now()),
52
+ };
53
+ }
54
+
55
+ function retryAfterMs(value, now) {
56
+ if (!value) return null;
57
+ const seconds = Number(value);
58
+ if (Number.isFinite(seconds) && seconds >= 0) return seconds * 1000;
59
+ const date = Date.parse(value);
60
+ if (Number.isNaN(date)) return null;
61
+ return Math.max(0, date - now());
62
+ }
63
+
64
+ function networkFailure(err) {
65
+ if (err instanceof TypeError) return true;
66
+ return RETRYABLE_NETWORK_CODES.has(err?.code) || RETRYABLE_NETWORK_CODES.has(err?.cause?.code);
67
+ }
68
+
69
+ function timeoutError(err, timeout) {
70
+ if (err?.name !== 'TimeoutError' && err?.name !== 'AbortError') return null;
71
+ return new Error(
72
+ `Request timed out after ${Math.round(timeout / 1000)}s — the model took too long to respond. ` +
73
+ `Raise "timeoutMs" in your config if this keeps happening.`,
74
+ );
75
+ }
76
+
77
+ async function fetchWithRetry(
78
+ apiUrl,
79
+ init,
80
+ timeout,
81
+ configuredPolicy,
82
+ consume,
83
+ beforeAttempt = null,
84
+ ) {
85
+ const policy = retryPolicy(configuredPolicy);
86
+ let attempt = 0;
87
+
88
+ while (attempt < policy.maxAttempts) {
89
+ attempt += 1;
90
+ beforeAttempt?.();
91
+ let response;
92
+ try {
93
+ response = await fetch(apiUrl, {
94
+ ...init,
95
+ signal: init.signal
96
+ ? AbortSignal.any([init.signal, AbortSignal.timeout(timeout)])
97
+ : AbortSignal.timeout(timeout),
98
+ });
99
+ } catch (err) {
100
+ const wrappedTimeout = timeoutError(err, timeout);
101
+ if (wrappedTimeout) throw wrappedTimeout;
102
+ if (!networkFailure(err) || attempt >= policy.maxAttempts) throw err;
103
+ const delay = Math.min(policy.baseDelayMs * 2 ** (attempt - 1), policy.maxDelayMs);
104
+ await policy.sleep(delay);
105
+ continue;
106
+ }
107
+
108
+ if (response.ok) {
109
+ try {
110
+ return { value: await consume(response), attempts: attempt };
111
+ } catch (err) {
112
+ const wrappedTimeout = timeoutError(err, timeout);
113
+ if (wrappedTimeout) throw wrappedTimeout;
114
+ // Once the provider has accepted a generation request, replaying it is
115
+ // unsafe: the first request may already have completed and been billed
116
+ // even though its response body was interrupted locally.
117
+ throw err;
118
+ }
119
+ }
120
+ if (RETRYABLE_STATUS.has(response.status) && attempt < policy.maxAttempts) {
121
+ const requestedDelay = retryAfterMs(response.headers.get('retry-after'), policy.now);
122
+ if (requestedDelay !== null && requestedDelay > policy.maxDelayMs) {
123
+ await response.body?.cancel().catch(() => {});
124
+ throw new Error(
125
+ `HTTP ${response.status}: provider requested a retry after ${Math.ceil(
126
+ requestedDelay / 1000,
127
+ )}s, exceeding the configured retry.maxDelayMs limit.`,
128
+ );
129
+ }
130
+ const delay =
131
+ requestedDelay ?? Math.min(policy.baseDelayMs * 2 ** (attempt - 1), policy.maxDelayMs);
132
+ await response.body?.cancel().catch(() => {});
133
+ await policy.sleep(delay);
134
+ continue;
135
+ }
136
+
137
+ const errText = await response.text();
138
+ throw new Error(`HTTP ${response.status}: ${errText.slice(0, 400)}`);
139
+ }
140
+
141
+ throw new Error('Provider request exhausted its retry budget.');
142
+ }
143
+
144
+ function piContext(messages, model) {
145
+ return {
146
+ systemPrompt:
147
+ messages
148
+ .filter((m) => m.role === 'system')
149
+ .map((m) => m.content)
150
+ .join('\n\n') || undefined,
151
+ messages: messages
152
+ .filter((m) => m.role !== 'system')
153
+ .map((message) => {
154
+ if (message.role === 'user') return { ...message, timestamp: Date.now() };
155
+ if (message.role !== 'assistant')
156
+ throw new Error(`Unsupported generation message role: ${message.role}`);
157
+ return {
158
+ role: 'assistant',
159
+ content: [{ type: 'text', text: message.content }],
160
+ api: model.api,
161
+ provider: model.provider,
162
+ model: model.id,
163
+ stopReason: 'stop',
164
+ usage: {
165
+ input: 0,
166
+ output: 0,
167
+ cacheRead: 0,
168
+ cacheWrite: 0,
169
+ totalTokens: 0,
170
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
171
+ },
172
+ timestamp: Date.now(),
173
+ };
174
+ }),
175
+ };
176
+ }
177
+
178
+ function nativeOllamaPayload(payload, apiUrl) {
179
+ const { max_tokens, temperature, stream_options: _streamOptions, options, ...rest } = payload;
180
+ const body = {
181
+ ...rest,
182
+ stream: false,
183
+ options: { temperature, num_predict: max_tokens, ...options },
184
+ };
185
+ if (/\/api\/generate\/?$/i.test(new URL(apiUrl).pathname)) {
186
+ // /generate accepts a prompt, not the messages array used by /chat.
187
+ body.system = body.messages
188
+ .filter((m) => m.role === 'system')
189
+ .map((m) => m.content)
190
+ .join('\n\n');
191
+ body.prompt = body.messages
192
+ .filter((m) => m.role !== 'system')
193
+ .map((m) => `${m.role}: ${m.content}`)
194
+ .join('\n\n');
195
+ delete body.messages;
196
+ }
197
+ return body;
198
+ }
199
+
200
+ function transport(config, adapter, state) {
201
+ return async (_sdkUrl, init) => {
202
+ try {
203
+ const headers = new globalThis.Headers(init.headers);
204
+ // Never let SDK defaults resolve a different credential or follow a redirect
205
+ // carrying repository content to an endpoint the user did not configure.
206
+ if (config.apiKey) headers.set('Authorization', `Bearer ${config.apiKey}`);
207
+ else headers.delete('Authorization');
208
+ let payload = JSON.parse(init.body);
209
+ if (adapter.nativeOllama) payload = nativeOllamaPayload(payload, config.apiUrl);
210
+ else if (config.extraBody?.stream === false) {
211
+ payload.stream = false;
212
+ delete payload.stream_options;
213
+ }
214
+ const result = await fetchWithRetry(
215
+ config.apiUrl,
216
+ {
217
+ ...init,
218
+ headers,
219
+ body: JSON.stringify(payload),
220
+ redirect: 'error',
221
+ },
222
+ config.timeoutMs || DEFAULT_TIMEOUT_MS,
223
+ config.analysisBudget
224
+ ? {
225
+ ...config.retry,
226
+ sleep: async (ms) => {
227
+ if (ms >= config.analysisBudget.remainingMs())
228
+ throw new Error('Analysis timed out during retry backoff.');
229
+ await new Promise((resolve) => {
230
+ setTimeout(resolve, ms);
231
+ });
232
+ },
233
+ }
234
+ : config.retry,
235
+ async (response) => {
236
+ if (
237
+ (response.headers.get('content-type') || '').toLowerCase().includes('text/event-stream')
238
+ )
239
+ return normalizeEventStream(response);
240
+ let data;
241
+ try {
242
+ data = await response.json();
243
+ } catch (err) {
244
+ if (err instanceof SyntaxError)
245
+ throw fail(ERROR_CATEGORIES.RESPONSE_FORMAT, 'Provider returned invalid JSON.', {
246
+ cause: err,
247
+ });
248
+ throw err;
249
+ }
250
+ const event = completionEvent(data);
251
+ state.raw = data;
252
+ return new Response(`data: ${JSON.stringify(event)}\n\ndata: [DONE]\n\n`, {
253
+ headers: { 'Content-Type': 'text/event-stream' },
254
+ });
255
+ },
256
+ config.analysisBudget
257
+ ? () => {
258
+ state.ticket = config.analysisBudget.reserve(
259
+ config.analysisInputTokens,
260
+ config.analysisOutputTokens,
261
+ );
262
+ }
263
+ : null,
264
+ );
265
+ state.attempts = result.attempts;
266
+ return result.value;
267
+ } catch (err) {
268
+ state.error = err;
269
+ throw err;
270
+ }
271
+ };
272
+ }
273
+
274
+ export async function requestGeneration(config, request) {
275
+ secureEndpoint(config.apiUrl);
276
+ if (config.analysisBudget) {
277
+ config = {
278
+ ...config,
279
+ analysisInputTokens: estimateTokens(JSON.stringify(request.messages)),
280
+ analysisOutputTokens: request.maxTokens,
281
+ timeoutMs: Math.max(
282
+ 1,
283
+ Math.min(config.timeoutMs || DEFAULT_TIMEOUT_MS, config.analysisBudget.remainingMs()),
284
+ ),
285
+ };
286
+ }
287
+ const adapter = getProviderAdapter(config);
288
+ const options = adapter.options({
289
+ ...request,
290
+ extraBody: config.extraBody,
291
+ reasoning: request.reasoning ?? config.reasoning,
292
+ });
293
+ const state = { attempts: 0, raw: null, error: null };
294
+ const startedAt = performance.now();
295
+ const controller = new AbortController();
296
+ const events = streamPi(adapter.model, piContext(request.messages, adapter.model), {
297
+ ...options,
298
+ // Pi's transport requires a key even for a keyless server. The placeholder
299
+ // never leaves the process: transport installs only the resolved config key.
300
+ apiKey: config.apiKey || 'aicommit-keyless',
301
+ headers: adapter.headers,
302
+ env: {},
303
+ maxRetries: 0,
304
+ timeoutMs: config.timeoutMs || DEFAULT_TIMEOUT_MS,
305
+ signal: config.analysisBudget
306
+ ? AbortSignal.any([controller.signal, config.analysisBudget.signal])
307
+ : controller.signal,
308
+ fetch: transport(config, adapter, state),
309
+ });
310
+ let result;
311
+ try {
312
+ for await (const event of events) {
313
+ if (event.type === 'thinking_delta') request.stream?.onReasoningDelta?.(event.delta);
314
+ if (event.type === 'error') {
315
+ if (state.error) throw state.error;
316
+ const message = event.error.errorMessage || 'Provider request failed.';
317
+ if (/without finish_reason/.test(message))
318
+ throw fail(
319
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
320
+ 'Streaming response ended before the provider sent a finish_reason. The partial response was discarded; retry the request.',
321
+ );
322
+ if (/timed out|timeout/i.test(message))
323
+ throw fail(
324
+ ERROR_CATEGORIES.NETWORK,
325
+ `Request timed out after ${Math.round((config.timeoutMs || DEFAULT_TIMEOUT_MS) / 1000)}s — the model took too long to respond. Raise "timeoutMs" in your config if this keeps happening.`,
326
+ );
327
+ if (/socket|network|fetch failed|terminated|econn/i.test(message))
328
+ throw fail(ERROR_CATEGORIES.NETWORK, message);
329
+ if (/JSON|Unexpected token/i.test(message))
330
+ throw fail(
331
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
332
+ `Provider returned invalid JSON: ${message}`,
333
+ );
334
+ throw fail(ERROR_CATEGORIES.PROVIDER, `Provider request failed: ${message}`);
335
+ }
336
+ if (event.type === 'done') result = event.message;
337
+ }
338
+ } finally {
339
+ controller.abort();
340
+ }
341
+ if (!result)
342
+ throw fail(
343
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
344
+ 'Provider returned an invalid response: no completed generation.',
345
+ );
346
+ const content = result.content
347
+ .filter((block) => block.type === 'text')
348
+ .map((block) => block.text)
349
+ .join('');
350
+ const reasoning =
351
+ result.content
352
+ .filter((block) => block.type === 'thinking')
353
+ .map((block) => block.thinking)
354
+ .filter(Boolean)
355
+ .join('\n') || null;
356
+ const usage = state.raw
357
+ ? normalizeUsage(state.raw.usage || state.raw)
358
+ : result.usage.totalTokens ||
359
+ result.usage.input ||
360
+ result.usage.output ||
361
+ result.usage.cacheRead ||
362
+ result.usage.cacheWrite
363
+ ? {
364
+ inputTokens: result.usage.input + result.usage.cacheRead + result.usage.cacheWrite,
365
+ outputTokens: result.usage.output,
366
+ totalTokens: result.usage.totalTokens,
367
+ }
368
+ : null;
369
+ const finishReason =
370
+ state.raw?.choices?.[0]?.finish_reason ??
371
+ state.raw?.stop_reason ??
372
+ state.raw?.done_reason ??
373
+ result.rawStopReason ??
374
+ result.stopReason;
375
+ config.analysisBudget?.settle(state.ticket, usage);
376
+ return {
377
+ provider: adapter.id,
378
+ model: result.responseModel || result.model,
379
+ content,
380
+ reasoning,
381
+ usage,
382
+ finishReason,
383
+ // Preserve callAPI's Chat Completions-shaped compatibility return. Pi's full
384
+ // normalized message is also available for future protocol-specific callers.
385
+ raw: state.raw || {
386
+ model: result.responseModel || result.model,
387
+ choices: [
388
+ { message: { content, reasoning_content: reasoning }, finish_reason: finishReason },
389
+ ],
390
+ usage: usage
391
+ ? {
392
+ prompt_tokens: usage.inputTokens,
393
+ completion_tokens: usage.outputTokens,
394
+ total_tokens: usage.totalTokens,
395
+ }
396
+ : null,
397
+ },
398
+ piMessage: result,
399
+ capabilities: adapter.capabilities,
400
+ attempts: state.attempts,
401
+ latencyMs: performance.now() - startedAt,
402
+ };
403
+ }
@@ -0,0 +1,148 @@
1
+ import { EventSourceParserStream } from 'eventsource-parser/stream';
2
+ import { normalizeUsage } from './providers.js';
3
+ import { ERROR_CATEGORIES, fail } from './errors.js';
4
+
5
+ function textValue(value, separator = '\n') {
6
+ if (typeof value === 'string') return value;
7
+ if (Array.isArray(value))
8
+ return value
9
+ .map((part) => textValue(part, separator))
10
+ .filter(Boolean)
11
+ .join(separator);
12
+ if (value && typeof value === 'object')
13
+ return textValue(value.text ?? value.summary ?? value.content, separator);
14
+ return '';
15
+ }
16
+
17
+ // Some compatible endpoints ignore stream=true; native Ollama uses complete
18
+ // JSON here as before. Convert that single response into one SDK-readable event.
19
+ // The event normalizer below uses the same field mappings for real SSE responses.
20
+ export function completionEvent(data) {
21
+ if (!data || typeof data !== 'object' || Array.isArray(data)) {
22
+ throw fail(
23
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
24
+ 'Provider returned an invalid response: expected a JSON object.',
25
+ );
26
+ }
27
+ if (data.error)
28
+ throw fail(
29
+ ERROR_CATEGORIES.PROVIDER,
30
+ `Provider request failed: ${textValue(data.error.message) || JSON.stringify(data.error)}`,
31
+ );
32
+ const choice = data.choices?.[0];
33
+ const message = choice?.message ?? data.message;
34
+ const usage = normalizeUsage(data.usage || data);
35
+ const finish = choice?.finish_reason ?? data.stop_reason ?? data.done_reason ?? 'stop';
36
+ return {
37
+ model: data.model,
38
+ choices: [
39
+ {
40
+ index: 0,
41
+ delta: {
42
+ content:
43
+ textValue(message?.content, '') ||
44
+ textValue(data.content, '') ||
45
+ textValue(data.response, ''),
46
+ reasoning_content: reasoningDelta(message) || textValue(data.thinking),
47
+ },
48
+ finish_reason: normalizeFinishReason(finish),
49
+ },
50
+ ],
51
+ ...(usage
52
+ ? {
53
+ usage: {
54
+ prompt_tokens: usage.inputTokens,
55
+ completion_tokens: usage.outputTokens,
56
+ total_tokens: usage.totalTokens,
57
+ },
58
+ }
59
+ : {}),
60
+ };
61
+ }
62
+
63
+ // Keep provider stop aliases out of Pi's error branch so the business recovery
64
+ // path receives a truncated result. Unknown and safety-related reasons stay intact.
65
+ function normalizeFinishReason(reason) {
66
+ if (['max_tokens', 'max_output_tokens', 'token_limit'].includes(reason)) return 'length';
67
+ return reason === 'end_turn' ? 'stop' : reason;
68
+ }
69
+
70
+ function detailText(value) {
71
+ if (Array.isArray(value)) return value.map(detailText).filter(Boolean).join('\n');
72
+ if (value?.type === 'reasoning.encrypted') return '';
73
+ return textValue(value);
74
+ }
75
+
76
+ function reasoningDelta(delta) {
77
+ if (!delta) return '';
78
+ const primary =
79
+ textValue(delta.reasoning_content) ||
80
+ textValue(delta.reasoning) ||
81
+ textValue(delta.reasoning_text) ||
82
+ textValue(delta.thinking);
83
+ const details = detailText(delta.reasoning_details);
84
+ // Providers sometimes expose the same delta in both representations. Only
85
+ // deduplicate within this event: repeated text in a later delta can be intentional.
86
+ if (!primary) return details;
87
+ if (!details || primary.startsWith(details)) return primary;
88
+ if (details.startsWith(primary)) return details;
89
+ return primary + details;
90
+ }
91
+
92
+ function normalizeChunk(value) {
93
+ if (!value || !Array.isArray(value.choices)) return value;
94
+ for (const choice of value.choices) {
95
+ if (!choice || typeof choice !== 'object') continue;
96
+ choice.finish_reason = normalizeFinishReason(choice.finish_reason);
97
+ const delta = choice.delta ?? choice.message;
98
+ if (!delta || typeof delta !== 'object') continue;
99
+ choice.delta = { ...delta, reasoning_content: reasoningDelta(delta) };
100
+ // Retain reasoning_details for Pi's replay metadata while also exposing all
101
+ // its textual segments as ordinary thinking deltas, including legacy shapes.
102
+ }
103
+ return value;
104
+ }
105
+
106
+ // eventsource-parser handles UTF-8-decoded SSE framing; this boundary adjusts
107
+ // only vendor fields. Pi still owns model-result assembly and finish validation.
108
+ // Web Stream piping preserves backpressure, cancellation and body read errors.
109
+ export function normalizeEventStream(response) {
110
+ if (!response.body)
111
+ throw fail(ERROR_CATEGORIES.RESPONSE_FORMAT, 'Streaming response did not include a body.');
112
+ let done = false;
113
+ const encoder = new globalThis.TextEncoder();
114
+ const body = response.body
115
+ .pipeThrough(new globalThis.TextDecoderStream())
116
+ .pipeThrough(new EventSourceParserStream())
117
+ .pipeThrough(
118
+ new globalThis.TransformStream({
119
+ transform(event, controller) {
120
+ if (done) return;
121
+ let data = event.data.trim();
122
+ if (data === '[DONE]') {
123
+ done = true;
124
+ } else {
125
+ let value;
126
+ try {
127
+ value = JSON.parse(data);
128
+ } catch (cause) {
129
+ throw fail(
130
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
131
+ 'Provider returned invalid JSON in streaming response.',
132
+ { cause },
133
+ );
134
+ }
135
+ data = JSON.stringify(normalizeChunk(value));
136
+ }
137
+ controller.enqueue(
138
+ encoder.encode(`${event.event ? `event: ${event.event}\n` : ''}data: ${data}\n\n`),
139
+ );
140
+ },
141
+ }),
142
+ );
143
+ const headers = new globalThis.Headers(response.headers);
144
+ headers.delete('content-length');
145
+ headers.delete('content-encoding');
146
+ headers.set('content-type', 'text/event-stream');
147
+ return new Response(body, { status: response.status, statusText: response.statusText, headers });
148
+ }