@hifullmoon/aicommit 2.6.6 → 2.6.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
- import { stream as streamPi } from '@earendil-works/pi-ai/api/openai-completions';
1
+ import OpenAI from 'openai';
2
2
  import { getProviderAdapter, normalizeUsage } from './providers.js';
3
3
  import { ERROR_CATEGORIES, fail } from './errors.js';
4
4
  import { completionEvent, normalizeEventStream } from './provider-response.js';
@@ -141,40 +141,6 @@ async function fetchWithRetry(
141
141
  throw new Error('Provider request exhausted its retry budget.');
142
142
  }
143
143
 
144
- function piContext(messages, model) {
145
- return {
146
- systemPrompt:
147
- messages
148
- .filter((m) => m.role === 'system')
149
- .map((m) => m.content)
150
- .join('\n\n') || undefined,
151
- messages: messages
152
- .filter((m) => m.role !== 'system')
153
- .map((message) => {
154
- if (message.role === 'user') return { ...message, timestamp: Date.now() };
155
- if (message.role !== 'assistant')
156
- throw new Error(`Unsupported generation message role: ${message.role}`);
157
- return {
158
- role: 'assistant',
159
- content: [{ type: 'text', text: message.content }],
160
- api: model.api,
161
- provider: model.provider,
162
- model: model.id,
163
- stopReason: 'stop',
164
- usage: {
165
- input: 0,
166
- output: 0,
167
- cacheRead: 0,
168
- cacheWrite: 0,
169
- totalTokens: 0,
170
- cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
171
- },
172
- timestamp: Date.now(),
173
- };
174
- }),
175
- };
176
- }
177
-
178
144
  function nativeOllamaPayload(payload, apiUrl) {
179
145
  const { max_tokens, temperature, stream_options: _streamOptions, options, ...rest } = payload;
180
146
  const body = {
@@ -200,7 +166,13 @@ function nativeOllamaPayload(payload, apiUrl) {
200
166
  function transport(config, adapter, state) {
201
167
  return async (_sdkUrl, init) => {
202
168
  try {
203
- const headers = new globalThis.Headers(init.headers);
169
+ // Build outbound headers from application-owned values: SDK headers can
170
+ // include arbitrary credentials from OPENAI_CUSTOM_HEADERS.
171
+ const headers = new globalThis.Headers({
172
+ Accept: 'application/json',
173
+ 'Content-Type': 'application/json',
174
+ ...adapter.headers,
175
+ });
204
176
  // Never let SDK defaults resolve a different credential or follow a redirect
205
177
  // carrying repository content to an endpoint the user did not configure.
206
178
  if (config.apiKey) headers.set('Authorization', `Bearer ${config.apiKey}`);
@@ -271,6 +243,77 @@ function transport(config, adapter, state) {
271
243
  };
272
244
  }
273
245
 
246
+ function requestPayload(adapter, request, options) {
247
+ const system = request.messages
248
+ .filter((message) => message.role === 'system')
249
+ .map((message) => message.content)
250
+ .join('\n\n');
251
+ const messages = request.messages
252
+ .filter((message) => message.role !== 'system')
253
+ .map((message) => {
254
+ if (!['user', 'assistant'].includes(message.role))
255
+ throw new Error(`Unsupported generation message role: ${message.role}`);
256
+ return {
257
+ role: message.role,
258
+ content: message.content,
259
+ ...((adapter.id === 'deepseek' ||
260
+ adapter.model.requiresReasoningContentOnAssistantMessages) &&
261
+ adapter.model.reasoning &&
262
+ message.role === 'assistant'
263
+ ? { reasoning_content: '' }
264
+ : {}),
265
+ };
266
+ });
267
+ if (system) messages.unshift({ role: 'system', content: system });
268
+ const payload = {
269
+ model: adapter.model.id,
270
+ messages,
271
+ stream: true,
272
+ [adapter.capabilities.tokenBudget === 'max_completion_tokens'
273
+ ? 'max_completion_tokens'
274
+ : 'max_tokens']: options.maxTokens,
275
+ ...(options.temperature !== undefined ? { temperature: options.temperature } : {}),
276
+ };
277
+ const model = adapter.model;
278
+ const effort = options.reasoningEffort;
279
+ const mapped = model.thinkingLevelMap?.[effort] ?? effort;
280
+ if (model.reasoning) {
281
+ if (adapter.id === 'deepseek') {
282
+ if (effort) payload.thinking = { type: 'enabled' };
283
+ else if (model.thinkingLevelMap?.off !== null) payload.thinking = { type: 'disabled' };
284
+ if (effort) payload.reasoning_effort = mapped;
285
+ } else if (adapter.id === 'openrouter') {
286
+ if (effort) payload.reasoning = { effort: mapped };
287
+ else if (model.thinkingLevelMap?.off !== null)
288
+ payload.reasoning = { effort: model.thinkingLevelMap?.off ?? 'none' };
289
+ } else if (adapter.id === 'openai') {
290
+ if (effort) payload.reasoning_effort = mapped;
291
+ else if (typeof model.thinkingLevelMap?.off === 'string')
292
+ payload.reasoning_effort = model.thinkingLevelMap.off;
293
+ }
294
+ }
295
+ options.onPayload(payload);
296
+ return payload;
297
+ }
298
+
299
+ function generationError(err, state, config) {
300
+ if (err.category) return err;
301
+ const cause = state.error || err;
302
+ if (cause.category) return cause;
303
+ const timeout = timeoutError(cause, config.timeoutMs || DEFAULT_TIMEOUT_MS);
304
+ if (timeout || /timed out|timeout/i.test(cause.message))
305
+ return fail(ERROR_CATEGORIES.NETWORK, timeout?.message || cause.message, { cause });
306
+ if (networkFailure(cause) || /socket|network|fetch failed|terminated|econn/i.test(cause.message))
307
+ return fail(ERROR_CATEGORIES.NETWORK, cause.message, { cause });
308
+ if (cause instanceof SyntaxError)
309
+ return fail(
310
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
311
+ `Provider returned invalid JSON: ${cause.message}`,
312
+ { cause },
313
+ );
314
+ return cause;
315
+ }
316
+
274
317
  export async function requestGeneration(config, request) {
275
318
  secureEndpoint(config.apiUrl);
276
319
  if (config.analysisBudget) {
@@ -293,97 +336,111 @@ export async function requestGeneration(config, request) {
293
336
  const state = { attempts: 0, raw: null, error: null };
294
337
  const startedAt = performance.now();
295
338
  const controller = new AbortController();
296
- const events = streamPi(adapter.model, piContext(request.messages, adapter.model), {
297
- ...options,
298
- // Pi's transport requires a key even for a keyless server. The placeholder
299
- // never leaves the process: transport installs only the resolved config key.
339
+ // An explicit key prevents SDK environment credential discovery. The transport
340
+ // removes this placeholder and sends only the resolved AICommit credential.
341
+ const client = new OpenAI({
300
342
  apiKey: config.apiKey || 'aicommit-keyless',
301
- headers: adapter.headers,
302
- env: {},
343
+ organization: null,
344
+ project: null,
345
+ logLevel: 'off',
346
+ baseURL: adapter.model.baseUrl,
347
+ defaultHeaders: adapter.headers,
303
348
  maxRetries: 0,
304
- timeoutMs: config.timeoutMs || DEFAULT_TIMEOUT_MS,
305
- signal: config.analysisBudget
306
- ? AbortSignal.any([controller.signal, config.analysisBudget.signal])
307
- : controller.signal,
349
+ timeout: config.timeoutMs || DEFAULT_TIMEOUT_MS,
308
350
  fetch: transport(config, adapter, state),
309
351
  });
310
- let result;
352
+ let content = '';
353
+ let reasoningText = '';
354
+ let responseModel = config.modelId;
355
+ let reportedUsage = null;
356
+ let finishReason = null;
311
357
  try {
358
+ const events = await client.chat.completions.create(requestPayload(adapter, request, options), {
359
+ signal: config.analysisBudget
360
+ ? AbortSignal.any([controller.signal, config.analysisBudget.signal])
361
+ : controller.signal,
362
+ });
312
363
  for await (const event of events) {
313
- if (event.type === 'thinking_delta') request.stream?.onReasoningDelta?.(event.delta);
314
- if (event.type === 'error') {
315
- if (state.error) throw state.error;
316
- const message = event.error.errorMessage || 'Provider request failed.';
317
- if (/without finish_reason/.test(message))
318
- throw fail(
319
- ERROR_CATEGORIES.RESPONSE_FORMAT,
320
- 'Streaming response ended before the provider sent a finish_reason. The partial response was discarded; retry the request.',
321
- );
322
- if (/timed out|timeout/i.test(message))
323
- throw fail(
324
- ERROR_CATEGORIES.NETWORK,
325
- `Request timed out after ${Math.round((config.timeoutMs || DEFAULT_TIMEOUT_MS) / 1000)}s — the model took too long to respond. Raise "timeoutMs" in your config if this keeps happening.`,
326
- );
327
- if (/socket|network|fetch failed|terminated|econn/i.test(message))
328
- throw fail(ERROR_CATEGORIES.NETWORK, message);
329
- if (/JSON|Unexpected token/i.test(message))
330
- throw fail(
331
- ERROR_CATEGORIES.RESPONSE_FORMAT,
332
- `Provider returned invalid JSON: ${message}`,
333
- );
334
- throw fail(ERROR_CATEGORIES.PROVIDER, `Provider request failed: ${message}`);
364
+ if (event.model) responseModel = event.model;
365
+ if (event.usage) reportedUsage = event.usage;
366
+ const choice = event.choices?.find((value) => (value.index ?? 0) === 0);
367
+ if (!choice) continue;
368
+ if (!event.usage && choice.usage) reportedUsage = choice.usage;
369
+ const delta = choice.delta;
370
+ if (typeof delta?.content === 'string') content += delta.content;
371
+ if (typeof delta?.reasoning_content === 'string' && delta.reasoning_content) {
372
+ reasoningText += delta.reasoning_content;
373
+ request.stream?.onReasoningDelta?.(delta.reasoning_content);
374
+ }
375
+ if (choice.finish_reason != null) {
376
+ finishReason = choice.finish_reason;
377
+ if (!['stop', 'end', 'length', 'tool_calls', 'function_call'].includes(finishReason))
378
+ throw fail(ERROR_CATEGORIES.PROVIDER, `Provider finish_reason: ${finishReason}`);
335
379
  }
336
- if (event.type === 'done') result = event.message;
337
380
  }
381
+ // The SDK treats an AbortError as the end of iteration. An interrupted
382
+ // response must still fail, even if a finish marker arrived before it.
383
+ if (events.controller.signal.aborted)
384
+ throw fail(ERROR_CATEGORIES.NETWORK, 'Provider response stream was aborted.');
385
+ if (finishReason === null)
386
+ throw fail(
387
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
388
+ 'Streaming response ended before the provider sent a finish_reason. The partial response was discarded; retry the request.',
389
+ );
390
+ } catch (err) {
391
+ throw generationError(err, state, config);
338
392
  } finally {
339
393
  controller.abort();
340
394
  }
341
- if (!result)
342
- throw fail(
343
- ERROR_CATEGORIES.RESPONSE_FORMAT,
344
- 'Provider returned an invalid response: no completed generation.',
345
- );
346
- const content = result.content
347
- .filter((block) => block.type === 'text')
348
- .map((block) => block.text)
349
- .join('');
350
- const reasoning =
351
- result.content
352
- .filter((block) => block.type === 'thinking')
353
- .map((block) => block.thinking)
354
- .filter(Boolean)
355
- .join('\n') || null;
356
395
  const usage = state.raw
357
- ? normalizeUsage(state.raw.usage || state.raw)
358
- : result.usage.totalTokens ||
359
- result.usage.input ||
360
- result.usage.output ||
361
- result.usage.cacheRead ||
362
- result.usage.cacheWrite
363
- ? {
364
- inputTokens: result.usage.input + result.usage.cacheRead + result.usage.cacheWrite,
365
- outputTokens: result.usage.output,
366
- totalTokens: result.usage.totalTokens,
367
- }
368
- : null;
369
- const finishReason =
370
- state.raw?.choices?.[0]?.finish_reason ??
371
- state.raw?.stop_reason ??
372
- state.raw?.done_reason ??
373
- result.rawStopReason ??
374
- result.stopReason;
396
+ ? normalizeUsage(state.raw.usage || state.raw.choices?.[0]?.usage || state.raw)
397
+ : normalizeUsage(reportedUsage);
398
+ const reasoning = reasoningText || null;
399
+ const cacheRead =
400
+ reportedUsage?.prompt_tokens_details?.cached_tokens ??
401
+ reportedUsage?.prompt_cache_hit_tokens ??
402
+ reportedUsage?.cached_tokens ??
403
+ 0;
404
+ const cacheWrite = reportedUsage?.prompt_tokens_details?.cache_write_tokens || 0;
405
+ const blocks = [];
406
+ if (content) blocks.push({ type: 'text', text: content });
407
+ if (reasoning) blocks.push({ type: 'thinking', thinking: reasoning });
408
+ // Retain the previous public result shape for callers during the SDK migration.
409
+ const assistantMessage = {
410
+ role: 'assistant',
411
+ api: 'openai-completions',
412
+ provider: adapter.id,
413
+ model: config.modelId,
414
+ responseModel,
415
+ content: blocks,
416
+ usage: {
417
+ input: Math.max(0, (usage?.inputTokens || 0) - cacheRead - cacheWrite),
418
+ output: usage?.outputTokens || 0,
419
+ cacheRead,
420
+ cacheWrite,
421
+ totalTokens: usage?.totalTokens || 0,
422
+ },
423
+ stopReason: ['tool_calls', 'function_call'].includes(finishReason)
424
+ ? 'toolUse'
425
+ : finishReason === 'end'
426
+ ? 'stop'
427
+ : finishReason,
428
+ timestamp: Date.now(),
429
+ };
375
430
  config.analysisBudget?.settle(state.ticket, usage);
376
431
  return {
377
432
  provider: adapter.id,
378
- model: result.responseModel || result.model,
433
+ model: responseModel,
379
434
  content,
380
435
  reasoning,
381
436
  usage,
382
- finishReason,
383
- // Preserve callAPI's Chat Completions-shaped compatibility return. Pi's full
384
- // normalized message is also available for future protocol-specific callers.
437
+ finishReason:
438
+ state.raw?.choices?.[0]?.finish_reason ??
439
+ state.raw?.stop_reason ??
440
+ state.raw?.done_reason ??
441
+ finishReason,
385
442
  raw: state.raw || {
386
- model: result.responseModel || result.model,
443
+ model: responseModel,
387
444
  choices: [
388
445
  { message: { content, reasoning_content: reasoning }, finish_reason: finishReason },
389
446
  ],
@@ -395,7 +452,7 @@ export async function requestGeneration(config, request) {
395
452
  }
396
453
  : null,
397
454
  },
398
- piMessage: result,
455
+ piMessage: assistantMessage,
399
456
  capabilities: adapter.capabilities,
400
457
  attempts: state.attempts,
401
458
  latencyMs: performance.now() - startedAt,
@@ -31,7 +31,7 @@ export function completionEvent(data) {
31
31
  );
32
32
  const choice = data.choices?.[0];
33
33
  const message = choice?.message ?? data.message;
34
- const usage = normalizeUsage(data.usage || data);
34
+ const usage = normalizeUsage(data.usage || choice?.usage || data);
35
35
  const finish = choice?.finish_reason ?? data.stop_reason ?? data.done_reason ?? 'stop';
36
36
  return {
37
37
  model: data.model,
@@ -60,7 +60,7 @@ export function completionEvent(data) {
60
60
  };
61
61
  }
62
62
 
63
- // Keep provider stop aliases out of Pi's error branch so the business recovery
63
+ // Keep provider stop aliases out of the error branch so the business recovery
64
64
  // path receives a truncated result. Unknown and safety-related reasons stay intact.
65
65
  function normalizeFinishReason(reason) {
66
66
  if (['max_tokens', 'max_output_tokens', 'token_limit'].includes(reason)) return 'length';
@@ -110,7 +110,18 @@ function miniMaxIncrement(value, state) {
110
110
  }
111
111
 
112
112
  function normalizeChunk(value, miniMaxChoices = null) {
113
- if (!value || !Array.isArray(value.choices)) return value;
113
+ if (
114
+ !value ||
115
+ typeof value !== 'object' ||
116
+ Array.isArray(value) ||
117
+ (value.choices !== undefined && !Array.isArray(value.choices)) ||
118
+ value.choices?.some((choice) => !choice || typeof choice !== 'object' || Array.isArray(choice))
119
+ )
120
+ throw fail(
121
+ ERROR_CATEGORIES.RESPONSE_FORMAT,
122
+ 'Provider returned an invalid streaming response: expected an object with a choices array.',
123
+ );
124
+ if (!value.choices) return value;
114
125
  for (const choice of value.choices) {
115
126
  if (!choice || typeof choice !== 'object') continue;
116
127
  choice.finish_reason = normalizeFinishReason(choice.finish_reason);
@@ -133,14 +144,14 @@ function normalizeChunk(value, miniMaxChoices = null) {
133
144
  } else {
134
145
  choice.delta = { ...delta, reasoning_content: reasoning };
135
146
  }
136
- // Retain reasoning_details for Pi's replay metadata while also exposing all
147
+ // Retain reasoning_details metadata while also exposing all
137
148
  // its textual segments as ordinary thinking deltas, including legacy shapes.
138
149
  }
139
150
  return value;
140
151
  }
141
152
 
142
153
  // eventsource-parser handles UTF-8-decoded SSE framing; this boundary adjusts
143
- // only vendor fields. Pi still owns model-result assembly and finish validation.
154
+ // only vendor fields. The client assembles results and validates completion.
144
155
  // Web Stream piping preserves backpressure, cancellation and body read errors.
145
156
  export function normalizeEventStream(response, provider = '') {
146
157
  if (!response.body)
@@ -169,6 +180,11 @@ export function normalizeEventStream(response, provider = '') {
169
180
  { cause },
170
181
  );
171
182
  }
183
+ if (value?.error || event.event === 'error')
184
+ throw fail(
185
+ ERROR_CATEGORIES.PROVIDER,
186
+ `Provider request failed: ${textValue(value?.error?.message ?? value?.message) || JSON.stringify(value)}`,
187
+ );
172
188
  data = JSON.stringify(normalizeChunk(value, miniMaxChoices));
173
189
  }
174
190
  controller.enqueue(
package/src/providers.js CHANGED
@@ -1,7 +1,27 @@
1
- import { clampThinkingLevel, getSupportedThinkingLevels } from '@earendil-works/pi-ai';
2
- import { OPENAI_MODELS } from '@earendil-works/pi-ai/providers/openai.models';
3
- import { DEEPSEEK_MODELS } from '@earendil-works/pi-ai/providers/deepseek.models';
4
- import { OPENROUTER_MODELS } from '@earendil-works/pi-ai/providers/openrouter.models';
1
+ import modelCapabilities from './model-capabilities.json' with { type: 'json' };
2
+
3
+ function getSupportedThinkingLevels(model) {
4
+ if (!model.reasoning) return ['off'];
5
+ return ['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max'].filter((level) => {
6
+ const mapped = model.thinkingLevelMap?.[level];
7
+ return mapped !== null && (!['xhigh', 'max'].includes(level) || mapped !== undefined);
8
+ });
9
+ }
10
+
11
+ function clampThinkingLevel(model, level) {
12
+ const levels = ['off', 'minimal', 'low', 'medium', 'high', 'xhigh', 'max'];
13
+ const supported = getSupportedThinkingLevels(model);
14
+ if (supported.includes(level)) return level;
15
+ const index = levels.indexOf(level);
16
+ return (
17
+ levels.slice(index).find((candidate) => supported.includes(candidate)) ??
18
+ levels
19
+ .slice(0, index)
20
+ .reverse()
21
+ .find((candidate) => supported.includes(candidate)) ??
22
+ 'off'
23
+ );
24
+ }
5
25
  export const PROVIDER_TYPES = Object.freeze([
6
26
  'openai',
7
27
  'openrouter',
@@ -86,15 +106,17 @@ export function normalizeUsage(usage) {
86
106
  return Object.keys(normalized).length ? normalized : null;
87
107
  }
88
108
 
89
- // Read Pi's bundled catalog locally. Credentials and network model discovery remain
109
+ // Read the local capability snapshot. Credentials and network model discovery remain
90
110
  // outside this layer; a configured model ID need not be present in the catalog.
91
111
  function catalogModel(provider, modelId) {
92
- if (provider === 'openai') {
93
- return OPENAI_MODELS[modelId] || OPENAI_MODELS[modelId.replace(/-codex$/, '')];
94
- }
95
- if (provider === 'deepseek') return DEEPSEEK_MODELS[modelId];
96
- if (provider === 'openrouter') return OPENROUTER_MODELS[modelId];
97
- return undefined;
112
+ const models = modelCapabilities.providers[provider];
113
+ if (!models) return undefined;
114
+ const id = Object.hasOwn(models, modelId)
115
+ ? modelId
116
+ : provider === 'openai'
117
+ ? modelId.replace(/-codex$/, '')
118
+ : modelId;
119
+ return Object.hasOwn(models, id) ? modelCapabilities.profiles[models[id]] : undefined;
98
120
  }
99
121
 
100
122
  export function isOpenAIReasoningModel(modelId = '') {
@@ -139,8 +161,7 @@ function resolveEffort(provider, model, reasoning) {
139
161
  return level === 'off' ? undefined : level;
140
162
  }
141
163
 
142
- // The application selects a Pi model and applies only configuration compatibility
143
- // overrides. Pi owns message conversion, model effort mapping and wire protocols.
164
+ // Keep vendor parameter mappings separate from the OpenAI SDK transport.
144
165
  export function getProviderAdapter({ apiUrl, providerType = '', modelId = '' }) {
145
166
  const provider = detectProviderType(apiUrl, providerType);
146
167
  const nativeOllama =
@@ -151,28 +172,14 @@ export function getProviderAdapter({ apiUrl, providerType = '', modelId = '' })
151
172
  const model = {
152
173
  ...known,
153
174
  id: modelId,
154
- name: known?.name || modelId,
155
175
  api: 'openai-completions',
156
176
  provider,
157
177
  // The transport pins the full configured URL, including nonstandard proxy paths.
158
178
  baseUrl: endpoint(apiUrl)?.origin || '',
159
179
  reasoning:
160
180
  known?.reasoning ?? (openAIReasoning || ['deepseek', 'openrouter'].includes(provider)),
161
- input: ['text'],
162
- cost: known?.cost || { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
163
181
  contextWindow: known?.contextWindow || 128000,
164
182
  maxTokens: known?.maxTokens || 16384,
165
- compat: {
166
- ...(known?.api === 'openai-completions' ? known.compat : {}),
167
- supportsStore: false,
168
- supportsDeveloperRole: false,
169
- supportsReasoningEffort: ['openai', 'deepseek', 'openrouter'].includes(provider),
170
- supportsUsageInStreaming: provider === 'openai',
171
- supportsFinishReason: true,
172
- maxTokensField: tokenField,
173
- thinkingFormat:
174
- provider === 'deepseek' ? 'deepseek' : provider === 'openrouter' ? 'openrouter' : 'openai',
175
- },
176
183
  };
177
184
  const capabilities = Object.freeze({
178
185
  streaming: !nativeOllama,
@@ -200,9 +207,8 @@ export function getProviderAdapter({ apiUrl, providerType = '', modelId = '' })
200
207
  temperature:
201
208
  openAIReasoning || (provider === 'deepseek' && mode === 'on') ? undefined : temperature,
202
209
  reasoningEffort,
203
- cacheRetention: 'none',
204
210
  onPayload(payload) {
205
- // Pi's absent effort means "off"; AICommit's auto means server defaults.
211
+ // Auto preserves server defaults; explicit modes take precedence over extras.
206
212
  const reasoningKeys = ['thinking', 'reasoning', 'reasoning_effort'];
207
213
  const mapped = Object.fromEntries(
208
214
  reasoningKeys