@hifullmoon/aicommit 2.3.0 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.aicommit.config.example.json +31 -8
- package/CHANGELOG.md +14 -1
- package/README.md +31 -5
- package/README.zh-CN.md +31 -5
- package/bin/aicommit.js +6 -1
- package/docs/distribution.md +1 -1
- package/docs/large-change-implementation-plan.md +193 -0
- package/docs/privacy.md +6 -0
- package/docs/provider-compatibility.md +44 -16
- package/docs/troubleshooting.md +10 -0
- package/package.json +4 -2
- package/schemas/aicommit-output.schema.json +120 -18
- package/src/analysis-budget.js +76 -0
- package/src/api.js +6 -316
- package/src/change-analysis.js +558 -0
- package/src/config.js +16 -0
- package/src/doctor.js +2 -4
- package/src/git-spool.js +108 -0
- package/src/git.js +25 -4
- package/src/local-analysis.js +271 -0
- package/src/main.js +114 -25
- package/src/model-client.js +403 -0
- package/src/provider-response.js +148 -0
- package/src/providers.js +134 -206
- package/src/runtime.js +6 -0
- package/src/split.js +193 -42
|
@@ -19,9 +19,15 @@
|
|
|
19
19
|
"error"
|
|
20
20
|
],
|
|
21
21
|
"properties": {
|
|
22
|
-
"schemaVersion": {
|
|
23
|
-
|
|
24
|
-
|
|
22
|
+
"schemaVersion": {
|
|
23
|
+
"const": "1.0"
|
|
24
|
+
},
|
|
25
|
+
"ok": {
|
|
26
|
+
"type": "boolean"
|
|
27
|
+
},
|
|
28
|
+
"message": {
|
|
29
|
+
"type": ["string", "null"]
|
|
30
|
+
},
|
|
25
31
|
"plan": {
|
|
26
32
|
"type": ["array", "null"],
|
|
27
33
|
"items": {
|
|
@@ -29,8 +35,15 @@
|
|
|
29
35
|
"additionalProperties": false,
|
|
30
36
|
"required": ["message", "files"],
|
|
31
37
|
"properties": {
|
|
32
|
-
"message": {
|
|
33
|
-
|
|
38
|
+
"message": {
|
|
39
|
+
"type": "string"
|
|
40
|
+
},
|
|
41
|
+
"files": {
|
|
42
|
+
"type": "array",
|
|
43
|
+
"items": {
|
|
44
|
+
"type": "string"
|
|
45
|
+
}
|
|
46
|
+
},
|
|
34
47
|
"hunks": {
|
|
35
48
|
"type": "array",
|
|
36
49
|
"items": {
|
|
@@ -38,29 +51,62 @@
|
|
|
38
51
|
"additionalProperties": false,
|
|
39
52
|
"required": ["path", "ids"],
|
|
40
53
|
"properties": {
|
|
41
|
-
"path": {
|
|
42
|
-
|
|
54
|
+
"path": {
|
|
55
|
+
"type": "string"
|
|
56
|
+
},
|
|
57
|
+
"ids": {
|
|
58
|
+
"type": "array",
|
|
59
|
+
"items": {
|
|
60
|
+
"type": "string"
|
|
61
|
+
}
|
|
62
|
+
}
|
|
43
63
|
}
|
|
44
64
|
}
|
|
45
65
|
}
|
|
46
66
|
}
|
|
47
67
|
}
|
|
48
68
|
},
|
|
49
|
-
"provider": {
|
|
50
|
-
|
|
51
|
-
|
|
69
|
+
"provider": {
|
|
70
|
+
"type": ["string", "null"]
|
|
71
|
+
},
|
|
72
|
+
"model": {
|
|
73
|
+
"type": ["string", "null"]
|
|
74
|
+
},
|
|
75
|
+
"latencyMs": {
|
|
76
|
+
"type": ["number", "null"],
|
|
77
|
+
"minimum": 0
|
|
78
|
+
},
|
|
52
79
|
"usage": {
|
|
53
80
|
"type": ["object", "null"],
|
|
54
81
|
"additionalProperties": false,
|
|
55
82
|
"properties": {
|
|
56
|
-
"inputTokens": {
|
|
57
|
-
|
|
58
|
-
|
|
83
|
+
"inputTokens": {
|
|
84
|
+
"type": "number",
|
|
85
|
+
"minimum": 0
|
|
86
|
+
},
|
|
87
|
+
"outputTokens": {
|
|
88
|
+
"type": "number",
|
|
89
|
+
"minimum": 0
|
|
90
|
+
},
|
|
91
|
+
"totalTokens": {
|
|
92
|
+
"type": "number",
|
|
93
|
+
"minimum": 0
|
|
94
|
+
}
|
|
59
95
|
}
|
|
60
96
|
},
|
|
61
|
-
"warnings": {
|
|
62
|
-
|
|
63
|
-
|
|
97
|
+
"warnings": {
|
|
98
|
+
"type": "array",
|
|
99
|
+
"items": {
|
|
100
|
+
"type": "string"
|
|
101
|
+
}
|
|
102
|
+
},
|
|
103
|
+
"exitReason": {
|
|
104
|
+
"type": "string",
|
|
105
|
+
"minLength": 1
|
|
106
|
+
},
|
|
107
|
+
"committed": {
|
|
108
|
+
"type": "boolean"
|
|
109
|
+
},
|
|
64
110
|
"error": {
|
|
65
111
|
"type": ["object", "null"],
|
|
66
112
|
"additionalProperties": false,
|
|
@@ -78,9 +124,65 @@
|
|
|
78
124
|
"internal"
|
|
79
125
|
]
|
|
80
126
|
},
|
|
81
|
-
"message": {
|
|
127
|
+
"message": {
|
|
128
|
+
"type": "string"
|
|
129
|
+
}
|
|
82
130
|
}
|
|
83
131
|
},
|
|
84
|
-
"data": {
|
|
132
|
+
"data": {
|
|
133
|
+
"type": "object",
|
|
134
|
+
"properties": {
|
|
135
|
+
"analysis": {
|
|
136
|
+
"type": "object",
|
|
137
|
+
"description": "Large-change coverage and shared request budget; never contains diff or intermediate summaries.",
|
|
138
|
+
"properties": {
|
|
139
|
+
"totalFiles": {
|
|
140
|
+
"type": "number",
|
|
141
|
+
"minimum": 0
|
|
142
|
+
},
|
|
143
|
+
"analyzedFiles": {
|
|
144
|
+
"type": "number",
|
|
145
|
+
"minimum": 0
|
|
146
|
+
},
|
|
147
|
+
"sampledFiles": {
|
|
148
|
+
"type": "number",
|
|
149
|
+
"minimum": 0,
|
|
150
|
+
"description": "Files represented by partial excerpts; not fully analyzed files."
|
|
151
|
+
},
|
|
152
|
+
"strategy": {
|
|
153
|
+
"enum": ["auto", "deep"]
|
|
154
|
+
},
|
|
155
|
+
"metadataOnlyFiles": {
|
|
156
|
+
"type": "number",
|
|
157
|
+
"minimum": 0
|
|
158
|
+
},
|
|
159
|
+
"failedFiles": {
|
|
160
|
+
"type": "number",
|
|
161
|
+
"minimum": 0
|
|
162
|
+
},
|
|
163
|
+
"completedChunks": {
|
|
164
|
+
"type": "number",
|
|
165
|
+
"minimum": 0
|
|
166
|
+
},
|
|
167
|
+
"requests": {
|
|
168
|
+
"type": "number",
|
|
169
|
+
"minimum": 0
|
|
170
|
+
},
|
|
171
|
+
"budgetedTokens": {
|
|
172
|
+
"type": "number",
|
|
173
|
+
"minimum": 0
|
|
174
|
+
},
|
|
175
|
+
"maxTotalTokens": {
|
|
176
|
+
"type": "number",
|
|
177
|
+
"minimum": 0
|
|
178
|
+
},
|
|
179
|
+
"elapsedMs": {
|
|
180
|
+
"type": "number",
|
|
181
|
+
"minimum": 0
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
}
|
|
85
187
|
}
|
|
86
188
|
}
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
import { ERROR_CATEGORIES, fail } from './errors.js';
|
|
2
|
+
|
|
3
|
+
export const DEFAULT_LARGE_CHANGE = Object.freeze({
|
|
4
|
+
strategy: 'auto',
|
|
5
|
+
chunkInputTokens: 12000,
|
|
6
|
+
maxTotalTokens: 200000,
|
|
7
|
+
concurrency: 2,
|
|
8
|
+
timeoutMs: 180000,
|
|
9
|
+
});
|
|
10
|
+
|
|
11
|
+
export function estimateTokens(text) {
|
|
12
|
+
// Deliberately conservative fallback for providers without a tokenizer.
|
|
13
|
+
return Math.ceil(Buffer.byteLength(String(text), 'utf8') / 2);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export function createAnalysisBudget(settings = {}) {
|
|
17
|
+
const limits = { ...DEFAULT_LARGE_CHANGE, ...settings };
|
|
18
|
+
const started = Date.now();
|
|
19
|
+
let charged = 0;
|
|
20
|
+
let requests = 0;
|
|
21
|
+
let inputTokens = 0;
|
|
22
|
+
let outputTokens = 0;
|
|
23
|
+
let reserveFinal = 0;
|
|
24
|
+
return {
|
|
25
|
+
limits,
|
|
26
|
+
signal: AbortSignal.timeout(limits.timeoutMs),
|
|
27
|
+
reserveFinal(tokens) {
|
|
28
|
+
reserveFinal = tokens;
|
|
29
|
+
},
|
|
30
|
+
remainingMs() {
|
|
31
|
+
return Math.max(0, limits.timeoutMs - (Date.now() - started));
|
|
32
|
+
},
|
|
33
|
+
snapshot() {
|
|
34
|
+
return {
|
|
35
|
+
requests,
|
|
36
|
+
budgetedTokens: charged,
|
|
37
|
+
maxTotalTokens: limits.maxTotalTokens,
|
|
38
|
+
elapsedMs: Date.now() - started,
|
|
39
|
+
usage: { inputTokens, outputTokens, totalTokens: inputTokens + outputTokens },
|
|
40
|
+
};
|
|
41
|
+
},
|
|
42
|
+
reserve(input, output) {
|
|
43
|
+
if (
|
|
44
|
+
!this.remainingMs() ||
|
|
45
|
+
charged + input + output + reserveFinal > limits.maxTotalTokens ||
|
|
46
|
+
requests >= 256
|
|
47
|
+
) {
|
|
48
|
+
throw fail(
|
|
49
|
+
ERROR_CATEGORIES.PROVIDER,
|
|
50
|
+
'Large-change analysis reached its token, request, or time budget. No incomplete plan will be committed. Increase the personal largeChange budget or stage a smaller change.',
|
|
51
|
+
{ data: { analysis: this.snapshot() } },
|
|
52
|
+
);
|
|
53
|
+
}
|
|
54
|
+
if (input > limits.chunkInputTokens) {
|
|
55
|
+
throw fail(
|
|
56
|
+
ERROR_CATEGORIES.PROVIDER,
|
|
57
|
+
'Analysis request exceeds largeChange.chunkInputTokens; shorten repository context or increase the personal input budget.',
|
|
58
|
+
);
|
|
59
|
+
}
|
|
60
|
+
charged += input + output;
|
|
61
|
+
requests += 1;
|
|
62
|
+
inputTokens += input;
|
|
63
|
+
outputTokens += output;
|
|
64
|
+
return { input, output };
|
|
65
|
+
},
|
|
66
|
+
settle(ticket, usage) {
|
|
67
|
+
if (!ticket || !usage) return;
|
|
68
|
+
const actualInput = usage.inputTokens ?? ticket.input;
|
|
69
|
+
const actualOutput = usage.outputTokens ?? ticket.output;
|
|
70
|
+
charged +=
|
|
71
|
+
Math.max(usage.totalTokens || 0, actualInput + actualOutput) - ticket.input - ticket.output;
|
|
72
|
+
inputTokens += actualInput - ticket.input;
|
|
73
|
+
outputTokens += actualOutput - ticket.output;
|
|
74
|
+
},
|
|
75
|
+
};
|
|
76
|
+
}
|
package/src/api.js
CHANGED
|
@@ -1,5 +1,7 @@
|
|
|
1
1
|
import { cleanCommitMessage } from './utils.js';
|
|
2
2
|
import { getProviderAdapter, normalizeUsage } from './providers.js';
|
|
3
|
+
import { requestGeneration } from './model-client.js';
|
|
4
|
+
export { requestGeneration } from './model-client.js';
|
|
3
5
|
import { ERROR_CATEGORIES, fail } from './errors.js';
|
|
4
6
|
import {
|
|
5
7
|
buildCommitPolicyPrompt,
|
|
@@ -9,9 +11,6 @@ import {
|
|
|
9
11
|
} from './policy.js';
|
|
10
12
|
import { encodeUntrustedData } from './trust.js';
|
|
11
13
|
|
|
12
|
-
// Default per-request timeout; overridable via the "timeoutMs" config key.
|
|
13
|
-
const DEFAULT_TIMEOUT_MS = 120_000;
|
|
14
|
-
|
|
15
14
|
export async function callAPI(
|
|
16
15
|
apiUrl,
|
|
17
16
|
apiKey,
|
|
@@ -41,294 +40,6 @@ export async function callAPI(
|
|
|
41
40
|
return result.raw;
|
|
42
41
|
}
|
|
43
42
|
|
|
44
|
-
const DEFAULT_RETRY_POLICY = Object.freeze({
|
|
45
|
-
maxAttempts: 3,
|
|
46
|
-
baseDelayMs: 500,
|
|
47
|
-
maxDelayMs: 5000,
|
|
48
|
-
});
|
|
49
|
-
const RETRYABLE_STATUS = new Set([429, 500, 502, 503, 504]);
|
|
50
|
-
const RETRYABLE_NETWORK_CODES = new Set([
|
|
51
|
-
'ECONNRESET',
|
|
52
|
-
'ECONNREFUSED',
|
|
53
|
-
'EHOSTUNREACH',
|
|
54
|
-
'ENETUNREACH',
|
|
55
|
-
'EPIPE',
|
|
56
|
-
'UND_ERR_CONNECT_TIMEOUT',
|
|
57
|
-
'UND_ERR_SOCKET',
|
|
58
|
-
]);
|
|
59
|
-
|
|
60
|
-
function secureEndpoint(apiUrl) {
|
|
61
|
-
const endpoint = new URL(apiUrl);
|
|
62
|
-
const loopback =
|
|
63
|
-
endpoint.hostname === 'localhost' ||
|
|
64
|
-
endpoint.hostname === '127.0.0.1' ||
|
|
65
|
-
endpoint.hostname.startsWith('127.') ||
|
|
66
|
-
endpoint.hostname === '[::1]';
|
|
67
|
-
if (endpoint.protocol !== 'https:' && !(endpoint.protocol === 'http:' && loopback)) {
|
|
68
|
-
throw new Error(
|
|
69
|
-
'Refusing insecure API endpoint: use HTTPS, or HTTP only for localhost/loopback.',
|
|
70
|
-
);
|
|
71
|
-
}
|
|
72
|
-
return endpoint;
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
function retryPolicy(value = {}) {
|
|
76
|
-
return {
|
|
77
|
-
maxAttempts: value?.maxAttempts ?? DEFAULT_RETRY_POLICY.maxAttempts,
|
|
78
|
-
baseDelayMs: value?.baseDelayMs ?? DEFAULT_RETRY_POLICY.baseDelayMs,
|
|
79
|
-
maxDelayMs: value?.maxDelayMs ?? DEFAULT_RETRY_POLICY.maxDelayMs,
|
|
80
|
-
sleep:
|
|
81
|
-
value?.sleep ??
|
|
82
|
-
((delayMs) =>
|
|
83
|
-
new Promise((resolve) => {
|
|
84
|
-
globalThis.setTimeout(resolve, delayMs);
|
|
85
|
-
})),
|
|
86
|
-
now: value?.now ?? (() => Date.now()),
|
|
87
|
-
};
|
|
88
|
-
}
|
|
89
|
-
|
|
90
|
-
function retryAfterMs(value, now) {
|
|
91
|
-
if (!value) return null;
|
|
92
|
-
const seconds = Number(value);
|
|
93
|
-
if (Number.isFinite(seconds) && seconds >= 0) return seconds * 1000;
|
|
94
|
-
const date = Date.parse(value);
|
|
95
|
-
if (Number.isNaN(date)) return null;
|
|
96
|
-
return Math.max(0, date - now());
|
|
97
|
-
}
|
|
98
|
-
|
|
99
|
-
function networkFailure(err) {
|
|
100
|
-
if (err instanceof TypeError) return true;
|
|
101
|
-
return RETRYABLE_NETWORK_CODES.has(err?.code) || RETRYABLE_NETWORK_CODES.has(err?.cause?.code);
|
|
102
|
-
}
|
|
103
|
-
|
|
104
|
-
function timeoutError(err, timeout) {
|
|
105
|
-
if (err?.name !== 'TimeoutError' && err?.name !== 'AbortError') return null;
|
|
106
|
-
return new Error(
|
|
107
|
-
`Request timed out after ${Math.round(timeout / 1000)}s — the model took too long to respond. ` +
|
|
108
|
-
`Raise "timeoutMs" in your config if this keeps happening.`,
|
|
109
|
-
);
|
|
110
|
-
}
|
|
111
|
-
|
|
112
|
-
async function fetchWithRetry(apiUrl, init, timeout, configuredPolicy, consume) {
|
|
113
|
-
const policy = retryPolicy(configuredPolicy);
|
|
114
|
-
let attempt = 0;
|
|
115
|
-
|
|
116
|
-
while (attempt < policy.maxAttempts) {
|
|
117
|
-
attempt += 1;
|
|
118
|
-
let response;
|
|
119
|
-
try {
|
|
120
|
-
response = await fetch(apiUrl, {
|
|
121
|
-
...init,
|
|
122
|
-
signal: AbortSignal.timeout(timeout),
|
|
123
|
-
});
|
|
124
|
-
} catch (err) {
|
|
125
|
-
const wrappedTimeout = timeoutError(err, timeout);
|
|
126
|
-
if (wrappedTimeout) throw wrappedTimeout;
|
|
127
|
-
if (!networkFailure(err) || attempt >= policy.maxAttempts) throw err;
|
|
128
|
-
const delay = Math.min(policy.baseDelayMs * 2 ** (attempt - 1), policy.maxDelayMs);
|
|
129
|
-
await policy.sleep(delay);
|
|
130
|
-
continue;
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
if (response.ok) {
|
|
134
|
-
try {
|
|
135
|
-
return { value: await consume(response), attempts: attempt };
|
|
136
|
-
} catch (err) {
|
|
137
|
-
const wrappedTimeout = timeoutError(err, timeout);
|
|
138
|
-
if (wrappedTimeout) throw wrappedTimeout;
|
|
139
|
-
// Once the provider has accepted a generation request, replaying it is
|
|
140
|
-
// unsafe: the first request may already have completed and been billed
|
|
141
|
-
// even though its response body was interrupted locally.
|
|
142
|
-
throw err;
|
|
143
|
-
}
|
|
144
|
-
}
|
|
145
|
-
if (RETRYABLE_STATUS.has(response.status) && attempt < policy.maxAttempts) {
|
|
146
|
-
const requestedDelay = retryAfterMs(response.headers.get('retry-after'), policy.now);
|
|
147
|
-
if (requestedDelay !== null && requestedDelay > policy.maxDelayMs) {
|
|
148
|
-
await response.body?.cancel().catch(() => {});
|
|
149
|
-
throw new Error(
|
|
150
|
-
`HTTP ${response.status}: provider requested a retry after ${Math.ceil(
|
|
151
|
-
requestedDelay / 1000,
|
|
152
|
-
)}s, exceeding the configured retry.maxDelayMs limit.`,
|
|
153
|
-
);
|
|
154
|
-
}
|
|
155
|
-
const delay =
|
|
156
|
-
requestedDelay ?? Math.min(policy.baseDelayMs * 2 ** (attempt - 1), policy.maxDelayMs);
|
|
157
|
-
await response.body?.cancel().catch(() => {});
|
|
158
|
-
await policy.sleep(delay);
|
|
159
|
-
continue;
|
|
160
|
-
}
|
|
161
|
-
|
|
162
|
-
const errText = await response.text();
|
|
163
|
-
throw new Error(`HTTP ${response.status}: ${errText.slice(0, 400)}`);
|
|
164
|
-
}
|
|
165
|
-
|
|
166
|
-
throw new Error('Provider request exhausted its retry budget.');
|
|
167
|
-
}
|
|
168
|
-
|
|
169
|
-
// Unified provider request contract. Provider adapters own request dialects
|
|
170
|
-
// and response normalization; callers receive the same shape regardless of
|
|
171
|
-
// whether the endpoint is OpenAI-compatible or native Ollama.
|
|
172
|
-
export async function requestGeneration(config, request) {
|
|
173
|
-
secureEndpoint(config.apiUrl);
|
|
174
|
-
const timeout = config.timeoutMs || DEFAULT_TIMEOUT_MS;
|
|
175
|
-
const adapter = getProviderAdapter(config);
|
|
176
|
-
const payload = await adapter.buildRequest({
|
|
177
|
-
messages: request.messages,
|
|
178
|
-
temperature: request.temperature,
|
|
179
|
-
maxTokens: request.maxTokens,
|
|
180
|
-
extraBody: config.extraBody,
|
|
181
|
-
reasoning: request.reasoning ?? config.reasoning,
|
|
182
|
-
streaming: Boolean(request.stream?.onReasoningDelta),
|
|
183
|
-
});
|
|
184
|
-
const startedAt = performance.now();
|
|
185
|
-
const { value: consumed, attempts } = await fetchWithRetry(
|
|
186
|
-
config.apiUrl,
|
|
187
|
-
{
|
|
188
|
-
method: 'POST',
|
|
189
|
-
headers: {
|
|
190
|
-
'Content-Type': 'application/json',
|
|
191
|
-
...(config.apiKey ? { Authorization: `Bearer ${config.apiKey}` } : {}),
|
|
192
|
-
...adapter.headers,
|
|
193
|
-
},
|
|
194
|
-
body: JSON.stringify(payload),
|
|
195
|
-
},
|
|
196
|
-
timeout,
|
|
197
|
-
config.retry,
|
|
198
|
-
async (response) => {
|
|
199
|
-
const contentType = response.headers.get('content-type') || '';
|
|
200
|
-
if (payload.stream && contentType.includes('text/event-stream')) {
|
|
201
|
-
return {
|
|
202
|
-
data: await consumeEventStream(response, request.stream.onReasoningDelta),
|
|
203
|
-
eventStream: true,
|
|
204
|
-
};
|
|
205
|
-
}
|
|
206
|
-
try {
|
|
207
|
-
return { data: await response.json(), eventStream: false };
|
|
208
|
-
} catch (err) {
|
|
209
|
-
if (err instanceof SyntaxError) {
|
|
210
|
-
throw fail(ERROR_CATEGORIES.RESPONSE_FORMAT, 'Provider returned invalid JSON.', {
|
|
211
|
-
cause: err,
|
|
212
|
-
});
|
|
213
|
-
}
|
|
214
|
-
throw err;
|
|
215
|
-
}
|
|
216
|
-
},
|
|
217
|
-
);
|
|
218
|
-
|
|
219
|
-
const { data, eventStream } = consumed;
|
|
220
|
-
const normalized = await adapter.normalizeResponse(data);
|
|
221
|
-
if (request.stream?.onReasoningDelta && !eventStream && normalized.reasoning) {
|
|
222
|
-
request.stream.onReasoningDelta(normalized.reasoning);
|
|
223
|
-
}
|
|
224
|
-
return {
|
|
225
|
-
...normalized,
|
|
226
|
-
capabilities: adapter.capabilities,
|
|
227
|
-
attempts,
|
|
228
|
-
latencyMs: performance.now() - startedAt,
|
|
229
|
-
};
|
|
230
|
-
}
|
|
231
|
-
|
|
232
|
-
function streamContent(value) {
|
|
233
|
-
if (typeof value === 'string') return value;
|
|
234
|
-
if (Array.isArray(value)) {
|
|
235
|
-
return value
|
|
236
|
-
.map((part) => part?.text ?? part?.content ?? '')
|
|
237
|
-
.filter(Boolean)
|
|
238
|
-
.join('');
|
|
239
|
-
}
|
|
240
|
-
return value?.text ?? '';
|
|
241
|
-
}
|
|
242
|
-
|
|
243
|
-
// Consume OpenAI-compatible SSE (`data: {...}` / `data: [DONE]`) while
|
|
244
|
-
// assembling a normal Chat Completions-shaped response for the existing
|
|
245
|
-
// parsing and retry pipeline. Reasoning fields differ by provider, so every
|
|
246
|
-
// delta goes through the same normalization used for non-stream responses.
|
|
247
|
-
async function consumeEventStream(response, onReasoningDelta) {
|
|
248
|
-
if (!response.body) throw new Error('Streaming response did not include a body.');
|
|
249
|
-
|
|
250
|
-
const reader = response.body.getReader();
|
|
251
|
-
const decoder = new TextDecoder();
|
|
252
|
-
let buffer = '';
|
|
253
|
-
let dataLines = [];
|
|
254
|
-
let content = '';
|
|
255
|
-
let reasoning = '';
|
|
256
|
-
let usage = null;
|
|
257
|
-
let model = null;
|
|
258
|
-
let finishReason = null;
|
|
259
|
-
let completed = false;
|
|
260
|
-
|
|
261
|
-
const consumeData = (raw) => {
|
|
262
|
-
const payloadText = raw.trim();
|
|
263
|
-
if (!payloadText) return;
|
|
264
|
-
if (payloadText === '[DONE]') {
|
|
265
|
-
completed = true;
|
|
266
|
-
return;
|
|
267
|
-
}
|
|
268
|
-
|
|
269
|
-
let event;
|
|
270
|
-
try {
|
|
271
|
-
event = JSON.parse(payloadText);
|
|
272
|
-
} catch {
|
|
273
|
-
throw new Error(`Invalid JSON in streaming response: ${payloadText.slice(0, 200)}`);
|
|
274
|
-
}
|
|
275
|
-
if (event.error) {
|
|
276
|
-
const message = event.error.message || JSON.stringify(event.error);
|
|
277
|
-
throw new Error(`Streaming API error: ${message}`);
|
|
278
|
-
}
|
|
279
|
-
|
|
280
|
-
model ||= event.model || null;
|
|
281
|
-
if (event.usage) usage = event.usage;
|
|
282
|
-
const finishedChoice = event?.choices?.find((choice) => choice?.finish_reason != null);
|
|
283
|
-
if (finishedChoice) {
|
|
284
|
-
completed = true;
|
|
285
|
-
finishReason ||= finishedChoice.finish_reason;
|
|
286
|
-
}
|
|
287
|
-
const delta = event?.choices?.[0]?.delta ?? event?.choices?.[0]?.message;
|
|
288
|
-
if (!delta) return;
|
|
289
|
-
|
|
290
|
-
content += streamContent(delta.content);
|
|
291
|
-
const reasoningDelta = extractReasoning(delta);
|
|
292
|
-
if (reasoningDelta) {
|
|
293
|
-
reasoning += reasoningDelta;
|
|
294
|
-
onReasoningDelta(reasoningDelta);
|
|
295
|
-
}
|
|
296
|
-
};
|
|
297
|
-
|
|
298
|
-
const consumeLine = (line) => {
|
|
299
|
-
if (line === '') {
|
|
300
|
-
if (dataLines.length) consumeData(dataLines.join('\n'));
|
|
301
|
-
dataLines = [];
|
|
302
|
-
return;
|
|
303
|
-
}
|
|
304
|
-
if (line.startsWith('data:')) dataLines.push(line.slice(5).trimStart());
|
|
305
|
-
};
|
|
306
|
-
|
|
307
|
-
while (true) {
|
|
308
|
-
const { value, done } = await reader.read();
|
|
309
|
-
if (done) break;
|
|
310
|
-
buffer += decoder.decode(value, { stream: true });
|
|
311
|
-
const lines = buffer.split(/\r?\n/);
|
|
312
|
-
buffer = lines.pop() || '';
|
|
313
|
-
for (const line of lines) consumeLine(line);
|
|
314
|
-
}
|
|
315
|
-
|
|
316
|
-
buffer += decoder.decode();
|
|
317
|
-
if (buffer) consumeLine(buffer);
|
|
318
|
-
if (dataLines.length) consumeData(dataLines.join('\n'));
|
|
319
|
-
|
|
320
|
-
if (!completed) {
|
|
321
|
-
throw new Error(
|
|
322
|
-
'Streaming response ended before the provider sent [DONE] or a finish_reason. ' +
|
|
323
|
-
'The partial response was discarded; retry the request.',
|
|
324
|
-
);
|
|
325
|
-
}
|
|
326
|
-
|
|
327
|
-
const message = { content: content || null };
|
|
328
|
-
if (reasoning) message.reasoning_content = reasoning;
|
|
329
|
-
return { model, choices: [{ message, finish_reason: finishReason }], usage };
|
|
330
|
-
}
|
|
331
|
-
|
|
332
43
|
// Minimal "ping" request to verify the endpoint, API key, and model are all
|
|
333
44
|
// reachable. Throws on HTTP errors (same as callAPI); returns latency, the
|
|
334
45
|
// echoed model id, and a preview of the model's reply. Uses the same request
|
|
@@ -368,7 +79,10 @@ const MAX_REASONING_CHARS = 8000;
|
|
|
368
79
|
// normal `stop` (or a provider omitting finish_reason) remains untouched.
|
|
369
80
|
function hitTokenLimit(data) {
|
|
370
81
|
const reason = data?.finishReason;
|
|
371
|
-
return
|
|
82
|
+
return (
|
|
83
|
+
typeof reason === 'string' &&
|
|
84
|
+
/^(?:length|max_tokens|max_output_tokens|token_limit)$/i.test(reason)
|
|
85
|
+
);
|
|
372
86
|
}
|
|
373
87
|
|
|
374
88
|
// A formatting follow-up does not need to repeat reasoning that has already
|
|
@@ -397,30 +111,6 @@ function regeneratePrompt(previousMessage, policy) {
|
|
|
397
111
|
].join('\n');
|
|
398
112
|
}
|
|
399
113
|
|
|
400
|
-
// Normalize reasoning from the vendor-specific fields that can carry it:
|
|
401
|
-
// OpenAI-style `reasoning_content` (DeepSeek), OpenRouter-style `reasoning`,
|
|
402
|
-
// MiniMax/OpenRouter-style `reasoning_details` ([{ type: 'thinking', text }],
|
|
403
|
-
// possibly multiple segments with interleaved thinking).
|
|
404
|
-
function reasoningText(value) {
|
|
405
|
-
if (typeof value === 'string') return value;
|
|
406
|
-
if (Array.isArray(value)) {
|
|
407
|
-
return value.map(reasoningText).filter(Boolean).join('\n');
|
|
408
|
-
}
|
|
409
|
-
if (value && typeof value === 'object') {
|
|
410
|
-
return reasoningText(value.text ?? value.summary ?? value.content);
|
|
411
|
-
}
|
|
412
|
-
return '';
|
|
413
|
-
}
|
|
414
|
-
|
|
415
|
-
function extractReasoning(msg0) {
|
|
416
|
-
return (
|
|
417
|
-
reasoningText(msg0?.reasoning_content) ||
|
|
418
|
-
reasoningText(msg0?.reasoning) ||
|
|
419
|
-
reasoningText(msg0?.reasoning_details) ||
|
|
420
|
-
null
|
|
421
|
-
);
|
|
422
|
-
}
|
|
423
|
-
|
|
424
114
|
// Last-ditch extraction from raw reasoning text: prefer the first line that
|
|
425
115
|
// carries a conventional-commit prefix, else fall back to the last non-empty
|
|
426
116
|
// line.
|