bunnyquery 1.8.6 → 1.8.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bunnyquery.css +95 -5
- package/bunnyquery.js +1400 -101
- package/dist/engine.cjs +1018 -78
- package/dist/engine.cjs.map +1 -1
- package/dist/engine.d.mts +506 -163
- package/dist/engine.d.ts +506 -163
- package/dist/engine.mjs +1002 -79
- package/dist/engine.mjs.map +1 -1
- package/package.json +1 -1
- package/src/engine/budget.ts +207 -68
- package/src/engine/config.ts +48 -0
- package/src/engine/errors.ts +38 -0
- package/src/engine/history.ts +546 -6
- package/src/engine/host.ts +7 -0
- package/src/engine/index.ts +15 -1
- package/src/engine/indexing_groups.ts +248 -3
- package/src/engine/office.ts +24 -1
- package/src/engine/prompts/chat_system_prompt.ts +1 -1
- package/src/engine/requests.ts +164 -11
- package/src/engine/session.ts +544 -35
- package/src/widget.css +35 -1
- package/styles/chat.css +60 -4
package/package.json
CHANGED
package/src/engine/budget.ts
CHANGED
|
@@ -7,23 +7,81 @@
|
|
|
7
7
|
import { sanitizeAttachmentLinksForHistory } from './links';
|
|
8
8
|
|
|
9
9
|
export var CONTEXT_WINDOW_DEFAULT: Record<string, number> = { claude: 200000, openai: 128000 };
|
|
10
|
-
// Exact model ids first, then family keys (see
|
|
11
|
-
// falls back to its family rather than to the platform default). Claude
|
|
12
|
-
// come from Anthropic's published model table; `max_input_tokens` from
|
|
10
|
+
// Exact model ids first, then family keys (see getModelContextWindow: a suffixed
|
|
11
|
+
// id falls back to its family rather than to the platform default). Claude
|
|
12
|
+
// figures come from Anthropic's published model table; `max_input_tokens` from
|
|
13
13
|
// /v1/models overrides all of this at runtime when a listing has been seen.
|
|
14
|
-
// '
|
|
15
|
-
//
|
|
16
|
-
//
|
|
17
|
-
//
|
|
14
|
+
// OpenAI's /v1/models carries NO context field at all, so every gpt entry here
|
|
15
|
+
// is hand-maintained and is the only source for those models.
|
|
16
|
+
// These are TOTAL windows (input + output), which is what the published tables
|
|
17
|
+
// report and what getInputTokenBudget assumes when it subtracts the output
|
|
18
|
+
// reserve. The two readings reconcile: gpt-5.6-luna is 1,050,000 total with a
|
|
19
|
+
// 128,000 output cap, i.e. the ~922,000 of usable input quoted for it elsewhere.
|
|
18
20
|
export var CONTEXT_WINDOW_BY_MODEL: Record<string, number> = {
|
|
19
|
-
// exact ids
|
|
20
|
-
'claude-
|
|
21
|
-
'claude-
|
|
22
|
-
'claude-
|
|
21
|
+
// claude, exact ids
|
|
22
|
+
'claude-fable-5': 1000000, 'claude-opus-5': 1000000,
|
|
23
|
+
'claude-opus-4-8': 1000000, 'claude-opus-4-7': 1000000,
|
|
24
|
+
'claude-opus-4-6': 1000000, 'claude-opus-4-5': 200000,
|
|
25
|
+
'claude-sonnet-5': 1000000, 'claude-sonnet-4-6': 1000000,
|
|
26
|
+
'claude-sonnet-4-5': 1000000, 'claude-sonnet-4': 200000,
|
|
27
|
+
'claude-haiku-4-5': 200000, 'claude-3-5-sonnet': 200000,
|
|
28
|
+
// openai, exact ids
|
|
29
|
+
'gpt-5.6-sol': 1050000, 'gpt-5.6-terra': 1050000, 'gpt-5.6-luna': 1050000,
|
|
30
|
+
'gpt-5.5': 1000000, 'gpt-5.4': 1050000,
|
|
31
|
+
'gpt-5.4-mini': 400000, 'gpt-5.4-nano': 400000,
|
|
32
|
+
'gpt-4.1': 1040000, 'gpt-4o': 128000, 'o1': 200000, 'o1-pro': 200000,
|
|
23
33
|
// family keys
|
|
24
|
-
'claude-
|
|
25
|
-
'gpt-5.6':
|
|
34
|
+
'claude-fable': 1000000, 'claude-opus': 1000000, 'claude-sonnet': 1000000,
|
|
35
|
+
'claude-haiku': 200000, 'gpt-5.6': 1050000, 'gpt-5': 128000,
|
|
26
36
|
};
|
|
37
|
+
// Two rows above are load-bearing rather than redundant, both because the family
|
|
38
|
+
// walk drops trailing segments:
|
|
39
|
+
// 'claude-opus-4-5' — a dated id like 'claude-opus-4-5-20251101' would
|
|
40
|
+
// otherwise walk past it to the 'claude-opus' family and resolve 1000000 for
|
|
41
|
+
// a model whose real window is 200000.
|
|
42
|
+
// 'gpt-5.4-mini' and 'gpt-5.4-nano' — would otherwise walk to 'gpt-5.4' and
|
|
43
|
+
// resolve 1050000 instead of their actual 400000. nano is the one that
|
|
44
|
+
// matters most in practice: the indexing path selects it by name.
|
|
45
|
+
// The bare 'gpt-5' family stays deliberately low: it is the catch-all for gpt-5
|
|
46
|
+
// variants not listed here, and a mini-class variant is the likelier unknown.
|
|
47
|
+
//
|
|
48
|
+
// One asymmetry these rows expose: a total minus our output reserve can exceed a
|
|
49
|
+
// model's separately-published INPUT ceiling (gpt-5.4-nano is 400000 total but
|
|
50
|
+
// caps input at 272000, where 400000 - 25000 - 4000 reads as 371000). Harmless
|
|
51
|
+
// today because INPUT_CAP_RATIO binds far below either figure (59360 for nano),
|
|
52
|
+
// but it is why the ratio is not something to remove.
|
|
53
|
+
|
|
54
|
+
// Provider hard ceilings on output tokens per request. Only used to clamp what
|
|
55
|
+
// we ask for (see getMaxOutputTokens) — we never request more than
|
|
56
|
+
// MAX_OUTPUT_TOKENS anyway, so this matters exactly where a model's cap is
|
|
57
|
+
// BELOW that: gpt-4o at 4000 and legacy 3.5 Sonnet at 8000 would reject the
|
|
58
|
+
// flat 25000 the request builder used to send unconditionally.
|
|
59
|
+
export var MAX_OUTPUT_BY_MODEL: Record<string, number> = {
|
|
60
|
+
// claude
|
|
61
|
+
'claude-fable-5': 128000, 'claude-opus-5': 128000,
|
|
62
|
+
'claude-opus-4-8': 128000, 'claude-sonnet-5': 128000,
|
|
63
|
+
'claude-sonnet-4-6': 64000, 'claude-haiku-4-5': 64000,
|
|
64
|
+
'claude-3-5-sonnet': 8000,
|
|
65
|
+
// openai
|
|
66
|
+
'gpt-5.6-sol': 128000, 'gpt-5.6-terra': 128000, 'gpt-5.6-luna': 128000,
|
|
67
|
+
'gpt-5.5': 128000, 'gpt-5.4': 128000,
|
|
68
|
+
'gpt-5.4-mini': 128000, 'gpt-5.4-nano': 128000,
|
|
69
|
+
'gpt-4.1': 16000, 'gpt-4o': 4000, 'o1': 100000, 'o1-pro': 100000,
|
|
70
|
+
// family keys
|
|
71
|
+
'claude-fable': 128000, 'claude-opus': 128000, 'claude-sonnet': 64000,
|
|
72
|
+
'claude-haiku': 64000, 'gpt-5.6': 128000, 'gpt-5': 128000,
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
// The window a project runs at when nobody has touched the setting, clamped per
|
|
76
|
+
// model by getContextWindow. Deliberately BELOW every frontier ceiling it can
|
|
77
|
+
// resolve against (1,000,000 on the Claude 5 line, 1,050,000 on gpt-5.6) because
|
|
78
|
+
// the client-side budget covers only the FIRST request: after that the model
|
|
79
|
+
// runs a server-side tool loop whose web_fetch and MCP results accumulate in the
|
|
80
|
+
// same conversation and are not counted here. The gap (120k and 170k
|
|
81
|
+
// respectively) is that loop's room. It matters because no compaction beta is
|
|
82
|
+
// enabled, so overrunning the real window is a hard error, not a graceful
|
|
83
|
+
// degrade.
|
|
84
|
+
export var DEFAULT_CONTEXT_WINDOW = 880000;
|
|
27
85
|
|
|
28
86
|
// Context windows reported by a provider's own models listing, keyed by model id.
|
|
29
87
|
// Anthropic's GET /v1/models returns `max_input_tokens` per model, which is the
|
|
@@ -31,21 +89,35 @@ export var CONTEXT_WINDOW_BY_MODEL: Record<string, number> = {
|
|
|
31
89
|
// OpenAI models resolve from the table above. Populated by
|
|
32
90
|
// registerModelContextWindows() when a client fetches its model list.
|
|
33
91
|
var apiReportedContextWindows: Record<string, number> = {};
|
|
92
|
+
var apiReportedMaxOutput: Record<string, number> = {};
|
|
34
93
|
|
|
35
94
|
/**
|
|
36
|
-
* Record context windows from a provider models listing.
|
|
37
|
-
*
|
|
38
|
-
* so passing an OpenAI listing is a no-op rather than an error.
|
|
95
|
+
* Record context windows and output caps from a provider models listing. Reads
|
|
96
|
+
* `max_input_tokens` and `max_tokens` (Anthropic); items without them are
|
|
97
|
+
* skipped, so passing an OpenAI listing is a no-op rather than an error.
|
|
98
|
+
*
|
|
99
|
+
* Note the asymmetry against the static table: Anthropic reports
|
|
100
|
+
* `max_input_tokens` (input only) where CONTEXT_WINDOW_BY_MODEL holds totals, so
|
|
101
|
+
* a registered Claude window is treated as a total and loses its output cap
|
|
102
|
+
* worth of budget. That is deliberate — under-spending the window is safe, and
|
|
103
|
+
* with no compaction beta enabled overrunning it is a hard error.
|
|
39
104
|
*/
|
|
40
|
-
export function registerModelContextWindows(
|
|
105
|
+
export function registerModelContextWindows(
|
|
106
|
+
models: Array<{ id?: string; max_input_tokens?: number; max_tokens?: number }> | null | undefined,
|
|
107
|
+
): void {
|
|
41
108
|
if (!Array.isArray(models)) return;
|
|
42
109
|
for (var i = 0; i < models.length; i++) {
|
|
43
110
|
var m = models[i];
|
|
44
111
|
var id = (m && m.id ? String(m.id) : '').trim().toLowerCase();
|
|
112
|
+
if (!id) continue;
|
|
45
113
|
var reported = m ? Number(m.max_input_tokens) : NaN;
|
|
46
|
-
if (
|
|
114
|
+
if (Number.isFinite(reported) && reported > 0) {
|
|
47
115
|
apiReportedContextWindows[id] = Math.floor(reported);
|
|
48
116
|
}
|
|
117
|
+
var out = m ? Number(m.max_tokens) : NaN;
|
|
118
|
+
if (Number.isFinite(out) && out > 0) {
|
|
119
|
+
apiReportedMaxOutput[id] = Math.floor(out);
|
|
120
|
+
}
|
|
49
121
|
}
|
|
50
122
|
}
|
|
51
123
|
|
|
@@ -64,18 +136,37 @@ export function getProjectContextWindow(projectId: string): number | null {
|
|
|
64
136
|
var key = (projectId || '').trim();
|
|
65
137
|
return key && projectContextWindows[key] ? projectContextWindows[key] : null;
|
|
66
138
|
}
|
|
67
|
-
|
|
139
|
+
// `max_tokens` sent on every chat request. Exported so the reserve below and the
|
|
140
|
+
// actual request cannot drift: they were 22000 and 25000 respectively, which
|
|
141
|
+
// under-reserved by 3k on a window spent to the last token.
|
|
142
|
+
export var MAX_OUTPUT_TOKENS = 25000;
|
|
143
|
+
export var OUTPUT_TOKEN_RESERVE = MAX_OUTPUT_TOKENS;
|
|
68
144
|
export var TOOL_AND_RESPONSE_BUFFER = 4000;
|
|
69
145
|
export var MIN_INPUT_TOKEN_BUDGET = 8000;
|
|
70
|
-
|
|
146
|
+
// Floor under INPUT_CAP_RATIO. Anthropic's default tier enforces 30,000 input
|
|
147
|
+
// tokens per MINUTE on Opus, which is where the number comes from; it is applied
|
|
148
|
+
// on OpenAI too so a small-window model keeps a usable attachment budget instead
|
|
149
|
+
// of collapsing to the ratio.
|
|
150
|
+
export var MIN_PER_REQUEST_INPUT_CAP = 28000;
|
|
151
|
+
/** @deprecated renamed to {@link MIN_PER_REQUEST_INPUT_CAP} (no longer Claude-only). */
|
|
152
|
+
export var CLAUDE_PER_REQUEST_INPUT_CAP = MIN_PER_REQUEST_INPUT_CAP;
|
|
71
153
|
export var MAX_HISTORY_MESSAGES = 20;
|
|
72
154
|
export var HISTORY_TOKEN_BUDGET = 8000;
|
|
73
155
|
// Ratios that scale the two ceilings above off the resolved context window.
|
|
74
|
-
//
|
|
75
|
-
// claude 200000 -> cap 27840
|
|
76
|
-
//
|
|
77
|
-
//
|
|
78
|
-
|
|
156
|
+
// Originally calibrated to reproduce the previous fixed values at the old
|
|
157
|
+
// default windows (claude 200000 -> cap 27840, openai 128000 -> history 8160).
|
|
158
|
+
// Both are floored at the old constants, so they can only ever raise a budget,
|
|
159
|
+
// never lower one, which is what keeps a small-window model (claude-opus-4-5 at
|
|
160
|
+
// 200000) behaving as it always did.
|
|
161
|
+
//
|
|
162
|
+
// INPUT_CAP_RATIO now applies to BOTH platforms. It was Claude-only because it
|
|
163
|
+
// encoded a Claude rate limit, but at a 880000 default it does a second job that
|
|
164
|
+
// matters just as much on OpenAI: it is the headroom. Uncapped, gpt-5.6-luna
|
|
165
|
+
// resolved an input budget of 851000 against a 922000 ceiling, leaving 71000 for
|
|
166
|
+
// the whole server-side tool loop, and ONE web_fetch result can be 200000.
|
|
167
|
+
export var INPUT_CAP_RATIO = 0.16;
|
|
168
|
+
/** @deprecated renamed to {@link INPUT_CAP_RATIO} (no longer Claude-only). */
|
|
169
|
+
export var CLAUDE_INPUT_CAP_RATIO = INPUT_CAP_RATIO;
|
|
79
170
|
export var HISTORY_BUDGET_RATIO = 0.08;
|
|
80
171
|
|
|
81
172
|
export function estimateTextTokens(text: string): number {
|
|
@@ -87,34 +178,91 @@ export function estimateMessageTokens(msg: { role: string; content: string }): n
|
|
|
87
178
|
}
|
|
88
179
|
|
|
89
180
|
/**
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
181
|
+
* The model's own HARD ceiling: the largest window it can be asked for at all.
|
|
182
|
+
* Resolved most specific source first:
|
|
183
|
+
* 1. the provider's own models listing (Anthropic `max_input_tokens`)
|
|
184
|
+
* 2. an exact entry in CONTEXT_WINDOW_BY_MODEL
|
|
185
|
+
* 3. a family entry, by dropping trailing '-' segments off the id
|
|
186
|
+
* 4. the platform default
|
|
96
187
|
*
|
|
97
|
-
* Step
|
|
188
|
+
* Step 3 is why a new or suffixed id no longer drops straight to the platform
|
|
98
189
|
* default: 'gpt-5.6-luna' resolves via 'gpt-5.6', and a dated Claude snapshot
|
|
99
190
|
* such as 'claude-opus-4-7-20260101' resolves via 'claude-opus-4-7'. The walk
|
|
100
191
|
* stops at the first hit, so a more specific entry always wins over its family.
|
|
192
|
+
*
|
|
193
|
+
* This is what the settings UI offers presets against; it is NOT what a request
|
|
194
|
+
* is budgeted at. For that see getContextWindow.
|
|
195
|
+
*/
|
|
196
|
+
/**
|
|
197
|
+
* Shared id resolution for both per-model tables: provider listing, then exact
|
|
198
|
+
* table entry, then family entries by dropping trailing '-' segments. Returns 0
|
|
199
|
+
* when nothing matches so callers can apply their own default.
|
|
200
|
+
*/
|
|
201
|
+
function resolveByModelId(
|
|
202
|
+
apiTable: Record<string, number>,
|
|
203
|
+
staticTable: Record<string, number>,
|
|
204
|
+
model?: string,
|
|
205
|
+
): number {
|
|
206
|
+
var normalized = (model || '').trim().toLowerCase();
|
|
207
|
+
if (!normalized) return 0;
|
|
208
|
+
if (apiTable[normalized]) return apiTable[normalized];
|
|
209
|
+
if (staticTable[normalized]) return staticTable[normalized];
|
|
210
|
+
var parts = normalized.split('-');
|
|
211
|
+
for (var end = parts.length - 1; end > 0; end--) {
|
|
212
|
+
var family = parts.slice(0, end).join('-');
|
|
213
|
+
if (staticTable[family]) return staticTable[family];
|
|
214
|
+
}
|
|
215
|
+
return 0;
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
export function getModelContextWindow(platform: string, model?: string): number {
|
|
219
|
+
return resolveByModelId(apiReportedContextWindows, CONTEXT_WINDOW_BY_MODEL, model)
|
|
220
|
+
|| CONTEXT_WINDOW_DEFAULT[platform];
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* How many output tokens to ask for. We never want more than MAX_OUTPUT_TOKENS,
|
|
225
|
+
* but a model whose own cap is lower rejects the request outright, so clamp to
|
|
226
|
+
* whichever is smaller. Models with no known cap keep MAX_OUTPUT_TOKENS.
|
|
227
|
+
*/
|
|
228
|
+
export function getMaxOutputTokens(platform: string, model?: string): number {
|
|
229
|
+
var cap = resolveByModelId(apiReportedMaxOutput, MAX_OUTPUT_BY_MODEL, model);
|
|
230
|
+
return cap ? Math.min(MAX_OUTPUT_TOKENS, cap) : MAX_OUTPUT_TOKENS;
|
|
231
|
+
}
|
|
232
|
+
|
|
233
|
+
/**
|
|
234
|
+
* The window a request is actually budgeted at: the per-project override when
|
|
235
|
+
* one is set, otherwise DEFAULT_CONTEXT_WINDOW. Both are clamped to the model's
|
|
236
|
+
* hard ceiling, because a budget above the ceiling builds a request the provider
|
|
237
|
+
* rejects, and a stored override outlives the model it was chosen under.
|
|
101
238
|
*/
|
|
102
239
|
export function getContextWindow(platform: string, model?: string, projectId?: string): number {
|
|
240
|
+
var ceiling = getModelContextWindow(platform, model);
|
|
103
241
|
var override = projectId ? getProjectContextWindow(projectId) : null;
|
|
104
|
-
|
|
242
|
+
return Math.min(override || DEFAULT_CONTEXT_WINDOW, ceiling);
|
|
243
|
+
}
|
|
105
244
|
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
245
|
+
/**
|
|
246
|
+
* Per-request input-token budget, i.e. how much of the resolved window this turn
|
|
247
|
+
* may spend on system prompt + history + the latest message. The single
|
|
248
|
+
* implementation behind both buildBoundedChatMessages and the composer's
|
|
249
|
+
* pre-send guard, which used to be separate copies that disagreed.
|
|
250
|
+
*/
|
|
251
|
+
/**
|
|
252
|
+
* The share of the resolved window left after reserving what this request may
|
|
253
|
+
* emit. The reserve is per-model rather than the flat OUTPUT_TOKEN_RESERVE
|
|
254
|
+
* because getMaxOutputTokens clamps small-cap models below it.
|
|
255
|
+
*/
|
|
256
|
+
function contextBasedBudgetFor(platform: string, model?: string, projectId?: string): number {
|
|
257
|
+
var contextWindow = getContextWindow(platform, model, projectId);
|
|
258
|
+
return Math.max(MIN_INPUT_TOKEN_BUDGET,
|
|
259
|
+
contextWindow - getMaxOutputTokens(platform, model) - TOOL_AND_RESPONSE_BUFFER);
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
export function getInputTokenBudget(platform: string, model?: string, projectId?: string): number {
|
|
263
|
+
var contextBasedBudget = contextBasedBudgetFor(platform, model, projectId);
|
|
264
|
+
return Math.min(contextBasedBudget,
|
|
265
|
+
Math.max(MIN_PER_REQUEST_INPUT_CAP, Math.round(contextBasedBudget * INPUT_CAP_RATIO)));
|
|
118
266
|
}
|
|
119
267
|
|
|
120
268
|
export function stripFileBlocksFromHistory(content: string): string {
|
|
@@ -132,33 +280,24 @@ export type BoundedChatOptions = {
|
|
|
132
280
|
};
|
|
133
281
|
|
|
134
282
|
export function buildBoundedChatMessages(options: BoundedChatOptions) {
|
|
135
|
-
var
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
//
|
|
139
|
-
//
|
|
140
|
-
//
|
|
141
|
-
//
|
|
142
|
-
//
|
|
143
|
-
//
|
|
144
|
-
|
|
145
|
-
// scale symmetrically rather than the Claude cap compounding the ratio down.
|
|
146
|
-
var scaled = !!(options.projectId && getProjectContextWindow(options.projectId));
|
|
147
|
-
var claudeInputCap = scaled
|
|
148
|
-
? Math.max(CLAUDE_PER_REQUEST_INPUT_CAP, Math.round(contextBasedBudget * CLAUDE_INPUT_CAP_RATIO))
|
|
149
|
-
: CLAUDE_PER_REQUEST_INPUT_CAP;
|
|
150
|
-
var availableInputBudget = options.platform === 'claude'
|
|
151
|
-
? Math.min(contextBasedBudget, claudeInputCap) : contextBasedBudget;
|
|
283
|
+
var contextBasedBudget = contextBasedBudgetFor(options.platform, options.model, options.projectId);
|
|
284
|
+
// Scaling used to be gated on an EXPLICIT per-project override so that a
|
|
285
|
+
// project nobody had configured kept the old fixed ceilings. That gate is
|
|
286
|
+
// gone now that DEFAULT_CONTEXT_WINDOW is itself the default: leaving it in
|
|
287
|
+
// would mean an unconfigured project resolved 880000 and then spent 28000 of
|
|
288
|
+
// it, making the new default purely cosmetic, and would keep the trap where
|
|
289
|
+
// "Default (880K)" and an explicit 880K behaved differently.
|
|
290
|
+
// Both ceilings derive from contextBasedBudget (pre-Claude-cap) so the two
|
|
291
|
+
// platforms scale symmetrically rather than the Claude cap compounding down.
|
|
292
|
+
var availableInputBudget = getInputTokenBudget(options.platform, options.model, options.projectId);
|
|
152
293
|
var systemCost = estimateTextTokens(options.systemPrompt) + 12;
|
|
153
|
-
var historyAllowance =
|
|
154
|
-
|
|
155
|
-
: HISTORY_TOKEN_BUDGET;
|
|
294
|
+
var historyAllowance = Math.max(HISTORY_TOKEN_BUDGET,
|
|
295
|
+
Math.round(contextBasedBudget * HISTORY_BUDGET_RATIO));
|
|
156
296
|
var budgetForHistory = Math.max(1000, Math.min(historyAllowance, availableInputBudget - systemCost));
|
|
157
297
|
// The message count scales alongside the token budget; otherwise 20 messages
|
|
158
298
|
// is a second ceiling that swallows the extra budget on a raised window.
|
|
159
|
-
var maxHistoryMessages =
|
|
160
|
-
|
|
161
|
-
: MAX_HISTORY_MESSAGES;
|
|
299
|
+
var maxHistoryMessages = Math.max(MAX_HISTORY_MESSAGES,
|
|
300
|
+
Math.round(MAX_HISTORY_MESSAGES * (budgetForHistory / HISTORY_TOKEN_BUDGET)));
|
|
162
301
|
var windowed = options.history.slice(-maxHistoryMessages);
|
|
163
302
|
var latestIndex = windowed.length - 1;
|
|
164
303
|
var trimmed = windowed.map(function (m, i) {
|
package/src/engine/config.ts
CHANGED
|
@@ -47,6 +47,54 @@ export interface ChatEngineConfig {
|
|
|
47
47
|
* until the worker is deployed, then flip it per environment.
|
|
48
48
|
*/
|
|
49
49
|
windowedIndexing?: boolean;
|
|
50
|
+
/**
|
|
51
|
+
* Mint the durable "indexing finished" marker record ("done::<path>",
|
|
52
|
+
* reference "src::<path>", table __INDEXING__ — same shape the backend
|
|
53
|
+
* worker writes via /internal/index-complete) for runs whose completion
|
|
54
|
+
* THIS CLIENT knows deterministically: a single-pass file's settled pass,
|
|
55
|
+
* or a client-driven chain whose reply carried the completion token.
|
|
56
|
+
* Worker-driven chains are NOT minted from the client (their completion is
|
|
57
|
+
* only ever inferred here); the worker writes their marker itself.
|
|
58
|
+
* Must be best-effort: tolerate the marker already existing and never
|
|
59
|
+
* throw. Optional so older consumers keep the pre-marker inference.
|
|
60
|
+
*/
|
|
61
|
+
mintIndexDoneMarker?: (info: { service: string; storagePath: string }) => void;
|
|
62
|
+
/**
|
|
63
|
+
* Create-or-update the per-file indexing RUN record ("run::<path>",
|
|
64
|
+
* reference "src::<path>", table __INDEXING__). The record is the durable
|
|
65
|
+
* "a run exists and this is its status" signal that lets chat rows and
|
|
66
|
+
* files-page badges paint without scanning bg history.
|
|
67
|
+
*
|
|
68
|
+
* The consumer implements upsert semantics (the records API has none):
|
|
69
|
+
* create, and on "is already taken" look the record up by unique_id and
|
|
70
|
+
* re-post with its record_id, merging `patch` over the stored data.
|
|
71
|
+
* Status precedence is the consumer's job too: 'working' must NEVER
|
|
72
|
+
* overwrite a terminal status (done/error/cancelled) — a late create from
|
|
73
|
+
* a slow enqueue must not resurrect a run another writer already closed.
|
|
74
|
+
* Must be best-effort and never throw. Optional: without it the engine
|
|
75
|
+
* behaves exactly as before (legacy scan/probe path).
|
|
76
|
+
*/
|
|
77
|
+
upsertIndexRunRecord?: (info: {
|
|
78
|
+
service: string;
|
|
79
|
+
storagePath: string;
|
|
80
|
+
patch: {
|
|
81
|
+
status: 'working' | 'done' | 'error' | 'cancelled';
|
|
82
|
+
filename?: string;
|
|
83
|
+
started?: number;
|
|
84
|
+
finished?: number;
|
|
85
|
+
error?: string;
|
|
86
|
+
queue?: string;
|
|
87
|
+
};
|
|
88
|
+
}) => void;
|
|
89
|
+
/**
|
|
90
|
+
* Single-item csr-poll point lookup (skapi.util.request('csr-poll', {id,
|
|
91
|
+
* service, owner}, {auth:true})). For a RESOLVED item the backend returns
|
|
92
|
+
* the provider response body itself; for a failed one, the resolved error.
|
|
93
|
+
* Used by ChatSession.hydrateCompactItems to fetch the real bodies of
|
|
94
|
+
* compact history stubs when the user expands an indexing row. Optional:
|
|
95
|
+
* without it, stubs keep their server-extracted heads.
|
|
96
|
+
*/
|
|
97
|
+
csrHistoryItemLookup?: (fullId: string, service: string, owner: string) => Promise<any>;
|
|
50
98
|
}
|
|
51
99
|
|
|
52
100
|
let _config: ChatEngineConfig | null = null;
|
package/src/engine/errors.ts
CHANGED
|
@@ -139,3 +139,41 @@ export function isAuthExpiredError(input: any): boolean {
|
|
|
139
139
|
hay.indexOf('unauthorized') !== -1 || hay.indexOf('not authorized') !== -1 ||
|
|
140
140
|
(hay.indexOf('invalid_request') !== -1 && hay.indexOf('token') !== -1);
|
|
141
141
|
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* True when the AI PROVIDER rejected the project's own API key.
|
|
145
|
+
*
|
|
146
|
+
* Deliberately narrow, and deliberately NOT the same question as
|
|
147
|
+
* isAuthExpiredError: that one is about OUR session/MCP bearer going stale,
|
|
148
|
+
* which the client fixes by refreshing and resending. This one means the key
|
|
149
|
+
* the project owner pasted is wrong or revoked, which only a human can fix.
|
|
150
|
+
*
|
|
151
|
+
* A bare 401 is NOT enough to conclude it (the MCP bearer expiring is also a
|
|
152
|
+
* 401), so this matches only the provider's key-specific markers:
|
|
153
|
+
* Anthropic -> `authentication_error`, "invalid x-api-key"
|
|
154
|
+
* OpenAI -> `invalid_api_key`, "Incorrect API key provided"
|
|
155
|
+
* Accepts a response object, a thrown error, or the message string those get
|
|
156
|
+
* reduced to by getErrorMessage, because the view usually only keeps the text.
|
|
157
|
+
*/
|
|
158
|
+
export function isProviderApiKeyError(input: any): boolean {
|
|
159
|
+
if (!input) return false;
|
|
160
|
+
var blobs: string[] = [];
|
|
161
|
+
var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
|
|
162
|
+
if (typeof input === 'string') push(input);
|
|
163
|
+
else {
|
|
164
|
+
push(input.message); push(input.code); push(input.type);
|
|
165
|
+
if (input.error) { push(input.error.message); push(input.error.code); push(input.error.type); }
|
|
166
|
+
if (input.body) {
|
|
167
|
+
push(input.body.message); push(input.body.type);
|
|
168
|
+
if (input.body.error) { push(input.body.error.message); push(input.body.error.code); push(input.body.error.type); }
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
var hay = blobs.join(' | ').toLowerCase();
|
|
172
|
+
if (!hay) return false;
|
|
173
|
+
return hay.indexOf('authentication_error') !== -1 ||
|
|
174
|
+
hay.indexOf('invalid_api_key') !== -1 ||
|
|
175
|
+
hay.indexOf('invalid x-api-key') !== -1 ||
|
|
176
|
+
hay.indexOf('incorrect api key') !== -1 ||
|
|
177
|
+
hay.indexOf('invalid api key') !== -1 ||
|
|
178
|
+
hay.indexOf('no api key provided') !== -1;
|
|
179
|
+
}
|