bunnyquery 1.8.6 → 1.8.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "bunnyquery",
3
- "version": "1.8.6",
3
+ "version": "1.8.8",
4
4
  "description": "Embeddable BunnyQuery AI chat widget + its framework-agnostic chat engine",
5
5
  "main": "bunnyquery.js",
6
6
  "exports": {
@@ -7,23 +7,81 @@
7
7
  import { sanitizeAttachmentLinksForHistory } from './links';
8
8
 
9
9
  export var CONTEXT_WINDOW_DEFAULT: Record<string, number> = { claude: 200000, openai: 128000 };
10
- // Exact model ids first, then family keys (see getContextWindow: a suffixed id
11
- // falls back to its family rather than to the platform default). Claude figures
12
- // come from Anthropic's published model table; `max_input_tokens` from
10
+ // Exact model ids first, then family keys (see getModelContextWindow: a suffixed
11
+ // id falls back to its family rather than to the platform default). Claude
12
+ // figures come from Anthropic's published model table; `max_input_tokens` from
13
13
  // /v1/models overrides all of this at runtime when a listing has been seen.
14
- // 'claude-opus-4-7' read 200000 here, which was wrong by 5x, and the default
15
- // Claude model 'claude-sonnet-4-6' was absent entirely so it fell through to
16
- // CONTEXT_WINDOW_DEFAULT.claude. Neither was observable, because the history
17
- // budget and the Claude per-request cap below bind long before the window does.
14
+ // OpenAI's /v1/models carries NO context field at all, so every gpt entry here
15
+ // is hand-maintained and is the only source for those models.
16
+ // These are TOTAL windows (input + output), which is what the published tables
17
+ // report and what getInputTokenBudget assumes when it subtracts the output
18
+ // reserve. The two readings reconcile: gpt-5.6-luna is 1,050,000 total with a
19
+ // 128,000 output cap, i.e. the ~922,000 of usable input quoted for it elsewhere.
18
20
  export var CONTEXT_WINDOW_BY_MODEL: Record<string, number> = {
19
- // exact ids
20
- 'claude-opus-5': 1000000, 'claude-opus-4-8': 1000000, 'claude-opus-4-7': 1000000,
21
- 'claude-sonnet-5': 1000000, 'claude-sonnet-4-6': 1000000, 'claude-sonnet-4': 200000,
22
- 'claude-haiku-4-5': 200000, 'gpt-5.4': 128000, 'gpt-5.6-luna': 128000,
21
+ // claude, exact ids
22
+ 'claude-fable-5': 1000000, 'claude-opus-5': 1000000,
23
+ 'claude-opus-4-8': 1000000, 'claude-opus-4-7': 1000000,
24
+ 'claude-opus-4-6': 1000000, 'claude-opus-4-5': 200000,
25
+ 'claude-sonnet-5': 1000000, 'claude-sonnet-4-6': 1000000,
26
+ 'claude-sonnet-4-5': 1000000, 'claude-sonnet-4': 200000,
27
+ 'claude-haiku-4-5': 200000, 'claude-3-5-sonnet': 200000,
28
+ // openai, exact ids
29
+ 'gpt-5.6-sol': 1050000, 'gpt-5.6-terra': 1050000, 'gpt-5.6-luna': 1050000,
30
+ 'gpt-5.5': 1000000, 'gpt-5.4': 1050000,
31
+ 'gpt-5.4-mini': 400000, 'gpt-5.4-nano': 400000,
32
+ 'gpt-4.1': 1040000, 'gpt-4o': 128000, 'o1': 200000, 'o1-pro': 200000,
23
33
  // family keys
24
- 'claude-opus': 1000000, 'claude-sonnet': 1000000, 'claude-haiku': 200000,
25
- 'gpt-5.6': 128000, 'gpt-5': 128000,
34
+ 'claude-fable': 1000000, 'claude-opus': 1000000, 'claude-sonnet': 1000000,
35
+ 'claude-haiku': 200000, 'gpt-5.6': 1050000, 'gpt-5': 128000,
26
36
  };
37
+ // Two rows above are load-bearing rather than redundant, both because the family
38
+ // walk drops trailing segments:
39
+ // 'claude-opus-4-5' — a dated id like 'claude-opus-4-5-20251101' would
40
+ // otherwise walk past it to the 'claude-opus' family and resolve 1000000 for
41
+ // a model whose real window is 200000.
42
+ // 'gpt-5.4-mini' and 'gpt-5.4-nano' — would otherwise walk to 'gpt-5.4' and
43
+ // resolve 1050000 instead of their actual 400000. nano is the one that
44
+ // matters most in practice: the indexing path selects it by name.
45
+ // The bare 'gpt-5' family stays deliberately low: it is the catch-all for gpt-5
46
+ // variants not listed here, and a mini-class variant is the likelier unknown.
47
+ //
48
+ // One asymmetry these rows expose: a total minus our output reserve can exceed a
49
+ // model's separately-published INPUT ceiling (gpt-5.4-nano is 400000 total but
50
+ // caps input at 272000, where 400000 - 25000 - 4000 reads as 371000). Harmless
51
+ // today because INPUT_CAP_RATIO binds far below either figure (59360 for nano),
52
+ // but it is why the ratio is not something to remove.
53
+
54
+ // Provider hard ceilings on output tokens per request. Only used to clamp what
55
+ // we ask for (see getMaxOutputTokens) — we never request more than
56
+ // MAX_OUTPUT_TOKENS anyway, so this matters exactly where a model's cap is
57
+ // BELOW that: gpt-4o at 4000 and legacy 3.5 Sonnet at 8000 would reject the
58
+ // flat 25000 the request builder used to send unconditionally.
59
+ export var MAX_OUTPUT_BY_MODEL: Record<string, number> = {
60
+ // claude
61
+ 'claude-fable-5': 128000, 'claude-opus-5': 128000,
62
+ 'claude-opus-4-8': 128000, 'claude-sonnet-5': 128000,
63
+ 'claude-sonnet-4-6': 64000, 'claude-haiku-4-5': 64000,
64
+ 'claude-3-5-sonnet': 8000,
65
+ // openai
66
+ 'gpt-5.6-sol': 128000, 'gpt-5.6-terra': 128000, 'gpt-5.6-luna': 128000,
67
+ 'gpt-5.5': 128000, 'gpt-5.4': 128000,
68
+ 'gpt-5.4-mini': 128000, 'gpt-5.4-nano': 128000,
69
+ 'gpt-4.1': 16000, 'gpt-4o': 4000, 'o1': 100000, 'o1-pro': 100000,
70
+ // family keys
71
+ 'claude-fable': 128000, 'claude-opus': 128000, 'claude-sonnet': 64000,
72
+ 'claude-haiku': 64000, 'gpt-5.6': 128000, 'gpt-5': 128000,
73
+ };
74
+
75
+ // The window a project runs at when nobody has touched the setting, clamped per
76
+ // model by getContextWindow. Deliberately BELOW every frontier ceiling it can
77
+ // resolve against (1,000,000 on the Claude 5 line, 1,050,000 on gpt-5.6) because
78
+ // the client-side budget covers only the FIRST request: after that the model
79
+ // runs a server-side tool loop whose web_fetch and MCP results accumulate in the
80
+ // same conversation and are not counted here. The gap (120k and 170k
81
+ // respectively) is that loop's room. It matters because no compaction beta is
82
+ // enabled, so overrunning the real window is a hard error, not a graceful
83
+ // degrade.
84
+ export var DEFAULT_CONTEXT_WINDOW = 880000;
27
85
 
28
86
  // Context windows reported by a provider's own models listing, keyed by model id.
29
87
  // Anthropic's GET /v1/models returns `max_input_tokens` per model, which is the
@@ -31,21 +89,35 @@ export var CONTEXT_WINDOW_BY_MODEL: Record<string, number> = {
31
89
  // OpenAI models resolve from the table above. Populated by
32
90
  // registerModelContextWindows() when a client fetches its model list.
33
91
  var apiReportedContextWindows: Record<string, number> = {};
92
+ var apiReportedMaxOutput: Record<string, number> = {};
34
93
 
35
94
  /**
36
- * Record context windows from a provider models listing. Accepts the raw list
37
- * items and reads `max_input_tokens` (Anthropic); items without it are skipped,
38
- * so passing an OpenAI listing is a no-op rather than an error.
95
+ * Record context windows and output caps from a provider models listing. Reads
96
+ * `max_input_tokens` and `max_tokens` (Anthropic); items without them are
97
+ * skipped, so passing an OpenAI listing is a no-op rather than an error.
98
+ *
99
+ * Note the asymmetry against the static table: Anthropic reports
100
+ * `max_input_tokens` (input only) where CONTEXT_WINDOW_BY_MODEL holds totals, so
101
+ * a registered Claude window is treated as a total and loses its output cap
102
+ * worth of budget. That is deliberate — under-spending the window is safe, and
103
+ * with no compaction beta enabled overrunning it is a hard error.
39
104
  */
40
- export function registerModelContextWindows(models: Array<{ id?: string; max_input_tokens?: number }> | null | undefined): void {
105
+ export function registerModelContextWindows(
106
+ models: Array<{ id?: string; max_input_tokens?: number; max_tokens?: number }> | null | undefined,
107
+ ): void {
41
108
  if (!Array.isArray(models)) return;
42
109
  for (var i = 0; i < models.length; i++) {
43
110
  var m = models[i];
44
111
  var id = (m && m.id ? String(m.id) : '').trim().toLowerCase();
112
+ if (!id) continue;
45
113
  var reported = m ? Number(m.max_input_tokens) : NaN;
46
- if (id && Number.isFinite(reported) && reported > 0) {
114
+ if (Number.isFinite(reported) && reported > 0) {
47
115
  apiReportedContextWindows[id] = Math.floor(reported);
48
116
  }
117
+ var out = m ? Number(m.max_tokens) : NaN;
118
+ if (Number.isFinite(out) && out > 0) {
119
+ apiReportedMaxOutput[id] = Math.floor(out);
120
+ }
49
121
  }
50
122
  }
51
123
 
@@ -64,18 +136,37 @@ export function getProjectContextWindow(projectId: string): number | null {
64
136
  var key = (projectId || '').trim();
65
137
  return key && projectContextWindows[key] ? projectContextWindows[key] : null;
66
138
  }
67
- export var OUTPUT_TOKEN_RESERVE = 22000;
139
+ // `max_tokens` sent on every chat request. Exported so the reserve below and the
140
+ // actual request cannot drift: they were 22000 and 25000 respectively, which
141
+ // under-reserved by 3k on a window spent to the last token.
142
+ export var MAX_OUTPUT_TOKENS = 25000;
143
+ export var OUTPUT_TOKEN_RESERVE = MAX_OUTPUT_TOKENS;
68
144
  export var TOOL_AND_RESPONSE_BUFFER = 4000;
69
145
  export var MIN_INPUT_TOKEN_BUDGET = 8000;
70
- export var CLAUDE_PER_REQUEST_INPUT_CAP = 28000;
146
+ // Floor under INPUT_CAP_RATIO. Anthropic's default tier enforces 30,000 input
147
+ // tokens per MINUTE on Opus, which is where the number comes from; it is applied
148
+ // on OpenAI too so a small-window model keeps a usable attachment budget instead
149
+ // of collapsing to the ratio.
150
+ export var MIN_PER_REQUEST_INPUT_CAP = 28000;
151
+ /** @deprecated renamed to {@link MIN_PER_REQUEST_INPUT_CAP} (no longer Claude-only). */
152
+ export var CLAUDE_PER_REQUEST_INPUT_CAP = MIN_PER_REQUEST_INPUT_CAP;
71
153
  export var MAX_HISTORY_MESSAGES = 20;
72
154
  export var HISTORY_TOKEN_BUDGET = 8000;
73
155
  // Ratios that scale the two ceilings above off the resolved context window.
74
- // Calibrated to reproduce the previous fixed values at the default windows:
75
- // claude 200000 -> cap 27840 (was 28000, so the 28000 floor holds), and
76
- // openai 128000 -> history 8160 (~the previous 8000). Both are floored at the
77
- // old constants, so these can only ever raise a budget, never lower one.
78
- export var CLAUDE_INPUT_CAP_RATIO = 0.16;
156
+ // Originally calibrated to reproduce the previous fixed values at the old
157
+ // default windows (claude 200000 -> cap 27840, openai 128000 -> history 8160).
158
+ // Both are floored at the old constants, so they can only ever raise a budget,
159
+ // never lower one, which is what keeps a small-window model (claude-opus-4-5 at
160
+ // 200000) behaving as it always did.
161
+ //
162
+ // INPUT_CAP_RATIO now applies to BOTH platforms. It was Claude-only because it
163
+ // encoded a Claude rate limit, but at a 880000 default it does a second job that
164
+ // matters just as much on OpenAI: it is the headroom. Uncapped, gpt-5.6-luna
165
+ // resolved an input budget of 851000 against a 922000 ceiling, leaving 71000 for
166
+ // the whole server-side tool loop, and ONE web_fetch result can be 200000.
167
+ export var INPUT_CAP_RATIO = 0.16;
168
+ /** @deprecated renamed to {@link INPUT_CAP_RATIO} (no longer Claude-only). */
169
+ export var CLAUDE_INPUT_CAP_RATIO = INPUT_CAP_RATIO;
79
170
  export var HISTORY_BUDGET_RATIO = 0.08;
80
171
 
81
172
  export function estimateTextTokens(text: string): number {
@@ -87,34 +178,91 @@ export function estimateMessageTokens(msg: { role: string; content: string }): n
87
178
  }
88
179
 
89
180
  /**
90
- * Resolve a model's context window, most specific source first:
91
- * 1. per-project override (project settings)
92
- * 2. the provider's own models listing (Anthropic `max_input_tokens`)
93
- * 3. an exact entry in CONTEXT_WINDOW_BY_MODEL
94
- * 4. a family entry, by dropping trailing '-' segments off the id
95
- * 5. the platform default
181
+ * The model's own HARD ceiling: the largest window it can be asked for at all.
182
+ * Resolved most specific source first:
183
+ * 1. the provider's own models listing (Anthropic `max_input_tokens`)
184
+ * 2. an exact entry in CONTEXT_WINDOW_BY_MODEL
185
+ * 3. a family entry, by dropping trailing '-' segments off the id
186
+ * 4. the platform default
96
187
  *
97
- * Step 4 is why a new or suffixed id no longer drops straight to the platform
188
+ * Step 3 is why a new or suffixed id no longer drops straight to the platform
98
189
  * default: 'gpt-5.6-luna' resolves via 'gpt-5.6', and a dated Claude snapshot
99
190
  * such as 'claude-opus-4-7-20260101' resolves via 'claude-opus-4-7'. The walk
100
191
  * stops at the first hit, so a more specific entry always wins over its family.
192
+ *
193
+ * This is what the settings UI offers presets against; it is NOT what a request
194
+ * is budgeted at. For that see getContextWindow.
195
+ */
196
+ /**
197
+ * Shared id resolution for both per-model tables: provider listing, then exact
198
+ * table entry, then family entries by dropping trailing '-' segments. Returns 0
199
+ * when nothing matches so callers can apply their own default.
200
+ */
201
+ function resolveByModelId(
202
+ apiTable: Record<string, number>,
203
+ staticTable: Record<string, number>,
204
+ model?: string,
205
+ ): number {
206
+ var normalized = (model || '').trim().toLowerCase();
207
+ if (!normalized) return 0;
208
+ if (apiTable[normalized]) return apiTable[normalized];
209
+ if (staticTable[normalized]) return staticTable[normalized];
210
+ var parts = normalized.split('-');
211
+ for (var end = parts.length - 1; end > 0; end--) {
212
+ var family = parts.slice(0, end).join('-');
213
+ if (staticTable[family]) return staticTable[family];
214
+ }
215
+ return 0;
216
+ }
217
+
218
+ export function getModelContextWindow(platform: string, model?: string): number {
219
+ return resolveByModelId(apiReportedContextWindows, CONTEXT_WINDOW_BY_MODEL, model)
220
+ || CONTEXT_WINDOW_DEFAULT[platform];
221
+ }
222
+
223
+ /**
224
+ * How many output tokens to ask for. We never want more than MAX_OUTPUT_TOKENS,
225
+ * but a model whose own cap is lower rejects the request outright, so clamp to
226
+ * whichever is smaller. Models with no known cap keep MAX_OUTPUT_TOKENS.
227
+ */
228
+ export function getMaxOutputTokens(platform: string, model?: string): number {
229
+ var cap = resolveByModelId(apiReportedMaxOutput, MAX_OUTPUT_BY_MODEL, model);
230
+ return cap ? Math.min(MAX_OUTPUT_TOKENS, cap) : MAX_OUTPUT_TOKENS;
231
+ }
232
+
233
+ /**
234
+ * The window a request is actually budgeted at: the per-project override when
235
+ * one is set, otherwise DEFAULT_CONTEXT_WINDOW. Both are clamped to the model's
236
+ * hard ceiling, because a budget above the ceiling builds a request the provider
237
+ * rejects, and a stored override outlives the model it was chosen under.
101
238
  */
102
239
  export function getContextWindow(platform: string, model?: string, projectId?: string): number {
240
+ var ceiling = getModelContextWindow(platform, model);
103
241
  var override = projectId ? getProjectContextWindow(projectId) : null;
104
- if (override) return override;
242
+ return Math.min(override || DEFAULT_CONTEXT_WINDOW, ceiling);
243
+ }
105
244
 
106
- var normalized = (model || '').trim().toLowerCase();
107
- if (normalized) {
108
- if (apiReportedContextWindows[normalized]) return apiReportedContextWindows[normalized];
109
- if (CONTEXT_WINDOW_BY_MODEL[normalized]) return CONTEXT_WINDOW_BY_MODEL[normalized];
110
-
111
- var parts = normalized.split('-');
112
- for (var end = parts.length - 1; end > 0; end--) {
113
- var family = parts.slice(0, end).join('-');
114
- if (CONTEXT_WINDOW_BY_MODEL[family]) return CONTEXT_WINDOW_BY_MODEL[family];
115
- }
116
- }
117
- return CONTEXT_WINDOW_DEFAULT[platform];
245
+ /**
246
+ * Per-request input-token budget, i.e. how much of the resolved window this turn
247
+ * may spend on system prompt + history + the latest message. The single
248
+ * implementation behind both buildBoundedChatMessages and the composer's
249
+ * pre-send guard, which used to be separate copies that disagreed.
250
+ */
251
+ /**
252
+ * The share of the resolved window left after reserving what this request may
253
+ * emit. The reserve is per-model rather than the flat OUTPUT_TOKEN_RESERVE
254
+ * because getMaxOutputTokens clamps small-cap models below it.
255
+ */
256
+ function contextBasedBudgetFor(platform: string, model?: string, projectId?: string): number {
257
+ var contextWindow = getContextWindow(platform, model, projectId);
258
+ return Math.max(MIN_INPUT_TOKEN_BUDGET,
259
+ contextWindow - getMaxOutputTokens(platform, model) - TOOL_AND_RESPONSE_BUFFER);
260
+ }
261
+
262
+ export function getInputTokenBudget(platform: string, model?: string, projectId?: string): number {
263
+ var contextBasedBudget = contextBasedBudgetFor(platform, model, projectId);
264
+ return Math.min(contextBasedBudget,
265
+ Math.max(MIN_PER_REQUEST_INPUT_CAP, Math.round(contextBasedBudget * INPUT_CAP_RATIO)));
118
266
  }
119
267
 
120
268
  export function stripFileBlocksFromHistory(content: string): string {
@@ -132,33 +280,24 @@ export type BoundedChatOptions = {
132
280
  };
133
281
 
134
282
  export function buildBoundedChatMessages(options: BoundedChatOptions) {
135
- var contextWindow = getContextWindow(options.platform, options.model, options.projectId);
136
- var contextBasedBudget = Math.max(MIN_INPUT_TOKEN_BUDGET,
137
- contextWindow - OUTPUT_TOKEN_RESERVE - TOOL_AND_RESPONSE_BUFFER);
138
- // Scaling is gated on an EXPLICIT per-project window set from project
139
- // settings. Without one, the fixed ceilings apply exactly as before, so every
140
- // existing project keeps byte-identical behavior and nobody's token spend
141
- // moves because a table value was corrected. With one, the Claude per-request
142
- // cap and the history budget scale off the window, so raising it genuinely
143
- // sends more history instead of being absorbed by a hardcoded ceiling.
144
- // Both derive from contextBasedBudget (pre-Claude-cap) so the two platforms
145
- // scale symmetrically rather than the Claude cap compounding the ratio down.
146
- var scaled = !!(options.projectId && getProjectContextWindow(options.projectId));
147
- var claudeInputCap = scaled
148
- ? Math.max(CLAUDE_PER_REQUEST_INPUT_CAP, Math.round(contextBasedBudget * CLAUDE_INPUT_CAP_RATIO))
149
- : CLAUDE_PER_REQUEST_INPUT_CAP;
150
- var availableInputBudget = options.platform === 'claude'
151
- ? Math.min(contextBasedBudget, claudeInputCap) : contextBasedBudget;
283
+ var contextBasedBudget = contextBasedBudgetFor(options.platform, options.model, options.projectId);
284
+ // Scaling used to be gated on an EXPLICIT per-project override so that a
285
+ // project nobody had configured kept the old fixed ceilings. That gate is
286
+ // gone now that DEFAULT_CONTEXT_WINDOW is itself the default: leaving it in
287
+ // would mean an unconfigured project resolved 880000 and then spent 28000 of
288
+ // it, making the new default purely cosmetic, and would keep the trap where
289
+ // "Default (880K)" and an explicit 880K behaved differently.
290
+ // Both ceilings derive from contextBasedBudget (pre-Claude-cap) so the two
291
+ // platforms scale symmetrically rather than the Claude cap compounding down.
292
+ var availableInputBudget = getInputTokenBudget(options.platform, options.model, options.projectId);
152
293
  var systemCost = estimateTextTokens(options.systemPrompt) + 12;
153
- var historyAllowance = scaled
154
- ? Math.max(HISTORY_TOKEN_BUDGET, Math.round(contextBasedBudget * HISTORY_BUDGET_RATIO))
155
- : HISTORY_TOKEN_BUDGET;
294
+ var historyAllowance = Math.max(HISTORY_TOKEN_BUDGET,
295
+ Math.round(contextBasedBudget * HISTORY_BUDGET_RATIO));
156
296
  var budgetForHistory = Math.max(1000, Math.min(historyAllowance, availableInputBudget - systemCost));
157
297
  // The message count scales alongside the token budget; otherwise 20 messages
158
298
  // is a second ceiling that swallows the extra budget on a raised window.
159
- var maxHistoryMessages = scaled
160
- ? Math.max(MAX_HISTORY_MESSAGES, Math.round(MAX_HISTORY_MESSAGES * (budgetForHistory / HISTORY_TOKEN_BUDGET)))
161
- : MAX_HISTORY_MESSAGES;
299
+ var maxHistoryMessages = Math.max(MAX_HISTORY_MESSAGES,
300
+ Math.round(MAX_HISTORY_MESSAGES * (budgetForHistory / HISTORY_TOKEN_BUDGET)));
162
301
  var windowed = options.history.slice(-maxHistoryMessages);
163
302
  var latestIndex = windowed.length - 1;
164
303
  var trimmed = windowed.map(function (m, i) {
@@ -47,6 +47,54 @@ export interface ChatEngineConfig {
47
47
  * until the worker is deployed, then flip it per environment.
48
48
  */
49
49
  windowedIndexing?: boolean;
50
+ /**
51
+ * Mint the durable "indexing finished" marker record ("done::<path>",
52
+ * reference "src::<path>", table __INDEXING__ — same shape the backend
53
+ * worker writes via /internal/index-complete) for runs whose completion
54
+ * THIS CLIENT knows deterministically: a single-pass file's settled pass,
55
+ * or a client-driven chain whose reply carried the completion token.
56
+ * Worker-driven chains are NOT minted from the client (their completion is
57
+ * only ever inferred here); the worker writes their marker itself.
58
+ * Must be best-effort: tolerate the marker already existing and never
59
+ * throw. Optional so older consumers keep the pre-marker inference.
60
+ */
61
+ mintIndexDoneMarker?: (info: { service: string; storagePath: string }) => void;
62
+ /**
63
+ * Create-or-update the per-file indexing RUN record ("run::<path>",
64
+ * reference "src::<path>", table __INDEXING__). The record is the durable
65
+ * "a run exists and this is its status" signal that lets chat rows and
66
+ * files-page badges paint without scanning bg history.
67
+ *
68
+ * The consumer implements upsert semantics (the records API has none):
69
+ * create, and on "is already taken" look the record up by unique_id and
70
+ * re-post with its record_id, merging `patch` over the stored data.
71
+ * Status precedence is the consumer's job too: 'working' must NEVER
72
+ * overwrite a terminal status (done/error/cancelled) — a late create from
73
+ * a slow enqueue must not resurrect a run another writer already closed.
74
+ * Must be best-effort and never throw. Optional: without it the engine
75
+ * behaves exactly as before (legacy scan/probe path).
76
+ */
77
+ upsertIndexRunRecord?: (info: {
78
+ service: string;
79
+ storagePath: string;
80
+ patch: {
81
+ status: 'working' | 'done' | 'error' | 'cancelled';
82
+ filename?: string;
83
+ started?: number;
84
+ finished?: number;
85
+ error?: string;
86
+ queue?: string;
87
+ };
88
+ }) => void;
89
+ /**
90
+ * Single-item csr-poll point lookup (skapi.util.request('csr-poll', {id,
91
+ * service, owner}, {auth:true})). For a RESOLVED item the backend returns
92
+ * the provider response body itself; for a failed one, the resolved error.
93
+ * Used by ChatSession.hydrateCompactItems to fetch the real bodies of
94
+ * compact history stubs when the user expands an indexing row. Optional:
95
+ * without it, stubs keep their server-extracted heads.
96
+ */
97
+ csrHistoryItemLookup?: (fullId: string, service: string, owner: string) => Promise<any>;
50
98
  }
51
99
 
52
100
  let _config: ChatEngineConfig | null = null;
@@ -139,3 +139,41 @@ export function isAuthExpiredError(input: any): boolean {
139
139
  hay.indexOf('unauthorized') !== -1 || hay.indexOf('not authorized') !== -1 ||
140
140
  (hay.indexOf('invalid_request') !== -1 && hay.indexOf('token') !== -1);
141
141
  }
142
+
143
+ /**
144
+ * True when the AI PROVIDER rejected the project's own API key.
145
+ *
146
+ * Deliberately narrow, and deliberately NOT the same question as
147
+ * isAuthExpiredError: that one is about OUR session/MCP bearer going stale,
148
+ * which the client fixes by refreshing and resending. This one means the key
149
+ * the project owner pasted is wrong or revoked, which only a human can fix.
150
+ *
151
+ * A bare 401 is NOT enough to conclude it (the MCP bearer expiring is also a
152
+ * 401), so this matches only the provider's key-specific markers:
153
+ * Anthropic -> `authentication_error`, "invalid x-api-key"
154
+ * OpenAI -> `invalid_api_key`, "Incorrect API key provided"
155
+ * Accepts a response object, a thrown error, or the message string those get
156
+ * reduced to by getErrorMessage, because the view usually only keeps the text.
157
+ */
158
+ export function isProviderApiKeyError(input: any): boolean {
159
+ if (!input) return false;
160
+ var blobs: string[] = [];
161
+ var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
162
+ if (typeof input === 'string') push(input);
163
+ else {
164
+ push(input.message); push(input.code); push(input.type);
165
+ if (input.error) { push(input.error.message); push(input.error.code); push(input.error.type); }
166
+ if (input.body) {
167
+ push(input.body.message); push(input.body.type);
168
+ if (input.body.error) { push(input.body.error.message); push(input.body.error.code); push(input.body.error.type); }
169
+ }
170
+ }
171
+ var hay = blobs.join(' | ').toLowerCase();
172
+ if (!hay) return false;
173
+ return hay.indexOf('authentication_error') !== -1 ||
174
+ hay.indexOf('invalid_api_key') !== -1 ||
175
+ hay.indexOf('invalid x-api-key') !== -1 ||
176
+ hay.indexOf('incorrect api key') !== -1 ||
177
+ hay.indexOf('invalid api key') !== -1 ||
178
+ hay.indexOf('no api key provided') !== -1;
179
+ }