bunnyquery 1.9.7 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "bunnyquery",
3
- "version": "1.9.7",
3
+ "version": "1.10.0",
4
4
  "description": "Embeddable BunnyQuery AI chat widget + its framework-agnostic chat engine",
5
5
  "main": "bunnyquery.js",
6
6
  "exports": {
@@ -140,6 +140,40 @@ export function getProjectContextWindow(projectId: string): number | null {
140
140
  // actual request cannot drift: they were 22000 and 25000 respectively, which
141
141
  // under-reserved by 3k on a window spent to the last token.
142
142
  export var MAX_OUTPUT_TOKENS = 25000;
143
+
144
+ /**
145
+ * The same ceiling for an INDEXING pass, which is a different job with a different shape.
146
+ *
147
+ * A chat turn has a person waiting, so a long reply is a worse outcome than a truncated
148
+ * one and 25,000 is generous for it. An indexing pass has nobody waiting and one job:
149
+ * emit records for the window it was shown. When it runs out of budget mid-window the
150
+ * worker halves the window and re-sends it (`window_scale 1.0 -> 0.5`), so the file pays
151
+ * roughly twice the passes for the rest of its length.
152
+ *
153
+ * WHY 64,000, measured on a live 465-row spreadsheet (9 passes, project ap21U8y5byIbkGSv):
154
+ * - gpt-5.6-luna allows 128,000 output tokens, so 25,000 was 19.5% of what it permits.
155
+ * - Only 1 of those 9 passes hit the cap; the median used 6,370. A cap is a CEILING, not
156
+ * a target: the model stops when it is done, so raising it costs nothing on the eight
157
+ * passes that never approach it. It only changes the one that would have truncated.
158
+ * - Fitting duration against output tokens across the nine: `55.4s + output / 131 tok/s`
159
+ * (R^2 0.906). The binding constraint is TIME, not the model.
160
+ * - The worker's upstream timeout is 870s and its Lambda is 900s with a 30s settle
161
+ * reserve, which puts the theoretical ceiling near 103,000 tokens. 64,000 lands at
162
+ * about 544s and leaves roughly 300s of margin for a slow provider hour or a pass with
163
+ * more tool round trips. Spending that margin buys nothing: 64,000 is already enough
164
+ * for a full 250-row window to finish in one pass, which is the whole point.
165
+ *
166
+ * It does NOT feed OUTPUT_TOKEN_RESERVE below. That reserve exists to size the INPUT
167
+ * budget for buildBoundedChatMessages, which only the chat path calls; an indexing message
168
+ * is built from a file window, not from bounded history. Wiring this into the reserve would
169
+ * shrink a budget this number has nothing to do with.
170
+ *
171
+ * Re-derive rather than nudge: re-run the fit if the model, the Lambda timeout or
172
+ * REQUEST_TIMEOUT changes, and note that a model whose own ceiling is lower still wins
173
+ * (getMaxOutputTokens clamps, so gpt-4o stays at its 4,000).
174
+ */
175
+ export var INDEXING_MAX_OUTPUT_TOKENS = 64000;
176
+
143
177
  export var OUTPUT_TOKEN_RESERVE = MAX_OUTPUT_TOKENS;
144
178
  export var TOOL_AND_RESPONSE_BUFFER = 4000;
145
179
  export var MIN_INPUT_TOKEN_BUDGET = 8000;
@@ -225,9 +259,19 @@ export function getModelContextWindow(platform: string, model?: string): number
225
259
  * but a model whose own cap is lower rejects the request outright, so clamp to
226
260
  * whichever is smaller. Models with no known cap keep MAX_OUTPUT_TOKENS.
227
261
  */
228
- export function getMaxOutputTokens(platform: string, model?: string): number {
262
+ export function getMaxOutputTokens(
263
+ platform: string,
264
+ model?: string,
265
+ /** 'indexing' asks for INDEXING_MAX_OUTPUT_TOKENS instead. Omitted means chat, so every
266
+ * existing caller keeps the number it had. */
267
+ purpose?: 'chat' | 'indexing',
268
+ ): number {
269
+ var want = purpose === 'indexing' ? INDEXING_MAX_OUTPUT_TOKENS : MAX_OUTPUT_TOKENS;
270
+ // The model's own ceiling still wins. This is the whole reason the clamp lives here and
271
+ // not at the call sites: gpt-4o caps output at 4,000 and legacy 3.5 Sonnet at 8,000, and
272
+ // asking either for 64,000 is a rejected request, not a long answer.
229
273
  var cap = resolveByModelId(apiReportedMaxOutput, MAX_OUTPUT_BY_MODEL, model);
230
- return cap ? Math.min(MAX_OUTPUT_TOKENS, cap) : MAX_OUTPUT_TOKENS;
274
+ return cap ? Math.min(want, cap) : want;
231
275
  }
232
276
 
233
277
  /**
@@ -15,6 +15,52 @@
15
15
 
16
16
  import { type AttachmentParser, registerAttachmentParser } from './attachment_parsers';
17
17
 
18
+ /**
19
+ * One report about a turn that is streaming, as handed to `onLiveStreamUpdate`.
20
+ *
21
+ * Deliberately a flat snapshot rather than the SseParser itself: the hook is a
22
+ * VIEW seam, and handing a client the parser would invite it to drive the stream
23
+ * (feed it, end it, read the assembled body) behind the session's back.
24
+ */
25
+ export interface LiveStreamUpdate {
26
+ /** Server item id of the turn, the same id its bubbles carry as _serverItemId. */
27
+ serverItemId: string;
28
+ /** History cache key (`projectId#platform`) the turn belongs to. A host that
29
+ * renders several projects must ignore an update for a chat it is not showing. */
30
+ ownerKey: string;
31
+ /** 'start' on the first paint, 'update' on every later one, 'end' once the
32
+ * stream is over and nothing more will be painted - whether because the turn
33
+ * settled (its authoritative answer is about to replace the live text) or
34
+ * because it was stopped. 'end' is only sent to a host that was told 'start'. */
35
+ phase: 'start' | 'update' | 'end';
36
+ /** Answer text so far, already trimmed to a safe reveal boundary: it never
37
+ * ends inside a half-arrived link, fence or url. Empty on 'end'. */
38
+ text: string;
39
+ /** Extended-thinking text so far. Separate from `text` and never part of it. */
40
+ thinkingText: string;
41
+ /** Tools reached for, in order of appearance, duplicates kept. */
42
+ toolNames: string[];
43
+ /** A terminal event arrived. False on 'end' means the stream was cut. */
44
+ complete: boolean;
45
+ /** The terminal event that arrived meant the answer FINISHED rather than DIED:
46
+ * false while running, false on a cut stream, and false when the stream ended
47
+ * on a provider error. `complete` answers "is anything more coming?"; this one
48
+ * answers "is this the whole answer?", and they differ on exactly the case
49
+ * that costs text - an `error` frame is terminal and truncating at once. A host
50
+ * drawing a "partial answer" affordance wants THIS one. Added after `complete`
51
+ * and always present: a host that ignores it reads as it did before. */
52
+ answerComplete: boolean;
53
+ /** The stream ended in a provider error. */
54
+ errored: boolean;
55
+ /** How many chunks of this turn each of skapi's two transports carried FIRST:
56
+ * `socket` for the websocket relay, `poll` for the chunk-table read. Both feed
57
+ * the same sink by design, so a turn that streamed perfectly over the socket
58
+ * and one that was polled the whole way are otherwise indistinguishable. A
59
+ * host that does not care can ignore it; a host showing a live/degraded
60
+ * indicator, or just logging which path it got, reads this. */
61
+ transport: { socket: number; poll: number };
62
+ }
63
+
18
64
  export interface ChatEngineConfig {
19
65
  /** skapi.clientSecretRequest, bound to the consumer's skapi instance. */
20
66
  clientSecretRequest: (opts: any) => Promise<any>;
@@ -86,6 +132,116 @@ export interface ChatEngineConfig {
86
132
  queue?: string;
87
133
  };
88
134
  }) => void;
135
+ /**
136
+ * Opt in to LIVE STREAMING of chat turns.
137
+ *
138
+ * Off by default, and for the same shipping-order reason `windowedIndexing`
139
+ * is: THE BACKEND MUST SHIP FIRST. When on, every chat turn carries two
140
+ * `stream` flags (see requests.ts chatStreamWiring) and the polling row
141
+ * settles with a STATUS AND NO BODY, because the answer was the stream. On a
142
+ * region whose polling worker does not relay, that same request either has
143
+ * its unknown `since` cursor rejected or stores an SSE transcript where the
144
+ * readers expect a parsed document, and the turn reads back as an empty
145
+ * answer. So it stays off until the worker is deployed, then flips per
146
+ * environment.
147
+ *
148
+ * It also needs `clientSecretRequestFinalize` below: without it a streamed
149
+ * turn is never finalized, so its row keeps a status and no body forever and
150
+ * a later history load shows the question with an empty answer.
151
+ */
152
+ liveStreaming?: boolean;
153
+ /**
154
+ * Also push each relayed chunk over skapi's websocket, so text lands as it is
155
+ * relayed instead of on the next poll tick. Requires `liveStreaming`.
156
+ *
157
+ * SEPARATE FROM `liveStreaming` ON PURPOSE, and off unless a host asks. It is a
158
+ * pure accelerator with a safe fallback, so the reason is not risk to the chat, it
159
+ * is what it does to the HOST'S OWN realtime: skapi's joinRealtime REPLACES the
160
+ * connection's group rather than adding to it, so for the length of a turn this
161
+ * takes the room. The dashboard owns its skapi instance and uses realtime for
162
+ * nothing else, so it opts in. The embeddable widget is handed the EMBEDDER'S
163
+ * instance and cannot know what their app does with it, so it stays off there
164
+ * unless the embedder turns it on.
165
+ */
166
+ liveStreamingRealtime?: boolean;
167
+ /**
168
+ * skapi.clientSecretRequestFinalize, bound to the consumer's skapi instance.
169
+ * Stores the version of a streamed turn that history should keep (the engine
170
+ * sends the ASSEMBLED provider body, so history reads it exactly as it reads
171
+ * a buffered turn) and releases that request's chunks. Optional: a host
172
+ * without it can still stream, it just leaves the chunks and an empty row.
173
+ */
174
+ clientSecretRequestFinalize?: (
175
+ requestId: string,
176
+ data: any,
177
+ options: { url: string; method: string; service?: string; owner?: string },
178
+ ) => Promise<any>;
179
+ /**
180
+ * skapi.clientSecretRequestStream, bound to the consumer's skapi instance.
181
+ *
182
+ * THE SECOND HALF OF THE DURABILITY GUARANTEE, and without it a streamed turn
183
+ * is only as durable as the tab that started it. A streamed row settles with a
184
+ * status and NO body; the answer is stored as chunks until
185
+ * clientSecretRequestFinalize says what to keep. A row that settles while no
186
+ * poll is attached (the user closed the tab, a mobile browser discarded it,
187
+ * the device slept and the interval stopped) is therefore never finalized, and
188
+ * a later history load sees a terminal row with no body and used to emit no
189
+ * assistant bubble at all: the answer simply gone from the conversation, with
190
+ * every byte of it still sitting in the chunk table.
191
+ *
192
+ * This is the documented way back to it. Given the request id it fetches every
193
+ * chunk of an already-finished turn in one pass (paging internally on `more`)
194
+ * and delivers them in order through `onStream`, then resolves. The engine
195
+ * feeds those into a fresh SSE parser and treats the assembled body exactly as
196
+ * it treats a live one, including finalizing it, which stores the answer as
197
+ * ordinary history and releases the chunks, so each row is recovered at most
198
+ * once ever.
199
+ *
200
+ * Optional. Without it the engine still marks such turns (`_streamPending` on
201
+ * the bubble) but has no way to read them back, so a host that ignores this
202
+ * behaves as it does today.
203
+ *
204
+ * NOTE THAT THIS HOOK, NOT `liveStreaming`, IS WHAT ARMS RECOVERY. See
205
+ * streamRecoveryEnabled() below for why the two decisions are separate.
206
+ */
207
+ clientSecretRequestStream?: (
208
+ requestId: string,
209
+ options: {
210
+ url: string;
211
+ method: string;
212
+ onStream?: (chunk: string, seq: number) => void;
213
+ since?: number;
214
+ poll?: number;
215
+ service?: string;
216
+ owner?: string;
217
+ },
218
+ ) => Promise<any>;
219
+ /**
220
+ * Observation hook for a live-streaming turn, called at most once per paint
221
+ * (about once a second) plus once when the turn settles.
222
+ *
223
+ * The engine already paints the answer text into the pending bubble itself,
224
+ * so a host needs this ONLY for the affordances the engine deliberately does
225
+ * not decide the presentation of: a "thinking..." line, or a "querying sales
226
+ * table..." row drawn from the tools the model reached for before any answer
227
+ * text exists. Optional, and a host without it behaves exactly as today.
228
+ *
229
+ * Never throw from it: it is called on the paint path and a throw would cost
230
+ * the user the rest of their answer. The engine guards it anyway.
231
+ */
232
+ onLiveStreamUpdate?: (update: LiveStreamUpdate) => void;
233
+ /**
234
+ * Force the read-back of already-streamed turns OFF, even though the chunk
235
+ * reader is injected.
236
+ *
237
+ * There is no need to set it to turn recovery ON: injecting
238
+ * `clientSecretRequestStream` is what arms it (see streamRecoveryEnabled).
239
+ * This exists only as the way back out for a host that wants byte-for-byte the
240
+ * pre-recovery rendering of a terminal-but-empty row - no bubble, no marker, no
241
+ * chunk read - while keeping the reader available for its own use. Omit it and
242
+ * nothing changes.
243
+ */
244
+ streamRecovery?: boolean;
89
245
  /**
90
246
  * Single-item csr-poll point lookup (skapi.util.request('csr-poll', {id,
91
247
  * service, owner}, {auth:true})). For a RESOLVED item the backend returns
@@ -120,6 +276,93 @@ export function windowedIndexingEnabled(): boolean {
120
276
  return _config?.windowedIndexing === true;
121
277
  }
122
278
 
279
+ /** True when the consumer has opted in to live streaming of chat turns. */
280
+ export function liveStreamingRealtimeEnabled(): boolean {
281
+ return liveStreamingEnabled() && _config?.liveStreamingRealtime === true;
282
+ }
283
+
284
+ export function liveStreamingEnabled(): boolean {
285
+ return _config?.liveStreaming === true;
286
+ }
287
+
288
+ /**
289
+ * True when a streamed turn whose answer never reached its row can be READ BACK.
290
+ *
291
+ * IT ASKS FOR THE READER AND NOT FOR `liveStreaming`, AND THAT SPLIT IS THE WHOLE
292
+ * POINT. "Should NEW turns stream?" and "can an ALREADY streamed row be recovered?"
293
+ * are two different questions about two different sets of rows, and answering both
294
+ * with one flag strands the second set the moment the first answer changes.
295
+ *
296
+ * The failure, and it is not hypothetical - it is what turning the feature off
297
+ * does. A row streamed yesterday holds its answer in the chunk table and a status
298
+ * and no body on the row; only csr-finalize ever copies one onto it. Flip
299
+ * `liveStreaming` off today (an embedder drops the option, a dev rolls the flag
300
+ * back after a bad deploy, a client's skapiSupportsStreaming probe degrades the
301
+ * instance to buffered) and every one of those rows instantly becomes unmarked,
302
+ * unrecoverable and unreadable: the mapper emits no bubble for it, the recovery
303
+ * never looks at it, and its answer is unreachable with every byte of it still
304
+ * stored. Rolling a rendering flag back must not delete anybody's history.
305
+ *
306
+ * The reader is the honest test because it is the CAPABILITY the recovery needs.
307
+ * Without it the engine could mark such a turn and never fill it in, trading a
308
+ * missing bubble for a permanently empty one, which is strictly worse than the
309
+ * bug - so the marker is still only ever minted when something can act on it.
310
+ *
311
+ * The cost of asking the wider question is one wasted read, once, on a row that
312
+ * was terminal and empty for some reason other than streaming (a buffered turn
313
+ * whose body the worker never managed to spill). That read finds no chunks, the
314
+ * bubble is dropped, and the list looks exactly as it did before. Set
315
+ * `streamRecovery: false` to opt out of even that.
316
+ */
317
+ export function streamRecoveryEnabled(): boolean {
318
+ if (_config?.streamRecovery === false) return false;
319
+ return typeof _config?.clientSecretRequestStream === 'function';
320
+ }
321
+
322
+ /**
323
+ * Does a given skapi INSTANCE support the streaming half of the protocol? Ask this
324
+ * before honouring a `liveStreaming: true` opt-in, and degrade to buffered when the
325
+ * answer is no.
326
+ *
327
+ * THE FAILURE THIS PREVENTS, and it is the one the SDK's own docs call quiet. A
328
+ * streamed turn carries TWO `stream` flags: skapi's (relay the destination's bytes
329
+ * into the chunk table) and the DESTINATION's own field inside `data`, which
330
+ * BunnyQuery is the party that sets, because skapi relays bytes and knows no
331
+ * vendor. clientSecretRequest validates its params against a schema and KEEPS ONLY
332
+ * THE KEYS IN THAT SCHEMA, so an skapi-js predating the feature does not reject
333
+ * `stream` - it silently DROPS it. What ships is then the exact split
334
+ * chatStreamWiring exists to make impossible: the destination is asked to answer in
335
+ * SSE frames, skapi waits and stores the whole transcript on the row, and
336
+ * extractClaudeText / extractOpenAIText read a wall of `data: {...}` lines where a
337
+ * document should be and find no answer at all. Nothing throws and nothing logs;
338
+ * the user gets an empty reply on every single turn.
339
+ *
340
+ * It lives in the ENGINE rather than in each client because the two clients are
341
+ * diffed against each other and this is precisely the kind of predicate that forks:
342
+ * the widget must ask it (init() takes the EMBEDDER's instance, and an embed page
343
+ * pins its own skapi-js version, so `liveStreaming: true` is a REQUEST and this is
344
+ * what grants it) and agent.vue must ask it too (its instance is the repo's own, so
345
+ * only a stale node_modules or an unbuilt skapi-js can fail it - which is exactly
346
+ * the state a dev flipping the constant is most likely to be in, and a silently
347
+ * empty chat is the worst possible way to find out).
348
+ *
349
+ * Probed by the two public METHODS rather than by a version string, for two
350
+ * reasons. They ship in the same change as the `stream` key (one feature: the
351
+ * relay, the finalize that stores what to keep, and the read-back), so an SDK
352
+ * missing them is exactly the SDK that would drop the flag. And they are not merely
353
+ * a proxy for the capability, they ARE half of it - the engine needs finalize to
354
+ * store a streamed answer onto its row and stream to read an unfinalized one back,
355
+ * and streaming without either leaves every answer in the chunk table with nothing
356
+ * able to fetch it. There is no cheap DIRECT probe of the schema: the only way to
357
+ * learn that `stream` was dropped is to send a real request and read an empty
358
+ * answer, which is the bug itself.
359
+ */
360
+ export function skapiSupportsStreaming(sk: any): boolean {
361
+ return !!sk
362
+ && typeof sk.clientSecretRequestStream === 'function'
363
+ && typeof sk.clientSecretRequestFinalize === 'function';
364
+ }
365
+
123
366
  /** Spread helper: `{ ...pollOpt() }` adds `poll` only when configured. */
124
367
  export function pollOpt(): { poll?: number } {
125
368
  const p = _config?.poll;
@@ -20,7 +20,73 @@ function isTransientStatus(status: number): boolean {
20
20
  return status === 408 || status === 425 || status === 429 || status >= 500;
21
21
  }
22
22
 
23
+ /**
24
+ * True when a csr-poll answer is the STATUS ENVELOPE rather than a stored body.
25
+ *
26
+ * Duck-typed, because the engine does not import skapi-js and the SDK does not
27
+ * export its own copy. The rule is the SDK's (isPollEnvelope): a request that has
28
+ * a stored result hands that result back verbatim, every other state hands back
29
+ * `{ id, status, in_queue, ... }`. A finalized body is the caller's own content
30
+ * and can itself carry a `status` key (OpenAI's Responses object does), so the
31
+ * id/in_queue pair is demanded too: a provider body would have to reproduce all
32
+ * three to be mistaken for an envelope.
33
+ *
34
+ * It lives HERE, next to the error readers, rather than in session.ts, because
35
+ * both of the things that have to recognise an envelope (the settle that
36
+ * substitutes an assembled body for one, and the error readers below) must agree
37
+ * on what one is. It was written twice once; that is how the error readers came to
38
+ * look one level too shallow.
39
+ */
40
+ export function isCsrStatusEnvelope(res: any): boolean {
41
+ return !!res && typeof res === 'object' && !Array.isArray(res) &&
42
+ typeof res.status === 'string' && typeof res.id === 'string' && ('in_queue' in res);
43
+ }
44
+
45
+ /**
46
+ * The real error payload inside a FAILED csr-poll status envelope, or undefined
47
+ * when `input` is not one.
48
+ *
49
+ * THE FAILURE THIS PREVENTS, verbatim from the wire. A buffered turn that fails
50
+ * polls back as the worker's failed payload itself:
51
+ *
52
+ * { status_code: 401, body: { error: { type, message } }, truncated: false }
53
+ *
54
+ * ...which every predicate below reads correctly. A STREAMED turn that fails does
55
+ * not: the poller (client_secret_key_request_polling) cannot return the error
56
+ * early for a streamed row, because the chunks that arrived before the stream died
57
+ * have to come back in the same response, so it falls through and ships
58
+ *
59
+ * { id, status: 'failed', queue_name, in_queue, stream, chunks, last_seq,
60
+ * more, error: <the payload above> }
61
+ *
62
+ * The payload is one level deeper, and every predicate here looked at the top
63
+ * level: `response.error.message` is undefined on it, `response.status_code` is
64
+ * absent, and `response.status` is the string 'failed' rather than a number. So a
65
+ * wrong API key on a streamed turn read as "not an error, and no answer either",
66
+ * which the caller renders as "No text response received from AI provider": the
67
+ * one message that tells the user nothing.
68
+ *
69
+ * Unwrapping HERE, once, is what makes a streamed error and a buffered error take
70
+ * the same path through every reader below. Deliberately not recursive: the value
71
+ * inside an envelope is a provider payload, never another envelope, and a single
72
+ * unwrap cannot loop on a malformed one.
73
+ *
74
+ * A failed envelope with a NULL payload (the worker recorded no detail, or its
75
+ * spill could not be fetched) still yields an object, because the row's status is
76
+ * itself the fact: 'failed' with nothing attached must not read as a clean turn.
77
+ */
78
+ export function csrEnvelopeError(input: any): any {
79
+ if (!isCsrStatusEnvelope(input)) return undefined;
80
+ // Only a failure carries one. 'resolved' means the body IS the answer (the
81
+ // settle substitutes it), and 'cancelled' is settled by its own predicate
82
+ // upstream, which must keep seeing the envelope it recognises.
83
+ if (input.status !== 'failed') return undefined;
84
+ return input.error != null ? input.error : { message: 'The AI provider request failed.' };
85
+ }
86
+
23
87
  export function getErrorMessage(input: any): string {
88
+ var envErr = csrEnvelopeError(input);
89
+ if (envErr !== undefined) input = envErr;
24
90
  if (!input) return 'Something went wrong.';
25
91
  if (typeof input === 'string') return input;
26
92
  if (input.error && input.error.message) return input.error.message;
@@ -48,7 +114,11 @@ export function getErrorMessage(input: any): string {
48
114
  }
49
115
 
50
116
  export function isErrorResponseBody(response: any): boolean {
51
- if (!response || typeof response !== 'object') return false;
117
+ // A FAILED streamed row is an error however empty its payload turned out to
118
+ // be: the status is the fact. See csrEnvelopeError.
119
+ var envErr = csrEnvelopeError(response);
120
+ if (envErr !== undefined) response = envErr;
121
+ if (!response || typeof response !== 'object') return envErr !== undefined;
52
122
  if (typeof response.status_code === 'number' && response.status_code >= 400) return true;
53
123
  if (response.type === 'error') return true;
54
124
  if (response.error && (response.error.message || response.error.type)) return true;
@@ -78,6 +148,11 @@ export function isErrorResponseBody(response: any): boolean {
78
148
  // retryable after a token refresh — so 401 (auth) and 429/5xx (transient) are
79
149
  // intentionally NOT treated as non-retryable here.
80
150
  export function isNonRetryableRequestError(input: any): boolean {
151
+ // Same one-level unwrap as the readers above: a streamed turn's failure is
152
+ // wrapped in its poll envelope, and a retry gate that cannot see the payload
153
+ // would re-send a request the provider has already rejected as malformed.
154
+ var envErr = csrEnvelopeError(input);
155
+ if (envErr !== undefined) input = envErr;
81
156
  if (!input || typeof input !== 'object') return false;
82
157
 
83
158
  var status = typeof input.status_code === 'number' ? input.status_code
@@ -118,6 +193,12 @@ export function isNonRetryableRequestError(input: any): boolean {
118
193
  }
119
194
 
120
195
  export function isAuthExpiredError(input: any): boolean {
196
+ // A streamed turn whose bearer expired fails exactly like a buffered one, one
197
+ // level deeper. Without this unwrap the refresh-and-resend gate never fires on
198
+ // the streaming path, so an expiry that costs a buffered turn nothing costs a
199
+ // streamed turn the whole answer.
200
+ var envErr = csrEnvelopeError(input);
201
+ if (envErr !== undefined) input = envErr;
121
202
  if (!input) return false;
122
203
  var blobs: string[] = [];
123
204
  var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
@@ -156,6 +237,11 @@ export function isAuthExpiredError(input: any): boolean {
156
237
  * reduced to by getErrorMessage, because the view usually only keeps the text.
157
238
  */
158
239
  export function isProviderApiKeyError(input: any): boolean {
240
+ // Same unwrap. A wrong API key on a streamed turn is the exact case CRITICAL 2
241
+ // was reported for: without this the "check the project's key" affordance never
242
+ // appears on the platform the user is actually streaming with.
243
+ var envErr = csrEnvelopeError(input);
244
+ if (envErr !== undefined) input = envErr;
159
245
  if (!input) return false;
160
246
  var blobs: string[] = [];
161
247
  var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
@@ -6,6 +6,7 @@
6
6
  */
7
7
  import { extractClaudeText, extractOpenAIText, INDEXING_COMPLETE_MARKER, EMPTY_INDEXING_REPLY, getChatHistory, bgIndexingQueueName } from './requests';
8
8
  import { isErrorResponseBody, getErrorMessage } from './errors';
9
+ import { streamRecoveryEnabled } from './config';
9
10
  import { sanitizeAttachmentLinksForHistory } from './links';
10
11
  import type { ChatMessage } from './host';
11
12
 
@@ -702,8 +703,11 @@ export type MapHistoryOptions = {
702
703
  };
703
704
 
704
705
  export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'openai', opts: MapHistoryOptions) {
705
- var mapped: any[] = [], runningItemIds: string[] = [];
706
+ var mapped: any[] = [], runningItemIds: string[] = [], streamPendingItemIds: string[] = [];
706
707
  var extractAssistantText = platform === 'openai' ? extractOpenAIText : extractClaudeText;
708
+ // See the `isStreamPending` block below. Read ONCE per call rather than per
709
+ // item: it is a config lookup, and the answer cannot change mid-list.
710
+ var canRecoverStreams = streamRecoveryEnabled();
707
711
  var filtered = filterListByClearHorizon(list, opts.clearedAt);
708
712
  filtered.slice().reverse().forEach(function (item) {
709
713
  var requestBody = item && item.request_body;
@@ -727,6 +731,45 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
727
731
  ? ((typeof item.response_text === 'string' ? item.response_text : '').trim())
728
732
  : ((extractAssistantText(response) || '').trim() || ''));
729
733
  var isErrorResponse = !isPending && (isFailed || (!isCompact && isErrorResponseBody(response)));
734
+ // AN ANSWER THAT IS UNKNOWN, NOT EMPTY.
735
+ //
736
+ // A STREAMED turn stores nothing on its row: the relay appends the
737
+ // destination's bytes to the chunk table and settles the row with a status
738
+ // and no body, and only csr-finalize ever copies an answer onto the row. So a
739
+ // row that is terminal, carries no body and carries no error is not a turn
740
+ // that answered nothing: it is a turn nobody finalized. That happens for one
741
+ // ordinary reason: the row settled while no poll was attached (the tab was
742
+ // closed, a mobile browser discarded it, the device slept and the interval
743
+ // stopped), so the client that would have finalized it was not there.
744
+ //
745
+ // Read as "empty" (which is what the `assistantText` guard further down did),
746
+ // such a row produces NO assistant bubble at all and the answer is simply gone
747
+ // from the conversation, with every byte of it still sitting in the chunk
748
+ // table, reachable through clientSecretRequestStream. So it is marked instead,
749
+ // and the two things that act on the mark are the merge (an unknown answer
750
+ // never overwrites a known one) and the recovery (an unknown answer is
751
+ // resolved by reading the chunks back). See ChatMessage._streamPending.
752
+ //
753
+ // Deliberately narrow, so nothing that is genuinely an empty turn is caught:
754
+ // - only when this host streams AND can read chunks back (see
755
+ // streamRecoveryEnabled); a host that does neither behaves exactly as it
756
+ // does today, and one that cannot read them back would only trade a
757
+ // missing bubble for a permanently empty one;
758
+ // - never on the background queue. chatStreamWiring turns streaming OFF for
759
+ // every bg-queue turn (an indexing pass must not stream, and an attachment
760
+ // turn is left buffered on purpose), so a bg row with no body really did
761
+ // answer nothing;
762
+ // - never on a compact stub, whose body was withheld by the server rather
763
+ // than never stored;
764
+ // - 'resolved' only. A FAILED row already renders its error, which is the
765
+ // authoritative account of the turn (and, once csrEnvelopeError is in
766
+ // play, a truthful one). Its chunks are deliberately kept but not
767
+ // rendered. See _finalizeStreamedTurn.
768
+ var isStreamPending = canRecoverStreams && !isCompact && !isPending && !isCancelledItem
769
+ && !isErrorResponse && !item._isBgTask && !item._isOnBgQueue
770
+ && item.status === 'resolved'
771
+ && (item.response_body == null) && (item.error == null)
772
+ && !assistantText;
730
773
  // Record the completion marker, then STRIP it — both, and in that order.
731
774
  // Recording gives the display layer a structured signal instead of a substring
732
775
  // search over model prose. Stripping matches the live resolution path: without
@@ -810,6 +853,14 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
810
853
  if (serverItemId !== undefined) em._serverItemId = serverItemId;
811
854
  if (replyTs !== undefined) em._ts = replyTs;
812
855
  mapped.push(em);
856
+ } else if (isStreamPending) {
857
+ // The bubble stands for the turn so the merge has something to key on and
858
+ // the recovery has somewhere to write. Its content is UNKNOWN: empty here
859
+ // is the absence of an answer on the row, not the absence of an answer.
860
+ var sp: any = { role: 'assistant', content: '', _streamPending: true };
861
+ if (serverItemId !== undefined) { sp._serverItemId = serverItemId; streamPendingItemIds.push(serverItemId); }
862
+ if (replyTs !== undefined) sp._ts = replyTs;
863
+ mapped.push(sp);
813
864
  // `|| reportedComplete`: a pass whose ENTIRE answer was the completion token
814
865
  // strips down to an empty string, and the plain `assistantText` guard then
815
866
  // emitted no bubble at all — while the live path emitted one. The run read as
@@ -837,7 +888,54 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
837
888
  var ownerKey = chatCacheKey(opts.projectId, platform, opts.userId);
838
889
  for (var oi = 0; oi < mapped.length; oi++) mapped[oi]._ownerKey = ownerKey;
839
890
  }
840
- return { messages: mapped, runningItemIds: runningItemIds };
891
+ // `streamPendingItemIds` is additive: a consumer that only destructures
892
+ // { messages, runningItemIds } (agent.vue's own call site does) is unaffected.
893
+ // The engine drives recovery off the BUBBLES rather than this list (the merge
894
+ // runs between the two and can adopt a local answer onto one of them), so this
895
+ // is reporting, not the mechanism.
896
+ return { messages: mapped, runningItemIds: runningItemIds, streamPendingItemIds: streamPendingItemIds };
897
+ }
898
+
899
+ /**
900
+ * Let a LOCAL copy of a turn survive a page whose copy of it is
901
+ * AUTHORITATIVE-BUT-EMPTY. Mutates `incoming`; returns true when it took anything.
902
+ *
903
+ * THE FAILURE THIS PREVENTS. A streamed turn's row goes 'resolved' the moment the
904
+ * relay finishes, and its answer reaches the row only when csr-finalize stores it,
905
+ * one poll interval plus a round trip later. A first-page history refetch landing
906
+ * inside that window maps the row to a `_streamPending` bubble with no content, and
907
+ * the merge, which believes the server, throws away the local bubble holding the
908
+ * answer the reader is looking at. The window opens on EVERY streamed turn, and a
909
+ * refetch fires from visibilitychange, so it is not a corner case.
910
+ *
911
+ * The rule is the same one the recovery reads: an UNKNOWN answer never overwrites a
912
+ * KNOWN one. Where the local copy is still live (pending, or being painted into),
913
+ * its live-ness is adopted too: without it the merge would hand back a settled
914
+ * bubble the painter can no longer find (_liveTargetIndex wants isPending or
915
+ * _streaming) and that _turnAlreadyRendered would then read as already answered, so
916
+ * the settle would drop the real answer on the floor.
917
+ */
918
+ export function adoptLocalAnswerIntoPage(incoming: ChatMessage, local: ChatMessage): boolean {
919
+ if (!incoming || !local || !incoming._streamPending) return false;
920
+ if (incoming.role !== 'assistant' || local.role !== 'assistant') return false;
921
+ var hasText = typeof local.content === 'string' && local.content.length > 0;
922
+ var isLive = !!(local.isPending || local._streaming);
923
+ if (!hasText && !isLive) return false;
924
+ if (hasText) {
925
+ incoming.content = local.content;
926
+ // The answer is on screen, so nothing has to be read back. Whether the row
927
+ // itself ever gets a stored body is the live path's business (its finalize is
928
+ // in flight); if that fails, the NEXT load meets an empty row again and
929
+ // recovers it then.
930
+ incoming._streamPending = false;
931
+ }
932
+ // Carried whatever the text says: these are what keep a still-running turn
933
+ // reachable by the painter and by the settle.
934
+ if (local._localId !== undefined) incoming._localId = local._localId;
935
+ if (local.isPending) incoming.isPending = true;
936
+ if (local.isPendingInProcess) incoming.isPendingInProcess = true;
937
+ if (local._streaming) incoming._streaming = true;
938
+ return true;
841
939
  }
842
940
 
843
941
  /* ---- rescuing in-flight bubbles across a first-page refetch ---------------
@@ -886,6 +984,19 @@ export function shouldRescueInFlightMessage(m: ChatMessage, ctx: RescueDecisionC
886
984
  // it. Unconditional: applying the pending-assistant test here would delete the
887
985
  // user's message mid-upload whenever some other turn happened to be in flight.
888
986
  if (m._stageId) return true;
987
+ // A bubble being painted from a LIVE stream, before its dispatch has reported a
988
+ // server id. Same shape of reason as the staged bubble above: the page identifies
989
+ // turns by id, so nothing in it can stand for this one, and dropping it would
990
+ // leave the stream with no bubble to paint into (the painter finds its target by
991
+ // _serverItemId, and this bubble has none to be found by) - the turn would sit on
992
+ // "Thinking..." until it settled, with every relayed byte already spent.
993
+ //
994
+ // Gated on the missing id rather than on _streaming alone, and deliberately: once
995
+ // the id IS on the bubble, the tests below are right and the local copy really is
996
+ // redundant. The page carries the same turn as a pending placeholder with that id,
997
+ // the painter finds THAT one on its next paint (about a second later), and rescuing
998
+ // as well would put the same turn on screen twice.
999
+ if (m._streaming && !m._serverItemId) return true;
889
1000
  if (!m._serverItemId && ctx.pageHasPendingAssistant) return false;
890
1001
  // In flight by its own flags, id or no id. The immediate-send pair is stamped
891
1002
  // with its server id as soon as the dispatch reports one, and that id is there