bunnyquery 1.9.6 → 1.9.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "bunnyquery",
3
- "version": "1.9.6",
3
+ "version": "1.9.10",
4
4
  "description": "Embeddable BunnyQuery AI chat widget + its framework-agnostic chat engine",
5
5
  "main": "bunnyquery.js",
6
6
  "exports": {
@@ -15,6 +15,52 @@
15
15
 
16
16
  import { type AttachmentParser, registerAttachmentParser } from './attachment_parsers';
17
17
 
18
+ /**
19
+ * One report about a turn that is streaming, as handed to `onLiveStreamUpdate`.
20
+ *
21
+ * Deliberately a flat snapshot rather than the SseParser itself: the hook is a
22
+ * VIEW seam, and handing a client the parser would invite it to drive the stream
23
+ * (feed it, end it, read the assembled body) behind the session's back.
24
+ */
25
+ export interface LiveStreamUpdate {
26
+ /** Server item id of the turn, the same id its bubbles carry as _serverItemId. */
27
+ serverItemId: string;
28
+ /** History cache key (`projectId#platform`) the turn belongs to. A host that
29
+ * renders several projects must ignore an update for a chat it is not showing. */
30
+ ownerKey: string;
31
+ /** 'start' on the first paint, 'update' on every later one, 'end' once the
32
+ * stream is over and nothing more will be painted - whether because the turn
33
+ * settled (its authoritative answer is about to replace the live text) or
34
+ * because it was stopped. 'end' is only sent to a host that was told 'start'. */
35
+ phase: 'start' | 'update' | 'end';
36
+ /** Answer text so far, already trimmed to a safe reveal boundary: it never
37
+ * ends inside a half-arrived link, fence or url. Empty on 'end'. */
38
+ text: string;
39
+ /** Extended-thinking text so far. Separate from `text` and never part of it. */
40
+ thinkingText: string;
41
+ /** Tools reached for, in order of appearance, duplicates kept. */
42
+ toolNames: string[];
43
+ /** A terminal event arrived. False on 'end' means the stream was cut. */
44
+ complete: boolean;
45
+ /** The terminal event that arrived meant the answer FINISHED rather than DIED:
46
+ * false while running, false on a cut stream, and false when the stream ended
47
+ * on a provider error. `complete` answers "is anything more coming?"; this one
48
+ * answers "is this the whole answer?", and they differ on exactly the case
49
+ * that costs text - an `error` frame is terminal and truncating at once. A host
50
+ * drawing a "partial answer" affordance wants THIS one. Added after `complete`
51
+ * and always present: a host that ignores it reads as it did before. */
52
+ answerComplete: boolean;
53
+ /** The stream ended in a provider error. */
54
+ errored: boolean;
55
+ /** How many chunks of this turn each of skapi's two transports carried FIRST:
56
+ * `socket` for the websocket relay, `poll` for the chunk-table read. Both feed
57
+ * the same sink by design, so a turn that streamed perfectly over the socket
58
+ * and one that was polled the whole way are otherwise indistinguishable. A
59
+ * host that does not care can ignore it; a host showing a live/degraded
60
+ * indicator, or just logging which path it got, reads this. */
61
+ transport: { socket: number; poll: number };
62
+ }
63
+
18
64
  export interface ChatEngineConfig {
19
65
  /** skapi.clientSecretRequest, bound to the consumer's skapi instance. */
20
66
  clientSecretRequest: (opts: any) => Promise<any>;
@@ -86,6 +132,116 @@ export interface ChatEngineConfig {
86
132
  queue?: string;
87
133
  };
88
134
  }) => void;
135
+ /**
136
+ * Opt in to LIVE STREAMING of chat turns.
137
+ *
138
+ * Off by default, and for the same shipping-order reason `windowedIndexing`
139
+ * is: THE BACKEND MUST SHIP FIRST. When on, every chat turn carries two
140
+ * `stream` flags (see requests.ts chatStreamWiring) and the polling row
141
+ * settles with a STATUS AND NO BODY, because the answer was the stream. On a
142
+ * region whose polling worker does not relay, that same request either has
143
+ * its unknown `since` cursor rejected or stores an SSE transcript where the
144
+ * readers expect a parsed document, and the turn reads back as an empty
145
+ * answer. So it stays off until the worker is deployed, then flips per
146
+ * environment.
147
+ *
148
+ * It also needs `clientSecretRequestFinalize` below: without it a streamed
149
+ * turn is never finalized, so its row keeps a status and no body forever and
150
+ * a later history load shows the question with an empty answer.
151
+ */
152
+ liveStreaming?: boolean;
153
+ /**
154
+ * Also push each relayed chunk over skapi's websocket, so text lands as it is
155
+ * relayed instead of on the next poll tick. Requires `liveStreaming`.
156
+ *
157
+ * SEPARATE FROM `liveStreaming` ON PURPOSE, and off unless a host asks. It is a
158
+ * pure accelerator with a safe fallback, so the reason is not risk to the chat, it
159
+ * is what it does to the HOST'S OWN realtime: skapi's joinRealtime REPLACES the
160
+ * connection's group rather than adding to it, so for the length of a turn this
161
+ * takes the room. The dashboard owns its skapi instance and uses realtime for
162
+ * nothing else, so it opts in. The embeddable widget is handed the EMBEDDER'S
163
+ * instance and cannot know what their app does with it, so it stays off there
164
+ * unless the embedder turns it on.
165
+ */
166
+ liveStreamingRealtime?: boolean;
167
+ /**
168
+ * skapi.clientSecretRequestFinalize, bound to the consumer's skapi instance.
169
+ * Stores the version of a streamed turn that history should keep (the engine
170
+ * sends the ASSEMBLED provider body, so history reads it exactly as it reads
171
+ * a buffered turn) and releases that request's chunks. Optional: a host
172
+ * without it can still stream, it just leaves the chunks and an empty row.
173
+ */
174
+ clientSecretRequestFinalize?: (
175
+ requestId: string,
176
+ data: any,
177
+ options: { url: string; method: string; service?: string; owner?: string },
178
+ ) => Promise<any>;
179
+ /**
180
+ * skapi.clientSecretRequestStream, bound to the consumer's skapi instance.
181
+ *
182
+ * THE SECOND HALF OF THE DURABILITY GUARANTEE, and without it a streamed turn
183
+ * is only as durable as the tab that started it. A streamed row settles with a
184
+ * status and NO body; the answer is stored as chunks until
185
+ * clientSecretRequestFinalize says what to keep. A row that settles while no
186
+ * poll is attached (the user closed the tab, a mobile browser discarded it,
187
+ * the device slept and the interval stopped) is therefore never finalized, and
188
+ * a later history load sees a terminal row with no body and used to emit no
189
+ * assistant bubble at all: the answer simply gone from the conversation, with
190
+ * every byte of it still sitting in the chunk table.
191
+ *
192
+ * This is the documented way back to it. Given the request id it fetches every
193
+ * chunk of an already-finished turn in one pass (paging internally on `more`)
194
+ * and delivers them in order through `onStream`, then resolves. The engine
195
+ * feeds those into a fresh SSE parser and treats the assembled body exactly as
196
+ * it treats a live one, including finalizing it, which stores the answer as
197
+ * ordinary history and releases the chunks, so each row is recovered at most
198
+ * once ever.
199
+ *
200
+ * Optional. Without it the engine still marks such turns (`_streamPending` on
201
+ * the bubble) but has no way to read them back, so a host that ignores this
202
+ * behaves as it does today.
203
+ *
204
+ * NOTE THAT THIS HOOK, NOT `liveStreaming`, IS WHAT ARMS RECOVERY. See
205
+ * streamRecoveryEnabled() below for why the two decisions are separate.
206
+ */
207
+ clientSecretRequestStream?: (
208
+ requestId: string,
209
+ options: {
210
+ url: string;
211
+ method: string;
212
+ onStream?: (chunk: string, seq: number) => void;
213
+ since?: number;
214
+ poll?: number;
215
+ service?: string;
216
+ owner?: string;
217
+ },
218
+ ) => Promise<any>;
219
+ /**
220
+ * Observation hook for a live-streaming turn, called at most once per paint
221
+ * (about once a second) plus once when the turn settles.
222
+ *
223
+ * The engine already paints the answer text into the pending bubble itself,
224
+ * so a host needs this ONLY for the affordances the engine deliberately does
225
+ * not decide the presentation of: a "thinking..." line, or a "querying sales
226
+ * table..." row drawn from the tools the model reached for before any answer
227
+ * text exists. Optional, and a host without it behaves exactly as today.
228
+ *
229
+ * Never throw from it: it is called on the paint path and a throw would cost
230
+ * the user the rest of their answer. The engine guards it anyway.
231
+ */
232
+ onLiveStreamUpdate?: (update: LiveStreamUpdate) => void;
233
+ /**
234
+ * Force the read-back of already-streamed turns OFF, even though the chunk
235
+ * reader is injected.
236
+ *
237
+ * There is no need to set it to turn recovery ON: injecting
238
+ * `clientSecretRequestStream` is what arms it (see streamRecoveryEnabled).
239
+ * This exists only as the way back out for a host that wants byte-for-byte the
240
+ * pre-recovery rendering of a terminal-but-empty row - no bubble, no marker, no
241
+ * chunk read - while keeping the reader available for its own use. Omit it and
242
+ * nothing changes.
243
+ */
244
+ streamRecovery?: boolean;
89
245
  /**
90
246
  * Single-item csr-poll point lookup (skapi.util.request('csr-poll', {id,
91
247
  * service, owner}, {auth:true})). For a RESOLVED item the backend returns
@@ -120,6 +276,93 @@ export function windowedIndexingEnabled(): boolean {
120
276
  return _config?.windowedIndexing === true;
121
277
  }
122
278
 
279
+ /** True when the consumer has opted in to live streaming of chat turns. */
280
+ export function liveStreamingRealtimeEnabled(): boolean {
281
+ return liveStreamingEnabled() && _config?.liveStreamingRealtime === true;
282
+ }
283
+
284
+ export function liveStreamingEnabled(): boolean {
285
+ return _config?.liveStreaming === true;
286
+ }
287
+
288
+ /**
289
+ * True when a streamed turn whose answer never reached its row can be READ BACK.
290
+ *
291
+ * IT ASKS FOR THE READER AND NOT FOR `liveStreaming`, AND THAT SPLIT IS THE WHOLE
292
+ * POINT. "Should NEW turns stream?" and "can an ALREADY streamed row be recovered?"
293
+ * are two different questions about two different sets of rows, and answering both
294
+ * with one flag strands the second set the moment the first answer changes.
295
+ *
296
+ * The failure, and it is not hypothetical - it is what turning the feature off
297
+ * does. A row streamed yesterday holds its answer in the chunk table and a status
298
+ * and no body on the row; only csr-finalize ever copies one onto it. Flip
299
+ * `liveStreaming` off today (an embedder drops the option, a dev rolls the flag
300
+ * back after a bad deploy, a client's skapiSupportsStreaming probe degrades the
301
+ * instance to buffered) and every one of those rows instantly becomes unmarked,
302
+ * unrecoverable and unreadable: the mapper emits no bubble for it, the recovery
303
+ * never looks at it, and its answer is unreachable with every byte of it still
304
+ * stored. Rolling a rendering flag back must not delete anybody's history.
305
+ *
306
+ * The reader is the honest test because it is the CAPABILITY the recovery needs.
307
+ * Without it the engine could mark such a turn and never fill it in, trading a
308
+ * missing bubble for a permanently empty one, which is strictly worse than the
309
+ * bug - so the marker is still only ever minted when something can act on it.
310
+ *
311
+ * The cost of asking the wider question is one wasted read, once, on a row that
312
+ * was terminal and empty for some reason other than streaming (a buffered turn
313
+ * whose body the worker never managed to spill). That read finds no chunks, the
314
+ * bubble is dropped, and the list looks exactly as it did before. Set
315
+ * `streamRecovery: false` to opt out of even that.
316
+ */
317
+ export function streamRecoveryEnabled(): boolean {
318
+ if (_config?.streamRecovery === false) return false;
319
+ return typeof _config?.clientSecretRequestStream === 'function';
320
+ }
321
+
322
+ /**
323
+ * Does a given skapi INSTANCE support the streaming half of the protocol? Ask this
324
+ * before honouring a `liveStreaming: true` opt-in, and degrade to buffered when the
325
+ * answer is no.
326
+ *
327
+ * THE FAILURE THIS PREVENTS, and it is the one the SDK's own docs call quiet. A
328
+ * streamed turn carries TWO `stream` flags: skapi's (relay the destination's bytes
329
+ * into the chunk table) and the DESTINATION's own field inside `data`, which
330
+ * BunnyQuery is the party that sets, because skapi relays bytes and knows no
331
+ * vendor. clientSecretRequest validates its params against a schema and KEEPS ONLY
332
+ * THE KEYS IN THAT SCHEMA, so an skapi-js predating the feature does not reject
333
+ * `stream` - it silently DROPS it. What ships is then the exact split
334
+ * chatStreamWiring exists to make impossible: the destination is asked to answer in
335
+ * SSE frames, skapi waits and stores the whole transcript on the row, and
336
+ * extractClaudeText / extractOpenAIText read a wall of `data: {...}` lines where a
337
+ * document should be and find no answer at all. Nothing throws and nothing logs;
338
+ * the user gets an empty reply on every single turn.
339
+ *
340
+ * It lives in the ENGINE rather than in each client because the two clients are
341
+ * diffed against each other and this is precisely the kind of predicate that forks:
342
+ * the widget must ask it (init() takes the EMBEDDER's instance, and an embed page
343
+ * pins its own skapi-js version, so `liveStreaming: true` is a REQUEST and this is
344
+ * what grants it) and agent.vue must ask it too (its instance is the repo's own, so
345
+ * only a stale node_modules or an unbuilt skapi-js can fail it - which is exactly
346
+ * the state a dev flipping the constant is most likely to be in, and a silently
347
+ * empty chat is the worst possible way to find out).
348
+ *
349
+ * Probed by the two public METHODS rather than by a version string, for two
350
+ * reasons. They ship in the same change as the `stream` key (one feature: the
351
+ * relay, the finalize that stores what to keep, and the read-back), so an SDK
352
+ * missing them is exactly the SDK that would drop the flag. And they are not merely
353
+ * a proxy for the capability, they ARE half of it - the engine needs finalize to
354
+ * store a streamed answer onto its row and stream to read an unfinalized one back,
355
+ * and streaming without either leaves every answer in the chunk table with nothing
356
+ * able to fetch it. There is no cheap DIRECT probe of the schema: the only way to
357
+ * learn that `stream` was dropped is to send a real request and read an empty
358
+ * answer, which is the bug itself.
359
+ */
360
+ export function skapiSupportsStreaming(sk: any): boolean {
361
+ return !!sk
362
+ && typeof sk.clientSecretRequestStream === 'function'
363
+ && typeof sk.clientSecretRequestFinalize === 'function';
364
+ }
365
+
123
366
  /** Spread helper: `{ ...pollOpt() }` adds `poll` only when configured. */
124
367
  export function pollOpt(): { poll?: number } {
125
368
  const p = _config?.poll;
@@ -20,7 +20,73 @@ function isTransientStatus(status: number): boolean {
20
20
  return status === 408 || status === 425 || status === 429 || status >= 500;
21
21
  }
22
22
 
23
+ /**
24
+ * True when a csr-poll answer is the STATUS ENVELOPE rather than a stored body.
25
+ *
26
+ * Duck-typed, because the engine does not import skapi-js and the SDK does not
27
+ * export its own copy. The rule is the SDK's (isPollEnvelope): a request that has
28
+ * a stored result hands that result back verbatim, every other state hands back
29
+ * `{ id, status, in_queue, ... }`. A finalized body is the caller's own content
30
+ * and can itself carry a `status` key (OpenAI's Responses object does), so the
31
+ * id/in_queue pair is demanded too: a provider body would have to reproduce all
32
+ * three to be mistaken for an envelope.
33
+ *
34
+ * It lives HERE, next to the error readers, rather than in session.ts, because
35
+ * both of the things that have to recognise an envelope (the settle that
36
+ * substitutes an assembled body for one, and the error readers below) must agree
37
+ * on what one is. It was written twice once; that is how the error readers came to
38
+ * look one level too shallow.
39
+ */
40
+ export function isCsrStatusEnvelope(res: any): boolean {
41
+ return !!res && typeof res === 'object' && !Array.isArray(res) &&
42
+ typeof res.status === 'string' && typeof res.id === 'string' && ('in_queue' in res);
43
+ }
44
+
45
+ /**
46
+ * The real error payload inside a FAILED csr-poll status envelope, or undefined
47
+ * when `input` is not one.
48
+ *
49
+ * THE FAILURE THIS PREVENTS, verbatim from the wire. A buffered turn that fails
50
+ * polls back as the worker's failed payload itself:
51
+ *
52
+ * { status_code: 401, body: { error: { type, message } }, truncated: false }
53
+ *
54
+ * ...which every predicate below reads correctly. A STREAMED turn that fails does
55
+ * not: the poller (client_secret_key_request_polling) cannot return the error
56
+ * early for a streamed row, because the chunks that arrived before the stream died
57
+ * have to come back in the same response, so it falls through and ships
58
+ *
59
+ * { id, status: 'failed', queue_name, in_queue, stream, chunks, last_seq,
60
+ * more, error: <the payload above> }
61
+ *
62
+ * The payload is one level deeper, and every predicate here looked at the top
63
+ * level: `response.error.message` is undefined on it, `response.status_code` is
64
+ * absent, and `response.status` is the string 'failed' rather than a number. So a
65
+ * wrong API key on a streamed turn read as "not an error, and no answer either",
66
+ * which the caller renders as "No text response received from AI provider": the
67
+ * one message that tells the user nothing.
68
+ *
69
+ * Unwrapping HERE, once, is what makes a streamed error and a buffered error take
70
+ * the same path through every reader below. Deliberately not recursive: the value
71
+ * inside an envelope is a provider payload, never another envelope, and a single
72
+ * unwrap cannot loop on a malformed one.
73
+ *
74
+ * A failed envelope with a NULL payload (the worker recorded no detail, or its
75
+ * spill could not be fetched) still yields an object, because the row's status is
76
+ * itself the fact: 'failed' with nothing attached must not read as a clean turn.
77
+ */
78
+ export function csrEnvelopeError(input: any): any {
79
+ if (!isCsrStatusEnvelope(input)) return undefined;
80
+ // Only a failure carries one. 'resolved' means the body IS the answer (the
81
+ // settle substitutes it), and 'cancelled' is settled by its own predicate
82
+ // upstream, which must keep seeing the envelope it recognises.
83
+ if (input.status !== 'failed') return undefined;
84
+ return input.error != null ? input.error : { message: 'The AI provider request failed.' };
85
+ }
86
+
23
87
  export function getErrorMessage(input: any): string {
88
+ var envErr = csrEnvelopeError(input);
89
+ if (envErr !== undefined) input = envErr;
24
90
  if (!input) return 'Something went wrong.';
25
91
  if (typeof input === 'string') return input;
26
92
  if (input.error && input.error.message) return input.error.message;
@@ -48,7 +114,11 @@ export function getErrorMessage(input: any): string {
48
114
  }
49
115
 
50
116
  export function isErrorResponseBody(response: any): boolean {
51
- if (!response || typeof response !== 'object') return false;
117
+ // A FAILED streamed row is an error however empty its payload turned out to
118
+ // be: the status is the fact. See csrEnvelopeError.
119
+ var envErr = csrEnvelopeError(response);
120
+ if (envErr !== undefined) response = envErr;
121
+ if (!response || typeof response !== 'object') return envErr !== undefined;
52
122
  if (typeof response.status_code === 'number' && response.status_code >= 400) return true;
53
123
  if (response.type === 'error') return true;
54
124
  if (response.error && (response.error.message || response.error.type)) return true;
@@ -78,6 +148,11 @@ export function isErrorResponseBody(response: any): boolean {
78
148
  // retryable after a token refresh — so 401 (auth) and 429/5xx (transient) are
79
149
  // intentionally NOT treated as non-retryable here.
80
150
  export function isNonRetryableRequestError(input: any): boolean {
151
+ // Same one-level unwrap as the readers above: a streamed turn's failure is
152
+ // wrapped in its poll envelope, and a retry gate that cannot see the payload
153
+ // would re-send a request the provider has already rejected as malformed.
154
+ var envErr = csrEnvelopeError(input);
155
+ if (envErr !== undefined) input = envErr;
81
156
  if (!input || typeof input !== 'object') return false;
82
157
 
83
158
  var status = typeof input.status_code === 'number' ? input.status_code
@@ -118,6 +193,12 @@ export function isNonRetryableRequestError(input: any): boolean {
118
193
  }
119
194
 
120
195
  export function isAuthExpiredError(input: any): boolean {
196
+ // A streamed turn whose bearer expired fails exactly like a buffered one, one
197
+ // level deeper. Without this unwrap the refresh-and-resend gate never fires on
198
+ // the streaming path, so an expiry that costs a buffered turn nothing costs a
199
+ // streamed turn the whole answer.
200
+ var envErr = csrEnvelopeError(input);
201
+ if (envErr !== undefined) input = envErr;
121
202
  if (!input) return false;
122
203
  var blobs: string[] = [];
123
204
  var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
@@ -156,6 +237,11 @@ export function isAuthExpiredError(input: any): boolean {
156
237
  * reduced to by getErrorMessage, because the view usually only keeps the text.
157
238
  */
158
239
  export function isProviderApiKeyError(input: any): boolean {
240
+ // Same unwrap. A wrong API key on a streamed turn is the exact case CRITICAL 2
241
+ // was reported for: without this the "check the project's key" affordance never
242
+ // appears on the platform the user is actually streaming with.
243
+ var envErr = csrEnvelopeError(input);
244
+ if (envErr !== undefined) input = envErr;
159
245
  if (!input) return false;
160
246
  var blobs: string[] = [];
161
247
  var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
@@ -6,6 +6,7 @@
6
6
  */
7
7
  import { extractClaudeText, extractOpenAIText, INDEXING_COMPLETE_MARKER, EMPTY_INDEXING_REPLY, getChatHistory, bgIndexingQueueName } from './requests';
8
8
  import { isErrorResponseBody, getErrorMessage } from './errors';
9
+ import { streamRecoveryEnabled } from './config';
9
10
  import { sanitizeAttachmentLinksForHistory } from './links';
10
11
  import type { ChatMessage } from './host';
11
12
 
@@ -702,8 +703,11 @@ export type MapHistoryOptions = {
702
703
  };
703
704
 
704
705
  export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'openai', opts: MapHistoryOptions) {
705
- var mapped: any[] = [], runningItemIds: string[] = [];
706
+ var mapped: any[] = [], runningItemIds: string[] = [], streamPendingItemIds: string[] = [];
706
707
  var extractAssistantText = platform === 'openai' ? extractOpenAIText : extractClaudeText;
708
+ // See the `isStreamPending` block below. Read ONCE per call rather than per
709
+ // item: it is a config lookup, and the answer cannot change mid-list.
710
+ var canRecoverStreams = streamRecoveryEnabled();
707
711
  var filtered = filterListByClearHorizon(list, opts.clearedAt);
708
712
  filtered.slice().reverse().forEach(function (item) {
709
713
  var requestBody = item && item.request_body;
@@ -727,6 +731,45 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
727
731
  ? ((typeof item.response_text === 'string' ? item.response_text : '').trim())
728
732
  : ((extractAssistantText(response) || '').trim() || ''));
729
733
  var isErrorResponse = !isPending && (isFailed || (!isCompact && isErrorResponseBody(response)));
734
+ // AN ANSWER THAT IS UNKNOWN, NOT EMPTY.
735
+ //
736
+ // A STREAMED turn stores nothing on its row: the relay appends the
737
+ // destination's bytes to the chunk table and settles the row with a status
738
+ // and no body, and only csr-finalize ever copies an answer onto the row. So a
739
+ // row that is terminal, carries no body and carries no error is not a turn
740
+ // that answered nothing: it is a turn nobody finalized. That happens for one
741
+ // ordinary reason: the row settled while no poll was attached (the tab was
742
+ // closed, a mobile browser discarded it, the device slept and the interval
743
+ // stopped), so the client that would have finalized it was not there.
744
+ //
745
+ // Read as "empty" (which is what the `assistantText` guard further down did),
746
+ // such a row produces NO assistant bubble at all and the answer is simply gone
747
+ // from the conversation, with every byte of it still sitting in the chunk
748
+ // table, reachable through clientSecretRequestStream. So it is marked instead,
749
+ // and the two things that act on the mark are the merge (an unknown answer
750
+ // never overwrites a known one) and the recovery (an unknown answer is
751
+ // resolved by reading the chunks back). See ChatMessage._streamPending.
752
+ //
753
+ // Deliberately narrow, so nothing that is genuinely an empty turn is caught:
754
+ // - only when this host streams AND can read chunks back (see
755
+ // streamRecoveryEnabled); a host that does neither behaves exactly as it
756
+ // does today, and one that cannot read them back would only trade a
757
+ // missing bubble for a permanently empty one;
758
+ // - never on the background queue. chatStreamWiring turns streaming OFF for
759
+ // every bg-queue turn (an indexing pass must not stream, and an attachment
760
+ // turn is left buffered on purpose), so a bg row with no body really did
761
+ // answer nothing;
762
+ // - never on a compact stub, whose body was withheld by the server rather
763
+ // than never stored;
764
+ // - 'resolved' only. A FAILED row already renders its error, which is the
765
+ // authoritative account of the turn (and, once csrEnvelopeError is in
766
+ // play, a truthful one). Its chunks are deliberately kept but not
767
+ // rendered. See _finalizeStreamedTurn.
768
+ var isStreamPending = canRecoverStreams && !isCompact && !isPending && !isCancelledItem
769
+ && !isErrorResponse && !item._isBgTask && !item._isOnBgQueue
770
+ && item.status === 'resolved'
771
+ && (item.response_body == null) && (item.error == null)
772
+ && !assistantText;
730
773
  // Record the completion marker, then STRIP it — both, and in that order.
731
774
  // Recording gives the display layer a structured signal instead of a substring
732
775
  // search over model prose. Stripping matches the live resolution path: without
@@ -810,6 +853,14 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
810
853
  if (serverItemId !== undefined) em._serverItemId = serverItemId;
811
854
  if (replyTs !== undefined) em._ts = replyTs;
812
855
  mapped.push(em);
856
+ } else if (isStreamPending) {
857
+ // The bubble stands for the turn so the merge has something to key on and
858
+ // the recovery has somewhere to write. Its content is UNKNOWN: empty here
859
+ // is the absence of an answer on the row, not the absence of an answer.
860
+ var sp: any = { role: 'assistant', content: '', _streamPending: true };
861
+ if (serverItemId !== undefined) { sp._serverItemId = serverItemId; streamPendingItemIds.push(serverItemId); }
862
+ if (replyTs !== undefined) sp._ts = replyTs;
863
+ mapped.push(sp);
813
864
  // `|| reportedComplete`: a pass whose ENTIRE answer was the completion token
814
865
  // strips down to an empty string, and the plain `assistantText` guard then
815
866
  // emitted no bubble at all — while the live path emitted one. The run read as
@@ -837,7 +888,54 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
837
888
  var ownerKey = chatCacheKey(opts.projectId, platform, opts.userId);
838
889
  for (var oi = 0; oi < mapped.length; oi++) mapped[oi]._ownerKey = ownerKey;
839
890
  }
840
- return { messages: mapped, runningItemIds: runningItemIds };
891
+ // `streamPendingItemIds` is additive: a consumer that only destructures
892
+ // { messages, runningItemIds } (agent.vue's own call site does) is unaffected.
893
+ // The engine drives recovery off the BUBBLES rather than this list (the merge
894
+ // runs between the two and can adopt a local answer onto one of them), so this
895
+ // is reporting, not the mechanism.
896
+ return { messages: mapped, runningItemIds: runningItemIds, streamPendingItemIds: streamPendingItemIds };
897
+ }
898
+
899
+ /**
900
+ * Let a LOCAL copy of a turn survive a page whose copy of it is
901
+ * AUTHORITATIVE-BUT-EMPTY. Mutates `incoming`; returns true when it took anything.
902
+ *
903
+ * THE FAILURE THIS PREVENTS. A streamed turn's row goes 'resolved' the moment the
904
+ * relay finishes, and its answer reaches the row only when csr-finalize stores it,
905
+ * one poll interval plus a round trip later. A first-page history refetch landing
906
+ * inside that window maps the row to a `_streamPending` bubble with no content, and
907
+ * the merge, which believes the server, throws away the local bubble holding the
908
+ * answer the reader is looking at. The window opens on EVERY streamed turn, and a
909
+ * refetch fires from visibilitychange, so it is not a corner case.
910
+ *
911
+ * The rule is the same one the recovery reads: an UNKNOWN answer never overwrites a
912
+ * KNOWN one. Where the local copy is still live (pending, or being painted into),
913
+ * its live-ness is adopted too: without it the merge would hand back a settled
914
+ * bubble the painter can no longer find (_liveTargetIndex wants isPending or
915
+ * _streaming) and that _turnAlreadyRendered would then read as already answered, so
916
+ * the settle would drop the real answer on the floor.
917
+ */
918
+ export function adoptLocalAnswerIntoPage(incoming: ChatMessage, local: ChatMessage): boolean {
919
+ if (!incoming || !local || !incoming._streamPending) return false;
920
+ if (incoming.role !== 'assistant' || local.role !== 'assistant') return false;
921
+ var hasText = typeof local.content === 'string' && local.content.length > 0;
922
+ var isLive = !!(local.isPending || local._streaming);
923
+ if (!hasText && !isLive) return false;
924
+ if (hasText) {
925
+ incoming.content = local.content;
926
+ // The answer is on screen, so nothing has to be read back. Whether the row
927
+ // itself ever gets a stored body is the live path's business (its finalize is
928
+ // in flight); if that fails, the NEXT load meets an empty row again and
929
+ // recovers it then.
930
+ incoming._streamPending = false;
931
+ }
932
+ // Carried whatever the text says: these are what keep a still-running turn
933
+ // reachable by the painter and by the settle.
934
+ if (local._localId !== undefined) incoming._localId = local._localId;
935
+ if (local.isPending) incoming.isPending = true;
936
+ if (local.isPendingInProcess) incoming.isPendingInProcess = true;
937
+ if (local._streaming) incoming._streaming = true;
938
+ return true;
841
939
  }
842
940
 
843
941
  /* ---- rescuing in-flight bubbles across a first-page refetch ---------------
@@ -886,6 +984,19 @@ export function shouldRescueInFlightMessage(m: ChatMessage, ctx: RescueDecisionC
886
984
  // it. Unconditional: applying the pending-assistant test here would delete the
887
985
  // user's message mid-upload whenever some other turn happened to be in flight.
888
986
  if (m._stageId) return true;
987
+ // A bubble being painted from a LIVE stream, before its dispatch has reported a
988
+ // server id. Same shape of reason as the staged bubble above: the page identifies
989
+ // turns by id, so nothing in it can stand for this one, and dropping it would
990
+ // leave the stream with no bubble to paint into (the painter finds its target by
991
+ // _serverItemId, and this bubble has none to be found by) - the turn would sit on
992
+ // "Thinking..." until it settled, with every relayed byte already spent.
993
+ //
994
+ // Gated on the missing id rather than on _streaming alone, and deliberately: once
995
+ // the id IS on the bubble, the tests below are right and the local copy really is
996
+ // redundant. The page carries the same turn as a pending placeholder with that id,
997
+ // the painter finds THAT one on its next paint (about a second later), and rescuing
998
+ // as well would put the same turn on screen twice.
999
+ if (m._streaming && !m._serverItemId) return true;
889
1000
  if (!m._serverItemId && ctx.pageHasPendingAssistant) return false;
890
1001
  // In flight by its own flags, id or no id. The immediate-send pair is stamped
891
1002
  // with its server id as soon as the dispatch reports one, and that id is there
@@ -130,6 +130,91 @@ export interface ChatMessage {
130
130
  * still dimmed. Cleared (with _dimSending) by markStagedMessageReady the moment
131
131
  * the queue drains, which is when the turn genuinely becomes "(In queue)". */
132
132
  isAwaitingIndexing?: boolean;
133
+ /** PROTOCOL + PRESENTATIONAL: this bubble is being painted from a LIVE stream.
134
+ *
135
+ * It sits alongside `isPending`, never instead of it. The bubble is still the
136
+ * turn's "Thinking..." placeholder as far as every queue mechanism is concerned
137
+ * (_ownThinkingIndex, resolveQueuedUserBubble, typewriteLatestReply and the
138
+ * stray-pending sweep all find their target by isPending), and clearing that flag
139
+ * to mean "it has text now" would strand the turn's real answer beside an orphan.
140
+ * What this adds is the one thing those mechanisms do not care about and the VIEW
141
+ * does: `content` is already worth rendering, so draw the text instead of the
142
+ * spinner.
143
+ *
144
+ * Cleared the moment the turn settles, BEFORE the authoritative answer replaces
145
+ * the live text: from that instant the bubble is an ordinary reply being typed.
146
+ * The partially painted `content` is deliberately left in place across that clear,
147
+ * because it is what the typewriter resumes from instead of replaying from zero.
148
+ *
149
+ * Also read by shouldRescueInFlightMessage: a streaming bubble that has no server
150
+ * id yet is unrepresentable in a freshly fetched page, so it must survive the
151
+ * merge or its stream is orphaned with nothing left to paint into. */
152
+ _streaming?: boolean;
153
+ /**
154
+ * This turn's row is TERMINAL but carries no stored answer, because the answer
155
+ * was streamed and nobody ever finalized it: the bytes are in the chunk store,
156
+ * not on the row. The bubble's `content` is therefore UNKNOWN, not empty.
157
+ *
158
+ * That distinction is the whole of the fix for two bugs that looked unrelated.
159
+ * A refetch landing in the window between the row going 'resolved' and finalize
160
+ * storing the body used to ERASE the answer off the screen (the server copy has
161
+ * no content, so the merge dropped the local bubble that did); and a turn that
162
+ * settled while no poll was attached (closed tab, slept device) used to be
163
+ * unrecoverable, because the mapper emitted no assistant bubble at all for a
164
+ * terminal-but-empty row. Both are the same question, "what should the merge
165
+ * believe when the server copy is authoritative but empty", and the answer is
166
+ * this flag: an unknown answer NEVER overwrites a known one, and an unknown one
167
+ * left over after the merge is resolved by reading the chunks back
168
+ * (ChatSession.recoverStreamedAnswer).
169
+ *
170
+ * For a view it is a rendering hint and nothing more: a bubble carrying it with
171
+ * empty content is being fetched, so draw whatever this client draws for a
172
+ * loading answer. A host that ignores it renders an empty bubble for the second
173
+ * or two the recovery takes, which is what it would have rendered anyway.
174
+ *
175
+ * WHEN IT COMES OFF, because that is the half that loses answers. It comes off
176
+ * for a FACT about the turn and never for an event in the client: an answer was
177
+ * recovered and written in, or the chunks were read and were genuinely empty (in
178
+ * which case the empty bubble is removed as well, restoring the list the mapper
179
+ * used to produce). It stays ON when the read FAILED, when the read was STOPPED,
180
+ * and when a live turn settled having painted nothing - three states that say
181
+ * nothing whatever about the turn, and in which the chunks are all still there.
182
+ * A marker cleared on one of those is an answer nothing will ever go back for,
183
+ * so a host may see the same bubble marked across several loads while the reads
184
+ * keep failing; that is the recoverable state, not a stuck one.
185
+ */
186
+ _streamPending?: boolean;
187
+ /**
188
+ * IS ANYTHING ACTUALLY DRIVING THIS BUBBLE RIGHT NOW? The second half of
189
+ * `_streamPending`, and the half a view cannot do without.
190
+ *
191
+ * `_streamPending` says the answer is elsewhere; it does NOT say somebody is on
192
+ * their way to fetch it, and the two are different states that used to render
193
+ * identically. Recovery is capped per history load (STREAM_RECOVERY_PER_LOAD),
194
+ * so the third and later marked turns on a page are marked and nobody is reading
195
+ * them; a read that FAILED leaves the marker on with the attempt forgotten, which
196
+ * is also nobody. Both drew the same loader as a live turn, so a bubble could
197
+ * spin for the rest of the session with nothing behind it - the one thing a
198
+ * spinner must never do, because it is a promise that something is coming.
199
+ *
200
+ * Three states, and only the first of them may draw a spinner:
201
+ * 'active' a chunk read is in flight or queued for this turn. Something IS
202
+ * coming; the loader is honest.
203
+ * 'failed' the last read failed. Nothing is coming until somebody asks
204
+ * again, so the view owes the reader a way to ask.
205
+ * undefined nothing has been tried, or the attempt is over. Same obligation.
206
+ *
207
+ * Written by the engine only, and never persisted anywhere: it describes THIS
208
+ * session's fetching, not the turn. A fresh history page therefore arrives
209
+ * without it, and _adoptLocalAnswers re-stamps the page's still-marked bubbles
210
+ * from the session's own bookkeeping, so a reload during a read does not turn a
211
+ * live loader into a button (and back a second later).
212
+ *
213
+ * Read it through streamRecoveryPhase(msg), never directly: the phase folds in
214
+ * "does this bubble need the affordance at all", and both clients must not
215
+ * answer that twice.
216
+ */
217
+ _streamRecovery?: 'active' | 'failed';
133
218
  _serverItemId?: string;
134
219
  _localId?: string;
135
220
  _cancelling?: boolean;