bunnyquery 1.9.7 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -22
- package/bunnyquery.css +53 -2
- package/bunnyquery.js +1880 -197
- package/dist/engine.cjs +1806 -133
- package/dist/engine.cjs.map +1 -1
- package/dist/engine.d.mts +1335 -11
- package/dist/engine.d.ts +1335 -11
- package/dist/engine.mjs +1765 -134
- package/dist/engine.mjs.map +1 -1
- package/package.json +1 -1
- package/src/engine/budget.ts +46 -2
- package/src/engine/config.ts +243 -0
- package/src/engine/errors.ts +87 -1
- package/src/engine/history.ts +113 -2
- package/src/engine/host.ts +85 -0
- package/src/engine/index.ts +111 -2
- package/src/engine/office.ts +16 -4
- package/src/engine/project_settings.ts +303 -0
- package/src/engine/prompts/chat_system_prompt.ts +21 -6
- package/src/engine/prompts/indexing_system_prompt.ts +8 -4
- package/src/engine/prompts/indexing_user_message.ts +39 -3
- package/src/engine/requests.ts +203 -8
- package/src/engine/session.ts +1553 -29
- package/src/engine/sse.ts +1054 -0
- package/src/widget.css +21 -2
- package/styles/chat.css +32 -0
package/package.json
CHANGED
package/src/engine/budget.ts
CHANGED
|
@@ -140,6 +140,40 @@ export function getProjectContextWindow(projectId: string): number | null {
|
|
|
140
140
|
// actual request cannot drift: they were 22000 and 25000 respectively, which
|
|
141
141
|
// under-reserved by 3k on a window spent to the last token.
|
|
142
142
|
export var MAX_OUTPUT_TOKENS = 25000;
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* The same ceiling for an INDEXING pass, which is a different job with a different shape.
|
|
146
|
+
*
|
|
147
|
+
* A chat turn has a person waiting, so a long reply is a worse outcome than a truncated
|
|
148
|
+
* one and 25,000 is generous for it. An indexing pass has nobody waiting and one job:
|
|
149
|
+
* emit records for the window it was shown. When it runs out of budget mid-window the
|
|
150
|
+
* worker halves the window and re-sends it (`window_scale 1.0 -> 0.5`), so the file pays
|
|
151
|
+
* roughly twice the passes for the rest of its length.
|
|
152
|
+
*
|
|
153
|
+
* WHY 64,000, measured on a live 465-row spreadsheet (9 passes, project ap21U8y5byIbkGSv):
|
|
154
|
+
* - gpt-5.6-luna allows 128,000 output tokens, so 25,000 was 19.5% of what it permits.
|
|
155
|
+
* - Only 1 of those 9 passes hit the cap; the median used 6,370. A cap is a CEILING, not
|
|
156
|
+
* a target: the model stops when it is done, so raising it costs nothing on the eight
|
|
157
|
+
* passes that never approach it. It only changes the one that would have truncated.
|
|
158
|
+
* - Fitting duration against output tokens across the nine: `55.4s + output / 131 tok/s`
|
|
159
|
+
* (R^2 0.906). The binding constraint is TIME, not the model.
|
|
160
|
+
* - The worker's upstream timeout is 870s and its Lambda is 900s with a 30s settle
|
|
161
|
+
* reserve, which puts the theoretical ceiling near 103,000 tokens. 64,000 lands at
|
|
162
|
+
* about 544s and leaves roughly 300s of margin for a slow provider hour or a pass with
|
|
163
|
+
* more tool round trips. Spending that margin buys nothing: 64,000 is already enough
|
|
164
|
+
* for a full 250-row window to finish in one pass, which is the whole point.
|
|
165
|
+
*
|
|
166
|
+
* It does NOT feed OUTPUT_TOKEN_RESERVE below. That reserve exists to size the INPUT
|
|
167
|
+
* budget for buildBoundedChatMessages, which only the chat path calls; an indexing message
|
|
168
|
+
* is built from a file window, not from bounded history. Wiring this into the reserve would
|
|
169
|
+
* shrink a budget this number has nothing to do with.
|
|
170
|
+
*
|
|
171
|
+
* Re-derive rather than nudge: re-run the fit if the model, the Lambda timeout or
|
|
172
|
+
* REQUEST_TIMEOUT changes, and note that a model whose own ceiling is lower still wins
|
|
173
|
+
* (getMaxOutputTokens clamps, so gpt-4o stays at its 4,000).
|
|
174
|
+
*/
|
|
175
|
+
export var INDEXING_MAX_OUTPUT_TOKENS = 64000;
|
|
176
|
+
|
|
143
177
|
export var OUTPUT_TOKEN_RESERVE = MAX_OUTPUT_TOKENS;
|
|
144
178
|
export var TOOL_AND_RESPONSE_BUFFER = 4000;
|
|
145
179
|
export var MIN_INPUT_TOKEN_BUDGET = 8000;
|
|
@@ -225,9 +259,19 @@ export function getModelContextWindow(platform: string, model?: string): number
|
|
|
225
259
|
* but a model whose own cap is lower rejects the request outright, so clamp to
|
|
226
260
|
* whichever is smaller. Models with no known cap keep MAX_OUTPUT_TOKENS.
|
|
227
261
|
*/
|
|
228
|
-
export function getMaxOutputTokens(
|
|
262
|
+
export function getMaxOutputTokens(
|
|
263
|
+
platform: string,
|
|
264
|
+
model?: string,
|
|
265
|
+
/** 'indexing' asks for INDEXING_MAX_OUTPUT_TOKENS instead. Omitted means chat, so every
|
|
266
|
+
* existing caller keeps the number it had. */
|
|
267
|
+
purpose?: 'chat' | 'indexing',
|
|
268
|
+
): number {
|
|
269
|
+
var want = purpose === 'indexing' ? INDEXING_MAX_OUTPUT_TOKENS : MAX_OUTPUT_TOKENS;
|
|
270
|
+
// The model's own ceiling still wins. This is the whole reason the clamp lives here and
|
|
271
|
+
// not at the call sites: gpt-4o caps output at 4,000 and legacy 3.5 Sonnet at 8,000, and
|
|
272
|
+
// asking either for 64,000 is a rejected request, not a long answer.
|
|
229
273
|
var cap = resolveByModelId(apiReportedMaxOutput, MAX_OUTPUT_BY_MODEL, model);
|
|
230
|
-
return cap ? Math.min(
|
|
274
|
+
return cap ? Math.min(want, cap) : want;
|
|
231
275
|
}
|
|
232
276
|
|
|
233
277
|
/**
|
package/src/engine/config.ts
CHANGED
|
@@ -15,6 +15,52 @@
|
|
|
15
15
|
|
|
16
16
|
import { type AttachmentParser, registerAttachmentParser } from './attachment_parsers';
|
|
17
17
|
|
|
18
|
+
/**
|
|
19
|
+
* One report about a turn that is streaming, as handed to `onLiveStreamUpdate`.
|
|
20
|
+
*
|
|
21
|
+
* Deliberately a flat snapshot rather than the SseParser itself: the hook is a
|
|
22
|
+
* VIEW seam, and handing a client the parser would invite it to drive the stream
|
|
23
|
+
* (feed it, end it, read the assembled body) behind the session's back.
|
|
24
|
+
*/
|
|
25
|
+
export interface LiveStreamUpdate {
|
|
26
|
+
/** Server item id of the turn, the same id its bubbles carry as _serverItemId. */
|
|
27
|
+
serverItemId: string;
|
|
28
|
+
/** History cache key (`projectId#platform`) the turn belongs to. A host that
|
|
29
|
+
* renders several projects must ignore an update for a chat it is not showing. */
|
|
30
|
+
ownerKey: string;
|
|
31
|
+
/** 'start' on the first paint, 'update' on every later one, 'end' once the
|
|
32
|
+
* stream is over and nothing more will be painted - whether because the turn
|
|
33
|
+
* settled (its authoritative answer is about to replace the live text) or
|
|
34
|
+
* because it was stopped. 'end' is only sent to a host that was told 'start'. */
|
|
35
|
+
phase: 'start' | 'update' | 'end';
|
|
36
|
+
/** Answer text so far, already trimmed to a safe reveal boundary: it never
|
|
37
|
+
* ends inside a half-arrived link, fence or url. Empty on 'end'. */
|
|
38
|
+
text: string;
|
|
39
|
+
/** Extended-thinking text so far. Separate from `text` and never part of it. */
|
|
40
|
+
thinkingText: string;
|
|
41
|
+
/** Tools reached for, in order of appearance, duplicates kept. */
|
|
42
|
+
toolNames: string[];
|
|
43
|
+
/** A terminal event arrived. False on 'end' means the stream was cut. */
|
|
44
|
+
complete: boolean;
|
|
45
|
+
/** The terminal event that arrived meant the answer FINISHED rather than DIED:
|
|
46
|
+
* false while running, false on a cut stream, and false when the stream ended
|
|
47
|
+
* on a provider error. `complete` answers "is anything more coming?"; this one
|
|
48
|
+
* answers "is this the whole answer?", and they differ on exactly the case
|
|
49
|
+
* that costs text - an `error` frame is terminal and truncating at once. A host
|
|
50
|
+
* drawing a "partial answer" affordance wants THIS one. Added after `complete`
|
|
51
|
+
* and always present: a host that ignores it reads as it did before. */
|
|
52
|
+
answerComplete: boolean;
|
|
53
|
+
/** The stream ended in a provider error. */
|
|
54
|
+
errored: boolean;
|
|
55
|
+
/** How many chunks of this turn each of skapi's two transports carried FIRST:
|
|
56
|
+
* `socket` for the websocket relay, `poll` for the chunk-table read. Both feed
|
|
57
|
+
* the same sink by design, so a turn that streamed perfectly over the socket
|
|
58
|
+
* and one that was polled the whole way are otherwise indistinguishable. A
|
|
59
|
+
* host that does not care can ignore it; a host showing a live/degraded
|
|
60
|
+
* indicator, or just logging which path it got, reads this. */
|
|
61
|
+
transport: { socket: number; poll: number };
|
|
62
|
+
}
|
|
63
|
+
|
|
18
64
|
export interface ChatEngineConfig {
|
|
19
65
|
/** skapi.clientSecretRequest, bound to the consumer's skapi instance. */
|
|
20
66
|
clientSecretRequest: (opts: any) => Promise<any>;
|
|
@@ -86,6 +132,116 @@ export interface ChatEngineConfig {
|
|
|
86
132
|
queue?: string;
|
|
87
133
|
};
|
|
88
134
|
}) => void;
|
|
135
|
+
/**
|
|
136
|
+
* Opt in to LIVE STREAMING of chat turns.
|
|
137
|
+
*
|
|
138
|
+
* Off by default, and for the same shipping-order reason `windowedIndexing`
|
|
139
|
+
* is: THE BACKEND MUST SHIP FIRST. When on, every chat turn carries two
|
|
140
|
+
* `stream` flags (see requests.ts chatStreamWiring) and the polling row
|
|
141
|
+
* settles with a STATUS AND NO BODY, because the answer was the stream. On a
|
|
142
|
+
* region whose polling worker does not relay, that same request either has
|
|
143
|
+
* its unknown `since` cursor rejected or stores an SSE transcript where the
|
|
144
|
+
* readers expect a parsed document, and the turn reads back as an empty
|
|
145
|
+
* answer. So it stays off until the worker is deployed, then flips per
|
|
146
|
+
* environment.
|
|
147
|
+
*
|
|
148
|
+
* It also needs `clientSecretRequestFinalize` below: without it a streamed
|
|
149
|
+
* turn is never finalized, so its row keeps a status and no body forever and
|
|
150
|
+
* a later history load shows the question with an empty answer.
|
|
151
|
+
*/
|
|
152
|
+
liveStreaming?: boolean;
|
|
153
|
+
/**
|
|
154
|
+
* Also push each relayed chunk over skapi's websocket, so text lands as it is
|
|
155
|
+
* relayed instead of on the next poll tick. Requires `liveStreaming`.
|
|
156
|
+
*
|
|
157
|
+
* SEPARATE FROM `liveStreaming` ON PURPOSE, and off unless a host asks. It is a
|
|
158
|
+
* pure accelerator with a safe fallback, so the reason is not risk to the chat, it
|
|
159
|
+
* is what it does to the HOST'S OWN realtime: skapi's joinRealtime REPLACES the
|
|
160
|
+
* connection's group rather than adding to it, so for the length of a turn this
|
|
161
|
+
* takes the room. The dashboard owns its skapi instance and uses realtime for
|
|
162
|
+
* nothing else, so it opts in. The embeddable widget is handed the EMBEDDER'S
|
|
163
|
+
* instance and cannot know what their app does with it, so it stays off there
|
|
164
|
+
* unless the embedder turns it on.
|
|
165
|
+
*/
|
|
166
|
+
liveStreamingRealtime?: boolean;
|
|
167
|
+
/**
|
|
168
|
+
* skapi.clientSecretRequestFinalize, bound to the consumer's skapi instance.
|
|
169
|
+
* Stores the version of a streamed turn that history should keep (the engine
|
|
170
|
+
* sends the ASSEMBLED provider body, so history reads it exactly as it reads
|
|
171
|
+
* a buffered turn) and releases that request's chunks. Optional: a host
|
|
172
|
+
* without it can still stream, it just leaves the chunks and an empty row.
|
|
173
|
+
*/
|
|
174
|
+
clientSecretRequestFinalize?: (
|
|
175
|
+
requestId: string,
|
|
176
|
+
data: any,
|
|
177
|
+
options: { url: string; method: string; service?: string; owner?: string },
|
|
178
|
+
) => Promise<any>;
|
|
179
|
+
/**
|
|
180
|
+
* skapi.clientSecretRequestStream, bound to the consumer's skapi instance.
|
|
181
|
+
*
|
|
182
|
+
* THE SECOND HALF OF THE DURABILITY GUARANTEE, and without it a streamed turn
|
|
183
|
+
* is only as durable as the tab that started it. A streamed row settles with a
|
|
184
|
+
* status and NO body; the answer is stored as chunks until
|
|
185
|
+
* clientSecretRequestFinalize says what to keep. A row that settles while no
|
|
186
|
+
* poll is attached (the user closed the tab, a mobile browser discarded it,
|
|
187
|
+
* the device slept and the interval stopped) is therefore never finalized, and
|
|
188
|
+
* a later history load sees a terminal row with no body and used to emit no
|
|
189
|
+
* assistant bubble at all: the answer simply gone from the conversation, with
|
|
190
|
+
* every byte of it still sitting in the chunk table.
|
|
191
|
+
*
|
|
192
|
+
* This is the documented way back to it. Given the request id it fetches every
|
|
193
|
+
* chunk of an already-finished turn in one pass (paging internally on `more`)
|
|
194
|
+
* and delivers them in order through `onStream`, then resolves. The engine
|
|
195
|
+
* feeds those into a fresh SSE parser and treats the assembled body exactly as
|
|
196
|
+
* it treats a live one, including finalizing it, which stores the answer as
|
|
197
|
+
* ordinary history and releases the chunks, so each row is recovered at most
|
|
198
|
+
* once ever.
|
|
199
|
+
*
|
|
200
|
+
* Optional. Without it the engine still marks such turns (`_streamPending` on
|
|
201
|
+
* the bubble) but has no way to read them back, so a host that ignores this
|
|
202
|
+
* behaves as it does today.
|
|
203
|
+
*
|
|
204
|
+
* NOTE THAT THIS HOOK, NOT `liveStreaming`, IS WHAT ARMS RECOVERY. See
|
|
205
|
+
* streamRecoveryEnabled() below for why the two decisions are separate.
|
|
206
|
+
*/
|
|
207
|
+
clientSecretRequestStream?: (
|
|
208
|
+
requestId: string,
|
|
209
|
+
options: {
|
|
210
|
+
url: string;
|
|
211
|
+
method: string;
|
|
212
|
+
onStream?: (chunk: string, seq: number) => void;
|
|
213
|
+
since?: number;
|
|
214
|
+
poll?: number;
|
|
215
|
+
service?: string;
|
|
216
|
+
owner?: string;
|
|
217
|
+
},
|
|
218
|
+
) => Promise<any>;
|
|
219
|
+
/**
|
|
220
|
+
* Observation hook for a live-streaming turn, called at most once per paint
|
|
221
|
+
* (about once a second) plus once when the turn settles.
|
|
222
|
+
*
|
|
223
|
+
* The engine already paints the answer text into the pending bubble itself,
|
|
224
|
+
* so a host needs this ONLY for the affordances the engine deliberately does
|
|
225
|
+
* not decide the presentation of: a "thinking..." line, or a "querying sales
|
|
226
|
+
* table..." row drawn from the tools the model reached for before any answer
|
|
227
|
+
* text exists. Optional, and a host without it behaves exactly as today.
|
|
228
|
+
*
|
|
229
|
+
* Never throw from it: it is called on the paint path and a throw would cost
|
|
230
|
+
* the user the rest of their answer. The engine guards it anyway.
|
|
231
|
+
*/
|
|
232
|
+
onLiveStreamUpdate?: (update: LiveStreamUpdate) => void;
|
|
233
|
+
/**
|
|
234
|
+
* Force the read-back of already-streamed turns OFF, even though the chunk
|
|
235
|
+
* reader is injected.
|
|
236
|
+
*
|
|
237
|
+
* There is no need to set it to turn recovery ON: injecting
|
|
238
|
+
* `clientSecretRequestStream` is what arms it (see streamRecoveryEnabled).
|
|
239
|
+
* This exists only as the way back out for a host that wants byte-for-byte the
|
|
240
|
+
* pre-recovery rendering of a terminal-but-empty row - no bubble, no marker, no
|
|
241
|
+
* chunk read - while keeping the reader available for its own use. Omit it and
|
|
242
|
+
* nothing changes.
|
|
243
|
+
*/
|
|
244
|
+
streamRecovery?: boolean;
|
|
89
245
|
/**
|
|
90
246
|
* Single-item csr-poll point lookup (skapi.util.request('csr-poll', {id,
|
|
91
247
|
* service, owner}, {auth:true})). For a RESOLVED item the backend returns
|
|
@@ -120,6 +276,93 @@ export function windowedIndexingEnabled(): boolean {
|
|
|
120
276
|
return _config?.windowedIndexing === true;
|
|
121
277
|
}
|
|
122
278
|
|
|
279
|
+
/** True when the consumer has opted in to live streaming of chat turns. */
|
|
280
|
+
export function liveStreamingRealtimeEnabled(): boolean {
|
|
281
|
+
return liveStreamingEnabled() && _config?.liveStreamingRealtime === true;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
export function liveStreamingEnabled(): boolean {
|
|
285
|
+
return _config?.liveStreaming === true;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/**
|
|
289
|
+
* True when a streamed turn whose answer never reached its row can be READ BACK.
|
|
290
|
+
*
|
|
291
|
+
* IT ASKS FOR THE READER AND NOT FOR `liveStreaming`, AND THAT SPLIT IS THE WHOLE
|
|
292
|
+
* POINT. "Should NEW turns stream?" and "can an ALREADY streamed row be recovered?"
|
|
293
|
+
* are two different questions about two different sets of rows, and answering both
|
|
294
|
+
* with one flag strands the second set the moment the first answer changes.
|
|
295
|
+
*
|
|
296
|
+
* The failure, and it is not hypothetical - it is what turning the feature off
|
|
297
|
+
* does. A row streamed yesterday holds its answer in the chunk table and a status
|
|
298
|
+
* and no body on the row; only csr-finalize ever copies one onto it. Flip
|
|
299
|
+
* `liveStreaming` off today (an embedder drops the option, a dev rolls the flag
|
|
300
|
+
* back after a bad deploy, a client's skapiSupportsStreaming probe degrades the
|
|
301
|
+
* instance to buffered) and every one of those rows instantly becomes unmarked,
|
|
302
|
+
* unrecoverable and unreadable: the mapper emits no bubble for it, the recovery
|
|
303
|
+
* never looks at it, and its answer is unreachable with every byte of it still
|
|
304
|
+
* stored. Rolling a rendering flag back must not delete anybody's history.
|
|
305
|
+
*
|
|
306
|
+
* The reader is the honest test because it is the CAPABILITY the recovery needs.
|
|
307
|
+
* Without it the engine could mark such a turn and never fill it in, trading a
|
|
308
|
+
* missing bubble for a permanently empty one, which is strictly worse than the
|
|
309
|
+
* bug - so the marker is still only ever minted when something can act on it.
|
|
310
|
+
*
|
|
311
|
+
* The cost of asking the wider question is one wasted read, once, on a row that
|
|
312
|
+
* was terminal and empty for some reason other than streaming (a buffered turn
|
|
313
|
+
* whose body the worker never managed to spill). That read finds no chunks, the
|
|
314
|
+
* bubble is dropped, and the list looks exactly as it did before. Set
|
|
315
|
+
* `streamRecovery: false` to opt out of even that.
|
|
316
|
+
*/
|
|
317
|
+
export function streamRecoveryEnabled(): boolean {
|
|
318
|
+
if (_config?.streamRecovery === false) return false;
|
|
319
|
+
return typeof _config?.clientSecretRequestStream === 'function';
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* Does a given skapi INSTANCE support the streaming half of the protocol? Ask this
|
|
324
|
+
* before honouring a `liveStreaming: true` opt-in, and degrade to buffered when the
|
|
325
|
+
* answer is no.
|
|
326
|
+
*
|
|
327
|
+
* THE FAILURE THIS PREVENTS, and it is the one the SDK's own docs call quiet. A
|
|
328
|
+
* streamed turn carries TWO `stream` flags: skapi's (relay the destination's bytes
|
|
329
|
+
* into the chunk table) and the DESTINATION's own field inside `data`, which
|
|
330
|
+
* BunnyQuery is the party that sets, because skapi relays bytes and knows no
|
|
331
|
+
* vendor. clientSecretRequest validates its params against a schema and KEEPS ONLY
|
|
332
|
+
* THE KEYS IN THAT SCHEMA, so an skapi-js predating the feature does not reject
|
|
333
|
+
* `stream` - it silently DROPS it. What ships is then the exact split
|
|
334
|
+
* chatStreamWiring exists to make impossible: the destination is asked to answer in
|
|
335
|
+
* SSE frames, skapi waits and stores the whole transcript on the row, and
|
|
336
|
+
* extractClaudeText / extractOpenAIText read a wall of `data: {...}` lines where a
|
|
337
|
+
* document should be and find no answer at all. Nothing throws and nothing logs;
|
|
338
|
+
* the user gets an empty reply on every single turn.
|
|
339
|
+
*
|
|
340
|
+
* It lives in the ENGINE rather than in each client because the two clients are
|
|
341
|
+
* diffed against each other and this is precisely the kind of predicate that forks:
|
|
342
|
+
* the widget must ask it (init() takes the EMBEDDER's instance, and an embed page
|
|
343
|
+
* pins its own skapi-js version, so `liveStreaming: true` is a REQUEST and this is
|
|
344
|
+
* what grants it) and agent.vue must ask it too (its instance is the repo's own, so
|
|
345
|
+
* only a stale node_modules or an unbuilt skapi-js can fail it - which is exactly
|
|
346
|
+
* the state a dev flipping the constant is most likely to be in, and a silently
|
|
347
|
+
* empty chat is the worst possible way to find out).
|
|
348
|
+
*
|
|
349
|
+
* Probed by the two public METHODS rather than by a version string, for two
|
|
350
|
+
* reasons. They ship in the same change as the `stream` key (one feature: the
|
|
351
|
+
* relay, the finalize that stores what to keep, and the read-back), so an SDK
|
|
352
|
+
* missing them is exactly the SDK that would drop the flag. And they are not merely
|
|
353
|
+
* a proxy for the capability, they ARE half of it - the engine needs finalize to
|
|
354
|
+
* store a streamed answer onto its row and stream to read an unfinalized one back,
|
|
355
|
+
* and streaming without either leaves every answer in the chunk table with nothing
|
|
356
|
+
* able to fetch it. There is no cheap DIRECT probe of the schema: the only way to
|
|
357
|
+
* learn that `stream` was dropped is to send a real request and read an empty
|
|
358
|
+
* answer, which is the bug itself.
|
|
359
|
+
*/
|
|
360
|
+
export function skapiSupportsStreaming(sk: any): boolean {
|
|
361
|
+
return !!sk
|
|
362
|
+
&& typeof sk.clientSecretRequestStream === 'function'
|
|
363
|
+
&& typeof sk.clientSecretRequestFinalize === 'function';
|
|
364
|
+
}
|
|
365
|
+
|
|
123
366
|
/** Spread helper: `{ ...pollOpt() }` adds `poll` only when configured. */
|
|
124
367
|
export function pollOpt(): { poll?: number } {
|
|
125
368
|
const p = _config?.poll;
|
package/src/engine/errors.ts
CHANGED
|
@@ -20,7 +20,73 @@ function isTransientStatus(status: number): boolean {
|
|
|
20
20
|
return status === 408 || status === 425 || status === 429 || status >= 500;
|
|
21
21
|
}
|
|
22
22
|
|
|
23
|
+
/**
|
|
24
|
+
* True when a csr-poll answer is the STATUS ENVELOPE rather than a stored body.
|
|
25
|
+
*
|
|
26
|
+
* Duck-typed, because the engine does not import skapi-js and the SDK does not
|
|
27
|
+
* export its own copy. The rule is the SDK's (isPollEnvelope): a request that has
|
|
28
|
+
* a stored result hands that result back verbatim, every other state hands back
|
|
29
|
+
* `{ id, status, in_queue, ... }`. A finalized body is the caller's own content
|
|
30
|
+
* and can itself carry a `status` key (OpenAI's Responses object does), so the
|
|
31
|
+
* id/in_queue pair is demanded too: a provider body would have to reproduce all
|
|
32
|
+
* three to be mistaken for an envelope.
|
|
33
|
+
*
|
|
34
|
+
* It lives HERE, next to the error readers, rather than in session.ts, because
|
|
35
|
+
* both of the things that have to recognise an envelope (the settle that
|
|
36
|
+
* substitutes an assembled body for one, and the error readers below) must agree
|
|
37
|
+
* on what one is. It was written twice once; that is how the error readers came to
|
|
38
|
+
* look one level too shallow.
|
|
39
|
+
*/
|
|
40
|
+
export function isCsrStatusEnvelope(res: any): boolean {
|
|
41
|
+
return !!res && typeof res === 'object' && !Array.isArray(res) &&
|
|
42
|
+
typeof res.status === 'string' && typeof res.id === 'string' && ('in_queue' in res);
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* The real error payload inside a FAILED csr-poll status envelope, or undefined
|
|
47
|
+
* when `input` is not one.
|
|
48
|
+
*
|
|
49
|
+
* THE FAILURE THIS PREVENTS, verbatim from the wire. A buffered turn that fails
|
|
50
|
+
* polls back as the worker's failed payload itself:
|
|
51
|
+
*
|
|
52
|
+
* { status_code: 401, body: { error: { type, message } }, truncated: false }
|
|
53
|
+
*
|
|
54
|
+
* ...which every predicate below reads correctly. A STREAMED turn that fails does
|
|
55
|
+
* not: the poller (client_secret_key_request_polling) cannot return the error
|
|
56
|
+
* early for a streamed row, because the chunks that arrived before the stream died
|
|
57
|
+
* have to come back in the same response, so it falls through and ships
|
|
58
|
+
*
|
|
59
|
+
* { id, status: 'failed', queue_name, in_queue, stream, chunks, last_seq,
|
|
60
|
+
* more, error: <the payload above> }
|
|
61
|
+
*
|
|
62
|
+
* The payload is one level deeper, and every predicate here looked at the top
|
|
63
|
+
* level: `response.error.message` is undefined on it, `response.status_code` is
|
|
64
|
+
* absent, and `response.status` is the string 'failed' rather than a number. So a
|
|
65
|
+
* wrong API key on a streamed turn read as "not an error, and no answer either",
|
|
66
|
+
* which the caller renders as "No text response received from AI provider": the
|
|
67
|
+
* one message that tells the user nothing.
|
|
68
|
+
*
|
|
69
|
+
* Unwrapping HERE, once, is what makes a streamed error and a buffered error take
|
|
70
|
+
* the same path through every reader below. Deliberately not recursive: the value
|
|
71
|
+
* inside an envelope is a provider payload, never another envelope, and a single
|
|
72
|
+
* unwrap cannot loop on a malformed one.
|
|
73
|
+
*
|
|
74
|
+
* A failed envelope with a NULL payload (the worker recorded no detail, or its
|
|
75
|
+
* spill could not be fetched) still yields an object, because the row's status is
|
|
76
|
+
* itself the fact: 'failed' with nothing attached must not read as a clean turn.
|
|
77
|
+
*/
|
|
78
|
+
export function csrEnvelopeError(input: any): any {
|
|
79
|
+
if (!isCsrStatusEnvelope(input)) return undefined;
|
|
80
|
+
// Only a failure carries one. 'resolved' means the body IS the answer (the
|
|
81
|
+
// settle substitutes it), and 'cancelled' is settled by its own predicate
|
|
82
|
+
// upstream, which must keep seeing the envelope it recognises.
|
|
83
|
+
if (input.status !== 'failed') return undefined;
|
|
84
|
+
return input.error != null ? input.error : { message: 'The AI provider request failed.' };
|
|
85
|
+
}
|
|
86
|
+
|
|
23
87
|
export function getErrorMessage(input: any): string {
|
|
88
|
+
var envErr = csrEnvelopeError(input);
|
|
89
|
+
if (envErr !== undefined) input = envErr;
|
|
24
90
|
if (!input) return 'Something went wrong.';
|
|
25
91
|
if (typeof input === 'string') return input;
|
|
26
92
|
if (input.error && input.error.message) return input.error.message;
|
|
@@ -48,7 +114,11 @@ export function getErrorMessage(input: any): string {
|
|
|
48
114
|
}
|
|
49
115
|
|
|
50
116
|
export function isErrorResponseBody(response: any): boolean {
|
|
51
|
-
|
|
117
|
+
// A FAILED streamed row is an error however empty its payload turned out to
|
|
118
|
+
// be: the status is the fact. See csrEnvelopeError.
|
|
119
|
+
var envErr = csrEnvelopeError(response);
|
|
120
|
+
if (envErr !== undefined) response = envErr;
|
|
121
|
+
if (!response || typeof response !== 'object') return envErr !== undefined;
|
|
52
122
|
if (typeof response.status_code === 'number' && response.status_code >= 400) return true;
|
|
53
123
|
if (response.type === 'error') return true;
|
|
54
124
|
if (response.error && (response.error.message || response.error.type)) return true;
|
|
@@ -78,6 +148,11 @@ export function isErrorResponseBody(response: any): boolean {
|
|
|
78
148
|
// retryable after a token refresh — so 401 (auth) and 429/5xx (transient) are
|
|
79
149
|
// intentionally NOT treated as non-retryable here.
|
|
80
150
|
export function isNonRetryableRequestError(input: any): boolean {
|
|
151
|
+
// Same one-level unwrap as the readers above: a streamed turn's failure is
|
|
152
|
+
// wrapped in its poll envelope, and a retry gate that cannot see the payload
|
|
153
|
+
// would re-send a request the provider has already rejected as malformed.
|
|
154
|
+
var envErr = csrEnvelopeError(input);
|
|
155
|
+
if (envErr !== undefined) input = envErr;
|
|
81
156
|
if (!input || typeof input !== 'object') return false;
|
|
82
157
|
|
|
83
158
|
var status = typeof input.status_code === 'number' ? input.status_code
|
|
@@ -118,6 +193,12 @@ export function isNonRetryableRequestError(input: any): boolean {
|
|
|
118
193
|
}
|
|
119
194
|
|
|
120
195
|
export function isAuthExpiredError(input: any): boolean {
|
|
196
|
+
// A streamed turn whose bearer expired fails exactly like a buffered one, one
|
|
197
|
+
// level deeper. Without this unwrap the refresh-and-resend gate never fires on
|
|
198
|
+
// the streaming path, so an expiry that costs a buffered turn nothing costs a
|
|
199
|
+
// streamed turn the whole answer.
|
|
200
|
+
var envErr = csrEnvelopeError(input);
|
|
201
|
+
if (envErr !== undefined) input = envErr;
|
|
121
202
|
if (!input) return false;
|
|
122
203
|
var blobs: string[] = [];
|
|
123
204
|
var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
|
|
@@ -156,6 +237,11 @@ export function isAuthExpiredError(input: any): boolean {
|
|
|
156
237
|
* reduced to by getErrorMessage, because the view usually only keeps the text.
|
|
157
238
|
*/
|
|
158
239
|
export function isProviderApiKeyError(input: any): boolean {
|
|
240
|
+
// Same unwrap. A wrong API key on a streamed turn is the exact case CRITICAL 2
|
|
241
|
+
// was reported for: without this the "check the project's key" affordance never
|
|
242
|
+
// appears on the platform the user is actually streaming with.
|
|
243
|
+
var envErr = csrEnvelopeError(input);
|
|
244
|
+
if (envErr !== undefined) input = envErr;
|
|
159
245
|
if (!input) return false;
|
|
160
246
|
var blobs: string[] = [];
|
|
161
247
|
var push = function (v: any) { if (typeof v === 'string' && v) blobs.push(v); };
|
package/src/engine/history.ts
CHANGED
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
import { extractClaudeText, extractOpenAIText, INDEXING_COMPLETE_MARKER, EMPTY_INDEXING_REPLY, getChatHistory, bgIndexingQueueName } from './requests';
|
|
8
8
|
import { isErrorResponseBody, getErrorMessage } from './errors';
|
|
9
|
+
import { streamRecoveryEnabled } from './config';
|
|
9
10
|
import { sanitizeAttachmentLinksForHistory } from './links';
|
|
10
11
|
import type { ChatMessage } from './host';
|
|
11
12
|
|
|
@@ -702,8 +703,11 @@ export type MapHistoryOptions = {
|
|
|
702
703
|
};
|
|
703
704
|
|
|
704
705
|
export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'openai', opts: MapHistoryOptions) {
|
|
705
|
-
var mapped: any[] = [], runningItemIds: string[] = [];
|
|
706
|
+
var mapped: any[] = [], runningItemIds: string[] = [], streamPendingItemIds: string[] = [];
|
|
706
707
|
var extractAssistantText = platform === 'openai' ? extractOpenAIText : extractClaudeText;
|
|
708
|
+
// See the `isStreamPending` block below. Read ONCE per call rather than per
|
|
709
|
+
// item: it is a config lookup, and the answer cannot change mid-list.
|
|
710
|
+
var canRecoverStreams = streamRecoveryEnabled();
|
|
707
711
|
var filtered = filterListByClearHorizon(list, opts.clearedAt);
|
|
708
712
|
filtered.slice().reverse().forEach(function (item) {
|
|
709
713
|
var requestBody = item && item.request_body;
|
|
@@ -727,6 +731,45 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
|
|
|
727
731
|
? ((typeof item.response_text === 'string' ? item.response_text : '').trim())
|
|
728
732
|
: ((extractAssistantText(response) || '').trim() || ''));
|
|
729
733
|
var isErrorResponse = !isPending && (isFailed || (!isCompact && isErrorResponseBody(response)));
|
|
734
|
+
// AN ANSWER THAT IS UNKNOWN, NOT EMPTY.
|
|
735
|
+
//
|
|
736
|
+
// A STREAMED turn stores nothing on its row: the relay appends the
|
|
737
|
+
// destination's bytes to the chunk table and settles the row with a status
|
|
738
|
+
// and no body, and only csr-finalize ever copies an answer onto the row. So a
|
|
739
|
+
// row that is terminal, carries no body and carries no error is not a turn
|
|
740
|
+
// that answered nothing: it is a turn nobody finalized. That happens for one
|
|
741
|
+
// ordinary reason: the row settled while no poll was attached (the tab was
|
|
742
|
+
// closed, a mobile browser discarded it, the device slept and the interval
|
|
743
|
+
// stopped), so the client that would have finalized it was not there.
|
|
744
|
+
//
|
|
745
|
+
// Read as "empty" (which is what the `assistantText` guard further down did),
|
|
746
|
+
// such a row produces NO assistant bubble at all and the answer is simply gone
|
|
747
|
+
// from the conversation, with every byte of it still sitting in the chunk
|
|
748
|
+
// table, reachable through clientSecretRequestStream. So it is marked instead,
|
|
749
|
+
// and the two things that act on the mark are the merge (an unknown answer
|
|
750
|
+
// never overwrites a known one) and the recovery (an unknown answer is
|
|
751
|
+
// resolved by reading the chunks back). See ChatMessage._streamPending.
|
|
752
|
+
//
|
|
753
|
+
// Deliberately narrow, so nothing that is genuinely an empty turn is caught:
|
|
754
|
+
// - only when this host streams AND can read chunks back (see
|
|
755
|
+
// streamRecoveryEnabled); a host that does neither behaves exactly as it
|
|
756
|
+
// does today, and one that cannot read them back would only trade a
|
|
757
|
+
// missing bubble for a permanently empty one;
|
|
758
|
+
// - never on the background queue. chatStreamWiring turns streaming OFF for
|
|
759
|
+
// every bg-queue turn (an indexing pass must not stream, and an attachment
|
|
760
|
+
// turn is left buffered on purpose), so a bg row with no body really did
|
|
761
|
+
// answer nothing;
|
|
762
|
+
// - never on a compact stub, whose body was withheld by the server rather
|
|
763
|
+
// than never stored;
|
|
764
|
+
// - 'resolved' only. A FAILED row already renders its error, which is the
|
|
765
|
+
// authoritative account of the turn (and, once csrEnvelopeError is in
|
|
766
|
+
// play, a truthful one). Its chunks are deliberately kept but not
|
|
767
|
+
// rendered. See _finalizeStreamedTurn.
|
|
768
|
+
var isStreamPending = canRecoverStreams && !isCompact && !isPending && !isCancelledItem
|
|
769
|
+
&& !isErrorResponse && !item._isBgTask && !item._isOnBgQueue
|
|
770
|
+
&& item.status === 'resolved'
|
|
771
|
+
&& (item.response_body == null) && (item.error == null)
|
|
772
|
+
&& !assistantText;
|
|
730
773
|
// Record the completion marker, then STRIP it — both, and in that order.
|
|
731
774
|
// Recording gives the display layer a structured signal instead of a substring
|
|
732
775
|
// search over model prose. Stripping matches the live resolution path: without
|
|
@@ -810,6 +853,14 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
|
|
|
810
853
|
if (serverItemId !== undefined) em._serverItemId = serverItemId;
|
|
811
854
|
if (replyTs !== undefined) em._ts = replyTs;
|
|
812
855
|
mapped.push(em);
|
|
856
|
+
} else if (isStreamPending) {
|
|
857
|
+
// The bubble stands for the turn so the merge has something to key on and
|
|
858
|
+
// the recovery has somewhere to write. Its content is UNKNOWN: empty here
|
|
859
|
+
// is the absence of an answer on the row, not the absence of an answer.
|
|
860
|
+
var sp: any = { role: 'assistant', content: '', _streamPending: true };
|
|
861
|
+
if (serverItemId !== undefined) { sp._serverItemId = serverItemId; streamPendingItemIds.push(serverItemId); }
|
|
862
|
+
if (replyTs !== undefined) sp._ts = replyTs;
|
|
863
|
+
mapped.push(sp);
|
|
813
864
|
// `|| reportedComplete`: a pass whose ENTIRE answer was the completion token
|
|
814
865
|
// strips down to an empty string, and the plain `assistantText` guard then
|
|
815
866
|
// emitted no bubble at all — while the live path emitted one. The run read as
|
|
@@ -837,7 +888,54 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
|
|
|
837
888
|
var ownerKey = chatCacheKey(opts.projectId, platform, opts.userId);
|
|
838
889
|
for (var oi = 0; oi < mapped.length; oi++) mapped[oi]._ownerKey = ownerKey;
|
|
839
890
|
}
|
|
840
|
-
|
|
891
|
+
// `streamPendingItemIds` is additive: a consumer that only destructures
|
|
892
|
+
// { messages, runningItemIds } (agent.vue's own call site does) is unaffected.
|
|
893
|
+
// The engine drives recovery off the BUBBLES rather than this list (the merge
|
|
894
|
+
// runs between the two and can adopt a local answer onto one of them), so this
|
|
895
|
+
// is reporting, not the mechanism.
|
|
896
|
+
return { messages: mapped, runningItemIds: runningItemIds, streamPendingItemIds: streamPendingItemIds };
|
|
897
|
+
}
|
|
898
|
+
|
|
899
|
+
/**
|
|
900
|
+
* Let a LOCAL copy of a turn survive a page whose copy of it is
|
|
901
|
+
* AUTHORITATIVE-BUT-EMPTY. Mutates `incoming`; returns true when it took anything.
|
|
902
|
+
*
|
|
903
|
+
* THE FAILURE THIS PREVENTS. A streamed turn's row goes 'resolved' the moment the
|
|
904
|
+
* relay finishes, and its answer reaches the row only when csr-finalize stores it,
|
|
905
|
+
* one poll interval plus a round trip later. A first-page history refetch landing
|
|
906
|
+
* inside that window maps the row to a `_streamPending` bubble with no content, and
|
|
907
|
+
* the merge, which believes the server, throws away the local bubble holding the
|
|
908
|
+
* answer the reader is looking at. The window opens on EVERY streamed turn, and a
|
|
909
|
+
* refetch fires from visibilitychange, so it is not a corner case.
|
|
910
|
+
*
|
|
911
|
+
* The rule is the same one the recovery reads: an UNKNOWN answer never overwrites a
|
|
912
|
+
* KNOWN one. Where the local copy is still live (pending, or being painted into),
|
|
913
|
+
* its live-ness is adopted too: without it the merge would hand back a settled
|
|
914
|
+
* bubble the painter can no longer find (_liveTargetIndex wants isPending or
|
|
915
|
+
* _streaming) and that _turnAlreadyRendered would then read as already answered, so
|
|
916
|
+
* the settle would drop the real answer on the floor.
|
|
917
|
+
*/
|
|
918
|
+
export function adoptLocalAnswerIntoPage(incoming: ChatMessage, local: ChatMessage): boolean {
|
|
919
|
+
if (!incoming || !local || !incoming._streamPending) return false;
|
|
920
|
+
if (incoming.role !== 'assistant' || local.role !== 'assistant') return false;
|
|
921
|
+
var hasText = typeof local.content === 'string' && local.content.length > 0;
|
|
922
|
+
var isLive = !!(local.isPending || local._streaming);
|
|
923
|
+
if (!hasText && !isLive) return false;
|
|
924
|
+
if (hasText) {
|
|
925
|
+
incoming.content = local.content;
|
|
926
|
+
// The answer is on screen, so nothing has to be read back. Whether the row
|
|
927
|
+
// itself ever gets a stored body is the live path's business (its finalize is
|
|
928
|
+
// in flight); if that fails, the NEXT load meets an empty row again and
|
|
929
|
+
// recovers it then.
|
|
930
|
+
incoming._streamPending = false;
|
|
931
|
+
}
|
|
932
|
+
// Carried whatever the text says: these are what keep a still-running turn
|
|
933
|
+
// reachable by the painter and by the settle.
|
|
934
|
+
if (local._localId !== undefined) incoming._localId = local._localId;
|
|
935
|
+
if (local.isPending) incoming.isPending = true;
|
|
936
|
+
if (local.isPendingInProcess) incoming.isPendingInProcess = true;
|
|
937
|
+
if (local._streaming) incoming._streaming = true;
|
|
938
|
+
return true;
|
|
841
939
|
}
|
|
842
940
|
|
|
843
941
|
/* ---- rescuing in-flight bubbles across a first-page refetch ---------------
|
|
@@ -886,6 +984,19 @@ export function shouldRescueInFlightMessage(m: ChatMessage, ctx: RescueDecisionC
|
|
|
886
984
|
// it. Unconditional: applying the pending-assistant test here would delete the
|
|
887
985
|
// user's message mid-upload whenever some other turn happened to be in flight.
|
|
888
986
|
if (m._stageId) return true;
|
|
987
|
+
// A bubble being painted from a LIVE stream, before its dispatch has reported a
|
|
988
|
+
// server id. Same shape of reason as the staged bubble above: the page identifies
|
|
989
|
+
// turns by id, so nothing in it can stand for this one, and dropping it would
|
|
990
|
+
// leave the stream with no bubble to paint into (the painter finds its target by
|
|
991
|
+
// _serverItemId, and this bubble has none to be found by) - the turn would sit on
|
|
992
|
+
// "Thinking..." until it settled, with every relayed byte already spent.
|
|
993
|
+
//
|
|
994
|
+
// Gated on the missing id rather than on _streaming alone, and deliberately: once
|
|
995
|
+
// the id IS on the bubble, the tests below are right and the local copy really is
|
|
996
|
+
// redundant. The page carries the same turn as a pending placeholder with that id,
|
|
997
|
+
// the painter finds THAT one on its next paint (about a second later), and rescuing
|
|
998
|
+
// as well would put the same turn on screen twice.
|
|
999
|
+
if (m._streaming && !m._serverItemId) return true;
|
|
889
1000
|
if (!m._serverItemId && ctx.pageHasPendingAssistant) return false;
|
|
890
1001
|
// In flight by its own flags, id or no id. The immediate-send pair is stamped
|
|
891
1002
|
// with its server id as soon as the dispatch reports one, and that id is there
|