bunnyquery 1.8.6 → 1.8.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bunnyquery.css +95 -5
- package/bunnyquery.js +1470 -111
- package/dist/engine.cjs +1071 -80
- package/dist/engine.cjs.map +1 -1
- package/dist/engine.d.mts +608 -169
- package/dist/engine.d.ts +608 -169
- package/dist/engine.mjs +1048 -81
- package/dist/engine.mjs.map +1 -1
- package/package.json +1 -1
- package/src/engine/budget.ts +207 -68
- package/src/engine/config.ts +48 -0
- package/src/engine/errors.ts +38 -0
- package/src/engine/history.ts +546 -6
- package/src/engine/host.ts +7 -0
- package/src/engine/image_preview.ts +0 -0
- package/src/engine/index.ts +15 -1
- package/src/engine/indexing_groups.ts +248 -3
- package/src/engine/links.ts +143 -6
- package/src/engine/office.ts +24 -1
- package/src/engine/prompts/chat_system_prompt.ts +1 -1
- package/src/engine/requests.ts +164 -11
- package/src/engine/session.ts +544 -35
- package/src/widget.css +35 -1
- package/styles/chat.css +60 -4
package/src/engine/index.ts
CHANGED
|
@@ -43,7 +43,7 @@ export * from './prompts';
|
|
|
43
43
|
|
|
44
44
|
// Pure helpers (Tier-1.5): error detection, token budgeting, link/path
|
|
45
45
|
// normalization, and history mapping — shared so both consumers stay identical.
|
|
46
|
-
export { getErrorMessage, isErrorResponseBody, isAuthExpiredError, isNonRetryableRequestError } from './errors';
|
|
46
|
+
export { getErrorMessage, isErrorResponseBody, isAuthExpiredError, isNonRetryableRequestError, isProviderApiKeyError } from './errors';
|
|
47
47
|
export * from './budget';
|
|
48
48
|
// Per-format UTF-8 declaration for files offered as a download. Shared so a fenced
|
|
49
49
|
// block and a server-published file open identically in Excel, Word and a browser.
|
|
@@ -64,6 +64,12 @@ export {
|
|
|
64
64
|
// read it the same way or the two will not group together.
|
|
65
65
|
isIndexingRequestText,
|
|
66
66
|
parseIndexingRequestText,
|
|
67
|
+
// One bounded look at the bg-indexing queue: which files still have a live
|
|
68
|
+
// pass. The dbfile browser's "indexed" badge uses this so a file only goes
|
|
69
|
+
// green once the run is confirmed over, not when its src:: record appears.
|
|
70
|
+
fetchLiveIndexingKeys,
|
|
71
|
+
getSplitChatHistory,
|
|
72
|
+
__resetSplitHistoryState,
|
|
67
73
|
type IndexingRequestRef,
|
|
68
74
|
type MapHistoryOptions,
|
|
69
75
|
} from './history';
|
|
@@ -94,6 +100,8 @@ export {
|
|
|
94
100
|
type IndexingGroup,
|
|
95
101
|
type IndexingGroupStatus,
|
|
96
102
|
type BuildDisplayListOptions,
|
|
103
|
+
type RunStubInfo,
|
|
104
|
+
RUN_RECORD_WORKING_STALE_MS,
|
|
97
105
|
} from './indexing_groups';
|
|
98
106
|
|
|
99
107
|
export {
|
|
@@ -105,6 +113,11 @@ export {
|
|
|
105
113
|
BG_INDEXING_QUEUE_SUFFIX,
|
|
106
114
|
bgIndexingQueueName,
|
|
107
115
|
isBgIndexingQueue,
|
|
116
|
+
indexDoneUniqueId,
|
|
117
|
+
runIndexUniqueId,
|
|
118
|
+
upsertIndexRunRecordSafe,
|
|
119
|
+
type IndexRunStatus,
|
|
120
|
+
type IndexRunPatch,
|
|
108
121
|
MCP_NAME,
|
|
109
122
|
DEFAULT_CLAUDE_MODEL,
|
|
110
123
|
DEFAULT_OPENAI_MODEL,
|
|
@@ -116,6 +129,7 @@ export {
|
|
|
116
129
|
listClaudeModels,
|
|
117
130
|
listOpenAIModels,
|
|
118
131
|
getChatHistory,
|
|
132
|
+
buildHistoryItemFullId,
|
|
119
133
|
// response extraction
|
|
120
134
|
extractClaudeText,
|
|
121
135
|
extractOpenAIText,
|
|
@@ -198,6 +198,15 @@ export type IndexingGroup = {
|
|
|
198
198
|
* loaded ones. 'status': the queue has not yet said whether this file is still
|
|
199
199
|
* being worked on, which is the only thing that can end a worker-driven run. */
|
|
200
200
|
resolvingReason?: 'history' | 'status';
|
|
201
|
+
/** Synthesized from a durable run:: record: none of the run's passes are
|
|
202
|
+
* among the loaded messages (bg history still deferred, or the run is older
|
|
203
|
+
* than the paging cap). Header-and-status only — members/visibleMembers are
|
|
204
|
+
* empty and there is nothing to cancel; the row is replaced by the real
|
|
205
|
+
* group the moment actual passes load (same `key`, so expansion state
|
|
206
|
+
* carries over). */
|
|
207
|
+
stub?: boolean;
|
|
208
|
+
/** The run:: record's stored error text, for a stub row's meta line. */
|
|
209
|
+
stubError?: string;
|
|
201
210
|
};
|
|
202
211
|
|
|
203
212
|
export type DisplayEntry =
|
|
@@ -222,6 +231,14 @@ export type BuildDisplayListOptions = {
|
|
|
222
231
|
/** Whether `liveIndexKeys` has been answered at least once for this chat. False
|
|
223
232
|
* is "we do not know", and a worker-driven run stays unfinished on it. */
|
|
224
233
|
liveIndexChecked?: boolean;
|
|
234
|
+
/** Files carrying the durable done:: completion marker (one prefix sweep),
|
|
235
|
+
* keyed like IndexingGroup.key (storage path; the bare-name fallback keys
|
|
236
|
+
* of very old prompts simply never match — they keep the queue inference).
|
|
237
|
+
* A marker is PROOF the file was read to the end: it settles a worker-run
|
|
238
|
+
* green without waiting for the queue answer, and it is never withheld by
|
|
239
|
+
* the resolving logic. A live queue hit still outranks it (a re-index in
|
|
240
|
+
* flight whose marker-cascade delete lagged). */
|
|
241
|
+
doneKeys?: { [fileKey: string]: boolean };
|
|
225
242
|
/** Server item ids of passes that were on a row when the user STOPPED it
|
|
226
243
|
* (ChatSession.state.stoppedIndexIds). A run holding any of them is a run the
|
|
227
244
|
* user stopped — see the status derivation for why a stop usually leaves no
|
|
@@ -232,8 +249,52 @@ export type BuildDisplayListOptions = {
|
|
|
232
249
|
* windowedIndexing). Passed in rather than read from config so this stays a
|
|
233
250
|
* pure function of its inputs and can be exercised for both settings. */
|
|
234
251
|
windowedIndexing?: boolean;
|
|
252
|
+
/** Durable run:: records, keyed by STORAGE PATH (the consumer's marker
|
|
253
|
+
* sweep). Each key with no real group in the loaded messages gets a
|
|
254
|
+
* synthesized header-only row (see IndexingGroup.stub), placed by its
|
|
255
|
+
* `started` timestamp. A real group for the same file — matched by key,
|
|
256
|
+
* path, or name — always suppresses the stub: loaded passes are evidence,
|
|
257
|
+
* the record is only a summary. */
|
|
258
|
+
runStubs?: { [storagePath: string]: RunStubInfo };
|
|
259
|
+
/** The platform whose chat this list is for. run:: records are per-FILE,
|
|
260
|
+
* but a chat is per (project, platform): a run started under Claude has
|
|
261
|
+
* its passes in the Claude conversation and is invisible to the
|
|
262
|
+
* OpenAI-scoped queue probe, so its stub could never be covered and never
|
|
263
|
+
* be confirmed — it just sat there, in a chat it did not belong to.
|
|
264
|
+
* Records minted before this was stamped carry no platform and are shown
|
|
265
|
+
* in both, which keeps the leak to the historical set. */
|
|
266
|
+
stubPlatform?: 'claude' | 'openai';
|
|
267
|
+
/** The chat's clear-history horizon (ms epoch). Run records are service-
|
|
268
|
+
* wide and know nothing about a cleared chat, so without this every
|
|
269
|
+
* "Clear chat history" resurrects one row per indexed file. A stub whose
|
|
270
|
+
* run ended (or, unfinished, began) at or before this moment is dropped —
|
|
271
|
+
* unless the queue says the file is live RIGHT NOW, which no horizon can
|
|
272
|
+
* make untrue. */
|
|
273
|
+
stubClearedAt?: number;
|
|
274
|
+
/** Clock injection for tests; defaults to Date.now(). Only run-stub
|
|
275
|
+
* staleness reads it. */
|
|
276
|
+
now?: number;
|
|
235
277
|
};
|
|
236
278
|
|
|
279
|
+
/** The display-relevant fields of a run:: record (see requests.ts
|
|
280
|
+
* runIndexUniqueId for the record's contract). */
|
|
281
|
+
export type RunStubInfo = {
|
|
282
|
+
status: 'working' | 'done' | 'error' | 'cancelled';
|
|
283
|
+
filename?: string;
|
|
284
|
+
started?: number;
|
|
285
|
+
finished?: number;
|
|
286
|
+
error?: string;
|
|
287
|
+
/** Chat that owns this run. Absent on records minted before it was
|
|
288
|
+
* stamped; see BuildDisplayListOptions.stubPlatform. */
|
|
289
|
+
platform?: 'claude' | 'openai';
|
|
290
|
+
};
|
|
291
|
+
|
|
292
|
+
/** A 'working' run record older than this with no live-queue confirmation is
|
|
293
|
+
* treated as unknown rather than live: a chain that died without reaching any
|
|
294
|
+
* error path leaves 'working' dangling, and a row must not spin forever on a
|
|
295
|
+
* claim nothing can end. */
|
|
296
|
+
export const RUN_RECORD_WORKING_STALE_MS = 6 * 60 * 60 * 1000;
|
|
297
|
+
|
|
237
298
|
// The indexing label is view-formatted (formatIndexingLabel), so parsing it is
|
|
238
299
|
// the FALLBACK path only: it exists for bubbles restored from a history cache
|
|
239
300
|
// written before `_indexFile` was stamped. Live and freshly-mapped bubbles carry
|
|
@@ -324,12 +385,14 @@ export function buildChatDisplayList(
|
|
|
324
385
|
var list = Array.isArray(messages) ? messages : [];
|
|
325
386
|
var liveIndexKeys = (opts && opts.liveIndexKeys) || {};
|
|
326
387
|
var liveIndexChecked = !!(opts && opts.liveIndexChecked);
|
|
388
|
+
var doneKeys = (opts && opts.doneKeys) || {};
|
|
327
389
|
var stoppedIndexIds = (opts && opts.stoppedIndexIds) || {};
|
|
328
390
|
var windowedIndexing = opts && opts.windowedIndexing !== undefined
|
|
329
391
|
? !!opts.windowedIndexing
|
|
330
392
|
: windowedIndexingEnabled();
|
|
331
393
|
var hasMoreHistory = !!(opts && opts.hasMoreHistory);
|
|
332
394
|
var loadingOlderHistory = !!(opts && opts.loadingOlderHistory);
|
|
395
|
+
var stubPlatform = opts && opts.stubPlatform;
|
|
333
396
|
|
|
334
397
|
// One entry per RUN (see IndexingGroup.runKey), addressed by an internal id
|
|
335
398
|
// while the list is being walked; runKey is assigned at the end, once the
|
|
@@ -629,7 +692,13 @@ export function buildChatDisplayList(
|
|
|
629
692
|
// permanently false and the two tests below were already carrying the whole
|
|
630
693
|
// decision. It is gone rather than left as a hook, so this reads as what it
|
|
631
694
|
// actually is: for a worker-driven run, ONLY the queue can say it is over.
|
|
695
|
+
// The done:: marker is the third — and strongest — disjunct: written
|
|
696
|
+
// only at a chain's true end (backend worker, or a client's own
|
|
697
|
+
// deterministic completion), it answers without any queue round
|
|
698
|
+
// trip. Gated on the queue NOT claiming the file live, so a
|
|
699
|
+
// re-index whose marker-cascade delete lagged still shows yellow.
|
|
632
700
|
grp.finished = !newestRunOfKey[order[oi]] ||
|
|
701
|
+
(!!doneKeys[grp.key] && !liveIndexKeys[grp.key]) ||
|
|
633
702
|
(liveIndexChecked && !liveIndexKeys[grp.key]);
|
|
634
703
|
}
|
|
635
704
|
|
|
@@ -645,7 +714,7 @@ export function buildChatDisplayList(
|
|
|
645
714
|
// newest pass's own outcome, and newest-first paging always has that pass.
|
|
646
715
|
grp.resolving = false;
|
|
647
716
|
} else if (grp.mayHaveOlder && loadingOlderHistory &&
|
|
648
|
-
!liveIndexKeys[grp.key] && newestRunOfKey[order[oi]]) {
|
|
717
|
+
!liveIndexKeys[grp.key] && !doneKeys[grp.key] && newestRunOfKey[order[oi]]) {
|
|
649
718
|
// The line this draws, and it is the same line the 'status' branch draws:
|
|
650
719
|
// withhold what is UNKNOWN, and what is only INFERRED settled. Never
|
|
651
720
|
// withhold what is PROVEN, in either direction. Two proofs, and an older
|
|
@@ -692,15 +761,191 @@ export function buildChatDisplayList(
|
|
|
692
761
|
}
|
|
693
762
|
}
|
|
694
763
|
|
|
764
|
+
// --- durable run:: stubs ---------------------------------------------------
|
|
765
|
+
// A run record whose passes are not among the loaded messages still deserves
|
|
766
|
+
// a row: on a fresh open the bg history is deferred, and a run older than the
|
|
767
|
+
// paging cap never loads at all. Synthesized as header-only groups and placed
|
|
768
|
+
// by `started` among the messages' own timestamps; a real group for the same
|
|
769
|
+
// file always wins (see BuildDisplayListOptions.runStubs).
|
|
770
|
+
var stubList: { started: number; group: IndexingGroup }[] = [];
|
|
771
|
+
var runStubs = opts && opts.runStubs;
|
|
772
|
+
if (runStubs) {
|
|
773
|
+
// Two namespaces, deliberately NOT one map: storage paths are
|
|
774
|
+
// project-relative and folder trees repeat basenames routinely, so a
|
|
775
|
+
// flat map let a real group for "2025/report.pdf" delete the stub for
|
|
776
|
+
// "2026/report.pdf" with nothing replacing it. A bare NAME only covers
|
|
777
|
+
// a stub when the group itself has no path (the legacy compact-label
|
|
778
|
+
// case) and the record has no filename of its own to disambiguate.
|
|
779
|
+
var coveredPaths: { [k: string]: boolean } = {};
|
|
780
|
+
var coveredPathlessNames: { [k: string]: boolean } = {};
|
|
781
|
+
for (var ci = 0; ci < order.length; ci++) {
|
|
782
|
+
var cg = groups[order[ci]];
|
|
783
|
+
if (cg.path) { coveredPaths[cg.path] = true; if (cg.key) coveredPaths[cg.key] = true; }
|
|
784
|
+
else if (cg.name) coveredPathlessNames[cg.name] = true;
|
|
785
|
+
else if (cg.key) coveredPaths[cg.key] = true;
|
|
786
|
+
}
|
|
787
|
+
var now = opts && typeof opts.now === 'number' ? opts.now : Date.now();
|
|
788
|
+
var stubClearedAt = (opts && typeof opts.stubClearedAt === 'number' && opts.stubClearedAt > 0)
|
|
789
|
+
? opts.stubClearedAt : 0;
|
|
790
|
+
for (var sp in runStubs) {
|
|
791
|
+
var rec = runStubs[sp];
|
|
792
|
+
if (!sp || !rec || !rec.status || coveredPaths[sp]) continue;
|
|
793
|
+
var fname = rec.filename || sp.split('/').pop() || sp;
|
|
794
|
+
// A PATHLESS group (legacy compact label) can only be matched by
|
|
795
|
+
// name, and that is still the right match: the file already has a
|
|
796
|
+
// row, so suppressing avoids a duplicate rather than creating a
|
|
797
|
+
// void. What is gone is the reverse — a group WITH a path no longer
|
|
798
|
+
// writes into the name namespace, so it can no longer delete a
|
|
799
|
+
// same-basename stub from another folder.
|
|
800
|
+
if (coveredPathlessNames[fname]) continue;
|
|
801
|
+
// A run recorded under the OTHER platform's chat belongs to that
|
|
802
|
+
// conversation: its passes live in that history (so no real group
|
|
803
|
+
// can ever cover this stub) and the queue probe is platform-scoped
|
|
804
|
+
// (so it can never confirm it). Records with no platform are
|
|
805
|
+
// legacy and keep the old behaviour rather than vanishing.
|
|
806
|
+
if (stubPlatform && rec.platform && rec.platform !== stubPlatform) continue;
|
|
807
|
+
// Mirrors the real groups' status precedence: a live queue hit outranks
|
|
808
|
+
// whatever the record says (a lagged update must not paint a verdict over
|
|
809
|
+
// visible work), then the record's own terminal statuses, then the
|
|
810
|
+
// queue's authoritative ABSENCE, and only then the stated wait.
|
|
811
|
+
var live = !!liveIndexKeys[sp] || !!liveIndexKeys[fname];
|
|
812
|
+
// Cleared-history horizon: a run that ended at or before the clear is
|
|
813
|
+
// part of what the user asked to forget. A live queue hit survives it
|
|
814
|
+
// (see BuildDisplayListOptions.stubClearedAt). A record carrying NO
|
|
815
|
+
// timestamp at all is unplaceable, not old — the horizon cannot judge
|
|
816
|
+
// it, so it is never dropped by one.
|
|
817
|
+
var recWhen = typeof rec.finished === 'number' ? rec.finished
|
|
818
|
+
: typeof rec.started === 'number' ? rec.started : undefined;
|
|
819
|
+
if (stubClearedAt && !live && recWhen !== undefined && recWhen <= stubClearedAt) continue;
|
|
820
|
+
var st: IndexingGroupStatus = 'active';
|
|
821
|
+
var fin = false;
|
|
822
|
+
var res = false;
|
|
823
|
+
var reason: 'history' | 'status' | undefined;
|
|
824
|
+
if (!live) {
|
|
825
|
+
// A done:: marker from the same sweep is as terminal as the record's
|
|
826
|
+
// own 'done' — it settles a dangling 'working' the chain never
|
|
827
|
+
// flipped (mirrors the real-group doneKeys disjunct).
|
|
828
|
+
if (rec.status === 'done' || doneKeys[sp] || doneKeys[fname]) { st = 'done'; fin = true; }
|
|
829
|
+
else if (rec.status === 'error') { st = 'error'; fin = true; }
|
|
830
|
+
else if (rec.status === 'cancelled') { st = 'cancelled'; fin = true; }
|
|
831
|
+
else if (liveIndexChecked) {
|
|
832
|
+
// 'working' is a CLAIM. The queue has now answered, untruncated,
|
|
833
|
+
// and holds nothing for this file — the same proof the real-group
|
|
834
|
+
// ladder settles on. Without this disjunct a stub had no
|
|
835
|
+
// terminator at all: worker-driven runs are never flipped by this
|
|
836
|
+
// client, so the row sat grey forever (the reported "checking
|
|
837
|
+
// status that never resolves").
|
|
838
|
+
st = 'done'; fin = true;
|
|
839
|
+
} else if (typeof rec.started === 'number' && now - rec.started > RUN_RECORD_WORKING_STALE_MS) {
|
|
840
|
+
// Older than any real run and still unconfirmed: a dead claim, not
|
|
841
|
+
// an open question. Same horizon the files page applies.
|
|
842
|
+
st = 'error'; fin = true;
|
|
843
|
+
} else {
|
|
844
|
+
// Genuinely unanswered — the queue reply is one round trip away
|
|
845
|
+
// and this self-heals the moment it lands.
|
|
846
|
+
res = true; reason = 'status';
|
|
847
|
+
}
|
|
848
|
+
}
|
|
849
|
+
var sg: IndexingGroup = {
|
|
850
|
+
key: sp,
|
|
851
|
+
// ONE identity for the run whether it renders from the record or
|
|
852
|
+
// from its loaded passes: the views key the DOM off runKey, so a
|
|
853
|
+
// 'stub:'-prefixed key meant every handoff was an unmount plus a
|
|
854
|
+
// remount somewhere else. Named after the record's start, which
|
|
855
|
+
// the real group below reuses when it has one.
|
|
856
|
+
runKey: 'run:' + sp + '#' + (typeof rec.started === 'number' ? rec.started : 'n'),
|
|
857
|
+
name: fname,
|
|
858
|
+
path: sp,
|
|
859
|
+
mime: undefined,
|
|
860
|
+
size: undefined,
|
|
861
|
+
isReindex: false,
|
|
862
|
+
members: [],
|
|
863
|
+
passCount: 0,
|
|
864
|
+
status: st,
|
|
865
|
+
cancellableIds: [],
|
|
866
|
+
cancelling: false,
|
|
867
|
+
stopped: st === 'cancelled',
|
|
868
|
+
mayHaveOlder: hasMoreHistory,
|
|
869
|
+
anchorIndex: -1,
|
|
870
|
+
anchorId: '',
|
|
871
|
+
visibleMembers: [],
|
|
872
|
+
driver: !isPagedReadFile(fname, undefined) ? 'single'
|
|
873
|
+
: isImageVisionFile(fname, undefined) ? 'worker'
|
|
874
|
+
: (windowedIndexing ? 'worker' : 'client'),
|
|
875
|
+
finished: fin,
|
|
876
|
+
resolving: res,
|
|
877
|
+
resolvingReason: reason,
|
|
878
|
+
stub: true,
|
|
879
|
+
stubError: rec.error || (st === 'error' && !rec.error
|
|
880
|
+
? 'Indexing did not finish.' : undefined),
|
|
881
|
+
};
|
|
882
|
+
// A record with no `started` cannot be placed in the conversation.
|
|
883
|
+
// Infinity sorts it to the END (nearest the newest turns) instead of
|
|
884
|
+
// above every message the user has ever sent — the row is about
|
|
885
|
+
// something recent, not about the beginning of time.
|
|
886
|
+
stubList.push({ started: typeof rec.started === 'number' ? rec.started : Infinity, group: sg });
|
|
887
|
+
}
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
// --- timestamp-anchor INCOMPLETE real runs ---------------------------------
|
|
891
|
+
// A real run's row renders at its first LOADED pass, and while older history
|
|
892
|
+
// pages in (newest-first) that is a moving target: the run first appears at
|
|
893
|
+
// its newest pass — down by the recent bubbles — then relocates upward as
|
|
894
|
+
// earlier passes arrive, so the row visibly jumps. The run:: record's
|
|
895
|
+
// `started` is a stable anchor for exactly that window: emit the row at its
|
|
896
|
+
// timestamp position instead, and once the true first pass loads
|
|
897
|
+
// (mayHaveOlder false) emission returns to the anchor, which by then IS the
|
|
898
|
+
// same spot. Newest run of the file only — the record describes it, and an
|
|
899
|
+
// older run's passes are already fully placed around it.
|
|
900
|
+
//
|
|
901
|
+
// NOT gated on mayHaveOlder: that made the position depend on WHICH passes
|
|
902
|
+
// happened to be loaded, so a row moved when the sweep resolved, again when
|
|
903
|
+
// the first pass paged in, and again when paging ended. A record with a
|
|
904
|
+
// `started` places the run at ONE spot for as long as the record exists.
|
|
905
|
+
var suppressAnchor: { [runId: string]: boolean } = {};
|
|
906
|
+
if (runStubs) {
|
|
907
|
+
for (var ti2 = 0; ti2 < order.length; ti2++) {
|
|
908
|
+
var tg = groups[order[ti2]];
|
|
909
|
+
if (!newestRunOfKey[order[ti2]]) continue;
|
|
910
|
+
var trec = (tg.path && runStubs[tg.path]) || runStubs[tg.key];
|
|
911
|
+
if (!trec || typeof trec.started !== 'number') continue;
|
|
912
|
+
if (stubPlatform && trec.platform && trec.platform !== stubPlatform) continue;
|
|
913
|
+
suppressAnchor[order[ti2]] = true;
|
|
914
|
+
// Same identity the stub form uses, so a row that upgrades from
|
|
915
|
+
// record-only to loaded-passes keeps its DOM node and its position.
|
|
916
|
+
tg.runKey = 'run:' + (tg.path || tg.key) + '#' + trec.started;
|
|
917
|
+
stubList.push({ started: trec.started, group: tg });
|
|
918
|
+
}
|
|
919
|
+
}
|
|
920
|
+
stubList.sort(function (a, b) { return a.started - b.started; });
|
|
921
|
+
|
|
695
922
|
var out: DisplayEntry[] = [];
|
|
923
|
+
var si = 0;
|
|
696
924
|
for (var j = 0; j < list.length; j++) {
|
|
925
|
+
// Splice stubs in by time: everything that started before this message
|
|
926
|
+
// goes above it. Messages without a timestamp decide nothing.
|
|
927
|
+
var mts = list[j] && typeof list[j]._ts === 'number' ? (list[j]._ts as number) : undefined;
|
|
928
|
+
if (mts !== undefined) {
|
|
929
|
+
while (si < stubList.length && stubList[si].started <= mts) {
|
|
930
|
+
out.push({ kind: 'indexing', group: stubList[si].group, index: -1 - si });
|
|
931
|
+
si++;
|
|
932
|
+
}
|
|
933
|
+
}
|
|
697
934
|
var r = runOfIndex[j];
|
|
698
935
|
if (r === undefined) {
|
|
699
936
|
out.push({ kind: 'message', msg: list[j], index: j });
|
|
700
937
|
continue;
|
|
701
938
|
}
|
|
702
|
-
// Every other member of the run is represented by the row at the anchor
|
|
703
|
-
|
|
939
|
+
// Every other member of the run is represented by the row at the anchor —
|
|
940
|
+
// unless the row is timestamp-anchored (incomplete run with a run::
|
|
941
|
+
// record), in which case the flush above already emitted it.
|
|
942
|
+
if (groups[r].anchorIndex === j && !suppressAnchor[r]) {
|
|
943
|
+
out.push({ kind: 'indexing', group: groups[r], index: j });
|
|
944
|
+
}
|
|
945
|
+
}
|
|
946
|
+
while (si < stubList.length) {
|
|
947
|
+
out.push({ kind: 'indexing', group: stubList[si].group, index: -1 - si });
|
|
948
|
+
si++;
|
|
704
949
|
}
|
|
705
950
|
return out;
|
|
706
951
|
}
|
package/src/engine/links.ts
CHANGED
|
@@ -21,6 +21,24 @@ export var LINK_LABEL_MAX_DISPLAY_CHARS = 32;
|
|
|
21
21
|
*/
|
|
22
22
|
export var EXPIRED_LINK_REFRESH_EXPIRES_SECONDS = 20 * 60;
|
|
23
23
|
|
|
24
|
+
/**
|
|
25
|
+
* Lifetime of the url minted for an inline image PREVIEW.
|
|
26
|
+
*
|
|
27
|
+
* Longer than the click url above, and for a different reason. A click hands the
|
|
28
|
+
* user a url they may keep, so it stays short. A preview url is consumed by the
|
|
29
|
+
* page itself and never leaves it, and it is the ONE lever on how long the
|
|
30
|
+
* downloaded picture stays reusable: get_signed_url will not cache a mint for
|
|
31
|
+
* longer than the credential inside it survives, so `browser_cache` cannot buy
|
|
32
|
+
* local availability that `expires` has not paid for. Twenty minutes meant every
|
|
33
|
+
* image re-downloaded three times an hour of ordinary reading.
|
|
34
|
+
*
|
|
35
|
+
* An hour, giving 55 minutes of cache once the server's five minute headroom is
|
|
36
|
+
* taken off. Short enough that a leaked preview url is not a standing grant, long
|
|
37
|
+
* enough that a conversation does not re-fetch its own pictures while the user is
|
|
38
|
+
* still reading it.
|
|
39
|
+
*/
|
|
40
|
+
export var PREVIEW_URL_EXPIRES_SECONDS = 60 * 60;
|
|
41
|
+
|
|
24
42
|
/**
|
|
25
43
|
* Seconds the browser may reuse a minted preview url (`browser_cache`).
|
|
26
44
|
*
|
|
@@ -30,12 +48,17 @@ export var EXPIRED_LINK_REFRESH_EXPIRES_SECONDS = 20 * 60;
|
|
|
30
48
|
* url comes back out of the browser cache, so the body already on disk stays
|
|
31
49
|
* addressable.
|
|
32
50
|
*
|
|
33
|
-
*
|
|
34
|
-
*
|
|
35
|
-
*
|
|
36
|
-
*
|
|
37
|
-
*
|
|
38
|
-
*
|
|
51
|
+
* A CEILING, not a promise. get_signed_url caps what it grants at the lifetime of
|
|
52
|
+
* the url inside the response (expires minus headroom, so 15 minutes for the
|
|
53
|
+
* platform's 20 minute url), because a mint cached for longer than its own
|
|
54
|
+
* credential is a guaranteed 403 that the browser keeps serving from its own
|
|
55
|
+
* store. Asking for the week is still right: it says what this client would
|
|
56
|
+
* reuse if the url were stable by construction, and the server decides.
|
|
57
|
+
*
|
|
58
|
+
* What keeps an image painting is the cached BODY, not a live url. Once the
|
|
59
|
+
* browser evicts that body it refetches with a url that has since expired, gets a
|
|
60
|
+
* 403, and the error path re-mints with `refresh` and mintCacheBustStamp. That
|
|
61
|
+
* path is load-bearing, not a rare fallback.
|
|
39
62
|
*
|
|
40
63
|
* A week is the platform default for reading a private file, not a number chosen
|
|
41
64
|
* here: skapi-js reads every private record file with
|
|
@@ -60,6 +83,102 @@ export var PREVIEW_BROWSER_CACHE_SECONDS = 7 * 24 * 60 * 60;
|
|
|
60
83
|
*/
|
|
61
84
|
export var LINK_REFRESH_WINDOW_MS = (EXPIRED_LINK_REFRESH_EXPIRES_SECONDS - 5 * 60) * 1000;
|
|
62
85
|
|
|
86
|
+
/**
|
|
87
|
+
* Cache generation for the mint request url. BUMP THIS to abandon every mint
|
|
88
|
+
* response browsers are currently holding.
|
|
89
|
+
*
|
|
90
|
+
* Generation 2 retires the entries written before 2026-08-11. Those were stored
|
|
91
|
+
* with `max-age=604800` around a presign that dies in twenty minutes, so from
|
|
92
|
+
* minute 21 each one is a guaranteed 403 that the browser keeps serving from its
|
|
93
|
+
* own store for the rest of the week. The server no longer grants a lifetime a
|
|
94
|
+
* url cannot back (get_signed_url resolve_browser_cache), but that fixes what is
|
|
95
|
+
* written from now on and cannot reach what is already stored on a user's
|
|
96
|
+
* device. Changing the url is the only thing that can: an entry nobody requests
|
|
97
|
+
* again is an entry that cannot answer again.
|
|
98
|
+
*/
|
|
99
|
+
export var MINT_CACHE_GENERATION = 2;
|
|
100
|
+
|
|
101
|
+
/**
|
|
102
|
+
* Window stamp for a REFRESH mint.
|
|
103
|
+
*
|
|
104
|
+
* WINDOWED, not Date.now(): a per-call stamp is a new cache key per image per
|
|
105
|
+
* retry, which is what made the original `nocache` parameter worse than the
|
|
106
|
+
* disease. One stamp per refresh window means every repair inside those minutes
|
|
107
|
+
* shares a single entry, and it rotates before the url it carries can die.
|
|
108
|
+
*/
|
|
109
|
+
export function mintCacheBustStamp(now?: number): number {
|
|
110
|
+
return Math.floor((now == null ? Date.now() : now) / LINK_REFRESH_WINDOW_MS);
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* The `nocache` value for a preview mint: the generation, plus a window stamp
|
|
115
|
+
* when this mint is a repair.
|
|
116
|
+
*
|
|
117
|
+
* A repair MUST reach the origin, and the request header the clients used to
|
|
118
|
+
* rely on cannot do it. `Cache-Control: no-cache` is not a CORS-safelisted
|
|
119
|
+
* request header, and the record gateway's preflight answers
|
|
120
|
+
* `Access-Control-Allow-Headers` WITHOUT it (verified against the live api on
|
|
121
|
+
* 2026-08-11), so a mint carrying that header is rejected by the browser before
|
|
122
|
+
* it is ever sent. Every repair therefore failed, in every browser, and the chip
|
|
123
|
+
* went straight to "(unavailable)". Only a phone noticed, because only a phone
|
|
124
|
+
* drops image bodies often enough to need the repair at all.
|
|
125
|
+
*
|
|
126
|
+
* A query parameter has no such problem: it is part of the url, so it needs no
|
|
127
|
+
* preflight and no cooperation from the cache.
|
|
128
|
+
*/
|
|
129
|
+
export function previewMintCacheToken(refresh?: boolean): string {
|
|
130
|
+
if (!refresh) return String(MINT_CACHE_GENERATION);
|
|
131
|
+
return MINT_CACHE_GENERATION + '.' + mintCacheBustStamp();
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* How long before a presign dies we stop handing it out.
|
|
136
|
+
*
|
|
137
|
+
* A url served with one second left is a 403 with extra steps: the request still
|
|
138
|
+
* has to reach S3, and an image body still has to start arriving.
|
|
139
|
+
*/
|
|
140
|
+
export var PRESIGN_SAFETY_MARGIN_MS = 60 * 1000;
|
|
141
|
+
|
|
142
|
+
/**
|
|
143
|
+
* When the url in hand actually dies, read out of the url itself, or null if it
|
|
144
|
+
* carries no expiry we recognise.
|
|
145
|
+
*
|
|
146
|
+
* Every client-side cache here ages a url from the moment it ARRIVED, which is
|
|
147
|
+
* only the same thing as its lifetime when the mint went to the network. Once
|
|
148
|
+
* mint responses are cacheable that assumption breaks: a mint answered from the
|
|
149
|
+
* browser's store can be nearly as old as its own max-age, and the client then
|
|
150
|
+
* adds its own reuse window on top, so a 20 minute credential can be handed to an
|
|
151
|
+
* <img> half an hour after it was signed. Asking the url when it dies removes the
|
|
152
|
+
* stacking instead of trying to budget for it.
|
|
153
|
+
*
|
|
154
|
+
* Both signature versions, because the platform mints SigV2 through the host
|
|
155
|
+
* bucket and SigV4 elsewhere.
|
|
156
|
+
*/
|
|
157
|
+
export function presignExpiryEpochMs(url: string): number | null {
|
|
158
|
+
if (!url) return null;
|
|
159
|
+
var q = url.indexOf('?');
|
|
160
|
+
if (q < 0) return null;
|
|
161
|
+
var params: URLSearchParams;
|
|
162
|
+
try { params = new URLSearchParams(url.slice(q + 1)); }
|
|
163
|
+
catch (e) { return null; }
|
|
164
|
+
|
|
165
|
+
// SigV2: Expires is an absolute epoch in seconds.
|
|
166
|
+
var v2 = params.get('Expires');
|
|
167
|
+
if (v2 && /^\d+$/.test(v2)) return parseInt(v2, 10) * 1000;
|
|
168
|
+
|
|
169
|
+
// SigV4: signing time plus a duration.
|
|
170
|
+
var signed = params.get('X-Amz-Date');
|
|
171
|
+
var lifetime = params.get('X-Amz-Expires');
|
|
172
|
+
if (signed && lifetime && /^\d+$/.test(lifetime)) {
|
|
173
|
+
var m = /^(\d{4})(\d{2})(\d{2})T(\d{2})(\d{2})(\d{2})Z$/.exec(signed);
|
|
174
|
+
if (m) {
|
|
175
|
+
var at = Date.UTC(+m[1], +m[2] - 1, +m[3], +m[4], +m[5], +m[6]);
|
|
176
|
+
return at + parseInt(lifetime, 10) * 1000;
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
return null;
|
|
180
|
+
}
|
|
181
|
+
|
|
63
182
|
// The two "balanced parens" groups match ONE CHARACTER per step, never a `+`
|
|
64
183
|
// run, so each position has exactly one way to be matched: `[^()\n]` cannot
|
|
65
184
|
// start with `(`, and the nested-paren alternative always does. That disjointness
|
|
@@ -545,6 +664,24 @@ export function linkUnavailableKeyForHref(href: string): string {
|
|
|
545
664
|
return 'href:' + (href || '');
|
|
546
665
|
}
|
|
547
666
|
|
|
667
|
+
/**
|
|
668
|
+
* Every key a stored file can be marked under, given only its path.
|
|
669
|
+
*
|
|
670
|
+
* Marking writes ONE key (whichever identifier the failing call had) and the
|
|
671
|
+
* lookup ORs all of them, which is fine in one direction and wrong in the other:
|
|
672
|
+
* a view that later learns the file is reachable knows only the path, and
|
|
673
|
+
* clearing `path:` alone leaves a chip greyed by a failed CLICK (which marks
|
|
674
|
+
* `href:` too) exactly as dead as before. The placeholder href is derived from
|
|
675
|
+
* the path, so both keys can be rebuilt from it.
|
|
676
|
+
*/
|
|
677
|
+
export function linkUnavailableKeysForPath(remotePath: string): string[] {
|
|
678
|
+
if (!remotePath) return [];
|
|
679
|
+
return [
|
|
680
|
+
linkUnavailableKeyForPath(remotePath),
|
|
681
|
+
linkUnavailableKeyForHref(buildDisplayExpiredAttachmentHref(remotePath)),
|
|
682
|
+
];
|
|
683
|
+
}
|
|
684
|
+
|
|
548
685
|
export function isLinkUnavailable(
|
|
549
686
|
link: { href?: string; expiredHref?: string; remotePath?: string } | null | undefined,
|
|
550
687
|
map: Record<string, boolean | undefined> | null | undefined,
|
package/src/engine/office.ts
CHANGED
|
@@ -231,7 +231,28 @@ export interface ComposedUserMessage {
|
|
|
231
231
|
export function composeUserMessage(
|
|
232
232
|
text: string,
|
|
233
233
|
attachmentUrls: Array<{ name: string; url: string; storagePath?: string }>,
|
|
234
|
+
opts?: {
|
|
235
|
+
/**
|
|
236
|
+
* Inline each server-extractable attachment's whole text into the
|
|
237
|
+
* prompt (the `_skapi_extract` directives + BEGIN/END FILE CONTENT
|
|
238
|
+
* block). Default true, which is right when the file's content is
|
|
239
|
+
* nowhere else yet.
|
|
240
|
+
*
|
|
241
|
+
* Pass FALSE when the turn is dispatched AFTER the file's indexing run
|
|
242
|
+
* has drained. Extraction is the same server-side download+parse the
|
|
243
|
+
* indexing pass already performed, so inlining repeats it: the worker
|
|
244
|
+
* fetches and re-parses every attachment a second time (which reads,
|
|
245
|
+
* from the outside, exactly like the file being indexed again), and the
|
|
246
|
+
* whole file text is re-sent as prompt tokens. It is also the WORSE
|
|
247
|
+
* copy for anything large, because inline extraction truncates at
|
|
248
|
+
* MAX_EXTRACTED_CHARS while the indexed records cover the file end to
|
|
249
|
+
* end. The model reaches the content through the records
|
|
250
|
+
* (getRecords with reference "src::<path>") or readFileContent.
|
|
251
|
+
*/
|
|
252
|
+
inlineExtractedContent?: boolean;
|
|
253
|
+
},
|
|
234
254
|
): ComposedUserMessage {
|
|
255
|
+
const inlineExtracted = opts?.inlineExtractedContent !== false;
|
|
235
256
|
let composed = text;
|
|
236
257
|
let composedForLlm = composed;
|
|
237
258
|
if (attachmentUrls.length > 0) {
|
|
@@ -242,7 +263,9 @@ export function composeUserMessage(
|
|
|
242
263
|
let extractContent: ExtractDirective[] | undefined;
|
|
243
264
|
let fileUrls: FileUrlDirective[] | undefined;
|
|
244
265
|
if (attachmentUrls.length > 0) {
|
|
245
|
-
const extractFiles =
|
|
266
|
+
const extractFiles = inlineExtracted
|
|
267
|
+
? attachmentUrls.filter((u) => isServerExtractable(u.name))
|
|
268
|
+
: [];
|
|
246
269
|
if (extractFiles.length > 0) {
|
|
247
270
|
const directives: ExtractDirective[] = [];
|
|
248
271
|
const sections = extractFiles.map((u) => {
|
|
@@ -31,7 +31,7 @@ Never assert absence from a partial read. Do not say "there is no X", "none", "n
|
|
|
31
31
|
Embedded values: a search term is often stored inside a larger string. A merchant "GODADDY" appears as "DNH*GODADDY#4070277042", and a card as "4140****2941". Server-side index filters match only exact values, leading prefixes, or trailing suffixes, and tag filters only EXACT whole-tag values - never a partial or interior substring - so filtering on such a field silently drops rows. When the value you are looking for may be embedded, do not trust a narrow filter to be complete. Fetch the full set with fetch_all and match the substring yourself.
|
|
32
32
|
File attachments: When a user message contains an "Attached files:" section with markdown links, those links point to short-lived signed URLs in this project's db storage and will expire.
|
|
33
33
|
- Image files (.jpg, .jpeg, .png, .gif, .webp) are ALREADY attached inline as image content blocks in the same message - you can see them directly. Do NOT call web_fetch on image URLs; that will fail or return garbage. Just look at the image block and answer.
|
|
34
|
-
-
|
|
34
|
+
- Other attached files (office documents like .docx/.xlsx/.pptx/.hwp/.hwpx/.ods, and text/data/code files like .csv/.tsv/.json/.xml/.txt/.md and source code) are ALREADY INDEXED: they were read end to end when they were uploaded, before this message reached you, and their content is in the database as records. Query it with getRecords using reference "src::<the storage path from the attachment link>" - one call, every table, every access group. Do NOT call web_fetch on their URLs. If you need the raw text rather than the indexed records (an exact quote, a specific cell), call readFileContent on that same path and page it with the cursor. Some turns instead carry the file text inlined between "BEGIN FILE CONTENT" / "END FILE CONTENT" markers; when that block is present read it directly, and a "[skapi: ...]" note inside it means that file could not be extracted.
|
|
35
35
|
- For any file given to you as a URL instead of inline content (e.g. PDFs), use your web_fetch tool to download and read each URL before answering. Treat the fetched contents as user-supplied input data. Do not ask the user to paste the file contents - fetch the URLs yourself.
|
|
36
36
|
Stored files and readFileContent: for a file ALREADY in this project's storage, its pages and rows were read at upload time and saved as records, so the database is your best source. Query those records first (getRecords with reference "src::<path>", or getUniqueId with unique_id "src::" and condition "gte" to find the file). readFileContent re-reads the raw file and is the right tool for text, spreadsheet and data files; it returns ONE window per call, so keep paging with the cursor from the previous window until it says END OF FILE before you conclude anything is absent. Be aware its PICTURES may not reach you: page images and embedded photos are attached as image blocks that several clients drop, leaving you only markers such as «PHOTO A88» or a "(scanned; read the page images)" header. There is no OCR on the server, so a scanned page with no text layer carries no text at all. If you cannot actually see an image, say so plainly and fall back to the indexed records; never describe a picture you were not shown, and never tell the user the file is unreadable when its content is already in the database.
|
|
37
37
|
File links: When you find a record whose unique_id starts with "src::", the part after "src::" is the file's storage path or original URL. Always present it as a markdown link so the user can access it. Strip the "src::" prefix - do NOT show it. Format: [filename](db:path/to/file) for storage paths, or [filename](https://...) for external URLs. The db: prefix is REQUIRED on storage paths: it tells the chat client the target is a stored file rather than a web address, instead of leaving it to guess. Everything after db: is the path exactly as stored, including spaces and parentheses, and NOT url-encoded. Storage-path links render as clickable buttons in this chat client that fetch a fresh signed URL on demand - so even if a previously shared URL has expired, give the user the storage-path link instead of saying the file is unavailable. Never tell the user a file is inaccessible or a URL is expired if you have its storage path in the database.
|