bunnyquery 1.8.6 → 1.8.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -43,7 +43,7 @@ export * from './prompts';
43
43
 
44
44
  // Pure helpers (Tier-1.5): error detection, token budgeting, link/path
45
45
  // normalization, and history mapping — shared so both consumers stay identical.
46
- export { getErrorMessage, isErrorResponseBody, isAuthExpiredError, isNonRetryableRequestError } from './errors';
46
+ export { getErrorMessage, isErrorResponseBody, isAuthExpiredError, isNonRetryableRequestError, isProviderApiKeyError } from './errors';
47
47
  export * from './budget';
48
48
  // Per-format UTF-8 declaration for files offered as a download. Shared so a fenced
49
49
  // block and a server-published file open identically in Excel, Word and a browser.
@@ -64,6 +64,12 @@ export {
64
64
  // read it the same way or the two will not group together.
65
65
  isIndexingRequestText,
66
66
  parseIndexingRequestText,
67
+ // One bounded look at the bg-indexing queue: which files still have a live
68
+ // pass. The dbfile browser's "indexed" badge uses this so a file only goes
69
+ // green once the run is confirmed over, not when its src:: record appears.
70
+ fetchLiveIndexingKeys,
71
+ getSplitChatHistory,
72
+ __resetSplitHistoryState,
67
73
  type IndexingRequestRef,
68
74
  type MapHistoryOptions,
69
75
  } from './history';
@@ -94,6 +100,8 @@ export {
94
100
  type IndexingGroup,
95
101
  type IndexingGroupStatus,
96
102
  type BuildDisplayListOptions,
103
+ type RunStubInfo,
104
+ RUN_RECORD_WORKING_STALE_MS,
97
105
  } from './indexing_groups';
98
106
 
99
107
  export {
@@ -105,6 +113,11 @@ export {
105
113
  BG_INDEXING_QUEUE_SUFFIX,
106
114
  bgIndexingQueueName,
107
115
  isBgIndexingQueue,
116
+ indexDoneUniqueId,
117
+ runIndexUniqueId,
118
+ upsertIndexRunRecordSafe,
119
+ type IndexRunStatus,
120
+ type IndexRunPatch,
108
121
  MCP_NAME,
109
122
  DEFAULT_CLAUDE_MODEL,
110
123
  DEFAULT_OPENAI_MODEL,
@@ -116,6 +129,7 @@ export {
116
129
  listClaudeModels,
117
130
  listOpenAIModels,
118
131
  getChatHistory,
132
+ buildHistoryItemFullId,
119
133
  // response extraction
120
134
  extractClaudeText,
121
135
  extractOpenAIText,
@@ -198,6 +198,15 @@ export type IndexingGroup = {
198
198
  * loaded ones. 'status': the queue has not yet said whether this file is still
199
199
  * being worked on, which is the only thing that can end a worker-driven run. */
200
200
  resolvingReason?: 'history' | 'status';
201
+ /** Synthesized from a durable run:: record: none of the run's passes are
202
+ * among the loaded messages (bg history still deferred, or the run is older
203
+ * than the paging cap). Header-and-status only — members/visibleMembers are
204
+ * empty and there is nothing to cancel; the row is replaced by the real
205
+ * group the moment actual passes load (same `key`, so expansion state
206
+ * carries over). */
207
+ stub?: boolean;
208
+ /** The run:: record's stored error text, for a stub row's meta line. */
209
+ stubError?: string;
201
210
  };
202
211
 
203
212
  export type DisplayEntry =
@@ -222,6 +231,14 @@ export type BuildDisplayListOptions = {
222
231
  /** Whether `liveIndexKeys` has been answered at least once for this chat. False
223
232
  * is "we do not know", and a worker-driven run stays unfinished on it. */
224
233
  liveIndexChecked?: boolean;
234
+ /** Files carrying the durable done:: completion marker (one prefix sweep),
235
+ * keyed like IndexingGroup.key (storage path; the bare-name fallback keys
236
+ * of very old prompts simply never match — they keep the queue inference).
237
+ * A marker is PROOF the file was read to the end: it settles a worker-run
238
+ * green without waiting for the queue answer, and it is never withheld by
239
+ * the resolving logic. A live queue hit still outranks it (a re-index in
240
+ * flight whose marker-cascade delete lagged). */
241
+ doneKeys?: { [fileKey: string]: boolean };
225
242
  /** Server item ids of passes that were on a row when the user STOPPED it
226
243
  * (ChatSession.state.stoppedIndexIds). A run holding any of them is a run the
227
244
  * user stopped — see the status derivation for why a stop usually leaves no
@@ -232,8 +249,52 @@ export type BuildDisplayListOptions = {
232
249
  * windowedIndexing). Passed in rather than read from config so this stays a
233
250
  * pure function of its inputs and can be exercised for both settings. */
234
251
  windowedIndexing?: boolean;
252
+ /** Durable run:: records, keyed by STORAGE PATH (the consumer's marker
253
+ * sweep). Each key with no real group in the loaded messages gets a
254
+ * synthesized header-only row (see IndexingGroup.stub), placed by its
255
+ * `started` timestamp. A real group for the same file — matched by key,
256
+ * path, or name — always suppresses the stub: loaded passes are evidence,
257
+ * the record is only a summary. */
258
+ runStubs?: { [storagePath: string]: RunStubInfo };
259
+ /** The platform whose chat this list is for. run:: records are per-FILE,
260
+ * but a chat is per (project, platform): a run started under Claude has
261
+ * its passes in the Claude conversation and is invisible to the
262
+ * OpenAI-scoped queue probe, so its stub could never be covered and never
263
+ * be confirmed — it just sat there, in a chat it did not belong to.
264
+ * Records minted before this was stamped carry no platform and are shown
265
+ * in both, which keeps the leak to the historical set. */
266
+ stubPlatform?: 'claude' | 'openai';
267
+ /** The chat's clear-history horizon (ms epoch). Run records are service-
268
+ * wide and know nothing about a cleared chat, so without this every
269
+ * "Clear chat history" resurrects one row per indexed file. A stub whose
270
+ * run ended (or, unfinished, began) at or before this moment is dropped —
271
+ * unless the queue says the file is live RIGHT NOW, which no horizon can
272
+ * make untrue. */
273
+ stubClearedAt?: number;
274
+ /** Clock injection for tests; defaults to Date.now(). Only run-stub
275
+ * staleness reads it. */
276
+ now?: number;
235
277
  };
236
278
 
279
+ /** The display-relevant fields of a run:: record (see requests.ts
280
+ * runIndexUniqueId for the record's contract). */
281
+ export type RunStubInfo = {
282
+ status: 'working' | 'done' | 'error' | 'cancelled';
283
+ filename?: string;
284
+ started?: number;
285
+ finished?: number;
286
+ error?: string;
287
+ /** Chat that owns this run. Absent on records minted before it was
288
+ * stamped; see BuildDisplayListOptions.stubPlatform. */
289
+ platform?: 'claude' | 'openai';
290
+ };
291
+
292
+ /** A 'working' run record older than this with no live-queue confirmation is
293
+ * treated as unknown rather than live: a chain that died without reaching any
294
+ * error path leaves 'working' dangling, and a row must not spin forever on a
295
+ * claim nothing can end. */
296
+ export const RUN_RECORD_WORKING_STALE_MS = 6 * 60 * 60 * 1000;
297
+
237
298
  // The indexing label is view-formatted (formatIndexingLabel), so parsing it is
238
299
  // the FALLBACK path only: it exists for bubbles restored from a history cache
239
300
  // written before `_indexFile` was stamped. Live and freshly-mapped bubbles carry
@@ -324,12 +385,14 @@ export function buildChatDisplayList(
324
385
  var list = Array.isArray(messages) ? messages : [];
325
386
  var liveIndexKeys = (opts && opts.liveIndexKeys) || {};
326
387
  var liveIndexChecked = !!(opts && opts.liveIndexChecked);
388
+ var doneKeys = (opts && opts.doneKeys) || {};
327
389
  var stoppedIndexIds = (opts && opts.stoppedIndexIds) || {};
328
390
  var windowedIndexing = opts && opts.windowedIndexing !== undefined
329
391
  ? !!opts.windowedIndexing
330
392
  : windowedIndexingEnabled();
331
393
  var hasMoreHistory = !!(opts && opts.hasMoreHistory);
332
394
  var loadingOlderHistory = !!(opts && opts.loadingOlderHistory);
395
+ var stubPlatform = opts && opts.stubPlatform;
333
396
 
334
397
  // One entry per RUN (see IndexingGroup.runKey), addressed by an internal id
335
398
  // while the list is being walked; runKey is assigned at the end, once the
@@ -629,7 +692,13 @@ export function buildChatDisplayList(
629
692
  // permanently false and the two tests below were already carrying the whole
630
693
  // decision. It is gone rather than left as a hook, so this reads as what it
631
694
  // actually is: for a worker-driven run, ONLY the queue can say it is over.
695
+ // The done:: marker is the third — and strongest — disjunct: written
696
+ // only at a chain's true end (backend worker, or a client's own
697
+ // deterministic completion), it answers without any queue round
698
+ // trip. Gated on the queue NOT claiming the file live, so a
699
+ // re-index whose marker-cascade delete lagged still shows yellow.
632
700
  grp.finished = !newestRunOfKey[order[oi]] ||
701
+ (!!doneKeys[grp.key] && !liveIndexKeys[grp.key]) ||
633
702
  (liveIndexChecked && !liveIndexKeys[grp.key]);
634
703
  }
635
704
 
@@ -645,7 +714,7 @@ export function buildChatDisplayList(
645
714
  // newest pass's own outcome, and newest-first paging always has that pass.
646
715
  grp.resolving = false;
647
716
  } else if (grp.mayHaveOlder && loadingOlderHistory &&
648
- !liveIndexKeys[grp.key] && newestRunOfKey[order[oi]]) {
717
+ !liveIndexKeys[grp.key] && !doneKeys[grp.key] && newestRunOfKey[order[oi]]) {
649
718
  // The line this draws, and it is the same line the 'status' branch draws:
650
719
  // withhold what is UNKNOWN, and what is only INFERRED settled. Never
651
720
  // withhold what is PROVEN, in either direction. Two proofs, and an older
@@ -692,15 +761,191 @@ export function buildChatDisplayList(
692
761
  }
693
762
  }
694
763
 
764
+ // --- durable run:: stubs ---------------------------------------------------
765
+ // A run record whose passes are not among the loaded messages still deserves
766
+ // a row: on a fresh open the bg history is deferred, and a run older than the
767
+ // paging cap never loads at all. Synthesized as header-only groups and placed
768
+ // by `started` among the messages' own timestamps; a real group for the same
769
+ // file always wins (see BuildDisplayListOptions.runStubs).
770
+ var stubList: { started: number; group: IndexingGroup }[] = [];
771
+ var runStubs = opts && opts.runStubs;
772
+ if (runStubs) {
773
+ // Two namespaces, deliberately NOT one map: storage paths are
774
+ // project-relative and folder trees repeat basenames routinely, so a
775
+ // flat map let a real group for "2025/report.pdf" delete the stub for
776
+ // "2026/report.pdf" with nothing replacing it. A bare NAME only covers
777
+ // a stub when the group itself has no path (the legacy compact-label
778
+ // case) and the record has no filename of its own to disambiguate.
779
+ var coveredPaths: { [k: string]: boolean } = {};
780
+ var coveredPathlessNames: { [k: string]: boolean } = {};
781
+ for (var ci = 0; ci < order.length; ci++) {
782
+ var cg = groups[order[ci]];
783
+ if (cg.path) { coveredPaths[cg.path] = true; if (cg.key) coveredPaths[cg.key] = true; }
784
+ else if (cg.name) coveredPathlessNames[cg.name] = true;
785
+ else if (cg.key) coveredPaths[cg.key] = true;
786
+ }
787
+ var now = opts && typeof opts.now === 'number' ? opts.now : Date.now();
788
+ var stubClearedAt = (opts && typeof opts.stubClearedAt === 'number' && opts.stubClearedAt > 0)
789
+ ? opts.stubClearedAt : 0;
790
+ for (var sp in runStubs) {
791
+ var rec = runStubs[sp];
792
+ if (!sp || !rec || !rec.status || coveredPaths[sp]) continue;
793
+ var fname = rec.filename || sp.split('/').pop() || sp;
794
+ // A PATHLESS group (legacy compact label) can only be matched by
795
+ // name, and that is still the right match: the file already has a
796
+ // row, so suppressing avoids a duplicate rather than creating a
797
+ // void. What is gone is the reverse — a group WITH a path no longer
798
+ // writes into the name namespace, so it can no longer delete a
799
+ // same-basename stub from another folder.
800
+ if (coveredPathlessNames[fname]) continue;
801
+ // A run recorded under the OTHER platform's chat belongs to that
802
+ // conversation: its passes live in that history (so no real group
803
+ // can ever cover this stub) and the queue probe is platform-scoped
804
+ // (so it can never confirm it). Records with no platform are
805
+ // legacy and keep the old behaviour rather than vanishing.
806
+ if (stubPlatform && rec.platform && rec.platform !== stubPlatform) continue;
807
+ // Mirrors the real groups' status precedence: a live queue hit outranks
808
+ // whatever the record says (a lagged update must not paint a verdict over
809
+ // visible work), then the record's own terminal statuses, then the
810
+ // queue's authoritative ABSENCE, and only then the stated wait.
811
+ var live = !!liveIndexKeys[sp] || !!liveIndexKeys[fname];
812
+ // Cleared-history horizon: a run that ended at or before the clear is
813
+ // part of what the user asked to forget. A live queue hit survives it
814
+ // (see BuildDisplayListOptions.stubClearedAt). A record carrying NO
815
+ // timestamp at all is unplaceable, not old — the horizon cannot judge
816
+ // it, so it is never dropped by one.
817
+ var recWhen = typeof rec.finished === 'number' ? rec.finished
818
+ : typeof rec.started === 'number' ? rec.started : undefined;
819
+ if (stubClearedAt && !live && recWhen !== undefined && recWhen <= stubClearedAt) continue;
820
+ var st: IndexingGroupStatus = 'active';
821
+ var fin = false;
822
+ var res = false;
823
+ var reason: 'history' | 'status' | undefined;
824
+ if (!live) {
825
+ // A done:: marker from the same sweep is as terminal as the record's
826
+ // own 'done' — it settles a dangling 'working' the chain never
827
+ // flipped (mirrors the real-group doneKeys disjunct).
828
+ if (rec.status === 'done' || doneKeys[sp] || doneKeys[fname]) { st = 'done'; fin = true; }
829
+ else if (rec.status === 'error') { st = 'error'; fin = true; }
830
+ else if (rec.status === 'cancelled') { st = 'cancelled'; fin = true; }
831
+ else if (liveIndexChecked) {
832
+ // 'working' is a CLAIM. The queue has now answered, untruncated,
833
+ // and holds nothing for this file — the same proof the real-group
834
+ // ladder settles on. Without this disjunct a stub had no
835
+ // terminator at all: worker-driven runs are never flipped by this
836
+ // client, so the row sat grey forever (the reported "checking
837
+ // status that never resolves").
838
+ st = 'done'; fin = true;
839
+ } else if (typeof rec.started === 'number' && now - rec.started > RUN_RECORD_WORKING_STALE_MS) {
840
+ // Older than any real run and still unconfirmed: a dead claim, not
841
+ // an open question. Same horizon the files page applies.
842
+ st = 'error'; fin = true;
843
+ } else {
844
+ // Genuinely unanswered — the queue reply is one round trip away
845
+ // and this self-heals the moment it lands.
846
+ res = true; reason = 'status';
847
+ }
848
+ }
849
+ var sg: IndexingGroup = {
850
+ key: sp,
851
+ // ONE identity for the run whether it renders from the record or
852
+ // from its loaded passes: the views key the DOM off runKey, so a
853
+ // 'stub:'-prefixed key meant every handoff was an unmount plus a
854
+ // remount somewhere else. Named after the record's start, which
855
+ // the real group below reuses when it has one.
856
+ runKey: 'run:' + sp + '#' + (typeof rec.started === 'number' ? rec.started : 'n'),
857
+ name: fname,
858
+ path: sp,
859
+ mime: undefined,
860
+ size: undefined,
861
+ isReindex: false,
862
+ members: [],
863
+ passCount: 0,
864
+ status: st,
865
+ cancellableIds: [],
866
+ cancelling: false,
867
+ stopped: st === 'cancelled',
868
+ mayHaveOlder: hasMoreHistory,
869
+ anchorIndex: -1,
870
+ anchorId: '',
871
+ visibleMembers: [],
872
+ driver: !isPagedReadFile(fname, undefined) ? 'single'
873
+ : isImageVisionFile(fname, undefined) ? 'worker'
874
+ : (windowedIndexing ? 'worker' : 'client'),
875
+ finished: fin,
876
+ resolving: res,
877
+ resolvingReason: reason,
878
+ stub: true,
879
+ stubError: rec.error || (st === 'error' && !rec.error
880
+ ? 'Indexing did not finish.' : undefined),
881
+ };
882
+ // A record with no `started` cannot be placed in the conversation.
883
+ // Infinity sorts it to the END (nearest the newest turns) instead of
884
+ // above every message the user has ever sent — the row is about
885
+ // something recent, not about the beginning of time.
886
+ stubList.push({ started: typeof rec.started === 'number' ? rec.started : Infinity, group: sg });
887
+ }
888
+ }
889
+
890
+ // --- timestamp-anchor INCOMPLETE real runs ---------------------------------
891
+ // A real run's row renders at its first LOADED pass, and while older history
892
+ // pages in (newest-first) that is a moving target: the run first appears at
893
+ // its newest pass — down by the recent bubbles — then relocates upward as
894
+ // earlier passes arrive, so the row visibly jumps. The run:: record's
895
+ // `started` is a stable anchor for exactly that window: emit the row at its
896
+ // timestamp position instead, and once the true first pass loads
897
+ // (mayHaveOlder false) emission returns to the anchor, which by then IS the
898
+ // same spot. Newest run of the file only — the record describes it, and an
899
+ // older run's passes are already fully placed around it.
900
+ //
901
+ // NOT gated on mayHaveOlder: that made the position depend on WHICH passes
902
+ // happened to be loaded, so a row moved when the sweep resolved, again when
903
+ // the first pass paged in, and again when paging ended. A record with a
904
+ // `started` places the run at ONE spot for as long as the record exists.
905
+ var suppressAnchor: { [runId: string]: boolean } = {};
906
+ if (runStubs) {
907
+ for (var ti2 = 0; ti2 < order.length; ti2++) {
908
+ var tg = groups[order[ti2]];
909
+ if (!newestRunOfKey[order[ti2]]) continue;
910
+ var trec = (tg.path && runStubs[tg.path]) || runStubs[tg.key];
911
+ if (!trec || typeof trec.started !== 'number') continue;
912
+ if (stubPlatform && trec.platform && trec.platform !== stubPlatform) continue;
913
+ suppressAnchor[order[ti2]] = true;
914
+ // Same identity the stub form uses, so a row that upgrades from
915
+ // record-only to loaded-passes keeps its DOM node and its position.
916
+ tg.runKey = 'run:' + (tg.path || tg.key) + '#' + trec.started;
917
+ stubList.push({ started: trec.started, group: tg });
918
+ }
919
+ }
920
+ stubList.sort(function (a, b) { return a.started - b.started; });
921
+
695
922
  var out: DisplayEntry[] = [];
923
+ var si = 0;
696
924
  for (var j = 0; j < list.length; j++) {
925
+ // Splice stubs in by time: everything that started before this message
926
+ // goes above it. Messages without a timestamp decide nothing.
927
+ var mts = list[j] && typeof list[j]._ts === 'number' ? (list[j]._ts as number) : undefined;
928
+ if (mts !== undefined) {
929
+ while (si < stubList.length && stubList[si].started <= mts) {
930
+ out.push({ kind: 'indexing', group: stubList[si].group, index: -1 - si });
931
+ si++;
932
+ }
933
+ }
697
934
  var r = runOfIndex[j];
698
935
  if (r === undefined) {
699
936
  out.push({ kind: 'message', msg: list[j], index: j });
700
937
  continue;
701
938
  }
702
- // Every other member of the run is represented by the row at the anchor.
703
- if (groups[r].anchorIndex === j) out.push({ kind: 'indexing', group: groups[r], index: j });
939
+ // Every other member of the run is represented by the row at the anchor —
940
+ // unless the row is timestamp-anchored (incomplete run with a run::
941
+ // record), in which case the flush above already emitted it.
942
+ if (groups[r].anchorIndex === j && !suppressAnchor[r]) {
943
+ out.push({ kind: 'indexing', group: groups[r], index: j });
944
+ }
945
+ }
946
+ while (si < stubList.length) {
947
+ out.push({ kind: 'indexing', group: stubList[si].group, index: -1 - si });
948
+ si++;
704
949
  }
705
950
  return out;
706
951
  }
@@ -231,7 +231,28 @@ export interface ComposedUserMessage {
231
231
  export function composeUserMessage(
232
232
  text: string,
233
233
  attachmentUrls: Array<{ name: string; url: string; storagePath?: string }>,
234
+ opts?: {
235
+ /**
236
+ * Inline each server-extractable attachment's whole text into the
237
+ * prompt (the `_skapi_extract` directives + BEGIN/END FILE CONTENT
238
+ * block). Default true, which is right when the file's content is
239
+ * nowhere else yet.
240
+ *
241
+ * Pass FALSE when the turn is dispatched AFTER the file's indexing run
242
+ * has drained. Extraction is the same server-side download+parse the
243
+ * indexing pass already performed, so inlining repeats it: the worker
244
+ * fetches and re-parses every attachment a second time (which reads,
245
+ * from the outside, exactly like the file being indexed again), and the
246
+ * whole file text is re-sent as prompt tokens. It is also the WORSE
247
+ * copy for anything large, because inline extraction truncates at
248
+ * MAX_EXTRACTED_CHARS while the indexed records cover the file end to
249
+ * end. The model reaches the content through the records
250
+ * (getRecords with reference "src::<path>") or readFileContent.
251
+ */
252
+ inlineExtractedContent?: boolean;
253
+ },
234
254
  ): ComposedUserMessage {
255
+ const inlineExtracted = opts?.inlineExtractedContent !== false;
235
256
  let composed = text;
236
257
  let composedForLlm = composed;
237
258
  if (attachmentUrls.length > 0) {
@@ -242,7 +263,9 @@ export function composeUserMessage(
242
263
  let extractContent: ExtractDirective[] | undefined;
243
264
  let fileUrls: FileUrlDirective[] | undefined;
244
265
  if (attachmentUrls.length > 0) {
245
- const extractFiles = attachmentUrls.filter((u) => isServerExtractable(u.name));
266
+ const extractFiles = inlineExtracted
267
+ ? attachmentUrls.filter((u) => isServerExtractable(u.name))
268
+ : [];
246
269
  if (extractFiles.length > 0) {
247
270
  const directives: ExtractDirective[] = [];
248
271
  const sections = extractFiles.map((u) => {
@@ -31,7 +31,7 @@ Never assert absence from a partial read. Do not say "there is no X", "none", "n
31
31
  Embedded values: a search term is often stored inside a larger string. A merchant "GODADDY" appears as "DNH*GODADDY#4070277042", and a card as "4140****2941". Server-side index filters match only exact values, leading prefixes, or trailing suffixes, and tag filters only EXACT whole-tag values - never a partial or interior substring - so filtering on such a field silently drops rows. When the value you are looking for may be embedded, do not trust a narrow filter to be complete. Fetch the full set with fetch_all and match the substring yourself.
32
32
  File attachments: When a user message contains an "Attached files:" section with markdown links, those links point to short-lived signed URLs in this project's db storage and will expire.
33
33
  - Image files (.jpg, .jpeg, .png, .gif, .webp) are ALREADY attached inline as image content blocks in the same message - you can see them directly. Do NOT call web_fetch on image URLs; that will fail or return garbage. Just look at the image block and answer.
34
- - Most attached files (office documents like .docx/.xlsx/.pptx/.hwp/.hwpx/.ods, and text/data/code files like .csv/.tsv/.json/.xml/.txt/.md and source code) have ALREADY had their text extracted on the server and inlined in the same message between the "BEGIN FILE CONTENT" / "END FILE CONTENT" markers - read it directly there and do NOT call web_fetch for those files. A "[skapi: ...]" note in that block means the file could not be extracted.
34
+ - Other attached files (office documents like .docx/.xlsx/.pptx/.hwp/.hwpx/.ods, and text/data/code files like .csv/.tsv/.json/.xml/.txt/.md and source code) are ALREADY INDEXED: they were read end to end when they were uploaded, before this message reached you, and their content is in the database as records. Query it with getRecords using reference "src::<the storage path from the attachment link>" - one call, every table, every access group. Do NOT call web_fetch on their URLs. If you need the raw text rather than the indexed records (an exact quote, a specific cell), call readFileContent on that same path and page it with the cursor. Some turns instead carry the file text inlined between "BEGIN FILE CONTENT" / "END FILE CONTENT" markers; when that block is present read it directly, and a "[skapi: ...]" note inside it means that file could not be extracted.
35
35
  - For any file given to you as a URL instead of inline content (e.g. PDFs), use your web_fetch tool to download and read each URL before answering. Treat the fetched contents as user-supplied input data. Do not ask the user to paste the file contents - fetch the URLs yourself.
36
36
  Stored files and readFileContent: for a file ALREADY in this project's storage, its pages and rows were read at upload time and saved as records, so the database is your best source. Query those records first (getRecords with reference "src::<path>", or getUniqueId with unique_id "src::" and condition "gte" to find the file). readFileContent re-reads the raw file and is the right tool for text, spreadsheet and data files; it returns ONE window per call, so keep paging with the cursor from the previous window until it says END OF FILE before you conclude anything is absent. Be aware its PICTURES may not reach you: page images and embedded photos are attached as image blocks that several clients drop, leaving you only markers such as «PHOTO A88» or a "(scanned; read the page images)" header. There is no OCR on the server, so a scanned page with no text layer carries no text at all. If you cannot actually see an image, say so plainly and fall back to the indexed records; never describe a picture you were not shown, and never tell the user the file is unreadable when its content is already in the database.
37
37
  File links: When you find a record whose unique_id starts with "src::", the part after "src::" is the file's storage path or original URL. Always present it as a markdown link so the user can access it. Strip the "src::" prefix - do NOT show it. Format: [filename](db:path/to/file) for storage paths, or [filename](https://...) for external URLs. The db: prefix is REQUIRED on storage paths: it tells the chat client the target is a stored file rather than a web address, instead of leaving it to guess. Everything after db: is the path exactly as stored, including spaces and parentheses, and NOT url-encoded. Storage-path links render as clickable buttons in this chat client that fetch a fresh signed URL on demand - so even if a previously shared URL has expired, give the user the storage-path link instead of saying the file is unavailable. Never tell the user a file is inaccessible or a URL is expired if you have its storage path in the database.