bunnyquery 1.9.1 → 1.9.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "bunnyquery",
3
- "version": "1.9.1",
3
+ "version": "1.9.5",
4
4
  "description": "Embeddable BunnyQuery AI chat widget + its framework-agnostic chat engine",
5
5
  "main": "bunnyquery.js",
6
6
  "exports": {
@@ -323,7 +323,34 @@ export type SplitHistoryResult = {
323
323
  };
324
324
 
325
325
  export async function getSplitChatHistory(
326
- params: { service: string; owner: string; platform: 'claude' | 'openai'; userId?: string },
326
+ params: {
327
+ service: string;
328
+ owner: string;
329
+ platform: 'claude' | 'openai';
330
+ userId?: string;
331
+ /**
332
+ * Scope the SURFACE fetch to this chat's own queue instead of "everything
333
+ * that is not the bg queue".
334
+ *
335
+ * Set for an ANONYMOUS visitor, and only for one. The backend identifies an
336
+ * unauthenticated caller as `ip + "(" + user_agent + ")"`, and the default
337
+ * surface fetch (queue_exclude, no queue) is scoped by exactly that string
338
+ * server side - so two anonymous visitors behind one NAT on the same browser
339
+ * build read each other's transcript, which is the thing per-device history
340
+ * exists to prevent. Reading the device's own queue instead scopes it by a
341
+ * value the client controls and the other device does not share.
342
+ *
343
+ * NOT used for a signed-in caller. Their turns are already scoped by their
344
+ * `sub`, and queue_exact would additionally hide any history sent under a
345
+ * different queue name than the current userId (an older fallback, a
346
+ * pre-rename row), which queue_exclude still returns.
347
+ *
348
+ * The queue name is unguessable but NOT secret: it travels on every request
349
+ * and queue listings are not user-scoped server side. Anonymous transcripts
350
+ * are non-confidential by construction.
351
+ */
352
+ scopeSurfaceToQueue?: boolean;
353
+ },
327
354
  fetchOptions: Record<string, any>,
328
355
  /** Test seam: replaces getChatHistory. Not for production callers. */
329
356
  _fetchImpl?: typeof getChatHistory,
@@ -345,7 +372,34 @@ export async function getSplitChatHistory(
345
372
 
346
373
  async function _getSplitChatHistoryLocked(
347
374
  key: string,
348
- params: { service: string; owner: string; platform: 'claude' | 'openai'; userId?: string },
375
+ params: {
376
+ service: string;
377
+ owner: string;
378
+ platform: 'claude' | 'openai';
379
+ userId?: string;
380
+ /**
381
+ * Scope the SURFACE fetch to this chat's own queue instead of "everything
382
+ * that is not the bg queue".
383
+ *
384
+ * Set for an ANONYMOUS visitor, and only for one. The backend identifies an
385
+ * unauthenticated caller as `ip + "(" + user_agent + ")"`, and the default
386
+ * surface fetch (queue_exclude, no queue) is scoped by exactly that string
387
+ * server side - so two anonymous visitors behind one NAT on the same browser
388
+ * build read each other's transcript, which is the thing per-device history
389
+ * exists to prevent. Reading the device's own queue instead scopes it by a
390
+ * value the client controls and the other device does not share.
391
+ *
392
+ * NOT used for a signed-in caller. Their turns are already scoped by their
393
+ * `sub`, and queue_exact would additionally hide any history sent under a
394
+ * different queue name than the current userId (an older fallback, a
395
+ * pre-rename row), which queue_exclude still returns.
396
+ *
397
+ * The queue name is unguessable but NOT secret: it travels on every request
398
+ * and queue listings are not user-scoped server side. Anonymous transcripts
399
+ * are non-confidential by construction.
400
+ */
401
+ scopeSurfaceToQueue?: boolean;
402
+ },
349
403
  fetchOptions: Record<string, any>,
350
404
  releaseLock: () => void,
351
405
  _fetchImpl?: typeof getChatHistory,
@@ -353,6 +407,14 @@ async function _getSplitChatHistoryLocked(
353
407
  const fetch = _fetchImpl || getChatHistory;
354
408
  const bgQueue = bgIndexingQueueName(params.userId, params.service);
355
409
  const base = { service: params.service, owner: params.owner, platform: params.platform };
410
+ // What the SURFACE (non-background) fetch filters on. `queue_exclude` is the
411
+ // default and returns every row that is not on the bg chain; the anonymous
412
+ // path narrows to this chat's OWN queue instead. Same queue name the chat
413
+ // turns are dispatched under (requests.ts `queue: userId || service`), so the
414
+ // two cannot drift.
415
+ const surfaceScope: Record<string, any> = params.scopeSurfaceToQueue && params.userId
416
+ ? { queue: params.userId, queue_exact: true }
417
+ : { queue_exclude: bgQueue };
356
418
  const fetchMore = !!(fetchOptions && fetchOptions.fetchMore);
357
419
  const limit = fetchOptions && fetchOptions.limit;
358
420
 
@@ -404,14 +466,14 @@ async function _getSplitChatHistoryLocked(
404
466
  } else {
405
467
  const sOpts: any = { fetchMore };
406
468
  if (limit) sOpts.limit = limit;
407
- let s = await fetch({ ...base, queue_exclude: bgQueue }, sOpts);
469
+ let s = await fetch({ ...base, ...surfaceScope }, sOpts);
408
470
  // Loop past empty-but-not-end pages (see SURFACE_EMPTY_MAX_PAGES).
409
471
  let hops = 0;
410
472
  while (s && !s.endOfList && !((s.list || []).length) && hops < SURFACE_EMPTY_MAX_PAGES) {
411
473
  hops++;
412
474
  const nOpts: any = { fetchMore: true };
413
475
  if (limit) nOpts.limit = limit;
414
- s = await fetch({ ...base, queue_exclude: bgQueue }, nOpts);
476
+ s = await fetch({ ...base, ...surfaceScope }, nOpts);
415
477
  }
416
478
  state.pendingSurface = {
417
479
  list: (s && Array.isArray(s.list)) ? s.list : [],
@@ -588,9 +650,53 @@ async function _getSplitChatHistoryLocked(
588
650
  }
589
651
 
590
652
 
653
+ /**
654
+ * THE chat key. Every cache, every ownership stamp and every "is this turn for
655
+ * the chat on screen?" comparison is built here and nowhere else.
656
+ *
657
+ * It used to be written out by hand in four places. When a third segment was
658
+ * added for the chat identity - a browser can hold an anonymous conversation and
659
+ * a signed-in one on the SAME project, and the two must not share a cache - only
660
+ * one of those four was updated. The rest kept producing the two-segment form,
661
+ * so `key !== getHistoryCacheKey()` became permanently true: every send was
662
+ * treated as belonging to another project, the optimistic bubble and the
663
+ * "Thinking..." placeholder were never pushed, and nothing appeared until the
664
+ * server history caught up. One function, so a twin cannot drift again.
665
+ */
666
+ export function chatCacheKey(
667
+ projectId: string | undefined,
668
+ platform: string | undefined,
669
+ userId?: string,
670
+ ): string {
671
+ if (!projectId || platform === 'none') return '';
672
+ return projectId + '#' + platform + '#' + (userId || '');
673
+ }
674
+
675
+ /**
676
+ * The INDEXING scope key: project + platform, deliberately WITHOUT the identity.
677
+ *
678
+ * Claiming, stopping and cancelling a file's indexing are scoped per project and
679
+ * platform because a storage path is project-relative and one ChatSession serves
680
+ * every project. They are NOT per user: an anonymous visitor cannot upload or
681
+ * index at all, so there is no second identity to separate, and folding the
682
+ * identity in here would only have to be threaded through BgTaskEntry to no end.
683
+ *
684
+ * Kept separate from chatCacheKey ON PURPOSE. These two were the same string
685
+ * once, which is exactly how adding a segment to one silently broke the other.
686
+ */
687
+ export function indexScopeKey(
688
+ projectId: string | undefined,
689
+ platform: string | undefined,
690
+ ): string {
691
+ if (!projectId || platform === 'none') return '';
692
+ return projectId + '#' + platform;
693
+ }
694
+
591
695
  export type MapHistoryOptions = {
592
696
  clearedAt: number;
593
697
  projectId: string;
698
+ /** Chat identity, so the `_ownerKey` stamp matches chatCacheKey(). */
699
+ userId?: string;
594
700
  /** View-side display formatter for "Indexing:/Reindexing: …" bubbles. */
595
701
  formatIndexingLabel: (name: string, mime?: string, size?: number | null, storagePath?: string, reindex?: boolean, continued?: boolean) => string;
596
702
  };
@@ -728,7 +834,7 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
728
834
  // unchallenged. That is what let one project's transcript survive on screen
729
835
  // into another project and be persisted under its key.
730
836
  if (opts.projectId) {
731
- var ownerKey = opts.projectId + '#' + platform;
837
+ var ownerKey = chatCacheKey(opts.projectId, platform, opts.userId);
732
838
  for (var oi = 0; oi < mapped.length; oi++) mapped[oi]._ownerKey = ownerKey;
733
839
  }
734
840
  return { messages: mapped, runningItemIds: runningItemIds };
@@ -20,6 +20,19 @@ export interface ChatIdentity {
20
20
  owner: string;
21
21
  /** Per-user queue name (falls back to projectId). */
22
22
  userId: string;
23
+ /**
24
+ * This chat is being used by a visitor with NO account, on a project whose
25
+ * owner allows that.
26
+ *
27
+ * It changes where the MCP tools point. A signed-in turn goes to the MCP
28
+ * server's root endpoint and authenticates with the caller's own token; an
29
+ * anonymous turn has no token to send, and an EMPTY one is worse than none
30
+ * (the server cannot identify a project from it, and an empty credential may
31
+ * be rejected by the provider before the request is even made). So an
32
+ * anonymous turn goes to the project-scoped endpoint instead, which is
33
+ * read-only, restricted to public records, and needs no credential at all.
34
+ */
35
+ anonymous?: boolean;
23
36
  platform: 'claude' | 'openai' | 'none';
24
37
  model?: string;
25
38
  serviceName?: string;
@@ -250,6 +263,22 @@ export interface ChatHost {
250
263
  * whichever pass got there first created the record and the others hoped it had.
251
264
  */
252
265
  ensureFileIndexRecord?(storagePath: string, meta?: { name?: string; mime?: string; size?: number }): Promise<any>;
266
+ /**
267
+ * The access group this file's records must be written at: the uploader's
268
+ * choice, which is the project default or a per-upload answer.
269
+ *
270
+ * Asked PER FILE, and asked AFTER ensureFileIndexRecord has run, because the
271
+ * host is what actually creates the "src::" record and it must report the
272
+ * group it really used. The engine threads the answer into the indexing
273
+ * prompts so the agent's own records land in the same group; a record saved
274
+ * under a different group is in a different table and never comes back with
275
+ * the rest of the file.
276
+ *
277
+ * Optional and may return a promise. A host without it (or one that returns
278
+ * nothing) gets "authorized", which is what every record used before the
279
+ * setting existed.
280
+ */
281
+ uploadAccessGroup?(storagePath: string): 'public' | 'authorized' | 'private' | undefined | Promise<'public' | 'authorized' | 'private' | undefined>;
253
282
  /** Map a relative path to the consumer's db storage key (e.g. uid-prefixed). */
254
283
  storagePathFor(relPath: string): string;
255
284
  getMimeType(name: string): string | null;
@@ -81,6 +81,11 @@ export {
81
81
  __resetSplitHistoryState,
82
82
  type IndexingRequestRef,
83
83
  type MapHistoryOptions,
84
+ // THE two key builders. Exported so a consumer (and a test) can assert the
85
+ // shape rather than rebuild it: a hand-built twin drifting out of step with
86
+ // getHistoryCacheKey is what stopped the chat rendering sent messages.
87
+ chatCacheKey,
88
+ indexScopeKey,
84
89
  } from './history';
85
90
 
86
91
  // Older history is reachable only by scrolling to the top of the message box, so
@@ -36,16 +36,35 @@ export type ChatSystemPromptParams = {
36
36
  * given when the caller says which one this is.
37
37
  */
38
38
  client?: 'console' | 'widget';
39
+ /**
40
+ * The access group THIS project's indexer writes its records at, from the
41
+ * project's `default_access_group` setting.
42
+ *
43
+ * The MCP auto-fills an index/tag query that names a table but no group with
44
+ * "authorized", which used to be right because every BunnyQuery record was
45
+ * hardcoded to it. Now a project can index at "public" (so an anonymous
46
+ * visitor can read it) or "private", and on those projects the auto-fill
47
+ * silently searches a group the data is not in and answers "nothing found".
48
+ * Defaults to 'authorized', which is what an unset project still uses.
49
+ */
50
+ indexAccessGroup?: string;
39
51
  };
40
52
 
41
53
  export function buildChatSystemPrompt(params: ChatSystemPromptParams): string {
42
54
  const { projectId, serviceName, serviceDescription, greeting, canUpload, client } = params;
55
+ // Rendered as the model would have to WRITE it in a tool call: the named
56
+ // aliases go in quotes, a raw group number does not.
57
+ const g = params.indexAccessGroup;
58
+ const indexGroupLiteral =
59
+ typeof g === 'number' ? String(g)
60
+ : (g === 'public' || g === 'private' || g === 'authorized' || g === 'admin') ? `"${g}"`
61
+ : '"authorized"';
43
62
 
44
63
  let systemPrompt = `
45
64
  You are a dedicated assistant for the project ID: "${projectId}".
46
65
  Scope: Only answer questions about this project and its data. Do not answer questions about other projects or topics unrelated to this project. When the user refers to "my database", "my data", or "my files", treat those as references to this project's database and file storage. The ONE exception is BunnyQuery itself - what this app is, what it can do, and how to use it - which is always in scope: answer it from the "About BunnyQuery" section at the end of this prompt.
47
66
  Knowledge lookup: Before saying you don't know or that something isn't in the chat history, ALWAYS query this project's database through the available MCP tools to look for the answer. The user's data is the source of truth - the chat transcript is not. Only respond with "I don't know" or "I couldn't find that" after you have actually searched the project's data and come back empty.
48
- Complete answers over stored data: The database holds one record per spreadsheet row, and each uploaded file becomes many records. ONE file is routinely SPLIT ACROSS SEVERAL TABLES - a summary row in one table, its page or row content in another, its extracted photos and other media in "__MEDIA__", and the indexer often invents a differently-named table on each pass. An index or tag filter matches inside ONE table only and requires table_name: on getRecords, an index or tag sent with table_name but no access_group is auto-filled with access_group "authorized" (where the indexer writes; pass access_group explicitly, including 0, to search another group), while an index or tag WITHOUT table_name FAILS with an error instead of answering, so read the error rather than guessing. Reference is the exception: reference ALONE spans EVERY table and EVERY access group, so getRecords with reference "src::<the file's storage path>" is the one call that returns a whole file's records wherever the indexer put them. Adding table_name narrows it to that table; access_group WITHOUT table_name fails with '"table" is required'; table_name on its own returns that whole table across all access groups. For anything NOT scoped to a single file, call getTables FIRST, run the query once per table that could hold the answer, and combine the results. For any request that counts, sums, totals, lists every match, compares across records, finds which one, or asks whether something is present or ABSENT (for example "how many", "total spent", "which card", "is there any", "없어?", "하나도 없나?"), you MUST read the COMPLETE matching set before answering. Query with fetch_all set to true, or page through getToolResponsePage until pagination.complete is true, across EVERY table and EVERY relevant file. A single default query returns only the first page (about 50 records). That is a SAMPLE. Never treat it as the whole dataset. If you already answered from one table and then realise another table holds more, do not simply apologise: re-run the sweep and give the complete answer.
67
+ Complete answers over stored data: The database holds one record per spreadsheet row, and each uploaded file becomes many records. ONE file is routinely SPLIT ACROSS SEVERAL TABLES - a summary row in one table, its page or row content in another, its extracted photos and other media in "__MEDIA__", and the indexer often invents a differently-named table on each pass. An index or tag filter matches inside ONE table only and requires table_name: on getRecords, an index or tag sent with table_name but no access_group is auto-filled with access_group "authorized", but THIS project indexes at access_group ${indexGroupLiteral}, so pass access_group ${indexGroupLiteral} EXPLICITLY on every index or tag query here - the auto-fill would search a group this project's data is not in and come back empty. Files uploaded before the project's setting changed may sit at another group, so when a scoped query comes back empty, retry it across the other groups (0, 1, "private") before concluding there is nothing, while an index or tag WITHOUT table_name FAILS with an error instead of answering, so read the error rather than guessing. Reference is the exception: reference ALONE spans EVERY table and EVERY access group, so getRecords with reference "src::<the file's storage path>" is the one call that returns a whole file's records wherever the indexer put them. Adding table_name narrows it to that table; access_group WITHOUT table_name fails with '"table" is required'; table_name on its own returns that whole table across all access groups. For anything NOT scoped to a single file, call getTables FIRST, run the query once per table that could hold the answer, and combine the results. For any request that counts, sums, totals, lists every match, compares across records, finds which one, or asks whether something is present or ABSENT (for example "how many", "total spent", "which card", "is there any", "없어?", "하나도 없나?"), you MUST read the COMPLETE matching set before answering. Query with fetch_all set to true, or page through getToolResponsePage until pagination.complete is true, across EVERY table and EVERY relevant file. A single default query returns only the first page (about 50 records). That is a SAMPLE. Never treat it as the whole dataset. If you already answered from one table and then realise another table holds more, do not simply apologise: re-run the sweep and give the complete answer.
49
68
  Never assert absence from a partial read. Do not say "there is no X", "none", "not found", or "아니요, 없습니다" until a complete scan has come back empty. If you have not finished scanning every relevant table and file, keep querying instead of guessing. A confident "no" that later turns out wrong is worse than telling the user you are still checking.
50
69
  Embedded values: a search term is often stored inside a larger string. A merchant "BAKSA" appears as "DNH*BAKSA#4070277042", and a card as "5860****5173". Server-side index filters match only exact values, leading prefixes, or trailing suffixes, and tag filters only EXACT whole-tag values - never a partial or interior substring - so filtering on such a field silently drops rows. When the value you are looking for may be embedded, do not trust a narrow filter to be complete. Fetch the full set with fetch_all and match the substring yourself.
51
70
  File attachments: When a user message contains an "Attached files:" section with markdown links, those links point to short-lived signed URLs in this project's db storage and will expire.
@@ -9,6 +9,7 @@ export { buildChatSystemPrompt, type ChatSystemPromptParams } from './chat_syste
9
9
  export { buildIndexingSystemPrompt, type IndexingSystemPromptParams } from './indexing_system_prompt';
10
10
  export {
11
11
  buildIndexingUserMessage,
12
+ indexingAccessGroup,
12
13
  buildIndexingContinueMessage,
13
14
  buildIndexingRenderMessage,
14
15
  buildIndexingRenderContinueTemplate,
@@ -14,10 +14,21 @@ export type IndexingSystemPromptParams = {
14
14
  serviceName?: string;
15
15
  /** Project description. When present, name + description are appended. */
16
16
  serviceDescription?: string;
17
+ /**
18
+ * Access group every record written during this run must carry. Chosen by the
19
+ * uploader (project default, or a per-upload prompt) and already applied to
20
+ * the "src::" file record before indexing starts. Defaults to "authorized",
21
+ * which is what every record written before this setting existed used.
22
+ */
23
+ accessGroup?: 'public' | 'authorized' | 'private';
17
24
  };
18
25
 
19
26
  export function buildIndexingSystemPrompt(params: IndexingSystemPromptParams): string {
20
27
  const { projectId, serviceName, serviceDescription } = params;
28
+ const accessGroup =
29
+ params.accessGroup === 'public' || params.accessGroup === 'private'
30
+ ? params.accessGroup
31
+ : 'authorized';
21
32
 
22
33
  let systemPrompt =
23
34
  `You are a background indexing agent for project ${projectId}.
@@ -28,7 +39,8 @@ export function buildIndexingSystemPrompt(params: IndexingSystemPromptParams): s
28
39
  - VISION: when the message (a readFileContent window, an embedded PDF page, or an inline attachment) includes IMAGES - scanned/rendered PDF pages, or photos embedded in a spreadsheet next to a row/block - LOOK at them and capture what they show as record data (the reading/values in a scanned table, the part/defect/condition visible in a photo). The image IS part of the data; correlate each photo with its labelled block ("PHOTO A3" markers tie a photo to that grid row).
29
40
  - TRANSCRIBE, DO NOT DESCRIBE. When an image contains ANY text - a label, tag, stamp, form field, serial/part number, handwriting - your FIRST job is to read the characters out and store them VERBATIM, not to describe the scene. A record saying "a red inspection tag with handwritten markings" is worthless: it is unsearchable and every such photo produces the same sentence. Put the characters you can actually read into these EXACT fields, not variations of them: "printed_text" (the pre-printed wording), "handwritten_text" (what a person wrote by hand), and, when you can resolve one, "part_no", "tag_id" and "date". Same reason as the fixed table names: a field called photo_text in one pass and visible_text_notes in the next cannot be queried together. Read PARTIAL values rather than skipping: "500.7402.52__" beats nothing. Only when a character is genuinely unreadable, leave that field null or mark the unreadable span - do NOT invent it, and do NOT replace the whole transcription with a description of what the object looks like. A scene description is a nice extra AFTER the text, never instead of it.
30
41
  - IMAGE FILES uploaded as the file itself: if ANY readable character appears ANYWHERE in the image (a label, a stamp, a sign in the background) it counts as an image WITH text - transcribe it per the rule above, and also capture the layout (what appears where) and every entity named. Only a truly text-free image gets description first: a one-line caption, then the objects present with their attributes (type, color, count, condition, position). Either way, save what you extract onto the file's "src::" record with updateRecords, TAG every entity and identifier visible, and INDEX the one number the image offers (a measured value, an amount, a count).
31
- - Whatever the file type, this file's identity is "src::" + its storage path (the "storage path" metadata line) - never the inline content or a temporary URL. That record ALREADY EXISTS: the upload pipeline creates it in table "file_summaries" (access group "authorized") before indexing starts, so posting it again is rejected as a duplicate unique_id. Reference it from every record you write, and add what you learn to it with updateRecords. If that update unexpectedly reports the record does not exist, post it yourself ONCE with that exact "src::" unique_id (table "file_summaries", access group "authorized") and carry on; this is the ONE exception to the do-NOT-post-the-file-record rules elsewhere in these instructions, because the source identity must never be dropped just because an update failed.
42
+ - Whatever the file type, this file's identity is "src::" + its storage path (the "storage path" metadata line) - never the inline content or a temporary URL. That record ALREADY EXISTS: the upload pipeline creates it in table "file_summaries" (access group "${accessGroup}") before indexing starts, so posting it again is rejected as a duplicate unique_id. Reference it from every record you write, and add what you learn to it with updateRecords. If that update unexpectedly reports the record does not exist, post it yourself ONCE with that exact "src::" unique_id (table "file_summaries", access group "${accessGroup}") and carry on; this is the ONE exception to the do-NOT-post-the-file-record rules elsewhere in these instructions, because the source identity must never be dropped just because an update failed.
43
+ - ACCESS GROUP (hard rule): every record you write for this file - the file record, per-row records, chapters, summaries, intermediates - MUST be posted with access group "${accessGroup}". Pass it explicitly on every postRecords call; do not leave it out and do not vary it between passes of the same file. An access group is part of a record's table key, so records saved under a different group than the file are in a different table and will not come back with the rest of it: a "public" file whose rows were saved as "authorized" is one an anonymous visitor can see the name of and none of the contents of, and a re-index cannot find the strays to clean them up. The one exception is the EXTRACTED MEDIA records in "__MEDIA__", which the pipeline creates for you - leave their group alone and only enrich them.
32
44
  - REACHABILITY (hard rule): every record you write while indexing this file MUST be reachable from the file's "src::<storage path>" record by following reference - either reference that record directly, or reference something that already reaches it. A record with no reference, or one pointing outside this file's chain, is an ORPHAN: deleting or re-indexing the file removes the reachable records and leaves the orphan behind forever, where it keeps turning up in later answers as stale data. If you create an intermediate record that OTHER records reference (a page record that rows hang off, a sheet or section record), set source.can_remove_referencing_records to true on it; the delete cascade passes a delete through a record only when that record carries the flag OR a unique_id starting "src::" (the file record cascades because its unique_id starts with "src::"; the intermediates you create carry no "src::" id, so they need the flag), and it cascades ONE LEVEL AT A TIME, so EVERY intermediate record in a chain needs its own marker - an unmarked link stops the cascade there and everything below it survives as orphans. When in doubt, reference the file record directly and keep the chain flat.
33
45
  - TABULAR data (any spreadsheet - .csv/.tsv/.xlsx/.xls/.ods, or sheet-like rows): you MUST save EVERY data row as its own record (ONE record per row) with that row's actual column values in the record's "data", keyed by the header names, in a table named EXACTLY "spreadsheet_rows". Do NOT summarize, sample only a few rows, or save just file metadata - index the whole sheet, window by window, until it ends. Make MULTIPLE postRecords calls in batches (e.g. 30-50 rows per call) rather than one oversized call. This per-row completeness OVERRIDES brevity. The file-level "src::" record ALREADY EXISTS - the upload pipeline creates it before indexing starts - so do NOT create it. Link EVERY per-row record to it via reference (set each row record's reference to exactly "src::" + the storage path, with NO sheet/window/summary suffix added; the row records themselves do NOT carry a src:: unique_id). Enrich that same record with sheet name(s), column headers and total row count via updateRecords rather than posting another one. The per-row records AND this reference linkage are BOTH mandatory: the linkage is what lets the whole sheet be found and cleaned up together when the file is re-indexed. INDEX each row record on the row's most useful NUMERIC column (named by its header) so rows sort and range-query; when the row has no numeric column, index the grid row number instead. TAG each row record with the sheet name, the file name, and the row's categorical values (a status, a category, a type) - tags are how rows are filtered without scanning the table.
34
46
  - ONE RECORD PER GRID ROW, ALWAYS. "Row" means the numbered row of the sheet (R37 is one record), never a visual block, item, section or left/right pair. Sheets that repeat the same columns side by side (an A/B block beside a C/D block, "paired" or "mirrored" layouts) still get ONE record per grid row, holding BOTH sides - suffix the keys to keep them apart (PART_NO_A / PART_NO_B). Collapsing a 16-row window into 2 or 3 "block" records is the single most damaging mistake here: it silently loses most of the cells and makes every later total wrong, because some windows were counted per row and others per block. If a window shows rows R37 to R52, you save records for R37..R52 and the count you report is the number of grid rows you actually wrote.
@@ -20,6 +20,18 @@ export type IndexingAttachmentInfo = {
20
20
  size?: number;
21
21
  /** Temporary signed URL the agent/MCP fetches to read the file contents. */
22
22
  url: string;
23
+ /**
24
+ * Access group every record extracted from this file must be written at.
25
+ *
26
+ * The uploader chooses it (project default, or a per-upload prompt), and the
27
+ * `src::` file record is already created at this group before indexing starts.
28
+ * The rows, chapters and summaries the agent writes have to MATCH it: skapi's
29
+ * access group is part of a record's table key, so a public file whose rows
30
+ * were saved as "authorized" is a file an anonymous visitor can see the name
31
+ * of and none of the contents of. Omitted means "authorized", which is what
32
+ * every record written before this setting existed used.
33
+ */
34
+ accessGroup?: 'public' | 'authorized' | 'private';
23
35
  };
24
36
 
25
37
  export type BuildIndexingUserMessageOptions = {
@@ -44,6 +56,15 @@ export type BuildIndexingUserMessageOptions = {
44
56
  pagedRead?: boolean;
45
57
  };
46
58
 
59
+ /**
60
+ * The access group to write this file's records at. One place, so the user
61
+ * message, the continue message and the system prompt cannot disagree.
62
+ */
63
+ export function indexingAccessGroup(attachment: { accessGroup?: string }): 'public' | 'authorized' | 'private' {
64
+ const g = attachment && attachment.accessGroup;
65
+ return g === 'public' || g === 'private' ? g : 'authorized';
66
+ }
67
+
47
68
  export function buildIndexingUserMessage(
48
69
  attachment: IndexingAttachmentInfo,
49
70
  options?: BuildIndexingUserMessageOptions,
@@ -54,7 +75,11 @@ export function buildIndexingUserMessage(
54
75
  `- name: ${attachment.name}\n` +
55
76
  `- storage path: ${attachment.storagePath}\n` +
56
77
  (attachment.mime ? `- mime type: ${attachment.mime}\n` : '') +
57
- (typeof attachment.size === 'number' ? `- size (bytes): ${attachment.size}\n` : '');
78
+ (typeof attachment.size === 'number' ? `- size (bytes): ${attachment.size}\n` : '') +
79
+ // Stated in the metadata block as well as the system prompt because this is
80
+ // the per-FILE value: one project can hold public and private files at once,
81
+ // and the system prompt is what is constant across the run.
82
+ `- access group (use this for EVERY record you write for this file): ${indexingAccessGroup(attachment)}\n`;
58
83
 
59
84
  if (options?.inlineContent) {
60
85
  // Parsed client-side (an attachment-parser plugin). The content is already
@@ -169,7 +194,8 @@ function buildRenderMeta(attachment: IndexingAttachmentInfo): string {
169
194
  `File metadata:\n` +
170
195
  `- name: ${attachment.name}\n` +
171
196
  `- storage path: ${attachment.storagePath}\n` +
172
- (attachment.mime ? `- mime type: ${attachment.mime}\n` : '')
197
+ (attachment.mime ? `- mime type: ${attachment.mime}\n` : '') +
198
+ `- access group (use this for EVERY record you write for this file): ${indexingAccessGroup(attachment)}\n`
173
199
  );
174
200
  }
175
201
 
@@ -262,6 +288,7 @@ export function buildIndexingContinueMessage(attachment: IndexingAttachmentInfo)
262
288
  `- name: ${attachment.name}\n` +
263
289
  `- storage path: ${attachment.storagePath}\n` +
264
290
  (attachment.mime ? `- mime type: ${attachment.mime}\n` : '') +
291
+ `- access group (use this for EVERY record you write for this file): ${indexingAccessGroup(attachment)}\n` +
265
292
  `\nRecords for the earlier windows/pages of this file are ALREADY saved (they reference "${src}"). ` +
266
293
  `First call getRecords with reference "${src}" to see how far the previous pass got (the furthest row/window already saved). The reference ALONE is the whole query: it returns every record written from this file across ALL tables and ALL access groups, so do NOT add table_name or access_group to narrow it. The response is PAGED, so keep fetching pages until it reports there are no more, and take the furthest point from the WHOLE set, never from the first page. ` +
267
294
  `Then call readFileContent with the storage path above and a CURSOR that RESUMES just after that point - do NOT start at the beginning. The cursor is derivable from what you already saved:\n` +
@@ -38,6 +38,38 @@ export const DEFAULT_CLAUDE_MODEL = 'claude-sonnet-5';
38
38
  export const DEFAULT_OPENAI_MODEL = 'gpt-5.6-luna';
39
39
 
40
40
  const mcpUrl = () => chatEngineConfig().mcpBaseUrl;
41
+
42
+ /**
43
+ * Where a chat turn's MCP tools point, and what they authenticate with.
44
+ *
45
+ * A SIGNED-IN turn uses the server's root endpoint with the literal
46
+ * '$ACCESS_TOKEN', which the backend substitutes from the caller's
47
+ * `x-access-token` header before the request leaves for the provider.
48
+ *
49
+ * An ANONYMOUS turn has no such header, so that substitution yields an EMPTY
50
+ * credential: `authorization_token: ""` for Claude, `Bearer ` for OpenAI. That is
51
+ * worse than sending none. The MCP server cannot identify a project from an empty
52
+ * token, so every tool call 401s — and a pass whose calls all 401 is exactly what
53
+ * the polling worker classifies as an auth outage, which STOPS the chain. A
54
+ * provider that validates the field would reject the whole request before that.
55
+ *
56
+ * So an anonymous turn points at the project-scoped endpoint `/p/<project id>`
57
+ * and sends NO credential. That route is anonymous by construction: read-only,
58
+ * one project, public records only, and a bearer on it is ignored rather than
59
+ * honoured.
60
+ *
61
+ * The project id is the PUBLIC compound token, matching the route's own pattern
62
+ * and the form the tools accept; the raw regional code would not match.
63
+ */
64
+ function mcpEndpointFor(
65
+ anonymous: boolean | undefined,
66
+ publicProjectId: string | undefined,
67
+ service: string,
68
+ ): { url: string; token?: string } {
69
+ if (!anonymous) return { url: mcpUrl(), token: '$ACCESS_TOKEN' };
70
+ const project = publicProjectId || service;
71
+ return { url: String(mcpUrl()).replace(/\/+$/, '') + '/p/' + project };
72
+ }
41
73
  const clientSecretRequest = (opts: any) => chatEngineConfig().clientSecretRequest(opts);
42
74
 
43
75
  // Resolve the per-image `detail` for OpenAI. The version match tolerates a
@@ -548,7 +580,9 @@ export async function callClaudeWithPublicMcp(
548
580
  fileUrls?: FileUrlDirective[],
549
581
  onResponse?: (res: any) => void,
550
582
  onError?: (err: any) => void,
583
+ mcpScope?: { anonymous?: boolean; publicProjectId?: string },
551
584
  ) {
585
+ const endpoint = mcpEndpointFor(mcpScope?.anonymous, mcpScope?.publicProjectId, service);
552
586
  return callClaudeWithMcp({
553
587
  prompt,
554
588
  messages,
@@ -562,8 +596,10 @@ export async function callClaudeWithPublicMcp(
562
596
  fileUrls,
563
597
  mcpServer: {
564
598
  name: MCP_NAME,
565
- url: mcpUrl(),
566
- authorizationToken: '$ACCESS_TOKEN',
599
+ url: endpoint.url,
600
+ // Omitted entirely for an anonymous turn; the `if (mcpServer.authorizationToken)`
601
+ // guard below drops the key rather than sending an empty one.
602
+ authorizationToken: endpoint.token,
567
603
  },
568
604
  onResponse,
569
605
  onError,
@@ -582,7 +618,9 @@ export async function callOpenAIWithPublicMcp(
582
618
  fileUrls?: FileUrlDirective[],
583
619
  onResponse?: (res: any) => void,
584
620
  onError?: (err: any) => void,
621
+ mcpScope?: { anonymous?: boolean; publicProjectId?: string },
585
622
  ) {
623
+ const endpoint = mcpEndpointFor(mcpScope?.anonymous, mcpScope?.publicProjectId, service);
586
624
  const resolvedModel = model || DEFAULT_OPENAI_MODEL;
587
625
  const imageDetail = getOpenAIImageDetail(resolvedModel);
588
626
  const messageList =
@@ -636,11 +674,14 @@ export async function callOpenAIWithPublicMcp(
636
674
  {
637
675
  type: 'mcp',
638
676
  server_label: MCP_NAME,
639
- server_url: mcpUrl(),
677
+ server_url: endpoint.url,
640
678
  require_approval: 'never',
641
- headers: {
642
- Authorization: 'Bearer $ACCESS_TOKEN',
643
- },
679
+ // No `headers` at all for an anonymous turn: `Bearer ` with an
680
+ // empty token is a credential the MCP server rejects, and the
681
+ // project-scoped endpoint needs none.
682
+ ...(endpoint.token
683
+ ? { headers: { Authorization: 'Bearer ' + endpoint.token } }
684
+ : {}),
644
685
  },
645
686
  ...(OPENAI_WEB_SEARCH_ENABLED
646
687
  ? [
@@ -681,6 +722,13 @@ export type AttachmentSaveInfo = {
681
722
  mime?: string;
682
723
  size?: number;
683
724
  url: string;
725
+ /**
726
+ * Access group this file's records are written at (the uploader's choice,
727
+ * already applied to the "src::" record). Threaded into the indexing
728
+ * prompts so the agent's own records land in the same group; omitted
729
+ * means "authorized", the group everything used before the setting existed.
730
+ */
731
+ accessGroup?: 'public' | 'authorized' | 'private';
684
732
  };
685
733
  /**
686
734
  * Content parsed CLIENT-SIDE by an attachment-parser plugin (e.g. an .hwp
@@ -892,6 +940,9 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
892
940
  projectId: info.publicProjectId || service,
893
941
  serviceName: info.serviceName,
894
942
  serviceDescription: info.serviceDescription,
943
+ // Per-FILE, not per-project: one project holds public and private files at
944
+ // once, so this travels on the attachment rather than the identity.
945
+ accessGroup: attachment.accessGroup,
895
946
  });
896
947
 
897
948
  if (platform === 'openai') {