bunnyquery 1.9.1 → 1.9.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bunnyquery.css +33 -0
- package/bunnyquery.js +331 -41
- package/dist/engine.cjs +107 -27
- package/dist/engine.cjs.map +1 -1
- package/dist/engine.d.mts +147 -3
- package/dist/engine.d.ts +147 -3
- package/dist/engine.mjs +105 -28
- package/dist/engine.mjs.map +1 -1
- package/package.json +1 -1
- package/src/engine/history.ts +111 -5
- package/src/engine/host.ts +29 -0
- package/src/engine/index.ts +5 -0
- package/src/engine/prompts/chat_system_prompt.ts +20 -1
- package/src/engine/prompts/index.ts +1 -0
- package/src/engine/prompts/indexing_system_prompt.ts +13 -1
- package/src/engine/prompts/indexing_user_message.ts +29 -2
- package/src/engine/requests.ts +57 -6
- package/src/engine/session.ts +59 -11
- package/src/widget.css +33 -0
package/package.json
CHANGED
package/src/engine/history.ts
CHANGED
|
@@ -323,7 +323,34 @@ export type SplitHistoryResult = {
|
|
|
323
323
|
};
|
|
324
324
|
|
|
325
325
|
export async function getSplitChatHistory(
|
|
326
|
-
params: {
|
|
326
|
+
params: {
|
|
327
|
+
service: string;
|
|
328
|
+
owner: string;
|
|
329
|
+
platform: 'claude' | 'openai';
|
|
330
|
+
userId?: string;
|
|
331
|
+
/**
|
|
332
|
+
* Scope the SURFACE fetch to this chat's own queue instead of "everything
|
|
333
|
+
* that is not the bg queue".
|
|
334
|
+
*
|
|
335
|
+
* Set for an ANONYMOUS visitor, and only for one. The backend identifies an
|
|
336
|
+
* unauthenticated caller as `ip + "(" + user_agent + ")"`, and the default
|
|
337
|
+
* surface fetch (queue_exclude, no queue) is scoped by exactly that string
|
|
338
|
+
* server side - so two anonymous visitors behind one NAT on the same browser
|
|
339
|
+
* build read each other's transcript, which is the thing per-device history
|
|
340
|
+
* exists to prevent. Reading the device's own queue instead scopes it by a
|
|
341
|
+
* value the client controls and the other device does not share.
|
|
342
|
+
*
|
|
343
|
+
* NOT used for a signed-in caller. Their turns are already scoped by their
|
|
344
|
+
* `sub`, and queue_exact would additionally hide any history sent under a
|
|
345
|
+
* different queue name than the current userId (an older fallback, a
|
|
346
|
+
* pre-rename row), which queue_exclude still returns.
|
|
347
|
+
*
|
|
348
|
+
* The queue name is unguessable but NOT secret: it travels on every request
|
|
349
|
+
* and queue listings are not user-scoped server side. Anonymous transcripts
|
|
350
|
+
* are non-confidential by construction.
|
|
351
|
+
*/
|
|
352
|
+
scopeSurfaceToQueue?: boolean;
|
|
353
|
+
},
|
|
327
354
|
fetchOptions: Record<string, any>,
|
|
328
355
|
/** Test seam: replaces getChatHistory. Not for production callers. */
|
|
329
356
|
_fetchImpl?: typeof getChatHistory,
|
|
@@ -345,7 +372,34 @@ export async function getSplitChatHistory(
|
|
|
345
372
|
|
|
346
373
|
async function _getSplitChatHistoryLocked(
|
|
347
374
|
key: string,
|
|
348
|
-
params: {
|
|
375
|
+
params: {
|
|
376
|
+
service: string;
|
|
377
|
+
owner: string;
|
|
378
|
+
platform: 'claude' | 'openai';
|
|
379
|
+
userId?: string;
|
|
380
|
+
/**
|
|
381
|
+
* Scope the SURFACE fetch to this chat's own queue instead of "everything
|
|
382
|
+
* that is not the bg queue".
|
|
383
|
+
*
|
|
384
|
+
* Set for an ANONYMOUS visitor, and only for one. The backend identifies an
|
|
385
|
+
* unauthenticated caller as `ip + "(" + user_agent + ")"`, and the default
|
|
386
|
+
* surface fetch (queue_exclude, no queue) is scoped by exactly that string
|
|
387
|
+
* server side - so two anonymous visitors behind one NAT on the same browser
|
|
388
|
+
* build read each other's transcript, which is the thing per-device history
|
|
389
|
+
* exists to prevent. Reading the device's own queue instead scopes it by a
|
|
390
|
+
* value the client controls and the other device does not share.
|
|
391
|
+
*
|
|
392
|
+
* NOT used for a signed-in caller. Their turns are already scoped by their
|
|
393
|
+
* `sub`, and queue_exact would additionally hide any history sent under a
|
|
394
|
+
* different queue name than the current userId (an older fallback, a
|
|
395
|
+
* pre-rename row), which queue_exclude still returns.
|
|
396
|
+
*
|
|
397
|
+
* The queue name is unguessable but NOT secret: it travels on every request
|
|
398
|
+
* and queue listings are not user-scoped server side. Anonymous transcripts
|
|
399
|
+
* are non-confidential by construction.
|
|
400
|
+
*/
|
|
401
|
+
scopeSurfaceToQueue?: boolean;
|
|
402
|
+
},
|
|
349
403
|
fetchOptions: Record<string, any>,
|
|
350
404
|
releaseLock: () => void,
|
|
351
405
|
_fetchImpl?: typeof getChatHistory,
|
|
@@ -353,6 +407,14 @@ async function _getSplitChatHistoryLocked(
|
|
|
353
407
|
const fetch = _fetchImpl || getChatHistory;
|
|
354
408
|
const bgQueue = bgIndexingQueueName(params.userId, params.service);
|
|
355
409
|
const base = { service: params.service, owner: params.owner, platform: params.platform };
|
|
410
|
+
// What the SURFACE (non-background) fetch filters on. `queue_exclude` is the
|
|
411
|
+
// default and returns every row that is not on the bg chain; the anonymous
|
|
412
|
+
// path narrows to this chat's OWN queue instead. Same queue name the chat
|
|
413
|
+
// turns are dispatched under (requests.ts `queue: userId || service`), so the
|
|
414
|
+
// two cannot drift.
|
|
415
|
+
const surfaceScope: Record<string, any> = params.scopeSurfaceToQueue && params.userId
|
|
416
|
+
? { queue: params.userId, queue_exact: true }
|
|
417
|
+
: { queue_exclude: bgQueue };
|
|
356
418
|
const fetchMore = !!(fetchOptions && fetchOptions.fetchMore);
|
|
357
419
|
const limit = fetchOptions && fetchOptions.limit;
|
|
358
420
|
|
|
@@ -404,14 +466,14 @@ async function _getSplitChatHistoryLocked(
|
|
|
404
466
|
} else {
|
|
405
467
|
const sOpts: any = { fetchMore };
|
|
406
468
|
if (limit) sOpts.limit = limit;
|
|
407
|
-
let s = await fetch({ ...base,
|
|
469
|
+
let s = await fetch({ ...base, ...surfaceScope }, sOpts);
|
|
408
470
|
// Loop past empty-but-not-end pages (see SURFACE_EMPTY_MAX_PAGES).
|
|
409
471
|
let hops = 0;
|
|
410
472
|
while (s && !s.endOfList && !((s.list || []).length) && hops < SURFACE_EMPTY_MAX_PAGES) {
|
|
411
473
|
hops++;
|
|
412
474
|
const nOpts: any = { fetchMore: true };
|
|
413
475
|
if (limit) nOpts.limit = limit;
|
|
414
|
-
s = await fetch({ ...base,
|
|
476
|
+
s = await fetch({ ...base, ...surfaceScope }, nOpts);
|
|
415
477
|
}
|
|
416
478
|
state.pendingSurface = {
|
|
417
479
|
list: (s && Array.isArray(s.list)) ? s.list : [],
|
|
@@ -588,9 +650,53 @@ async function _getSplitChatHistoryLocked(
|
|
|
588
650
|
}
|
|
589
651
|
|
|
590
652
|
|
|
653
|
+
/**
|
|
654
|
+
* THE chat key. Every cache, every ownership stamp and every "is this turn for
|
|
655
|
+
* the chat on screen?" comparison is built here and nowhere else.
|
|
656
|
+
*
|
|
657
|
+
* It used to be written out by hand in four places. When a third segment was
|
|
658
|
+
* added for the chat identity - a browser can hold an anonymous conversation and
|
|
659
|
+
* a signed-in one on the SAME project, and the two must not share a cache - only
|
|
660
|
+
* one of those four was updated. The rest kept producing the two-segment form,
|
|
661
|
+
* so `key !== getHistoryCacheKey()` became permanently true: every send was
|
|
662
|
+
* treated as belonging to another project, the optimistic bubble and the
|
|
663
|
+
* "Thinking..." placeholder were never pushed, and nothing appeared until the
|
|
664
|
+
* server history caught up. One function, so a twin cannot drift again.
|
|
665
|
+
*/
|
|
666
|
+
export function chatCacheKey(
|
|
667
|
+
projectId: string | undefined,
|
|
668
|
+
platform: string | undefined,
|
|
669
|
+
userId?: string,
|
|
670
|
+
): string {
|
|
671
|
+
if (!projectId || platform === 'none') return '';
|
|
672
|
+
return projectId + '#' + platform + '#' + (userId || '');
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
/**
|
|
676
|
+
* The INDEXING scope key: project + platform, deliberately WITHOUT the identity.
|
|
677
|
+
*
|
|
678
|
+
* Claiming, stopping and cancelling a file's indexing are scoped per project and
|
|
679
|
+
* platform because a storage path is project-relative and one ChatSession serves
|
|
680
|
+
* every project. They are NOT per user: an anonymous visitor cannot upload or
|
|
681
|
+
* index at all, so there is no second identity to separate, and folding the
|
|
682
|
+
* identity in here would only have to be threaded through BgTaskEntry to no end.
|
|
683
|
+
*
|
|
684
|
+
* Kept separate from chatCacheKey ON PURPOSE. These two were the same string
|
|
685
|
+
* once, which is exactly how adding a segment to one silently broke the other.
|
|
686
|
+
*/
|
|
687
|
+
export function indexScopeKey(
|
|
688
|
+
projectId: string | undefined,
|
|
689
|
+
platform: string | undefined,
|
|
690
|
+
): string {
|
|
691
|
+
if (!projectId || platform === 'none') return '';
|
|
692
|
+
return projectId + '#' + platform;
|
|
693
|
+
}
|
|
694
|
+
|
|
591
695
|
export type MapHistoryOptions = {
|
|
592
696
|
clearedAt: number;
|
|
593
697
|
projectId: string;
|
|
698
|
+
/** Chat identity, so the `_ownerKey` stamp matches chatCacheKey(). */
|
|
699
|
+
userId?: string;
|
|
594
700
|
/** View-side display formatter for "Indexing:/Reindexing: …" bubbles. */
|
|
595
701
|
formatIndexingLabel: (name: string, mime?: string, size?: number | null, storagePath?: string, reindex?: boolean, continued?: boolean) => string;
|
|
596
702
|
};
|
|
@@ -728,7 +834,7 @@ export function mapHistoryListToMessages(list: any[], platform: 'claude' | 'open
|
|
|
728
834
|
// unchallenged. That is what let one project's transcript survive on screen
|
|
729
835
|
// into another project and be persisted under its key.
|
|
730
836
|
if (opts.projectId) {
|
|
731
|
-
var ownerKey = opts.projectId
|
|
837
|
+
var ownerKey = chatCacheKey(opts.projectId, platform, opts.userId);
|
|
732
838
|
for (var oi = 0; oi < mapped.length; oi++) mapped[oi]._ownerKey = ownerKey;
|
|
733
839
|
}
|
|
734
840
|
return { messages: mapped, runningItemIds: runningItemIds };
|
package/src/engine/host.ts
CHANGED
|
@@ -20,6 +20,19 @@ export interface ChatIdentity {
|
|
|
20
20
|
owner: string;
|
|
21
21
|
/** Per-user queue name (falls back to projectId). */
|
|
22
22
|
userId: string;
|
|
23
|
+
/**
|
|
24
|
+
* This chat is being used by a visitor with NO account, on a project whose
|
|
25
|
+
* owner allows that.
|
|
26
|
+
*
|
|
27
|
+
* It changes where the MCP tools point. A signed-in turn goes to the MCP
|
|
28
|
+
* server's root endpoint and authenticates with the caller's own token; an
|
|
29
|
+
* anonymous turn has no token to send, and an EMPTY one is worse than none
|
|
30
|
+
* (the server cannot identify a project from it, and an empty credential may
|
|
31
|
+
* be rejected by the provider before the request is even made). So an
|
|
32
|
+
* anonymous turn goes to the project-scoped endpoint instead, which is
|
|
33
|
+
* read-only, restricted to public records, and needs no credential at all.
|
|
34
|
+
*/
|
|
35
|
+
anonymous?: boolean;
|
|
23
36
|
platform: 'claude' | 'openai' | 'none';
|
|
24
37
|
model?: string;
|
|
25
38
|
serviceName?: string;
|
|
@@ -250,6 +263,22 @@ export interface ChatHost {
|
|
|
250
263
|
* whichever pass got there first created the record and the others hoped it had.
|
|
251
264
|
*/
|
|
252
265
|
ensureFileIndexRecord?(storagePath: string, meta?: { name?: string; mime?: string; size?: number }): Promise<any>;
|
|
266
|
+
/**
|
|
267
|
+
* The access group this file's records must be written at: the uploader's
|
|
268
|
+
* choice, which is the project default or a per-upload answer.
|
|
269
|
+
*
|
|
270
|
+
* Asked PER FILE, and asked AFTER ensureFileIndexRecord has run, because the
|
|
271
|
+
* host is what actually creates the "src::" record and it must report the
|
|
272
|
+
* group it really used. The engine threads the answer into the indexing
|
|
273
|
+
* prompts so the agent's own records land in the same group; a record saved
|
|
274
|
+
* under a different group is in a different table and never comes back with
|
|
275
|
+
* the rest of the file.
|
|
276
|
+
*
|
|
277
|
+
* Optional and may return a promise. A host without it (or one that returns
|
|
278
|
+
* nothing) gets "authorized", which is what every record used before the
|
|
279
|
+
* setting existed.
|
|
280
|
+
*/
|
|
281
|
+
uploadAccessGroup?(storagePath: string): 'public' | 'authorized' | 'private' | undefined | Promise<'public' | 'authorized' | 'private' | undefined>;
|
|
253
282
|
/** Map a relative path to the consumer's db storage key (e.g. uid-prefixed). */
|
|
254
283
|
storagePathFor(relPath: string): string;
|
|
255
284
|
getMimeType(name: string): string | null;
|
package/src/engine/index.ts
CHANGED
|
@@ -81,6 +81,11 @@ export {
|
|
|
81
81
|
__resetSplitHistoryState,
|
|
82
82
|
type IndexingRequestRef,
|
|
83
83
|
type MapHistoryOptions,
|
|
84
|
+
// THE two key builders. Exported so a consumer (and a test) can assert the
|
|
85
|
+
// shape rather than rebuild it: a hand-built twin drifting out of step with
|
|
86
|
+
// getHistoryCacheKey is what stopped the chat rendering sent messages.
|
|
87
|
+
chatCacheKey,
|
|
88
|
+
indexScopeKey,
|
|
84
89
|
} from './history';
|
|
85
90
|
|
|
86
91
|
// Older history is reachable only by scrolling to the top of the message box, so
|
|
@@ -36,16 +36,35 @@ export type ChatSystemPromptParams = {
|
|
|
36
36
|
* given when the caller says which one this is.
|
|
37
37
|
*/
|
|
38
38
|
client?: 'console' | 'widget';
|
|
39
|
+
/**
|
|
40
|
+
* The access group THIS project's indexer writes its records at, from the
|
|
41
|
+
* project's `default_access_group` setting.
|
|
42
|
+
*
|
|
43
|
+
* The MCP auto-fills an index/tag query that names a table but no group with
|
|
44
|
+
* "authorized", which used to be right because every BunnyQuery record was
|
|
45
|
+
* hardcoded to it. Now a project can index at "public" (so an anonymous
|
|
46
|
+
* visitor can read it) or "private", and on those projects the auto-fill
|
|
47
|
+
* silently searches a group the data is not in and answers "nothing found".
|
|
48
|
+
* Defaults to 'authorized', which is what an unset project still uses.
|
|
49
|
+
*/
|
|
50
|
+
indexAccessGroup?: string;
|
|
39
51
|
};
|
|
40
52
|
|
|
41
53
|
export function buildChatSystemPrompt(params: ChatSystemPromptParams): string {
|
|
42
54
|
const { projectId, serviceName, serviceDescription, greeting, canUpload, client } = params;
|
|
55
|
+
// Rendered as the model would have to WRITE it in a tool call: the named
|
|
56
|
+
// aliases go in quotes, a raw group number does not.
|
|
57
|
+
const g = params.indexAccessGroup;
|
|
58
|
+
const indexGroupLiteral =
|
|
59
|
+
typeof g === 'number' ? String(g)
|
|
60
|
+
: (g === 'public' || g === 'private' || g === 'authorized' || g === 'admin') ? `"${g}"`
|
|
61
|
+
: '"authorized"';
|
|
43
62
|
|
|
44
63
|
let systemPrompt = `
|
|
45
64
|
You are a dedicated assistant for the project ID: "${projectId}".
|
|
46
65
|
Scope: Only answer questions about this project and its data. Do not answer questions about other projects or topics unrelated to this project. When the user refers to "my database", "my data", or "my files", treat those as references to this project's database and file storage. The ONE exception is BunnyQuery itself - what this app is, what it can do, and how to use it - which is always in scope: answer it from the "About BunnyQuery" section at the end of this prompt.
|
|
47
66
|
Knowledge lookup: Before saying you don't know or that something isn't in the chat history, ALWAYS query this project's database through the available MCP tools to look for the answer. The user's data is the source of truth - the chat transcript is not. Only respond with "I don't know" or "I couldn't find that" after you have actually searched the project's data and come back empty.
|
|
48
|
-
Complete answers over stored data: The database holds one record per spreadsheet row, and each uploaded file becomes many records. ONE file is routinely SPLIT ACROSS SEVERAL TABLES - a summary row in one table, its page or row content in another, its extracted photos and other media in "__MEDIA__", and the indexer often invents a differently-named table on each pass. An index or tag filter matches inside ONE table only and requires table_name: on getRecords, an index or tag sent with table_name but no access_group is auto-filled with access_group "authorized"
|
|
67
|
+
Complete answers over stored data: The database holds one record per spreadsheet row, and each uploaded file becomes many records. ONE file is routinely SPLIT ACROSS SEVERAL TABLES - a summary row in one table, its page or row content in another, its extracted photos and other media in "__MEDIA__", and the indexer often invents a differently-named table on each pass. An index or tag filter matches inside ONE table only and requires table_name: on getRecords, an index or tag sent with table_name but no access_group is auto-filled with access_group "authorized", but THIS project indexes at access_group ${indexGroupLiteral}, so pass access_group ${indexGroupLiteral} EXPLICITLY on every index or tag query here - the auto-fill would search a group this project's data is not in and come back empty. Files uploaded before the project's setting changed may sit at another group, so when a scoped query comes back empty, retry it across the other groups (0, 1, "private") before concluding there is nothing, while an index or tag WITHOUT table_name FAILS with an error instead of answering, so read the error rather than guessing. Reference is the exception: reference ALONE spans EVERY table and EVERY access group, so getRecords with reference "src::<the file's storage path>" is the one call that returns a whole file's records wherever the indexer put them. Adding table_name narrows it to that table; access_group WITHOUT table_name fails with '"table" is required'; table_name on its own returns that whole table across all access groups. For anything NOT scoped to a single file, call getTables FIRST, run the query once per table that could hold the answer, and combine the results. For any request that counts, sums, totals, lists every match, compares across records, finds which one, or asks whether something is present or ABSENT (for example "how many", "total spent", "which card", "is there any", "없어?", "하나도 없나?"), you MUST read the COMPLETE matching set before answering. Query with fetch_all set to true, or page through getToolResponsePage until pagination.complete is true, across EVERY table and EVERY relevant file. A single default query returns only the first page (about 50 records). That is a SAMPLE. Never treat it as the whole dataset. If you already answered from one table and then realise another table holds more, do not simply apologise: re-run the sweep and give the complete answer.
|
|
49
68
|
Never assert absence from a partial read. Do not say "there is no X", "none", "not found", or "아니요, 없습니다" until a complete scan has come back empty. If you have not finished scanning every relevant table and file, keep querying instead of guessing. A confident "no" that later turns out wrong is worse than telling the user you are still checking.
|
|
50
69
|
Embedded values: a search term is often stored inside a larger string. A merchant "BAKSA" appears as "DNH*BAKSA#4070277042", and a card as "5860****5173". Server-side index filters match only exact values, leading prefixes, or trailing suffixes, and tag filters only EXACT whole-tag values - never a partial or interior substring - so filtering on such a field silently drops rows. When the value you are looking for may be embedded, do not trust a narrow filter to be complete. Fetch the full set with fetch_all and match the substring yourself.
|
|
51
70
|
File attachments: When a user message contains an "Attached files:" section with markdown links, those links point to short-lived signed URLs in this project's db storage and will expire.
|
|
@@ -9,6 +9,7 @@ export { buildChatSystemPrompt, type ChatSystemPromptParams } from './chat_syste
|
|
|
9
9
|
export { buildIndexingSystemPrompt, type IndexingSystemPromptParams } from './indexing_system_prompt';
|
|
10
10
|
export {
|
|
11
11
|
buildIndexingUserMessage,
|
|
12
|
+
indexingAccessGroup,
|
|
12
13
|
buildIndexingContinueMessage,
|
|
13
14
|
buildIndexingRenderMessage,
|
|
14
15
|
buildIndexingRenderContinueTemplate,
|
|
@@ -14,10 +14,21 @@ export type IndexingSystemPromptParams = {
|
|
|
14
14
|
serviceName?: string;
|
|
15
15
|
/** Project description. When present, name + description are appended. */
|
|
16
16
|
serviceDescription?: string;
|
|
17
|
+
/**
|
|
18
|
+
* Access group every record written during this run must carry. Chosen by the
|
|
19
|
+
* uploader (project default, or a per-upload prompt) and already applied to
|
|
20
|
+
* the "src::" file record before indexing starts. Defaults to "authorized",
|
|
21
|
+
* which is what every record written before this setting existed used.
|
|
22
|
+
*/
|
|
23
|
+
accessGroup?: 'public' | 'authorized' | 'private';
|
|
17
24
|
};
|
|
18
25
|
|
|
19
26
|
export function buildIndexingSystemPrompt(params: IndexingSystemPromptParams): string {
|
|
20
27
|
const { projectId, serviceName, serviceDescription } = params;
|
|
28
|
+
const accessGroup =
|
|
29
|
+
params.accessGroup === 'public' || params.accessGroup === 'private'
|
|
30
|
+
? params.accessGroup
|
|
31
|
+
: 'authorized';
|
|
21
32
|
|
|
22
33
|
let systemPrompt =
|
|
23
34
|
`You are a background indexing agent for project ${projectId}.
|
|
@@ -28,7 +39,8 @@ export function buildIndexingSystemPrompt(params: IndexingSystemPromptParams): s
|
|
|
28
39
|
- VISION: when the message (a readFileContent window, an embedded PDF page, or an inline attachment) includes IMAGES - scanned/rendered PDF pages, or photos embedded in a spreadsheet next to a row/block - LOOK at them and capture what they show as record data (the reading/values in a scanned table, the part/defect/condition visible in a photo). The image IS part of the data; correlate each photo with its labelled block ("PHOTO A3" markers tie a photo to that grid row).
|
|
29
40
|
- TRANSCRIBE, DO NOT DESCRIBE. When an image contains ANY text - a label, tag, stamp, form field, serial/part number, handwriting - your FIRST job is to read the characters out and store them VERBATIM, not to describe the scene. A record saying "a red inspection tag with handwritten markings" is worthless: it is unsearchable and every such photo produces the same sentence. Put the characters you can actually read into these EXACT fields, not variations of them: "printed_text" (the pre-printed wording), "handwritten_text" (what a person wrote by hand), and, when you can resolve one, "part_no", "tag_id" and "date". Same reason as the fixed table names: a field called photo_text in one pass and visible_text_notes in the next cannot be queried together. Read PARTIAL values rather than skipping: "500.7402.52__" beats nothing. Only when a character is genuinely unreadable, leave that field null or mark the unreadable span - do NOT invent it, and do NOT replace the whole transcription with a description of what the object looks like. A scene description is a nice extra AFTER the text, never instead of it.
|
|
30
41
|
- IMAGE FILES uploaded as the file itself: if ANY readable character appears ANYWHERE in the image (a label, a stamp, a sign in the background) it counts as an image WITH text - transcribe it per the rule above, and also capture the layout (what appears where) and every entity named. Only a truly text-free image gets description first: a one-line caption, then the objects present with their attributes (type, color, count, condition, position). Either way, save what you extract onto the file's "src::" record with updateRecords, TAG every entity and identifier visible, and INDEX the one number the image offers (a measured value, an amount, a count).
|
|
31
|
-
- Whatever the file type, this file's identity is "src::" + its storage path (the "storage path" metadata line) - never the inline content or a temporary URL. That record ALREADY EXISTS: the upload pipeline creates it in table "file_summaries" (access group "
|
|
42
|
+
- Whatever the file type, this file's identity is "src::" + its storage path (the "storage path" metadata line) - never the inline content or a temporary URL. That record ALREADY EXISTS: the upload pipeline creates it in table "file_summaries" (access group "${accessGroup}") before indexing starts, so posting it again is rejected as a duplicate unique_id. Reference it from every record you write, and add what you learn to it with updateRecords. If that update unexpectedly reports the record does not exist, post it yourself ONCE with that exact "src::" unique_id (table "file_summaries", access group "${accessGroup}") and carry on; this is the ONE exception to the do-NOT-post-the-file-record rules elsewhere in these instructions, because the source identity must never be dropped just because an update failed.
|
|
43
|
+
- ACCESS GROUP (hard rule): every record you write for this file - the file record, per-row records, chapters, summaries, intermediates - MUST be posted with access group "${accessGroup}". Pass it explicitly on every postRecords call; do not leave it out and do not vary it between passes of the same file. An access group is part of a record's table key, so records saved under a different group than the file are in a different table and will not come back with the rest of it: a "public" file whose rows were saved as "authorized" is one an anonymous visitor can see the name of and none of the contents of, and a re-index cannot find the strays to clean them up. The one exception is the EXTRACTED MEDIA records in "__MEDIA__", which the pipeline creates for you - leave their group alone and only enrich them.
|
|
32
44
|
- REACHABILITY (hard rule): every record you write while indexing this file MUST be reachable from the file's "src::<storage path>" record by following reference - either reference that record directly, or reference something that already reaches it. A record with no reference, or one pointing outside this file's chain, is an ORPHAN: deleting or re-indexing the file removes the reachable records and leaves the orphan behind forever, where it keeps turning up in later answers as stale data. If you create an intermediate record that OTHER records reference (a page record that rows hang off, a sheet or section record), set source.can_remove_referencing_records to true on it; the delete cascade passes a delete through a record only when that record carries the flag OR a unique_id starting "src::" (the file record cascades because its unique_id starts with "src::"; the intermediates you create carry no "src::" id, so they need the flag), and it cascades ONE LEVEL AT A TIME, so EVERY intermediate record in a chain needs its own marker - an unmarked link stops the cascade there and everything below it survives as orphans. When in doubt, reference the file record directly and keep the chain flat.
|
|
33
45
|
- TABULAR data (any spreadsheet - .csv/.tsv/.xlsx/.xls/.ods, or sheet-like rows): you MUST save EVERY data row as its own record (ONE record per row) with that row's actual column values in the record's "data", keyed by the header names, in a table named EXACTLY "spreadsheet_rows". Do NOT summarize, sample only a few rows, or save just file metadata - index the whole sheet, window by window, until it ends. Make MULTIPLE postRecords calls in batches (e.g. 30-50 rows per call) rather than one oversized call. This per-row completeness OVERRIDES brevity. The file-level "src::" record ALREADY EXISTS - the upload pipeline creates it before indexing starts - so do NOT create it. Link EVERY per-row record to it via reference (set each row record's reference to exactly "src::" + the storage path, with NO sheet/window/summary suffix added; the row records themselves do NOT carry a src:: unique_id). Enrich that same record with sheet name(s), column headers and total row count via updateRecords rather than posting another one. The per-row records AND this reference linkage are BOTH mandatory: the linkage is what lets the whole sheet be found and cleaned up together when the file is re-indexed. INDEX each row record on the row's most useful NUMERIC column (named by its header) so rows sort and range-query; when the row has no numeric column, index the grid row number instead. TAG each row record with the sheet name, the file name, and the row's categorical values (a status, a category, a type) - tags are how rows are filtered without scanning the table.
|
|
34
46
|
- ONE RECORD PER GRID ROW, ALWAYS. "Row" means the numbered row of the sheet (R37 is one record), never a visual block, item, section or left/right pair. Sheets that repeat the same columns side by side (an A/B block beside a C/D block, "paired" or "mirrored" layouts) still get ONE record per grid row, holding BOTH sides - suffix the keys to keep them apart (PART_NO_A / PART_NO_B). Collapsing a 16-row window into 2 or 3 "block" records is the single most damaging mistake here: it silently loses most of the cells and makes every later total wrong, because some windows were counted per row and others per block. If a window shows rows R37 to R52, you save records for R37..R52 and the count you report is the number of grid rows you actually wrote.
|
|
@@ -20,6 +20,18 @@ export type IndexingAttachmentInfo = {
|
|
|
20
20
|
size?: number;
|
|
21
21
|
/** Temporary signed URL the agent/MCP fetches to read the file contents. */
|
|
22
22
|
url: string;
|
|
23
|
+
/**
|
|
24
|
+
* Access group every record extracted from this file must be written at.
|
|
25
|
+
*
|
|
26
|
+
* The uploader chooses it (project default, or a per-upload prompt), and the
|
|
27
|
+
* `src::` file record is already created at this group before indexing starts.
|
|
28
|
+
* The rows, chapters and summaries the agent writes have to MATCH it: skapi's
|
|
29
|
+
* access group is part of a record's table key, so a public file whose rows
|
|
30
|
+
* were saved as "authorized" is a file an anonymous visitor can see the name
|
|
31
|
+
* of and none of the contents of. Omitted means "authorized", which is what
|
|
32
|
+
* every record written before this setting existed used.
|
|
33
|
+
*/
|
|
34
|
+
accessGroup?: 'public' | 'authorized' | 'private';
|
|
23
35
|
};
|
|
24
36
|
|
|
25
37
|
export type BuildIndexingUserMessageOptions = {
|
|
@@ -44,6 +56,15 @@ export type BuildIndexingUserMessageOptions = {
|
|
|
44
56
|
pagedRead?: boolean;
|
|
45
57
|
};
|
|
46
58
|
|
|
59
|
+
/**
|
|
60
|
+
* The access group to write this file's records at. One place, so the user
|
|
61
|
+
* message, the continue message and the system prompt cannot disagree.
|
|
62
|
+
*/
|
|
63
|
+
export function indexingAccessGroup(attachment: { accessGroup?: string }): 'public' | 'authorized' | 'private' {
|
|
64
|
+
const g = attachment && attachment.accessGroup;
|
|
65
|
+
return g === 'public' || g === 'private' ? g : 'authorized';
|
|
66
|
+
}
|
|
67
|
+
|
|
47
68
|
export function buildIndexingUserMessage(
|
|
48
69
|
attachment: IndexingAttachmentInfo,
|
|
49
70
|
options?: BuildIndexingUserMessageOptions,
|
|
@@ -54,7 +75,11 @@ export function buildIndexingUserMessage(
|
|
|
54
75
|
`- name: ${attachment.name}\n` +
|
|
55
76
|
`- storage path: ${attachment.storagePath}\n` +
|
|
56
77
|
(attachment.mime ? `- mime type: ${attachment.mime}\n` : '') +
|
|
57
|
-
(typeof attachment.size === 'number' ? `- size (bytes): ${attachment.size}\n` : '')
|
|
78
|
+
(typeof attachment.size === 'number' ? `- size (bytes): ${attachment.size}\n` : '') +
|
|
79
|
+
// Stated in the metadata block as well as the system prompt because this is
|
|
80
|
+
// the per-FILE value: one project can hold public and private files at once,
|
|
81
|
+
// and the system prompt is what is constant across the run.
|
|
82
|
+
`- access group (use this for EVERY record you write for this file): ${indexingAccessGroup(attachment)}\n`;
|
|
58
83
|
|
|
59
84
|
if (options?.inlineContent) {
|
|
60
85
|
// Parsed client-side (an attachment-parser plugin). The content is already
|
|
@@ -169,7 +194,8 @@ function buildRenderMeta(attachment: IndexingAttachmentInfo): string {
|
|
|
169
194
|
`File metadata:\n` +
|
|
170
195
|
`- name: ${attachment.name}\n` +
|
|
171
196
|
`- storage path: ${attachment.storagePath}\n` +
|
|
172
|
-
(attachment.mime ? `- mime type: ${attachment.mime}\n` : '')
|
|
197
|
+
(attachment.mime ? `- mime type: ${attachment.mime}\n` : '') +
|
|
198
|
+
`- access group (use this for EVERY record you write for this file): ${indexingAccessGroup(attachment)}\n`
|
|
173
199
|
);
|
|
174
200
|
}
|
|
175
201
|
|
|
@@ -262,6 +288,7 @@ export function buildIndexingContinueMessage(attachment: IndexingAttachmentInfo)
|
|
|
262
288
|
`- name: ${attachment.name}\n` +
|
|
263
289
|
`- storage path: ${attachment.storagePath}\n` +
|
|
264
290
|
(attachment.mime ? `- mime type: ${attachment.mime}\n` : '') +
|
|
291
|
+
`- access group (use this for EVERY record you write for this file): ${indexingAccessGroup(attachment)}\n` +
|
|
265
292
|
`\nRecords for the earlier windows/pages of this file are ALREADY saved (they reference "${src}"). ` +
|
|
266
293
|
`First call getRecords with reference "${src}" to see how far the previous pass got (the furthest row/window already saved). The reference ALONE is the whole query: it returns every record written from this file across ALL tables and ALL access groups, so do NOT add table_name or access_group to narrow it. The response is PAGED, so keep fetching pages until it reports there are no more, and take the furthest point from the WHOLE set, never from the first page. ` +
|
|
267
294
|
`Then call readFileContent with the storage path above and a CURSOR that RESUMES just after that point - do NOT start at the beginning. The cursor is derivable from what you already saved:\n` +
|
package/src/engine/requests.ts
CHANGED
|
@@ -38,6 +38,38 @@ export const DEFAULT_CLAUDE_MODEL = 'claude-sonnet-5';
|
|
|
38
38
|
export const DEFAULT_OPENAI_MODEL = 'gpt-5.6-luna';
|
|
39
39
|
|
|
40
40
|
const mcpUrl = () => chatEngineConfig().mcpBaseUrl;
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Where a chat turn's MCP tools point, and what they authenticate with.
|
|
44
|
+
*
|
|
45
|
+
* A SIGNED-IN turn uses the server's root endpoint with the literal
|
|
46
|
+
* '$ACCESS_TOKEN', which the backend substitutes from the caller's
|
|
47
|
+
* `x-access-token` header before the request leaves for the provider.
|
|
48
|
+
*
|
|
49
|
+
* An ANONYMOUS turn has no such header, so that substitution yields an EMPTY
|
|
50
|
+
* credential: `authorization_token: ""` for Claude, `Bearer ` for OpenAI. That is
|
|
51
|
+
* worse than sending none. The MCP server cannot identify a project from an empty
|
|
52
|
+
* token, so every tool call 401s — and a pass whose calls all 401 is exactly what
|
|
53
|
+
* the polling worker classifies as an auth outage, which STOPS the chain. A
|
|
54
|
+
* provider that validates the field would reject the whole request before that.
|
|
55
|
+
*
|
|
56
|
+
* So an anonymous turn points at the project-scoped endpoint `/p/<project id>`
|
|
57
|
+
* and sends NO credential. That route is anonymous by construction: read-only,
|
|
58
|
+
* one project, public records only, and a bearer on it is ignored rather than
|
|
59
|
+
* honoured.
|
|
60
|
+
*
|
|
61
|
+
* The project id is the PUBLIC compound token, matching the route's own pattern
|
|
62
|
+
* and the form the tools accept; the raw regional code would not match.
|
|
63
|
+
*/
|
|
64
|
+
function mcpEndpointFor(
|
|
65
|
+
anonymous: boolean | undefined,
|
|
66
|
+
publicProjectId: string | undefined,
|
|
67
|
+
service: string,
|
|
68
|
+
): { url: string; token?: string } {
|
|
69
|
+
if (!anonymous) return { url: mcpUrl(), token: '$ACCESS_TOKEN' };
|
|
70
|
+
const project = publicProjectId || service;
|
|
71
|
+
return { url: String(mcpUrl()).replace(/\/+$/, '') + '/p/' + project };
|
|
72
|
+
}
|
|
41
73
|
const clientSecretRequest = (opts: any) => chatEngineConfig().clientSecretRequest(opts);
|
|
42
74
|
|
|
43
75
|
// Resolve the per-image `detail` for OpenAI. The version match tolerates a
|
|
@@ -548,7 +580,9 @@ export async function callClaudeWithPublicMcp(
|
|
|
548
580
|
fileUrls?: FileUrlDirective[],
|
|
549
581
|
onResponse?: (res: any) => void,
|
|
550
582
|
onError?: (err: any) => void,
|
|
583
|
+
mcpScope?: { anonymous?: boolean; publicProjectId?: string },
|
|
551
584
|
) {
|
|
585
|
+
const endpoint = mcpEndpointFor(mcpScope?.anonymous, mcpScope?.publicProjectId, service);
|
|
552
586
|
return callClaudeWithMcp({
|
|
553
587
|
prompt,
|
|
554
588
|
messages,
|
|
@@ -562,8 +596,10 @@ export async function callClaudeWithPublicMcp(
|
|
|
562
596
|
fileUrls,
|
|
563
597
|
mcpServer: {
|
|
564
598
|
name: MCP_NAME,
|
|
565
|
-
url:
|
|
566
|
-
authorizationToken
|
|
599
|
+
url: endpoint.url,
|
|
600
|
+
// Omitted entirely for an anonymous turn; the `if (mcpServer.authorizationToken)`
|
|
601
|
+
// guard below drops the key rather than sending an empty one.
|
|
602
|
+
authorizationToken: endpoint.token,
|
|
567
603
|
},
|
|
568
604
|
onResponse,
|
|
569
605
|
onError,
|
|
@@ -582,7 +618,9 @@ export async function callOpenAIWithPublicMcp(
|
|
|
582
618
|
fileUrls?: FileUrlDirective[],
|
|
583
619
|
onResponse?: (res: any) => void,
|
|
584
620
|
onError?: (err: any) => void,
|
|
621
|
+
mcpScope?: { anonymous?: boolean; publicProjectId?: string },
|
|
585
622
|
) {
|
|
623
|
+
const endpoint = mcpEndpointFor(mcpScope?.anonymous, mcpScope?.publicProjectId, service);
|
|
586
624
|
const resolvedModel = model || DEFAULT_OPENAI_MODEL;
|
|
587
625
|
const imageDetail = getOpenAIImageDetail(resolvedModel);
|
|
588
626
|
const messageList =
|
|
@@ -636,11 +674,14 @@ export async function callOpenAIWithPublicMcp(
|
|
|
636
674
|
{
|
|
637
675
|
type: 'mcp',
|
|
638
676
|
server_label: MCP_NAME,
|
|
639
|
-
server_url:
|
|
677
|
+
server_url: endpoint.url,
|
|
640
678
|
require_approval: 'never',
|
|
641
|
-
headers:
|
|
642
|
-
|
|
643
|
-
|
|
679
|
+
// No `headers` at all for an anonymous turn: `Bearer ` with an
|
|
680
|
+
// empty token is a credential the MCP server rejects, and the
|
|
681
|
+
// project-scoped endpoint needs none.
|
|
682
|
+
...(endpoint.token
|
|
683
|
+
? { headers: { Authorization: 'Bearer ' + endpoint.token } }
|
|
684
|
+
: {}),
|
|
644
685
|
},
|
|
645
686
|
...(OPENAI_WEB_SEARCH_ENABLED
|
|
646
687
|
? [
|
|
@@ -681,6 +722,13 @@ export type AttachmentSaveInfo = {
|
|
|
681
722
|
mime?: string;
|
|
682
723
|
size?: number;
|
|
683
724
|
url: string;
|
|
725
|
+
/**
|
|
726
|
+
* Access group this file's records are written at (the uploader's choice,
|
|
727
|
+
* already applied to the "src::" record). Threaded into the indexing
|
|
728
|
+
* prompts so the agent's own records land in the same group; omitted
|
|
729
|
+
* means "authorized", the group everything used before the setting existed.
|
|
730
|
+
*/
|
|
731
|
+
accessGroup?: 'public' | 'authorized' | 'private';
|
|
684
732
|
};
|
|
685
733
|
/**
|
|
686
734
|
* Content parsed CLIENT-SIDE by an attachment-parser plugin (e.g. an .hwp
|
|
@@ -892,6 +940,9 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
|
|
|
892
940
|
projectId: info.publicProjectId || service,
|
|
893
941
|
serviceName: info.serviceName,
|
|
894
942
|
serviceDescription: info.serviceDescription,
|
|
943
|
+
// Per-FILE, not per-project: one project holds public and private files at
|
|
944
|
+
// once, so this travels on the attachment rather than the identity.
|
|
945
|
+
accessGroup: attachment.accessGroup,
|
|
895
946
|
});
|
|
896
947
|
|
|
897
948
|
if (platform === 'openai') {
|