bunnyquery 1.8.5 → 1.8.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,9 @@
13
13
  import { buildIndexingSystemPrompt, buildIndexingUserMessage, buildIndexingContinueMessage, buildIndexingRenderMessage, buildIndexingRenderContinueTemplate, buildIndexingWindowMessage } from './prompts';
14
14
  import { isServerExtractable, isPagedReadFile, isImageVisionFile, isWindowedReadFile, makeExtractPlaceholder, makeRenderPlaceholder, makeWindowPlaceholder, RENDER_PAGES_PER_WINDOW, type ExtractDirective, type FileUrlDirective } from './office';
15
15
  import { chatEngineConfig, pollOpt, windowedIndexingEnabled } from './config';
16
+ // Output sizing lives in budget.ts so the request cap and the reserve the input
17
+ // budget subtracts cannot drift; getMaxOutputTokens also clamps per model.
18
+ import { getMaxOutputTokens } from './budget';
16
19
 
17
20
  export const ANTHROPIC_MESSAGES_API_URL = 'https://api.anthropic.com/v1/messages';
18
21
  const ANTHROPIC_MODELS_API_URL = 'https://api.anthropic.com/v1/models';
@@ -26,13 +29,12 @@ const WEB_FETCH_MAX_CONTENT_TOKENS = 200000;
26
29
 
27
30
  export const OPENAI_RESPONSES_API_URL = 'https://api.openai.com/v1/responses';
28
31
  const OPENAI_MODELS_API_URL = 'https://api.openai.com/v1/models';
29
- const MAX_TOKENS = 25000;
30
32
  const DEFAULT_OPENAI_IMAGE_DETAIL = 'auto';
31
33
  const OPENAI_WEB_SEARCH_ENABLED = true;
32
34
  const OPENAI_WEB_SEARCH_EXTERNAL_WEB_ACCESS = true;
33
35
  export const MCP_NAME = 'BunnyQuery';
34
36
 
35
- export const DEFAULT_CLAUDE_MODEL = 'claude-sonnet-4-6';
37
+ export const DEFAULT_CLAUDE_MODEL = 'claude-sonnet-5';
36
38
  export const DEFAULT_OPENAI_MODEL = 'gpt-5.6-luna';
37
39
 
38
40
  const mcpUrl = () => chatEngineConfig().mcpBaseUrl;
@@ -205,7 +207,7 @@ const isOldestNano = (model?: string) => {
205
207
  // full one.
206
208
  //
207
209
  // This is the lever that does not risk a 400. The output budget is one number for the whole
208
- // pass (MAX_TOKENS, and reasoning is billed against it), so a window of 5 dense pages leaves
210
+ // pass (getMaxOutputTokens, and reasoning is billed against it), so a window of 5 dense pages leaves
209
211
  // a small model a couple of thousand tokens per page and it starts sampling rows instead of
210
212
  // transcribing them - which is exactly the "saved 5 line items" on a page holding twenty.
211
213
  // Halving the window does not raise the cap, it just stops dividing it so many ways, and the
@@ -554,7 +556,7 @@ export async function callClaudeWithPublicMcp(
554
556
  owner,
555
557
  userId,
556
558
  model: model || DEFAULT_CLAUDE_MODEL,
557
- maxTokens: MAX_TOKENS,
559
+ maxTokens: getMaxOutputTokens('claude', model || DEFAULT_CLAUDE_MODEL),
558
560
  system,
559
561
  extractContent,
560
562
  fileUrls,
@@ -622,7 +624,7 @@ export async function callOpenAIWithPublicMcp(
622
624
  },
623
625
  data: {
624
626
  model: resolvedModel,
625
- max_output_tokens: MAX_TOKENS,
627
+ max_output_tokens: getMaxOutputTokens('openai', resolvedModel),
626
628
  ...(extractContent && extractContent.length
627
629
  ? { _skapi_extract: extractContent }
628
630
  : {}),
@@ -716,6 +718,37 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
716
718
  // A CONTINUE pass resumes a large file that a previous pass could not finish.
717
719
  const continuing = !!info.continueIndexing;
718
720
 
721
+ // Durable run record, minted the moment a run's FIRST pass is enqueued (every
722
+ // first-pass site funnels through this function, so one mint covers them all;
723
+ // resume passes belong to the existing run and must not touch it). Ordering is
724
+ // safe by construction: every caller runs its delete-then-repost + src:: mint
725
+ // BEFORE calling here, so the record's `reference: src::<path>` resolves. The
726
+ // dispatch promise is tapped below so an enqueue that never reached the queue
727
+ // closes the record as an error instead of leaving 'working' dangling.
728
+ if (!continuing) {
729
+ upsertIndexRunRecordSafe(service, attachment.storagePath, {
730
+ status: 'working',
731
+ filename: attachment.name,
732
+ started: Date.now(),
733
+ queue: bgIndexingQueueName(info.userId, service),
734
+ platform: platform,
735
+ });
736
+ }
737
+ const tapDispatchFailure = (p: Promise<any>): Promise<any> => {
738
+ if (continuing) return p;
739
+ return p.then(
740
+ (ack: any) => ack,
741
+ (err: any) => {
742
+ upsertIndexRunRecordSafe(service, attachment.storagePath, {
743
+ status: 'error',
744
+ finished: Date.now(),
745
+ error: (err && (err.message || String(err))) || 'The indexing request could not be enqueued.',
746
+ });
747
+ throw err;
748
+ },
749
+ );
750
+ };
751
+
719
752
  // VISION files (PDFs) are delivered as rendered page IMAGES injected into the message by
720
753
  // the worker (`_skapi_render`), because tool-result images render on neither provider.
721
754
  // Both the first pass and every resume pass use this; renderFrom advances the page window.
@@ -816,6 +849,25 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
816
849
  }
817
850
  : {};
818
851
 
852
+ // Ask the worker to re-sign this file's url at EXECUTION time.
853
+ //
854
+ // The url below was minted when the file was uploaded. Indexing for a project
855
+ // runs on ONE FIFO group, one pass at a time, so a row queued behind a bulk
856
+ // upload can wait hours or days before it fires - by which point that url can
857
+ // be dead, the provider's fetch returns nothing, and the file is recorded as
858
+ // indexed with an empty index behind it. The worker rebuilds the body from the
859
+ // row on every delivery, so re-signing there means the url only has to outlive
860
+ // ONE invocation however long the wait was.
861
+ //
862
+ // Sent for every indexing pass, including CONTINUE passes: a chain spanning
863
+ // days is exactly where the baked url is oldest. Harmless where the file's
864
+ // content arrives another way (server-side extraction, rendered pages) - the
865
+ // url still appears as the attachment link, and a dead link in front of the
866
+ // model is worth avoiding either way.
867
+ const skapiFileUrls = attachment.url && attachment.storagePath
868
+ ? { _skapi_file_urls: [{ path: attachment.storagePath, url: attachment.url }] }
869
+ : {};
870
+
819
871
  const userMessage = (visionFile && renderPlaceholder)
820
872
  ? buildIndexingRenderMessage(attachment, renderPlaceholder, renderFrom)
821
873
  : (windowedRead && windowPlaceholder)
@@ -845,7 +897,7 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
845
897
  if (platform === 'openai') {
846
898
  const resolvedModel = info.model || DEFAULT_OPENAI_MODEL;
847
899
  const imageDetail = getOpenAIImageDetail(resolvedModel);
848
- return clientSecretRequest({
900
+ return tapDispatchFailure(clientSecretRequest({
849
901
  clientSecretName: 'openai',
850
902
  queue: bgIndexingQueueName(info.userId, service),
851
903
  service,
@@ -859,12 +911,13 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
859
911
  },
860
912
  data: {
861
913
  model: resolvedModel,
862
- max_output_tokens: MAX_TOKENS,
914
+ max_output_tokens: getMaxOutputTokens('openai', resolvedModel),
863
915
  // Nano-only transcription knobs. Indexing only; see variantIndexingOptions.
864
916
  ...variantIndexingOptions(resolvedModel),
865
917
  ...skapiExtract,
866
918
  ...skapiRender,
867
919
  ...skapiWindow,
920
+ ...skapiFileUrls,
868
921
  input: [
869
922
  { role: 'system', content: systemPrompt },
870
923
  {
@@ -890,11 +943,11 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
890
943
  : []),
891
944
  ],
892
945
  },
893
- });
946
+ }));
894
947
  }
895
948
 
896
949
  const resolvedModel = info.model || DEFAULT_CLAUDE_MODEL;
897
- return clientSecretRequest({
950
+ return tapDispatchFailure(clientSecretRequest({
898
951
  clientSecretName: 'claude',
899
952
  queue: bgIndexingQueueName(info.userId, service),
900
953
  service,
@@ -910,10 +963,11 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
910
963
  },
911
964
  data: {
912
965
  model: resolvedModel,
913
- max_tokens: MAX_TOKENS,
966
+ max_tokens: getMaxOutputTokens('claude', resolvedModel),
914
967
  ...skapiExtract,
915
968
  ...skapiRender,
916
969
  ...skapiWindow,
970
+ ...skapiFileUrls,
917
971
  system: [
918
972
  {
919
973
  type: 'text',
@@ -949,7 +1003,7 @@ export async function notifyAgentSaveAttachment(info: AttachmentSaveInfo) {
949
1003
  },
950
1004
  ],
951
1005
  },
952
- });
1006
+ }));
953
1007
  }
954
1008
 
955
1009
  export function extractClaudeText(response: any) {
@@ -1038,6 +1092,79 @@ export async function listOpenAIModels(service: string, owner: string) {
1038
1092
  // so the chat-history BETWEEN query never includes bg-queue items. '-' (45) works.
1039
1093
  export const BG_INDEXING_QUEUE_SUFFIX = '-bg';
1040
1094
 
1095
+ /**
1096
+ * unique_id of the durable "indexing finished" marker record for a stored file.
1097
+ *
1098
+ * Written by the BACKEND (the polling worker calls the MCP server's
1099
+ * /internal/index-complete at the end of an auto_continue chain whose final
1100
+ * window reported no more content) and, for completions a client knows
1101
+ * deterministically (single-pass settle, client-chain completion token), by the
1102
+ * consumer's mintIndexDoneMarker hook — never by the model.
1103
+ * The marker record carries `reference: "src::<path>"`, so the reindex flow's
1104
+ * delete of the src:: record cascades to it and a re-run starts unmarked.
1105
+ *
1106
+ * Existence semantics: present = the whole file was read to the end. Absent =
1107
+ * unknown (still running, failed partway, indexed before this marker existed,
1108
+ * or a single-pass run minted before the client hook existed) - callers must
1109
+ * fall back to the live-queue probe (fetchLiveIndexingKeys) before reading
1110
+ * absence as anything.
1111
+ */
1112
+ export function indexDoneUniqueId(storagePath: string): string {
1113
+ return 'done::' + storagePath;
1114
+ }
1115
+
1116
+ /**
1117
+ * unique_id of the per-file indexing RUN record.
1118
+ *
1119
+ * One record per storage path, newest run wins (a reindex's delete-then-repost
1120
+ * of src:: cascade-deletes the old record first, exactly like done::). Minted
1121
+ * status='working' by the client the moment it enqueues a run's FIRST pass, and
1122
+ * closed (done/error/cancelled) by whichever side observes the ending: the
1123
+ * worker via the MCP internal routes for worker-driven chains, the client for
1124
+ * deterministic settles, cancels, and dispatch failures. It exists so chat rows
1125
+ * and files-page badges can answer "which runs exist and how did they end"
1126
+ * from ONE records query instead of scanning bg history.
1127
+ *
1128
+ * A 'working' record is a claim, not proof: a chain that dies without reaching
1129
+ * any error path leaves it dangling, so readers must treat a stale 'working'
1130
+ * (old `started`, no live-queue confirmation) as unknown, never as live.
1131
+ */
1132
+ export function runIndexUniqueId(storagePath: string): string {
1133
+ return 'run::' + storagePath;
1134
+ }
1135
+
1136
+ export type IndexRunStatus = 'working' | 'done' | 'error' | 'cancelled';
1137
+
1138
+ export type IndexRunPatch = {
1139
+ status: IndexRunStatus;
1140
+ filename?: string;
1141
+ started?: number;
1142
+ finished?: number;
1143
+ error?: string;
1144
+ queue?: string;
1145
+ /** Chat that owns this run. A run:: record is keyed by storage path alone,
1146
+ * but a chat is per (project, platform) — without this the Claude chat's
1147
+ * runs surfaced as rows in the same project's ChatGPT chat, where their
1148
+ * passes can never load and the queue probe can never see them. */
1149
+ platform?: 'claude' | 'openai';
1150
+ };
1151
+
1152
+ /**
1153
+ * Fire-and-forget wrapper over the consumer's upsertIndexRunRecord hook.
1154
+ * Safe everywhere: missing hook, unconfigured engine, and consumer throws all
1155
+ * reduce to a no-op — a run record must never be able to break the run itself.
1156
+ */
1157
+ export function upsertIndexRunRecordSafe(service: string, storagePath: string, patch: IndexRunPatch): void {
1158
+ if (!service || !storagePath) return;
1159
+ try {
1160
+ const hook = chatEngineConfig().upsertIndexRunRecord;
1161
+ if (typeof hook !== 'function') return;
1162
+ hook({ service, storagePath, patch });
1163
+ } catch (e) {
1164
+ // best-effort by contract
1165
+ }
1166
+ }
1167
+
1041
1168
  /**
1042
1169
  * The one place the background-indexing queue name is spelled out. The backend
1043
1170
  * serialises requests sharing a queue name and runs different names in PARALLEL,
@@ -1133,7 +1260,21 @@ export const CHAT_HISTORY_PAGE_LIMIT = 500;
1133
1260
  * everything already finished.
1134
1261
  */
1135
1262
  export async function getChatHistory(
1136
- params: { service?: string; owner?: string; platform: 'claude' | 'openai'; queue?: string; status?: 'pending' | 'running' | 'resolved' | 'failed' },
1263
+ params: {
1264
+ service?: string; owner?: string; platform: 'claude' | 'openai'; queue?: string;
1265
+ status?: 'pending' | 'running' | 'resolved' | 'failed';
1266
+ /** Exact-queue listing: without it the qid range is a PREFIX match, so
1267
+ * queue "u1" also returns "u1-bg" rows. Requires the updated polling
1268
+ * lambda; older backends ignore it (harmless, wider results). */
1269
+ queue_exact?: boolean;
1270
+ /** Label/marker STUBS instead of full bodies (see the polling lambda).
1271
+ * Older backends ignore it and return full items. */
1272
+ compact?: boolean;
1273
+ /** Drop one queue's rows from an id-prefix listing — how the surface
1274
+ * chat is fetched WITHOUT the bg-indexing queue while legacy items on
1275
+ * odd queue names survive. Older backends ignore it. */
1276
+ queue_exclude?: string;
1277
+ },
1137
1278
  fetchOptions: Record<string, any>,
1138
1279
  ) {
1139
1280
  const url =
@@ -1148,6 +1289,9 @@ export async function getChatHistory(
1148
1289
  { service: params.service, owner: params.owner },
1149
1290
  params.queue ? { queue: params.queue } : {},
1150
1291
  params.status ? { status: params.status } : {},
1292
+ params.queue_exact ? { queue_exact: true } : {},
1293
+ params.compact ? { compact: true } : {},
1294
+ params.queue_exclude ? { queue_exclude: params.queue_exclude } : {},
1151
1295
  );
1152
1296
 
1153
1297
  return chatEngineConfig().clientSecretRequestHistory(
@@ -1155,3 +1299,12 @@ export async function getChatHistory(
1155
1299
  Object.assign({ ascending: false, limit: CHAT_HISTORY_PAGE_LIMIT }, fetchOptions),
1156
1300
  );
1157
1301
  }
1302
+
1303
+ /** Full server-side id of one history item, for a csr-poll POINT LOOKUP (the
1304
+ * single-item path returns the item WITH bodies — how an expanded row fetches
1305
+ * the passes a compact listing stubbed out). Mirrors the id the SDK builds:
1306
+ * `[METHOD]url#service:` + the item's own `stamp:entropy` id. */
1307
+ export function buildHistoryItemFullId(platform: 'claude' | 'openai', service: string, itemId: string): string {
1308
+ const url = platform === 'claude' ? ANTHROPIC_MESSAGES_API_URL : OPENAI_RESPONSES_API_URL;
1309
+ return `[POST]${url.toLowerCase()}#${service}:${itemId}`;
1310
+ }