@k2b/cloud 0.25.0 → 0.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/package.json +3 -3
  2. package/src/_internal/capabilities.ts +12 -0
  3. package/src/_internal/define-app.ts +8 -1
  4. package/src/_internal/process-identity.ts +7 -1
  5. package/src/_internal/registry-validation.ts +3 -0
  6. package/src/_internal/registry.ts +1 -0
  7. package/src/_internal/runtime-context.ts +1 -0
  8. package/src/access/GroupCoverage.tsx +175 -0
  9. package/src/access/PermissionEditor.tsx +119 -99
  10. package/src/access/messages.ts +30 -0
  11. package/src/ai/admin.ts +1 -0
  12. package/src/ai/approval-routes.ts +5 -5
  13. package/src/ai/browser-code-contracts.ts +14 -2
  14. package/src/ai/browser.ts +8 -1
  15. package/src/ai/capabilities.ts +115 -33
  16. package/src/ai/chat/blocks.tsx +86 -179
  17. package/src/ai/chat/builtin-tools.tsx +83 -48
  18. package/src/ai/chat/file-tools.tsx +4 -1
  19. package/src/ai/chat/live-turn.browser-harness.tsx +44 -0
  20. package/src/ai/chat/message-actions.tsx +6 -2
  21. package/src/ai/chat/message-utils.ts +17 -14
  22. package/src/ai/chat/messages.ts +330 -2
  23. package/src/ai/chat/presentation.tsx +278 -104
  24. package/src/ai/chat/tool-groups.ts +55 -35
  25. package/src/ai/chat/turn-layout.ts +141 -0
  26. package/src/ai/chat/turn-view.tsx +644 -0
  27. package/src/ai/client/controller.ts +60 -31
  28. package/src/ai/client/file-source.ts +20 -3
  29. package/src/ai/client/projection.ts +42 -6
  30. package/src/ai/code-mode-skill.ts +27 -27
  31. package/src/ai/code-runtime-tools.ts +10 -1
  32. package/src/ai/code-source-contracts.ts +54 -4
  33. package/src/ai/code-source-tools.ts +10 -3
  34. package/src/ai/credentials.ts +17 -3
  35. package/src/ai/data-analysis-skill.ts +2 -2
  36. package/src/ai/default-tools.ts +2 -2
  37. package/src/ai/executor.ts +177 -80
  38. package/src/ai/file-context.ts +14 -2
  39. package/src/ai/file-tools.ts +17 -3
  40. package/src/ai/files-store.ts +134 -11
  41. package/src/ai/grids-skill.ts +2 -2
  42. package/src/ai/index.ts +7 -0
  43. package/src/ai/memories.ts +14 -0
  44. package/src/ai/migrate.ts +125 -0
  45. package/src/ai/model-request-settings.ts +98 -0
  46. package/src/ai/protocol.ts +26 -4
  47. package/src/ai/provider-fetch.ts +67 -15
  48. package/src/ai/provider-retry.ts +105 -0
  49. package/src/ai/provider.ts +7 -1
  50. package/src/ai/quota-provider.ts +16 -7
  51. package/src/ai/request-headers.ts +117 -0
  52. package/src/ai/routes.ts +34 -6
  53. package/src/ai/runtime.ts +1 -1
  54. package/src/ai/settings.ts +19 -2
  55. package/src/ai/skill-seeds.ts +31 -3
  56. package/src/ai/skills.ts +26 -0
  57. package/src/ai/solid.ts +1 -1
  58. package/src/ai/store.ts +202 -59
  59. package/src/ai/stream.ts +182 -37
  60. package/src/ai/structured.ts +20 -5
  61. package/src/ai/system-prompt.ts +25 -0
  62. package/src/ai/timeline.ts +9 -11
  63. package/src/ai/tool-call-names.ts +45 -0
  64. package/src/ai/turn-policy.ts +247 -0
  65. package/src/ai/turn-timing.ts +31 -3
  66. package/src/ai/types.ts +36 -5
  67. package/src/api/admin-ai-quotas.ts +36 -1
  68. package/src/api/admin-core-settings.ts +16 -23
  69. package/src/api/admin-outgoing-mail.ts +62 -0
  70. package/src/api/index.ts +2 -0
  71. package/src/cli/admin/ai-quotas.ts +70 -1
  72. package/src/cli/admin/index.ts +6 -0
  73. package/src/cli/admin/outgoing-mail.ts +118 -0
  74. package/src/contracts/app.ts +2 -0
  75. package/src/contracts/index.ts +1 -0
  76. package/src/contracts/outgoing-mail.ts +77 -0
  77. package/src/contracts/registry.ts +4 -0
  78. package/src/services/index.ts +3 -0
  79. package/src/services/notifications/email.ts +16 -26
  80. package/src/services/outgoing-mail/index.ts +19 -0
  81. package/src/services/outgoing-mail/store.ts +286 -0
  82. package/src/services/outgoing-mail/test-send.ts +40 -0
  83. package/src/services/outgoing-mail/transport.ts +13 -0
  84. package/src/services/settings/core-settings.ts +1 -38
  85. package/src/services/settings/store.ts +5 -1
  86. package/src/shared/ai-model-request-settings.ts +21 -0
  87. package/src/shared/ai-platform-prompt.ts +1 -1
  88. package/src/shared/ai-request-options.ts +185 -0
  89. package/src/shared/app-presentation.ts +10 -2
  90. package/src/ssr/admin-navigation.ts +1 -1
  91. package/src/ssr/platform-messages.ts +2 -0
  92. package/src/ssr/workspace-navigation.ts +7 -1
  93. package/src/styles/effects.css +69 -0
@@ -1,27 +1,63 @@
1
1
  import { AsyncLocalStorage } from "node:async_hooks";
2
2
 
3
3
  /**
4
- * Dates the provider response headers and the first body byte of the inference
5
- * call that is currently streaming, without touching the wire. nessi adapters
6
- * call the global `fetch`; the wrapper is installed once, on first use, and
7
- * only observes requests made inside `runWithProviderFetchMarks`.
4
+ * Observes the provider request of the inference call that is currently
5
+ * running, without touching the wire: when its headers and first body byte
6
+ * arrived, whether the provider certainly did not process it, and how long a
7
+ * rejecting provider asked the caller to wait. nessi adapters call the global
8
+ * `fetch`; the wrapper is installed once, on first use, and only observes
9
+ * requests made inside `runWithProviderFetchMarks`. Scopes nest, so an outer
10
+ * retry policy and the inner per-call accounting each see the same request.
8
11
  *
9
12
  * Nothing here logs: frames may carry reasoning text and headers carry keys.
10
13
  */
11
- export type ProviderFetchMarks = { headersAt?: number; firstByteAt?: number };
14
+ export type ProviderFetchMarks = {
15
+ headersAt?: number;
16
+ firstByteAt?: number;
17
+ /** The request never reached the provider, or the provider answered with a status that says it did not process it. */
18
+ refused?: boolean;
19
+ retryAfterMs?: number;
20
+ };
12
21
 
13
- const marks = new AsyncLocalStorage<ProviderFetchMarks>();
22
+ const scopes = new AsyncLocalStorage<readonly ProviderFetchMarks[]>();
14
23
  let installed = false;
15
24
 
16
25
  export const runWithProviderFetchMarks = <T>(store: ProviderFetchMarks, run: () => Promise<T>): Promise<T> => {
17
26
  install();
18
- return marks.run(store, run);
27
+ return scopes.run([...(scopes.getStore() ?? []), store], run);
28
+ };
29
+
30
+ /** `retry-after-ms` (OpenAI) wins over the standard `retry-after` seconds or HTTP date. */
31
+ export const retryAfterMs = (headers: Headers, now = Date.now()): number | undefined => {
32
+ const milliseconds = Number(headers.get("retry-after-ms")?.trim() || Number.NaN);
33
+ if (Number.isFinite(milliseconds) && milliseconds >= 0) return milliseconds;
34
+ const value = headers.get("retry-after")?.trim();
35
+ if (!value) return undefined;
36
+ const seconds = Number(value);
37
+ if (Number.isFinite(seconds)) return seconds >= 0 ? seconds * 1_000 : undefined;
38
+ const date = Date.parse(value);
39
+ return Number.isFinite(date) ? Math.max(0, date - now) : undefined;
19
40
  };
20
41
 
21
- const firstByteObserver = (store: ProviderFetchMarks) =>
42
+ /**
43
+ * A client error, 503, or Anthropic's overloaded 529 says the provider did not
44
+ * process the request. Other server errors, such as a gateway's 502 or 504, can
45
+ * follow processing upstream.
46
+ */
47
+ const refusedStatus = (status: number) => (status >= 400 && status < 500) || status === 503 || status === 529;
48
+
49
+ /** Bun's codes for a request that never left: no connection, or no address for the host. */
50
+ const unsentCodes = new Set(["ConnectionRefused", "ENOTFOUND"]);
51
+ const unsent = (error: unknown) =>
52
+ typeof error === "object" && error !== null && "code" in error && typeof error.code === "string" && unsentCodes.has(error.code);
53
+
54
+ const firstByteObserver = (stores: readonly ProviderFetchMarks[]) =>
22
55
  new TransformStream<Uint8Array, Uint8Array>({
23
56
  transform(chunk, controller) {
24
- if (chunk.byteLength > 0 && store.firstByteAt === undefined) store.firstByteAt = Date.now();
57
+ if (chunk.byteLength > 0)
58
+ for (const store of stores) {
59
+ store.firstByteAt ??= Date.now();
60
+ }
25
61
  controller.enqueue(chunk);
26
62
  },
27
63
  });
@@ -31,13 +67,29 @@ const install = (): void => {
31
67
  installed = true;
32
68
  const realFetch = globalThis.fetch;
33
69
  const instrumented = async (input: RequestInfo | URL, init?: RequestInit): Promise<Response> => {
34
- const store = marks.getStore();
35
- // Only the first request of a call is the provider request; nessi retries create a new call.
36
- if (store === undefined || store.headersAt !== undefined) return realFetch(input, init);
37
- const response = await realFetch(input, init);
38
- store.headersAt = Date.now();
70
+ // Only the first request of a scope is its provider request; a retry opens a new scope.
71
+ const stores = (scopes.getStore() ?? []).filter((store) => store.headersAt === undefined);
72
+ if (stores.length === 0) return realFetch(input, init);
73
+ let response: Response;
74
+ try {
75
+ response = await realFetch(input, init);
76
+ } catch (error) {
77
+ // A reset, an abort, or a timeout can follow delivery, so only an unsent request counts as refused.
78
+ if (unsent(error))
79
+ for (const store of stores) {
80
+ store.refused = true;
81
+ }
82
+ throw error;
83
+ }
84
+ const headersAt = Date.now();
85
+ const wait = response.ok ? undefined : retryAfterMs(response.headers, headersAt);
86
+ for (const store of stores) {
87
+ store.headersAt = headersAt;
88
+ store.refused = refusedStatus(response.status);
89
+ if (wait !== undefined) store.retryAfterMs = wait;
90
+ }
39
91
  if (!response.body) return response;
40
- return new Response(response.body.pipeThrough(firstByteObserver(store)), {
92
+ return new Response(response.body.pipeThrough(firstByteObserver(stores)), {
41
93
  status: response.status,
42
94
  statusText: response.statusText,
43
95
  headers: response.headers,
@@ -0,0 +1,105 @@
1
+ import { setTimeout as delay } from "node:timers/promises";
2
+ import type { NessiIssue, Provider, ProviderIssue, StreamEvent, TimeoutIssue } from "@k2b/nessi/ai";
3
+ import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
4
+
5
+ /** Waits before the first and the second retry when the provider names none. nessi itself never retries. */
6
+ export const AI_PROVIDER_RETRY_DELAYS_MS: readonly number[] = [1_000, 4_000];
7
+
8
+ /**
9
+ * The longest Retry-After that is honored, the bound the OpenAI and Anthropic
10
+ * SDKs apply. A provider that asks for a longer wait is unavailable for this
11
+ * turn, so the call fails now with the provider's message.
12
+ */
13
+ const AI_PROVIDER_MAX_RETRY_AFTER_MS = 60_000;
14
+
15
+ export type AiProviderRetry = {
16
+ /** 1 for the first retry. */
17
+ retry: number;
18
+ delayMs: number;
19
+ issue: ProviderIssue | TimeoutIssue;
20
+ };
21
+
22
+ const transientIssue = (issue: NessiIssue): issue is ProviderIssue | TimeoutIssue =>
23
+ (issue.kind === "provider_error" && issue.retryable && !issue.contextOverflow) ||
24
+ (issue.kind === "timeout" && issue.retryable && issue.scope !== "tool");
25
+
26
+ const isBlockEvent = (event: StreamEvent) => event.type === "block_start" || event.type === "block_delta" || event.type === "block_end";
27
+
28
+ /**
29
+ * Repeats a model call that failed transiently (429, 5xx, a lost connection, a
30
+ * provider timeout) before it produced any output. Each attempt passes through
31
+ * `provider` again, so quota admission and accounting see every request.
32
+ *
33
+ * Once a block event was emitted the attempt is final: streamed text cannot be
34
+ * taken back, so a failure mid-stream still ends the call. Context overflow is
35
+ * left to compaction. Waits follow the provider's Retry-After, otherwise
36
+ * `delaysMs`, end before `deadline`, and stop on the request's abort signal.
37
+ */
38
+ export function retryTransientProviderErrors(
39
+ provider: Provider,
40
+ options: {
41
+ /** Epoch milliseconds by which a wait must end; null when the turn has no run time limit. */
42
+ deadline: number | null;
43
+ delaysMs?: readonly number[];
44
+ /** Announces a wait before it starts. */
45
+ onRetry?: (retry: AiProviderRetry) => Promise<void>;
46
+ },
47
+ ): Provider {
48
+ const delays = options.delaysMs ?? AI_PROVIDER_RETRY_DELAYS_MS;
49
+ const waitFor = (retry: number, marks: ProviderFetchMarks, signal: AbortSignal | undefined): number | null => {
50
+ if (retry >= delays.length || signal?.aborted) return null;
51
+ const delayMs = marks.retryAfterMs ?? delays[retry]!;
52
+ if (delayMs > AI_PROVIDER_MAX_RETRY_AFTER_MS) return null;
53
+ if (options.deadline !== null && Date.now() + delayMs >= options.deadline) return null;
54
+ return delayMs;
55
+ };
56
+ return {
57
+ name: provider.name,
58
+ family: provider.family,
59
+ model: provider.model,
60
+ contextWindow: provider.contextWindow,
61
+ capabilities: provider.capabilities,
62
+ complete: (request) => provider.complete(request),
63
+ stream: async function* (request) {
64
+ for (let retry = 0; ; retry += 1) {
65
+ const marks: ProviderFetchMarks = {};
66
+ const events = provider.stream(request)[Symbol.asyncIterator]();
67
+ const next = () => runWithProviderFetchMarks(marks, () => events.next());
68
+ // Events before the first block (usage, issues) are held so that a retried attempt leaves no trace.
69
+ const held: StreamEvent[] = [];
70
+ let committed = false;
71
+ let pending: AiProviderRetry | null = null;
72
+ try {
73
+ for (let step = await next(); !step.done; step = await next()) {
74
+ const event = step.value;
75
+ if (!committed && event.type === "issue" && transientIssue(event.issue)) {
76
+ const delayMs = waitFor(retry, marks, request.signal);
77
+ if (delayMs !== null) {
78
+ pending = { retry: retry + 1, delayMs, issue: event.issue };
79
+ break;
80
+ }
81
+ }
82
+ if (!committed && !isBlockEvent(event)) {
83
+ held.push(event);
84
+ continue;
85
+ }
86
+ if (!committed) {
87
+ committed = true;
88
+ yield* held.splice(0);
89
+ }
90
+ yield event;
91
+ }
92
+ } finally {
93
+ // Settles the failed attempt's accounting before the wait starts.
94
+ await events.return?.();
95
+ }
96
+ if (!pending) {
97
+ yield* held;
98
+ return;
99
+ }
100
+ await options.onRetry?.(pending);
101
+ await delay(pending.delayMs, undefined, { signal: request.signal });
102
+ }
103
+ },
104
+ };
105
+ }
@@ -11,9 +11,10 @@ const commonOptions = (profile: AiModelProfile, apiKey?: string) => ({
11
11
  contextWindow: profile.contextWindow,
12
12
  temperature: profile.temperature,
13
13
  timeouts: PROVIDER_TIMEOUTS,
14
+ extraBody: profile.extraBody,
14
15
  });
15
16
 
16
- export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Provider => {
17
+ export const createAiProvider = (profile: AiModelProfile, apiKey?: string, headers?: Record<string, string>): Provider => {
17
18
  if (profile.capabilities.includes("transcription")) throw new Error("Transcription profiles cannot be used for chat generation.");
18
19
  switch (profile.provider) {
19
20
  case "openai":
@@ -32,6 +33,7 @@ export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Prov
32
33
  contextWindow: profile.contextWindow,
33
34
  temperature: profile.temperature,
34
35
  timeouts: PROVIDER_TIMEOUTS,
36
+ extraBody: profile.extraBody,
35
37
  });
36
38
  case "vllm":
37
39
  return openAICompatible({
@@ -39,9 +41,11 @@ export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Prov
39
41
  model: profile.model,
40
42
  baseURL: profile.baseURL ?? "http://localhost:8000/v1",
41
43
  apiKey,
44
+ headers,
42
45
  contextWindow: profile.contextWindow,
43
46
  temperature: profile.temperature,
44
47
  timeouts: PROVIDER_TIMEOUTS,
48
+ extraBody: profile.extraBody,
45
49
  compat: {
46
50
  toolCallIdPolicy: "passthrough",
47
51
  supportsUsageInStreaming: true,
@@ -58,9 +62,11 @@ export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Prov
58
62
  model: profile.model,
59
63
  baseURL: profile.baseURL,
60
64
  apiKey,
65
+ headers,
61
66
  contextWindow: profile.contextWindow,
62
67
  temperature: profile.temperature,
63
68
  timeouts: PROVIDER_TIMEOUTS,
69
+ extraBody: profile.extraBody,
64
70
  });
65
71
  }
66
72
  };
@@ -5,7 +5,7 @@ import type { AccessSubject } from "../server/services/access";
5
5
  import { logger } from "../services/logging";
6
6
  import { isAssistantChatTurn } from "./assistant-models";
7
7
  import { AiBackgroundAdmissionError, type AiCallContext, type AiCallDetails, beginAiCall, finishAiCall } from "./inference-calls";
8
- import { runWithProviderFetchMarks } from "./provider-fetch";
8
+ import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
9
9
  import type { AiModelProfile } from "./types";
10
10
 
11
11
  const log = logger("ai:quotas");
@@ -54,8 +54,8 @@ export function inferenceProvider(
54
54
  ctx,
55
55
  inputTokens,
56
56
  request.maxOutputTokens ?? profile.maxOutputTokens,
57
- // Nessi's Anthropic adapter defaults to 1024; other adapters use the model default.
58
- provider.family === "anthropic" ? 1024 : provider.contextWindow,
57
+ // Nessi's Anthropic adapter defaults to 8192; other adapters use the model default.
58
+ provider.family === "anthropic" ? 8192 : provider.contextWindow,
59
59
  );
60
60
  } catch (error) {
61
61
  if (!(error instanceof AiBackgroundAdmissionError) || !error.retryable || Date.now() >= deadline) throw error;
@@ -93,7 +93,7 @@ export function inferenceProvider(
93
93
  let status: "ok" | "failed" = "failed";
94
94
  let error: string | undefined;
95
95
  const requestStartedAt = Date.now();
96
- const marks: { headersAt?: number; firstByteAt?: number } = {};
96
+ const marks: ProviderFetchMarks = {};
97
97
  try {
98
98
  const result = await runWithProviderFetchMarks(marks, () =>
99
99
  provider.complete({ ...request, maxOutputTokens: call.maxOutputTokens }),
@@ -112,7 +112,9 @@ export function inferenceProvider(
112
112
  throw thrown;
113
113
  } finally {
114
114
  call.stop();
115
- if (!usage && status === "failed") usage = { input: call.inputTokens, output: 0, estimated: true };
115
+ // A request the provider refused or never received cost nothing; any other failure may have been processed.
116
+ if (!usage && status === "failed")
117
+ usage = marks.refused ? { input: 0, output: 0 } : { input: call.inputTokens, output: 0, estimated: true };
116
118
  const cancelled = request.signal?.aborted === true;
117
119
  await finish(call.id, usage, cancelled && status === "failed" ? "aborted" : status, {
118
120
  error: cancelled ? null : error,
@@ -133,7 +135,7 @@ export function inferenceProvider(
133
135
  const outputBlocks = new Map<string, number>();
134
136
  let generated = false;
135
137
  const requestStartedAt = Date.now();
136
- const marks: { headersAt?: number; firstByteAt?: number; firstBlockAt?: number } = {};
138
+ const marks: ProviderFetchMarks & { firstBlockAt?: number } = {};
137
139
  // The wrapped adapter reads lazily, so the request only leaves once the first pull runs inside the marked scope.
138
140
  const events = provider.stream({ ...request, maxOutputTokens: call.maxOutputTokens })[Symbol.asyncIterator]();
139
141
  const next = () => runWithProviderFetchMarks(marks, () => events.next());
@@ -158,7 +160,14 @@ export function inferenceProvider(
158
160
  outputBlocks.set(event.blockId, size);
159
161
  }
160
162
  if (event.type === "block_start" || event.type === "block_delta" || event.type === "block_end") generated = true;
161
- if (event.type === "issue" && event.issue.kind === "provider_error" && event.issue.contextOverflow && !generated && !usage)
163
+ // A provider that refused the request, or never received it, did not process it.
164
+ if (
165
+ event.type === "issue" &&
166
+ event.issue.kind === "provider_error" &&
167
+ (event.issue.contextOverflow || marks.refused) &&
168
+ !generated &&
169
+ !usage
170
+ )
162
171
  usage = { input: 0, output: 0 };
163
172
  if (event.type === "usage") {
164
173
  if (event.finishReason === "aborted" || event.finishReason === "interrupted" || event.finishReason === "error") failed = true;
@@ -0,0 +1,117 @@
1
+ import { sql } from "bun";
2
+ import { toPgTextArray } from "../services/postgres";
3
+ import { decryptValue, encryptValue } from "../services/settings/crypto";
4
+ import {
5
+ AI_REQUEST_HEADERS_PROVIDER_ERROR,
6
+ AiRequestHeadersSchema,
7
+ patchAiRequestHeaders,
8
+ providerSupportsRequestHeaders,
9
+ } from "../shared/ai-request-options";
10
+ import type { AiModelProfile } from "./types";
11
+
12
+ type SqlClient = typeof sql;
13
+ export type AiRequestHeaderPatch = { profileId: string; patch: unknown };
14
+
15
+ const decryptHeaders = async (profileId: string, secret: string): Promise<Record<string, string>> => {
16
+ try {
17
+ const parsed = AiRequestHeadersSchema.safeParse(await decryptValue(secret));
18
+ if (parsed.success && Object.values(parsed.data).every((value) => typeof value === "string"))
19
+ return patchAiRequestHeaders({}, parsed.data);
20
+ } catch {}
21
+ console.warn(`[ai] ignoring unreadable request headers for profile ${JSON.stringify(profileId)}`);
22
+ return {};
23
+ };
24
+
25
+ /** Secrets stay server-side, in a separate encrypted row like provider API keys. */
26
+ export const getAiRequestHeaders = async (profileId: string, db: SqlClient = sql): Promise<Record<string, string>> => {
27
+ const [row] = await db<{ secret: string }[]>`SELECT secret FROM ai.model_request_headers WHERE profile_id = ${profileId}`;
28
+ return row ? decryptHeaders(profileId, row.secret) : {};
29
+ };
30
+
31
+ export const setAiRequestHeaders = async (profileId: string, headers: Record<string, string>, db: SqlClient = sql): Promise<void> => {
32
+ const normalized = patchAiRequestHeaders({}, headers);
33
+ if (!Object.keys(normalized).length) {
34
+ await db`DELETE FROM ai.model_request_headers WHERE profile_id = ${profileId}`;
35
+ return;
36
+ }
37
+ const encrypted = await encryptValue(normalized);
38
+ await db`INSERT INTO ai.model_request_headers (profile_id, secret) VALUES (${profileId}, ${encrypted})
39
+ ON CONFLICT (profile_id) DO UPDATE SET secret = EXCLUDED.secret, updated_at = now()`;
40
+ };
41
+
42
+ /** Admin reads contain names only. Never serialize header values into profiles. */
43
+ export const listAiRequestHeaderNames = async (db: SqlClient = sql): Promise<Record<string, string[]>> => {
44
+ const rows = await db<{ profile_id: string; secret: string }[]>`SELECT profile_id, secret FROM ai.model_request_headers`;
45
+ const names: Record<string, string[]> = {};
46
+ for (const row of rows)
47
+ Object.defineProperty(names, row.profile_id, {
48
+ value: Object.keys(await decryptHeaders(row.profile_id, row.secret)).sort(),
49
+ enumerable: true,
50
+ });
51
+ return names;
52
+ };
53
+
54
+ export const pruneAiRequestHeaders = async (keepProfileIds: readonly string[], db: SqlClient = sql): Promise<void> => {
55
+ if (!keepProfileIds.length) await db`DELETE FROM ai.model_request_headers`;
56
+ else await db`DELETE FROM ai.model_request_headers WHERE profile_id <> ALL(${toPgTextArray([...keepProfileIds])}::text[])`;
57
+ };
58
+
59
+ /** Follow credentials: changing provider discards stored secrets; omission preserves them on the same provider. */
60
+ export const planAiProfileRequestHeaders = (input: {
61
+ currentProfiles: readonly AiModelProfile[];
62
+ nextProfiles: readonly AiModelProfile[];
63
+ existingNames: Record<string, string[]>;
64
+ submitted: readonly AiRequestHeaderPatch[];
65
+ }): { keepHeaderProfileIds: string[]; patches: { profileId: string; patch: Record<string, string | null> }[]; error?: string } => {
66
+ const current = new Map(input.currentProfiles.map((profile) => [profile.id, profile]));
67
+ const keepHeaderProfileIds = input.nextProfiles
68
+ .filter(
69
+ (profile) =>
70
+ providerSupportsRequestHeaders(profile.provider) &&
71
+ current.get(profile.id)?.provider === profile.provider &&
72
+ !profile.capabilities.includes("transcription"),
73
+ )
74
+ .map((profile) => profile.id);
75
+ const patches: { profileId: string; patch: Record<string, string | null> }[] = [];
76
+ for (const submitted of input.submitted) {
77
+ const profile = input.nextProfiles.find((profile) => profile.id === submitted.profileId);
78
+ const parsed = AiRequestHeadersSchema.safeParse(submitted.patch);
79
+ if (!parsed.success)
80
+ return { keepHeaderProfileIds: [], patches: [], error: `requestHeaders: ${zodHeaderMessage(parsed.error.issues)}` };
81
+ if (!profile || !providerSupportsRequestHeaders(profile.provider))
82
+ return { keepHeaderProfileIds: [], patches: [], error: AI_REQUEST_HEADERS_PROVIDER_ERROR };
83
+ if (profile.capabilities.includes("transcription"))
84
+ return {
85
+ keepHeaderProfileIds: [],
86
+ patches: [],
87
+ error: "requestHeaders: Request settings are not supported on transcription profiles.",
88
+ };
89
+ const oldNames =
90
+ keepHeaderProfileIds.includes(profile.id) && Object.hasOwn(input.existingNames, profile.id) ? input.existingNames[profile.id]! : [];
91
+ try {
92
+ patchAiRequestHeaders(Object.fromEntries(oldNames.map((name) => [name, ""])), parsed.data);
93
+ } catch {
94
+ return {
95
+ keepHeaderProfileIds: [],
96
+ patches: [],
97
+ error: "requestHeaders: At most 32 extra headers are allowed after applying the patch.",
98
+ };
99
+ }
100
+ patches.push({ profileId: profile.id, patch: parsed.data });
101
+ }
102
+ return { keepHeaderProfileIds, patches };
103
+ };
104
+
105
+ const zodHeaderMessage = (issues: readonly { path: readonly PropertyKey[]; message: string }[]) =>
106
+ issues.map((issue) => `${issue.path.map(String).join(".") || "headers"}: ${issue.message}`).join("; ");
107
+
108
+ /** Apply after pruning so a provider change cannot inherit old secrets. Call inside the settings transaction. */
109
+ export const storeAiRequestHeaderPlan = async (
110
+ plan: ReturnType<typeof planAiProfileRequestHeaders>,
111
+ db: SqlClient = sql,
112
+ ): Promise<void> => {
113
+ if (plan.error) throw new Error(plan.error);
114
+ await pruneAiRequestHeaders(plan.keepHeaderProfileIds, db);
115
+ for (const { profileId, patch } of plan.patches)
116
+ await setAiRequestHeaders(profileId, patchAiRequestHeaders(await getAiRequestHeaders(profileId, db), patch), db);
117
+ };
package/src/ai/routes.ts CHANGED
@@ -168,10 +168,26 @@ const resourcesQuerySchema = (scope: "conversation" | "user") =>
168
168
  limit: z.coerce.number().int().min(1).max(100).optional(),
169
169
  });
170
170
  const ConversationResourcesQuerySchema = resourcesQuerySchema("conversation");
171
+ const ConversationSourcesQuerySchema = ConversationResourcesQuerySchema.extend({
172
+ /** Kinds separated by commas, for example `web,activity,resource`. */
173
+ kind: z
174
+ .string()
175
+ .max(64)
176
+ .transform((value) => value.split(","))
177
+ .pipe(z.array(z.enum(["result", "web", "file", "resource", "activity"])).min(1))
178
+ .optional(),
179
+ observed: z.enum(["true", "false"]).optional(),
180
+ });
171
181
  const UserResourcesQuerySchema = resourcesQuerySchema("user");
172
182
 
173
- const FilesListQuerySchema = z.object({ prefix: z.string().optional() });
183
+ const FilesListQuerySchema = z.object({
184
+ prefix: z.string().optional(),
185
+ /** Pages by path: pass the last path of a page as `after`. Without `limit`, every file comes newest first. */
186
+ limit: z.coerce.number().int().min(1).max(1000).optional(),
187
+ after: z.string().max(4096).optional(),
188
+ });
174
189
  const FilePathQuerySchema = z.object({ path: z.string().min(1) });
190
+ const FileDeleteQuerySchema = FilePathQuerySchema.extend({ recursive: z.enum(["true", "false"]).optional() });
175
191
  const FileWriteSchema = z.object({
176
192
  path: z.string().min(1),
177
193
  content: z.string().max(12_000_000),
@@ -425,6 +441,7 @@ export const aiRoutes = (() => {
425
441
  memory: memory?.text,
426
442
  timeZone,
427
443
  locale: promptLocale,
444
+ skillCreatorAvailable: availableSkills.some((skill) => skill.name === "skill-creator"),
428
445
  });
429
446
  return respond(c, ok({ prompt, renderedAt: new Date().toISOString() }));
430
447
  })
@@ -629,7 +646,7 @@ export const aiRoutes = (() => {
629
646
  ),
630
647
  );
631
648
  })
632
- .get("/conversations/:conversationId/sources", v("query", ConversationResourcesQuerySchema), async (c) => {
649
+ .get("/conversations/:conversationId/sources", v("query", ConversationSourcesQuerySchema), async (c) => {
633
650
  const ctx = await resolveContext(c);
634
651
  if (ctx instanceof Response) return ctx;
635
652
  const conversation = await loadConversation(c, ctx);
@@ -643,6 +660,8 @@ export const aiRoutes = (() => {
643
660
  search: query.q,
644
661
  before: query.cursor,
645
662
  limit: query.limit,
663
+ kinds: query.kind,
664
+ observed: query.observed === "true",
646
665
  }),
647
666
  ),
648
667
  );
@@ -1127,7 +1146,12 @@ export const aiRoutes = (() => {
1127
1146
  if (ctx instanceof Response) return ctx;
1128
1147
  const conversation = await loadConversation(c, ctx);
1129
1148
  if (!conversation) return notFound(c);
1130
- const files = await aiFileStore.list({ conversationId: conversation.id, prefix: c.req.valid("query").prefix ?? "/" });
1149
+ const query = c.req.valid("query");
1150
+ const files = await aiFileStore.list({
1151
+ conversationId: conversation.id,
1152
+ prefix: query.prefix ?? "/",
1153
+ ...(query.limit ? { limit: query.limit, after: query.after } : {}),
1154
+ });
1131
1155
  return respond(c, ok({ files, totalBytes: await aiFileStore.totalBytes(conversation.id) }));
1132
1156
  })
1133
1157
  .post("/conversations/:conversationId/dictations", bodyLimit({ maxSize: AI_AUDIO_MAX_BYTES + 65_536 }), async (c) => {
@@ -1307,14 +1331,18 @@ export const aiRoutes = (() => {
1307
1331
  "Cache-Control": "private, no-store",
1308
1332
  });
1309
1333
  })
1310
- .delete("/conversations/:conversationId/files", v("query", FilePathQuerySchema), async (c) => {
1334
+ .delete("/conversations/:conversationId/files", v("query", FileDeleteQuerySchema), async (c) => {
1311
1335
  const ctx = await resolveContext(c);
1312
1336
  if (ctx instanceof Response) return ctx;
1313
1337
  const conversation = await loadConversation(c, ctx);
1314
1338
  if (!conversation) return notFound(c);
1315
- const path = normalizeAiFilePath(c.req.valid("query").path);
1339
+ const query = c.req.valid("query");
1340
+ const path = normalizeAiFilePath(query.path);
1316
1341
  if (!path) return fileNotFound(c);
1317
- const removed = await aiFileStore.remove({ conversationId: conversation.id, path, recursive: false });
1342
+ // A folder goes with everything below it, never the whole chat at once.
1343
+ const recursive = query.recursive === "true";
1344
+ if (recursive && path === "/") return respond(c, fail(err.badInput("Choose a folder below /.")));
1345
+ const removed = await aiFileStore.remove({ conversationId: conversation.id, path, recursive });
1318
1346
  if (removed === 0) return fileNotFound(c);
1319
1347
  return respond(c, ok({ deleted: true }));
1320
1348
  })
package/src/ai/runtime.ts CHANGED
@@ -13,6 +13,7 @@ import { startAiDictationRuntime } from "./dictation-runtime";
13
13
  import { AiTurnExecutor } from "./executor";
14
14
  import { canonicalizeAiConversationAttachments, snapshotAiConversationFiles } from "./file-context";
15
15
  import { drainQueuedMessages } from "./message-queue";
16
+ import { AI_TURN_LEASE_MS } from "./protocol";
16
17
  import { aiQuotas } from "./quotas";
17
18
  import { parseAiResourceMarker } from "./resource-markers";
18
19
  import { isAiVisionModelConfigured } from "./settings";
@@ -43,7 +44,6 @@ export { isAiSettingsError, validateAiTurnRequest } from "./validate";
43
44
  const log = logger("ai:runtime");
44
45
 
45
46
  const AI_WORKER_ID = `worker-${crypto.randomUUID()}`;
46
- const AI_TURN_LEASE_MS = 45_000;
47
47
  const AI_TURN_HEARTBEAT_MS = 3_000;
48
48
  const AI_TURN_WORKER_CONCURRENCY = 8;
49
49
  const AI_TURN_MAX_ATTEMPTS = 5;
@@ -1,9 +1,17 @@
1
1
  import { z } from "zod";
2
2
  import { coreSettings } from "../services";
3
3
  import { AiModelPricingSchema } from "../shared/ai-costs";
4
+ import {
5
+ AI_REQUEST_HEADERS_PROVIDER_ERROR,
6
+ AiExtraBodySchema,
7
+ AiReasoningEffortSchema,
8
+ AiRequestHeadersSchema,
9
+ providerSupportsRequestHeaders,
10
+ } from "../shared/ai-request-options";
4
11
  import { getAiCredential, listAiCredentialProfileIds } from "./credentials";
5
12
  import { AI_FIRECRAWL_API_KEY_SETTING_KEY } from "./firecrawl-tools";
6
13
  import { createAiProvider } from "./provider";
14
+ import { getAiRequestHeaders } from "./request-headers";
7
15
  import {
8
16
  AI_DATA_BOUNDARIES,
9
17
  AI_MODEL_CAPABILITIES,
@@ -93,12 +101,20 @@ const ModelProfileSchema = z
93
101
  contextWindow: z.number().int().positive().optional(),
94
102
  temperature: z.number().min(0).max(2).optional(),
95
103
  maxOutputTokens: z.number().int().positive().optional(),
104
+ reasoningEffort: AiReasoningEffortSchema,
105
+ extraBody: AiExtraBodySchema.optional(),
106
+ requestHeaders: AiRequestHeadersSchema.optional(),
96
107
  maxLoadedTools: z.number().int().optional(),
97
108
  maxToolRounds: z.number().int().optional(),
98
109
  pricing: AiModelPricingSchema.optional(),
99
110
  })
100
111
  .superRefine((profile, ctx) => {
112
+ if (profile.requestHeaders !== undefined && !providerSupportsRequestHeaders(profile.provider))
113
+ ctx.addIssue({ code: "custom", path: ["requestHeaders"], message: AI_REQUEST_HEADERS_PROVIDER_ERROR });
101
114
  if (profile.capabilities?.includes("transcription")) {
115
+ for (const key of ["reasoningEffort", "extraBody", "requestHeaders"] as const)
116
+ if (profile[key] !== undefined)
117
+ ctx.addIssue({ code: "custom", path: [key], message: "Request settings are not supported on transcription profiles." });
102
118
  if (profile.pricing) ctx.addIssue({ code: "custom", path: ["pricing"], message: "Audio pricing is not supported." });
103
119
  if (profile.capabilities.some((capability) => capability !== "transcription")) {
104
120
  ctx.addIssue({ code: "custom", path: ["capabilities"], message: "Audio transcription cannot be combined with chat capabilities." });
@@ -128,7 +144,7 @@ const profileToPublic = (profile: AiModelProfile): AiPublicModelProfile => ({
128
144
  });
129
145
 
130
146
  const normalizeProfile = (raw: z.infer<typeof ModelProfileSchema>): AiModelProfile => {
131
- const { capabilities, dataBoundary, dataPolicy: legacyDataPolicy, tags: _legacyTags, ...profile } = raw;
147
+ const { requestHeaders: _requestHeaders, capabilities, dataBoundary, dataPolicy: legacyDataPolicy, tags: _legacyTags, ...profile } = raw;
132
148
  return {
133
149
  ...profile,
134
150
  capabilities: normalizeCapabilities(capabilities),
@@ -522,7 +538,8 @@ export const resolveAiModelFromState = async (
522
538
  });
523
539
  }
524
540
 
525
- return { profile, provider: createAiProvider(profile, credential?.trim() || undefined) };
541
+ const headers = providerSupportsRequestHeaders(profile.provider) ? await getAiRequestHeaders(profile.id) : {};
542
+ return { profile, provider: createAiProvider(profile, credential?.trim() || undefined, headers) };
526
543
  };
527
544
 
528
545
  const isUsableProfile = (profile: AiModelProfile, credentialProfileIds: ReadonlySet<string>): boolean =>