@k2b/cloud 0.25.0 → 0.27.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +3 -3
- package/src/_internal/capabilities.ts +12 -0
- package/src/_internal/define-app.ts +8 -1
- package/src/_internal/process-identity.ts +7 -1
- package/src/_internal/registry-validation.ts +3 -0
- package/src/_internal/registry.ts +1 -0
- package/src/_internal/runtime-context.ts +1 -0
- package/src/access/GroupCoverage.tsx +175 -0
- package/src/access/PermissionEditor.tsx +119 -99
- package/src/access/messages.ts +30 -0
- package/src/ai/admin.ts +1 -0
- package/src/ai/approval-routes.ts +5 -5
- package/src/ai/browser-code-contracts.ts +14 -2
- package/src/ai/browser.ts +8 -1
- package/src/ai/capabilities.ts +115 -33
- package/src/ai/chat/blocks.tsx +86 -179
- package/src/ai/chat/builtin-tools.tsx +83 -48
- package/src/ai/chat/file-tools.tsx +4 -1
- package/src/ai/chat/live-turn.browser-harness.tsx +44 -0
- package/src/ai/chat/message-actions.tsx +6 -2
- package/src/ai/chat/message-utils.ts +17 -14
- package/src/ai/chat/messages.ts +330 -2
- package/src/ai/chat/presentation.tsx +278 -104
- package/src/ai/chat/tool-groups.ts +55 -35
- package/src/ai/chat/turn-layout.ts +141 -0
- package/src/ai/chat/turn-view.tsx +644 -0
- package/src/ai/client/controller.ts +60 -31
- package/src/ai/client/file-source.ts +20 -3
- package/src/ai/client/projection.ts +42 -6
- package/src/ai/code-mode-skill.ts +27 -27
- package/src/ai/code-runtime-tools.ts +10 -1
- package/src/ai/code-source-contracts.ts +54 -4
- package/src/ai/code-source-tools.ts +10 -3
- package/src/ai/credentials.ts +17 -3
- package/src/ai/data-analysis-skill.ts +2 -2
- package/src/ai/default-tools.ts +2 -2
- package/src/ai/executor.ts +177 -80
- package/src/ai/file-context.ts +14 -2
- package/src/ai/file-tools.ts +17 -3
- package/src/ai/files-store.ts +134 -11
- package/src/ai/grids-skill.ts +2 -2
- package/src/ai/index.ts +7 -0
- package/src/ai/memories.ts +14 -0
- package/src/ai/migrate.ts +125 -0
- package/src/ai/model-request-settings.ts +98 -0
- package/src/ai/protocol.ts +26 -4
- package/src/ai/provider-fetch.ts +67 -15
- package/src/ai/provider-retry.ts +105 -0
- package/src/ai/provider.ts +7 -1
- package/src/ai/quota-provider.ts +16 -7
- package/src/ai/request-headers.ts +117 -0
- package/src/ai/routes.ts +34 -6
- package/src/ai/runtime.ts +1 -1
- package/src/ai/settings.ts +19 -2
- package/src/ai/skill-seeds.ts +31 -3
- package/src/ai/skills.ts +26 -0
- package/src/ai/solid.ts +1 -1
- package/src/ai/store.ts +202 -59
- package/src/ai/stream.ts +182 -37
- package/src/ai/structured.ts +20 -5
- package/src/ai/system-prompt.ts +25 -0
- package/src/ai/timeline.ts +9 -11
- package/src/ai/tool-call-names.ts +45 -0
- package/src/ai/turn-policy.ts +247 -0
- package/src/ai/turn-timing.ts +31 -3
- package/src/ai/types.ts +36 -5
- package/src/api/admin-ai-quotas.ts +36 -1
- package/src/api/admin-core-settings.ts +16 -23
- package/src/api/admin-outgoing-mail.ts +62 -0
- package/src/api/index.ts +2 -0
- package/src/cli/admin/ai-quotas.ts +70 -1
- package/src/cli/admin/index.ts +6 -0
- package/src/cli/admin/outgoing-mail.ts +118 -0
- package/src/contracts/app.ts +2 -0
- package/src/contracts/index.ts +1 -0
- package/src/contracts/outgoing-mail.ts +77 -0
- package/src/contracts/registry.ts +4 -0
- package/src/services/index.ts +3 -0
- package/src/services/notifications/email.ts +16 -26
- package/src/services/outgoing-mail/index.ts +19 -0
- package/src/services/outgoing-mail/store.ts +286 -0
- package/src/services/outgoing-mail/test-send.ts +40 -0
- package/src/services/outgoing-mail/transport.ts +13 -0
- package/src/services/settings/core-settings.ts +1 -38
- package/src/services/settings/store.ts +5 -1
- package/src/shared/ai-model-request-settings.ts +21 -0
- package/src/shared/ai-platform-prompt.ts +1 -1
- package/src/shared/ai-request-options.ts +185 -0
- package/src/shared/app-presentation.ts +10 -2
- package/src/ssr/admin-navigation.ts +1 -1
- package/src/ssr/platform-messages.ts +2 -0
- package/src/ssr/workspace-navigation.ts +7 -1
- package/src/styles/effects.css +69 -0
package/src/ai/provider-fetch.ts
CHANGED
|
@@ -1,27 +1,63 @@
|
|
|
1
1
|
import { AsyncLocalStorage } from "node:async_hooks";
|
|
2
2
|
|
|
3
3
|
/**
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
4
|
+
* Observes the provider request of the inference call that is currently
|
|
5
|
+
* running, without touching the wire: when its headers and first body byte
|
|
6
|
+
* arrived, whether the provider certainly did not process it, and how long a
|
|
7
|
+
* rejecting provider asked the caller to wait. nessi adapters call the global
|
|
8
|
+
* `fetch`; the wrapper is installed once, on first use, and only observes
|
|
9
|
+
* requests made inside `runWithProviderFetchMarks`. Scopes nest, so an outer
|
|
10
|
+
* retry policy and the inner per-call accounting each see the same request.
|
|
8
11
|
*
|
|
9
12
|
* Nothing here logs: frames may carry reasoning text and headers carry keys.
|
|
10
13
|
*/
|
|
11
|
-
export type ProviderFetchMarks = {
|
|
14
|
+
export type ProviderFetchMarks = {
|
|
15
|
+
headersAt?: number;
|
|
16
|
+
firstByteAt?: number;
|
|
17
|
+
/** The request never reached the provider, or the provider answered with a status that says it did not process it. */
|
|
18
|
+
refused?: boolean;
|
|
19
|
+
retryAfterMs?: number;
|
|
20
|
+
};
|
|
12
21
|
|
|
13
|
-
const
|
|
22
|
+
const scopes = new AsyncLocalStorage<readonly ProviderFetchMarks[]>();
|
|
14
23
|
let installed = false;
|
|
15
24
|
|
|
16
25
|
export const runWithProviderFetchMarks = <T>(store: ProviderFetchMarks, run: () => Promise<T>): Promise<T> => {
|
|
17
26
|
install();
|
|
18
|
-
return
|
|
27
|
+
return scopes.run([...(scopes.getStore() ?? []), store], run);
|
|
28
|
+
};
|
|
29
|
+
|
|
30
|
+
/** `retry-after-ms` (OpenAI) wins over the standard `retry-after` seconds or HTTP date. */
|
|
31
|
+
export const retryAfterMs = (headers: Headers, now = Date.now()): number | undefined => {
|
|
32
|
+
const milliseconds = Number(headers.get("retry-after-ms")?.trim() || Number.NaN);
|
|
33
|
+
if (Number.isFinite(milliseconds) && milliseconds >= 0) return milliseconds;
|
|
34
|
+
const value = headers.get("retry-after")?.trim();
|
|
35
|
+
if (!value) return undefined;
|
|
36
|
+
const seconds = Number(value);
|
|
37
|
+
if (Number.isFinite(seconds)) return seconds >= 0 ? seconds * 1_000 : undefined;
|
|
38
|
+
const date = Date.parse(value);
|
|
39
|
+
return Number.isFinite(date) ? Math.max(0, date - now) : undefined;
|
|
19
40
|
};
|
|
20
41
|
|
|
21
|
-
|
|
42
|
+
/**
|
|
43
|
+
* A client error, 503, or Anthropic's overloaded 529 says the provider did not
|
|
44
|
+
* process the request. Other server errors, such as a gateway's 502 or 504, can
|
|
45
|
+
* follow processing upstream.
|
|
46
|
+
*/
|
|
47
|
+
const refusedStatus = (status: number) => (status >= 400 && status < 500) || status === 503 || status === 529;
|
|
48
|
+
|
|
49
|
+
/** Bun's codes for a request that never left: no connection, or no address for the host. */
|
|
50
|
+
const unsentCodes = new Set(["ConnectionRefused", "ENOTFOUND"]);
|
|
51
|
+
const unsent = (error: unknown) =>
|
|
52
|
+
typeof error === "object" && error !== null && "code" in error && typeof error.code === "string" && unsentCodes.has(error.code);
|
|
53
|
+
|
|
54
|
+
const firstByteObserver = (stores: readonly ProviderFetchMarks[]) =>
|
|
22
55
|
new TransformStream<Uint8Array, Uint8Array>({
|
|
23
56
|
transform(chunk, controller) {
|
|
24
|
-
if (chunk.byteLength > 0
|
|
57
|
+
if (chunk.byteLength > 0)
|
|
58
|
+
for (const store of stores) {
|
|
59
|
+
store.firstByteAt ??= Date.now();
|
|
60
|
+
}
|
|
25
61
|
controller.enqueue(chunk);
|
|
26
62
|
},
|
|
27
63
|
});
|
|
@@ -31,13 +67,29 @@ const install = (): void => {
|
|
|
31
67
|
installed = true;
|
|
32
68
|
const realFetch = globalThis.fetch;
|
|
33
69
|
const instrumented = async (input: RequestInfo | URL, init?: RequestInit): Promise<Response> => {
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
if (
|
|
37
|
-
|
|
38
|
-
|
|
70
|
+
// Only the first request of a scope is its provider request; a retry opens a new scope.
|
|
71
|
+
const stores = (scopes.getStore() ?? []).filter((store) => store.headersAt === undefined);
|
|
72
|
+
if (stores.length === 0) return realFetch(input, init);
|
|
73
|
+
let response: Response;
|
|
74
|
+
try {
|
|
75
|
+
response = await realFetch(input, init);
|
|
76
|
+
} catch (error) {
|
|
77
|
+
// A reset, an abort, or a timeout can follow delivery, so only an unsent request counts as refused.
|
|
78
|
+
if (unsent(error))
|
|
79
|
+
for (const store of stores) {
|
|
80
|
+
store.refused = true;
|
|
81
|
+
}
|
|
82
|
+
throw error;
|
|
83
|
+
}
|
|
84
|
+
const headersAt = Date.now();
|
|
85
|
+
const wait = response.ok ? undefined : retryAfterMs(response.headers, headersAt);
|
|
86
|
+
for (const store of stores) {
|
|
87
|
+
store.headersAt = headersAt;
|
|
88
|
+
store.refused = refusedStatus(response.status);
|
|
89
|
+
if (wait !== undefined) store.retryAfterMs = wait;
|
|
90
|
+
}
|
|
39
91
|
if (!response.body) return response;
|
|
40
|
-
return new Response(response.body.pipeThrough(firstByteObserver(
|
|
92
|
+
return new Response(response.body.pipeThrough(firstByteObserver(stores)), {
|
|
41
93
|
status: response.status,
|
|
42
94
|
statusText: response.statusText,
|
|
43
95
|
headers: response.headers,
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import { setTimeout as delay } from "node:timers/promises";
|
|
2
|
+
import type { NessiIssue, Provider, ProviderIssue, StreamEvent, TimeoutIssue } from "@k2b/nessi/ai";
|
|
3
|
+
import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
|
|
4
|
+
|
|
5
|
+
/** Waits before the first and the second retry when the provider names none. nessi itself never retries. */
|
|
6
|
+
export const AI_PROVIDER_RETRY_DELAYS_MS: readonly number[] = [1_000, 4_000];
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* The longest Retry-After that is honored, the bound the OpenAI and Anthropic
|
|
10
|
+
* SDKs apply. A provider that asks for a longer wait is unavailable for this
|
|
11
|
+
* turn, so the call fails now with the provider's message.
|
|
12
|
+
*/
|
|
13
|
+
const AI_PROVIDER_MAX_RETRY_AFTER_MS = 60_000;
|
|
14
|
+
|
|
15
|
+
export type AiProviderRetry = {
|
|
16
|
+
/** 1 for the first retry. */
|
|
17
|
+
retry: number;
|
|
18
|
+
delayMs: number;
|
|
19
|
+
issue: ProviderIssue | TimeoutIssue;
|
|
20
|
+
};
|
|
21
|
+
|
|
22
|
+
const transientIssue = (issue: NessiIssue): issue is ProviderIssue | TimeoutIssue =>
|
|
23
|
+
(issue.kind === "provider_error" && issue.retryable && !issue.contextOverflow) ||
|
|
24
|
+
(issue.kind === "timeout" && issue.retryable && issue.scope !== "tool");
|
|
25
|
+
|
|
26
|
+
const isBlockEvent = (event: StreamEvent) => event.type === "block_start" || event.type === "block_delta" || event.type === "block_end";
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Repeats a model call that failed transiently (429, 5xx, a lost connection, a
|
|
30
|
+
* provider timeout) before it produced any output. Each attempt passes through
|
|
31
|
+
* `provider` again, so quota admission and accounting see every request.
|
|
32
|
+
*
|
|
33
|
+
* Once a block event was emitted the attempt is final: streamed text cannot be
|
|
34
|
+
* taken back, so a failure mid-stream still ends the call. Context overflow is
|
|
35
|
+
* left to compaction. Waits follow the provider's Retry-After, otherwise
|
|
36
|
+
* `delaysMs`, end before `deadline`, and stop on the request's abort signal.
|
|
37
|
+
*/
|
|
38
|
+
export function retryTransientProviderErrors(
|
|
39
|
+
provider: Provider,
|
|
40
|
+
options: {
|
|
41
|
+
/** Epoch milliseconds by which a wait must end; null when the turn has no run time limit. */
|
|
42
|
+
deadline: number | null;
|
|
43
|
+
delaysMs?: readonly number[];
|
|
44
|
+
/** Announces a wait before it starts. */
|
|
45
|
+
onRetry?: (retry: AiProviderRetry) => Promise<void>;
|
|
46
|
+
},
|
|
47
|
+
): Provider {
|
|
48
|
+
const delays = options.delaysMs ?? AI_PROVIDER_RETRY_DELAYS_MS;
|
|
49
|
+
const waitFor = (retry: number, marks: ProviderFetchMarks, signal: AbortSignal | undefined): number | null => {
|
|
50
|
+
if (retry >= delays.length || signal?.aborted) return null;
|
|
51
|
+
const delayMs = marks.retryAfterMs ?? delays[retry]!;
|
|
52
|
+
if (delayMs > AI_PROVIDER_MAX_RETRY_AFTER_MS) return null;
|
|
53
|
+
if (options.deadline !== null && Date.now() + delayMs >= options.deadline) return null;
|
|
54
|
+
return delayMs;
|
|
55
|
+
};
|
|
56
|
+
return {
|
|
57
|
+
name: provider.name,
|
|
58
|
+
family: provider.family,
|
|
59
|
+
model: provider.model,
|
|
60
|
+
contextWindow: provider.contextWindow,
|
|
61
|
+
capabilities: provider.capabilities,
|
|
62
|
+
complete: (request) => provider.complete(request),
|
|
63
|
+
stream: async function* (request) {
|
|
64
|
+
for (let retry = 0; ; retry += 1) {
|
|
65
|
+
const marks: ProviderFetchMarks = {};
|
|
66
|
+
const events = provider.stream(request)[Symbol.asyncIterator]();
|
|
67
|
+
const next = () => runWithProviderFetchMarks(marks, () => events.next());
|
|
68
|
+
// Events before the first block (usage, issues) are held so that a retried attempt leaves no trace.
|
|
69
|
+
const held: StreamEvent[] = [];
|
|
70
|
+
let committed = false;
|
|
71
|
+
let pending: AiProviderRetry | null = null;
|
|
72
|
+
try {
|
|
73
|
+
for (let step = await next(); !step.done; step = await next()) {
|
|
74
|
+
const event = step.value;
|
|
75
|
+
if (!committed && event.type === "issue" && transientIssue(event.issue)) {
|
|
76
|
+
const delayMs = waitFor(retry, marks, request.signal);
|
|
77
|
+
if (delayMs !== null) {
|
|
78
|
+
pending = { retry: retry + 1, delayMs, issue: event.issue };
|
|
79
|
+
break;
|
|
80
|
+
}
|
|
81
|
+
}
|
|
82
|
+
if (!committed && !isBlockEvent(event)) {
|
|
83
|
+
held.push(event);
|
|
84
|
+
continue;
|
|
85
|
+
}
|
|
86
|
+
if (!committed) {
|
|
87
|
+
committed = true;
|
|
88
|
+
yield* held.splice(0);
|
|
89
|
+
}
|
|
90
|
+
yield event;
|
|
91
|
+
}
|
|
92
|
+
} finally {
|
|
93
|
+
// Settles the failed attempt's accounting before the wait starts.
|
|
94
|
+
await events.return?.();
|
|
95
|
+
}
|
|
96
|
+
if (!pending) {
|
|
97
|
+
yield* held;
|
|
98
|
+
return;
|
|
99
|
+
}
|
|
100
|
+
await options.onRetry?.(pending);
|
|
101
|
+
await delay(pending.delayMs, undefined, { signal: request.signal });
|
|
102
|
+
}
|
|
103
|
+
},
|
|
104
|
+
};
|
|
105
|
+
}
|
package/src/ai/provider.ts
CHANGED
|
@@ -11,9 +11,10 @@ const commonOptions = (profile: AiModelProfile, apiKey?: string) => ({
|
|
|
11
11
|
contextWindow: profile.contextWindow,
|
|
12
12
|
temperature: profile.temperature,
|
|
13
13
|
timeouts: PROVIDER_TIMEOUTS,
|
|
14
|
+
extraBody: profile.extraBody,
|
|
14
15
|
});
|
|
15
16
|
|
|
16
|
-
export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Provider => {
|
|
17
|
+
export const createAiProvider = (profile: AiModelProfile, apiKey?: string, headers?: Record<string, string>): Provider => {
|
|
17
18
|
if (profile.capabilities.includes("transcription")) throw new Error("Transcription profiles cannot be used for chat generation.");
|
|
18
19
|
switch (profile.provider) {
|
|
19
20
|
case "openai":
|
|
@@ -32,6 +33,7 @@ export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Prov
|
|
|
32
33
|
contextWindow: profile.contextWindow,
|
|
33
34
|
temperature: profile.temperature,
|
|
34
35
|
timeouts: PROVIDER_TIMEOUTS,
|
|
36
|
+
extraBody: profile.extraBody,
|
|
35
37
|
});
|
|
36
38
|
case "vllm":
|
|
37
39
|
return openAICompatible({
|
|
@@ -39,9 +41,11 @@ export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Prov
|
|
|
39
41
|
model: profile.model,
|
|
40
42
|
baseURL: profile.baseURL ?? "http://localhost:8000/v1",
|
|
41
43
|
apiKey,
|
|
44
|
+
headers,
|
|
42
45
|
contextWindow: profile.contextWindow,
|
|
43
46
|
temperature: profile.temperature,
|
|
44
47
|
timeouts: PROVIDER_TIMEOUTS,
|
|
48
|
+
extraBody: profile.extraBody,
|
|
45
49
|
compat: {
|
|
46
50
|
toolCallIdPolicy: "passthrough",
|
|
47
51
|
supportsUsageInStreaming: true,
|
|
@@ -58,9 +62,11 @@ export const createAiProvider = (profile: AiModelProfile, apiKey?: string): Prov
|
|
|
58
62
|
model: profile.model,
|
|
59
63
|
baseURL: profile.baseURL,
|
|
60
64
|
apiKey,
|
|
65
|
+
headers,
|
|
61
66
|
contextWindow: profile.contextWindow,
|
|
62
67
|
temperature: profile.temperature,
|
|
63
68
|
timeouts: PROVIDER_TIMEOUTS,
|
|
69
|
+
extraBody: profile.extraBody,
|
|
64
70
|
});
|
|
65
71
|
}
|
|
66
72
|
};
|
package/src/ai/quota-provider.ts
CHANGED
|
@@ -5,7 +5,7 @@ import type { AccessSubject } from "../server/services/access";
|
|
|
5
5
|
import { logger } from "../services/logging";
|
|
6
6
|
import { isAssistantChatTurn } from "./assistant-models";
|
|
7
7
|
import { AiBackgroundAdmissionError, type AiCallContext, type AiCallDetails, beginAiCall, finishAiCall } from "./inference-calls";
|
|
8
|
-
import { runWithProviderFetchMarks } from "./provider-fetch";
|
|
8
|
+
import { type ProviderFetchMarks, runWithProviderFetchMarks } from "./provider-fetch";
|
|
9
9
|
import type { AiModelProfile } from "./types";
|
|
10
10
|
|
|
11
11
|
const log = logger("ai:quotas");
|
|
@@ -54,8 +54,8 @@ export function inferenceProvider(
|
|
|
54
54
|
ctx,
|
|
55
55
|
inputTokens,
|
|
56
56
|
request.maxOutputTokens ?? profile.maxOutputTokens,
|
|
57
|
-
// Nessi's Anthropic adapter defaults to
|
|
58
|
-
provider.family === "anthropic" ?
|
|
57
|
+
// Nessi's Anthropic adapter defaults to 8192; other adapters use the model default.
|
|
58
|
+
provider.family === "anthropic" ? 8192 : provider.contextWindow,
|
|
59
59
|
);
|
|
60
60
|
} catch (error) {
|
|
61
61
|
if (!(error instanceof AiBackgroundAdmissionError) || !error.retryable || Date.now() >= deadline) throw error;
|
|
@@ -93,7 +93,7 @@ export function inferenceProvider(
|
|
|
93
93
|
let status: "ok" | "failed" = "failed";
|
|
94
94
|
let error: string | undefined;
|
|
95
95
|
const requestStartedAt = Date.now();
|
|
96
|
-
const marks:
|
|
96
|
+
const marks: ProviderFetchMarks = {};
|
|
97
97
|
try {
|
|
98
98
|
const result = await runWithProviderFetchMarks(marks, () =>
|
|
99
99
|
provider.complete({ ...request, maxOutputTokens: call.maxOutputTokens }),
|
|
@@ -112,7 +112,9 @@ export function inferenceProvider(
|
|
|
112
112
|
throw thrown;
|
|
113
113
|
} finally {
|
|
114
114
|
call.stop();
|
|
115
|
-
|
|
115
|
+
// A request the provider refused or never received cost nothing; any other failure may have been processed.
|
|
116
|
+
if (!usage && status === "failed")
|
|
117
|
+
usage = marks.refused ? { input: 0, output: 0 } : { input: call.inputTokens, output: 0, estimated: true };
|
|
116
118
|
const cancelled = request.signal?.aborted === true;
|
|
117
119
|
await finish(call.id, usage, cancelled && status === "failed" ? "aborted" : status, {
|
|
118
120
|
error: cancelled ? null : error,
|
|
@@ -133,7 +135,7 @@ export function inferenceProvider(
|
|
|
133
135
|
const outputBlocks = new Map<string, number>();
|
|
134
136
|
let generated = false;
|
|
135
137
|
const requestStartedAt = Date.now();
|
|
136
|
-
const marks:
|
|
138
|
+
const marks: ProviderFetchMarks & { firstBlockAt?: number } = {};
|
|
137
139
|
// The wrapped adapter reads lazily, so the request only leaves once the first pull runs inside the marked scope.
|
|
138
140
|
const events = provider.stream({ ...request, maxOutputTokens: call.maxOutputTokens })[Symbol.asyncIterator]();
|
|
139
141
|
const next = () => runWithProviderFetchMarks(marks, () => events.next());
|
|
@@ -158,7 +160,14 @@ export function inferenceProvider(
|
|
|
158
160
|
outputBlocks.set(event.blockId, size);
|
|
159
161
|
}
|
|
160
162
|
if (event.type === "block_start" || event.type === "block_delta" || event.type === "block_end") generated = true;
|
|
161
|
-
|
|
163
|
+
// A provider that refused the request, or never received it, did not process it.
|
|
164
|
+
if (
|
|
165
|
+
event.type === "issue" &&
|
|
166
|
+
event.issue.kind === "provider_error" &&
|
|
167
|
+
(event.issue.contextOverflow || marks.refused) &&
|
|
168
|
+
!generated &&
|
|
169
|
+
!usage
|
|
170
|
+
)
|
|
162
171
|
usage = { input: 0, output: 0 };
|
|
163
172
|
if (event.type === "usage") {
|
|
164
173
|
if (event.finishReason === "aborted" || event.finishReason === "interrupted" || event.finishReason === "error") failed = true;
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
import { sql } from "bun";
|
|
2
|
+
import { toPgTextArray } from "../services/postgres";
|
|
3
|
+
import { decryptValue, encryptValue } from "../services/settings/crypto";
|
|
4
|
+
import {
|
|
5
|
+
AI_REQUEST_HEADERS_PROVIDER_ERROR,
|
|
6
|
+
AiRequestHeadersSchema,
|
|
7
|
+
patchAiRequestHeaders,
|
|
8
|
+
providerSupportsRequestHeaders,
|
|
9
|
+
} from "../shared/ai-request-options";
|
|
10
|
+
import type { AiModelProfile } from "./types";
|
|
11
|
+
|
|
12
|
+
type SqlClient = typeof sql;
|
|
13
|
+
export type AiRequestHeaderPatch = { profileId: string; patch: unknown };
|
|
14
|
+
|
|
15
|
+
const decryptHeaders = async (profileId: string, secret: string): Promise<Record<string, string>> => {
|
|
16
|
+
try {
|
|
17
|
+
const parsed = AiRequestHeadersSchema.safeParse(await decryptValue(secret));
|
|
18
|
+
if (parsed.success && Object.values(parsed.data).every((value) => typeof value === "string"))
|
|
19
|
+
return patchAiRequestHeaders({}, parsed.data);
|
|
20
|
+
} catch {}
|
|
21
|
+
console.warn(`[ai] ignoring unreadable request headers for profile ${JSON.stringify(profileId)}`);
|
|
22
|
+
return {};
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
/** Secrets stay server-side, in a separate encrypted row like provider API keys. */
|
|
26
|
+
export const getAiRequestHeaders = async (profileId: string, db: SqlClient = sql): Promise<Record<string, string>> => {
|
|
27
|
+
const [row] = await db<{ secret: string }[]>`SELECT secret FROM ai.model_request_headers WHERE profile_id = ${profileId}`;
|
|
28
|
+
return row ? decryptHeaders(profileId, row.secret) : {};
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
export const setAiRequestHeaders = async (profileId: string, headers: Record<string, string>, db: SqlClient = sql): Promise<void> => {
|
|
32
|
+
const normalized = patchAiRequestHeaders({}, headers);
|
|
33
|
+
if (!Object.keys(normalized).length) {
|
|
34
|
+
await db`DELETE FROM ai.model_request_headers WHERE profile_id = ${profileId}`;
|
|
35
|
+
return;
|
|
36
|
+
}
|
|
37
|
+
const encrypted = await encryptValue(normalized);
|
|
38
|
+
await db`INSERT INTO ai.model_request_headers (profile_id, secret) VALUES (${profileId}, ${encrypted})
|
|
39
|
+
ON CONFLICT (profile_id) DO UPDATE SET secret = EXCLUDED.secret, updated_at = now()`;
|
|
40
|
+
};
|
|
41
|
+
|
|
42
|
+
/** Admin reads contain names only. Never serialize header values into profiles. */
|
|
43
|
+
export const listAiRequestHeaderNames = async (db: SqlClient = sql): Promise<Record<string, string[]>> => {
|
|
44
|
+
const rows = await db<{ profile_id: string; secret: string }[]>`SELECT profile_id, secret FROM ai.model_request_headers`;
|
|
45
|
+
const names: Record<string, string[]> = {};
|
|
46
|
+
for (const row of rows)
|
|
47
|
+
Object.defineProperty(names, row.profile_id, {
|
|
48
|
+
value: Object.keys(await decryptHeaders(row.profile_id, row.secret)).sort(),
|
|
49
|
+
enumerable: true,
|
|
50
|
+
});
|
|
51
|
+
return names;
|
|
52
|
+
};
|
|
53
|
+
|
|
54
|
+
export const pruneAiRequestHeaders = async (keepProfileIds: readonly string[], db: SqlClient = sql): Promise<void> => {
|
|
55
|
+
if (!keepProfileIds.length) await db`DELETE FROM ai.model_request_headers`;
|
|
56
|
+
else await db`DELETE FROM ai.model_request_headers WHERE profile_id <> ALL(${toPgTextArray([...keepProfileIds])}::text[])`;
|
|
57
|
+
};
|
|
58
|
+
|
|
59
|
+
/** Follow credentials: changing provider discards stored secrets; omission preserves them on the same provider. */
|
|
60
|
+
export const planAiProfileRequestHeaders = (input: {
|
|
61
|
+
currentProfiles: readonly AiModelProfile[];
|
|
62
|
+
nextProfiles: readonly AiModelProfile[];
|
|
63
|
+
existingNames: Record<string, string[]>;
|
|
64
|
+
submitted: readonly AiRequestHeaderPatch[];
|
|
65
|
+
}): { keepHeaderProfileIds: string[]; patches: { profileId: string; patch: Record<string, string | null> }[]; error?: string } => {
|
|
66
|
+
const current = new Map(input.currentProfiles.map((profile) => [profile.id, profile]));
|
|
67
|
+
const keepHeaderProfileIds = input.nextProfiles
|
|
68
|
+
.filter(
|
|
69
|
+
(profile) =>
|
|
70
|
+
providerSupportsRequestHeaders(profile.provider) &&
|
|
71
|
+
current.get(profile.id)?.provider === profile.provider &&
|
|
72
|
+
!profile.capabilities.includes("transcription"),
|
|
73
|
+
)
|
|
74
|
+
.map((profile) => profile.id);
|
|
75
|
+
const patches: { profileId: string; patch: Record<string, string | null> }[] = [];
|
|
76
|
+
for (const submitted of input.submitted) {
|
|
77
|
+
const profile = input.nextProfiles.find((profile) => profile.id === submitted.profileId);
|
|
78
|
+
const parsed = AiRequestHeadersSchema.safeParse(submitted.patch);
|
|
79
|
+
if (!parsed.success)
|
|
80
|
+
return { keepHeaderProfileIds: [], patches: [], error: `requestHeaders: ${zodHeaderMessage(parsed.error.issues)}` };
|
|
81
|
+
if (!profile || !providerSupportsRequestHeaders(profile.provider))
|
|
82
|
+
return { keepHeaderProfileIds: [], patches: [], error: AI_REQUEST_HEADERS_PROVIDER_ERROR };
|
|
83
|
+
if (profile.capabilities.includes("transcription"))
|
|
84
|
+
return {
|
|
85
|
+
keepHeaderProfileIds: [],
|
|
86
|
+
patches: [],
|
|
87
|
+
error: "requestHeaders: Request settings are not supported on transcription profiles.",
|
|
88
|
+
};
|
|
89
|
+
const oldNames =
|
|
90
|
+
keepHeaderProfileIds.includes(profile.id) && Object.hasOwn(input.existingNames, profile.id) ? input.existingNames[profile.id]! : [];
|
|
91
|
+
try {
|
|
92
|
+
patchAiRequestHeaders(Object.fromEntries(oldNames.map((name) => [name, ""])), parsed.data);
|
|
93
|
+
} catch {
|
|
94
|
+
return {
|
|
95
|
+
keepHeaderProfileIds: [],
|
|
96
|
+
patches: [],
|
|
97
|
+
error: "requestHeaders: At most 32 extra headers are allowed after applying the patch.",
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
patches.push({ profileId: profile.id, patch: parsed.data });
|
|
101
|
+
}
|
|
102
|
+
return { keepHeaderProfileIds, patches };
|
|
103
|
+
};
|
|
104
|
+
|
|
105
|
+
const zodHeaderMessage = (issues: readonly { path: readonly PropertyKey[]; message: string }[]) =>
|
|
106
|
+
issues.map((issue) => `${issue.path.map(String).join(".") || "headers"}: ${issue.message}`).join("; ");
|
|
107
|
+
|
|
108
|
+
/** Apply after pruning so a provider change cannot inherit old secrets. Call inside the settings transaction. */
|
|
109
|
+
export const storeAiRequestHeaderPlan = async (
|
|
110
|
+
plan: ReturnType<typeof planAiProfileRequestHeaders>,
|
|
111
|
+
db: SqlClient = sql,
|
|
112
|
+
): Promise<void> => {
|
|
113
|
+
if (plan.error) throw new Error(plan.error);
|
|
114
|
+
await pruneAiRequestHeaders(plan.keepHeaderProfileIds, db);
|
|
115
|
+
for (const { profileId, patch } of plan.patches)
|
|
116
|
+
await setAiRequestHeaders(profileId, patchAiRequestHeaders(await getAiRequestHeaders(profileId, db), patch), db);
|
|
117
|
+
};
|
package/src/ai/routes.ts
CHANGED
|
@@ -168,10 +168,26 @@ const resourcesQuerySchema = (scope: "conversation" | "user") =>
|
|
|
168
168
|
limit: z.coerce.number().int().min(1).max(100).optional(),
|
|
169
169
|
});
|
|
170
170
|
const ConversationResourcesQuerySchema = resourcesQuerySchema("conversation");
|
|
171
|
+
const ConversationSourcesQuerySchema = ConversationResourcesQuerySchema.extend({
|
|
172
|
+
/** Kinds separated by commas, for example `web,activity,resource`. */
|
|
173
|
+
kind: z
|
|
174
|
+
.string()
|
|
175
|
+
.max(64)
|
|
176
|
+
.transform((value) => value.split(","))
|
|
177
|
+
.pipe(z.array(z.enum(["result", "web", "file", "resource", "activity"])).min(1))
|
|
178
|
+
.optional(),
|
|
179
|
+
observed: z.enum(["true", "false"]).optional(),
|
|
180
|
+
});
|
|
171
181
|
const UserResourcesQuerySchema = resourcesQuerySchema("user");
|
|
172
182
|
|
|
173
|
-
const FilesListQuerySchema = z.object({
|
|
183
|
+
const FilesListQuerySchema = z.object({
|
|
184
|
+
prefix: z.string().optional(),
|
|
185
|
+
/** Pages by path: pass the last path of a page as `after`. Without `limit`, every file comes newest first. */
|
|
186
|
+
limit: z.coerce.number().int().min(1).max(1000).optional(),
|
|
187
|
+
after: z.string().max(4096).optional(),
|
|
188
|
+
});
|
|
174
189
|
const FilePathQuerySchema = z.object({ path: z.string().min(1) });
|
|
190
|
+
const FileDeleteQuerySchema = FilePathQuerySchema.extend({ recursive: z.enum(["true", "false"]).optional() });
|
|
175
191
|
const FileWriteSchema = z.object({
|
|
176
192
|
path: z.string().min(1),
|
|
177
193
|
content: z.string().max(12_000_000),
|
|
@@ -425,6 +441,7 @@ export const aiRoutes = (() => {
|
|
|
425
441
|
memory: memory?.text,
|
|
426
442
|
timeZone,
|
|
427
443
|
locale: promptLocale,
|
|
444
|
+
skillCreatorAvailable: availableSkills.some((skill) => skill.name === "skill-creator"),
|
|
428
445
|
});
|
|
429
446
|
return respond(c, ok({ prompt, renderedAt: new Date().toISOString() }));
|
|
430
447
|
})
|
|
@@ -629,7 +646,7 @@ export const aiRoutes = (() => {
|
|
|
629
646
|
),
|
|
630
647
|
);
|
|
631
648
|
})
|
|
632
|
-
.get("/conversations/:conversationId/sources", v("query",
|
|
649
|
+
.get("/conversations/:conversationId/sources", v("query", ConversationSourcesQuerySchema), async (c) => {
|
|
633
650
|
const ctx = await resolveContext(c);
|
|
634
651
|
if (ctx instanceof Response) return ctx;
|
|
635
652
|
const conversation = await loadConversation(c, ctx);
|
|
@@ -643,6 +660,8 @@ export const aiRoutes = (() => {
|
|
|
643
660
|
search: query.q,
|
|
644
661
|
before: query.cursor,
|
|
645
662
|
limit: query.limit,
|
|
663
|
+
kinds: query.kind,
|
|
664
|
+
observed: query.observed === "true",
|
|
646
665
|
}),
|
|
647
666
|
),
|
|
648
667
|
);
|
|
@@ -1127,7 +1146,12 @@ export const aiRoutes = (() => {
|
|
|
1127
1146
|
if (ctx instanceof Response) return ctx;
|
|
1128
1147
|
const conversation = await loadConversation(c, ctx);
|
|
1129
1148
|
if (!conversation) return notFound(c);
|
|
1130
|
-
const
|
|
1149
|
+
const query = c.req.valid("query");
|
|
1150
|
+
const files = await aiFileStore.list({
|
|
1151
|
+
conversationId: conversation.id,
|
|
1152
|
+
prefix: query.prefix ?? "/",
|
|
1153
|
+
...(query.limit ? { limit: query.limit, after: query.after } : {}),
|
|
1154
|
+
});
|
|
1131
1155
|
return respond(c, ok({ files, totalBytes: await aiFileStore.totalBytes(conversation.id) }));
|
|
1132
1156
|
})
|
|
1133
1157
|
.post("/conversations/:conversationId/dictations", bodyLimit({ maxSize: AI_AUDIO_MAX_BYTES + 65_536 }), async (c) => {
|
|
@@ -1307,14 +1331,18 @@ export const aiRoutes = (() => {
|
|
|
1307
1331
|
"Cache-Control": "private, no-store",
|
|
1308
1332
|
});
|
|
1309
1333
|
})
|
|
1310
|
-
.delete("/conversations/:conversationId/files", v("query",
|
|
1334
|
+
.delete("/conversations/:conversationId/files", v("query", FileDeleteQuerySchema), async (c) => {
|
|
1311
1335
|
const ctx = await resolveContext(c);
|
|
1312
1336
|
if (ctx instanceof Response) return ctx;
|
|
1313
1337
|
const conversation = await loadConversation(c, ctx);
|
|
1314
1338
|
if (!conversation) return notFound(c);
|
|
1315
|
-
const
|
|
1339
|
+
const query = c.req.valid("query");
|
|
1340
|
+
const path = normalizeAiFilePath(query.path);
|
|
1316
1341
|
if (!path) return fileNotFound(c);
|
|
1317
|
-
|
|
1342
|
+
// A folder goes with everything below it, never the whole chat at once.
|
|
1343
|
+
const recursive = query.recursive === "true";
|
|
1344
|
+
if (recursive && path === "/") return respond(c, fail(err.badInput("Choose a folder below /.")));
|
|
1345
|
+
const removed = await aiFileStore.remove({ conversationId: conversation.id, path, recursive });
|
|
1318
1346
|
if (removed === 0) return fileNotFound(c);
|
|
1319
1347
|
return respond(c, ok({ deleted: true }));
|
|
1320
1348
|
})
|
package/src/ai/runtime.ts
CHANGED
|
@@ -13,6 +13,7 @@ import { startAiDictationRuntime } from "./dictation-runtime";
|
|
|
13
13
|
import { AiTurnExecutor } from "./executor";
|
|
14
14
|
import { canonicalizeAiConversationAttachments, snapshotAiConversationFiles } from "./file-context";
|
|
15
15
|
import { drainQueuedMessages } from "./message-queue";
|
|
16
|
+
import { AI_TURN_LEASE_MS } from "./protocol";
|
|
16
17
|
import { aiQuotas } from "./quotas";
|
|
17
18
|
import { parseAiResourceMarker } from "./resource-markers";
|
|
18
19
|
import { isAiVisionModelConfigured } from "./settings";
|
|
@@ -43,7 +44,6 @@ export { isAiSettingsError, validateAiTurnRequest } from "./validate";
|
|
|
43
44
|
const log = logger("ai:runtime");
|
|
44
45
|
|
|
45
46
|
const AI_WORKER_ID = `worker-${crypto.randomUUID()}`;
|
|
46
|
-
const AI_TURN_LEASE_MS = 45_000;
|
|
47
47
|
const AI_TURN_HEARTBEAT_MS = 3_000;
|
|
48
48
|
const AI_TURN_WORKER_CONCURRENCY = 8;
|
|
49
49
|
const AI_TURN_MAX_ATTEMPTS = 5;
|
package/src/ai/settings.ts
CHANGED
|
@@ -1,9 +1,17 @@
|
|
|
1
1
|
import { z } from "zod";
|
|
2
2
|
import { coreSettings } from "../services";
|
|
3
3
|
import { AiModelPricingSchema } from "../shared/ai-costs";
|
|
4
|
+
import {
|
|
5
|
+
AI_REQUEST_HEADERS_PROVIDER_ERROR,
|
|
6
|
+
AiExtraBodySchema,
|
|
7
|
+
AiReasoningEffortSchema,
|
|
8
|
+
AiRequestHeadersSchema,
|
|
9
|
+
providerSupportsRequestHeaders,
|
|
10
|
+
} from "../shared/ai-request-options";
|
|
4
11
|
import { getAiCredential, listAiCredentialProfileIds } from "./credentials";
|
|
5
12
|
import { AI_FIRECRAWL_API_KEY_SETTING_KEY } from "./firecrawl-tools";
|
|
6
13
|
import { createAiProvider } from "./provider";
|
|
14
|
+
import { getAiRequestHeaders } from "./request-headers";
|
|
7
15
|
import {
|
|
8
16
|
AI_DATA_BOUNDARIES,
|
|
9
17
|
AI_MODEL_CAPABILITIES,
|
|
@@ -93,12 +101,20 @@ const ModelProfileSchema = z
|
|
|
93
101
|
contextWindow: z.number().int().positive().optional(),
|
|
94
102
|
temperature: z.number().min(0).max(2).optional(),
|
|
95
103
|
maxOutputTokens: z.number().int().positive().optional(),
|
|
104
|
+
reasoningEffort: AiReasoningEffortSchema,
|
|
105
|
+
extraBody: AiExtraBodySchema.optional(),
|
|
106
|
+
requestHeaders: AiRequestHeadersSchema.optional(),
|
|
96
107
|
maxLoadedTools: z.number().int().optional(),
|
|
97
108
|
maxToolRounds: z.number().int().optional(),
|
|
98
109
|
pricing: AiModelPricingSchema.optional(),
|
|
99
110
|
})
|
|
100
111
|
.superRefine((profile, ctx) => {
|
|
112
|
+
if (profile.requestHeaders !== undefined && !providerSupportsRequestHeaders(profile.provider))
|
|
113
|
+
ctx.addIssue({ code: "custom", path: ["requestHeaders"], message: AI_REQUEST_HEADERS_PROVIDER_ERROR });
|
|
101
114
|
if (profile.capabilities?.includes("transcription")) {
|
|
115
|
+
for (const key of ["reasoningEffort", "extraBody", "requestHeaders"] as const)
|
|
116
|
+
if (profile[key] !== undefined)
|
|
117
|
+
ctx.addIssue({ code: "custom", path: [key], message: "Request settings are not supported on transcription profiles." });
|
|
102
118
|
if (profile.pricing) ctx.addIssue({ code: "custom", path: ["pricing"], message: "Audio pricing is not supported." });
|
|
103
119
|
if (profile.capabilities.some((capability) => capability !== "transcription")) {
|
|
104
120
|
ctx.addIssue({ code: "custom", path: ["capabilities"], message: "Audio transcription cannot be combined with chat capabilities." });
|
|
@@ -128,7 +144,7 @@ const profileToPublic = (profile: AiModelProfile): AiPublicModelProfile => ({
|
|
|
128
144
|
});
|
|
129
145
|
|
|
130
146
|
const normalizeProfile = (raw: z.infer<typeof ModelProfileSchema>): AiModelProfile => {
|
|
131
|
-
const { capabilities, dataBoundary, dataPolicy: legacyDataPolicy, tags: _legacyTags, ...profile } = raw;
|
|
147
|
+
const { requestHeaders: _requestHeaders, capabilities, dataBoundary, dataPolicy: legacyDataPolicy, tags: _legacyTags, ...profile } = raw;
|
|
132
148
|
return {
|
|
133
149
|
...profile,
|
|
134
150
|
capabilities: normalizeCapabilities(capabilities),
|
|
@@ -522,7 +538,8 @@ export const resolveAiModelFromState = async (
|
|
|
522
538
|
});
|
|
523
539
|
}
|
|
524
540
|
|
|
525
|
-
|
|
541
|
+
const headers = providerSupportsRequestHeaders(profile.provider) ? await getAiRequestHeaders(profile.id) : {};
|
|
542
|
+
return { profile, provider: createAiProvider(profile, credential?.trim() || undefined, headers) };
|
|
526
543
|
};
|
|
527
544
|
|
|
528
545
|
const isUsableProfile = (profile: AiModelProfile, credentialProfileIds: ReadonlySet<string>): boolean =>
|