@intentface/latch-core 0.9.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +45 -0
- package/dist/agent.d.ts +200 -0
- package/dist/agent.d.ts.map +1 -0
- package/dist/agent.js +9 -0
- package/dist/agent.js.map +1 -0
- package/dist/compaction.d.ts +33 -0
- package/dist/compaction.d.ts.map +1 -0
- package/dist/compaction.js +104 -0
- package/dist/compaction.js.map +1 -0
- package/dist/connections.d.ts +16 -0
- package/dist/connections.d.ts.map +1 -0
- package/dist/connections.js +41 -0
- package/dist/connections.js.map +1 -0
- package/dist/context.d.ts +41 -0
- package/dist/context.d.ts.map +1 -0
- package/dist/context.js +25 -0
- package/dist/context.js.map +1 -0
- package/dist/current-date.d.ts +12 -0
- package/dist/current-date.d.ts.map +1 -0
- package/dist/current-date.js +25 -0
- package/dist/current-date.js.map +1 -0
- package/dist/extensions.d.ts +333 -0
- package/dist/extensions.d.ts.map +1 -0
- package/dist/extensions.js +569 -0
- package/dist/extensions.js.map +1 -0
- package/dist/harness/index.d.ts +17 -0
- package/dist/harness/index.d.ts.map +1 -0
- package/dist/harness/index.js +15 -0
- package/dist/harness/index.js.map +1 -0
- package/dist/harness/tools.d.ts +88 -0
- package/dist/harness/tools.d.ts.map +1 -0
- package/dist/harness/tools.js +296 -0
- package/dist/harness/tools.js.map +1 -0
- package/dist/harness/web-fetch.d.ts +47 -0
- package/dist/harness/web-fetch.d.ts.map +1 -0
- package/dist/harness/web-fetch.js +247 -0
- package/dist/harness/web-fetch.js.map +1 -0
- package/dist/index.d.ts +25 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +25 -0
- package/dist/index.js.map +1 -0
- package/dist/limits.d.ts +152 -0
- package/dist/limits.d.ts.map +1 -0
- package/dist/limits.js +97 -0
- package/dist/limits.js.map +1 -0
- package/dist/memory.d.ts +93 -0
- package/dist/memory.d.ts.map +1 -0
- package/dist/memory.js +13 -0
- package/dist/memory.js.map +1 -0
- package/dist/message.d.ts +46 -0
- package/dist/message.d.ts.map +1 -0
- package/dist/message.js +2 -0
- package/dist/message.js.map +1 -0
- package/dist/models/catalog.d.ts +56 -0
- package/dist/models/catalog.d.ts.map +1 -0
- package/dist/models/catalog.js +211 -0
- package/dist/models/catalog.js.map +1 -0
- package/dist/models/defaults.d.ts +23 -0
- package/dist/models/defaults.d.ts.map +1 -0
- package/dist/models/defaults.js +19 -0
- package/dist/models/defaults.js.map +1 -0
- package/dist/models/index.d.ts +23 -0
- package/dist/models/index.d.ts.map +1 -0
- package/dist/models/index.js +19 -0
- package/dist/models/index.js.map +1 -0
- package/dist/models/prompt-caching.d.ts +19 -0
- package/dist/models/prompt-caching.d.ts.map +1 -0
- package/dist/models/prompt-caching.js +18 -0
- package/dist/models/prompt-caching.js.map +1 -0
- package/dist/models/provider.d.ts +19 -0
- package/dist/models/provider.d.ts.map +1 -0
- package/dist/models/provider.js +22 -0
- package/dist/models/provider.js.map +1 -0
- package/dist/models/reasoning.d.ts +21 -0
- package/dist/models/reasoning.d.ts.map +1 -0
- package/dist/models/reasoning.js +59 -0
- package/dist/models/reasoning.js.map +1 -0
- package/dist/pricing.d.ts +52 -0
- package/dist/pricing.d.ts.map +1 -0
- package/dist/pricing.js +37 -0
- package/dist/pricing.js.map +1 -0
- package/dist/principal.d.ts +37 -0
- package/dist/principal.d.ts.map +1 -0
- package/dist/principal.js +30 -0
- package/dist/principal.js.map +1 -0
- package/dist/projections.d.ts +37 -0
- package/dist/projections.d.ts.map +1 -0
- package/dist/projections.js +128 -0
- package/dist/projections.js.map +1 -0
- package/dist/prompt-caching.d.ts +106 -0
- package/dist/prompt-caching.d.ts.map +1 -0
- package/dist/prompt-caching.js +165 -0
- package/dist/prompt-caching.js.map +1 -0
- package/dist/runtime.d.ts +670 -0
- package/dist/runtime.d.ts.map +1 -0
- package/dist/runtime.js +2425 -0
- package/dist/runtime.js.map +1 -0
- package/dist/scheduler.d.ts +31 -0
- package/dist/scheduler.d.ts.map +1 -0
- package/dist/scheduler.js +43 -0
- package/dist/scheduler.js.map +1 -0
- package/dist/storage.d.ts +389 -0
- package/dist/storage.d.ts.map +1 -0
- package/dist/storage.js +38 -0
- package/dist/storage.js.map +1 -0
- package/dist/telemetry.d.ts +155 -0
- package/dist/telemetry.d.ts.map +1 -0
- package/dist/telemetry.js +2 -0
- package/dist/telemetry.js.map +1 -0
- package/dist/vault-node.d.ts +20 -0
- package/dist/vault-node.d.ts.map +1 -0
- package/dist/vault-node.js +30 -0
- package/dist/vault-node.js.map +1 -0
- package/dist/vault.d.ts +62 -0
- package/dist/vault.d.ts.map +1 -0
- package/dist/vault.js +88 -0
- package/dist/vault.js.map +1 -0
- package/package.json +95 -0
|
@@ -0,0 +1,670 @@
|
|
|
1
|
+
import type { Experimental_SandboxSession } from "ai";
|
|
2
|
+
import type { AppMessage } from "./message.js";
|
|
3
|
+
import type { ReconstructPrincipal } from "./principal.js";
|
|
4
|
+
import type { ContextDefinition } from "./context.js";
|
|
5
|
+
import type { AgentConfig, AgentFactory, AgentInfo, ToolSource } from "./agent.js";
|
|
6
|
+
import type { MemoryProvider } from "./memory.js";
|
|
7
|
+
import { type Pricing } from "./pricing.js";
|
|
8
|
+
import { type PromptCachingPlan } from "./prompt-caching.js";
|
|
9
|
+
import type { TurnTrigger, LatchTelemetry } from "./telemetry.js";
|
|
10
|
+
import { type LatchLimits } from "./limits.js";
|
|
11
|
+
import type { ChatRecord, RunRecord, RunStatus, ScheduleRecord, StorageAdapter } from "./storage.js";
|
|
12
|
+
/**
|
|
13
|
+
* The Runtime — composition root + operations facade.
|
|
14
|
+
*
|
|
15
|
+
* `createRuntime(config)` wires the chosen seams (storage, context, agent
|
|
16
|
+
* registry) and exposes the operations. Routes/cron/CLI/tests all call the
|
|
17
|
+
* runtime, never the seams directly. It's stateless and re-creatable (holds
|
|
18
|
+
* adapters, no per-run state — that lives in the DB), so it's serverless-safe.
|
|
19
|
+
*
|
|
20
|
+
* T1: `handleChat` runs one turn with a `ToolLoopAgent` and persists via the
|
|
21
|
+
* stock AI SDK stream helper — `onStepEnd` → runs/usage, `onFinish` → messages.
|
|
22
|
+
*/
|
|
23
|
+
/**
|
|
24
|
+
* T2 durability (opt-in). When enabled, each turn claims a single-writer lease,
|
|
25
|
+
* heartbeats it while streaming, and aborts if the lease is lost. A reaper
|
|
26
|
+
* (`runtime.cron()` → `sweep`, driven by cron) reclaims runs whose owner died.
|
|
27
|
+
*/
|
|
28
|
+
export interface DurabilityConfig {
|
|
29
|
+
enabled?: boolean;
|
|
30
|
+
/** Lease lifetime in ms; a run is reapable once `now > leaseExpiresAt` (default 30s). */
|
|
31
|
+
leaseTtlMs?: number;
|
|
32
|
+
/** Heartbeat cadence in ms; must be comfortably < leaseTtlMs (default 10s). */
|
|
33
|
+
heartbeatMs?: number;
|
|
34
|
+
/** Stable owner id for this runtime instance (default: random per process). */
|
|
35
|
+
instanceId?: string;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* A connection registry seam: resolve a set of connection names to a single
|
|
39
|
+
* per-turn tool source. `@intentface/latch-mcp`'s `McpConnections` satisfies
|
|
40
|
+
* this structurally, so `core` stays decoupled from MCP. When set on the
|
|
41
|
+
* runtime, an agent's `connections: [...]` names are resolved through it.
|
|
42
|
+
*/
|
|
43
|
+
export interface ConnectionRegistry<P> {
|
|
44
|
+
hostFor(names: string[]): ToolSource<P>;
|
|
45
|
+
}
|
|
46
|
+
/**
|
|
47
|
+
* A source of agents resolved at request time (e.g. DB/UI-created), consulted
|
|
48
|
+
* when a name isn't in the static `agents` registry. Lets agents be data, not
|
|
49
|
+
* just code — scoped per `Principal`. `@intentface` platforms back this with a
|
|
50
|
+
* table; `resolve` builds the same `AgentConfig` a code agent would.
|
|
51
|
+
*/
|
|
52
|
+
export interface DynamicAgentStore<P> {
|
|
53
|
+
/** Build an agent's config for this caller, or undefined if not found. */
|
|
54
|
+
resolve(name: string, principal: P): AgentConfig<P> | undefined | Promise<AgentConfig<P> | undefined>;
|
|
55
|
+
/** The caller's dynamic agents, for the picker. */
|
|
56
|
+
list(principal: P): AgentInfo[] | Promise<AgentInfo[]>;
|
|
57
|
+
}
|
|
58
|
+
export interface RuntimeConfig<P, RuntimeContext = unknown> {
|
|
59
|
+
storage: StorageAdapter<P>;
|
|
60
|
+
context: ContextDefinition<P, RuntimeContext>;
|
|
61
|
+
/** The agent registry: name → factory (file-based or DB-driven upstream). */
|
|
62
|
+
agents: Record<string, AgentFactory<P, RuntimeContext>>;
|
|
63
|
+
durability?: DurabilityConfig;
|
|
64
|
+
/** Optional cost calculation. Without it, recorded cost is 0. */
|
|
65
|
+
pricing?: Pricing;
|
|
66
|
+
/**
|
|
67
|
+
* Connection registry used to resolve each agent's `connections: [...]` names
|
|
68
|
+
* into a per-turn tool source (MCP servers / integrations). Optional — without
|
|
69
|
+
* it, `connections` on an agent is ignored.
|
|
70
|
+
*/
|
|
71
|
+
connections?: ConnectionRegistry<P>;
|
|
72
|
+
/**
|
|
73
|
+
* Resolve agents not in the static `agents` registry (e.g. DB/UI-created),
|
|
74
|
+
* per caller. Optional — without it, only code-declared agents exist.
|
|
75
|
+
*/
|
|
76
|
+
dynamicAgents?: DynamicAgentStore<P>;
|
|
77
|
+
/**
|
|
78
|
+
* Memory seam: gives agents that declare `memory` persistent cross-session
|
|
79
|
+
* memory (compiled index injected into instructions + scope-bound
|
|
80
|
+
* `memory_search`/`memory_save` tools). Optional — without it, an agent's
|
|
81
|
+
* `memory` declaration is ignored. See `MemoryProvider`.
|
|
82
|
+
*/
|
|
83
|
+
memory?: MemoryProvider<P>;
|
|
84
|
+
/**
|
|
85
|
+
* Rebuild a `Principal` from the opaque identity blob a run stored, for
|
|
86
|
+
* resume/cron (no request). Default: passthrough (`identity as P`). Override to
|
|
87
|
+
* re-hydrate (e.g. re-fetch fresh identity) or validate on resume.
|
|
88
|
+
*/
|
|
89
|
+
reconstructPrincipal?: ReconstructPrincipal<P>;
|
|
90
|
+
/**
|
|
91
|
+
* Next-fire evaluator for scheduled agents: given a cron expression, a lower
|
|
92
|
+
* bound (epoch ms), and an optional IANA timezone the expression is
|
|
93
|
+
* interpreted in, return the next fire time (epoch ms) or null if it never
|
|
94
|
+
* fires again. Keeps core cron-library-free — wire e.g. `croner`. Required to
|
|
95
|
+
* create recurring schedules.
|
|
96
|
+
*/
|
|
97
|
+
cron?: (expr: string, afterMs: number, timezone?: string) => number | null;
|
|
98
|
+
/**
|
|
99
|
+
* Optional per-fire override for `runDue`, so a host can run a scheduled turn
|
|
100
|
+
* somewhere the default can't reach — typically a chat channel, where the
|
|
101
|
+
* report should land in the conversation the person actually replies in.
|
|
102
|
+
*
|
|
103
|
+
* Return true when the host ran (and delivered) this fire ITSELF; core then
|
|
104
|
+
* skips its own fresh-chat run. Anything else — false, absent, or a throw —
|
|
105
|
+
* falls through to the default, so a host whose delivery target has gone
|
|
106
|
+
* stale (bot disabled, user unlinked) still gets the work done in a chat
|
|
107
|
+
* rather than losing the fire.
|
|
108
|
+
*
|
|
109
|
+
* A host that wants the SAME accounting the default path gets can return the
|
|
110
|
+
* object form instead of a bare boolean: `chatId` lets the next occurrence's
|
|
111
|
+
* `onParked` check see this fire's conversation, and `parked` joins `runDue`'s
|
|
112
|
+
* `parked` count. Core cannot derive either itself — the host owns the chat a
|
|
113
|
+
* channel fire runs in. Omit them and this fire simply isn't accounted (the
|
|
114
|
+
* bare-boolean behavior), which is correct for a host that reports its own.
|
|
115
|
+
*
|
|
116
|
+
* Core keeps owning the claim, the lease and the advance either way, and
|
|
117
|
+
* stays entirely channel-blind: `schedule.delivery` is opaque here and only
|
|
118
|
+
* ever interpreted by this hook.
|
|
119
|
+
*/
|
|
120
|
+
runSchedule?: (input: {
|
|
121
|
+
schedule: ScheduleRecord;
|
|
122
|
+
principal: P;
|
|
123
|
+
}) => Promise<boolean | {
|
|
124
|
+
handled: boolean;
|
|
125
|
+
chatId?: string;
|
|
126
|
+
parked?: boolean;
|
|
127
|
+
}>;
|
|
128
|
+
/**
|
|
129
|
+
* Resolve a per-session sandbox for an agent that declares `sandbox: true`.
|
|
130
|
+
* Returns the AI SDK `Experimental_SandboxSession` passed to the model
|
|
131
|
+
* (`agent.stream({ experimental_sandbox })`) and read by sandbox tools. Keeps
|
|
132
|
+
* core free of any sandbox backend dep — the platform supplies the provider
|
|
133
|
+
* (e.g. just-bash / Daytona). The sandbox is per chat (`chatId`).
|
|
134
|
+
*/
|
|
135
|
+
sandbox?: (args: {
|
|
136
|
+
principal: P;
|
|
137
|
+
chatId: string;
|
|
138
|
+
agent: string;
|
|
139
|
+
provider?: string;
|
|
140
|
+
}) => Experimental_SandboxSession | Promise<Experimental_SandboxSession>;
|
|
141
|
+
/**
|
|
142
|
+
* The default harness tools, supplied by the platform (kept out of core so we
|
|
143
|
+
* carry no sandbox/QuickJS deps). `sandbox` tools (bash/read_file/write_file/
|
|
144
|
+
* glob/grep) operate on the resolved `experimental_sandbox`; `app` tools
|
|
145
|
+
* (web_fetch, askUser, scheduling, …) run in the app process. Merged into an
|
|
146
|
+
* agent's toolset per its `sandbox`/`defaultTools` flags.
|
|
147
|
+
*/
|
|
148
|
+
harnessTools?: {
|
|
149
|
+
sandbox?: Record<string, unknown>;
|
|
150
|
+
/**
|
|
151
|
+
* App-process tools. A static record for principal-independent tools
|
|
152
|
+
* (web_fetch, …), or a per-turn factory when the tools must bind to the
|
|
153
|
+
* caller (e.g. workspace file/db tools scoped to the user's permissions
|
|
154
|
+
* and the current chat).
|
|
155
|
+
*/
|
|
156
|
+
app?: Record<string, unknown> | ((args: {
|
|
157
|
+
principal: unknown;
|
|
158
|
+
chatId: string;
|
|
159
|
+
agent: string;
|
|
160
|
+
}) => Record<string, unknown> | Promise<Record<string, unknown>>);
|
|
161
|
+
/**
|
|
162
|
+
* Platform render tool(s) — attached only when an agent sets
|
|
163
|
+
* `renderTools: true`, never by `defaultTools`. Kept a separate capability
|
|
164
|
+
* (like `sandbox`) because it's platform-specific (needs a deployed render
|
|
165
|
+
* service) and should be opted into explicitly.
|
|
166
|
+
*/
|
|
167
|
+
render?: Record<string, unknown>;
|
|
168
|
+
/**
|
|
169
|
+
* Default approval policy for harness tools, by tool name (e.g.
|
|
170
|
+
* `{ create_schedule: "user-approval" }`). Applied only to tools actually
|
|
171
|
+
* attached to the agent, and merged like a tool source's: the AGENT WINS
|
|
172
|
+
* for names it lists in its own `toolApproval`, so a host can ship a
|
|
173
|
+
* side-effecting harness tool gated by default without freezing that
|
|
174
|
+
* choice.
|
|
175
|
+
*/
|
|
176
|
+
approval?: Record<string, unknown>;
|
|
177
|
+
};
|
|
178
|
+
/**
|
|
179
|
+
* Resolve the model provider's NATIVE web tools (server-side search/fetch) for
|
|
180
|
+
* a model id — kept out of core so it carries no provider SDK dep. Merged into
|
|
181
|
+
* an agent's toolset when it sets `providerWebTools`. Return `{}` for providers
|
|
182
|
+
* without native web tools.
|
|
183
|
+
*/
|
|
184
|
+
resolveProviderTools?: (modelId: string) => Record<string, unknown>;
|
|
185
|
+
/**
|
|
186
|
+
* Resolve an agent's `effort` to provider-specific reasoning options for a
|
|
187
|
+
* model id — kept out of core (no provider SDK dep). Returns the
|
|
188
|
+
* `providerOptions` (keyed by provider id) and an optional `maxOutputTokens`
|
|
189
|
+
* (token-budget providers need the budget to stay under the model's max).
|
|
190
|
+
* Return `undefined` to leave the provider default (e.g. non-reasoning model).
|
|
191
|
+
*/
|
|
192
|
+
resolveReasoningOptions?: (modelId: string, effort: string) => {
|
|
193
|
+
providerOptions?: Record<string, unknown>;
|
|
194
|
+
maxOutputTokens?: number;
|
|
195
|
+
} | undefined;
|
|
196
|
+
/**
|
|
197
|
+
* Resolve the prompt-caching plan for a model id. Supply it when you own a
|
|
198
|
+
* model catalog (you know the provider family better than the model object
|
|
199
|
+
* does) or need per-principal cache routing — typically
|
|
200
|
+
* `promptCachingPlanFor(providerOf(modelId), { agent, cacheKeySuffix })`.
|
|
201
|
+
* `memoryScoped` is true when the turn's system prompt is per-principal
|
|
202
|
+
* (memory-enabled agent) — use it to scope OpenAI's cache routing key.
|
|
203
|
+
*
|
|
204
|
+
* ABSENT hook → caching defaults ON for first-party Anthropic/OpenAI models
|
|
205
|
+
* (see `defaultPromptCachingPlan`; measured ~6x unit cost on MCP-heavy
|
|
206
|
+
* agents). Opt out with `resolvePromptCaching: () => undefined`, or per turn
|
|
207
|
+
* with `promptCaching: false`. Returning `undefined` for a model disables
|
|
208
|
+
* caching for it.
|
|
209
|
+
*/
|
|
210
|
+
resolvePromptCaching?: (modelId: string, args: {
|
|
211
|
+
agent: string;
|
|
212
|
+
memoryScoped: boolean;
|
|
213
|
+
principal: P;
|
|
214
|
+
}) => PromptCachingPlan | undefined;
|
|
215
|
+
/**
|
|
216
|
+
* The observation seam (kept out of core — no OTEL/vendor dep): per-run AI
|
|
217
|
+
* SDK telemetry options, an optional trace wrapper around the model call, a
|
|
218
|
+
* per-step hook and a post-turn hook. A full tracer
|
|
219
|
+
* (`@intentface/latch-langfuse`) or just the callbacks a host wants (usage
|
|
220
|
+
* rollups, per-tool analytics, "a scheduled fire parked — notify someone").
|
|
221
|
+
* Fires for subagent turns too — distinguish via `depth` and `trigger`.
|
|
222
|
+
* Optional — without it, nothing is observed and behavior is unchanged.
|
|
223
|
+
*/
|
|
224
|
+
telemetry?: LatchTelemetry<P>;
|
|
225
|
+
/**
|
|
226
|
+
* Per org / user / agent budgets on turns, tokens and cost (see `limits.ts`).
|
|
227
|
+
* The host names the counters a turn counts against and their caps; the
|
|
228
|
+
* runtime refuses at admission (`LimitExceededError`, HTTP 429), stops the
|
|
229
|
+
* tool loop at the tightest remaining token budget, and settles usage on
|
|
230
|
+
* finish. Optional — without it, nothing is limited.
|
|
231
|
+
*/
|
|
232
|
+
limits?: LatchLimits<P>;
|
|
233
|
+
}
|
|
234
|
+
export interface HandleChatInput<P> {
|
|
235
|
+
/** Which registered agent to run. */
|
|
236
|
+
agent: string;
|
|
237
|
+
/** Client-chosen chat id (create-or-continue). Omit to start a new chat. */
|
|
238
|
+
chatId?: string;
|
|
239
|
+
/** Resolved by the caller's ContextResolver (isolation flows from this). */
|
|
240
|
+
principal: P;
|
|
241
|
+
/** The incoming user message. */
|
|
242
|
+
message: AppMessage;
|
|
243
|
+
/** Optional original request, passed to context.build. */
|
|
244
|
+
request?: Request;
|
|
245
|
+
/**
|
|
246
|
+
* Per-turn request payload, passed to `context.build` and the agent factory.
|
|
247
|
+
* EPHEMERAL: never persisted with the run (only the principal is stored as
|
|
248
|
+
* `runs.identity`), so it is absent on resume/cron. Use it for request-scoped
|
|
249
|
+
* inputs (active document, UI selection, parsed body fields) — never identity.
|
|
250
|
+
*/
|
|
251
|
+
turnContext?: unknown;
|
|
252
|
+
}
|
|
253
|
+
export interface Runtime<P> {
|
|
254
|
+
/** The agents available to the caller (code-declared + dynamic) — for a picker. */
|
|
255
|
+
listAgents(principal: P): Promise<AgentInfo[]>;
|
|
256
|
+
/** Run one turn; returns a streaming UIMessage `Response`. */
|
|
257
|
+
handleChat(input: HandleChatInput<P>): Promise<Response>;
|
|
258
|
+
/**
|
|
259
|
+
* Run one agent to completion once, autonomously, in an isolated throwaway
|
|
260
|
+
* thread (no streaming) — for programmatic use like the Agent Builder testing
|
|
261
|
+
* an agent it just created. `blockGated` (default true) stubs approval-gated
|
|
262
|
+
* (mutating) tools so a test run has no real side effects; pass false for a
|
|
263
|
+
* true end-to-end run. Returns the throwaway threadId + the final answer.
|
|
264
|
+
*/
|
|
265
|
+
runAgent(input: {
|
|
266
|
+
principal: P;
|
|
267
|
+
agent: string;
|
|
268
|
+
prompt: string;
|
|
269
|
+
threadId?: string;
|
|
270
|
+
blockGated?: boolean;
|
|
271
|
+
/**
|
|
272
|
+
* Per-turn request payload, passed to `context.build` and the agent
|
|
273
|
+
* factory (see `HandleChatInput.turnContext`). Ephemeral — never persisted,
|
|
274
|
+
* absent on resume.
|
|
275
|
+
*/
|
|
276
|
+
turnContext?: unknown;
|
|
277
|
+
/**
|
|
278
|
+
* Per-turn prompt-caching override. `false` = set no caching provider
|
|
279
|
+
* options for this run (the control arm of an A/B comparison); default on
|
|
280
|
+
* when the runtime has a `resolvePromptCaching` hook.
|
|
281
|
+
*/
|
|
282
|
+
promptCaching?: boolean;
|
|
283
|
+
/**
|
|
284
|
+
* How to classify this run on the telemetry run meta (see `TurnTrigger`).
|
|
285
|
+
* Defaults to `"programmatic"` — there is no client stream here, so the
|
|
286
|
+
* safe assumption is that nobody is watching, and a host that knows better
|
|
287
|
+
* (a webhook with a user waiting) says so by passing `"chat"`.
|
|
288
|
+
*/
|
|
289
|
+
trigger?: TurnTrigger;
|
|
290
|
+
}): Promise<{
|
|
291
|
+
threadId: string;
|
|
292
|
+
answer: string;
|
|
293
|
+
}>;
|
|
294
|
+
/**
|
|
295
|
+
* Summarize a conversation and append a compaction boundary, so later turns
|
|
296
|
+
* send the model the summary instead of the turns it replaces. Stored history
|
|
297
|
+
* is untouched — this shrinks the model projection, not the record. Null when
|
|
298
|
+
* there's too little to compact (or the chat isn't the caller's); throws
|
|
299
|
+
* `PendingApprovalError` when the last turn is paused awaiting a human.
|
|
300
|
+
*/
|
|
301
|
+
compactChat(input: {
|
|
302
|
+
chatId: string;
|
|
303
|
+
principal: P;
|
|
304
|
+
/** Optional steer for the summary ("keep the training plan"). */
|
|
305
|
+
focus?: string;
|
|
306
|
+
request?: Request;
|
|
307
|
+
turnContext?: unknown;
|
|
308
|
+
}): Promise<{
|
|
309
|
+
summary: string;
|
|
310
|
+
compacted: number;
|
|
311
|
+
} | null>;
|
|
312
|
+
/**
|
|
313
|
+
* Client-projected conversation history (`toClientMessages`): the stored
|
|
314
|
+
* messages with `internal`/`redacted` ones filtered out. For rendering past
|
|
315
|
+
* turns on load.
|
|
316
|
+
*/
|
|
317
|
+
loadHistory(input: {
|
|
318
|
+
chatId: string;
|
|
319
|
+
principal: P;
|
|
320
|
+
}): Promise<AppMessage[]>;
|
|
321
|
+
/**
|
|
322
|
+
* The caller's chats, most-recent first (for a chat list / sidebar).
|
|
323
|
+
* `kind` filters to one chat kind (e.g. `'user'` to hide the runtime's
|
|
324
|
+
* internal `sub_…` threads); omitted → all kinds (back-compatible default).
|
|
325
|
+
*/
|
|
326
|
+
listChats(input: {
|
|
327
|
+
principal: P;
|
|
328
|
+
limit?: number;
|
|
329
|
+
kind?: string;
|
|
330
|
+
}): Promise<ChatRecord[]>;
|
|
331
|
+
/**
|
|
332
|
+
* One chat by id, scoped to the caller (null when it isn't theirs or doesn't
|
|
333
|
+
* exist). A direct lookup, not a capped scan — so "does this chat exist, and
|
|
334
|
+
* whose agent is it?" stays correct for a tenant with any number of chats.
|
|
335
|
+
*/
|
|
336
|
+
getChat(input: {
|
|
337
|
+
principal: P;
|
|
338
|
+
id: string;
|
|
339
|
+
}): Promise<ChatRecord | null>;
|
|
340
|
+
/** The caller's runs, most-recent first (status, usage, cost — for an ops view). */
|
|
341
|
+
listRuns(input: {
|
|
342
|
+
principal: P;
|
|
343
|
+
limit?: number;
|
|
344
|
+
}): Promise<RunRecord[]>;
|
|
345
|
+
/**
|
|
346
|
+
* One run by id, scoped to the caller: returns it only when the run's chat is
|
|
347
|
+
* owned by `principal` (else null). A direct lookup — no capped scan — so it's
|
|
348
|
+
* safe for ownership checks on arbitrarily old runs.
|
|
349
|
+
*/
|
|
350
|
+
getRun(input: {
|
|
351
|
+
principal: P;
|
|
352
|
+
runId: string;
|
|
353
|
+
}): Promise<RunRecord | null>;
|
|
354
|
+
/**
|
|
355
|
+
* Resume an interrupted run (requires durability). Re-claims it exactly-once,
|
|
356
|
+
* re-runs the turn from persisted history, and completes it server-side.
|
|
357
|
+
* Returns the run's final state, or null if it wasn't resumable (already
|
|
358
|
+
* completed, or claimed by another node).
|
|
359
|
+
*/
|
|
360
|
+
resume(runId: string): Promise<RunOutcome | null>;
|
|
361
|
+
/**
|
|
362
|
+
* Reaper + auto-resume in one call — the durability half of `cron()`. Reaps
|
|
363
|
+
* lease-expired runs, then resumes each whose attempt count is still under
|
|
364
|
+
* `maxAttempts` (default 5; the cap stops poison runs from looping forever).
|
|
365
|
+
* Safe to fire on every node: reap and reclaim are atomic, so each run is
|
|
366
|
+
* handled exactly once. Requires durability.
|
|
367
|
+
*/
|
|
368
|
+
sweep(opts?: {
|
|
369
|
+
now?: number;
|
|
370
|
+
maxAttempts?: number;
|
|
371
|
+
error?: string;
|
|
372
|
+
}): Promise<{
|
|
373
|
+
reaped: RunRecord[];
|
|
374
|
+
resumed: RunOutcome[];
|
|
375
|
+
}>;
|
|
376
|
+
/**
|
|
377
|
+
* THE cron entry point — point a platform cron or the in-process scheduler at
|
|
378
|
+
* this and nothing else. Fires due schedules (`runDue`), then, when
|
|
379
|
+
* durability is on, reaps lease-expired runs and resumes them (`sweep`). One
|
|
380
|
+
* call per tick, so a deploy can't wire half of the operational surface. The
|
|
381
|
+
* two phases are isolated: a failure in one lands in `errors` and the other
|
|
382
|
+
* still runs. Safe to fire on every node concurrently (claims are atomic).
|
|
383
|
+
*/
|
|
384
|
+
cron(opts?: {
|
|
385
|
+
now?: number;
|
|
386
|
+
maxAttempts?: number;
|
|
387
|
+
}): Promise<{
|
|
388
|
+
due: {
|
|
389
|
+
fired: number;
|
|
390
|
+
errors: number;
|
|
391
|
+
skipped: number;
|
|
392
|
+
parked: number;
|
|
393
|
+
};
|
|
394
|
+
reaped: RunRecord[];
|
|
395
|
+
resumed: RunOutcome[];
|
|
396
|
+
/** Phase failures (`"runDue: …"` / `"sweep: …"`) — reported, never thrown. */
|
|
397
|
+
errors: string[];
|
|
398
|
+
}>;
|
|
399
|
+
/**
|
|
400
|
+
* Apply HITL tool-approval decisions and continue the turn. Records each
|
|
401
|
+
* decision onto the persisted assistant message (approval-requested →
|
|
402
|
+
* approval-responded), then re-runs the agent from history: approved tools
|
|
403
|
+
* execute, denied ones are skipped. Returns the continuation's streaming
|
|
404
|
+
* Response. No suspension machinery — approvals round-trip as message state.
|
|
405
|
+
*/
|
|
406
|
+
applyApproval(input: {
|
|
407
|
+
chatId: string;
|
|
408
|
+
principal: P;
|
|
409
|
+
decisions: ApprovalDecision[];
|
|
410
|
+
request?: Request;
|
|
411
|
+
/** Per-turn request payload (see HandleChatInput.turnContext). */
|
|
412
|
+
turnContext?: unknown;
|
|
413
|
+
}): Promise<Response>;
|
|
414
|
+
/**
|
|
415
|
+
* Resume a turn paused on a client-handled tool (no `execute`, e.g.
|
|
416
|
+
* `askUser`): record the supplied output onto the matching tool call and
|
|
417
|
+
* re-run from history. Returns the continuation stream, or an empty 200 if
|
|
418
|
+
* other client tool calls in the step still need results.
|
|
419
|
+
*/
|
|
420
|
+
applyToolResult(input: {
|
|
421
|
+
chatId: string;
|
|
422
|
+
principal: P;
|
|
423
|
+
toolCallId: string;
|
|
424
|
+
output: unknown;
|
|
425
|
+
request?: Request;
|
|
426
|
+
/** Per-turn request payload (see HandleChatInput.turnContext). */
|
|
427
|
+
turnContext?: unknown;
|
|
428
|
+
}): Promise<Response>;
|
|
429
|
+
/**
|
|
430
|
+
* Schedule an agent to run on a cron expression (or `cron: null` for one-shot,
|
|
431
|
+
* fired on the next `runDue`). Captures the caller's identity for fire-time.
|
|
432
|
+
* Requires `cron` in the runtime config for recurring schedules.
|
|
433
|
+
*/
|
|
434
|
+
schedule(input: {
|
|
435
|
+
principal: P;
|
|
436
|
+
agent: string;
|
|
437
|
+
cron: string | null;
|
|
438
|
+
/** IANA timezone the cron is interpreted in (e.g. "Europe/Helsinki"). */
|
|
439
|
+
timezone?: string;
|
|
440
|
+
prompt: string;
|
|
441
|
+
/**
|
|
442
|
+
* Opaque, host-interpreted delivery target for each fire (see
|
|
443
|
+
* `RuntimeConfig.runSchedule`). Stored verbatim; core never reads it.
|
|
444
|
+
*/
|
|
445
|
+
delivery?: unknown;
|
|
446
|
+
/**
|
|
447
|
+
* What a fire does while the previous fire's chat is still parked waiting
|
|
448
|
+
* on a human (see `ScheduleRecord.onParked`). Default `"skip"`.
|
|
449
|
+
*/
|
|
450
|
+
onParked?: "skip" | "fire";
|
|
451
|
+
}): Promise<ScheduleRecord>;
|
|
452
|
+
/** The caller's schedules. */
|
|
453
|
+
listSchedules(input: {
|
|
454
|
+
principal: P;
|
|
455
|
+
}): Promise<ScheduleRecord[]>;
|
|
456
|
+
/** Delete one of the caller's schedules. */
|
|
457
|
+
unschedule(input: {
|
|
458
|
+
principal: P;
|
|
459
|
+
id: string;
|
|
460
|
+
}): Promise<void>;
|
|
461
|
+
/**
|
|
462
|
+
* Make one of the caller's schedules due immediately. It then fires through
|
|
463
|
+
* the ordinary `runDue` path — same claim, same delivery, same advance — so
|
|
464
|
+
* "run now" can never drift from what the cron actually does.
|
|
465
|
+
*
|
|
466
|
+
* False when nothing was armed: unknown id, not the caller's, or already
|
|
467
|
+
* spent.
|
|
468
|
+
*/
|
|
469
|
+
runScheduleNow(input: {
|
|
470
|
+
principal: P;
|
|
471
|
+
id: string;
|
|
472
|
+
}): Promise<boolean>;
|
|
473
|
+
/**
|
|
474
|
+
* Fire all due schedules across ALL tenants (system op — the cron entry point).
|
|
475
|
+
* Claims each, consumes the occurrence (advances `nextRunAt` BEFORE firing,
|
|
476
|
+
* so no concurrent claimer can fire it again), reconstructs its principal,
|
|
477
|
+
* and runs the turn in a fresh chat (or via `runSchedule`, when the host
|
|
478
|
+
* handles the fire). Point a platform cron (or the in-process scheduler) at
|
|
479
|
+
* this.
|
|
480
|
+
*
|
|
481
|
+
* Occurrences are at-most-once: a process that dies mid-fire does not retry
|
|
482
|
+
* the occurrence — the run row survives and `sweep()` resumes the turn
|
|
483
|
+
* instead, so recovery never means a second paid attempt.
|
|
484
|
+
*
|
|
485
|
+
* Awaits each fired turn to completion before returning. A driver that also
|
|
486
|
+
* gates on this promise (the in-process interval scheduler does) is stalled
|
|
487
|
+
* for the duration of the longest turn — fan out per tenant rather than
|
|
488
|
+
* looping serially if that matters to you.
|
|
489
|
+
*
|
|
490
|
+
* `skipped` counts occurrences consumed WITHOUT running because the previous
|
|
491
|
+
* fire's chat is still parked on a human (see `ScheduleRecord.onParked`) —
|
|
492
|
+
* distinct from `errors` so a quiet schedule is legible as "waiting on you",
|
|
493
|
+
* never mistaken for "ran fine" or "broken".
|
|
494
|
+
*
|
|
495
|
+
* `parked` counts fires that RAN and then stopped waiting on a human (an
|
|
496
|
+
* approval or question they cannot answer) — tokens spent, no answer
|
|
497
|
+
* delivered. They are counted in `fired` too: the turn did happen. A
|
|
498
|
+
* non-zero `parked` is the signal to go look — a cron whose every fire parks
|
|
499
|
+
* is paying for nothing, and for a one-shot schedule there is no later
|
|
500
|
+
* occurrence to notice. Chats awaiting a human are queryable via
|
|
501
|
+
* `ChatRecord.lastRunStatus`.
|
|
502
|
+
*/
|
|
503
|
+
runDue(opts?: {
|
|
504
|
+
now?: number;
|
|
505
|
+
}): Promise<{
|
|
506
|
+
fired: number;
|
|
507
|
+
errors: number;
|
|
508
|
+
skipped: number;
|
|
509
|
+
parked: number;
|
|
510
|
+
}>;
|
|
511
|
+
}
|
|
512
|
+
/** A user's decision on one pending tool-approval request. */
|
|
513
|
+
export interface ApprovalDecision {
|
|
514
|
+
approvalId: string;
|
|
515
|
+
approved: boolean;
|
|
516
|
+
reason?: string;
|
|
517
|
+
}
|
|
518
|
+
/** The terminal state of a resumed run. */
|
|
519
|
+
export interface RunOutcome {
|
|
520
|
+
runId: string;
|
|
521
|
+
status: RunStatus;
|
|
522
|
+
attempt: number | null;
|
|
523
|
+
}
|
|
524
|
+
/**
|
|
525
|
+
* The minimal `UIMessageStreamWriter` surface we use: `merge` to fold in the
|
|
526
|
+
* model's stream, and `write` for custom data parts (exposed to tools/context).
|
|
527
|
+
*/
|
|
528
|
+
export interface TurnWriter {
|
|
529
|
+
write(part: unknown): void;
|
|
530
|
+
merge(stream: ReadableStream<unknown>): void;
|
|
531
|
+
}
|
|
532
|
+
/**
|
|
533
|
+
* The `data-subagent-progress` part `spawn_agent` streams into the PARENT
|
|
534
|
+
* turn while a subagent runs (one part, reconciled in place by id) — lets the
|
|
535
|
+
* host render the delegated conversation live instead of a silent spinner.
|
|
536
|
+
* Written immediately on spawn (empty `messages`) so the client learns the
|
|
537
|
+
* sub-thread id up front, then re-written (throttled) as the subagent streams.
|
|
538
|
+
*/
|
|
539
|
+
export interface SubagentProgressData {
|
|
540
|
+
/** The spawn_agent tool call this progress belongs to. */
|
|
541
|
+
toolCallId?: string;
|
|
542
|
+
agent: string;
|
|
543
|
+
threadId: string;
|
|
544
|
+
/** The sub-thread's current exchange: the prompt + the streaming assistant snapshot. */
|
|
545
|
+
messages: AppMessage[];
|
|
546
|
+
}
|
|
547
|
+
/** A subagent's undecided approval, lifted so the parent turn can surface it. */
|
|
548
|
+
export interface SubagentPending {
|
|
549
|
+
approvalId: string;
|
|
550
|
+
toolName: string;
|
|
551
|
+
summary: string;
|
|
552
|
+
}
|
|
553
|
+
/**
|
|
554
|
+
* Thrown when work is attempted on a chat whose last turn is still paused
|
|
555
|
+
* awaiting a human — by `compactChat`, where the pause can also be a parked
|
|
556
|
+
* client tool call and there's no sensible way to summarize around it. Carries
|
|
557
|
+
* a stable `code` so the handler can map it to a 409 and the client can prompt
|
|
558
|
+
* the user to resolve it first.
|
|
559
|
+
*
|
|
560
|
+
* `handleChat` deliberately does NOT throw this: a new message supersedes the
|
|
561
|
+
* pause (see `declinePendingApprovals`).
|
|
562
|
+
*/
|
|
563
|
+
export declare class PendingApprovalError extends Error {
|
|
564
|
+
readonly code = "pending_approval";
|
|
565
|
+
constructor(message: string);
|
|
566
|
+
}
|
|
567
|
+
/**
|
|
568
|
+
* What crash recovery should DO with a run it just reclaimed, read off the tail
|
|
569
|
+
* of persisted history.
|
|
570
|
+
*
|
|
571
|
+
* The subtlety this exists for: finishing the tool loop and RECORDING that fact
|
|
572
|
+
* are two different writes. The per-step checkpoint is fire-and-forget at each
|
|
573
|
+
* step's end, while the run row only becomes `completed` in `onFinish`. Die in
|
|
574
|
+
* that gap and the row still says `active` with a lease — so the reaper takes a
|
|
575
|
+
* run whose work is already done and whose answer is already stored.
|
|
576
|
+
*
|
|
577
|
+
* Re-running the model there is worse than useless: it pays for a second
|
|
578
|
+
* generation, and (because the SDK reuses the message id) the new text lands in
|
|
579
|
+
* the same message as the old, so the reader can end up with the answer twice.
|
|
580
|
+
* Reconciling the row costs nothing and loses nothing.
|
|
581
|
+
*
|
|
582
|
+
* - `superseded` — not produced here but by `supersededBy`, from the runs
|
|
583
|
+
* table: the chat opened a later turn while this run sat reaped. Settle it as
|
|
584
|
+
* errored with the same "superseded" outcome `openTurn` gives a stale run.
|
|
585
|
+
* - `decision-pending` — a gate is awaiting a human. The row never got its
|
|
586
|
+
* `awaiting_input`; settle it there and let `applyApproval` drive the rest.
|
|
587
|
+
* (Reaping is scoped to `active`, so this only ever arrives via that gap.)
|
|
588
|
+
* - `loop-finished` — the last step called no tool, which is exactly the
|
|
589
|
+
* condition on which the tool loop stops; or it is waiting on a CLIENT-
|
|
590
|
+
* handled tool (askUser), which `onFinish` also records as completed, the
|
|
591
|
+
* answer arriving later via `applyToolResult`. The turn is done; settle it.
|
|
592
|
+
* - `continue` — there is real work left (a tool result to consume, a step that
|
|
593
|
+
* produced nothing, or no assistant message at all). Re-run the model.
|
|
594
|
+
*
|
|
595
|
+
* The yardstick for every verdict: what `onFinish` would have written had the
|
|
596
|
+
* process lived. That is what makes each one checkable — a `stopWhen` cut (step
|
|
597
|
+
* cap, token budget) leaves a tool part in the final step and so reads as
|
|
598
|
+
* `continue`, which costs a wasted turn but never loses work; the reverse
|
|
599
|
+
* mistake, calling an unfinished turn finished, would silently drop the user's
|
|
600
|
+
* request.
|
|
601
|
+
*
|
|
602
|
+
* The final step is the parts after the last `step-start`; a tool part there
|
|
603
|
+
* means the loop had gone round again and was still going.
|
|
604
|
+
*/
|
|
605
|
+
export type ResumeShape = "superseded" | "decision-pending" | "loop-finished" | "continue";
|
|
606
|
+
/**
|
|
607
|
+
* Has this chat opened another turn since `run` started? Then `run` was
|
|
608
|
+
* superseded — the conversation moved on while it sat reaped — and it must
|
|
609
|
+
* settle rather than run: re-running would generate against a conversation that
|
|
610
|
+
* has moved on and, because the message id is reused, append the result to the
|
|
611
|
+
* later turn's message.
|
|
612
|
+
*
|
|
613
|
+
* Read off the RUNS, not the messages. Every user message is persisted together
|
|
614
|
+
* with its run (`openTurn`), so a later run in the chat is the one fact that
|
|
615
|
+
* proves a later turn exists. Messages cannot tell us this: user messages carry
|
|
616
|
+
* no run id, and when a turn crashes before its first checkpoint the previous
|
|
617
|
+
* turn's assistant message is legitimately the last assistant in history —
|
|
618
|
+
* reading either as "someone else's" marks a live run superseded and drops the
|
|
619
|
+
* user's request without a model call, which is the one failure worse than the
|
|
620
|
+
* bug this all fixes.
|
|
621
|
+
*
|
|
622
|
+
* Compaction runs summarise the conversation so far; they are not turns.
|
|
623
|
+
*/
|
|
624
|
+
export declare function supersededBy(run: Pick<RunRecord, "id" | "chatId" | "startedAt">, runs: readonly RunRecord[]): boolean;
|
|
625
|
+
/**
|
|
626
|
+
* What a reclaimed run's OWN history says it needs. Pure: whether the chat has
|
|
627
|
+
* since moved on is `supersededBy`'s question, answered from the runs table.
|
|
628
|
+
*/
|
|
629
|
+
export declare function resumeShapeOf(messages: AppMessage[]): ResumeShape;
|
|
630
|
+
/**
|
|
631
|
+
* The resume paths ({@link applyApproval} / {@link applyToolResult}) re-run the
|
|
632
|
+
* model from history whose LAST message is the assistant turn that paused — no
|
|
633
|
+
* fresh user message is appended. If that turn streamed trailing user-facing
|
|
634
|
+
* text before pausing (e.g. "…ready to create the page but needs your approval
|
|
635
|
+
* first"), the model projection ends on an assistant message. Some providers
|
|
636
|
+
* reject that ("This model does not support assistant message prefill. The
|
|
637
|
+
* conversation must end with a user message.").
|
|
638
|
+
*
|
|
639
|
+
* Drop the trailing text/reasoning that follows the final TOOL part of the last
|
|
640
|
+
* assistant message, so the projection ends on the tool result (a user turn) and
|
|
641
|
+
* the model generates a fresh continuation. Returns a shallow copy touching only
|
|
642
|
+
* the last message.
|
|
643
|
+
*
|
|
644
|
+
* **The anchor must be a part the model projection keeps.** `toModelMessages`
|
|
645
|
+
* drops every `data-*` part, so anchoring on a bubbled `data-subagent-approval`
|
|
646
|
+
* kept the text BEFORE it — leaving the projection ending on assistant text,
|
|
647
|
+
* i.e. exactly the prefill this exists to prevent. That is the shape the
|
|
648
|
+
* agent-builder produces on every approval round: `… tool-spawn_agent,
|
|
649
|
+
* data-subagent-progress, data-subagent-approval, step-start, text,
|
|
650
|
+
* data-subagent-approval`. Only `tool-*` / `dynamic-tool` anchor.
|
|
651
|
+
*
|
|
652
|
+
* `data-*` parts after the anchor are KEPT: the resumed turn's `onFinish`
|
|
653
|
+
* re-persists this message, so removing them would erase the approval markers
|
|
654
|
+
* the UI renders and `hasPendingApproval` reads. Trailing `step-start` is
|
|
655
|
+
* DROPPED along with the text: it is invisible content, but it instructs
|
|
656
|
+
* `convertToModelMessages` to open a NEW assistant message — and with the text
|
|
657
|
+
* gone, that message is empty, so the projection would still end on the
|
|
658
|
+
* assistant (`{ role: "assistant", content: [] }`), trading the prefill
|
|
659
|
+
* rejection for an empty-content one. The only thing in that step was the
|
|
660
|
+
* narration being removed, so history loses nothing but an empty divider.
|
|
661
|
+
* The trailing text is dropped from stored history too — it is provisional
|
|
662
|
+
* commentary about a decision that has now been made; the resumed turn
|
|
663
|
+
* regenerates the real continuation.
|
|
664
|
+
*/
|
|
665
|
+
export declare function trimTrailingAssistantPrefill(messages: AppMessage[], opts?: {
|
|
666
|
+
unanchored?: boolean;
|
|
667
|
+
}): AppMessage[];
|
|
668
|
+
export declare function clientErrorMessage(error: unknown): string;
|
|
669
|
+
export declare function createRuntime<P, RC = unknown>(config: RuntimeConfig<P, RC>): Runtime<P>;
|
|
670
|
+
//# sourceMappingURL=runtime.d.ts.map
|