@empyria/restate 0.1.23 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/AGENTS.md CHANGED
@@ -24,6 +24,18 @@ is not the right default here.
24
24
  - Relative imports must include explicit `.js` extensions — Bun tolerates missing ones, Node's
25
25
  ESM resolver doesn't.
26
26
 
27
+ ## Agent entry point (`lib/agent/`)
28
+
29
+ `lib/agent/` (merged in from the former `@empyria/restate-llm`) is exported only through
30
+ `@empyria/restate/agent` ([agent.js](./agent.js)), never from [index.js](./index.js). The AI
31
+ SDK packages (`ai`, `@ai-sdk/*`) are **optional** peer dependencies (and devDependencies for
32
+ the tests): a consumer that only uses Admin/Cron/Pubsub must be able to import the package
33
+ root without them installed. Never import `lib/agent/` from a root-exported module.
34
+
35
+ Agent tests use `test/agent/fakes.js`'s `fakeCtx`, which mimics the journal's JSON round trip
36
+ and a bounded `ctx.run` turning an exhausted retry into a `TerminalError` — keep it in step
37
+ with the real SDK behaviour when touching either.
38
+
27
39
  ## No official Admin API client
28
40
 
29
41
  Verified directly against the published `@restatedev/restate-sdk`/`-clients` package
package/README.md CHANGED
@@ -51,6 +51,66 @@ import { checkServiceHandler } from '@empyria/restate/lib/Admin.js'
51
51
  | [lib/Pubsub.js](./lib/Pubsub.js) | `definePubsub` (register a pubsub Virtual Object on your endpoint), `pubsubPublisher` (publish into it from inside a handler, in-process), `pubsubClient` (publish/pull/subscribe to it from anywhere else, over the network) — thin wrappers around `@restatedev/pubsub`/`@restatedev/pubsub-client`. |
52
52
  | [lib/Validation.js](./lib/Validation.js) | `withValidation` — wraps a workflow handler with input/output JSON schema validation via `@empyria/common`. |
53
53
 
54
+ ## Agents: `@empyria/restate/agent`
55
+
56
+ LLM agents as Restate building blocks (formerly `@empyria/restate-llm`). A separate entry
57
+ point, so the root import never loads the AI SDK. It needs the optional peer dependencies:
58
+
59
+ ```bash
60
+ bun add ai @ai-sdk/openai-compatible @ai-sdk/provider
61
+ ```
62
+
63
+ ```js
64
+ import { defineAgent } from '@empyria/restate/agent'
65
+
66
+ export const triage = defineAgent({
67
+ name: 'CorporateActionTriage',
68
+ env: { LLM_BASE_URL: 'https://agent-bureau.vip/v1', MODEL_ID: 'vllm/ornith-1.5' },
69
+ system: 'You triage corporate action notices.',
70
+ tools: {
71
+ // a Restate call to another building block
72
+ lookupIssuer: {
73
+ description: 'Look up an issuer by ISIN',
74
+ inputSchema: {
75
+ type: 'object',
76
+ properties: { isin: { type: 'string' } },
77
+ required: ['isin'],
78
+ },
79
+ block: 'IssuerService',
80
+ handler: 'get',
81
+ },
82
+ // a direct side effect in its own bounded ctx.run, gated by approval
83
+ notifyDesk: {
84
+ description: 'Notify the operations desk',
85
+ inputSchema: { type: 'object', properties: { message: { type: 'string' } } },
86
+ execute: async ({ message }, { idempotencyKey }) => sendToDesk(message, idempotencyKey),
87
+ approval: {
88
+ timeout: 4 * 3600_000,
89
+ notify: { block: 'ApprovalInbox', handler: 'request' },
90
+ },
91
+ },
92
+ },
93
+ output: TriageResultSchema, // the model answers through a final_answer tool
94
+ memory: 'none', // or 'session': a virtual object keyed by session, plus a reset handler
95
+ limits: { maxSteps: 8, maxTokens: 50_000 },
96
+ })
97
+ // register `triage` on your endpoint, then: POST /CorporateActionTriage/ask {"prompt": "…"}
98
+ ```
99
+
100
+ | Module | Purpose |
101
+ | -------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
102
+ | [lib/agent/Agent.js](./lib/agent/Agent.js) | `defineAgent` — builds the service/object and its `ask` handler: validated I/O, `loadContext` that journals only the skill index, session memory (object state or an external `historyStore`), step/token budgets, invocation retry policy. `createAgentService` is the deprecated `AgentService` shortcut. |
103
+ | [lib/agent/Loop.js](./lib/agent/Loop.js) | `runAgentLoop` — one round at a time; every LLM call and tool call is its own durable step. Journals model responses, never prompts or prior history. |
104
+ | [lib/agent/Tools.js](./lib/agent/Tools.js) | Tool specs and execution: block tools (`ctx.genericCall`), execute tools (bounded `ctx.run` with an idempotency key), approval through an awakeable with a timeout, `load_skill`, `final_answer`. |
105
+ | [lib/agent/Skills.js](./lib/agent/Skills.js) | Skill bodies stay out of the journal: `load_skill` records `{name, sha256}`, and `expandSkillRefs` puts the body back inside each LLM step. A hash mismatch within one invocation is terminal; across session turns the current body is used. |
106
+ | [lib/agent/Errors.js](./lib/agent/Errors.js) | `toRestateLLMError` (AI SDK error → terminal / `RetryableError` / transient) and the default retry policies for LLM steps, tool steps and the agent invocation. |
107
+ | [lib/agent/Model.js](./lib/agent/Model.js) | `createModel` — AI SDK model for an OpenAI-compatible endpoint (e.g. Bifrost). |
108
+
109
+ Error policy: a tool that fails terminally (rejected or timed-out approval, a callee's
110
+ `TerminalError`, an execute tool out of retries) goes back to the model as an error result,
111
+ so the agent can adapt. Transient errors are left to Restate to retry, and cancellation
112
+ always propagates.
113
+
54
114
  There is no official Restate Admin API client (verified directly against the published
55
115
  `@restatedev/restate-sdk`/`-clients` packages) — `lib/Admin.js` is a hand-written wrapper kept
56
116
  in sync with the live Admin API's OpenAPI spec (`<admin-url>/openapi`) by hand.
package/agent.js ADDED
@@ -0,0 +1,11 @@
1
+ export * from './lib/agent/Agent.js'
2
+
3
+ export * from './lib/agent/Loop.js'
4
+
5
+ export * from './lib/agent/Tools.js'
6
+
7
+ export * from './lib/agent/Model.js'
8
+
9
+ export * from './lib/agent/Errors.js'
10
+
11
+ export * from './lib/agent/Skills.js'
@@ -0,0 +1,246 @@
1
+ import { TerminalError } from '@restatedev/restate-sdk'
2
+ import { defineObject, defineService } from '../Admin.js'
3
+ import { withValidation } from '../Validation.js'
4
+ import { DEFAULT_AGENT_RETRY_POLICY } from './Errors.js'
5
+ import { runAgentLoop } from './Loop.js'
6
+ import { createModel } from './Model.js'
7
+ import { assertToolSpecs } from './Tools.js'
8
+
9
+ const HISTORY = 'history'
10
+ const HISTORY_REF = 'historyRef'
11
+
12
+ export const AskInputSchema = {
13
+ type: 'object',
14
+ properties: {
15
+ prompt: { type: 'string', minLength: 1 },
16
+ skillsFolder: { type: 'string' },
17
+ department: { type: 'string' },
18
+ threadId: { type: 'string' },
19
+ resourceId: { type: 'string' },
20
+ },
21
+ required: ['prompt'],
22
+ additionalProperties: false,
23
+ }
24
+
25
+ export const AskOutputSchema = {
26
+ type: 'object',
27
+ properties: {
28
+ text: { type: 'string' },
29
+ steps: { type: 'number' },
30
+ },
31
+ required: ['text', 'steps'],
32
+ additionalProperties: false,
33
+ }
34
+
35
+ /**
36
+ * @typedef {Object} HistoryStore External storage for a session agent's conversation, so
37
+ * the journal and object state only ever hold a reference to it.
38
+ * @property {(ref: string) => Promise<Array<object>>} load Returns the messages saved under
39
+ * `ref`. Must return the same messages for the same `ref` every time: refs are immutable
40
+ * snapshots, e.g. a content hash or a versioned object key.
41
+ * @property {(sessionKey: string, messages: Array<object>) => Promise<string>} save Stores
42
+ * the full conversation as a new snapshot and returns its ref.
43
+ *
44
+ * @typedef {Object} AgentSpec
45
+ * @property {string} name Restate service / object name. Required: each agent is its own
46
+ * building block.
47
+ * @property {{LLM_BASE_URL: string, LLM_API_KEY?: string, MODEL_ID: string}} [env] Builds the
48
+ * model via {@link createModel}, unless `model` is given.
49
+ * @property {import('@ai-sdk/provider').LanguageModelV4} [model]
50
+ * @property {string} [system] Static system prompt, prepended to `loadContext`'s.
51
+ * @property {(input: any) => Promise<{systemPrompt?: string, skills?: Array<{name: string, description: string, body?: string}>}>} [loadContext]
52
+ * Resolves this request's system prompt and skills. Runs in its own `ctx.run`, which
53
+ * records only the system prompt and the skill index — never the skill bodies.
54
+ * @property {(name: string, input: any) => Promise<string|undefined>} [loadSkill] Fetches one
55
+ * skill body when the model asks for it. Defaults to re-running `loadContext` and
56
+ * picking the body by name. The journal records only the skill's name and hash; the
57
+ * body is fetched again before every model call. So it must return the same content
58
+ * for the whole invocation (skills shipped with the deployment do), or the invocation
59
+ * fails terminally rather than send the model changed instructions.
60
+ * @property {Record<string, import('./Tools.js').ToolSpec>} [tools]
61
+ * @property {object} [input] Input JSON Schema. Defaults to {@link AskInputSchema}.
62
+ * @property {(input: any) => string} [prompt] Builds the user prompt from the validated
63
+ * input. Defaults to `input.prompt`.
64
+ * @property {object} [output] Output JSON Schema. When given, the model must answer through
65
+ * the `final_answer` tool and `ask` returns its arguments. Defaults to
66
+ * {@link AskOutputSchema} (`{text, steps}`).
67
+ * @property {'none'|'session'} [memory] `none` (default): a stateless service. `session`: a
68
+ * virtual object keyed by session ID, so turns of one session run one at a time and see
69
+ * the conversation so far.
70
+ * @property {HistoryStore} [historyStore] Session memory only. Without it, the conversation
71
+ * is kept in object state, trimmed to `limits.maxHistoryTurns` — fine for short
72
+ * sessions; long ones should use a store.
73
+ * @property {{maxSteps?: number, maxTokens?: number, maxHistoryTurns?: number}} [limits]
74
+ * @property {import('@restatedev/restate-sdk').RunOptions<any>} [llmRetry] Retry policy of
75
+ * each LLM call step. Defaults to `DEFAULT_LLM_RETRY`.
76
+ * @property {import('@restatedev/restate-sdk').RetryPolicy} [retryPolicy] Invocation-level
77
+ * retry policy. Defaults to {@link DEFAULT_AGENT_RETRY_POLICY}.
78
+ */
79
+
80
+ /**
81
+ * Defines an LLM agent as a Restate building block. Restate itself has no agent
82
+ * primitive: this generates a regular service (`memory: 'none'`) or virtual object
83
+ * (`memory: 'session'`) with an `ask` handler that runs {@link runAgentLoop}, and enforces
84
+ * what a hand-written agent service tends to get wrong:
85
+ *
86
+ * - every LLM call and tool call is its own bounded, durable step;
87
+ * - tools are Restate calls to other building blocks, or bounded `ctx.run` side effects
88
+ * that get an idempotency key;
89
+ * - tools can require approval through an awakeable, with a timeout;
90
+ * - input and output are validated, and failures are terminal;
91
+ * - step and token budgets, and an explicit invocation retry policy;
92
+ * - the journal holds only small values: the skill index, never skill bodies; a session
93
+ * store's reference, never the conversation.
94
+ *
95
+ * Session agents also get a `reset` handler that forgets the conversation.
96
+ * @param {AgentSpec} spec
97
+ * @returns {import('@restatedev/restate-sdk').ServiceDefinition<string, unknown> | import('@restatedev/restate-sdk').VirtualObjectDefinition<string, unknown>}
98
+ * @throws {TypeError} If the spec is invalid.
99
+ */
100
+ export function defineAgent(spec) {
101
+ const {
102
+ name,
103
+ env,
104
+ model = env ? createModel(env) : undefined,
105
+ system = '',
106
+ loadContext,
107
+ loadSkill,
108
+ tools = {},
109
+ input = AskInputSchema,
110
+ prompt = (validatedInput) => validatedInput.prompt,
111
+ output,
112
+ memory = 'none',
113
+ historyStore,
114
+ limits = {},
115
+ llmRetry,
116
+ retryPolicy = DEFAULT_AGENT_RETRY_POLICY,
117
+ } = spec
118
+
119
+ if (typeof name !== 'string' || name === '') {
120
+ throw new TypeError('defineAgent needs a name')
121
+ }
122
+ if (!model) throw new TypeError(`Agent '${name}' needs a model or an env to build one from`)
123
+ if (memory !== 'none' && memory !== 'session') {
124
+ throw new TypeError(`Agent '${name}': memory must be 'none' or 'session'`)
125
+ }
126
+ if (historyStore && memory !== 'session') {
127
+ throw new TypeError(`Agent '${name}': historyStore needs memory: 'session'`)
128
+ }
129
+ assertToolSpecs(tools)
130
+
131
+ const { maxSteps, maxTokens, maxHistoryTurns = 20 } = limits
132
+
133
+ const ask = withValidation(input, output ?? AskOutputSchema, async (ctx, validatedInput) => {
134
+ const userPrompt = prompt(validatedInput)
135
+ if (typeof userPrompt !== 'string' || userPrompt === '') {
136
+ throw new TerminalError(`Agent '${name}': the input produced no prompt`, {
137
+ errorCode: 400,
138
+ })
139
+ }
140
+
141
+ const context = loadContext
142
+ ? await ctx.run('load-context', async () => {
143
+ const { systemPrompt = '', skills = [] } =
144
+ (await loadContext(validatedInput)) ?? {}
145
+ return {
146
+ systemPrompt,
147
+ skills: skills.map((skill) => ({
148
+ name: skill.name,
149
+ description: skill.description,
150
+ })),
151
+ }
152
+ })
153
+ : { systemPrompt: '', skills: [] }
154
+
155
+ const session = memory === 'session' ? await openSession(ctx, historyStore) : undefined
156
+
157
+ const result = await runAgentLoop(ctx, {
158
+ model,
159
+ system: [system, context.systemPrompt].filter(Boolean).join('\n\n'),
160
+ prompt: userPrompt,
161
+ tools,
162
+ skills: context.skills,
163
+ loadSkill: skillLoader({ loadSkill, loadContext, input: validatedInput }),
164
+ loadHistory: session?.load,
165
+ outputSchema: output,
166
+ maxSteps,
167
+ maxTokens,
168
+ retry: llmRetry,
169
+ })
170
+
171
+ if (session) await session.save(result.messages, maxHistoryTurns)
172
+
173
+ return output ? result.output : { text: result.text, steps: result.steps }
174
+ })
175
+
176
+ const options = { retryPolicy }
177
+
178
+ if (memory === 'none') {
179
+ return defineService({ name, handlers: { ask }, options })
180
+ }
181
+
182
+ return defineObject({
183
+ name,
184
+ handlers: {
185
+ ask,
186
+ reset: async (ctx) => {
187
+ ctx.clear(HISTORY)
188
+ ctx.clear(HISTORY_REF)
189
+ },
190
+ },
191
+ options,
192
+ })
193
+ }
194
+
195
+ /**
196
+ * The session's conversation: `load` returns the previous turns' messages (memoised for
197
+ * this execution; called inside each LLM step, so never journaled), `save` appends this
198
+ * turn's messages.
199
+ * @param {import('@restatedev/restate-sdk').ObjectContext} ctx
200
+ * @param {HistoryStore} [store]
201
+ */
202
+ async function openSession(ctx, store) {
203
+ if (store) {
204
+ const ref = await ctx.get(HISTORY_REF)
205
+ let cached
206
+ const load = async () => (cached ??= ref ? await store.load(ref) : [])
207
+ return {
208
+ load,
209
+ save: async (messages) => {
210
+ const newRef = await ctx.run('save-history', async () =>
211
+ store.save(ctx.key, [...(await load()), ...messages]),
212
+ )
213
+ ctx.set(HISTORY_REF, newRef)
214
+ },
215
+ }
216
+ }
217
+
218
+ // Without a store, history lives in object state as a list of turns, so trimming never
219
+ // separates a tool call from its result.
220
+ const turns = (await ctx.get(HISTORY)) ?? []
221
+ return {
222
+ load: async () => turns.flat(),
223
+ save: async (messages, maxHistoryTurns) => {
224
+ ctx.set(HISTORY, [...turns, messages].slice(-maxHistoryTurns))
225
+ },
226
+ }
227
+ }
228
+
229
+ function skillLoader({ loadSkill, loadContext, input }) {
230
+ if (loadSkill) return (skillName) => loadSkill(skillName, input)
231
+ if (!loadContext) return undefined
232
+ return async (skillName) => {
233
+ const { skills = [] } = (await loadContext(input)) ?? {}
234
+ return skills.find((skill) => skill.name === skillName)?.body
235
+ }
236
+ }
237
+
238
+ /**
239
+ * @deprecated Use {@link defineAgent} with an explicit `name`. Kept so callers migrating
240
+ * from `@empyria/restate-llm` only change their import: defines a stateless agent named
241
+ * `AgentService` with the default `ask` input/output.
242
+ * @param {Omit<AgentSpec, 'name'>} spec
243
+ */
244
+ export function createAgentService(spec) {
245
+ return defineAgent({ name: 'AgentService', ...spec })
246
+ }
@@ -0,0 +1,86 @@
1
+ import { TerminalError, RetryableError } from '@restatedev/restate-sdk'
2
+ import { APICallError } from '@ai-sdk/provider'
3
+
4
+ /**
5
+ * Default bounded retry policy for an LLM call's durable `ctx.run` step.
6
+ *
7
+ * `ctx.run`'s own SDK-wide defaults (`initialRetryInterval: 50ms`, unbounded attempts)
8
+ * are tuned for cheap, fast-failing side effects, not a rate-limited LLM gateway — a
9
+ * 50ms-then-double backoff burns through a 429's typical multi-second cooldown in a
10
+ * handful of attempts, and with no `maxRetryAttempts` it never gives up on a
11
+ * genuinely-broken endpoint. `maxRetryAttempts` also matters independently of the
12
+ * interval tuning: without it, any error this module fails to classify as
13
+ * {@link TerminalError} (see {@link toRestateLLMError}) retries forever.
14
+ */
15
+ export const DEFAULT_LLM_RETRY = {
16
+ maxRetryAttempts: 5,
17
+ initialRetryInterval: 1_000,
18
+ maxRetryInterval: 30_000,
19
+ retryIntervalFactor: 2,
20
+ }
21
+
22
+ /**
23
+ * Default bounded retry policy for an `execute` tool's durable `ctx.run` step. Once it's
24
+ * exhausted, `ctx.run` throws a `TerminalError`, which the agent loop hands back to the
25
+ * model as a tool error instead of failing the whole invocation — see `Tools.js`.
26
+ */
27
+ export const DEFAULT_TOOL_RETRY = {
28
+ maxRetryAttempts: 3,
29
+ initialRetryInterval: 500,
30
+ maxRetryInterval: 10_000,
31
+ retryIntervalFactor: 2,
32
+ }
33
+
34
+ /**
35
+ * Default invocation-level retry policy for an agent service/object. Applies to failures
36
+ * outside any bounded `ctx.run` (e.g. a bug in the handler itself): after `maxAttempts`
37
+ * the invocation is paused for an operator to fix and resume, never retried forever and
38
+ * never silently killed.
39
+ */
40
+ export const DEFAULT_AGENT_RETRY_POLICY = {
41
+ maxAttempts: 10,
42
+ onMaxAttempts: 'pause',
43
+ }
44
+
45
+ /**
46
+ * Maps an error thrown by an AI SDK Core call (`generateText`/`streamText`/...) to the
47
+ * error a `ctx.run(name, fn, RunOptions)` closure should throw, so Restate retries
48
+ * exactly the failures worth retrying and stops immediately on the ones that aren't.
49
+ *
50
+ * `@ai-sdk/provider`'s `APICallError` already carries an `isRetryable` flag the AI SDK
51
+ * itself derives from the HTTP status code (true for 408/409/429/5xx, false for
52
+ * 400/401/403/404/422/...) — this reuses that classification instead of re-deriving it
53
+ * from status codes by hand, and layers Restate's two escape hatches on top of it:
54
+ *
55
+ * - Not retryable → {@link TerminalError}, so Restate fails the invocation immediately
56
+ * instead of retrying an auth/config/bad-request error forever.
57
+ * - Retryable AND the response carried a `Retry-After` header → {@link RetryableError},
58
+ * so Restate honors the gateway's own requested cooldown instead of guessing one.
59
+ * - Retryable with no such header → the original error, unchanged, so `ctx.run`'s own
60
+ * `RunOptions` backoff (see {@link DEFAULT_LLM_RETRY}) applies.
61
+ * - Anything that isn't an `APICallError` at all (a thrown `TypeError`, a network-layer
62
+ * error `fetch` itself throws, ...) is returned unchanged — Restate's default is to
63
+ * treat any non-`TerminalError` throw as retryable, which is the right default for an
64
+ * error shape this module doesn't recognize.
65
+ *
66
+ * Call sites re-throw the result — this function never throws itself, only classifies:
67
+ * `catch (error) { throw toRestateLLMError(error) }`.
68
+ * @param {unknown} error
69
+ * @returns {Error}
70
+ */
71
+ export function toRestateLLMError(error) {
72
+ if (!APICallError.isInstance(error)) return error
73
+
74
+ if (!error.isRetryable) {
75
+ return new TerminalError(error.message, { errorCode: error.statusCode })
76
+ }
77
+
78
+ const retryAfterHeader = error.responseHeaders?.['retry-after']
79
+ const retryAfterSeconds = retryAfterHeader ? Number(retryAfterHeader) : undefined
80
+
81
+ if (retryAfterSeconds !== undefined && Number.isFinite(retryAfterSeconds)) {
82
+ return RetryableError.from(error, { retryAfter: { seconds: retryAfterSeconds } })
83
+ }
84
+
85
+ return error
86
+ }
@@ -0,0 +1,157 @@
1
+ import { generateText } from 'ai'
2
+ import { TerminalError } from '@restatedev/restate-sdk'
3
+ import { DEFAULT_LLM_RETRY, toRestateLLMError } from './Errors.js'
4
+ import { cachedSkillFetcher, expandSkillRefs } from './Skills.js'
5
+ import { FINAL_ANSWER_TOOL, buildToolDefs, executeTool } from './Tools.js'
6
+
7
+ const FINAL_ANSWER_NUDGE = `Respond by calling the ${FINAL_ANSWER_TOOL} tool with your answer.`
8
+
9
+ /**
10
+ * Drives an AI SDK Core tool-calling loop ONE round at a time, from inside a Restate
11
+ * handler — deliberately NOT `generateText`'s own multi-step loop wrapped in a single
12
+ * `ctx.run()`. That would make the whole loop atomic to Restate: a crash mid-loop would
13
+ * replay it from scratch, re-billing finished LLM calls and re-executing tools that
14
+ * already ran. Here every LLM call and every tool call is its own durable step, so a
15
+ * crash resumes exactly where it left off, and each step shows up by name in Restate's
16
+ * invocation UI.
17
+ *
18
+ * Journal discipline: an LLM step records only the model's response (text, tool calls,
19
+ * token usage), never the prompt it was sent. Prior conversation comes from
20
+ * `loadHistory`, which is called inside each LLM step — so it must return the same
21
+ * messages every time it's called during one invocation (e.g. load an immutable
22
+ * snapshot by reference). Loaded skills work the same way: the journal holds a
23
+ * reference and hash, and the body is fetched again inside each LLM step (see Skills.js).
24
+ * @param {import('@restatedev/restate-sdk').Context} ctx
25
+ * @param {{
26
+ * model: import('@ai-sdk/provider').LanguageModelV4,
27
+ * system: string,
28
+ * prompt: string,
29
+ * tools?: Record<string, import('./Tools.js').ToolSpec>,
30
+ * skills?: Array<{name: string, description: string}>,
31
+ * loadSkill?: (name: string) => Promise<string|undefined>,
32
+ * loadHistory?: () => Promise<Array<object>>,
33
+ * outputSchema?: object,
34
+ * maxSteps?: number,
35
+ * maxTokens?: number,
36
+ * retry?: import('@restatedev/restate-sdk').RunOptions<any>,
37
+ * }} params `skills` is the index only (name + description); `loadSkill` fetches a body
38
+ * when the model asks for it. With `outputSchema`, the model must answer through the
39
+ * `final_answer` tool and the loop returns that tool's input as `output`.
40
+ * @returns {Promise<{text: string, output?: unknown, steps: number, totalTokens: number, messages: Array<object>}>}
41
+ * `messages` are the new messages of this turn only (prompt, responses, tool results).
42
+ */
43
+ export async function runAgentLoop(
44
+ ctx,
45
+ {
46
+ model,
47
+ system,
48
+ prompt,
49
+ tools = {},
50
+ skills = [],
51
+ loadSkill,
52
+ loadHistory = async () => [],
53
+ outputSchema,
54
+ maxSteps = 8,
55
+ maxTokens,
56
+ retry = DEFAULT_LLM_RETRY,
57
+ },
58
+ ) {
59
+ const fullSystem = buildSystemPrompt(system, skills)
60
+ const toolDefs = buildToolDefs({ tools, hasSkills: skills.length > 0, outputSchema })
61
+ const messages = [{ role: 'user', content: prompt }]
62
+ let totalTokens = 0
63
+ const fetchSkill = cachedSkillFetcher(loadSkill)
64
+
65
+ for (let step = 0; step < maxSteps; step++) {
66
+ // `ctx.run` JSON-serialises the callback's result for the journal and hands back that
67
+ // copy. AI SDK's `response` is a non-enumerable getter `JSON.stringify` drops, so the
68
+ // plain fields the loop needs are extracted inside the callback.
69
+ const result = await ctx.run(
70
+ `llm-step-${step}`,
71
+ async () => {
72
+ try {
73
+ // `maxRetries: 0`: `ctx.run`'s RunOptions are the single retry authority,
74
+ // instead of AI SDK's own retries stacking invisibly underneath them.
75
+ const r = await generateText({
76
+ model,
77
+ system: fullSystem,
78
+ messages: [
79
+ ...(await expandSkillRefs(await loadHistory(), {
80
+ fetchSkill,
81
+ strict: false,
82
+ })),
83
+ ...(await expandSkillRefs(messages, { fetchSkill, strict: true })),
84
+ ],
85
+ tools: toolDefs,
86
+ maxRetries: 0,
87
+ })
88
+ return {
89
+ text: r.text,
90
+ toolCalls: r.toolCalls.map(({ toolCallId, toolName, input }) => ({
91
+ toolCallId,
92
+ toolName,
93
+ input,
94
+ })),
95
+ responseMessages: r.response.messages,
96
+ totalTokens: r.usage?.totalTokens ?? 0,
97
+ }
98
+ } catch (error) {
99
+ throw toRestateLLMError(error)
100
+ }
101
+ },
102
+ retry,
103
+ )
104
+
105
+ messages.push(...result.responseMessages)
106
+ totalTokens += result.totalTokens
107
+ if (maxTokens !== undefined && totalTokens > maxTokens) {
108
+ throw new TerminalError(
109
+ `Agent exceeded its token budget (${totalTokens} > ${maxTokens})`,
110
+ { errorCode: 429 },
111
+ )
112
+ }
113
+
114
+ if (result.toolCalls.length === 0) {
115
+ if (!outputSchema) {
116
+ return { text: result.text, steps: step + 1, totalTokens, messages }
117
+ }
118
+ messages.push({ role: 'user', content: FINAL_ANSWER_NUDGE })
119
+ continue
120
+ }
121
+
122
+ for (const [index, call] of result.toolCalls.entries()) {
123
+ if (outputSchema && call.toolName === FINAL_ANSWER_TOOL) {
124
+ return {
125
+ text: result.text,
126
+ output: call.input,
127
+ steps: step + 1,
128
+ totalTokens,
129
+ messages,
130
+ }
131
+ }
132
+ const output = await executeTool(ctx, call, { tools, loadSkill, step, index })
133
+ messages.push({
134
+ role: 'tool',
135
+ content: [
136
+ {
137
+ type: 'tool-result',
138
+ toolCallId: call.toolCallId,
139
+ toolName: call.toolName,
140
+ output,
141
+ },
142
+ ],
143
+ })
144
+ }
145
+ }
146
+
147
+ // Exceeding maxSteps isn't a transient fault a retry could fix.
148
+ throw new TerminalError(`Agent loop exceeded maxSteps (${maxSteps}) without a final answer`)
149
+ }
150
+
151
+ function buildSystemPrompt(system, skills) {
152
+ if (skills.length === 0) return system
153
+
154
+ const index = skills.map((skill) => `- ${skill.name}: ${skill.description}`).join('\n')
155
+ const instructions = `Available skills — call load_skill with a skill's name to read its full instructions before following them:\n${index}`
156
+ return [system, instructions].filter(Boolean).join('\n\n')
157
+ }
@@ -0,0 +1,19 @@
1
+ import { createOpenAICompatible } from '@ai-sdk/openai-compatible'
2
+
3
+ /**
4
+ * AI SDK Core model, pointed at an OpenAI-compatible endpoint — Bifrost
5
+ * (llm-system's gateway) by default, fully overridable via env. `LLM_BASE_URL`
6
+ * already includes the `/v1` suffix Bifrost's own docs use (`https://agent-bureau.vip/v1`);
7
+ * `MODEL_ID` defaults to `vllm/ornith-1.5`, the currently-live routed model ID on that
8
+ * gateway (see llm-system's README, "Verifying").
9
+ * @param {{LLM_BASE_URL: string, LLM_API_KEY?: string, MODEL_ID: string}} env
10
+ * @returns {import('@ai-sdk/provider').LanguageModelV4}
11
+ */
12
+ export function createModel(env) {
13
+ const provider = createOpenAICompatible({
14
+ name: 'bifrost',
15
+ baseURL: env.LLM_BASE_URL,
16
+ apiKey: env.LLM_API_KEY,
17
+ })
18
+ return provider(env.MODEL_ID)
19
+ }