@empyria/restate 0.1.23 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +12 -0
- package/README.md +60 -0
- package/agent.js +11 -0
- package/lib/agent/Agent.js +246 -0
- package/lib/agent/Errors.js +86 -0
- package/lib/agent/Loop.js +157 -0
- package/lib/agent/Model.js +19 -0
- package/lib/agent/Skills.js +81 -0
- package/lib/agent/Tools.js +215 -0
- package/package.json +21 -1
- package/test/agent/Agent.test.js +195 -0
- package/test/agent/Errors.test.js +66 -0
- package/test/agent/Loop.test.js +173 -0
- package/test/agent/Skills.test.js +147 -0
- package/test/agent/Tools.test.js +277 -0
- package/test/agent/fakes.js +84 -0
package/AGENTS.md
CHANGED
|
@@ -24,6 +24,18 @@ is not the right default here.
|
|
|
24
24
|
- Relative imports must include explicit `.js` extensions — Bun tolerates missing ones, Node's
|
|
25
25
|
ESM resolver doesn't.
|
|
26
26
|
|
|
27
|
+
## Agent entry point (`lib/agent/`)
|
|
28
|
+
|
|
29
|
+
`lib/agent/` (merged in from the former `@empyria/restate-llm`) is exported only through
|
|
30
|
+
`@empyria/restate/agent` ([agent.js](./agent.js)), never from [index.js](./index.js). The AI
|
|
31
|
+
SDK packages (`ai`, `@ai-sdk/*`) are **optional** peer dependencies (and devDependencies for
|
|
32
|
+
the tests): a consumer that only uses Admin/Cron/Pubsub must be able to import the package
|
|
33
|
+
root without them installed. Never import `lib/agent/` from a root-exported module.
|
|
34
|
+
|
|
35
|
+
Agent tests use `test/agent/fakes.js`'s `fakeCtx`, which mimics the journal's JSON round trip
|
|
36
|
+
and a bounded `ctx.run` turning an exhausted retry into a `TerminalError` — keep it in step
|
|
37
|
+
with the real SDK behaviour when touching either.
|
|
38
|
+
|
|
27
39
|
## No official Admin API client
|
|
28
40
|
|
|
29
41
|
Verified directly against the published `@restatedev/restate-sdk`/`-clients` package
|
package/README.md
CHANGED
|
@@ -51,6 +51,66 @@ import { checkServiceHandler } from '@empyria/restate/lib/Admin.js'
|
|
|
51
51
|
| [lib/Pubsub.js](./lib/Pubsub.js) | `definePubsub` (register a pubsub Virtual Object on your endpoint), `pubsubPublisher` (publish into it from inside a handler, in-process), `pubsubClient` (publish/pull/subscribe to it from anywhere else, over the network) — thin wrappers around `@restatedev/pubsub`/`@restatedev/pubsub-client`. |
|
|
52
52
|
| [lib/Validation.js](./lib/Validation.js) | `withValidation` — wraps a workflow handler with input/output JSON schema validation via `@empyria/common`. |
|
|
53
53
|
|
|
54
|
+
## Agents: `@empyria/restate/agent`
|
|
55
|
+
|
|
56
|
+
LLM agents as Restate building blocks (formerly `@empyria/restate-llm`). A separate entry
|
|
57
|
+
point, so the root import never loads the AI SDK. It needs the optional peer dependencies:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
bun add ai @ai-sdk/openai-compatible @ai-sdk/provider
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
```js
|
|
64
|
+
import { defineAgent } from '@empyria/restate/agent'
|
|
65
|
+
|
|
66
|
+
export const triage = defineAgent({
|
|
67
|
+
name: 'CorporateActionTriage',
|
|
68
|
+
env: { LLM_BASE_URL: 'https://agent-bureau.vip/v1', MODEL_ID: 'vllm/ornith-1.5' },
|
|
69
|
+
system: 'You triage corporate action notices.',
|
|
70
|
+
tools: {
|
|
71
|
+
// a Restate call to another building block
|
|
72
|
+
lookupIssuer: {
|
|
73
|
+
description: 'Look up an issuer by ISIN',
|
|
74
|
+
inputSchema: {
|
|
75
|
+
type: 'object',
|
|
76
|
+
properties: { isin: { type: 'string' } },
|
|
77
|
+
required: ['isin'],
|
|
78
|
+
},
|
|
79
|
+
block: 'IssuerService',
|
|
80
|
+
handler: 'get',
|
|
81
|
+
},
|
|
82
|
+
// a direct side effect in its own bounded ctx.run, gated by approval
|
|
83
|
+
notifyDesk: {
|
|
84
|
+
description: 'Notify the operations desk',
|
|
85
|
+
inputSchema: { type: 'object', properties: { message: { type: 'string' } } },
|
|
86
|
+
execute: async ({ message }, { idempotencyKey }) => sendToDesk(message, idempotencyKey),
|
|
87
|
+
approval: {
|
|
88
|
+
timeout: 4 * 3600_000,
|
|
89
|
+
notify: { block: 'ApprovalInbox', handler: 'request' },
|
|
90
|
+
},
|
|
91
|
+
},
|
|
92
|
+
},
|
|
93
|
+
output: TriageResultSchema, // the model answers through a final_answer tool
|
|
94
|
+
memory: 'none', // or 'session': a virtual object keyed by session, plus a reset handler
|
|
95
|
+
limits: { maxSteps: 8, maxTokens: 50_000 },
|
|
96
|
+
})
|
|
97
|
+
// register `triage` on your endpoint, then: POST /CorporateActionTriage/ask {"prompt": "…"}
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
| Module | Purpose |
|
|
101
|
+
| -------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
|
|
102
|
+
| [lib/agent/Agent.js](./lib/agent/Agent.js) | `defineAgent` — builds the service/object and its `ask` handler: validated I/O, `loadContext` that journals only the skill index, session memory (object state or an external `historyStore`), step/token budgets, invocation retry policy. `createAgentService` is the deprecated `AgentService` shortcut. |
|
|
103
|
+
| [lib/agent/Loop.js](./lib/agent/Loop.js) | `runAgentLoop` — one round at a time; every LLM call and tool call is its own durable step. Journals model responses, never prompts or prior history. |
|
|
104
|
+
| [lib/agent/Tools.js](./lib/agent/Tools.js) | Tool specs and execution: block tools (`ctx.genericCall`), execute tools (bounded `ctx.run` with an idempotency key), approval through an awakeable with a timeout, `load_skill`, `final_answer`. |
|
|
105
|
+
| [lib/agent/Skills.js](./lib/agent/Skills.js) | Skill bodies stay out of the journal: `load_skill` records `{name, sha256}`, and `expandSkillRefs` puts the body back inside each LLM step. A hash mismatch within one invocation is terminal; across session turns the current body is used. |
|
|
106
|
+
| [lib/agent/Errors.js](./lib/agent/Errors.js) | `toRestateLLMError` (AI SDK error → terminal / `RetryableError` / transient) and the default retry policies for LLM steps, tool steps and the agent invocation. |
|
|
107
|
+
| [lib/agent/Model.js](./lib/agent/Model.js) | `createModel` — AI SDK model for an OpenAI-compatible endpoint (e.g. Bifrost). |
|
|
108
|
+
|
|
109
|
+
Error policy: a tool that fails terminally (rejected or timed-out approval, a callee's
|
|
110
|
+
`TerminalError`, an execute tool out of retries) goes back to the model as an error result,
|
|
111
|
+
so the agent can adapt. Transient errors are left to Restate to retry, and cancellation
|
|
112
|
+
always propagates.
|
|
113
|
+
|
|
54
114
|
There is no official Restate Admin API client (verified directly against the published
|
|
55
115
|
`@restatedev/restate-sdk`/`-clients` packages) — `lib/Admin.js` is a hand-written wrapper kept
|
|
56
116
|
in sync with the live Admin API's OpenAPI spec (`<admin-url>/openapi`) by hand.
|
package/agent.js
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
import { TerminalError } from '@restatedev/restate-sdk'
|
|
2
|
+
import { defineObject, defineService } from '../Admin.js'
|
|
3
|
+
import { withValidation } from '../Validation.js'
|
|
4
|
+
import { DEFAULT_AGENT_RETRY_POLICY } from './Errors.js'
|
|
5
|
+
import { runAgentLoop } from './Loop.js'
|
|
6
|
+
import { createModel } from './Model.js'
|
|
7
|
+
import { assertToolSpecs } from './Tools.js'
|
|
8
|
+
|
|
9
|
+
const HISTORY = 'history'
|
|
10
|
+
const HISTORY_REF = 'historyRef'
|
|
11
|
+
|
|
12
|
+
export const AskInputSchema = {
|
|
13
|
+
type: 'object',
|
|
14
|
+
properties: {
|
|
15
|
+
prompt: { type: 'string', minLength: 1 },
|
|
16
|
+
skillsFolder: { type: 'string' },
|
|
17
|
+
department: { type: 'string' },
|
|
18
|
+
threadId: { type: 'string' },
|
|
19
|
+
resourceId: { type: 'string' },
|
|
20
|
+
},
|
|
21
|
+
required: ['prompt'],
|
|
22
|
+
additionalProperties: false,
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export const AskOutputSchema = {
|
|
26
|
+
type: 'object',
|
|
27
|
+
properties: {
|
|
28
|
+
text: { type: 'string' },
|
|
29
|
+
steps: { type: 'number' },
|
|
30
|
+
},
|
|
31
|
+
required: ['text', 'steps'],
|
|
32
|
+
additionalProperties: false,
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @typedef {Object} HistoryStore External storage for a session agent's conversation, so
|
|
37
|
+
* the journal and object state only ever hold a reference to it.
|
|
38
|
+
* @property {(ref: string) => Promise<Array<object>>} load Returns the messages saved under
|
|
39
|
+
* `ref`. Must return the same messages for the same `ref` every time: refs are immutable
|
|
40
|
+
* snapshots, e.g. a content hash or a versioned object key.
|
|
41
|
+
* @property {(sessionKey: string, messages: Array<object>) => Promise<string>} save Stores
|
|
42
|
+
* the full conversation as a new snapshot and returns its ref.
|
|
43
|
+
*
|
|
44
|
+
* @typedef {Object} AgentSpec
|
|
45
|
+
* @property {string} name Restate service / object name. Required: each agent is its own
|
|
46
|
+
* building block.
|
|
47
|
+
* @property {{LLM_BASE_URL: string, LLM_API_KEY?: string, MODEL_ID: string}} [env] Builds the
|
|
48
|
+
* model via {@link createModel}, unless `model` is given.
|
|
49
|
+
* @property {import('@ai-sdk/provider').LanguageModelV4} [model]
|
|
50
|
+
* @property {string} [system] Static system prompt, prepended to `loadContext`'s.
|
|
51
|
+
* @property {(input: any) => Promise<{systemPrompt?: string, skills?: Array<{name: string, description: string, body?: string}>}>} [loadContext]
|
|
52
|
+
* Resolves this request's system prompt and skills. Runs in its own `ctx.run`, which
|
|
53
|
+
* records only the system prompt and the skill index — never the skill bodies.
|
|
54
|
+
* @property {(name: string, input: any) => Promise<string|undefined>} [loadSkill] Fetches one
|
|
55
|
+
* skill body when the model asks for it. Defaults to re-running `loadContext` and
|
|
56
|
+
* picking the body by name. The journal records only the skill's name and hash; the
|
|
57
|
+
* body is fetched again before every model call. So it must return the same content
|
|
58
|
+
* for the whole invocation (skills shipped with the deployment do), or the invocation
|
|
59
|
+
* fails terminally rather than send the model changed instructions.
|
|
60
|
+
* @property {Record<string, import('./Tools.js').ToolSpec>} [tools]
|
|
61
|
+
* @property {object} [input] Input JSON Schema. Defaults to {@link AskInputSchema}.
|
|
62
|
+
* @property {(input: any) => string} [prompt] Builds the user prompt from the validated
|
|
63
|
+
* input. Defaults to `input.prompt`.
|
|
64
|
+
* @property {object} [output] Output JSON Schema. When given, the model must answer through
|
|
65
|
+
* the `final_answer` tool and `ask` returns its arguments. Defaults to
|
|
66
|
+
* {@link AskOutputSchema} (`{text, steps}`).
|
|
67
|
+
* @property {'none'|'session'} [memory] `none` (default): a stateless service. `session`: a
|
|
68
|
+
* virtual object keyed by session ID, so turns of one session run one at a time and see
|
|
69
|
+
* the conversation so far.
|
|
70
|
+
* @property {HistoryStore} [historyStore] Session memory only. Without it, the conversation
|
|
71
|
+
* is kept in object state, trimmed to `limits.maxHistoryTurns` — fine for short
|
|
72
|
+
* sessions; long ones should use a store.
|
|
73
|
+
* @property {{maxSteps?: number, maxTokens?: number, maxHistoryTurns?: number}} [limits]
|
|
74
|
+
* @property {import('@restatedev/restate-sdk').RunOptions<any>} [llmRetry] Retry policy of
|
|
75
|
+
* each LLM call step. Defaults to `DEFAULT_LLM_RETRY`.
|
|
76
|
+
* @property {import('@restatedev/restate-sdk').RetryPolicy} [retryPolicy] Invocation-level
|
|
77
|
+
* retry policy. Defaults to {@link DEFAULT_AGENT_RETRY_POLICY}.
|
|
78
|
+
*/
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Defines an LLM agent as a Restate building block. Restate itself has no agent
|
|
82
|
+
* primitive: this generates a regular service (`memory: 'none'`) or virtual object
|
|
83
|
+
* (`memory: 'session'`) with an `ask` handler that runs {@link runAgentLoop}, and enforces
|
|
84
|
+
* what a hand-written agent service tends to get wrong:
|
|
85
|
+
*
|
|
86
|
+
* - every LLM call and tool call is its own bounded, durable step;
|
|
87
|
+
* - tools are Restate calls to other building blocks, or bounded `ctx.run` side effects
|
|
88
|
+
* that get an idempotency key;
|
|
89
|
+
* - tools can require approval through an awakeable, with a timeout;
|
|
90
|
+
* - input and output are validated, and failures are terminal;
|
|
91
|
+
* - step and token budgets, and an explicit invocation retry policy;
|
|
92
|
+
* - the journal holds only small values: the skill index, never skill bodies; a session
|
|
93
|
+
* store's reference, never the conversation.
|
|
94
|
+
*
|
|
95
|
+
* Session agents also get a `reset` handler that forgets the conversation.
|
|
96
|
+
* @param {AgentSpec} spec
|
|
97
|
+
* @returns {import('@restatedev/restate-sdk').ServiceDefinition<string, unknown> | import('@restatedev/restate-sdk').VirtualObjectDefinition<string, unknown>}
|
|
98
|
+
* @throws {TypeError} If the spec is invalid.
|
|
99
|
+
*/
|
|
100
|
+
export function defineAgent(spec) {
|
|
101
|
+
const {
|
|
102
|
+
name,
|
|
103
|
+
env,
|
|
104
|
+
model = env ? createModel(env) : undefined,
|
|
105
|
+
system = '',
|
|
106
|
+
loadContext,
|
|
107
|
+
loadSkill,
|
|
108
|
+
tools = {},
|
|
109
|
+
input = AskInputSchema,
|
|
110
|
+
prompt = (validatedInput) => validatedInput.prompt,
|
|
111
|
+
output,
|
|
112
|
+
memory = 'none',
|
|
113
|
+
historyStore,
|
|
114
|
+
limits = {},
|
|
115
|
+
llmRetry,
|
|
116
|
+
retryPolicy = DEFAULT_AGENT_RETRY_POLICY,
|
|
117
|
+
} = spec
|
|
118
|
+
|
|
119
|
+
if (typeof name !== 'string' || name === '') {
|
|
120
|
+
throw new TypeError('defineAgent needs a name')
|
|
121
|
+
}
|
|
122
|
+
if (!model) throw new TypeError(`Agent '${name}' needs a model or an env to build one from`)
|
|
123
|
+
if (memory !== 'none' && memory !== 'session') {
|
|
124
|
+
throw new TypeError(`Agent '${name}': memory must be 'none' or 'session'`)
|
|
125
|
+
}
|
|
126
|
+
if (historyStore && memory !== 'session') {
|
|
127
|
+
throw new TypeError(`Agent '${name}': historyStore needs memory: 'session'`)
|
|
128
|
+
}
|
|
129
|
+
assertToolSpecs(tools)
|
|
130
|
+
|
|
131
|
+
const { maxSteps, maxTokens, maxHistoryTurns = 20 } = limits
|
|
132
|
+
|
|
133
|
+
const ask = withValidation(input, output ?? AskOutputSchema, async (ctx, validatedInput) => {
|
|
134
|
+
const userPrompt = prompt(validatedInput)
|
|
135
|
+
if (typeof userPrompt !== 'string' || userPrompt === '') {
|
|
136
|
+
throw new TerminalError(`Agent '${name}': the input produced no prompt`, {
|
|
137
|
+
errorCode: 400,
|
|
138
|
+
})
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const context = loadContext
|
|
142
|
+
? await ctx.run('load-context', async () => {
|
|
143
|
+
const { systemPrompt = '', skills = [] } =
|
|
144
|
+
(await loadContext(validatedInput)) ?? {}
|
|
145
|
+
return {
|
|
146
|
+
systemPrompt,
|
|
147
|
+
skills: skills.map((skill) => ({
|
|
148
|
+
name: skill.name,
|
|
149
|
+
description: skill.description,
|
|
150
|
+
})),
|
|
151
|
+
}
|
|
152
|
+
})
|
|
153
|
+
: { systemPrompt: '', skills: [] }
|
|
154
|
+
|
|
155
|
+
const session = memory === 'session' ? await openSession(ctx, historyStore) : undefined
|
|
156
|
+
|
|
157
|
+
const result = await runAgentLoop(ctx, {
|
|
158
|
+
model,
|
|
159
|
+
system: [system, context.systemPrompt].filter(Boolean).join('\n\n'),
|
|
160
|
+
prompt: userPrompt,
|
|
161
|
+
tools,
|
|
162
|
+
skills: context.skills,
|
|
163
|
+
loadSkill: skillLoader({ loadSkill, loadContext, input: validatedInput }),
|
|
164
|
+
loadHistory: session?.load,
|
|
165
|
+
outputSchema: output,
|
|
166
|
+
maxSteps,
|
|
167
|
+
maxTokens,
|
|
168
|
+
retry: llmRetry,
|
|
169
|
+
})
|
|
170
|
+
|
|
171
|
+
if (session) await session.save(result.messages, maxHistoryTurns)
|
|
172
|
+
|
|
173
|
+
return output ? result.output : { text: result.text, steps: result.steps }
|
|
174
|
+
})
|
|
175
|
+
|
|
176
|
+
const options = { retryPolicy }
|
|
177
|
+
|
|
178
|
+
if (memory === 'none') {
|
|
179
|
+
return defineService({ name, handlers: { ask }, options })
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
return defineObject({
|
|
183
|
+
name,
|
|
184
|
+
handlers: {
|
|
185
|
+
ask,
|
|
186
|
+
reset: async (ctx) => {
|
|
187
|
+
ctx.clear(HISTORY)
|
|
188
|
+
ctx.clear(HISTORY_REF)
|
|
189
|
+
},
|
|
190
|
+
},
|
|
191
|
+
options,
|
|
192
|
+
})
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* The session's conversation: `load` returns the previous turns' messages (memoised for
|
|
197
|
+
* this execution; called inside each LLM step, so never journaled), `save` appends this
|
|
198
|
+
* turn's messages.
|
|
199
|
+
* @param {import('@restatedev/restate-sdk').ObjectContext} ctx
|
|
200
|
+
* @param {HistoryStore} [store]
|
|
201
|
+
*/
|
|
202
|
+
async function openSession(ctx, store) {
|
|
203
|
+
if (store) {
|
|
204
|
+
const ref = await ctx.get(HISTORY_REF)
|
|
205
|
+
let cached
|
|
206
|
+
const load = async () => (cached ??= ref ? await store.load(ref) : [])
|
|
207
|
+
return {
|
|
208
|
+
load,
|
|
209
|
+
save: async (messages) => {
|
|
210
|
+
const newRef = await ctx.run('save-history', async () =>
|
|
211
|
+
store.save(ctx.key, [...(await load()), ...messages]),
|
|
212
|
+
)
|
|
213
|
+
ctx.set(HISTORY_REF, newRef)
|
|
214
|
+
},
|
|
215
|
+
}
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// Without a store, history lives in object state as a list of turns, so trimming never
|
|
219
|
+
// separates a tool call from its result.
|
|
220
|
+
const turns = (await ctx.get(HISTORY)) ?? []
|
|
221
|
+
return {
|
|
222
|
+
load: async () => turns.flat(),
|
|
223
|
+
save: async (messages, maxHistoryTurns) => {
|
|
224
|
+
ctx.set(HISTORY, [...turns, messages].slice(-maxHistoryTurns))
|
|
225
|
+
},
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
function skillLoader({ loadSkill, loadContext, input }) {
|
|
230
|
+
if (loadSkill) return (skillName) => loadSkill(skillName, input)
|
|
231
|
+
if (!loadContext) return undefined
|
|
232
|
+
return async (skillName) => {
|
|
233
|
+
const { skills = [] } = (await loadContext(input)) ?? {}
|
|
234
|
+
return skills.find((skill) => skill.name === skillName)?.body
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
|
|
238
|
+
/**
|
|
239
|
+
* @deprecated Use {@link defineAgent} with an explicit `name`. Kept so callers migrating
|
|
240
|
+
* from `@empyria/restate-llm` only change their import: defines a stateless agent named
|
|
241
|
+
* `AgentService` with the default `ask` input/output.
|
|
242
|
+
* @param {Omit<AgentSpec, 'name'>} spec
|
|
243
|
+
*/
|
|
244
|
+
export function createAgentService(spec) {
|
|
245
|
+
return defineAgent({ name: 'AgentService', ...spec })
|
|
246
|
+
}
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import { TerminalError, RetryableError } from '@restatedev/restate-sdk'
|
|
2
|
+
import { APICallError } from '@ai-sdk/provider'
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Default bounded retry policy for an LLM call's durable `ctx.run` step.
|
|
6
|
+
*
|
|
7
|
+
* `ctx.run`'s own SDK-wide defaults (`initialRetryInterval: 50ms`, unbounded attempts)
|
|
8
|
+
* are tuned for cheap, fast-failing side effects, not a rate-limited LLM gateway — a
|
|
9
|
+
* 50ms-then-double backoff burns through a 429's typical multi-second cooldown in a
|
|
10
|
+
* handful of attempts, and with no `maxRetryAttempts` it never gives up on a
|
|
11
|
+
* genuinely-broken endpoint. `maxRetryAttempts` also matters independently of the
|
|
12
|
+
* interval tuning: without it, any error this module fails to classify as
|
|
13
|
+
* {@link TerminalError} (see {@link toRestateLLMError}) retries forever.
|
|
14
|
+
*/
|
|
15
|
+
export const DEFAULT_LLM_RETRY = {
|
|
16
|
+
maxRetryAttempts: 5,
|
|
17
|
+
initialRetryInterval: 1_000,
|
|
18
|
+
maxRetryInterval: 30_000,
|
|
19
|
+
retryIntervalFactor: 2,
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Default bounded retry policy for an `execute` tool's durable `ctx.run` step. Once it's
|
|
24
|
+
* exhausted, `ctx.run` throws a `TerminalError`, which the agent loop hands back to the
|
|
25
|
+
* model as a tool error instead of failing the whole invocation — see `Tools.js`.
|
|
26
|
+
*/
|
|
27
|
+
export const DEFAULT_TOOL_RETRY = {
|
|
28
|
+
maxRetryAttempts: 3,
|
|
29
|
+
initialRetryInterval: 500,
|
|
30
|
+
maxRetryInterval: 10_000,
|
|
31
|
+
retryIntervalFactor: 2,
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/**
|
|
35
|
+
* Default invocation-level retry policy for an agent service/object. Applies to failures
|
|
36
|
+
* outside any bounded `ctx.run` (e.g. a bug in the handler itself): after `maxAttempts`
|
|
37
|
+
* the invocation is paused for an operator to fix and resume, never retried forever and
|
|
38
|
+
* never silently killed.
|
|
39
|
+
*/
|
|
40
|
+
export const DEFAULT_AGENT_RETRY_POLICY = {
|
|
41
|
+
maxAttempts: 10,
|
|
42
|
+
onMaxAttempts: 'pause',
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* Maps an error thrown by an AI SDK Core call (`generateText`/`streamText`/...) to the
|
|
47
|
+
* error a `ctx.run(name, fn, RunOptions)` closure should throw, so Restate retries
|
|
48
|
+
* exactly the failures worth retrying and stops immediately on the ones that aren't.
|
|
49
|
+
*
|
|
50
|
+
* `@ai-sdk/provider`'s `APICallError` already carries an `isRetryable` flag the AI SDK
|
|
51
|
+
* itself derives from the HTTP status code (true for 408/409/429/5xx, false for
|
|
52
|
+
* 400/401/403/404/422/...) — this reuses that classification instead of re-deriving it
|
|
53
|
+
* from status codes by hand, and layers Restate's two escape hatches on top of it:
|
|
54
|
+
*
|
|
55
|
+
* - Not retryable → {@link TerminalError}, so Restate fails the invocation immediately
|
|
56
|
+
* instead of retrying an auth/config/bad-request error forever.
|
|
57
|
+
* - Retryable AND the response carried a `Retry-After` header → {@link RetryableError},
|
|
58
|
+
* so Restate honors the gateway's own requested cooldown instead of guessing one.
|
|
59
|
+
* - Retryable with no such header → the original error, unchanged, so `ctx.run`'s own
|
|
60
|
+
* `RunOptions` backoff (see {@link DEFAULT_LLM_RETRY}) applies.
|
|
61
|
+
* - Anything that isn't an `APICallError` at all (a thrown `TypeError`, a network-layer
|
|
62
|
+
* error `fetch` itself throws, ...) is returned unchanged — Restate's default is to
|
|
63
|
+
* treat any non-`TerminalError` throw as retryable, which is the right default for an
|
|
64
|
+
* error shape this module doesn't recognize.
|
|
65
|
+
*
|
|
66
|
+
* Call sites re-throw the result — this function never throws itself, only classifies:
|
|
67
|
+
* `catch (error) { throw toRestateLLMError(error) }`.
|
|
68
|
+
* @param {unknown} error
|
|
69
|
+
* @returns {Error}
|
|
70
|
+
*/
|
|
71
|
+
export function toRestateLLMError(error) {
|
|
72
|
+
if (!APICallError.isInstance(error)) return error
|
|
73
|
+
|
|
74
|
+
if (!error.isRetryable) {
|
|
75
|
+
return new TerminalError(error.message, { errorCode: error.statusCode })
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
const retryAfterHeader = error.responseHeaders?.['retry-after']
|
|
79
|
+
const retryAfterSeconds = retryAfterHeader ? Number(retryAfterHeader) : undefined
|
|
80
|
+
|
|
81
|
+
if (retryAfterSeconds !== undefined && Number.isFinite(retryAfterSeconds)) {
|
|
82
|
+
return RetryableError.from(error, { retryAfter: { seconds: retryAfterSeconds } })
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
return error
|
|
86
|
+
}
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
import { generateText } from 'ai'
|
|
2
|
+
import { TerminalError } from '@restatedev/restate-sdk'
|
|
3
|
+
import { DEFAULT_LLM_RETRY, toRestateLLMError } from './Errors.js'
|
|
4
|
+
import { cachedSkillFetcher, expandSkillRefs } from './Skills.js'
|
|
5
|
+
import { FINAL_ANSWER_TOOL, buildToolDefs, executeTool } from './Tools.js'
|
|
6
|
+
|
|
7
|
+
const FINAL_ANSWER_NUDGE = `Respond by calling the ${FINAL_ANSWER_TOOL} tool with your answer.`
|
|
8
|
+
|
|
9
|
+
/**
|
|
10
|
+
* Drives an AI SDK Core tool-calling loop ONE round at a time, from inside a Restate
|
|
11
|
+
* handler — deliberately NOT `generateText`'s own multi-step loop wrapped in a single
|
|
12
|
+
* `ctx.run()`. That would make the whole loop atomic to Restate: a crash mid-loop would
|
|
13
|
+
* replay it from scratch, re-billing finished LLM calls and re-executing tools that
|
|
14
|
+
* already ran. Here every LLM call and every tool call is its own durable step, so a
|
|
15
|
+
* crash resumes exactly where it left off, and each step shows up by name in Restate's
|
|
16
|
+
* invocation UI.
|
|
17
|
+
*
|
|
18
|
+
* Journal discipline: an LLM step records only the model's response (text, tool calls,
|
|
19
|
+
* token usage), never the prompt it was sent. Prior conversation comes from
|
|
20
|
+
* `loadHistory`, which is called inside each LLM step — so it must return the same
|
|
21
|
+
* messages every time it's called during one invocation (e.g. load an immutable
|
|
22
|
+
* snapshot by reference). Loaded skills work the same way: the journal holds a
|
|
23
|
+
* reference and hash, and the body is fetched again inside each LLM step (see Skills.js).
|
|
24
|
+
* @param {import('@restatedev/restate-sdk').Context} ctx
|
|
25
|
+
* @param {{
|
|
26
|
+
* model: import('@ai-sdk/provider').LanguageModelV4,
|
|
27
|
+
* system: string,
|
|
28
|
+
* prompt: string,
|
|
29
|
+
* tools?: Record<string, import('./Tools.js').ToolSpec>,
|
|
30
|
+
* skills?: Array<{name: string, description: string}>,
|
|
31
|
+
* loadSkill?: (name: string) => Promise<string|undefined>,
|
|
32
|
+
* loadHistory?: () => Promise<Array<object>>,
|
|
33
|
+
* outputSchema?: object,
|
|
34
|
+
* maxSteps?: number,
|
|
35
|
+
* maxTokens?: number,
|
|
36
|
+
* retry?: import('@restatedev/restate-sdk').RunOptions<any>,
|
|
37
|
+
* }} params `skills` is the index only (name + description); `loadSkill` fetches a body
|
|
38
|
+
* when the model asks for it. With `outputSchema`, the model must answer through the
|
|
39
|
+
* `final_answer` tool and the loop returns that tool's input as `output`.
|
|
40
|
+
* @returns {Promise<{text: string, output?: unknown, steps: number, totalTokens: number, messages: Array<object>}>}
|
|
41
|
+
* `messages` are the new messages of this turn only (prompt, responses, tool results).
|
|
42
|
+
*/
|
|
43
|
+
export async function runAgentLoop(
|
|
44
|
+
ctx,
|
|
45
|
+
{
|
|
46
|
+
model,
|
|
47
|
+
system,
|
|
48
|
+
prompt,
|
|
49
|
+
tools = {},
|
|
50
|
+
skills = [],
|
|
51
|
+
loadSkill,
|
|
52
|
+
loadHistory = async () => [],
|
|
53
|
+
outputSchema,
|
|
54
|
+
maxSteps = 8,
|
|
55
|
+
maxTokens,
|
|
56
|
+
retry = DEFAULT_LLM_RETRY,
|
|
57
|
+
},
|
|
58
|
+
) {
|
|
59
|
+
const fullSystem = buildSystemPrompt(system, skills)
|
|
60
|
+
const toolDefs = buildToolDefs({ tools, hasSkills: skills.length > 0, outputSchema })
|
|
61
|
+
const messages = [{ role: 'user', content: prompt }]
|
|
62
|
+
let totalTokens = 0
|
|
63
|
+
const fetchSkill = cachedSkillFetcher(loadSkill)
|
|
64
|
+
|
|
65
|
+
for (let step = 0; step < maxSteps; step++) {
|
|
66
|
+
// `ctx.run` JSON-serialises the callback's result for the journal and hands back that
|
|
67
|
+
// copy. AI SDK's `response` is a non-enumerable getter `JSON.stringify` drops, so the
|
|
68
|
+
// plain fields the loop needs are extracted inside the callback.
|
|
69
|
+
const result = await ctx.run(
|
|
70
|
+
`llm-step-${step}`,
|
|
71
|
+
async () => {
|
|
72
|
+
try {
|
|
73
|
+
// `maxRetries: 0`: `ctx.run`'s RunOptions are the single retry authority,
|
|
74
|
+
// instead of AI SDK's own retries stacking invisibly underneath them.
|
|
75
|
+
const r = await generateText({
|
|
76
|
+
model,
|
|
77
|
+
system: fullSystem,
|
|
78
|
+
messages: [
|
|
79
|
+
...(await expandSkillRefs(await loadHistory(), {
|
|
80
|
+
fetchSkill,
|
|
81
|
+
strict: false,
|
|
82
|
+
})),
|
|
83
|
+
...(await expandSkillRefs(messages, { fetchSkill, strict: true })),
|
|
84
|
+
],
|
|
85
|
+
tools: toolDefs,
|
|
86
|
+
maxRetries: 0,
|
|
87
|
+
})
|
|
88
|
+
return {
|
|
89
|
+
text: r.text,
|
|
90
|
+
toolCalls: r.toolCalls.map(({ toolCallId, toolName, input }) => ({
|
|
91
|
+
toolCallId,
|
|
92
|
+
toolName,
|
|
93
|
+
input,
|
|
94
|
+
})),
|
|
95
|
+
responseMessages: r.response.messages,
|
|
96
|
+
totalTokens: r.usage?.totalTokens ?? 0,
|
|
97
|
+
}
|
|
98
|
+
} catch (error) {
|
|
99
|
+
throw toRestateLLMError(error)
|
|
100
|
+
}
|
|
101
|
+
},
|
|
102
|
+
retry,
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
messages.push(...result.responseMessages)
|
|
106
|
+
totalTokens += result.totalTokens
|
|
107
|
+
if (maxTokens !== undefined && totalTokens > maxTokens) {
|
|
108
|
+
throw new TerminalError(
|
|
109
|
+
`Agent exceeded its token budget (${totalTokens} > ${maxTokens})`,
|
|
110
|
+
{ errorCode: 429 },
|
|
111
|
+
)
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
if (result.toolCalls.length === 0) {
|
|
115
|
+
if (!outputSchema) {
|
|
116
|
+
return { text: result.text, steps: step + 1, totalTokens, messages }
|
|
117
|
+
}
|
|
118
|
+
messages.push({ role: 'user', content: FINAL_ANSWER_NUDGE })
|
|
119
|
+
continue
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
for (const [index, call] of result.toolCalls.entries()) {
|
|
123
|
+
if (outputSchema && call.toolName === FINAL_ANSWER_TOOL) {
|
|
124
|
+
return {
|
|
125
|
+
text: result.text,
|
|
126
|
+
output: call.input,
|
|
127
|
+
steps: step + 1,
|
|
128
|
+
totalTokens,
|
|
129
|
+
messages,
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
const output = await executeTool(ctx, call, { tools, loadSkill, step, index })
|
|
133
|
+
messages.push({
|
|
134
|
+
role: 'tool',
|
|
135
|
+
content: [
|
|
136
|
+
{
|
|
137
|
+
type: 'tool-result',
|
|
138
|
+
toolCallId: call.toolCallId,
|
|
139
|
+
toolName: call.toolName,
|
|
140
|
+
output,
|
|
141
|
+
},
|
|
142
|
+
],
|
|
143
|
+
})
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// Exceeding maxSteps isn't a transient fault a retry could fix.
|
|
148
|
+
throw new TerminalError(`Agent loop exceeded maxSteps (${maxSteps}) without a final answer`)
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
function buildSystemPrompt(system, skills) {
|
|
152
|
+
if (skills.length === 0) return system
|
|
153
|
+
|
|
154
|
+
const index = skills.map((skill) => `- ${skill.name}: ${skill.description}`).join('\n')
|
|
155
|
+
const instructions = `Available skills — call load_skill with a skill's name to read its full instructions before following them:\n${index}`
|
|
156
|
+
return [system, instructions].filter(Boolean).join('\n\n')
|
|
157
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
import { createOpenAICompatible } from '@ai-sdk/openai-compatible'
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* AI SDK Core model, pointed at an OpenAI-compatible endpoint — Bifrost
|
|
5
|
+
* (llm-system's gateway) by default, fully overridable via env. `LLM_BASE_URL`
|
|
6
|
+
* already includes the `/v1` suffix Bifrost's own docs use (`https://agent-bureau.vip/v1`);
|
|
7
|
+
* `MODEL_ID` defaults to `vllm/ornith-1.5`, the currently-live routed model ID on that
|
|
8
|
+
* gateway (see llm-system's README, "Verifying").
|
|
9
|
+
* @param {{LLM_BASE_URL: string, LLM_API_KEY?: string, MODEL_ID: string}} env
|
|
10
|
+
* @returns {import('@ai-sdk/provider').LanguageModelV4}
|
|
11
|
+
*/
|
|
12
|
+
export function createModel(env) {
|
|
13
|
+
const provider = createOpenAICompatible({
|
|
14
|
+
name: 'bifrost',
|
|
15
|
+
baseURL: env.LLM_BASE_URL,
|
|
16
|
+
apiKey: env.LLM_API_KEY,
|
|
17
|
+
})
|
|
18
|
+
return provider(env.MODEL_ID)
|
|
19
|
+
}
|