@owlmeans/agent 0.1.18-rc.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -0
- package/agent-meta/manifest.json +16 -0
- package/agent-meta/skills/agent/SKILL.md +122 -0
- package/build/consts.d.ts +16 -0
- package/build/consts.d.ts.map +1 -0
- package/build/consts.js +16 -0
- package/build/consts.js.map +1 -0
- package/build/errors.d.ts +16 -0
- package/build/errors.d.ts.map +1 -0
- package/build/errors.js +27 -0
- package/build/errors.js.map +1 -0
- package/build/helpers/compaction.d.ts +46 -0
- package/build/helpers/compaction.d.ts.map +1 -0
- package/build/helpers/compaction.js +119 -0
- package/build/helpers/compaction.js.map +1 -0
- package/build/helpers/index.d.ts +4 -0
- package/build/helpers/index.d.ts.map +1 -0
- package/build/helpers/index.js +4 -0
- package/build/helpers/index.js.map +1 -0
- package/build/helpers/rolling.d.ts +25 -0
- package/build/helpers/rolling.d.ts.map +1 -0
- package/build/helpers/rolling.js +45 -0
- package/build/helpers/rolling.js.map +1 -0
- package/build/helpers/tools.d.ts +29 -0
- package/build/helpers/tools.d.ts.map +1 -0
- package/build/helpers/tools.js +36 -0
- package/build/helpers/tools.js.map +1 -0
- package/build/index.d.ts +12 -0
- package/build/index.d.ts.map +1 -0
- package/build/index.js +11 -0
- package/build/index.js.map +1 -0
- package/build/model.d.ts +15 -0
- package/build/model.d.ts.map +1 -0
- package/build/model.js +208 -0
- package/build/model.js.map +1 -0
- package/build/plugins/export.d.ts +8 -0
- package/build/plugins/export.d.ts.map +1 -0
- package/build/plugins/export.js +4 -0
- package/build/plugins/export.js.map +1 -0
- package/build/plugins/memory-events.d.ts +36 -0
- package/build/plugins/memory-events.d.ts.map +1 -0
- package/build/plugins/memory-events.js +83 -0
- package/build/plugins/memory-events.js.map +1 -0
- package/build/plugins/memory-graph.d.ts +52 -0
- package/build/plugins/memory-graph.d.ts.map +1 -0
- package/build/plugins/memory-graph.js +155 -0
- package/build/plugins/memory-graph.js.map +1 -0
- package/build/plugins/summarize.d.ts +44 -0
- package/build/plugins/summarize.d.ts.map +1 -0
- package/build/plugins/summarize.js +77 -0
- package/build/plugins/summarize.js.map +1 -0
- package/build/runtime/checkpoint.d.ts +37 -0
- package/build/runtime/checkpoint.d.ts.map +1 -0
- package/build/runtime/checkpoint.js +52 -0
- package/build/runtime/checkpoint.js.map +1 -0
- package/build/runtime/provider.d.ts +18 -0
- package/build/runtime/provider.d.ts.map +1 -0
- package/build/runtime/provider.js +30 -0
- package/build/runtime/provider.js.map +1 -0
- package/build/runtime/transport.d.ts +27 -0
- package/build/runtime/transport.d.ts.map +1 -0
- package/build/runtime/transport.js +29 -0
- package/build/runtime/transport.js.map +1 -0
- package/build/service.d.ts +14 -0
- package/build/service.d.ts.map +1 -0
- package/build/service.js +61 -0
- package/build/service.js.map +1 -0
- package/build/stores/index.d.ts +3 -0
- package/build/stores/index.d.ts.map +1 -0
- package/build/stores/index.js +2 -0
- package/build/stores/index.js.map +1 -0
- package/build/stores/memory.d.ts +6 -0
- package/build/stores/memory.d.ts.map +1 -0
- package/build/stores/memory.js +0 -0
- package/build/stores/memory.js.map +1 -0
- package/build/stores/types.d.ts +38 -0
- package/build/stores/types.d.ts.map +1 -0
- package/build/stores/types.js +2 -0
- package/build/stores/types.js.map +1 -0
- package/build/types.d.ts +130 -0
- package/build/types.d.ts.map +1 -0
- package/build/types.js +2 -0
- package/build/types.js.map +1 -0
- package/package.json +72 -0
- package/src/consts.ts +19 -0
- package/src/errors.ts +33 -0
- package/src/helpers/compaction.ts +172 -0
- package/src/helpers/index.ts +3 -0
- package/src/helpers/rolling.ts +68 -0
- package/src/helpers/tools.ts +46 -0
- package/src/index.ts +11 -0
- package/src/model.ts +269 -0
- package/src/plugins/export.ts +12 -0
- package/src/plugins/memory-events.ts +129 -0
- package/src/plugins/memory-graph.ts +217 -0
- package/src/plugins/summarize.ts +129 -0
- package/src/runtime/checkpoint.ts +89 -0
- package/src/runtime/provider.ts +35 -0
- package/src/runtime/transport.ts +50 -0
- package/src/service.ts +97 -0
- package/src/stores/index.ts +2 -0
- package/src/stores/memory.ts +0 -0
- package/src/stores/types.ts +45 -0
- package/src/types.ts +144 -0
- package/tests/_tools/model.ts +50 -0
- package/tests/agent.spec.ts +218 -0
- package/tests/plugins.spec.ts +257 -0
- package/tests/runtime.spec.ts +140 -0
- package/tests/summary.spec.ts +143 -0
- package/tests/tools.spec.ts +68 -0
- package/tsconfig.json +19 -0
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
import { describe, expect, test } from 'bun:test'
|
|
2
|
+
import { tool } from '@langchain/core/tools'
|
|
3
|
+
import * as z from 'zod'
|
|
4
|
+
import { ExecutionLevel, PromptBlock } from '@owlmeans/llm-common'
|
|
5
|
+
import { makePromptService } from '@owlmeans/llm'
|
|
6
|
+
import type { Execution, PromptService } from '@owlmeans/llm'
|
|
7
|
+
import { AgentRunStatus } from '@owlmeans/agent-common'
|
|
8
|
+
import { makeAgentModel } from '../src/index.js'
|
|
9
|
+
import type { AgentPlugin, AgentToolSet } from '../src/index.js'
|
|
10
|
+
import { scriptedModel } from './_tools/model.js'
|
|
11
|
+
|
|
12
|
+
let seq = 0
|
|
13
|
+
const promptService = (): PromptService =>
|
|
14
|
+
makePromptService({}, `agent-spec-prompts-${++seq}`)
|
|
15
|
+
|
|
16
|
+
const execution = (prompts?: () => PromptService): Execution => ({
|
|
17
|
+
level: ExecutionLevel.Helper,
|
|
18
|
+
purpose: { dedication: 'project:p1' },
|
|
19
|
+
policy: { effort: 'standard' as never },
|
|
20
|
+
prompt: { role: 'You are a test agent.' },
|
|
21
|
+
models: (() => { throw new Error('unused') }) as never,
|
|
22
|
+
...(prompts != null ? { prompts } : {}),
|
|
23
|
+
} as unknown as Execution)
|
|
24
|
+
|
|
25
|
+
const echo = tool(
|
|
26
|
+
async ({ value }: { value: string }) => `echoed:${value}`,
|
|
27
|
+
{ name: 'echo', description: 'Echoes its argument.', schema: z.object({ value: z.string() }) },
|
|
28
|
+
) as unknown as AgentToolSet[string]
|
|
29
|
+
|
|
30
|
+
describe('agent — the tool loop', () => {
|
|
31
|
+
test('answers without tools when the model asks for none', async () => {
|
|
32
|
+
const scripted = scriptedModel([{ content: 'done' }])
|
|
33
|
+
const agent = makeAgentModel({
|
|
34
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
35
|
+
})
|
|
36
|
+
|
|
37
|
+
const result = await agent.invoke('do the thing')
|
|
38
|
+
|
|
39
|
+
expect(result.message.content).toBe('done')
|
|
40
|
+
expect(scripted.turns()).toBe(1)
|
|
41
|
+
})
|
|
42
|
+
|
|
43
|
+
test('runs a requested tool and asks again with its result', async () => {
|
|
44
|
+
const scripted = scriptedModel([
|
|
45
|
+
{ toolCalls: [{ name: 'echo', args: { value: 'hi' } }] },
|
|
46
|
+
{ content: 'finished' },
|
|
47
|
+
])
|
|
48
|
+
const agent = makeAgentModel({
|
|
49
|
+
exec: execution(), agentModel: scripted.model, tools: { echo },
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
const result = await agent.invoke('use the tool')
|
|
53
|
+
|
|
54
|
+
expect(result.message.content).toBe('finished')
|
|
55
|
+
expect(scripted.turns()).toBe(2)
|
|
56
|
+
// The tool's answer has to reach the second ask, or the loop is a no-op with extra steps.
|
|
57
|
+
expect(JSON.stringify(scripted.asked()[1])).toContain('echoed:hi')
|
|
58
|
+
})
|
|
59
|
+
|
|
60
|
+
test('survives a tool that fails, feeding the error back to the model', async () => {
|
|
61
|
+
const explodes = tool(
|
|
62
|
+
async () => { throw new Error('nope') },
|
|
63
|
+
{ name: 'explodes', description: 'Throws.', schema: z.object({}) },
|
|
64
|
+
) as unknown as AgentToolSet[string]
|
|
65
|
+
const scripted = scriptedModel([
|
|
66
|
+
{ toolCalls: [{ name: 'explodes', args: {} }] },
|
|
67
|
+
{ content: 'recovered' },
|
|
68
|
+
])
|
|
69
|
+
const agent = makeAgentModel({
|
|
70
|
+
exec: execution(), agentModel: scripted.model, tools: { explodes },
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
const result = await agent.invoke('break it')
|
|
74
|
+
|
|
75
|
+
expect(result.message.content).toBe('recovered')
|
|
76
|
+
expect(JSON.stringify(scripted.asked()[1])).toContain('nope')
|
|
77
|
+
})
|
|
78
|
+
|
|
79
|
+
test('stops at the turn ceiling instead of looping on a model that never answers', async () => {
|
|
80
|
+
// A model that keeps calling tools forever is the ordinary failure of an unsatisfied loop.
|
|
81
|
+
// Without a ceiling it spends the caller's budget until something else kills it.
|
|
82
|
+
const scripted = scriptedModel([{ toolCalls: [{ name: 'echo', args: { value: 'again' } }] }])
|
|
83
|
+
const agent = makeAgentModel({
|
|
84
|
+
exec: execution(), agentModel: scripted.model, tools: { echo }, maxTurns: 3,
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
expect(agent.invoke('spin')).rejects.toThrow(/loop-exhausted/)
|
|
88
|
+
})
|
|
89
|
+
})
|
|
90
|
+
|
|
91
|
+
describe('agent — plugins', () => {
|
|
92
|
+
const collecting = (): { plugin: AgentPlugin, finished: unknown[] } => {
|
|
93
|
+
const finished: unknown[] = []
|
|
94
|
+
return {
|
|
95
|
+
finished,
|
|
96
|
+
plugin: {
|
|
97
|
+
alias: 'collector',
|
|
98
|
+
context: async () => ['REMEMBERED CONTEXT'],
|
|
99
|
+
onFinish: async (_run, _result, outcome) => { finished.push(outcome) },
|
|
100
|
+
},
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
test('contributes context into the prompt and fires on finish', async () => {
|
|
105
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
106
|
+
const { plugin, finished } = collecting()
|
|
107
|
+
const agent = makeAgentModel({
|
|
108
|
+
exec: execution(promptService), agentModel: scripted.model, tools: {}, plugins: [plugin],
|
|
109
|
+
})
|
|
110
|
+
|
|
111
|
+
await agent.invoke('go')
|
|
112
|
+
|
|
113
|
+
expect(JSON.stringify(scripted.asked()[0])).toContain('REMEMBERED CONTEXT')
|
|
114
|
+
expect(finished).toHaveLength(1)
|
|
115
|
+
expect((finished[0] as { status: string }).status).toBe(AgentRunStatus.Ok)
|
|
116
|
+
})
|
|
117
|
+
|
|
118
|
+
test('contributed context lands in the volatile block, never in the cached prefix', async () => {
|
|
119
|
+
// The Context block is the only one a provider will not put a cache breakpoint on. Volatile
|
|
120
|
+
// material anywhere above it invalidates the prefix every call that shares the persona pays for.
|
|
121
|
+
const prompts = promptService()
|
|
122
|
+
const composed = await prompts.compose(
|
|
123
|
+
{ role: 'You are a test agent.', context: ['REMEMBERED CONTEXT'] },
|
|
124
|
+
[],
|
|
125
|
+
{ model: scriptedModel([]).model },
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
const context = composed.blocks.find(block => block.block === PromptBlock.Context)
|
|
129
|
+
const role = composed.blocks.find(block => block.block === PromptBlock.Role)
|
|
130
|
+
|
|
131
|
+
expect(context?.text).toContain('REMEMBERED CONTEXT')
|
|
132
|
+
expect(role?.text).not.toContain('REMEMBERED CONTEXT')
|
|
133
|
+
})
|
|
134
|
+
|
|
135
|
+
test('a plugin that throws while contributing does not stop the run', async () => {
|
|
136
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
137
|
+
const agent = makeAgentModel({
|
|
138
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
139
|
+
plugins: [{ alias: 'broken', context: async () => { throw new Error('no memory today') } }],
|
|
140
|
+
})
|
|
141
|
+
|
|
142
|
+
expect((await agent.invoke('go')).message.content).toBe('ok')
|
|
143
|
+
})
|
|
144
|
+
|
|
145
|
+
test('reports a failed run to its plugins before rethrowing', async () => {
|
|
146
|
+
// A run that vanishes from the history is one the next session repeats verbatim.
|
|
147
|
+
const scripted = scriptedModel([{ toolCalls: [{ name: 'echo', args: { value: 'x' } }] }])
|
|
148
|
+
const { plugin, finished } = collecting()
|
|
149
|
+
const agent = makeAgentModel({
|
|
150
|
+
exec: execution(), agentModel: scripted.model, tools: { echo }, maxTurns: 1,
|
|
151
|
+
plugins: [plugin],
|
|
152
|
+
})
|
|
153
|
+
|
|
154
|
+
expect(agent.invoke('spin')).rejects.toThrow()
|
|
155
|
+
await new Promise(resolve => setTimeout(resolve, 50))
|
|
156
|
+
|
|
157
|
+
expect((finished[0] as { status: string })?.status).toBe(AgentRunStatus.Failed)
|
|
158
|
+
})
|
|
159
|
+
|
|
160
|
+
test('seats a re-registered plugin in place rather than emitting it twice', async () => {
|
|
161
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
162
|
+
const agent = makeAgentModel({ exec: execution(promptService), agentModel: scripted.model, tools: {} })
|
|
163
|
+
|
|
164
|
+
agent.use({ alias: 'ctx', context: async () => ['ONCE'] })
|
|
165
|
+
agent.use({ alias: 'ctx', context: async () => ['ONCE'] })
|
|
166
|
+
|
|
167
|
+
await agent.invoke('go')
|
|
168
|
+
|
|
169
|
+
const asked = JSON.stringify(scripted.asked()[0])
|
|
170
|
+
expect(asked.split('ONCE').length - 1).toBe(1)
|
|
171
|
+
})
|
|
172
|
+
})
|
|
173
|
+
|
|
174
|
+
describe('agent — finalization', () => {
|
|
175
|
+
test('defers finalization when the caller owns the outcome', async () => {
|
|
176
|
+
// Free flight validates and repairs AFTER the model stops talking; a compaction written before
|
|
177
|
+
// that describes a state which did not survive it.
|
|
178
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
179
|
+
const finished: unknown[] = []
|
|
180
|
+
const agent = makeAgentModel({
|
|
181
|
+
exec: execution(), agentModel: scripted.model, tools: {}, autoFinish: false,
|
|
182
|
+
plugins: [{ alias: 'c', onFinish: async (_r, _res, outcome) => { finished.push(outcome) } }],
|
|
183
|
+
})
|
|
184
|
+
|
|
185
|
+
const result = await agent.invoke('go')
|
|
186
|
+
expect(finished).toHaveLength(0)
|
|
187
|
+
|
|
188
|
+
await result.run.finish({ status: AgentRunStatus.Ok, note: 'fixers clean' })
|
|
189
|
+
expect((finished[0] as { note: string }).note).toBe('fixers clean')
|
|
190
|
+
})
|
|
191
|
+
|
|
192
|
+
test('a second finish is a no-op, never a second event', async () => {
|
|
193
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
194
|
+
const finished: unknown[] = []
|
|
195
|
+
const agent = makeAgentModel({
|
|
196
|
+
exec: execution(), agentModel: scripted.model, tools: {}, autoFinish: false,
|
|
197
|
+
plugins: [{ alias: 'c', onFinish: async () => { finished.push(1) } }],
|
|
198
|
+
})
|
|
199
|
+
|
|
200
|
+
const result = await agent.invoke('go')
|
|
201
|
+
await result.run.finish({ status: AgentRunStatus.Ok })
|
|
202
|
+
await result.run.finish({ status: AgentRunStatus.Ok })
|
|
203
|
+
|
|
204
|
+
expect(finished).toHaveLength(1)
|
|
205
|
+
})
|
|
206
|
+
|
|
207
|
+
test('a failing plugin does not fail the run it is finalizing', async () => {
|
|
208
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
209
|
+
const agent = makeAgentModel({
|
|
210
|
+
exec: execution(), agentModel: scripted.model, tools: {}, autoFinish: false,
|
|
211
|
+
plugins: [{ alias: 'c', onFinish: async () => { throw new Error('store is down') } }],
|
|
212
|
+
})
|
|
213
|
+
|
|
214
|
+
const result = await agent.invoke('go')
|
|
215
|
+
|
|
216
|
+
expect(result.run.finish({ status: AgentRunStatus.Ok })).resolves.toBeUndefined()
|
|
217
|
+
})
|
|
218
|
+
})
|
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
import { describe, expect, test } from 'bun:test'
|
|
2
|
+
import { ExecutionLevel } from '@owlmeans/llm-common'
|
|
3
|
+
import type { Execution, LlmModel, PromptService } from '@owlmeans/llm'
|
|
4
|
+
import { makePromptService } from '@owlmeans/llm'
|
|
5
|
+
import { AgentRunStatus } from '@owlmeans/agent-common'
|
|
6
|
+
import {
|
|
7
|
+
createMemoryConversationStore, createMemoryEventStore, createMemoryGraphStore, makeAgentModel,
|
|
8
|
+
memoryEvents, memoryEventsPlugin, memoryGraph, memoryGraphPlugin, summarizePlugin,
|
|
9
|
+
} from '../src/index.js'
|
|
10
|
+
import { scriptedModel } from './_tools/model.js'
|
|
11
|
+
|
|
12
|
+
let seq = 0
|
|
13
|
+
const execution = (): Execution => ({
|
|
14
|
+
level: ExecutionLevel.Helper,
|
|
15
|
+
purpose: { dedication: 'project:p1' },
|
|
16
|
+
policy: { effort: 'standard' as never },
|
|
17
|
+
prompt: { role: 'You are a test agent.' },
|
|
18
|
+
models: (() => { throw new Error('unused') }) as never,
|
|
19
|
+
prompts: () => makePromptService({}, `plugins-spec-${++seq}`) as PromptService,
|
|
20
|
+
} as unknown as Execution)
|
|
21
|
+
|
|
22
|
+
const answering = (answer: unknown): LlmModel =>
|
|
23
|
+
({ invoke: async () => answer, ask: async () => answer as string } as unknown as LlmModel)
|
|
24
|
+
|
|
25
|
+
describe('agent — the summarize plugin', () => {
|
|
26
|
+
test('stores a two-part compaction when a run finishes', async () => {
|
|
27
|
+
const store = createMemoryConversationStore()
|
|
28
|
+
const scripted = scriptedModel([{ content: 'renamed it' }])
|
|
29
|
+
const agent = makeAgentModel({
|
|
30
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
31
|
+
plugins: [summarizePlugin({
|
|
32
|
+
store, model: () => answering({ summary: 'renamed the header', advice: 'run the build' }),
|
|
33
|
+
})],
|
|
34
|
+
})
|
|
35
|
+
|
|
36
|
+
await agent.invoke('rename the header')
|
|
37
|
+
|
|
38
|
+
const [event] = await store.last({ conversationId: 'project:p1', scope: 'p1' }, 5)
|
|
39
|
+
expect(event).toMatchObject({
|
|
40
|
+
summary: 'renamed the header', advice: 'run the build', status: AgentRunStatus.Ok,
|
|
41
|
+
prompt: 'rename the header',
|
|
42
|
+
})
|
|
43
|
+
})
|
|
44
|
+
|
|
45
|
+
test('puts the last sessions and the newest advice back into the next run', async () => {
|
|
46
|
+
const store = createMemoryConversationStore()
|
|
47
|
+
for (const [summary, advice] of [['did A', 'then B'], ['did B', 'then C']]) {
|
|
48
|
+
await store.append({
|
|
49
|
+
conversationId: 'project:p1', scope: 'p1', summary, advice, status: AgentRunStatus.Ok,
|
|
50
|
+
})
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
54
|
+
const agent = makeAgentModel({
|
|
55
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
56
|
+
plugins: [summarizePlugin({ store, model: () => answering({ summary: 's', advice: 'a' }) })],
|
|
57
|
+
})
|
|
58
|
+
|
|
59
|
+
await agent.invoke('carry on')
|
|
60
|
+
|
|
61
|
+
const asked = JSON.stringify(scripted.asked()[0])
|
|
62
|
+
expect(asked).toContain('did A')
|
|
63
|
+
expect(asked).toContain('did B')
|
|
64
|
+
// Only the NEWEST advice: two conflicting "do this next" instructions is worse than none.
|
|
65
|
+
expect(asked).toContain('then C')
|
|
66
|
+
expect(asked.split('then B').length - 1).toBe(0)
|
|
67
|
+
})
|
|
68
|
+
|
|
69
|
+
test('honours the window rather than replaying the whole conversation', async () => {
|
|
70
|
+
const store = createMemoryConversationStore()
|
|
71
|
+
for (const summary of ['one', 'two', 'three', 'four']) {
|
|
72
|
+
await store.append({
|
|
73
|
+
conversationId: 'project:p1', scope: 'p1', summary, status: AgentRunStatus.Ok,
|
|
74
|
+
})
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
78
|
+
await makeAgentModel({
|
|
79
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
80
|
+
plugins: [summarizePlugin({ store, window: 2, model: () => answering({ summary: 's', advice: '' }) })],
|
|
81
|
+
}).invoke('go')
|
|
82
|
+
|
|
83
|
+
const asked = JSON.stringify(scripted.asked()[0])
|
|
84
|
+
expect(asked).toContain('four')
|
|
85
|
+
expect(asked).toContain('three')
|
|
86
|
+
expect(asked).not.toContain('"one"')
|
|
87
|
+
})
|
|
88
|
+
|
|
89
|
+
test('contributes nothing on the very first run', async () => {
|
|
90
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
91
|
+
await makeAgentModel({
|
|
92
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
93
|
+
plugins: [summarizePlugin({ store: createMemoryConversationStore() })],
|
|
94
|
+
}).invoke('go')
|
|
95
|
+
|
|
96
|
+
expect(JSON.stringify(scripted.asked()[0])).not.toContain('Earlier in this conversation')
|
|
97
|
+
})
|
|
98
|
+
|
|
99
|
+
test('is a no-op with no store bound', async () => {
|
|
100
|
+
// An application that has not wired persistence still runs agents; it just has no memory.
|
|
101
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
102
|
+
const agent = makeAgentModel({
|
|
103
|
+
exec: execution(), agentModel: scripted.model, tools: {}, plugins: [summarizePlugin({})],
|
|
104
|
+
})
|
|
105
|
+
|
|
106
|
+
expect((await agent.invoke('go')).message.content).toBe('ok')
|
|
107
|
+
})
|
|
108
|
+
|
|
109
|
+
test('records a failed run so the next session does not repeat it verbatim', async () => {
|
|
110
|
+
const store = createMemoryConversationStore()
|
|
111
|
+
const scripted = scriptedModel([{ toolCalls: [{ name: 'missing', args: {} }] }])
|
|
112
|
+
const agent = makeAgentModel({
|
|
113
|
+
exec: execution(), agentModel: scripted.model, tools: {}, maxTurns: 1,
|
|
114
|
+
plugins: [summarizePlugin({ store })],
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
expect(agent.invoke('do the impossible')).rejects.toThrow()
|
|
118
|
+
await new Promise(resolve => setTimeout(resolve, 50))
|
|
119
|
+
|
|
120
|
+
const [event] = await store.last({ conversationId: 'project:p1', scope: 'p1' }, 5)
|
|
121
|
+
expect(event?.status).toBe(AgentRunStatus.Failed)
|
|
122
|
+
})
|
|
123
|
+
})
|
|
124
|
+
|
|
125
|
+
describe('agent — the subsystem memory graph', () => {
|
|
126
|
+
test('merges into a node instead of replacing it', async () => {
|
|
127
|
+
// Replacing would make every write a potential act of forgetting.
|
|
128
|
+
const graph = memoryGraph(createMemoryGraphStore())
|
|
129
|
+
await graph.write('p1', 'auth', 'uses OIDC')
|
|
130
|
+
await graph.write('p1', 'auth', 'tokens live 1h')
|
|
131
|
+
|
|
132
|
+
const [node] = await graph.read('p1', 'auth')
|
|
133
|
+
expect(node.content).toContain('uses OIDC')
|
|
134
|
+
expect(node.content).toContain('tokens live 1h')
|
|
135
|
+
})
|
|
136
|
+
|
|
137
|
+
test('compacts a node that outgrows its budget', async () => {
|
|
138
|
+
const graph = memoryGraph(createMemoryGraphStore(), {
|
|
139
|
+
maxNodeChars: 60,
|
|
140
|
+
model: () => answering('the folded account'),
|
|
141
|
+
run: {} as never,
|
|
142
|
+
})
|
|
143
|
+
await graph.write('p1', 'auth', 'A'.repeat(50))
|
|
144
|
+
await graph.write('p1', 'auth', 'B'.repeat(50))
|
|
145
|
+
|
|
146
|
+
const [node] = await graph.read('p1', 'auth')
|
|
147
|
+
expect(node.content.length).toBeLessThanOrEqual(60)
|
|
148
|
+
expect(node.content).toBe('the folded account')
|
|
149
|
+
})
|
|
150
|
+
|
|
151
|
+
test('accumulates links and never links a node to itself', async () => {
|
|
152
|
+
const graph = memoryGraph(createMemoryGraphStore())
|
|
153
|
+
await graph.write('p1', 'auth', 'x', ['db'])
|
|
154
|
+
await graph.write('p1', 'auth', 'y', ['api', 'auth'])
|
|
155
|
+
|
|
156
|
+
const [node] = await graph.read('p1', 'auth')
|
|
157
|
+
expect(node.links.sort()).toEqual(['api', 'db'])
|
|
158
|
+
})
|
|
159
|
+
|
|
160
|
+
test('follows links to the requested depth and no further', async () => {
|
|
161
|
+
const graph = memoryGraph(createMemoryGraphStore())
|
|
162
|
+
await graph.write('p1', 'auth', 'a', ['db'])
|
|
163
|
+
await graph.write('p1', 'db', 'b', ['storage'])
|
|
164
|
+
await graph.write('p1', 'storage', 'c')
|
|
165
|
+
|
|
166
|
+
expect((await graph.read('p1', 'auth', 0)).map(node => node.subsystem)).toEqual(['auth'])
|
|
167
|
+
expect((await graph.read('p1', 'auth', 1)).map(node => node.subsystem)).toEqual(['auth', 'db'])
|
|
168
|
+
expect((await graph.read('p1', 'auth', 2)).map(node => node.subsystem))
|
|
169
|
+
.toEqual(['auth', 'db', 'storage'])
|
|
170
|
+
})
|
|
171
|
+
|
|
172
|
+
test('survives a cycle in the graph', async () => {
|
|
173
|
+
const graph = memoryGraph(createMemoryGraphStore())
|
|
174
|
+
await graph.write('p1', 'a', 'x', ['b'])
|
|
175
|
+
await graph.write('p1', 'b', 'y', ['a'])
|
|
176
|
+
|
|
177
|
+
expect((await graph.read('p1', 'a', 5)).map(node => node.subsystem)).toEqual(['a', 'b'])
|
|
178
|
+
})
|
|
179
|
+
|
|
180
|
+
test('injects the index but never the content of a node', async () => {
|
|
181
|
+
// Bulk-injecting notes spends the context window on knowledge the run cannot tell apart from
|
|
182
|
+
// what it needs; the agent pulls what it wants by name.
|
|
183
|
+
const store = createMemoryGraphStore()
|
|
184
|
+
await memoryGraph(store).write('p1', 'auth', 'a long private note', ['db'])
|
|
185
|
+
|
|
186
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
187
|
+
await makeAgentModel({
|
|
188
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
189
|
+
plugins: [memoryGraphPlugin({ store })],
|
|
190
|
+
}).invoke('go')
|
|
191
|
+
|
|
192
|
+
const asked = JSON.stringify(scripted.asked()[0])
|
|
193
|
+
expect(asked).toContain('auth')
|
|
194
|
+
expect(asked).not.toContain('a long private note')
|
|
195
|
+
})
|
|
196
|
+
|
|
197
|
+
test('offers its tools to the agent', async () => {
|
|
198
|
+
const store = createMemoryGraphStore()
|
|
199
|
+
const scripted = scriptedModel([
|
|
200
|
+
{ toolCalls: [{ name: 'memory_write', args: { subsystem: 'auth', content: 'noted' } }] },
|
|
201
|
+
{ content: 'done' },
|
|
202
|
+
])
|
|
203
|
+
|
|
204
|
+
await makeAgentModel({
|
|
205
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
206
|
+
plugins: [memoryGraphPlugin({ store })],
|
|
207
|
+
}).invoke('remember something')
|
|
208
|
+
|
|
209
|
+
expect((await store.read('p1', 'auth'))?.content).toBe('noted')
|
|
210
|
+
})
|
|
211
|
+
})
|
|
212
|
+
|
|
213
|
+
describe('agent — the event sequence memory', () => {
|
|
214
|
+
test('keeps only the most recent entries', async () => {
|
|
215
|
+
const events = memoryEvents(createMemoryEventStore(), { limit: 2 })
|
|
216
|
+
for (const content of ['a', 'b', 'c']) {
|
|
217
|
+
await events.append('p1', 'note', content)
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
expect((await events.read('p1', 10)).map(event => event.content)).toEqual(['c', 'b'])
|
|
221
|
+
})
|
|
222
|
+
|
|
223
|
+
test('caps a single entry — an event is a line, not a document', async () => {
|
|
224
|
+
const events = memoryEvents(createMemoryEventStore(), { maxEventChars: 20 })
|
|
225
|
+
await events.append('p1', 'note', 'x'.repeat(500))
|
|
226
|
+
|
|
227
|
+
expect((await events.read('p1', 1))[0].content.length).toBeLessThanOrEqual(20)
|
|
228
|
+
})
|
|
229
|
+
|
|
230
|
+
test('injects the recent window into a run', async () => {
|
|
231
|
+
const store = createMemoryEventStore()
|
|
232
|
+
await memoryEvents(store).append('p1', 'deploy', 'shipped version 3')
|
|
233
|
+
|
|
234
|
+
const scripted = scriptedModel([{ content: 'ok' }])
|
|
235
|
+
await makeAgentModel({
|
|
236
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
237
|
+
plugins: [memoryEventsPlugin({ store })],
|
|
238
|
+
}).invoke('go')
|
|
239
|
+
|
|
240
|
+
expect(JSON.stringify(scripted.asked()[0])).toContain('shipped version 3')
|
|
241
|
+
})
|
|
242
|
+
|
|
243
|
+
test('lets an agent append through its tool', async () => {
|
|
244
|
+
const store = createMemoryEventStore()
|
|
245
|
+
const scripted = scriptedModel([
|
|
246
|
+
{ toolCalls: [{ name: 'event_append', args: { kind: 'fix', content: 'patched the header' } }] },
|
|
247
|
+
{ content: 'done' },
|
|
248
|
+
])
|
|
249
|
+
|
|
250
|
+
await makeAgentModel({
|
|
251
|
+
exec: execution(), agentModel: scripted.model, tools: {},
|
|
252
|
+
plugins: [memoryEventsPlugin({ store })],
|
|
253
|
+
}).invoke('fix it')
|
|
254
|
+
|
|
255
|
+
expect((await store.read('p1', 5))[0].content).toBe('patched the header')
|
|
256
|
+
})
|
|
257
|
+
})
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
import { describe, expect, test } from 'bun:test'
|
|
2
|
+
import { makeFlowModel } from '@owlmeans/flow'
|
|
3
|
+
import { ExecutionLevel } from '@owlmeans/llm-common'
|
|
4
|
+
import type { ExecutionState } from '@owlmeans/llm-common'
|
|
5
|
+
import type { Execution } from '@owlmeans/llm'
|
|
6
|
+
import { AGENT_RUN_FLOW, AgentRunStep, AgentRunTransition, agentRunFlow } from '@owlmeans/agent-common'
|
|
7
|
+
import {
|
|
8
|
+
createMemoryConversationStore, createMemoryEventStore, createMemoryGraphStore,
|
|
9
|
+
createMemoryRunStateStore, inProcessTransport, makeAgentExecutionPlugin, makeStaticFlowProvider,
|
|
10
|
+
} from '../src/index.js'
|
|
11
|
+
|
|
12
|
+
const state = (extra: Record<string, unknown> = {}): ExecutionState => ({
|
|
13
|
+
level: ExecutionLevel.Project,
|
|
14
|
+
purpose: { dedication: 'project:p1' },
|
|
15
|
+
policy: { effort: 'standard' as never },
|
|
16
|
+
...extra,
|
|
17
|
+
} as ExecutionState)
|
|
18
|
+
|
|
19
|
+
const exec = (): Execution => ({ ...state(), models: (() => {}) as never } as unknown as Execution)
|
|
20
|
+
|
|
21
|
+
describe('agent — the static flow provider', () => {
|
|
22
|
+
test('serves a declared flow and restores a serialized run', async () => {
|
|
23
|
+
const provider = makeStaticFlowProvider([agentRunFlow])
|
|
24
|
+
const model = await makeFlowModel(agentRunFlow)
|
|
25
|
+
model.transit(AgentRunTransition.Prepare, true)
|
|
26
|
+
const token = model.transit(AgentRunTransition.Work, true)
|
|
27
|
+
|
|
28
|
+
expect((await provider(AGENT_RUN_FLOW)).flow).toBe(AGENT_RUN_FLOW)
|
|
29
|
+
expect((await makeFlowModel(token, provider)).step().step).toBe(AgentRunStep.Working)
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
test('throws on an unknown flow, which is what makes token recovery work', async () => {
|
|
33
|
+
// `makeFlowModel` reads a string as a flow NAME first and only re-reads it as a serialized
|
|
34
|
+
// token once the provider throws. Returning null here would break every restore.
|
|
35
|
+
expect(makeStaticFlowProvider([])('nothing-declared')).rejects.toThrow()
|
|
36
|
+
})
|
|
37
|
+
})
|
|
38
|
+
|
|
39
|
+
describe('agent — the execution checkpoint plugin', () => {
|
|
40
|
+
test('persists a checkpointed state and restores it by key', async () => {
|
|
41
|
+
const store = createMemoryRunStateStore()
|
|
42
|
+
const plugin = makeAgentExecutionPlugin({ store })
|
|
43
|
+
|
|
44
|
+
await plugin.onCheckpoint!(state({ phase: 'develop' }), exec(), 'run-1')
|
|
45
|
+
|
|
46
|
+
expect(await plugin.onRestore!('run-1')).toMatchObject({ phase: 'develop' })
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
test('recovers the conversation from the execution dedication', async () => {
|
|
50
|
+
const store = createMemoryRunStateStore()
|
|
51
|
+
await makeAgentExecutionPlugin({ store }).onCheckpoint!(state(), exec(), 'run-2')
|
|
52
|
+
|
|
53
|
+
expect((await store.load('run-2'))?.conversationId).toBe('project:p1')
|
|
54
|
+
})
|
|
55
|
+
|
|
56
|
+
test('refuses a state too large to belong in a checkpoint', async () => {
|
|
57
|
+
// A project execution carries the whole specification. Writing that on every checkpoint is a
|
|
58
|
+
// storage problem that surfaces much later and much worse than a skipped write.
|
|
59
|
+
const store = createMemoryRunStateStore()
|
|
60
|
+
const plugin = makeAgentExecutionPlugin({ store, maxStateChars: 200 })
|
|
61
|
+
|
|
62
|
+
await plugin.onCheckpoint!(state({ project: { specification: 'x'.repeat(5_000) } }), exec(), 'run-3')
|
|
63
|
+
|
|
64
|
+
expect(await store.load('run-3')).toBeNull()
|
|
65
|
+
})
|
|
66
|
+
|
|
67
|
+
test('returns null for a key that was never checkpointed', async () => {
|
|
68
|
+
const plugin = makeAgentExecutionPlugin({ store: createMemoryRunStateStore() })
|
|
69
|
+
|
|
70
|
+
expect(await plugin.onRestore!('never-seen')).toBeNull()
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
test('announces a saved checkpoint on the transport by reference', async () => {
|
|
74
|
+
// The message carries a reference, not the state: a queue whose messages hold a whole project
|
|
75
|
+
// specification falls over on the first large project.
|
|
76
|
+
const transport = inProcessTransport()
|
|
77
|
+
const seen: unknown[] = []
|
|
78
|
+
await transport.consume(async message => { seen.push(message) })
|
|
79
|
+
|
|
80
|
+
await makeAgentExecutionPlugin({ store: createMemoryRunStateStore(), transport })
|
|
81
|
+
.onCheckpoint!(state(), exec(), 'run-4')
|
|
82
|
+
|
|
83
|
+
expect(seen).toEqual([{ id: 'run-4', conversationId: 'project:p1', flow: '', stateRef: 'run-4' }])
|
|
84
|
+
})
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
describe('agent — in-memory stores', () => {
|
|
88
|
+
test('a conversation allocates monotonic sequence numbers per thread', async () => {
|
|
89
|
+
const store = createMemoryConversationStore()
|
|
90
|
+
const base = { scope: 'p1', summary: 's', status: 'ok' as never }
|
|
91
|
+
|
|
92
|
+
await store.append({ ...base, conversationId: 'a' })
|
|
93
|
+
await store.append({ ...base, conversationId: 'b' })
|
|
94
|
+
const third = await store.append({ ...base, conversationId: 'a' })
|
|
95
|
+
|
|
96
|
+
expect(third.seq).toBe(2)
|
|
97
|
+
expect((await store.last({ conversationId: 'b', scope: 'p1' }, 10))).toHaveLength(1)
|
|
98
|
+
})
|
|
99
|
+
|
|
100
|
+
test('a conversation reads back newest first', async () => {
|
|
101
|
+
const store = createMemoryConversationStore()
|
|
102
|
+
const ref = { conversationId: 'a', scope: 'p1' }
|
|
103
|
+
for (const summary of ['first', 'second', 'third']) {
|
|
104
|
+
await store.append({ conversationId: 'a', scope: 'p1', summary, status: 'ok' as never })
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
expect((await store.last(ref, 2)).map(event => event.summary)).toEqual(['third', 'second'])
|
|
108
|
+
})
|
|
109
|
+
|
|
110
|
+
test('a memory graph node is replaced in place rather than appended', async () => {
|
|
111
|
+
const store = createMemoryGraphStore()
|
|
112
|
+
await store.write({ scope: 'p1', subsystem: 'auth', content: 'old', links: [] })
|
|
113
|
+
await store.write({ scope: 'p1', subsystem: 'auth', content: 'new', links: ['db'] })
|
|
114
|
+
|
|
115
|
+
expect(await store.index('p1')).toHaveLength(1)
|
|
116
|
+
expect((await store.read('p1', 'auth'))?.content).toBe('new')
|
|
117
|
+
})
|
|
118
|
+
|
|
119
|
+
test('the graph index carries names and links but not content', async () => {
|
|
120
|
+
const store = createMemoryGraphStore()
|
|
121
|
+
await store.write({ scope: 'p1', subsystem: 'auth', content: 'secret detail', links: ['db'] })
|
|
122
|
+
|
|
123
|
+
const index = await store.index('p1')
|
|
124
|
+
|
|
125
|
+
expect(index[0]).toEqual({ subsystem: 'auth', links: ['db'], updatedAt: expect.any(String) })
|
|
126
|
+
expect(JSON.stringify(index)).not.toContain('secret detail')
|
|
127
|
+
})
|
|
128
|
+
|
|
129
|
+
test('event memory prunes its own scope only', async () => {
|
|
130
|
+
// A shared cap would let a chatty subject evict a quiet one's whole history.
|
|
131
|
+
const store = createMemoryEventStore()
|
|
132
|
+
await store.append({ scope: 'quiet', kind: 'note', content: 'keep me' }, 2)
|
|
133
|
+
for (const content of ['a', 'b', 'c', 'd']) {
|
|
134
|
+
await store.append({ scope: 'busy', kind: 'note', content }, 2)
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
expect((await store.read('busy', 10)).map(event => event.content)).toEqual(['d', 'c'])
|
|
138
|
+
expect(await store.read('quiet', 10)).toHaveLength(1)
|
|
139
|
+
})
|
|
140
|
+
})
|